Author: Hongyu Chen Date: 2026-10-02T17:01:56Z New Revision: 08e245062b1028d09aadf8344d5d7e9177ce19d6
URL: https://github.com/llvm/llvm-project/commit/08e245062b1028d09aadf8344d5d7e9177ce19d6 DIFF: https://github.com/llvm/llvm-project/commit/08e245062b1028d09aadf8344d5d7e9177ce19d6.diff LOG: [RISCV][P-ext] Support Packed Slide 1 up/down (#228049) This patch supports packed slide 1 up/down intrinsics and codegen. Added: Modified: clang/lib/Headers/riscv_packed_simd.h clang/test/CodeGen/RISCV/rvp-intrinsics.c cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c llvm/lib/Target/RISCV/RISCVISelLowering.cpp llvm/lib/Target/RISCV/RISCVInstrInfoP.td llvm/test/CodeGen/RISCV/rvp-simd-32.ll llvm/test/CodeGen/RISCV/rvp-simd-64.ll Removed: ################################################################################ diff --git a/clang/lib/Headers/riscv_packed_simd.h b/clang/lib/Headers/riscv_packed_simd.h index 4148b165e28f3..9ae0c45d60674 100644 --- a/clang/lib/Headers/riscv_packed_simd.h +++ b/clang/lib/Headers/riscv_packed_simd.h @@ -274,6 +274,12 @@ typedef uint32_t uint32x2_t __attribute__((__vector_size__(8))); return __builtin_shufflevector(__lo, __hi, 0, 1, 2, 3, 4, 5, 6, 7); \ } +#define __packed_slide1(name, ty, elt_ty, ...) \ + static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rd, \ + elt_ty __rs1) { \ + return __builtin_shufflevector(__rd, (ty){__rs1}, __VA_ARGS__); \ + } + #define __packed_pair_ee2(name, ty) \ static __inline__ ty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1, ty __rs2) { \ return __builtin_shufflevector(__rs1, __rs2, 0, 2); \ @@ -1310,6 +1316,30 @@ __packed_concat4(pjoin2_u8x8, uint8x8_t, uint8x4_t) __packed_concat2(pjoin2_i16x4, int16x4_t, int16x2_t) __packed_concat2(pjoin2_u16x4, uint16x4_t, uint16x2_t) +/* Packed Slide 1 up/down (32-bit) */ +__packed_slide1(pslide1up_i8x4, int8x4_t, int8_t, 4, 0, 1, 2) +__packed_slide1(pslide1up_u8x4, uint8x4_t, uint8_t, 4, 0, 1, 2) +__packed_slide1(pslide1up_i16x2, int16x2_t, int16_t, 2, 0) +__packed_slide1(pslide1up_u16x2, uint16x2_t, uint16_t, 2, 0) +__packed_slide1(pslide1down_i8x4, int8x4_t, int8_t, 1, 2, 3, 4) +__packed_slide1(pslide1down_u8x4, uint8x4_t, uint8_t, 1, 2, 3, 4) +__packed_slide1(pslide1down_i16x2, int16x2_t, int16_t, 1, 2) +__packed_slide1(pslide1down_u16x2, uint16x2_t, uint16_t, 1, 2) + +/* Packed Slide 1 up/down (64-bit) */ +__packed_slide1(pslide1up_i8x8, int8x8_t, int8_t, 8, 0, 1, 2, 3, 4, 5, 6) +__packed_slide1(pslide1up_u8x8, uint8x8_t, uint8_t, 8, 0, 1, 2, 3, 4, 5, 6) +__packed_slide1(pslide1up_i16x4, int16x4_t, int16_t, 4, 0, 1, 2) +__packed_slide1(pslide1up_u16x4, uint16x4_t, uint16_t, 4, 0, 1, 2) +__packed_slide1(pslide1up_i32x2, int32x2_t, int32_t, 2, 0) +__packed_slide1(pslide1up_u32x2, uint32x2_t, uint32_t, 2, 0) +__packed_slide1(pslide1down_i8x8, int8x8_t, int8_t, 1, 2, 3, 4, 5, 6, 7, 8) +__packed_slide1(pslide1down_u8x8, uint8x8_t, uint8_t, 1, 2, 3, 4, 5, 6, 7, 8) +__packed_slide1(pslide1down_i16x4, int16x4_t, int16_t, 1, 2, 3, 4) +__packed_slide1(pslide1down_u16x4, uint16x4_t, uint16_t, 1, 2, 3, 4) +__packed_slide1(pslide1down_i32x2, int32x2_t, int32_t, 1, 2) +__packed_slide1(pslide1down_u32x2, uint32x2_t, uint32_t, 1, 2) + /* Packed Subvector Extract */ __packed_subvector_extract8(pget_i8x8_i8x4, int8x4_t, int8x8_t) __packed_subvector_extract8(pget_u8x8_u8x4, uint8x4_t, uint8x8_t) @@ -1531,6 +1561,7 @@ __packed_reinterpret(u32x2_i32x2, int32x2_t, uint32x2_t) #undef __packed_unzipo4 #undef __packed_concat2 #undef __packed_concat4 +#undef __packed_slide1 #undef __packed_pair_ee2 #undef __packed_pair_eo2 #undef __packed_pair_oe2 diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c b/clang/test/CodeGen/RISCV/rvp-intrinsics.c index 5f57545cf6c9f..d9ad53c373da2 100644 --- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c +++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c @@ -9096,3 +9096,447 @@ int32x2_t test_pwunzipho_i32x2(int16x4_t a) { return __riscv_pwunzipho_i32x2(a); uint32x2_t test_pwunzipho_u32x2(uint16x4_t a) { return __riscv_pwunzipho_u32x2(a); } + +/* Packed Slide 1 up/down (32-bit) */ + +// RV32-LABEL: define dso_local i32 @test_pslide1up_i8x4( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1up_i8x4( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +int8x4_t test_pslide1up_i8x4(int8x4_t rd, int8_t rs1) { + return __riscv_pslide1up_i8x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1up_u8x4( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1up_u8x4( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +uint8x4_t test_pslide1up_u8x4(uint8x4_t rd, uint8_t rs1) { + return __riscv_pslide1up_u8x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1up_i16x2( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1up_i16x2( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +int16x2_t test_pslide1up_i16x2(int16x2_t rd, int16_t rs1) { + return __riscv_pslide1up_i16x2(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1up_u16x2( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1up_u16x2( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +uint16x2_t test_pslide1up_u16x2(uint16x2_t rd, uint16_t rs1) { + return __riscv_pslide1up_u16x2(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1down_i8x4( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1down_i8x4( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +int8x4_t test_pslide1down_i8x4(int8x4_t rd, int8_t rs1) { + return __riscv_pslide1down_i8x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1down_u8x4( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1down_u8x4( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <4 x i8> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i8> [[TMP0]], <4 x i8> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i8> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +uint8x4_t test_pslide1down_u8x4(uint8x4_t rd, uint8_t rs1) { + return __riscv_pslide1down_u8x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1down_i16x2( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1down_i16x2( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +int16x2_t test_pslide1down_i16x2(int16x2_t rd, int16_t rs1) { + return __riscv_pslide1down_i16x2(rd, rs1); +} + +// RV32-LABEL: define dso_local i32 @test_pslide1down_u16x2( +// RV32-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV32-NEXT: ret i32 [[TMP1]] +// +// RV64-LABEL: define dso_local i32 @test_pslide1down_u16x2( +// RV64-SAME: i32 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i32 [[RD_COERCE]] to <2 x i16> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i16> [[TMP0]], <2 x i16> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i16> [[SHUFFLE_I]] to i32 +// RV64-NEXT: ret i32 [[TMP1]] +// +uint16x2_t test_pslide1down_u16x2(uint16x2_t rd, uint16_t rs1) { + return __riscv_pslide1down_u16x2(rd, rs1); +} + +/* Packed Slide 1 up/down (64-bit) */ + +// RV32-LABEL: define dso_local i64 @test_pslide1up_i8x8( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV32-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1up_i8x8( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV64-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +int8x8_t test_pslide1up_i8x8(int8x8_t rd, int8_t rs1) { + return __riscv_pslide1up_i8x8(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1up_u8x8( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV32-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1up_u8x8( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV64-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +uint8x8_t test_pslide1up_u8x8(uint8x8_t rd, uint8_t rs1) { + return __riscv_pslide1up_u8x8(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1up_i16x4( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1up_i16x4( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +int16x4_t test_pslide1up_i16x4(int16x4_t rd, int16_t rs1) { + return __riscv_pslide1up_i16x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1up_u16x4( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1up_u16x4( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 4, i32 0, i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +uint16x4_t test_pslide1up_u16x4(uint16x4_t rd, uint16_t rs1) { + return __riscv_pslide1up_u16x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1up_i32x2( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1up_i32x2( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +int32x2_t test_pslide1up_i32x2(int32x2_t rd, int32_t rs1) { + return __riscv_pslide1up_i32x2(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1up_u32x2( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1up_u32x2( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 2, i32 0> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +uint32x2_t test_pslide1up_u32x2(uint32x2_t rd, uint32_t rs1) { + return __riscv_pslide1up_u32x2(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1down_i8x8( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV32-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1down_i8x8( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV64-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +int8x8_t test_pslide1down_i8x8(int8x8_t rd, int8_t rs1) { + return __riscv_pslide1down_i8x8(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1down_u8x8( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV32-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1down_u8x8( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i8 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <8 x i8> +// RV64-NEXT: [[VECINIT8_I:%.*]] = insertelement <8 x i8> poison, i8 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <8 x i8> [[TMP0]], <8 x i8> [[VECINIT8_I]], <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <8 x i8> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +uint8x8_t test_pslide1down_u8x8(uint8x8_t rd, uint8_t rs1) { + return __riscv_pslide1down_u8x8(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1down_i16x4( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1down_i16x4( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +int16x4_t test_pslide1down_i16x4(int16x4_t rd, int16_t rs1) { + return __riscv_pslide1down_i16x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1down_u16x4( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV32-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1down_u16x4( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i16 noundef zeroext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <4 x i16> +// RV64-NEXT: [[VECINIT4_I:%.*]] = insertelement <4 x i16> poison, i16 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <4 x i16> [[TMP0]], <4 x i16> [[VECINIT4_I]], <4 x i32> <i32 1, i32 2, i32 3, i32 4> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <4 x i16> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +uint16x4_t test_pslide1down_u16x4(uint16x4_t rd, uint16_t rs1) { + return __riscv_pslide1down_u16x4(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1down_i32x2( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1down_i32x2( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +int32x2_t test_pslide1down_i32x2(int32x2_t rd, int32_t rs1) { + return __riscv_pslide1down_i32x2(rd, rs1); +} + +// RV32-LABEL: define dso_local i64 @test_pslide1down_u32x2( +// RV32-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef [[RS1:%.*]]) #[[ATTR0]] { +// RV32-NEXT: [[ENTRY:.*:]] +// RV32-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV32-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV32-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV32-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV32-NEXT: ret i64 [[TMP1]] +// +// RV64-LABEL: define dso_local i64 @test_pslide1down_u32x2( +// RV64-SAME: i64 noundef [[RD_COERCE:%.*]], i32 noundef signext [[RS1:%.*]]) #[[ATTR0]] { +// RV64-NEXT: [[ENTRY:.*:]] +// RV64-NEXT: [[TMP0:%.*]] = bitcast i64 [[RD_COERCE]] to <2 x i32> +// RV64-NEXT: [[VECINIT2_I:%.*]] = insertelement <2 x i32> poison, i32 [[RS1]], i64 0 +// RV64-NEXT: [[SHUFFLE_I:%.*]] = shufflevector <2 x i32> [[TMP0]], <2 x i32> [[VECINIT2_I]], <2 x i32> <i32 1, i32 2> +// RV64-NEXT: [[TMP1:%.*]] = bitcast <2 x i32> [[SHUFFLE_I]] to i64 +// RV64-NEXT: ret i64 [[TMP1]] +// +uint32x2_t test_pslide1down_u32x2(uint32x2_t rd, uint32_t rs1) { + return __riscv_pslide1down_u32x2(rd, rs1); +} diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c index e7ad4af120270..6d92dd5990261 100644 --- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c +++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c @@ -5547,3 +5547,137 @@ int32x2_t test_pwunzipho_i32x2(int16x4_t a) { uint32x2_t test_pwunzipho_u32x2(uint16x4_t a) { return __riscv_pwunzipho_u32x2(a); } + +/* Packed Slide 1 up/down (32-bit) */ + +// CHECK-LABEL: test_pslide1up_i8x4: +// CHECK: slx +int8x4_t test_pslide1up_i8x4(int8x4_t rd, int8_t rs1) { + return __riscv_pslide1up_i8x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_u8x4: +// CHECK: slx +uint8x4_t test_pslide1up_u8x4(uint8x4_t rd, uint8_t rs1) { + return __riscv_pslide1up_u8x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_i16x2: +// RV32: pack +// RV64: ppaire.h +int16x2_t test_pslide1up_i16x2(int16x2_t rd, int16_t rs1) { + return __riscv_pslide1up_i16x2(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_u16x2: +// RV32: pack +// RV64: ppaire.h +uint16x2_t test_pslide1up_u16x2(uint16x2_t rd, uint16_t rs1) { + return __riscv_pslide1up_u16x2(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_i8x4: +// RV32: srx +// RV64: pack +// RV64: srli +int8x4_t test_pslide1down_i8x4(int8x4_t rd, int8_t rs1) { + return __riscv_pslide1down_i8x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_u8x4: +// RV32: srx +// RV64: pack +// RV64: srli +uint8x4_t test_pslide1down_u8x4(uint8x4_t rd, uint8_t rs1) { + return __riscv_pslide1down_u8x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_i16x2: +// CHECK: ppairoe.h +int16x2_t test_pslide1down_i16x2(int16x2_t rd, int16_t rs1) { + return __riscv_pslide1down_i16x2(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_u16x2: +// CHECK: ppairoe.h +uint16x2_t test_pslide1down_u16x2(uint16x2_t rd, uint16_t rs1) { + return __riscv_pslide1down_u16x2(rd, rs1); +} + +/* Packed Slide 1 up/down (64-bit) */ + +// CHECK-LABEL: test_pslide1up_i8x8: +// CHECK: slx +int8x8_t test_pslide1up_i8x8(int8x8_t rd, int8_t rs1) { + return __riscv_pslide1up_i8x8(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_u8x8: +// CHECK: slx +uint8x8_t test_pslide1up_u8x8(uint8x8_t rd, uint8_t rs1) { + return __riscv_pslide1up_u8x8(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_i16x4: +// CHECK: slx +int16x4_t test_pslide1up_i16x4(int16x4_t rd, int16_t rs1) { + return __riscv_pslide1up_i16x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_u16x4: +// CHECK: slx +uint16x4_t test_pslide1up_u16x4(uint16x4_t rd, uint16_t rs1) { + return __riscv_pslide1up_u16x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_i32x2: +// RV32: mv +// RV64: pack +int32x2_t test_pslide1up_i32x2(int32x2_t rd, int32_t rs1) { + return __riscv_pslide1up_i32x2(rd, rs1); +} + +// CHECK-LABEL: test_pslide1up_u32x2: +// RV32: mv +// RV64: pack +uint32x2_t test_pslide1up_u32x2(uint32x2_t rd, uint32_t rs1) { + return __riscv_pslide1up_u32x2(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_i8x8: +// CHECK: srx +int8x8_t test_pslide1down_i8x8(int8x8_t rd, int8_t rs1) { + return __riscv_pslide1down_i8x8(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_u8x8: +// CHECK: srx +uint8x8_t test_pslide1down_u8x8(uint8x8_t rd, uint8_t rs1) { + return __riscv_pslide1down_u8x8(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_i16x4: +// CHECK: srx +int16x4_t test_pslide1down_i16x4(int16x4_t rd, int16_t rs1) { + return __riscv_pslide1down_i16x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_u16x4: +// CHECK: srx +uint16x4_t test_pslide1down_u16x4(uint16x4_t rd, uint16_t rs1) { + return __riscv_pslide1down_u16x4(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_i32x2: +// RV32: mv +// RV64: ppairoe.w +int32x2_t test_pslide1down_i32x2(int32x2_t rd, int32_t rs1) { + return __riscv_pslide1down_i32x2(rd, rs1); +} + +// CHECK-LABEL: test_pslide1down_u32x2: +// RV32: mv +// RV64: ppairoe.w +uint32x2_t test_pslide1down_u32x2(uint32x2_t rd, uint32_t rs1) { + return __riscv_pslide1down_u32x2(rd, rs1); +} diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp index fbdd4e48c9f63..c96a481705cec 100644 --- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp +++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp @@ -6497,6 +6497,111 @@ static SDValue lowerVECTOR_SHUFFLEAsPUnzip(ShuffleVectorSDNode *SVN, return DAG.getNode(Opc, DL, VT, V1, V2); } +// Match a slide by one element with a scalar inserted at either end: +// <b0, a0, ..., aN-2> -> slide1up +// <a1, ..., aN-1, b0> -> slide1down +static SDValue lowerVECTOR_SHUFFLEAsPSlide1(ShuffleVectorSDNode *SVN, + const RISCVSubtarget &Subtarget, + SelectionDAG &DAG) { + MVT VT = SVN->getSimpleValueType(0); + if (VT != MVT::v4i8 && VT != MVT::v8i8 && VT != MVT::v2i16 && + VT != MVT::v4i16 && VT != MVT::v2i32) + return SDValue(); + if (!Subtarget.is64Bit() && VT == MVT::v2i32) + return SDValue(); + + unsigned NumElts = VT.getVectorNumElements(); + ArrayRef<int> Mask = SVN->getMask(); + unsigned ActiveElts = NumElts; + if (Subtarget.is64Bit() && + all_of(Mask.drop_front(NumElts / 2), [](int M) { return M < 0; })) + ActiveElts /= 2; + + ArrayRef<int> ActiveMask = Mask.take_front(ActiveElts); + bool SlideUp = ActiveMask[0] == (int)NumElts && + all_of(enumerate(ActiveMask.drop_front()), [](const auto &M) { + return M.value() == (int)M.index(); + }); + bool SlideDown = ActiveMask.back() == (int)NumElts && + all_of(enumerate(ActiveMask.drop_back()), [](const auto &M) { + return M.value() == (int)M.index() + 1; + }); + if (!SlideUp && !SlideDown) + return SDValue(); + + SDValue V1 = SVN->getOperand(0); + SDValue V2 = SVN->getOperand(1); + SDLoc DL(SVN); + unsigned EltBits = VT.getScalarSizeInBits(); + unsigned SlideBits = ActiveElts * EltBits; + MVT XLenVT = Subtarget.getXLenVT(); + SDValue Scalar = DAG.getExtractVectorElt(DL, XLenVT, V2, 0); + + // A two-element slide is exactly a packed pair operation. + if (ActiveElts == 2) { + SDValue ScalarVec = DAG.getBitcast(VT, Scalar); + return DAG.getNode(SlideUp ? RISCVISD::PPAIRE : RISCVISD::PPAIROE, DL, VT, + SlideUp ? ScalarVec : V1, SlideUp ? V1 : ScalarVec); + } + + // Lower a full-register slide to the funnel shift instruction that + // implements it. + if (SlideBits == Subtarget.getXLen()) { + SDValue Bits = DAG.getBitcast(XLenVT, V1); + SDValue Shamt = DAG.getConstant(EltBits, DL, XLenVT); + if (SlideUp) { + Scalar = DAG.getNode(ISD::SHL, DL, XLenVT, Scalar, + DAG.getConstant(SlideBits - EltBits, DL, XLenVT)); + Bits = DAG.getNode(ISD::FSHL, DL, XLenVT, Bits, Scalar, Shamt); + } else { + Bits = DAG.getNode(ISD::FSHR, DL, XLenVT, Scalar, Bits, Shamt); + } + return DAG.getBitcast(VT, Bits); + } + + // A 32-bit packed slide on RV64 is carried in the low word of a GPR. Expand + // it using XLEN operations or packed pair instructions. + if (Subtarget.is64Bit()) { + assert(SlideBits == 32 && "Unexpected RV64 packed slide width"); + SDValue Shamt = DAG.getConstant(EltBits, DL, MVT::i64); + SDValue Bits = DAG.getBitcast(MVT::i64, V1); + if (SlideUp) { + Scalar = DAG.getNode(ISD::SHL, DL, MVT::i64, Scalar, + DAG.getConstant(64 - EltBits, DL, MVT::i64)); + Bits = DAG.getNode(ISD::FSHL, DL, MVT::i64, Bits, Scalar, Shamt); + } else { + SDValue Pair = DAG.getNode(RISCVISD::PPAIRE, DL, MVT::v2i32, + DAG.getBitcast(MVT::v2i32, V1), + DAG.getBitcast(MVT::v2i32, Scalar)); + Bits = DAG.getNode(ISD::SRL, DL, MVT::i64, DAG.getBitcast(MVT::i64, Pair), + Shamt); + } + return DAG.getBitcast(VT, Bits); + } + + // RV32 represents 64-bit packed values as two GPRs. Form each shifted half + // directly so the slide uses two XLEN funnel shifts. + assert(SlideBits == 64 && "Unexpected RV32 packed slide width"); + auto [LoVec, HiVec] = DAG.SplitVector(V1, DL); + MVT HalfVT = LoVec.getSimpleValueType(); + SDValue Lo = DAG.getBitcast(MVT::i32, LoVec); + SDValue Hi = DAG.getBitcast(MVT::i32, HiVec); + SDValue Shamt = DAG.getConstant(EltBits, DL, MVT::i32); + SDValue NewLo; + SDValue NewHi; + if (SlideUp) { + SDValue Insert = DAG.getNode(ISD::SHL, DL, MVT::i32, Scalar, + DAG.getConstant(32 - EltBits, DL, MVT::i32)); + NewLo = DAG.getNode(ISD::FSHL, DL, MVT::i32, Lo, Insert, Shamt); + NewHi = DAG.getNode(ISD::FSHL, DL, MVT::i32, Hi, Lo, Shamt); + } else { + NewLo = DAG.getNode(ISD::FSHR, DL, MVT::i32, Hi, Lo, Shamt); + NewHi = DAG.getNode(ISD::FSHR, DL, MVT::i32, Scalar, Hi, Shamt); + } + return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, DAG.getBitcast(HalfVT, NewLo), + DAG.getBitcast(HalfVT, NewHi)); +} + // Match the packed zero-extend shuffle mask <0, N, 2, N+2, ...>: even result // lanes keep operand 0's even lanes and odd result lanes come from operand 1. // The odd lanes may select any element of operand 1, which is looser than a @@ -6670,6 +6775,8 @@ SDValue RISCVTargetLowering::lowerVECTOR_SHUFFLE(SDValue Op, return DAG.getBitcast(VT, Srl); } + if (SDValue V = lowerVECTOR_SHUFFLEAsPSlide1(SVN, Subtarget, DAG)) + return V; if (SDValue V = lowerVECTOR_SHUFFLEAsPUnzip(SVN, DAG, Subtarget.is64Bit())) return V; if (SDValue V = lowerVECTOR_SHUFFLEAsPZip(SVN, Subtarget, DAG)) diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td index 67fee79db0582..ee5b17749cb73 100644 --- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td +++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td @@ -2011,6 +2011,7 @@ def riscv_ppaire : RVSDNode<"PPAIRE", SDT_RISCVPPair>; def riscv_ppaireo : RVSDNode<"PPAIREO", SDT_RISCVPPair>; def riscv_ppairoe : RVSDNode<"PPAIROE", SDT_RISCVPPair>; def riscv_ppairo : RVSDNode<"PPAIRO", SDT_RISCVPPair>; + def SDT_RISCVPackedBinary : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisSameAs<0, 1>, SDTCisSameAs<0, 2>]>; @@ -3641,6 +3642,10 @@ let append Predicates = [IsRV64] in { (ZIP16P GPR:$rs1, GPR:$rs2)>; // Packed pair: pair the even/odd-position elements of rs1 and rs2. + def : Pat<(v2i32 (riscv_ppaire (v2i32 GPR:$rs1), (v2i32 GPR:$rs2))), + (PACK GPR:$rs1, GPR:$rs2)>; + def : Pat<(v2i32 (riscv_ppairoe (v2i32 GPR:$rs1), (v2i32 GPR:$rs2))), + (PPAIROE_W GPR:$rs1, GPR:$rs2)>; def : Pat<(v8i8 (riscv_ppaire (v8i8 GPR:$rs1), (v8i8 GPR:$rs2))), (PPAIRE_B GPR:$rs1, GPR:$rs2)>; def : Pat<(v8i8 (riscv_ppaireo (v8i8 GPR:$rs1), (v8i8 GPR:$rs2))), diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll index e947cebb90a91..bdeef32a84cb9 100644 --- a/llvm/test/CodeGen/RISCV/rvp-simd-32.ll +++ b/llvm/test/CodeGen/RISCV/rvp-simd-32.ll @@ -4184,3 +4184,69 @@ define <2 x i16> @test_pusati_u16x2_max_width(<2 x i16> %a) { %res = call <2 x i16> @llvm.riscv.pusati.v2i16.i32(<2 x i16> %a, i32 15) ret <2 x i16> %res } + +define <4 x i8> @test_pslide1up_v4i8(<4 x i8> %rd, i8 %rs1) { +; RV32-LABEL: test_pslide1up_v4i8: +; RV32: # %bb.0: +; RV32-NEXT: slli a1, a1, 24 +; RV32-NEXT: li a2, 8 +; RV32-NEXT: slx a0, a1, a2 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1up_v4i8: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 56 +; RV64-NEXT: li a2, 8 +; RV64-NEXT: slx a0, a1, a2 +; RV64-NEXT: ret + %scalar = insertelement <4 x i8> poison, i8 %rs1, i64 0 + %res = shufflevector <4 x i8> %rd, <4 x i8> %scalar, <4 x i32> <i32 4, i32 0, i32 1, i32 2> + ret <4 x i8> %res +} + +define <4 x i8> @test_pslide1down_v4i8(<4 x i8> %rd, i8 %rs1) { +; RV32-LABEL: test_pslide1down_v4i8: +; RV32: # %bb.0: +; RV32-NEXT: li a2, 8 +; RV32-NEXT: srx a0, a1, a2 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1down_v4i8: +; RV64: # %bb.0: +; RV64-NEXT: pack a0, a0, a1 +; RV64-NEXT: srli a0, a0, 8 +; RV64-NEXT: ret + %scalar = insertelement <4 x i8> poison, i8 %rs1, i64 0 + %res = shufflevector <4 x i8> %rd, <4 x i8> %scalar, <4 x i32> <i32 1, i32 2, i32 3, i32 4> + ret <4 x i8> %res +} + +define <2 x i16> @test_pslide1up_v2i16(<2 x i16> %rd, i16 %rs1) { +; RV32-LABEL: test_pslide1up_v2i16: +; RV32: # %bb.0: +; RV32-NEXT: pack a0, a1, a0 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1up_v2i16: +; RV64: # %bb.0: +; RV64-NEXT: ppaire.h a0, a1, a0 +; RV64-NEXT: ret + %scalar = insertelement <2 x i16> poison, i16 %rs1, i64 0 + %res = shufflevector <2 x i16> %rd, <2 x i16> %scalar, <2 x i32> <i32 2, i32 0> + ret <2 x i16> %res +} + +define <2 x i16> @test_pslide1down_v2i16(<2 x i16> %rd, i16 %rs1) { +; RV32-LABEL: test_pslide1down_v2i16: +; RV32: # %bb.0: +; RV32-NEXT: ppairoe.h a0, a0, a1 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1down_v2i16: +; RV64: # %bb.0: +; RV64-NEXT: ppairoe.h a0, a0, a1 +; RV64-NEXT: ret + %scalar = insertelement <2 x i16> poison, i16 %rs1, i64 0 + %res = shufflevector <2 x i16> %rd, <2 x i16> %scalar, <2 x i32> <i32 1, i32 2> + ret <2 x i16> %res +} diff --git a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll index 3ba711d2082e0..b930302b8bfe2 100644 --- a/llvm/test/CodeGen/RISCV/rvp-simd-64.ll +++ b/llvm/test/CodeGen/RISCV/rvp-simd-64.ll @@ -8650,3 +8650,111 @@ define <2 x i32> @test_pusati_u32x2_max_width(<2 x i32> %a) { %res = call <2 x i32> @llvm.riscv.pusati.v2i32.i32(<2 x i32> %a, i32 31) ret <2 x i32> %res } + +define <8 x i8> @test_pslide1up_v8i8(<8 x i8> %rd, i8 %rs1) { +; RV32-LABEL: test_pslide1up_v8i8: +; RV32: # %bb.0: +; RV32-NEXT: li a3, 8 +; RV32-NEXT: slli a2, a2, 24 +; RV32-NEXT: slx a1, a0, a3 +; RV32-NEXT: slx a0, a2, a3 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1up_v8i8: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 56 +; RV64-NEXT: li a2, 8 +; RV64-NEXT: slx a0, a1, a2 +; RV64-NEXT: ret + %scalar = insertelement <8 x i8> poison, i8 %rs1, i64 0 + %res = shufflevector <8 x i8> %rd, <8 x i8> %scalar, <8 x i32> <i32 8, i32 0, i32 1, i32 2, i32 3, i32 4, i32 5, i32 6> + ret <8 x i8> %res +} + +define <8 x i8> @test_pslide1down_v8i8(<8 x i8> %rd, i8 %rs1) { +; RV32-LABEL: test_pslide1down_v8i8: +; RV32: # %bb.0: +; RV32-NEXT: li a3, 8 +; RV32-NEXT: srx a0, a1, a3 +; RV32-NEXT: srx a1, a2, a3 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1down_v8i8: +; RV64: # %bb.0: +; RV64-NEXT: li a2, 8 +; RV64-NEXT: srx a0, a1, a2 +; RV64-NEXT: ret + %scalar = insertelement <8 x i8> poison, i8 %rs1, i64 0 + %res = shufflevector <8 x i8> %rd, <8 x i8> %scalar, <8 x i32> <i32 1, i32 2, i32 3, i32 4, i32 5, i32 6, i32 7, i32 8> + ret <8 x i8> %res +} + +define <4 x i16> @test_pslide1up_v4i16(<4 x i16> %rd, i16 %rs1) { +; RV32-LABEL: test_pslide1up_v4i16: +; RV32: # %bb.0: +; RV32-NEXT: li a3, 16 +; RV32-NEXT: slli a2, a2, 16 +; RV32-NEXT: slx a1, a0, a3 +; RV32-NEXT: slx a0, a2, a3 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1up_v4i16: +; RV64: # %bb.0: +; RV64-NEXT: slli a1, a1, 48 +; RV64-NEXT: li a2, 16 +; RV64-NEXT: slx a0, a1, a2 +; RV64-NEXT: ret + %scalar = insertelement <4 x i16> poison, i16 %rs1, i64 0 + %res = shufflevector <4 x i16> %rd, <4 x i16> %scalar, <4 x i32> <i32 4, i32 0, i32 1, i32 2> + ret <4 x i16> %res +} + +define <4 x i16> @test_pslide1down_v4i16(<4 x i16> %rd, i16 %rs1) { +; RV32-LABEL: test_pslide1down_v4i16: +; RV32: # %bb.0: +; RV32-NEXT: li a3, 16 +; RV32-NEXT: srx a0, a1, a3 +; RV32-NEXT: srx a1, a2, a3 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1down_v4i16: +; RV64: # %bb.0: +; RV64-NEXT: li a2, 16 +; RV64-NEXT: srx a0, a1, a2 +; RV64-NEXT: ret + %scalar = insertelement <4 x i16> poison, i16 %rs1, i64 0 + %res = shufflevector <4 x i16> %rd, <4 x i16> %scalar, <4 x i32> <i32 1, i32 2, i32 3, i32 4> + ret <4 x i16> %res +} + +define <2 x i32> @test_pslide1up_v2i32(<2 x i32> %rd, i32 %rs1) { +; RV32-LABEL: test_pslide1up_v2i32: +; RV32: # %bb.0: +; RV32-NEXT: mv a1, a0 +; RV32-NEXT: mv a0, a2 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1up_v2i32: +; RV64: # %bb.0: +; RV64-NEXT: pack a0, a1, a0 +; RV64-NEXT: ret + %scalar = insertelement <2 x i32> poison, i32 %rs1, i64 0 + %res = shufflevector <2 x i32> %rd, <2 x i32> %scalar, <2 x i32> <i32 2, i32 0> + ret <2 x i32> %res +} + +define <2 x i32> @test_pslide1down_v2i32(<2 x i32> %rd, i32 %rs1) { +; RV32-LABEL: test_pslide1down_v2i32: +; RV32: # %bb.0: +; RV32-NEXT: mv a0, a1 +; RV32-NEXT: mv a1, a2 +; RV32-NEXT: ret +; +; RV64-LABEL: test_pslide1down_v2i32: +; RV64: # %bb.0: +; RV64-NEXT: ppairoe.w a0, a0, a1 +; RV64-NEXT: ret + %scalar = insertelement <2 x i32> poison, i32 %rs1, i64 0 + %res = shufflevector <2 x i32> %rd, <2 x i32> %scalar, <2 x i32> <i32 1, i32 2> + ret <2 x i32> %res +} _______________________________________________ cfe-commits mailing list [email protected] https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits
