https://github.com/Michael-Chen-NJU updated 
https://github.com/llvm/llvm-project/pull/224790

From 34fcde06043bab04888cebd7c45273b43b2887fd Mon Sep 17 00:00:00 2001
From: Michael-Chen-NJU <[email protected]>
Date: Fri, 18 Sep 2026 11:13:19 +0800
Subject: [PATCH 1/3] [Clang][RISCV] Add packed widening shift intrinsics

Add the 32-bit forms of the RISC-V P-extension packed widening shift intrinsics 
to riscv_packed_simd.h using generic extend-and-shift IR.\n\nRecognize the 
generic widening shift pattern in the RISC-V backend and select the spec-listed 
RV32 instructions while retaining composed RV64 sequences.\n\nAdd Clang 
CodeGen, LLVM CodeGen, and intrinsic header tests for register, immediate, and 
masked shift amounts.
---
 clang/lib/Headers/riscv_packed_simd.h         |  12 ++
 clang/test/CodeGen/RISCV/rvp-intrinsics.c     | 116 ++++++++++++++
 .../riscv_packed_simd.c                       |  86 ++++++++++
 llvm/lib/Target/RISCV/RISCVISelLowering.cpp   |  23 +++
 llvm/lib/Target/RISCV/RISCVInstrInfoP.td      |  27 ++++
 llvm/test/CodeGen/RISCV/rvp-widening-shift.ll | 147 ++++++++++++++++++
 6 files changed, 411 insertions(+)
 create mode 100644 llvm/test/CodeGen/RISCV/rvp-widening-shift.ll

diff --git a/clang/lib/Headers/riscv_packed_simd.h 
b/clang/lib/Headers/riscv_packed_simd.h
index db6d0d37c2e8a..96413a68919f3 100644
--- a/clang/lib/Headers/riscv_packed_simd.h
+++ b/clang/lib/Headers/riscv_packed_simd.h
@@ -175,6 +175,11 @@ typedef uint32_t uint32x2_t 
__attribute__((__vector_size__(8)));
   static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) {          
\
     return __builtin_convertvector(__rs1, rty);                                
\
   }
+#define __packed_widen_shift(name, rty, ty, mask)                              
\
+  static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1,            
\
+                                                          unsigned __shamt) {  
\
+    return __builtin_convertvector(__rs1, rty) << (__shamt & (mask));          
\
+  }
 #define __packed_widen_binary_op(name, rty, ty, op)                            
\
   static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1,            
\
                                                           ty __rs2) {          
\
@@ -730,6 +735,12 @@ __packed_widen_high4(pwcvth_u16x4, uint16x4_t, uint8x4_t)
 __packed_widen_high2(pwcvth_i32x2, int32x2_t, int16x2_t)
 __packed_widen_high2(pwcvth_u32x2, uint32x2_t, uint16x2_t)
 
+/* Packed Widening Shift */
+__packed_widen_shift(pwsll_s_u16x4, uint16x4_t, uint8x4_t, 0xf)
+__packed_widen_shift(pwsll_s_u32x2, uint32x2_t, uint16x2_t, 0x1f)
+__packed_widen_shift(pwsla_s_i16x4, int16x4_t, int8x4_t, 0xf)
+__packed_widen_shift(pwsla_s_i32x2, int32x2_t, int16x2_t, 0x1f)
+
 /* Packed Widening Addition and Subtraction */
 __packed_widen_binary_op(pwadd_i16x4, int16x4_t, int8x4_t, +)
 __packed_widen_binary_op(pwadd_i32x2, int32x2_t, int16x2_t, +)
@@ -1314,6 +1325,7 @@ __packed_reinterpret(u32x2_i32x2, int32x2_t, uint32x2_t)
 #undef __packed_merge_builtin
 #undef __packed_unary_builtin
 #undef __packed_widen_convert
+#undef __packed_widen_shift
 #undef __packed_widen_binary_op
 #undef __packed_widen_binary_acc_op
 #undef __packed_widen_mul
diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c 
b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
index c6721dbeb5db8..0946f4885d938 100644
--- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c
+++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
@@ -8158,6 +8158,122 @@ uint32x2_t test_pwcvtu_u32x2(uint16x2_t rs1) {
   return __riscv_pwcvtu_u32x2(rs1);
 }
 
+// RV32-LABEL: define dso_local i64 @test_pwsll_s_u16x4(
+// RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) 
#[[ATTR0]] {
+// RV32-NEXT:  [[ENTRY:.*:]]
+// RV32-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8>
+// RV32-NEXT:    [[CONV_I:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i16>
+// RV32-NEXT:    [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16
+// RV32-NEXT:    [[TMP2:%.*]] = and i16 [[TMP1]], 15
+// RV32-NEXT:    [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], 
i64 0
+// RV32-NEXT:    [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x 
i16> poison, <4 x i32> zeroinitializer
+// RV32-NEXT:    [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]]
+// RV32-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64
+// RV32-NEXT:    ret i64 [[TMP4]]
+//
+// RV64-LABEL: define dso_local i64 @test_pwsll_s_u16x4(
+// RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext 
[[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT:  [[ENTRY:.*:]]
+// RV64-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8>
+// RV64-NEXT:    [[CONV_I:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i16>
+// RV64-NEXT:    [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16
+// RV64-NEXT:    [[TMP2:%.*]] = and i16 [[TMP1]], 15
+// RV64-NEXT:    [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], 
i64 0
+// RV64-NEXT:    [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x 
i16> poison, <4 x i32> zeroinitializer
+// RV64-NEXT:    [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]]
+// RV64-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64
+// RV64-NEXT:    ret i64 [[TMP4]]
+//
+uint16x4_t test_pwsll_s_u16x4(uint8x4_t rs1, unsigned shamt) {
+  return __riscv_pwsll_s_u16x4(rs1, shamt);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pwsll_s_u32x2(
+// RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) 
#[[ATTR0]] {
+// RV32-NEXT:  [[ENTRY:.*:]]
+// RV32-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV32-NEXT:    [[CONV_I:%.*]] = zext <2 x i16> [[TMP0]] to <2 x i32>
+// RV32-NEXT:    [[AND_I:%.*]] = and i32 [[SHAMT]], 31
+// RV32-NEXT:    [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, 
i32 [[AND_I]], i64 0
+// RV32-NEXT:    [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> 
[[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer
+// RV32-NEXT:    [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]]
+// RV32-NEXT:    [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64
+// RV32-NEXT:    ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pwsll_s_u32x2(
+// RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext 
[[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT:  [[ENTRY:.*:]]
+// RV64-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV64-NEXT:    [[CONV_I:%.*]] = zext <2 x i16> [[TMP0]] to <2 x i32>
+// RV64-NEXT:    [[AND_I:%.*]] = and i32 [[SHAMT]], 31
+// RV64-NEXT:    [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, 
i32 [[AND_I]], i64 0
+// RV64-NEXT:    [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> 
[[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer
+// RV64-NEXT:    [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]]
+// RV64-NEXT:    [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64
+// RV64-NEXT:    ret i64 [[TMP1]]
+//
+uint32x2_t test_pwsll_s_u32x2(uint16x2_t rs1, unsigned shamt) {
+  return __riscv_pwsll_s_u32x2(rs1, shamt);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pwsla_s_i16x4(
+// RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) 
#[[ATTR0]] {
+// RV32-NEXT:  [[ENTRY:.*:]]
+// RV32-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8>
+// RV32-NEXT:    [[CONV_I:%.*]] = sext <4 x i8> [[TMP0]] to <4 x i16>
+// RV32-NEXT:    [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16
+// RV32-NEXT:    [[TMP2:%.*]] = and i16 [[TMP1]], 15
+// RV32-NEXT:    [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], 
i64 0
+// RV32-NEXT:    [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x 
i16> poison, <4 x i32> zeroinitializer
+// RV32-NEXT:    [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]]
+// RV32-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64
+// RV32-NEXT:    ret i64 [[TMP4]]
+//
+// RV64-LABEL: define dso_local i64 @test_pwsla_s_i16x4(
+// RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext 
[[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT:  [[ENTRY:.*:]]
+// RV64-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8>
+// RV64-NEXT:    [[CONV_I:%.*]] = sext <4 x i8> [[TMP0]] to <4 x i16>
+// RV64-NEXT:    [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16
+// RV64-NEXT:    [[TMP2:%.*]] = and i16 [[TMP1]], 15
+// RV64-NEXT:    [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], 
i64 0
+// RV64-NEXT:    [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x 
i16> poison, <4 x i32> zeroinitializer
+// RV64-NEXT:    [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]]
+// RV64-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64
+// RV64-NEXT:    ret i64 [[TMP4]]
+//
+int16x4_t test_pwsla_s_i16x4(int8x4_t rs1, unsigned shamt) {
+  return __riscv_pwsla_s_i16x4(rs1, shamt);
+}
+
+// RV32-LABEL: define dso_local i64 @test_pwsla_s_i32x2(
+// RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) 
#[[ATTR0]] {
+// RV32-NEXT:  [[ENTRY:.*:]]
+// RV32-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV32-NEXT:    [[CONV_I:%.*]] = sext <2 x i16> [[TMP0]] to <2 x i32>
+// RV32-NEXT:    [[AND_I:%.*]] = and i32 [[SHAMT]], 31
+// RV32-NEXT:    [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, 
i32 [[AND_I]], i64 0
+// RV32-NEXT:    [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> 
[[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer
+// RV32-NEXT:    [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]]
+// RV32-NEXT:    [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64
+// RV32-NEXT:    ret i64 [[TMP1]]
+//
+// RV64-LABEL: define dso_local i64 @test_pwsla_s_i32x2(
+// RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext 
[[SHAMT:%.*]]) #[[ATTR0]] {
+// RV64-NEXT:  [[ENTRY:.*:]]
+// RV64-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
+// RV64-NEXT:    [[CONV_I:%.*]] = sext <2 x i16> [[TMP0]] to <2 x i32>
+// RV64-NEXT:    [[AND_I:%.*]] = and i32 [[SHAMT]], 31
+// RV64-NEXT:    [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, 
i32 [[AND_I]], i64 0
+// RV64-NEXT:    [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> 
[[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer
+// RV64-NEXT:    [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]]
+// RV64-NEXT:    [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64
+// RV64-NEXT:    ret i64 [[TMP1]]
+//
+int32x2_t test_pwsla_s_i32x2(int16x2_t rs1, unsigned shamt) {
+  return __riscv_pwsla_s_i32x2(rs1, shamt);
+}
+
 // RV32-LABEL: define dso_local i64 @test_pwadd_i16x4(
 // RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[RS2_COERCE:%.*]]) 
#[[ATTR0]] {
 // RV32-NEXT:  [[ENTRY:.*:]]
diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c 
b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
index 14e64c3c3584b..910e1b8b85500 100644
--- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
+++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
@@ -2317,6 +2317,92 @@ uint32x2_t test_pwcvtu_u32x2(uint16x2_t rs1) {
   return __riscv_pwcvtu_u32x2(rs1);
 }
 
+// CHECK-LABEL: test_pwsll_s_u16x4:
+// RV32:        pwsll.bs
+// RV64:        pwcvtu.wb
+// RV64:        psll.hs
+uint16x4_t test_pwsll_s_u16x4(uint8x4_t rs1, unsigned shamt) {
+  return __riscv_pwsll_s_u16x4(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pwsll_s_u32x2:
+// RV32:        pwsll.hs
+// RV64:        pwcvtu.wh
+// RV64:        psll.ws
+uint32x2_t test_pwsll_s_u32x2(uint16x2_t rs1, unsigned shamt) {
+  return __riscv_pwsll_s_u32x2(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pwsla_s_i16x4:
+// RV32:        pwsla.bs
+// RV64:        pwcvtu.wb
+// RV64:        psext.h.b
+// RV64:        psll.hs
+int16x4_t test_pwsla_s_i16x4(int8x4_t rs1, unsigned shamt) {
+  return __riscv_pwsla_s_i16x4(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pwsla_s_i32x2:
+// RV32:        pwsla.hs
+// RV64:        pwcvtu.wh
+// RV64:        psext.w.h
+// RV64:        psll.ws
+int32x2_t test_pwsla_s_i32x2(int16x2_t rs1, unsigned shamt) {
+  return __riscv_pwsla_s_i32x2(rs1, shamt);
+}
+
+// CHECK-LABEL: test_pwsll_s_u16x4_imm:
+// RV32:        pwslli.b{{[[:space:]]}}a0, a0, 3
+// RV64:        pwcvtu.wb
+// RV64:        pslli.h{{[[:space:]]}}a0, a0, 3
+uint16x4_t test_pwsll_s_u16x4_imm(uint8x4_t rs1) {
+  return __riscv_pwsll_s_u16x4(rs1, 3);
+}
+
+// CHECK-LABEL: test_pwsll_s_u32x2_imm:
+// RV32:        pwslli.h{{[[:space:]]}}a0, a0, 7
+// RV64:        pwcvtu.wh
+// RV64:        pslli.w{{[[:space:]]}}a0, a0, 7
+uint32x2_t test_pwsll_s_u32x2_imm(uint16x2_t rs1) {
+  return __riscv_pwsll_s_u32x2(rs1, 7);
+}
+
+// CHECK-LABEL: test_pwsla_s_i16x4_imm:
+// RV32:        pwslai.b{{[[:space:]]}}a0, a0, 3
+// RV64:        pwcvtu.wb
+// RV64:        psext.h.b
+// RV64:        pslli.h{{[[:space:]]}}a0, a0, 3
+int16x4_t test_pwsla_s_i16x4_imm(int8x4_t rs1) {
+  return __riscv_pwsla_s_i16x4(rs1, 3);
+}
+
+// CHECK-LABEL: test_pwsla_s_i32x2_imm:
+// RV32:        pwslai.h{{[[:space:]]}}a0, a0, 7
+// RV64:        pwcvtu.wh
+// RV64:        psext.w.h
+// RV64:        pslli.w{{[[:space:]]}}a0, a0, 7
+int32x2_t test_pwsla_s_i32x2_imm(int16x2_t rs1) {
+  return __riscv_pwsla_s_i32x2(rs1, 7);
+}
+
+// Verify that an out-of-range constant shift amount is masked to the maximum
+// in-range value by the header implementation.
+// CHECK-LABEL: test_pwsll_s_u16x4_masked_imm:
+// RV32:        pslli.dh{{[[:space:]]}}a0, a0, 15
+// RV64:        pwcvtu.wb
+// RV64:        pslli.h{{[[:space:]]}}a0, a0, 15
+uint16x4_t test_pwsll_s_u16x4_masked_imm(uint8x4_t rs1) {
+  return __riscv_pwsll_s_u16x4(rs1, 31);
+}
+
+// CHECK-LABEL: test_pwsll_s_u32x2_masked_imm:
+// RV32:        pslli.dw{{[[:space:]]}}a0, a0, 31
+// RV64:        pwcvtu.wh
+// RV64:        pslli.w{{[[:space:]]}}a0, a0, 31
+uint32x2_t test_pwsll_s_u32x2_masked_imm(uint16x2_t rs1) {
+  return __riscv_pwsll_s_u32x2(rs1, 63);
+}
+
 // CHECK-LABEL: test_pwadd_i16x4:
 // RV32:        pwadd.b
 // RV64:        zip8p
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp 
b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 2c05e3000c5eb..03966db5d6021 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -9545,6 +9545,29 @@ SDValue RISCVTargetLowering::LowerOperation(SDValue Op,
         if (!SplatVal)
           return SDValue();
 
+        // The 32-bit packed widening shift intrinsics produce extend followed
+        // by a scalar-splat shift. Preserve that shape as a widening shift
+        // before generic packed-shift lowering loses the narrow source.
+        if (!Subtarget.is64Bit() && Op.getOpcode() == ISD::SHL) {
+          using namespace SDPatternMatch;
+          MVT VT = Op.getSimpleValueType();
+          if (VT == MVT::v4i16 || VT == MVT::v2i32) {
+            MVT SrcVT = VT == MVT::v4i16 ? MVT::v4i8 : MVT::v2i16;
+            SDValue Src;
+            unsigned ExtendOpcode = Op.getOperand(0).getOpcode();
+            if ((ExtendOpcode == ISD::SIGN_EXTEND ||
+                 ExtendOpcode == ISD::ZERO_EXTEND) &&
+                sd_match(Op.getOperand(0),
+                         m_OneUse(m_Node(ExtendOpcode,
+                                         m_Value(Src, m_SpecificVT(SrcVT)))))) 
{
+              unsigned Opc = ExtendOpcode == ISD::SIGN_EXTEND ? RISCVISD::PWSLA
+                                                              : 
RISCVISD::PWSLL;
+              SplatVal = DAG.getZExtOrTrunc(SplatVal, SDLoc(Op), MVT::i32);
+              return DAG.getNode(Opc, SDLoc(Op), VT, Src, SplatVal);
+            }
+          }
+        }
+
         unsigned Opc;
         switch (Op.getOpcode()) {
         default:
diff --git a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td 
b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
index 831a591cf6283..81bfbc7bd4326 100644
--- a/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
+++ b/llvm/lib/Target/RISCV/RISCVInstrInfoP.td
@@ -2052,6 +2052,15 @@ def riscv_pnsrl : RVSDNode<"PNSRL", 
SDT_RISCVPackedNarrowingShift>;
 def riscv_pnclip  : RVSDNode<"PNCLIP", SDT_RISCVPackedNarrowingShift>;
 def riscv_pnclipu : RVSDNode<"PNCLIPU", SDT_RISCVPackedNarrowingShift>;
 
+// RV32 packed widening shift.
+def SDT_RISCVPackedWideningShift
+    : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisVec<1>,
+                           SDTCisOpSmallerThanOp<1, 0>,
+                           SDTCisSameNumEltsAs<0, 1>,
+                           SDTCisVT<2, XLenVT>]>;
+def riscv_pwsll : RVSDNode<"PWSLL", SDT_RISCVPackedWideningShift>;
+def riscv_pwsla : RVSDNode<"PWSLA", SDT_RISCVPackedWideningShift>;
+
 // Packed narrowing clip pair.
 def SDT_RISCVPackedNarrowingClip
     : SDTypeProfile<1, 2, [SDTCisVec<0>, SDTCisVec<1>,
@@ -2581,6 +2590,24 @@ let append Predicates = [IsRV32] in {
   def : Pat<(v2i32 (sub (zext (v2i16 GPR:$rs1)), (zext (v2i16 GPR:$rs2)))),
             (PWSUBU_H GPR:$rs1, GPR:$rs2)>;
 
+  // Packed widening shift patterns.
+  def : Pat<(v4i16 (riscv_pwsll (v4i8 GPR:$rs1), uimm4:$imm)),
+            (PWSLLI_B GPR:$rs1, uimm4:$imm)>;
+  def : Pat<(v2i32 (riscv_pwsll (v2i16 GPR:$rs1), uimm5:$imm)),
+            (PWSLLI_H GPR:$rs1, uimm5:$imm)>;
+  def : Pat<(v4i16 (riscv_pwsla (v4i8 GPR:$rs1), uimm4:$imm)),
+            (PWSLAI_B GPR:$rs1, uimm4:$imm)>;
+  def : Pat<(v2i32 (riscv_pwsla (v2i16 GPR:$rs1), uimm5:$imm)),
+            (PWSLAI_H GPR:$rs1, uimm5:$imm)>;
+  def : Pat<(v4i16 (riscv_pwsll (v4i8 GPR:$rs1), shiftMask32:$rs2)),
+            (PWSLL_BS GPR:$rs1, shiftMask32:$rs2)>;
+  def : Pat<(v2i32 (riscv_pwsll (v2i16 GPR:$rs1), shiftMask32:$rs2)),
+            (PWSLL_HS GPR:$rs1, shiftMask32:$rs2)>;
+  def : Pat<(v4i16 (riscv_pwsla (v4i8 GPR:$rs1), shiftMask32:$rs2)),
+            (PWSLA_BS GPR:$rs1, shiftMask32:$rs2)>;
+  def : Pat<(v2i32 (riscv_pwsla (v2i16 GPR:$rs1), shiftMask32:$rs2)),
+            (PWSLA_HS GPR:$rs1, shiftMask32:$rs2)>;
+
   // Packed widening multiply patterns.
   def : Pat<(v4i16 (riscv_pwmul (v4i8 GPR:$rs1), (v4i8 GPR:$rs2))),
             (PWMUL_B GPR:$rs1, GPR:$rs2)>;
diff --git a/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll 
b/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll
new file mode 100644
index 0000000000000..ec6a0e0032c87
--- /dev/null
+++ b/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll
@@ -0,0 +1,147 @@
+; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py 
UTC_ARGS: --version 6
+; RUN: llc -mtriple=riscv32 -mattr=+experimental-p -verify-machineinstrs < %s 
| \
+; RUN:   FileCheck %s --check-prefixes=CHECK,RV32
+; RUN: llc -mtriple=riscv64 -mattr=+experimental-p -verify-machineinstrs < %s 
| \
+; RUN:   FileCheck %s --check-prefixes=CHECK,RV64
+
+define <4 x i16> @pwsll_v4i8(<4 x i8> %x, i16 %shamt) {
+; RV32-LABEL: pwsll_v4i8:
+; RV32:       # %bb.0:
+; RV32-NEXT:    pwsll.bs a0, a0, a1
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: pwsll_v4i8:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pwcvtu.wb a0, a0
+; RV64-NEXT:    psll.hs a0, a0, a1
+; RV64-NEXT:    ret
+  %ext = zext <4 x i8> %x to <4 x i16>
+  %splat.ins = insertelement <4 x i16> poison, i16 %shamt, i64 0
+  %splat = shufflevector <4 x i16> %splat.ins, <4 x i16> poison, <4 x i32> 
zeroinitializer
+  %res = shl <4 x i16> %ext, %splat
+  ret <4 x i16> %res
+}
+
+define <2 x i32> @pwsll_v2i16(<2 x i16> %x, i32 %shamt) {
+; RV32-LABEL: pwsll_v2i16:
+; RV32:       # %bb.0:
+; RV32-NEXT:    pwsll.hs a0, a0, a1
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: pwsll_v2i16:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pwcvtu.wh a0, a0
+; RV64-NEXT:    psll.ws a0, a0, a1
+; RV64-NEXT:    ret
+  %ext = zext <2 x i16> %x to <2 x i32>
+  %splat.ins = insertelement <2 x i32> poison, i32 %shamt, i64 0
+  %splat = shufflevector <2 x i32> %splat.ins, <2 x i32> poison, <2 x i32> 
zeroinitializer
+  %res = shl <2 x i32> %ext, %splat
+  ret <2 x i32> %res
+}
+
+define <4 x i16> @pwsla_v4i8(<4 x i8> %x, i16 %shamt) {
+; RV32-LABEL: pwsla_v4i8:
+; RV32:       # %bb.0:
+; RV32-NEXT:    pwsla.bs a0, a0, a1
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: pwsla_v4i8:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pwcvtu.wb a0, a0
+; RV64-NEXT:    psext.h.b a0, a0
+; RV64-NEXT:    psll.hs a0, a0, a1
+; RV64-NEXT:    ret
+  %ext = sext <4 x i8> %x to <4 x i16>
+  %splat.ins = insertelement <4 x i16> poison, i16 %shamt, i64 0
+  %splat = shufflevector <4 x i16> %splat.ins, <4 x i16> poison, <4 x i32> 
zeroinitializer
+  %res = shl <4 x i16> %ext, %splat
+  ret <4 x i16> %res
+}
+
+define <2 x i32> @pwsla_v2i16(<2 x i16> %x, i32 %shamt) {
+; RV32-LABEL: pwsla_v2i16:
+; RV32:       # %bb.0:
+; RV32-NEXT:    pwsla.hs a0, a0, a1
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: pwsla_v2i16:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pwcvtu.wh a0, a0
+; RV64-NEXT:    psext.w.h a0, a0
+; RV64-NEXT:    psll.ws a0, a0, a1
+; RV64-NEXT:    ret
+  %ext = sext <2 x i16> %x to <2 x i32>
+  %splat.ins = insertelement <2 x i32> poison, i32 %shamt, i64 0
+  %splat = shufflevector <2 x i32> %splat.ins, <2 x i32> poison, <2 x i32> 
zeroinitializer
+  %res = shl <2 x i32> %ext, %splat
+  ret <2 x i32> %res
+}
+
+define <4 x i16> @pwslli_v4i8(<4 x i8> %x) {
+; RV32-LABEL: pwslli_v4i8:
+; RV32:       # %bb.0:
+; RV32-NEXT:    pwslli.b a0, a0, 3
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: pwslli_v4i8:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pwcvtu.wb a0, a0
+; RV64-NEXT:    pslli.h a0, a0, 3
+; RV64-NEXT:    ret
+  %ext = zext <4 x i8> %x to <4 x i16>
+  %res = shl <4 x i16> %ext, <i16 3, i16 3, i16 3, i16 3>
+  ret <4 x i16> %res
+}
+
+define <2 x i32> @pwslli_v2i16(<2 x i16> %x) {
+; RV32-LABEL: pwslli_v2i16:
+; RV32:       # %bb.0:
+; RV32-NEXT:    pwslli.h a0, a0, 7
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: pwslli_v2i16:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pwcvtu.wh a0, a0
+; RV64-NEXT:    pslli.w a0, a0, 7
+; RV64-NEXT:    ret
+  %ext = zext <2 x i16> %x to <2 x i32>
+  %res = shl <2 x i32> %ext, <i32 7, i32 7>
+  ret <2 x i32> %res
+}
+
+define <4 x i16> @pwslai_v4i8(<4 x i8> %x) {
+; RV32-LABEL: pwslai_v4i8:
+; RV32:       # %bb.0:
+; RV32-NEXT:    pwslai.b a0, a0, 3
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: pwslai_v4i8:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pwcvtu.wb a0, a0
+; RV64-NEXT:    psext.h.b a0, a0
+; RV64-NEXT:    pslli.h a0, a0, 3
+; RV64-NEXT:    ret
+  %ext = sext <4 x i8> %x to <4 x i16>
+  %res = shl <4 x i16> %ext, <i16 3, i16 3, i16 3, i16 3>
+  ret <4 x i16> %res
+}
+
+define <2 x i32> @pwslai_v2i16(<2 x i16> %x) {
+; RV32-LABEL: pwslai_v2i16:
+; RV32:       # %bb.0:
+; RV32-NEXT:    pwslai.h a0, a0, 7
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: pwslai_v2i16:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pwcvtu.wh a0, a0
+; RV64-NEXT:    psext.w.h a0, a0
+; RV64-NEXT:    pslli.w a0, a0, 7
+; RV64-NEXT:    ret
+  %ext = sext <2 x i16> %x to <2 x i32>
+  %res = shl <2 x i32> %ext, <i32 7, i32 7>
+  ret <2 x i32> %res
+}
+;; NOTE: These prefixes are unused and the list is autogenerated. Do not add 
tests below this line:
+; CHECK: {{.*}}

From 3809d0a966eb8199ae2efee5e1ac0b2fb111257a Mon Sep 17 00:00:00 2001
From: Michael-Chen-NJU <[email protected]>
Date: Tue, 22 Sep 2026 11:59:35 +0800
Subject: [PATCH 2/3] [Clang][RISCV] Use intrinsics for packed widening shifts

---
 clang/include/clang/Basic/BuiltinsRISCV.td    |  6 ++
 clang/lib/CodeGen/TargetBuiltins/RISCV.cpp    | 12 +++
 clang/lib/Headers/riscv_packed_simd.h         | 18 ++--
 clang/test/CodeGen/RISCV/rvp-intrinsics.c     | 88 ++++++------------
 .../riscv_packed_simd.c                       | 24 ++---
 llvm/include/llvm/IR/IntrinsicsRISCV.td       |  8 ++
 llvm/lib/Target/RISCV/RISCVISelLowering.cpp   | 43 +++++----
 llvm/test/CodeGen/RISCV/rvp-widening-shift.ll | 89 ++++++++++++++-----
 8 files changed, 159 insertions(+), 129 deletions(-)

diff --git a/clang/include/clang/Basic/BuiltinsRISCV.td 
b/clang/include/clang/Basic/BuiltinsRISCV.td
index ee840e45a65ba..2382f87ba0242 100644
--- a/clang/include/clang/Basic/BuiltinsRISCV.td
+++ b/clang/include/clang/Basic/BuiltinsRISCV.td
@@ -465,6 +465,12 @@ def psext_h_i32x2 : RISCVBuiltin<"_Vector<2, 
int>(_Vector<2, int>)">;
 def pzext_b_u16x4 : RISCVBuiltin<"_Vector<4, unsigned short>(_Vector<4, 
unsigned short>)">;
 def pzext_h_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned 
int>)">;
 
+// Packed Widening Shifts
+def pwsll_s_u16x4 : RISCVBuiltin<"_Vector<4, unsigned short>(_Vector<4, 
unsigned char>, unsigned int)">;
+def pwsll_s_u32x2 : RISCVBuiltin<"_Vector<2, unsigned int>(_Vector<2, unsigned 
short>, unsigned int)">;
+def pwsla_s_i16x4 : RISCVBuiltin<"_Vector<4, short>(_Vector<4, signed char>, 
unsigned int)">;
+def pwsla_s_i32x2 : RISCVBuiltin<"_Vector<2, int>(_Vector<2, short>, unsigned 
int)">;
+
 
 // Packed Narrowing Clip Pair (32-bit)
 def pnclipp_i8x4   : RISCVBuiltin<"_Vector<4, signed char>(_Vector<2, short>, 
_Vector<2, short>)">;
diff --git a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp 
b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
index f99a05ce673aa..c1cead61a0108 100644
--- a/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
+++ b/clang/lib/CodeGen/TargetBuiltins/RISCV.cpp
@@ -1199,6 +1199,18 @@ Value *CodeGenFunction::EmitRISCVBuiltinExpr(unsigned 
BuiltinID,
     break;
   }
 
+  // Packed Widening Shifts
+  case RISCV::BI__builtin_riscv_pwsll_s_u16x4:
+  case RISCV::BI__builtin_riscv_pwsll_s_u32x2:
+    ID = Intrinsic::riscv_pwsll;
+    IntrinsicTypes = {ResultType, Ops[0]->getType()};
+    break;
+  case RISCV::BI__builtin_riscv_pwsla_s_i16x4:
+  case RISCV::BI__builtin_riscv_pwsla_s_i32x2:
+    ID = Intrinsic::riscv_pwsla;
+    IntrinsicTypes = {ResultType, Ops[0]->getType()};
+    break;
+
   // Packed Averaging Addition and Subtraction
   case RISCV::BI__builtin_riscv_paadd_i8x4:
   case RISCV::BI__builtin_riscv_paadd_i16x2:
diff --git a/clang/lib/Headers/riscv_packed_simd.h 
b/clang/lib/Headers/riscv_packed_simd.h
index 96413a68919f3..aafad863c7efd 100644
--- a/clang/lib/Headers/riscv_packed_simd.h
+++ b/clang/lib/Headers/riscv_packed_simd.h
@@ -175,11 +175,6 @@ typedef uint32_t uint32x2_t 
__attribute__((__vector_size__(8)));
   static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1) {          
\
     return __builtin_convertvector(__rs1, rty);                                
\
   }
-#define __packed_widen_shift(name, rty, ty, mask)                              
\
-  static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1,            
\
-                                                          unsigned __shamt) {  
\
-    return __builtin_convertvector(__rs1, rty) << (__shamt & (mask));          
\
-  }
 #define __packed_widen_binary_op(name, rty, ty, op)                            
\
   static __inline__ rty __DEFAULT_FN_ATTRS __riscv_##name(ty __rs1,            
\
                                                           ty __rs2) {          
\
@@ -736,10 +731,14 @@ __packed_widen_high2(pwcvth_i32x2, int32x2_t, int16x2_t)
 __packed_widen_high2(pwcvth_u32x2, uint32x2_t, uint16x2_t)
 
 /* Packed Widening Shift */
-__packed_widen_shift(pwsll_s_u16x4, uint16x4_t, uint8x4_t, 0xf)
-__packed_widen_shift(pwsll_s_u32x2, uint32x2_t, uint16x2_t, 0x1f)
-__packed_widen_shift(pwsla_s_i16x4, int16x4_t, int8x4_t, 0xf)
-__packed_widen_shift(pwsla_s_i32x2, int32x2_t, int16x2_t, 0x1f)
+__packed_binary_builtin_mixed(pwsll_s_u16x4, uint16x4_t, uint8x4_t, unsigned,
+                              __builtin_riscv_pwsll_s_u16x4)
+__packed_binary_builtin_mixed(pwsll_s_u32x2, uint32x2_t, uint16x2_t, unsigned,
+                              __builtin_riscv_pwsll_s_u32x2)
+__packed_binary_builtin_mixed(pwsla_s_i16x4, int16x4_t, int8x4_t, unsigned,
+                              __builtin_riscv_pwsla_s_i16x4)
+__packed_binary_builtin_mixed(pwsla_s_i32x2, int32x2_t, int16x2_t, unsigned,
+                              __builtin_riscv_pwsla_s_i32x2)
 
 /* Packed Widening Addition and Subtraction */
 __packed_widen_binary_op(pwadd_i16x4, int16x4_t, int8x4_t, +)
@@ -1325,7 +1324,6 @@ __packed_reinterpret(u32x2_i32x2, int32x2_t, uint32x2_t)
 #undef __packed_merge_builtin
 #undef __packed_unary_builtin
 #undef __packed_widen_convert
-#undef __packed_widen_shift
 #undef __packed_widen_binary_op
 #undef __packed_widen_binary_acc_op
 #undef __packed_widen_mul
diff --git a/clang/test/CodeGen/RISCV/rvp-intrinsics.c 
b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
index 0946f4885d938..58faf9eaf486b 100644
--- a/clang/test/CodeGen/RISCV/rvp-intrinsics.c
+++ b/clang/test/CodeGen/RISCV/rvp-intrinsics.c
@@ -137,7 +137,7 @@ uint64_t test_abs_u64(int64_t a) {
 /* Packed Splat (32-bit) */
 
 // RV32-LABEL: define dso_local i32 @test_pmv_s_u8x4(
-// RV32-SAME: i8 noundef zeroext [[X:%.*]]) #[[ATTR0:[0-9]+]] {
+// RV32-SAME: i8 noundef zeroext [[X:%.*]]) #[[ATTR0]] {
 // RV32-NEXT:  [[ENTRY:.*:]]
 // RV32-NEXT:    [[VECINIT_I:%.*]] = insertelement <4 x i8> poison, i8 [[X]], 
i64 0
 // RV32-NEXT:    [[VECINIT3_I:%.*]] = shufflevector <4 x i8> [[VECINIT_I]], <4 
x i8> poison, <4 x i32> zeroinitializer
@@ -145,7 +145,7 @@ uint64_t test_abs_u64(int64_t a) {
 // RV32-NEXT:    ret i32 [[TMP0]]
 //
 // RV64-LABEL: define dso_local i32 @test_pmv_s_u8x4(
-// RV64-SAME: i8 noundef zeroext [[X:%.*]]) #[[ATTR0:[0-9]+]] {
+// RV64-SAME: i8 noundef zeroext [[X:%.*]]) #[[ATTR0]] {
 // RV64-NEXT:  [[ENTRY:.*:]]
 // RV64-NEXT:    [[VECINIT_I:%.*]] = insertelement <4 x i8> poison, i8 [[X]], 
i64 0
 // RV64-NEXT:    [[VECINIT3_I:%.*]] = shufflevector <4 x i8> [[VECINIT_I]], <4 
x i8> poison, <4 x i32> zeroinitializer
@@ -8162,27 +8162,17 @@ uint32x2_t test_pwcvtu_u32x2(uint16x2_t rs1) {
 // RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) 
#[[ATTR0]] {
 // RV32-NEXT:  [[ENTRY:.*:]]
 // RV32-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8>
-// RV32-NEXT:    [[CONV_I:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i16>
-// RV32-NEXT:    [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16
-// RV32-NEXT:    [[TMP2:%.*]] = and i16 [[TMP1]], 15
-// RV32-NEXT:    [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], 
i64 0
-// RV32-NEXT:    [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x 
i16> poison, <4 x i32> zeroinitializer
-// RV32-NEXT:    [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]]
-// RV32-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64
-// RV32-NEXT:    ret i64 [[TMP4]]
+// RV32-NEXT:    [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.pwsll.v4i16.v4i8(<4 
x i8> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT:    [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64
+// RV32-NEXT:    ret i64 [[TMP2]]
 //
 // RV64-LABEL: define dso_local i64 @test_pwsll_s_u16x4(
 // RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext 
[[SHAMT:%.*]]) #[[ATTR0]] {
 // RV64-NEXT:  [[ENTRY:.*:]]
 // RV64-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8>
-// RV64-NEXT:    [[CONV_I:%.*]] = zext <4 x i8> [[TMP0]] to <4 x i16>
-// RV64-NEXT:    [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16
-// RV64-NEXT:    [[TMP2:%.*]] = and i16 [[TMP1]], 15
-// RV64-NEXT:    [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], 
i64 0
-// RV64-NEXT:    [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x 
i16> poison, <4 x i32> zeroinitializer
-// RV64-NEXT:    [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]]
-// RV64-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64
-// RV64-NEXT:    ret i64 [[TMP4]]
+// RV64-NEXT:    [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.pwsll.v4i16.v4i8(<4 
x i8> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT:    [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64
+// RV64-NEXT:    ret i64 [[TMP2]]
 //
 uint16x4_t test_pwsll_s_u16x4(uint8x4_t rs1, unsigned shamt) {
   return __riscv_pwsll_s_u16x4(rs1, shamt);
@@ -8192,25 +8182,17 @@ uint16x4_t test_pwsll_s_u16x4(uint8x4_t rs1, unsigned 
shamt) {
 // RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) 
#[[ATTR0]] {
 // RV32-NEXT:  [[ENTRY:.*:]]
 // RV32-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
-// RV32-NEXT:    [[CONV_I:%.*]] = zext <2 x i16> [[TMP0]] to <2 x i32>
-// RV32-NEXT:    [[AND_I:%.*]] = and i32 [[SHAMT]], 31
-// RV32-NEXT:    [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, 
i32 [[AND_I]], i64 0
-// RV32-NEXT:    [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> 
[[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer
-// RV32-NEXT:    [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]]
-// RV32-NEXT:    [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64
-// RV32-NEXT:    ret i64 [[TMP1]]
+// RV32-NEXT:    [[TMP1:%.*]] = call <2 x i32> 
@llvm.riscv.pwsll.v2i32.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT:    [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64
+// RV32-NEXT:    ret i64 [[TMP2]]
 //
 // RV64-LABEL: define dso_local i64 @test_pwsll_s_u32x2(
 // RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext 
[[SHAMT:%.*]]) #[[ATTR0]] {
 // RV64-NEXT:  [[ENTRY:.*:]]
 // RV64-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
-// RV64-NEXT:    [[CONV_I:%.*]] = zext <2 x i16> [[TMP0]] to <2 x i32>
-// RV64-NEXT:    [[AND_I:%.*]] = and i32 [[SHAMT]], 31
-// RV64-NEXT:    [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, 
i32 [[AND_I]], i64 0
-// RV64-NEXT:    [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> 
[[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer
-// RV64-NEXT:    [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]]
-// RV64-NEXT:    [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64
-// RV64-NEXT:    ret i64 [[TMP1]]
+// RV64-NEXT:    [[TMP1:%.*]] = call <2 x i32> 
@llvm.riscv.pwsll.v2i32.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT:    [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64
+// RV64-NEXT:    ret i64 [[TMP2]]
 //
 uint32x2_t test_pwsll_s_u32x2(uint16x2_t rs1, unsigned shamt) {
   return __riscv_pwsll_s_u32x2(rs1, shamt);
@@ -8220,27 +8202,17 @@ uint32x2_t test_pwsll_s_u32x2(uint16x2_t rs1, unsigned 
shamt) {
 // RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) 
#[[ATTR0]] {
 // RV32-NEXT:  [[ENTRY:.*:]]
 // RV32-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8>
-// RV32-NEXT:    [[CONV_I:%.*]] = sext <4 x i8> [[TMP0]] to <4 x i16>
-// RV32-NEXT:    [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16
-// RV32-NEXT:    [[TMP2:%.*]] = and i16 [[TMP1]], 15
-// RV32-NEXT:    [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], 
i64 0
-// RV32-NEXT:    [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x 
i16> poison, <4 x i32> zeroinitializer
-// RV32-NEXT:    [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]]
-// RV32-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64
-// RV32-NEXT:    ret i64 [[TMP4]]
+// RV32-NEXT:    [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.pwsla.v4i16.v4i8(<4 
x i8> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT:    [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64
+// RV32-NEXT:    ret i64 [[TMP2]]
 //
 // RV64-LABEL: define dso_local i64 @test_pwsla_s_i16x4(
 // RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext 
[[SHAMT:%.*]]) #[[ATTR0]] {
 // RV64-NEXT:  [[ENTRY:.*:]]
 // RV64-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <4 x i8>
-// RV64-NEXT:    [[CONV_I:%.*]] = sext <4 x i8> [[TMP0]] to <4 x i16>
-// RV64-NEXT:    [[TMP1:%.*]] = trunc i32 [[SHAMT]] to i16
-// RV64-NEXT:    [[TMP2:%.*]] = and i16 [[TMP1]], 15
-// RV64-NEXT:    [[TMP3:%.*]] = insertelement <4 x i16> poison, i16 [[TMP2]], 
i64 0
-// RV64-NEXT:    [[SH_PROM_I:%.*]] = shufflevector <4 x i16> [[TMP3]], <4 x 
i16> poison, <4 x i32> zeroinitializer
-// RV64-NEXT:    [[SHL_I:%.*]] = shl <4 x i16> [[CONV_I]], [[SH_PROM_I]]
-// RV64-NEXT:    [[TMP4:%.*]] = bitcast <4 x i16> [[SHL_I]] to i64
-// RV64-NEXT:    ret i64 [[TMP4]]
+// RV64-NEXT:    [[TMP1:%.*]] = call <4 x i16> @llvm.riscv.pwsla.v4i16.v4i8(<4 
x i8> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT:    [[TMP2:%.*]] = bitcast <4 x i16> [[TMP1]] to i64
+// RV64-NEXT:    ret i64 [[TMP2]]
 //
 int16x4_t test_pwsla_s_i16x4(int8x4_t rs1, unsigned shamt) {
   return __riscv_pwsla_s_i16x4(rs1, shamt);
@@ -8250,25 +8222,17 @@ int16x4_t test_pwsla_s_i16x4(int8x4_t rs1, unsigned 
shamt) {
 // RV32-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef [[SHAMT:%.*]]) 
#[[ATTR0]] {
 // RV32-NEXT:  [[ENTRY:.*:]]
 // RV32-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
-// RV32-NEXT:    [[CONV_I:%.*]] = sext <2 x i16> [[TMP0]] to <2 x i32>
-// RV32-NEXT:    [[AND_I:%.*]] = and i32 [[SHAMT]], 31
-// RV32-NEXT:    [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, 
i32 [[AND_I]], i64 0
-// RV32-NEXT:    [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> 
[[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer
-// RV32-NEXT:    [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]]
-// RV32-NEXT:    [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64
-// RV32-NEXT:    ret i64 [[TMP1]]
+// RV32-NEXT:    [[TMP1:%.*]] = call <2 x i32> 
@llvm.riscv.pwsla.v2i32.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV32-NEXT:    [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64
+// RV32-NEXT:    ret i64 [[TMP2]]
 //
 // RV64-LABEL: define dso_local i64 @test_pwsla_s_i32x2(
 // RV64-SAME: i32 noundef [[RS1_COERCE:%.*]], i32 noundef signext 
[[SHAMT:%.*]]) #[[ATTR0]] {
 // RV64-NEXT:  [[ENTRY:.*:]]
 // RV64-NEXT:    [[TMP0:%.*]] = bitcast i32 [[RS1_COERCE]] to <2 x i16>
-// RV64-NEXT:    [[CONV_I:%.*]] = sext <2 x i16> [[TMP0]] to <2 x i32>
-// RV64-NEXT:    [[AND_I:%.*]] = and i32 [[SHAMT]], 31
-// RV64-NEXT:    [[SPLAT_SPLATINSERT_I:%.*]] = insertelement <2 x i32> poison, 
i32 [[AND_I]], i64 0
-// RV64-NEXT:    [[SPLAT_SPLAT_I:%.*]] = shufflevector <2 x i32> 
[[SPLAT_SPLATINSERT_I]], <2 x i32> poison, <2 x i32> zeroinitializer
-// RV64-NEXT:    [[SHL_I:%.*]] = shl <2 x i32> [[CONV_I]], [[SPLAT_SPLAT_I]]
-// RV64-NEXT:    [[TMP1:%.*]] = bitcast <2 x i32> [[SHL_I]] to i64
-// RV64-NEXT:    ret i64 [[TMP1]]
+// RV64-NEXT:    [[TMP1:%.*]] = call <2 x i32> 
@llvm.riscv.pwsla.v2i32.v2i16(<2 x i16> [[TMP0]], i32 [[SHAMT]])
+// RV64-NEXT:    [[TMP2:%.*]] = bitcast <2 x i32> [[TMP1]] to i64
+// RV64-NEXT:    ret i64 [[TMP2]]
 //
 int32x2_t test_pwsla_s_i32x2(int16x2_t rs1, unsigned shamt) {
   return __riscv_pwsla_s_i32x2(rs1, shamt);
diff --git a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c 
b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
index 910e1b8b85500..91ac4557204ff 100644
--- a/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
+++ b/cross-project-tests/intrinsic-header-tests/riscv_packed_simd.c
@@ -2385,21 +2385,25 @@ int32x2_t test_pwsla_s_i32x2_imm(int16x2_t rs1) {
   return __riscv_pwsla_s_i32x2(rs1, 7);
 }
 
-// Verify that an out-of-range constant shift amount is masked to the maximum
-// in-range value by the header implementation.
-// CHECK-LABEL: test_pwsll_s_u16x4_masked_imm:
-// RV32:        pslli.dh{{[[:space:]]}}a0, a0, 15
+// The intrinsic uses the low 5 bits of the register-form instruction. Values
+// that do not fit an immediate form must retain that register-form semantics.
+// CHECK-LABEL: test_pwsll_s_u16x4_low5:
+// RV32:        li{{[[:space:]]}}a1, 31
+// RV32-NEXT:   pwsll.bs{{[[:space:]]}}a0, a0, a1
+// RV64:        li{{[[:space:]]}}a1, 31
 // RV64:        pwcvtu.wb
-// RV64:        pslli.h{{[[:space:]]}}a0, a0, 15
-uint16x4_t test_pwsll_s_u16x4_masked_imm(uint8x4_t rs1) {
+// RV64:        psll.hs{{[[:space:]]}}a0, a0, a1
+uint16x4_t test_pwsll_s_u16x4_low5(uint8x4_t rs1) {
   return __riscv_pwsll_s_u16x4(rs1, 31);
 }
 
-// CHECK-LABEL: test_pwsll_s_u32x2_masked_imm:
-// RV32:        pslli.dw{{[[:space:]]}}a0, a0, 31
+// CHECK-LABEL: test_pwsll_s_u32x2_low5:
+// RV32:        li{{[[:space:]]}}a1, 63
+// RV32-NEXT:   pwsll.hs{{[[:space:]]}}a0, a0, a1
+// RV64:        li{{[[:space:]]}}a1, 63
 // RV64:        pwcvtu.wh
-// RV64:        pslli.w{{[[:space:]]}}a0, a0, 31
-uint32x2_t test_pwsll_s_u32x2_masked_imm(uint16x2_t rs1) {
+// RV64:        psll.ws{{[[:space:]]}}a0, a0, a1
+uint32x2_t test_pwsll_s_u32x2_low5(uint16x2_t rs1) {
   return __riscv_pwsll_s_u32x2(rs1, 63);
 }
 
diff --git a/llvm/include/llvm/IR/IntrinsicsRISCV.td 
b/llvm/include/llvm/IR/IntrinsicsRISCV.td
index 09399b0ea3f36..0236d614377dc 100644
--- a/llvm/include/llvm/IR/IntrinsicsRISCV.td
+++ b/llvm/include/llvm/IR/IntrinsicsRISCV.td
@@ -2076,6 +2076,14 @@ class RVPBinaryIntrinsic
   def int_riscv_psshl  : RVPShiftIntrinsic;
   def int_riscv_psshlr : RVPShiftIntrinsic;
 
+  // Packed Widening Shifts.
+  class RVPWideningShiftIntrinsic
+      : DefaultAttrsIntrinsic<[llvm_anyvector_ty],
+                              [llvm_anyvector_ty, llvm_i32_ty],
+                              [IntrNoMem, IntrSpeculatable]>;
+  def int_riscv_pwsll : RVPWideningShiftIntrinsic;
+  def int_riscv_pwsla : RVPWideningShiftIntrinsic;
+
   // Packed Exchanged Addition and Subtraction.
   def int_riscv_pas  : RVPBinaryIntrinsic;
   def int_riscv_psa  : RVPBinaryIntrinsic;
diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp 
b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 03966db5d6021..4a6f08f101c27 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -9545,29 +9545,6 @@ SDValue RISCVTargetLowering::LowerOperation(SDValue Op,
         if (!SplatVal)
           return SDValue();
 
-        // The 32-bit packed widening shift intrinsics produce extend followed
-        // by a scalar-splat shift. Preserve that shape as a widening shift
-        // before generic packed-shift lowering loses the narrow source.
-        if (!Subtarget.is64Bit() && Op.getOpcode() == ISD::SHL) {
-          using namespace SDPatternMatch;
-          MVT VT = Op.getSimpleValueType();
-          if (VT == MVT::v4i16 || VT == MVT::v2i32) {
-            MVT SrcVT = VT == MVT::v4i16 ? MVT::v4i8 : MVT::v2i16;
-            SDValue Src;
-            unsigned ExtendOpcode = Op.getOperand(0).getOpcode();
-            if ((ExtendOpcode == ISD::SIGN_EXTEND ||
-                 ExtendOpcode == ISD::ZERO_EXTEND) &&
-                sd_match(Op.getOperand(0),
-                         m_OneUse(m_Node(ExtendOpcode,
-                                         m_Value(Src, m_SpecificVT(SrcVT)))))) 
{
-              unsigned Opc = ExtendOpcode == ISD::SIGN_EXTEND ? RISCVISD::PWSLA
-                                                              : 
RISCVISD::PWSLL;
-              SplatVal = DAG.getZExtOrTrunc(SplatVal, SDLoc(Op), MVT::i32);
-              return DAG.getNode(Opc, SDLoc(Op), VT, Src, SplatVal);
-            }
-          }
-        }
-
         unsigned Opc;
         switch (Op.getOpcode()) {
         default:
@@ -13112,6 +13089,26 @@ SDValue 
RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
     return DAG.getNode(getRVPShiftOpcode(IntNo), DL, Op.getValueType(),
                        Op.getOperand(1), ShAmt);
   }
+  case Intrinsic::riscv_pwsll:
+  case Intrinsic::riscv_pwsla: {
+    MVT VT = Op.getSimpleValueType();
+    SDValue Src = Op.getOperand(1);
+    MVT SrcVT = Src.getSimpleValueType();
+    if (!((VT == MVT::v4i16 && SrcVT == MVT::v4i8) ||
+          (VT == MVT::v2i32 && SrcVT == MVT::v2i16)))
+      reportFatalUsageError("unsupported packed widening shift intrinsic");
+
+    SDValue ShAmt = DAG.getAnyExtOrTrunc(Op.getOperand(2), DL, XLenVT);
+    bool IsSigned = IntNo == Intrinsic::riscv_pwsla;
+    if (!Subtarget.is64Bit()) {
+      unsigned Opc = IsSigned ? RISCVISD::PWSLA : RISCVISD::PWSLL;
+      return DAG.getNode(Opc, DL, VT, Src, ShAmt);
+    }
+
+    unsigned ExtOpc = IsSigned ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND;
+    SDValue Wide = DAG.getNode(ExtOpc, DL, VT, Src);
+    return DAG.getNode(RISCVISD::PSHL, DL, VT, Wide, ShAmt);
+  }
   case Intrinsic::riscv_psext_b:
   case Intrinsic::riscv_psext_h: {
     EVT VT = Op.getValueType();
diff --git a/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll 
b/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll
index ec6a0e0032c87..a95ce8fabbadb 100644
--- a/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll
+++ b/llvm/test/CodeGen/RISCV/rvp-widening-shift.ll
@@ -4,6 +4,11 @@
 ; RUN: llc -mtriple=riscv64 -mattr=+experimental-p -verify-machineinstrs < %s 
| \
 ; RUN:   FileCheck %s --check-prefixes=CHECK,RV64
 
+declare <4 x i16> @llvm.riscv.pwsll.v4i16.v4i8(<4 x i8>, i32)
+declare <2 x i32> @llvm.riscv.pwsll.v2i32.v2i16(<2 x i16>, i32)
+declare <4 x i16> @llvm.riscv.pwsla.v4i16.v4i8(<4 x i8>, i32)
+declare <2 x i32> @llvm.riscv.pwsla.v2i32.v2i16(<2 x i16>, i32)
+
 define <4 x i16> @pwsll_v4i8(<4 x i8> %x, i16 %shamt) {
 ; RV32-LABEL: pwsll_v4i8:
 ; RV32:       # %bb.0:
@@ -15,10 +20,8 @@ define <4 x i16> @pwsll_v4i8(<4 x i8> %x, i16 %shamt) {
 ; RV64-NEXT:    pwcvtu.wb a0, a0
 ; RV64-NEXT:    psll.hs a0, a0, a1
 ; RV64-NEXT:    ret
-  %ext = zext <4 x i8> %x to <4 x i16>
-  %splat.ins = insertelement <4 x i16> poison, i16 %shamt, i64 0
-  %splat = shufflevector <4 x i16> %splat.ins, <4 x i16> poison, <4 x i32> 
zeroinitializer
-  %res = shl <4 x i16> %ext, %splat
+  %shamt.ext = zext i16 %shamt to i32
+  %res = call <4 x i16> @llvm.riscv.pwsll.v4i16.v4i8(<4 x i8> %x, i32 
%shamt.ext)
   ret <4 x i16> %res
 }
 
@@ -33,10 +36,7 @@ define <2 x i32> @pwsll_v2i16(<2 x i16> %x, i32 %shamt) {
 ; RV64-NEXT:    pwcvtu.wh a0, a0
 ; RV64-NEXT:    psll.ws a0, a0, a1
 ; RV64-NEXT:    ret
-  %ext = zext <2 x i16> %x to <2 x i32>
-  %splat.ins = insertelement <2 x i32> poison, i32 %shamt, i64 0
-  %splat = shufflevector <2 x i32> %splat.ins, <2 x i32> poison, <2 x i32> 
zeroinitializer
-  %res = shl <2 x i32> %ext, %splat
+  %res = call <2 x i32> @llvm.riscv.pwsll.v2i32.v2i16(<2 x i16> %x, i32 %shamt)
   ret <2 x i32> %res
 }
 
@@ -52,10 +52,8 @@ define <4 x i16> @pwsla_v4i8(<4 x i8> %x, i16 %shamt) {
 ; RV64-NEXT:    psext.h.b a0, a0
 ; RV64-NEXT:    psll.hs a0, a0, a1
 ; RV64-NEXT:    ret
-  %ext = sext <4 x i8> %x to <4 x i16>
-  %splat.ins = insertelement <4 x i16> poison, i16 %shamt, i64 0
-  %splat = shufflevector <4 x i16> %splat.ins, <4 x i16> poison, <4 x i32> 
zeroinitializer
-  %res = shl <4 x i16> %ext, %splat
+  %shamt.ext = zext i16 %shamt to i32
+  %res = call <4 x i16> @llvm.riscv.pwsla.v4i16.v4i8(<4 x i8> %x, i32 
%shamt.ext)
   ret <4 x i16> %res
 }
 
@@ -71,10 +69,7 @@ define <2 x i32> @pwsla_v2i16(<2 x i16> %x, i32 %shamt) {
 ; RV64-NEXT:    psext.w.h a0, a0
 ; RV64-NEXT:    psll.ws a0, a0, a1
 ; RV64-NEXT:    ret
-  %ext = sext <2 x i16> %x to <2 x i32>
-  %splat.ins = insertelement <2 x i32> poison, i32 %shamt, i64 0
-  %splat = shufflevector <2 x i32> %splat.ins, <2 x i32> poison, <2 x i32> 
zeroinitializer
-  %res = shl <2 x i32> %ext, %splat
+  %res = call <2 x i32> @llvm.riscv.pwsla.v2i32.v2i16(<2 x i16> %x, i32 %shamt)
   ret <2 x i32> %res
 }
 
@@ -89,8 +84,7 @@ define <4 x i16> @pwslli_v4i8(<4 x i8> %x) {
 ; RV64-NEXT:    pwcvtu.wb a0, a0
 ; RV64-NEXT:    pslli.h a0, a0, 3
 ; RV64-NEXT:    ret
-  %ext = zext <4 x i8> %x to <4 x i16>
-  %res = shl <4 x i16> %ext, <i16 3, i16 3, i16 3, i16 3>
+  %res = call <4 x i16> @llvm.riscv.pwsll.v4i16.v4i8(<4 x i8> %x, i32 3)
   ret <4 x i16> %res
 }
 
@@ -105,8 +99,7 @@ define <2 x i32> @pwslli_v2i16(<2 x i16> %x) {
 ; RV64-NEXT:    pwcvtu.wh a0, a0
 ; RV64-NEXT:    pslli.w a0, a0, 7
 ; RV64-NEXT:    ret
-  %ext = zext <2 x i16> %x to <2 x i32>
-  %res = shl <2 x i32> %ext, <i32 7, i32 7>
+  %res = call <2 x i32> @llvm.riscv.pwsll.v2i32.v2i16(<2 x i16> %x, i32 7)
   ret <2 x i32> %res
 }
 
@@ -122,8 +115,7 @@ define <4 x i16> @pwslai_v4i8(<4 x i8> %x) {
 ; RV64-NEXT:    psext.h.b a0, a0
 ; RV64-NEXT:    pslli.h a0, a0, 3
 ; RV64-NEXT:    ret
-  %ext = sext <4 x i8> %x to <4 x i16>
-  %res = shl <4 x i16> %ext, <i16 3, i16 3, i16 3, i16 3>
+  %res = call <4 x i16> @llvm.riscv.pwsla.v4i16.v4i8(<4 x i8> %x, i32 3)
   ret <4 x i16> %res
 }
 
@@ -139,8 +131,57 @@ define <2 x i32> @pwslai_v2i16(<2 x i16> %x) {
 ; RV64-NEXT:    psext.w.h a0, a0
 ; RV64-NEXT:    pslli.w a0, a0, 7
 ; RV64-NEXT:    ret
-  %ext = sext <2 x i16> %x to <2 x i32>
-  %res = shl <2 x i32> %ext, <i32 7, i32 7>
+  %res = call <2 x i32> @llvm.riscv.pwsla.v2i32.v2i16(<2 x i16> %x, i32 7)
+  ret <2 x i32> %res
+}
+
+define <4 x i16> @pwslli_v4i8_31(<4 x i8> %x) {
+; RV32-LABEL: pwslli_v4i8_31:
+; RV32:       # %bb.0:
+; RV32-NEXT:    li a1, 31
+; RV32-NEXT:    pwsll.bs a0, a0, a1
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: pwslli_v4i8_31:
+; RV64:       # %bb.0:
+; RV64-NEXT:    li a1, 31
+; RV64-NEXT:    pwcvtu.wb a0, a0
+; RV64-NEXT:    psll.hs a0, a0, a1
+; RV64-NEXT:    ret
+  %res = call <4 x i16> @llvm.riscv.pwsll.v4i16.v4i8(<4 x i8> %x, i32 31)
+  ret <4 x i16> %res
+}
+
+define <4 x i16> @pwslai_v4i8_16(<4 x i8> %x) {
+; RV32-LABEL: pwslai_v4i8_16:
+; RV32:       # %bb.0:
+; RV32-NEXT:    li a1, 16
+; RV32-NEXT:    pwsla.bs a0, a0, a1
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: pwslai_v4i8_16:
+; RV64:       # %bb.0:
+; RV64-NEXT:    li a1, 16
+; RV64-NEXT:    pwcvtu.wb a0, a0
+; RV64-NEXT:    psext.h.b a0, a0
+; RV64-NEXT:    psll.hs a0, a0, a1
+; RV64-NEXT:    ret
+  %res = call <4 x i16> @llvm.riscv.pwsla.v4i16.v4i8(<4 x i8> %x, i32 16)
+  ret <4 x i16> %res
+}
+
+define <2 x i32> @pwslli_v2i16_31(<2 x i16> %x) {
+; RV32-LABEL: pwslli_v2i16_31:
+; RV32:       # %bb.0:
+; RV32-NEXT:    pwslli.h a0, a0, 31
+; RV32-NEXT:    ret
+;
+; RV64-LABEL: pwslli_v2i16_31:
+; RV64:       # %bb.0:
+; RV64-NEXT:    pwcvtu.wh a0, a0
+; RV64-NEXT:    pslli.w a0, a0, 31
+; RV64-NEXT:    ret
+  %res = call <2 x i32> @llvm.riscv.pwsll.v2i32.v2i16(<2 x i16> %x, i32 31)
   ret <2 x i32> %res
 }
 ;; NOTE: These prefixes are unused and the list is autogenerated. Do not add 
tests below this line:

From 7ae05158b861705a83d4a64e730607e1b89dce02 Mon Sep 17 00:00:00 2001
From: Michael-Chen-NJU <[email protected]>
Date: Sun, 27 Sep 2026 11:29:12 +0800
Subject: [PATCH 3/3] [RISCV] Use ANY_EXTEND for packed widening shift amount

---
 llvm/lib/Target/RISCV/RISCVISelLowering.cpp | 3 ++-
 1 file changed, 2 insertions(+), 1 deletion(-)

diff --git a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp 
b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
index 6a988b82284a8..e76c0b692964c 100644
--- a/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
+++ b/llvm/lib/Target/RISCV/RISCVISelLowering.cpp
@@ -13167,7 +13167,8 @@ SDValue 
RISCVTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
           (VT == MVT::v2i32 && SrcVT == MVT::v2i16)))
       reportFatalUsageError("unsupported packed widening shift intrinsic");
 
-    SDValue ShAmt = DAG.getAnyExtOrTrunc(Op.getOperand(2), DL, XLenVT);
+    SDValue ShAmt =
+        DAG.getNode(ISD::ANY_EXTEND, DL, XLenVT, Op.getOperand(2));
     bool IsSigned = IntNo == Intrinsic::riscv_pwsla;
     if (!Subtarget.is64Bit()) {
       unsigned Opc = IsSigned ? RISCVISD::PWSLA : RISCVISD::PWSLL;

_______________________________________________
cfe-commits mailing list
[email protected]
https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits

Reply via email to