On Tue, Jul 21, 2026 at 5:11 PM Roger Sayle <[email protected]> wrote:
>
>
> This patch is a reorganization of x86's vector initialization (vec_init)
> functionality to generate more efficient implementations in most/many
> cases.  Previously, for most (128-bit and 256-bit) vectors types,
> i386-expand.cc made use of "concat" recursion to divide-and-conquor;
> splitting each vector into upper and lower halves, initializing them,
> then concatenating the results together.  Simple and orthogonal, but
> alas inefficient.  This idiom is unable to take advantage of SSE's
> zero extension semantics, shuffle/permutation instructions, byte-level
> shifts, element insertion instructions nor vector-mode logic operations.
> Unfortunately the reality is that these ISAs are irregular, as are the
> patterns provided by the backend expose their instructions (which are
> often available in one mode but not another).
>
> The patch below recognizes/accepts these asymmetries, and provides
> "custom" vector initialization functions for most 128-bit and 256-bit
> vector modes.  There are too many optimization/improvements to list
> them all, but some examples are given below:
>
> v4si f1(int x, int y) { return (v4si){x,y,0,0}; }
>
> Before with -O2:
> f1_old: movd    %edi, %xmm0
>         movd    %esi, %xmm1
>         punpckldq       %xmm1, %xmm0
>         movq    %xmm0, %xmm0
>         ret
>
> After with -O2:
> f1_new: movd    %edi, %xmm0
>         movd    %esi, %xmm1
>         punpckldq       %xmm1, %xmm0
>         ret
>
> v4si f2(int x) { return (v4si){0,x,x,0}; }
>
> Before with -O2:
> f2_old: movd    %edi, %xmm2
>         pxor    %xmm0, %xmm0
>         movd    %edi, %xmm1
>         punpckldq       %xmm2, %xmm0
>         punpcklqdq      %xmm1, %xmm0
>         ret
>
> f2_new: movd    %edi, %xmm0
>         shufps  $65, %xmm0, %xmm0
>         ret
>
> v4si f3(int x) { return (v4si){x,1,x,2}; }
>
> Before with -O2:
> f3_old: movl    $2, %eax
>         movd    %edi, %xmm0
>         movd    %eax, %xmm2
>         movl    $1, %eax
>         movdqa  %xmm0, %xmm1
>         movd    %eax, %xmm3
>         punpckldq       %xmm2, %xmm1
>         punpckldq       %xmm3, %xmm0
>         punpcklqdq      %xmm1, %xmm0
>         ret
>
> After with -O2:
> f3_new: movd    %edi, %xmm0
>         shufps  $68, %xmm0, %xmm0
>         por     .LC0(%rip), %xmm0
>         ret
>
> v16qi f4(char x) { return (v16qi){x,0,0,0,0,0,0,0,0,x,0,0,0,0,0,0}; }
>
> Before with -O2 -mavx2:
> f4_old: vmovd   %edi, %xmm0
>         xorl    %eax, %eax
>         vpxor   %xmm1, %xmm1, %xmm1
>         vpinsrb $1, %eax, %xmm0, %xmm0
>         vpinsrb $1, %edi, %xmm1, %xmm1
>         vpmovzxwd       %xmm0, %xmm0
>         vpmovzxwd       %xmm1, %xmm1
>         vpmovzxdq       %xmm1, %xmm1
>         vpmovzxdq       %xmm0, %xmm0
>         vpunpcklqdq     %xmm1, %xmm0, %xmm0
>         ret
>
> After with -O2 -mavx2:
> f4_new: movzbl  %dil, %eax
>         vmovd   %eax, %xmm0
>         vpinsrb $9, %edi, %xmm0, %xmm0
>         ret
>
> Unfortunately, despite all of the goodness there remains one testsuite
> regression: avx512vl-concatv4si-1.c whose f2 function currently expects
> 3 instructions before the return:
>
> orig:   vmovd   (%rdi), %xmm2
>         vpinsrd $1, 4(%rdi), %xmm2, %xmm1
>         vpunpcklqdq     %xmm1, %xmm0, %xmm16
>         ret
>
> where actually an optimal implementation should require only two:
>
> ideal:  vpinsrd $2, (%rdi), %xmm0, %xmm0
>         vpinsrd $3, 4(%rdi), %xmm0, %xmm0
>         ret
>
> but unfortunately with this patch we currently (for now) generate:
>
> curr:   vmovd   %xmm0, %eax
>         vpextrd $1, %xmm0, %edx
>         vmovd   %eax, %xmm0
>         vpinsrd $1, %edx, %xmm0, %xmm0
>         vpinsrd $2, (%rdi), %xmm0, %xmm0
>         vpinsrd $3, 4(%rdi), %xmm0, %xmm0
>         vmovdqa32       %xmm0, %xmm16
>         ret
>
> which actually contains our two optimal instructions, but between
> combine, simplify-rtx and sse.md's define_insn_and_splits, we fail
> to notice that the remaining operations (converting V2SI to V4SI)
> are a no-op.  I beg the reviewers'/maintainers' indulgence to allow
> this to FAIL for the time being, to be solved in a follow-up patch.
> This current patch is large enough already, and this remaining quirk
> needs to be resolved outside the RTL expansion pass, in the later
> RTL optimizers (where it is currently a missed optimization).
>


+          else if (TARGET_SSE2)
+            {
+              rtx tmp2 = gen_reg_rtx (V2DImode);
+              emit_move_insn (tmp2, gen_lowpart (V2DImode, tmp));
+              emit_insn (gen_vec_shl_v2di (target, tmp, GEN_INT (64)));

Looks like a typo, should be tmp2 instead of tmp?

+    case E_V2DFmode:
+      if (!REG_P (var) || !MEM_P (var))
+        var = force_reg (DFmode, var);
+      x = gen_rtx_VEC_CONCAT (V2DFmode, var, CONST0_RTX (DFmode));
+      if (!one_var)
+        emit_insn (gen_rtx_SET (target, x));
+      else if (TARGET_SSE2)
+        {
+          tmp = gen_reg_rtx (V2DFmode);
+          emit_insn (gen_rtx_SET (tmp, x));
+          emit_insn (gen_vec_shl_v2df (target, tmp, GEN_INT (64)));
+        }
+      else
+        {
+          rtx tmp1 = gen_reg_rtx (V2DFmode);
+          emit_insn (gen_rtx_SET (tmp1, x));
+          rtx tmp2 = gen_reg_rtx (V4SImode);
+          emit_move_insn (tmp2, gen_lowpart (V4SImode, tmp1));
+          emit_insn (gen_sse_shufps_v4si (tmp2, tmp2, tmp2,
+                                          GEN_INT (2), GEN_INT (3),
+                                          GEN_INT (4), GEN_INT (5)));
+          emit_move_insn (target, gen_lowpart (V2DFmode, tmp2));
+        }

vec_initv2df is guarded under TARGET_SSE2, so I think the else part is
redundant.

+static int
+nonzero_int_const_count (rtx *ops, int n)
+{
+  int result = 0;
+  int i;
+  for (i = 0; i < n; i++)
+    if (CONST_INT_P (ops[i]) && ops[i] != const0_rtx)  ---- here
check CONST_INT_P (ops[i])
+      result++;
+  return result;
+}

...
+  else if (nonzero_int_const_count (ops, 4) >= 2)
+    {
+      rtx csts[4];
+      int i;
+      for (i = 0; i < 4; i++)
+        if (CONST_DOUBLE_P (ops[i]))    ---- here check
CONST_DOUBLE_P (ops[i]))
+          {
+            csts[i] = ops[i];
+            vars[i] = CONST0_RTX (SFmode);
+          }
+        else
+          {
+            csts[i] = CONST0_RTX (SFmode);
+            vars[i] = ops[i];
+          }

nonzero_int_const_count  check CONST_INT_P, but in the float cases, it
should be CONST_DOUBLE_P? it's always 0 the float helpers(v4sf/v8f)?


+    case E_V4DImode:
+      if (TARGET_AVX2 && (TARGET_64BIT || MEM_P (var)))
+        {
+          if (!REG_P (var) || !MEM_P (var))
+            var = force_reg (DImode, var);

!REG_P (var) && !MEM_P (var)?

+    case E_V4DFmode:
+      if (TARGET_AVX2)
+        {
+          if (!REG_P (var) || !MEM_P (var))
+            var = force_reg (DFmode, var);
Ditto

+      return true;
+    case E_V2DFmode:
+      if (!REG_P (var) || !MEM_P (var))
+        var = force_reg (DFmode, var);
+      x = gen_rtx_VEC_CONCAT (V2DFmode, var, CONST0_RTX (DFmode));
Ditto.

+      if (TARGET_64BIT || MEM_P (var))
+        {
+          if (!REG_P (var) || !MEM_P (var))
+            var = force_reg (DImode, var);
+          x = gen_rtx_VEC_CONCAT (V2DImode, var, CONST0_RTX (DImode));
Ditto.

+/* Helper function.  Determine if the given OPS array of size N
+   contains only zeros and one other value (possible repeated).
+   If TRUE, *VAR returns the value, PERM[i] contains 0 for for

Double *for* in the comments.

+/* Helper function.  Determine if the given OPS array of size N
+   contains only zeros and two other values (possible repeated).
+   If TRUE, VARS returns the values, PERM[i] contains 0 for for

Ditto.

Could you also add some testcases to cover your changes, it can be in
a separate commit.

-- 
BR,
Hongtao

Reply via email to