This is an automated email from the git hooks/post-receive script.

Git pushed a commit to branch master
in repository ffmpeg.

commit a785ff4ca2a73d151d5a8b7ce5d101063fe3dc3c
Author:     Niklas Haas <[email protected]>
AuthorDate: Tue Jul 14 18:13:05 2026 +0200
Commit:     Niklas Haas <[email protected]>
CommitDate: Fri Jul 17 13:23:21 2026 +0000

    swscale/x86/ops: add AVX512 vpermb packed shuffle path
    
    We can quite easily generalize the existing AVX512 pushfb implementation to
    support vpermb when the shuffle mask is cross-lane.
    
      rgb24 1920x1080 -> rgba 1920x1080, speedup=1.372x faster
    
    This does require some finesse w.r.t. the MOVSIZE macro, to avoid reading or
    writing too many bits. In particular, for gray->rgba64le, the pattern even
    devolves into 8/64 which would trigger an illegal operand error when 
combining
    movq with zmm registers.
    
    Other than that, we can use {k1}{z} zero masking to cleanly recreate the
    semantics of pushfb's implicit zero masking, except for non-zero clear
    values where we continue doing what we already do.
    
    Lastly, while EVEX-encoded vpermb can take the source operand directly, this
    somewhat defeats the purpose of the READ size optimization; in practice it
    was better to keep the read and vpermb separate.
    
    Signed-off-by: Niklas Haas <[email protected]>
---
 libswscale/x86/ops.c          | 41 ++++++++++++++++++------------
 libswscale/x86/ops_common.asm | 58 ++++++++++++++++++++++++++++++++++---------
 2 files changed, 71 insertions(+), 28 deletions(-)

diff --git a/libswscale/x86/ops.c b/libswscale/x86/ops.c
index a55b201a16..b3eb00af65 100644
--- a/libswscale/x86/ops.c
+++ b/libswscale/x86/ops.c
@@ -489,9 +489,10 @@ SWS_DECL_FUNC(ff_sws_process3_x86);
 SWS_DECL_FUNC(ff_sws_process4_x86);
 
 /* Declare packed shuffle functions */
-SWS_FOR_STRUCT(U8, RW_SHUFFLE, DECL_ENTRY, _sse4,   NULL, NULL)
-SWS_FOR_STRUCT(U8, RW_SHUFFLE, DECL_ENTRY, _avx2,   NULL, NULL)
-SWS_FOR_STRUCT(U8, RW_SHUFFLE, DECL_ENTRY, _avx512, NULL, NULL)
+SWS_FOR_STRUCT(U8, RW_SHUFFLE, DECL_ENTRY, _sse4,       NULL, NULL)
+SWS_FOR_STRUCT(U8, RW_SHUFFLE, DECL_ENTRY, _avx2,       NULL, NULL)
+SWS_FOR_STRUCT(U8, RW_SHUFFLE, DECL_ENTRY, _avx512,     NULL, NULL)
+SWS_FOR_STRUCT(U8, RW_SHUFFLE, DECL_ENTRY, _avx512icl,  NULL, NULL)
 
 static int get_mmsize(void)
 {
@@ -508,19 +509,22 @@ static int get_mmsize(void)
 
 static int movsize(const int bytes, const int mmsize)
 {
-    return bytes <= 4 ? 4 : /* movd */
-           bytes <= 8 ? 8 : /* movq */
-           mmsize;          /* movu */
+    return bytes <= 4  ? 4  : /* movd */
+           bytes <= 8  ? 8  : /* movq */
+           bytes <= 16 ? 16 : /* xmm movu */
+           bytes <= 32 ? 32 : /* ymm movu */
+           mmsize;            /* zmm movu */
 }
 
 static int translate_shuffle(const SwsUOp *uop, int mmsize, SwsCompiledOp *out)
 {
     /* We can't shuffle across lanes, so restrict the vector size to XMM
-     * whenever the read/write size would be a subset of the full vector */
+     * whenever the read/write size would be a subset of the full vector,
+     * unless we have access to AVX-512 ICL vpermb */
     const SwsShuffleUOp *par = &uop->par.shuffle;
     const int lane_aligned = par->read_size == par->write_size &&
                              16 % par->read_size == 0;
-    if (!lane_aligned)
+    if (!lane_aligned && !EXTERNAL_AVX512ICL(av_get_cpu_flags()))
         mmsize = 16;
 
     /* Generate the shuffle mask */
@@ -547,22 +551,27 @@ static int translate_shuffle(const SwsUOp *uop, int 
mmsize, SwsCompiledOp *out)
         .block_size  = groups * uop->data.shuffle.pixels * num_lanes,
         .over_read   = { movsize(in_total,  mmsize) - in_total },
         .over_write  = { movsize(out_total, mmsize) - out_total },
-        .cpu_flags   = mmsize > 32 ? AV_CPU_FLAG_AVX512 :
-                       mmsize > 16 ? AV_CPU_FLAG_AVX2 :
-                                     AV_CPU_FLAG_SSE4,
     };
 
-#define ASSIGN_SHUFFLE_FUNC(EXT, NAME, ...)                                    
 \
+#define ASSIGN_SHUFFLE_FUNC(CPU, EXT, NAME, ...)                               
 \
 do {                                                                           
 \
     const SwsUOpEntry *entry = &uop_##NAME##EXT;                               
 \
-    if (!memcmp(&uop->par, &entry->par, sizeof(uop->par)))                     
 \
+    if (!memcmp(&uop->par, &entry->par, sizeof(uop->par))) {                   
 \
         out->func = (SwsOpFunc) entry->func;                                   
 \
+        out->cpu_flags = AV_CPU_FLAG_##CPU;                                    
 \
+    }                                                                          
 \
 } while (0);
 
     switch (mmsize) {
-    case 16: SWS_FOR(U8, RW_SHUFFLE, ASSIGN_SHUFFLE_FUNC, _sse4);   break;
-    case 32: SWS_FOR(U8, RW_SHUFFLE, ASSIGN_SHUFFLE_FUNC, _avx2);   break;
-    case 64: SWS_FOR(U8, RW_SHUFFLE, ASSIGN_SHUFFLE_FUNC, _avx512); break;
+    case 16: SWS_FOR(U8, RW_SHUFFLE, ASSIGN_SHUFFLE_FUNC, SSE4, _sse4); break;
+    case 32: SWS_FOR(U8, RW_SHUFFLE, ASSIGN_SHUFFLE_FUNC, AVX2, _avx2); break;
+    case 64:
+        if (lane_aligned) {
+            SWS_FOR(U8, RW_SHUFFLE, ASSIGN_SHUFFLE_FUNC, AVX512, _avx512);
+        } else { /* vpermb variant */
+            SWS_FOR(U8, RW_SHUFFLE, ASSIGN_SHUFFLE_FUNC, AVX512ICL, 
_avx512icl);
+        }
+        break;
     }
 
     if (!out->func) {
diff --git a/libswscale/x86/ops_common.asm b/libswscale/x86/ops_common.asm
index d54f7c6a13..e1f658b0e6 100644
--- a/libswscale/x86/ops_common.asm
+++ b/libswscale/x86/ops_common.asm
@@ -127,13 +127,31 @@ process_fn 4
 ; in which case we can also use `pshufb` on mmsize == 32/64. This is detected
 ; by the `LANE_ALIGNED` condition.
 
-%macro MOVSIZE 3 ; size, dst, src
-    %if %1 <= 4
-        movd %2, %3
-    %elif %1 <= 8
-        movq %2, %3
+%macro READ 2 ; dst, src
+    %if READ_SIZE <= 4
+        movd xmm%1, %2
+    %elif READ_SIZE <= 8
+        movq xmm%1, %2
+    %elif READ_SIZE <= 16
+        movu xmm%1, %2
+    %elif READ_SIZE <= 32
+        movu ymm%1, %2
     %else
-        movu %2, %3
+        movu zmm%1, %2
+    %endif
+%endmacro
+
+%macro WRITE 2 ; dst, src
+    %if WRITE_SIZE <= 4
+        movd %1, xmm%2
+    %elif WRITE_SIZE <= 8
+        movq %1, xmm%2
+    %elif WRITE_SIZE <= 16
+        movu %1, xmm%2
+    %elif WRITE_SIZE <= 32
+        movu %1, ymm%2
+    %else
+        movu %1, zmm%2
     %endif
 %endmacro
 
@@ -152,7 +170,7 @@ process_fn 4
 %assign WRITE_SIZE   (WRITE_SIZE * GROUPS)
 
 cglobal NAME, 6, 10, 3, exec, shuffle, bx, y, bxend, yend, src, dst, 
src_stride, dst_stride
-%if mmsize > 16 && !LANE_ALIGNED
+%if mmsize > 16 && !LANE_ALIGNED && !cpuflag(avx512icl)
             ud2 ; runtime checks should prevent this variant from being called
 %else
             mov srcq, [execq + SwsOpExec.in0]
@@ -161,9 +179,16 @@ cglobal NAME, 6, 10, 3, exec, shuffle, bx, y, bxend, yend, 
src, dst, src_stride,
             mov dst_strideq, [execq + SwsOpExec.out_stride0]
 
             ; setup shuffle mask
+    %if LANE_ALIGNED
             VBROADCASTI128 m0, [shuffleq]
-    %if cpuflag(avx512) && CLEAR_VALUE != 0
-            vpmovb2m k1, m0 ; needed for vpblendmb
+    %else
+            mova m0, [shuffleq]
+    %endif
+    %if cpuflag(avx512)
+            vpmovb2m k1, m0 ; needed for vpblendmb / vpermb
+        %if CLEAR_VALUE == 0
+            knotq k1, k1
+        %endif
     %endif
 
             ; setup clear value register if needed
@@ -173,7 +198,7 @@ cglobal NAME, 6, 10, 3, exec, shuffle, bx, y, bxend, yend, 
src, dst, src_stride,
         %else
             pcmpeqb m2, m2
         %endif
-    %elif CLEAR_VALUE != 0 ; clear-to-0 is implicitly handled by pshufb
+    %elif CLEAR_VALUE != 0 ; clear-to-0 is implicitly handled by pshufb / 
vpermb
             mov shuffled, CLEAR_VALUE * 0x1010101
             movd xm2, shuffled
             VPBROADCASTD m2, xm2
@@ -195,8 +220,14 @@ cglobal NAME, 6, 10, 3, exec, shuffle, bx, y, bxend, yend, 
src, dst, src_stride,
             sub dstq, dstidxq
 
 .loop:
-            MOVSIZE READ_SIZE, m1, [srcq + srcidxq]
+            READ m1, [srcq + srcidxq]
+    %if LANE_ALIGNED || mmsize == 16
             pshufb m1, m0
+    %elif CLEAR_VALUE == 0
+            vpermb m1{k1}{z}, m0, m1
+    %else
+            vpermb m1, m0, m1
+    %endif
 
     %if CLEAR_VALUE != 0
         %if cpuflag(avx512)
@@ -208,7 +239,7 @@ cglobal NAME, 6, 10, 3, exec, shuffle, bx, y, bxend, yend, 
src, dst, src_stride,
         %endif
     %endif
 
-            MOVSIZE WRITE_SIZE, [dstq + dstidxq], m1
+            WRITE [dstq + dstidxq], m1
             add srcidxq, READ_SIZE
     %if READ_SIZE != WRITE_SIZE
             add dstidxq, WRITE_SIZE
@@ -234,3 +265,6 @@ DECL_U8_RW_SHUFFLE (RW_SHUFFLE)
 
 INIT_ZMM avx512
 DECL_U8_RW_SHUFFLE (RW_SHUFFLE)
+
+INIT_ZMM avx512icl
+DECL_U8_RW_SHUFFLE (RW_SHUFFLE)

_______________________________________________
ffmpeg-cvslog mailing list -- [email protected]
To unsubscribe send an email to [email protected]

Reply via email to