[PR] avcodec/x86/hevc: add SSSE3 pred_planar (PR #24052)

Marcos Ashton via ffmpeg-devel <[email protected]>
Newsgroups gmane.comp.video.ffmpeg.devel
Message-ID <178618662965.59.11815518138915206518@29965ddac10e>
PR #24052 opened by Marcos Ashton (MarcosAsh)
URL: https://code.ffmpeg.org/FFmpeg/FFmpeg/pulls/24052
Patch URL: https://code.ffmpeg.org/FFmpeg/FFmpeg/pulls/24052.patch

Adds an x86 SIMD implementation of HEVC planar intra prediction (previously C-only), for 8 bit, as part of #23022. x86 has no HEVC intra prediction at all today; aarch64 NEON already implements the planar, DC and angular groups at 8 bit.

Everything in the planar expression except (size-1-x)*left[y] is linear in y, so for a fixed group of eight columns the remaining terms collapse into a single running accumulator whose increment left[size]-top[x] is loop invariant. Iterating columns outermost and rows innermost keeps that accumulator and its delta in registers, so a row costs one pmullw and two paddw with no reload of top[]. For 16x16 and 32x32 two column groups run together, which shares the left[y] broadcast and lets a row be written with a single 16 byte store. Accumulation is in words throughout since the largest value the expression can take is 31*255 + 32*255 + 31*255 + 32*255 + 32 = 32162. SSSE3 is the baseline rather than SSE2 because pshufb does both the byte broadcasts and the byte to word widening, which keeps t
 he row loop short and avoids a dedicated zero register. 4x4 keeps a separate fully unrolled path that does two rows per register.

checkasm -t hevc_pred passes bit-exact. Benchmarks (Core Ultra 7 155H, medians of five runs):
```
hevc_pred_planar_4x4_8_c:        18.4
hevc_pred_planar_4x4_8_ssse3:     5.5
hevc_pred_planar_8x8_8_c:        83.1
hevc_pred_planar_8x8_8_ssse3:    16.3
hevc_pred_planar_16x16_8_c:     224.8
hevc_pred_planar_16x16_8_ssse3:  47.3
hevc_pred_planar_32x32_8_c:     629.5
hevc_pred_planar_32x32_8_ssse3: 196.5
```


>From 47d10c66b9c9b35fd5f2dd6b1289a0183488a0f2 Mon Sep 17 00:00:00 2001
From: Marcos Ashton Iglesias <[email protected]>
Date: Fri, 7 Aug 2026 20:42:41 +0100
Subject: [PATCH] avcodec/x86/hevc: add SSSE3 pred_planar

Implements planar intra prediction for 8 bit. x86 has no HEVC intra
prediction at all; aarch64 NEON already covers planar, DC and angular.

Every term except (size-1-x)*left[y] is linear in y, so for a group of
eight columns they collapse into one running accumulator whose increment
left[size]-top[x] is loop invariant. Running columns outermost keeps that
accumulator and its delta in registers, so a row costs one pmullw and two
paddw with no reload of top[]. 16x16 and 32x32 run two groups together,
sharing the left[y] broadcast and writing a row with one 16 byte store.
Words suffice throughout: the largest value is 32162. SSSE3 rather than
SSE2 since pshufb does the broadcasts and the byte to word widening.

checkasm --bench on a Core Ultra 7 155H:
hevc_pred_planar_4x4_8_c:            18.4
hevc_pred_planar_4x4_8_ssse3:         5.5 ( 3.35x)
hevc_pred_planar_8x8_8_c:            83.1
hevc_pred_planar_8x8_8_ssse3:        16.3 ( 5.10x)
hevc_pred_planar_16x16_8_c:         224.8
hevc_pred_planar_16x16_8_ssse3:      47.3 ( 4.75x)
hevc_pred_planar_32x32_8_c:         629.5
hevc_pred_planar_32x32_8_ssse3:     196.5 ( 3.20x)

Part of #23022.
---
 libavcodec/hevc/pred.c          |   3 +
 libavcodec/hevc/pred.h          |   1 +
 libavcodec/x86/hevc/Makefile    |   2 +
 libavcodec/x86/hevc/pred.asm    | 205 ++++++++++++++++++++++++++++++++
 libavcodec/x86/hevc/pred_init.c |  51 ++++++++
 5 files changed, 262 insertions(+)
 create mode 100644 libavcodec/x86/hevc/pred.asm
 create mode 100644 libavcodec/x86/hevc/pred_init.c

diff --git a/libavcodec/hevc/pred.c b/libavcodec/hevc/pred.c
index 037cbc413f..673c02d856 100644
--- a/libavcodec/hevc/pred.c
+++ b/libavcodec/hevc/pred.c
@@ -85,4 +85,7 @@ void ff_hevc_pred_init(HEVCPredContext *hpc, int bit_depth)
 #if ARCH_MIPS
     ff_hevc_pred_init_mips(hpc, bit_depth);
 #endif
+#if ARCH_X86 && HAVE_X86ASM
+    ff_hevc_pred_init_x86(hpc, bit_depth);
+#endif
 }
diff --git a/libavcodec/hevc/pred.h b/libavcodec/hevc/pred.h
index 849806fefb..9c0cf17349 100644
--- a/libavcodec/hevc/pred.h
+++ b/libavcodec/hevc/pred.h
@@ -51,6 +51,7 @@ typedef struct HEVCPredContext {
 void ff_hevc_pred_init(HEVCPredContext *hpc, int bit_depth);
 void ff_hevc_pred_init_mips(HEVCPredContext *hpc, int bit_depth);
 void ff_hevc_pred_init_aarch64(HEVCPredContext *hpc, int bit_depth);
+void ff_hevc_pred_init_x86(HEVCPredContext *hpc, int bit_depth);
 
 /* C angular prediction fallbacks (non-static for arch-specific partial override) */
 #define HEVC_PRED_ANGULAR_DECL(depth)                                         \
diff --git a/libavcodec/x86/hevc/Makefile b/libavcodec/x86/hevc/Makefile
index d09c613a19..0ac5287bdc 100644
--- a/libavcodec/x86/hevc/Makefile
+++ b/libavcodec/x86/hevc/Makefile
@@ -7,6 +7,8 @@ X86ASM-OBJS-$(CONFIG_HEVC_DECODER)      += x86/hevc/dsp_init.o      \
                                            x86/hevc/dequant.o       \
                                            x86/hevc/idct.o          \
                                            x86/hevc/mc.o            \
+                                           x86/hevc/pred.o          \
+                                           x86/hevc/pred_init.o     \
                                            x86/hevc/sao.o           \
                                            x86/hevc/sao_10bit.o     \
                                            x86/h26x/h2656dsp.o      \
diff --git a/libavcodec/x86/hevc/pred.asm b/libavcodec/x86/hevc/pred.asm
new file mode 100644
index 0000000000..724a385a0c
--- /dev/null
+++ b/libavcodec/x86/hevc/pred.asm
@@ -0,0 +1,205 @@
+;******************************************************************************
+;* SIMD-optimized HEVC intra prediction
+;*
+;* This file is part of FFmpeg.
+;*
+;* FFmpeg is free software; you can redistribute it and/or
+;* modify it under the terms of the GNU Lesser General Public
+;* License as published by the Free Software Foundation; either
+;* version 2.1 of the License, or (at your option) any later version.
+;*
+;* FFmpeg is distributed in the hope that it will be useful,
+;* but WITHOUT ANY WARRANTY; without even the implied warranty of
+;* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
+;* Lesser General Public License for more details.
+;*
+;* You should have received a copy of the GNU Lesser General Public
+;* License along with FFmpeg; if not, write to the Free Software
+;* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
+;******************************************************************************
+
+%include "libavutil/x86/x86util.asm"
+
+SECTION_RODATA 16
+
+; planar weights: (size - 1 - x) is a suffix of pw_desc, (x + 1) a prefix of pw_asc
+pw_desc: dw 31, 30, 29, 28, 27, 26, 25, 24, 23, 22, 21, 20, 19, 18, 17, 16
+         dw 15, 14, 13, 12, 11, 10,  9,  8,  7,  6,  5,  4,  3,  2,  1,  0
+pw_asc:  dw  1,  2,  3,  4,  5,  6,  7,  8,  9, 10, 11, 12, 13, 14, 15, 16
+         dw 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31, 32
+
+; pshufb masks: zero-extend bytes to words without a dedicated zero register
+pb_bcastw0: times 8 db 0, -1            ; byte 0 -> all eight words
+pb_widen:   db 0, -1, 1, -1, 2, -1, 3, -1, 4, -1, 5, -1, 6, -1, 7, -1
+
+; 4x4-specific: two rows per register
+pb_top4dup: db 0, -1, 1, -1, 2, -1, 3, -1, 0, -1, 1, -1, 2, -1, 3, -1
+pb_left01:  db 0, -1, 0, -1, 0, -1, 0, -1, 1, -1, 1, -1, 1, -1, 1, -1
+pb_left23:  db 2, -1, 2, -1, 2, -1, 2, -1, 3, -1, 3, -1, 3, -1, 3, -1
+pw_planar4_a:   dw 3, 2, 1, 0, 3, 2, 1, 0   ; size - 1 - x
+pw_planar4_b:   dw 1, 2, 3, 4, 1, 2, 3, 4   ; x + 1
+pw_planar4_vy0: dw 3, 3, 3, 3, 2, 2, 2, 2   ; size - 1 - y, rows 0-1
+pw_planar4_vy1: dw 1, 1, 1, 1, 0, 0, 0, 0   ; rows 2-3
+pw_planar4_yb0: dw 1, 1, 1, 1, 2, 2, 2, 2   ; y + 1, rows 0-1
+pw_planar4_yb1: dw 3, 3, 3, 3, 4, 4, 4, 4   ; rows 2-3
+
+cextern pw_4
+cextern pw_8
+cextern pw_16
+cextern pw_32
+
+SECTION .text
+
+; void ff_hevc_pred_planar_<idx>_8_<opt>(uint8_t *src, const uint8_t *top,
+;                                        const uint8_t *left, ptrdiff_t stride)
+;
+; src[y*stride + x] = ((size-1-x)*left[y] + (x+1)*top[size] +
+;                      (size-1-y)*top[x]  + (y+1)*left[size] + size) >> (log2 + 1)
+;
+; Everything except (size-1-x)*left[y] depends on y only through two terms that
+; are both linear in y, so for a fixed group of eight columns they collapse into
+; a single running accumulator
+;
+;   sum(y) = (x+1)*top[size] + (y+1)*left[size] + size + (size-1-y)*top[x]
+;   sum(y+1) - sum(y) = left[size] - top[x]
+;
+; which is loop-invariant. Iterating columns outermost and rows innermost keeps
+; sum and that delta in registers, so a row costs one pmullw and two paddw and
+; no reload of top[]. All intermediates stay in words: the largest value the
+; expression can take is 31*255 + 32*255 + 31*255 + 32*255 + 32 = 32162 < 2^15.
+
+; Set up the running accumulator for the eight columns starting at %6.
+; (size-1)*top[x] is formed as (top[x] << log2) - top[x] so that no vector of
+; size-1 has to be materialised.
+; %1 = sum, %2 = delta, %3 = temp, %4 = size, %5 = index, %6 = first column
+%macro PLANAR_INIT 6
+    movd            %1, [topq + %4]
+    pshufb          %1, [pb_bcastw0]        ; top[size]
+    pmullw          %1, [pw_asc + (%6) * 2] ; (x+1)*top[size]
+    movd            %2, [leftq + %4]
+    pshufb          %2, [pb_bcastw0]        ; left[size]
+    paddw           %1, %2
+    paddw           %1, [pw_ %+ %4]         ; + left[size] + size
+    movq            %3, [topq + (%6)]
+    pshufb          %3, [pb_widen]          ; top[x]
+    psubw           %2, %3                  ; delta = left[size] - top[x]
+    psubw           %1, %3
+    psllw           %3, (%5) + 2
+    paddw           %1, %3                  ; + (size-1)*top[x] = sum(0)
+%endmacro
+
+; One group of eight columns, for the block size that is only eight wide.
+; %1 = block size, %2 = index, %3 = first column
+%macro PLANAR_CHUNK 3
+    PLANAR_INIT     m0, m1, m2, %1, %2, %3
+    lea           srcq, [baseq + %3]
+    xor             yd, yd
+%%loop:
+    movd            m2, [leftq + yq]
+    pshufb          m2, [pb_bcastw0]                    ; left[y]
+    pmullw          m2, [pw_desc + (32 - %1 + %3) * 2]  ; (size-1-x)*left[y]
+    paddw           m2, m0
+    psrlw           m2, %2 + 3
+    packuswb        m2, m2
+    movq        [srcq], m2
+    add           srcq, strideq
+    paddw           m0, m1
+    inc             yd
+    cmp             yd, %1
+    jl %%loop
+%endmacro
+
+; Two adjacent groups at once, so that a row is written with a single 16 byte
+; store and the left[y] broadcast is shared between them.
+; %1 = block size, %2 = index, %3 = first column
+%macro PLANAR_CHUNK2 3
+    PLANAR_INIT     m0, m1, m4, %1, %2, %3
+    PLANAR_INIT     m2, m3, m4, %1, %2, %3 + 8
+    lea           srcq, [baseq + %3]
+    xor             yd, yd
+%%loop:
+    movd            m4, [leftq + yq]
+    pshufb          m4, [pb_bcastw0]                        ; left[y]
+    mova            m5, m4
+    pmullw          m4, [pw_desc + (32 - %1 + %3) * 2]
+    paddw           m4, m0
+    psrlw           m4, %2 + 3
+    pmullw          m5, [pw_desc + (32 - %1 + %3 + 8) * 2]
+    paddw           m5, m2
+    psrlw           m5, %2 + 3
+    packuswb        m4, m5
+    movu        [srcq], m4
+    add           srcq, strideq
+    paddw           m0, m1
+    paddw           m2, m3
+    inc             yd
+    cmp             yd, %1
+    jl %%loop
+%endmacro
+
+; %1 = block size, %2 = pred_planar index (log2(size) - 2)
+%macro PRED_PLANAR 2
+cglobal hevc_pred_planar_%2_8, 4, 6, 6, src, top, left, stride, base, y
+    mov          baseq, srcq
+%if %1 == 8
+    PLANAR_CHUNK %1, %2, 0
+%else
+%assign %%off 0
+%rep %1 / 16
+    PLANAR_CHUNK2 %1, %2, %%off
+%assign %%off %%off + 16
+%endrep
+%endif
+    RET
+%endmacro
+
+; 4x4: fully unrolled, two rows per register
+%macro PRED_PLANAR4 0
+cglobal hevc_pred_planar_0_8, 4, 4, 5, src, top, left, stride
+    movd            m0, [topq + 4]
+    pshufb          m0, [pb_bcastw0]        ; top[4]
+    pmullw          m0, [pw_planar4_b]      ; (x+1)*top[4]
+    mova            m1, m0
+    movd            m2, [leftq + 4]
+    pshufb          m2, [pb_bcastw0]        ; left[4]
+    mova            m3, m2
+    pmullw          m3, [pw_planar4_yb0]
+    paddw           m0, m3
+    paddw           m0, [pw_4]              ; row-invariant terms, rows 0-1
+    pmullw          m2, [pw_planar4_yb1]
+    paddw           m1, m2
+    paddw           m1, [pw_4]              ; rows 2-3
+    movd            m2, [topq]
+    pshufb          m2, [pb_top4dup]        ; top[0..3], both halves
+    movd            m3, [leftq]
+    mova            m4, m3
+    pshufb          m3, [pb_left01]         ; left[0] x4 | left[1] x4
+    pshufb          m4, [pb_left23]         ; left[2] x4 | left[3] x4
+    pmullw          m3, [pw_planar4_a]
+    paddw           m3, m0
+    mova            m0, m2
+    pmullw          m0, [pw_planar4_vy0]
+    paddw           m3, m0
+    psrlw           m3, 3
+    pmullw          m4, [pw_planar4_a]
+    paddw           m4, m1
+    pmullw          m2, [pw_planar4_vy1]
+    paddw           m4, m2
+    psrlw           m4, 3
+    packuswb        m3, m4
+    movd        [srcq], m3
+    psrldq          m3, 4
+    movd [srcq + strideq], m3
+    lea           srcq, [srcq + strideq * 2]
+    psrldq          m3, 4
+    movd        [srcq], m3
+    psrldq          m3, 4
+    movd [srcq + strideq], m3
+    RET
+%endmacro
+
+INIT_XMM ssse3
+PRED_PLANAR4
+PRED_PLANAR  8, 1
+PRED_PLANAR 16, 2
+PRED_PLANAR 32, 3
diff --git a/libavcodec/x86/hevc/pred_init.c b/libavcodec/x86/hevc/pred_init.c
new file mode 100644
index 0000000000..980e28b861
--- /dev/null
+++ b/libavcodec/x86/hevc/pred_init.c
@@ -0,0 +1,51 @@
+/*
+ * SIMD-optimized HEVC intra prediction
+ *
+ * This file is part of FFmpeg.
+ *
+ * FFmpeg is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU Lesser General Public
+ * License as published by the Free Software Foundation; either
+ * version 2.1 of the License, or (at your option) any later version.
+ *
+ * FFmpeg is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
+ * Lesser General Public License for more details.
+ *
+ * You should have received a copy of the GNU Lesser General Public
+ * License along with FFmpeg; if not, write to the Free Software
+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
+ */
+
+#include "config.h"
+
+#include "libavutil/attributes.h"
+#include "libavutil/cpu.h"
+#include "libavutil/x86/cpu.h"
+#include "libavcodec/hevc/pred.h"
+
+#define PRED_PLANAR_FUNC(idx, opt)                                            \
+void ff_hevc_pred_planar_ ## idx ## _8_ ## opt(uint8_t *src,                  \
+                                               const uint8_t *top,            \
+                                               const uint8_t *left,           \
+                                               ptrdiff_t stride);
+
+PRED_PLANAR_FUNC(0, ssse3)
+PRED_PLANAR_FUNC(1, ssse3)
+PRED_PLANAR_FUNC(2, ssse3)
+PRED_PLANAR_FUNC(3, ssse3)
+
+av_cold void ff_hevc_pred_init_x86(HEVCPredContext *hpc, int bit_depth)
+{
+    int cpu_flags = av_get_cpu_flags();
+
+    if (bit_depth == 8) {
+        if (EXTERNAL_SSSE3(cpu_flags)) {
+            hpc->pred_planar[0] = ff_hevc_pred_planar_0_8_ssse3;
+            hpc->pred_planar[1] = ff_hevc_pred_planar_1_8_ssse3;
+            hpc->pred_planar[2] = ff_hevc_pred_planar_2_8_ssse3;
+            hpc->pred_planar[3] = ff_hevc_pred_planar_3_8_ssse3;
+        }
+    }
+}
-- 
2.52.0

_______________________________________________
ffmpeg-devel mailing list -- [email protected]
To unsubscribe send an email to [email protected]
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.