PR #24602 opened by Shreesh Adiga (tantei3)
URL: https://code.ffmpeg.org/FFmpeg/FFmpeg/pulls/24602
Patch URL: https://code.ffmpeg.org/FFmpeg/FFmpeg/pulls/24602.patch

Add SSSE3 implementation for base64 encode which reads 12 bytes and
splits the input into 16 6b elements, which is then translated.
The algorithm is described in https://arxiv.org/pdf/1704.00605 section
4.1 for base64 encode.

Checkasm output on AMD Zen4 reports 4x speedup for 1kB buffer size:
```
SSSE3:
 - base64.base64 [OK]
checkasm: all 1 tests passed
Benchmark results:
  name                      cycles (vs ref)
  av_base64_encode_c:       1354.4
  av_base64_encode_ssse3:    288.2 ( 4.70x)
```



>From cc545c7bf98b031d78bfaf9f876b7e9ecc89cc41 Mon Sep 17 00:00:00 2001
From: Shreesh Adiga <[email protected]>
Date: Mon, 21 Sep 2026 15:57:40 +0530
Subject: [PATCH] avutil/base64: add x86_64 SSSE3 implementation for base64
 encode

Add SSSE3 implementation for base64 encode which reads 12 bytes and
splits the input into 16 6b elements, which is then translated.
The algorithm is described in https://arxiv.org/pdf/1704.00605 section
4.1 for base64 encode.

Checkasm output on AMD Zen4 reports 4x speedup for 1kB buffer size:
SSSE3:
 - base64.base64 [OK]
checkasm: all 1 tests passed
Benchmark results:
  name                      cycles (vs ref)
  av_base64_encode_c:       1354.4
  av_base64_encode_ssse3:    288.2 ( 4.70x)
---
 libavutil/base64.c       |   7 ++
 libavutil/x86/Makefile   |   3 +-
 libavutil/x86/base64.asm | 208 +++++++++++++++++++++++++++++++++++++++
 libavutil/x86/base64.h   |  26 +++++
 tests/checkasm/base64.c  |   8 ++
 5 files changed, 251 insertions(+), 1 deletion(-)
 create mode 100644 libavutil/x86/base64.asm
 create mode 100644 libavutil/x86/base64.h

diff --git a/libavutil/base64.c b/libavutil/base64.c
index a6c21b05e7..875cfd5e1e 100644
--- a/libavutil/base64.c
+++ b/libavutil/base64.c
@@ -34,6 +34,9 @@
 #if ARCH_AARCH64
 #include "libavutil/aarch64/cpu.h"
 #include "libavutil/aarch64/base64.h"
+#elif ARCH_X86_64
+#include "libavutil/x86/cpu.h"
+#include "libavutil/x86/base64.h"
 #endif
 
 /* ---------------- private code */
@@ -161,6 +164,10 @@ char *av_base64_encode(char *out, int out_size, const 
uint8_t *in, int in_size)
     if (in_size >= 48 && have_neon(av_get_cpu_flags())) {
         return ff_base64_encode_neon(out, in, in_size, b64);
     }
+#elif ARCH_X86_64 && HAVE_SSSE3_EXTERNAL
+    if (in_size >= 48 && EXTERNAL_SSSE3(av_get_cpu_flags())) {
+        return ff_base64_encode_ssse3(out, in, in_size, b64);
+    }
 #endif
     return ff_base64_encode_c(out, in, in_size, b64);
 }
diff --git a/libavutil/x86/Makefile b/libavutil/x86/Makefile
index a305503c0f..16d78c2a03 100644
--- a/libavutil/x86/Makefile
+++ b/libavutil/x86/Makefile
@@ -4,7 +4,8 @@ 
EMMS_OBJS_$(HAVE_MMX_INLINE)_$(HAVE_MMX_EXTERNAL)_$(HAVE_MM_EMPTY) = x86/emms.o
 # For static builds, libavutil provides ff_emms for all libraries (if needed).
 STLIBOBJS   += $(EMMS_OBJS__yes_)
 
-X86ASM-OBJS += x86/cpuid.o                                              \
+X86ASM-OBJS += x86/base64.o                                             \
+               x86/cpuid.o                                              \
                x86/crc.o                                                \
                x86/fixed_dsp.o x86/fixed_dsp_init.o                     \
                x86/float_dsp.o x86/float_dsp_init.o                     \
diff --git a/libavutil/x86/base64.asm b/libavutil/x86/base64.asm
new file mode 100644
index 0000000000..a2297a040e
--- /dev/null
+++ b/libavutil/x86/base64.asm
@@ -0,0 +1,208 @@
+;*****************************************************************************
+;* Copyright (c) 2026 Shreesh Adiga <[email protected]>
+;*
+;* This file is part of FFmpeg.
+;*
+;* FFmpeg is free software; you can redistribute it and/or
+;* modify it under the terms of the GNU Lesser General Public
+;* License as published by the Free Software Foundation; either
+;* version 2.1 of the License, or (at your option) any later version.
+;*
+;* FFmpeg is distributed in the hope that it will be useful,
+;* but WITHOUT ANY WARRANTY; without even the implied warranty of
+;* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
+;* Lesser General Public License for more details.
+;*
+;* You should have received a copy of the GNU Lesser General Public
+;* License along with FFmpeg; if not, write to the Free Software
+;* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
+;******************************************************************************
+
+%include "x86util.asm"
+
+SECTION_RODATA
+bytes_shuffle: db 1, 0, 2, 1, 4, 3, 5, 4, 7, 6, 8, 7, 10, 9, 11, 10
+
+offset_lookup_table: db  65,  71, 252, 252, 252, 252, 252, 252,\
+                        252, 252, 252, 252, 237, 240,   0,   0
+
+SECTION .text
+
+%macro UNPACK_6b 2
+; Takes a 8b*12 XMM register and expands it to have 16 6-bit elements.
+    pshufb  %1, m0
+    mova    %2, %1
+    pand    %2, m1
+    pmulhuw %2, m2
+    pand    %1, m3
+    pmullw  %1, m4
+    por     %1, %2
+%endmacro
+
+%macro TRANSLATE_BASE64 3
+; takes a 6b * 16 XMM register and maps each 0-63 byte to base64 value.
+; maps 0-25 to 0, 26-51 to 1, 52-61 to 2-11, 62 to 12 and 63 to 13.
+; then uses lookup table in m7 to derive the delta values.
+; the delta values are then added to input to obtain the base64 translation.
+    mova    %2, %1
+    psubusb %2, m5
+    mova    %3, %1
+    pcmpgtb %3, m6
+    psubb   %2, %3
+    mova    %3, m7
+    pshufb  %3, %2
+    paddb   %1, %3
+%endmacro
+
+%macro BROADCAST_EPI32 3
+    mov    %2, %3
+    movd   %1, %2
+    pshufd %1, %1, 0
+%endmacro
+
+; SIMD Algorithm Reference: https://arxiv.org/pdf/1704.00605
+%macro BASE64_ENCODE 0
+;----------------------------------------------------------------------------------------
+; ff_base64_encode_ssse3(char *out, const uint8_t *in, int in_size, const char 
*tbl)
+;----------------------------------------------------------------------------------------
+cglobal base64_encode, 4, 7, 11
+; r0 - out
+; r1 - in
+; r2 - in_size
+; r3 - tbl
+; r4 - dst
+    mov r4, r0
+    cmp r2d, 16
+    jl .simd_loop_done
+
+    mova            m0, [bytes_shuffle]
+    BROADCAST_EPI32 m1, r5d, 0x0fc0fc00
+    BROADCAST_EPI32 m2, r6d, 0x04000040
+    BROADCAST_EPI32 m3, r5d, 0x003f03f0
+    BROADCAST_EPI32 m4, r6d, 0x01000010
+    BROADCAST_EPI32 m5, r5d, 0x33333333
+    BROADCAST_EPI32 m6, r6d, 0x19191919
+    mova            m7, [offset_lookup_table]
+
+.simd_loop:
+    movu               m8, [r1]
+    UNPACK_6b          m8, m9
+    TRANSLATE_BASE64   m8, m9, m10
+    movu             [r4], m8
+    add                r1, 12
+    sub               r2d, 12
+    add                r4, 16
+    cmp               r2d, 16
+    jge        .simd_loop
+
+.simd_loop_done:
+    cmp r2d, 4
+    jl .less_than_4b
+
+.loop_4b:
+    mov        r5d, [r1]
+    bswap      r5d
+    sub        r2d, 3
+    add         r1, 3
+    mov        r6d, r5d
+    shr        r6d, 26
+    movzx      r6d, byte [r3 + r6]
+    mov       [r4], r6b
+    mov        r6d, r5d
+    shr        r6d, 20
+    and        r6d, 63
+    movzx      r6d, byte [r3 + r6]
+    mov   [r4 + 1], r6b
+    mov        r6d, r5d
+    shr        r6d, 14
+    and        r6d, 63
+    movzx      r6d, byte [r3 + r6]
+    mov   [r4 + 2], r6b
+    shr        r5d, 8
+    and        r5d, 63
+    movzx      r5d, byte [r3 + r5]
+    mov   [r4 + 3], r5b
+    add         r4, 4
+    cmp        r2d, 3
+    jg    .loop_4b
+
+.less_than_4b:
+    cmp      r2d, 3
+    je  .tail_3b
+    cmp      r2d, 2
+    je  .tail_2b
+    cmp      r2d, 1
+    je  .tail_1b
+
+.last:
+    mov byte [r4], 0
+    mov       rax, r0
+    RET
+
+.tail_1b:
+    movzx           r5d, byte [r1]
+    mov             r6d, r5d
+    shr             r6d, 2
+    movzx           r6d, byte [r3 + r6]
+    mov            [r4], r6b
+    shl             r5d, 4
+    and             r5d, 48
+    movzx           r6d, byte [r3 + r5]
+    mov        [r4 + 1], r6b
+    mov   word [r4 + 2], 15677 ; '=='
+    add              r4, 4
+    jmp           .last
+
+.tail_2b:
+    movzx           r5d, byte [r1]
+    movzx           r2d, byte [r1 + 1]
+    mov             r6d, r5d
+    shr             r6d, 2
+    movzx           r6d, byte [r3 + r6]
+    mov            [r4], r6b
+    shl             r5d, 4
+    and             r5d, 48
+    mov             r6d, r2d
+    shr             r6d, 4
+    or              r6d, r5d
+    movzx           r6d, byte [r3 + r6]
+    mov        [r4 + 1], r6b
+    and             r2d, 15
+    movzx           r6d, byte [r3 + 4 * r2]
+    mov        [r4 + 2], r6b
+    mov   byte [r4 + 3], '='
+    add              r4, 4
+    jmp           .last
+
+.tail_3b:
+    movzx      r5d, byte [r1]
+    movzx      r2d, byte [r1 + 1]
+    movzx      r1d, byte [r1 + 2]
+    mov        r6d, r5d
+    shr        r6d, 2
+    movzx      r6d, byte [r3 + r6]
+    mov       [r4], r6b
+    shl        r5d, 4
+    and        r5d, 48
+    mov        r6d, r2d
+    shr        r6d, 4
+    or         r6d, r5d
+    movzx      r6d, byte [r3 + r6]
+    mov   [r4 + 1], r6b
+    and        r2d, 15
+    mov        r5d, r1d
+    shr        r5d, 6
+    lea        r6d, [r5 + 4 * r2]
+    movzx      r6d, byte [r3 + r6]
+    mov   [r4 + 2], r6b
+    and        r1d, 63
+    movzx      r6d, byte [r3 + r1]
+    mov   [r4 + 3], r6b
+    add         r4, 4
+    jmp      .last
+%endmacro
+
+%if ARCH_X86_64 && HAVE_SSSE3_EXTERNAL
+INIT_XMM ssse3
+BASE64_ENCODE
+%endif
diff --git a/libavutil/x86/base64.h b/libavutil/x86/base64.h
new file mode 100644
index 0000000000..0d1b4ba100
--- /dev/null
+++ b/libavutil/x86/base64.h
@@ -0,0 +1,26 @@
+/*
+ * This file is part of FFmpeg.
+ *
+ * FFmpeg is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU Lesser General Public
+ * License as published by the Free Software Foundation; either
+ * version 2.1 of the License, or (at your option) any later version.
+ *
+ * FFmpeg is distributed in the hope that it will be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
+ * Lesser General Public License for more details.
+ *
+ * You should have received a copy of the GNU Lesser General Public
+ * License along with FFmpeg; if not, write to the Free Software
+ * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
+ */
+
+#ifndef AVUTIL_X86_BASE64_H
+#define AVUTIL_X86_BASE64_H
+
+#include <stdint.h>
+
+char *ff_base64_encode_ssse3(char *out, const uint8_t *in, int in_size, const 
char *tbl);
+
+#endif /* AVUTIL_X86_BASE64_H */
diff --git a/tests/checkasm/base64.c b/tests/checkasm/base64.c
index dd75aed9e6..91bfd9acaf 100644
--- a/tests/checkasm/base64.c
+++ b/tests/checkasm/base64.c
@@ -25,6 +25,9 @@
 #if ARCH_AARCH64
 #include "libavutil/aarch64/cpu.h"
 #include "libavutil/aarch64/base64.h"
+#elif ARCH_X86_64
+#include "libavutil/x86/cpu.h"
+#include "libavutil/x86/base64.h"
 #endif
 #include <string.h>
 
@@ -36,6 +39,11 @@ static char *(*base64_encode_get_fn(void))(char *, const 
uint8_t *, int, const c
     if (have_neon(cpu_flags))
         return ff_base64_encode_neon;
     else
+#elif ARCH_X86_64
+    int cpu_flags = av_get_cpu_flags();
+    if (EXTERNAL_SSSE3(cpu_flags))
+        return ff_base64_encode_ssse3;
+    else
 #endif
         return ff_base64_encode_c;
 }
-- 
2.52.0

_______________________________________________
ffmpeg-devel mailing list -- [email protected]
To unsubscribe send an email to [email protected]

Reply via email to