PR #24602 opened by Shreesh Adiga (tantei3) URL: https://code.ffmpeg.org/FFmpeg/FFmpeg/pulls/24602 Patch URL: https://code.ffmpeg.org/FFmpeg/FFmpeg/pulls/24602.patch
Add SSSE3 implementation for base64 encode which reads 12 bytes and splits the input into 16 6b elements, which is then translated. The algorithm is described in https://arxiv.org/pdf/1704.00605 section 4.1 for base64 encode. Checkasm output on AMD Zen4 reports 4x speedup for 1kB buffer size: ``` SSSE3: - base64.base64 [OK] checkasm: all 1 tests passed Benchmark results: name cycles (vs ref) av_base64_encode_c: 1354.4 av_base64_encode_ssse3: 288.2 ( 4.70x) ``` >From cc545c7bf98b031d78bfaf9f876b7e9ecc89cc41 Mon Sep 17 00:00:00 2001 From: Shreesh Adiga <[email protected]> Date: Mon, 21 Sep 2026 15:57:40 +0530 Subject: [PATCH] avutil/base64: add x86_64 SSSE3 implementation for base64 encode Add SSSE3 implementation for base64 encode which reads 12 bytes and splits the input into 16 6b elements, which is then translated. The algorithm is described in https://arxiv.org/pdf/1704.00605 section 4.1 for base64 encode. Checkasm output on AMD Zen4 reports 4x speedup for 1kB buffer size: SSSE3: - base64.base64 [OK] checkasm: all 1 tests passed Benchmark results: name cycles (vs ref) av_base64_encode_c: 1354.4 av_base64_encode_ssse3: 288.2 ( 4.70x) --- libavutil/base64.c | 7 ++ libavutil/x86/Makefile | 3 +- libavutil/x86/base64.asm | 208 +++++++++++++++++++++++++++++++++++++++ libavutil/x86/base64.h | 26 +++++ tests/checkasm/base64.c | 8 ++ 5 files changed, 251 insertions(+), 1 deletion(-) create mode 100644 libavutil/x86/base64.asm create mode 100644 libavutil/x86/base64.h diff --git a/libavutil/base64.c b/libavutil/base64.c index a6c21b05e7..875cfd5e1e 100644 --- a/libavutil/base64.c +++ b/libavutil/base64.c @@ -34,6 +34,9 @@ #if ARCH_AARCH64 #include "libavutil/aarch64/cpu.h" #include "libavutil/aarch64/base64.h" +#elif ARCH_X86_64 +#include "libavutil/x86/cpu.h" +#include "libavutil/x86/base64.h" #endif /* ---------------- private code */ @@ -161,6 +164,10 @@ char *av_base64_encode(char *out, int out_size, const uint8_t *in, int in_size) if (in_size >= 48 && have_neon(av_get_cpu_flags())) { return ff_base64_encode_neon(out, in, in_size, b64); } +#elif ARCH_X86_64 && HAVE_SSSE3_EXTERNAL + if (in_size >= 48 && EXTERNAL_SSSE3(av_get_cpu_flags())) { + return ff_base64_encode_ssse3(out, in, in_size, b64); + } #endif return ff_base64_encode_c(out, in, in_size, b64); } diff --git a/libavutil/x86/Makefile b/libavutil/x86/Makefile index a305503c0f..16d78c2a03 100644 --- a/libavutil/x86/Makefile +++ b/libavutil/x86/Makefile @@ -4,7 +4,8 @@ EMMS_OBJS_$(HAVE_MMX_INLINE)_$(HAVE_MMX_EXTERNAL)_$(HAVE_MM_EMPTY) = x86/emms.o # For static builds, libavutil provides ff_emms for all libraries (if needed). STLIBOBJS += $(EMMS_OBJS__yes_) -X86ASM-OBJS += x86/cpuid.o \ +X86ASM-OBJS += x86/base64.o \ + x86/cpuid.o \ x86/crc.o \ x86/fixed_dsp.o x86/fixed_dsp_init.o \ x86/float_dsp.o x86/float_dsp_init.o \ diff --git a/libavutil/x86/base64.asm b/libavutil/x86/base64.asm new file mode 100644 index 0000000000..a2297a040e --- /dev/null +++ b/libavutil/x86/base64.asm @@ -0,0 +1,208 @@ +;***************************************************************************** +;* Copyright (c) 2026 Shreesh Adiga <[email protected]> +;* +;* This file is part of FFmpeg. +;* +;* FFmpeg is free software; you can redistribute it and/or +;* modify it under the terms of the GNU Lesser General Public +;* License as published by the Free Software Foundation; either +;* version 2.1 of the License, or (at your option) any later version. +;* +;* FFmpeg is distributed in the hope that it will be useful, +;* but WITHOUT ANY WARRANTY; without even the implied warranty of +;* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU +;* Lesser General Public License for more details. +;* +;* You should have received a copy of the GNU Lesser General Public +;* License along with FFmpeg; if not, write to the Free Software +;* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA +;****************************************************************************** + +%include "x86util.asm" + +SECTION_RODATA +bytes_shuffle: db 1, 0, 2, 1, 4, 3, 5, 4, 7, 6, 8, 7, 10, 9, 11, 10 + +offset_lookup_table: db 65, 71, 252, 252, 252, 252, 252, 252,\ + 252, 252, 252, 252, 237, 240, 0, 0 + +SECTION .text + +%macro UNPACK_6b 2 +; Takes a 8b*12 XMM register and expands it to have 16 6-bit elements. + pshufb %1, m0 + mova %2, %1 + pand %2, m1 + pmulhuw %2, m2 + pand %1, m3 + pmullw %1, m4 + por %1, %2 +%endmacro + +%macro TRANSLATE_BASE64 3 +; takes a 6b * 16 XMM register and maps each 0-63 byte to base64 value. +; maps 0-25 to 0, 26-51 to 1, 52-61 to 2-11, 62 to 12 and 63 to 13. +; then uses lookup table in m7 to derive the delta values. +; the delta values are then added to input to obtain the base64 translation. + mova %2, %1 + psubusb %2, m5 + mova %3, %1 + pcmpgtb %3, m6 + psubb %2, %3 + mova %3, m7 + pshufb %3, %2 + paddb %1, %3 +%endmacro + +%macro BROADCAST_EPI32 3 + mov %2, %3 + movd %1, %2 + pshufd %1, %1, 0 +%endmacro + +; SIMD Algorithm Reference: https://arxiv.org/pdf/1704.00605 +%macro BASE64_ENCODE 0 +;---------------------------------------------------------------------------------------- +; ff_base64_encode_ssse3(char *out, const uint8_t *in, int in_size, const char *tbl) +;---------------------------------------------------------------------------------------- +cglobal base64_encode, 4, 7, 11 +; r0 - out +; r1 - in +; r2 - in_size +; r3 - tbl +; r4 - dst + mov r4, r0 + cmp r2d, 16 + jl .simd_loop_done + + mova m0, [bytes_shuffle] + BROADCAST_EPI32 m1, r5d, 0x0fc0fc00 + BROADCAST_EPI32 m2, r6d, 0x04000040 + BROADCAST_EPI32 m3, r5d, 0x003f03f0 + BROADCAST_EPI32 m4, r6d, 0x01000010 + BROADCAST_EPI32 m5, r5d, 0x33333333 + BROADCAST_EPI32 m6, r6d, 0x19191919 + mova m7, [offset_lookup_table] + +.simd_loop: + movu m8, [r1] + UNPACK_6b m8, m9 + TRANSLATE_BASE64 m8, m9, m10 + movu [r4], m8 + add r1, 12 + sub r2d, 12 + add r4, 16 + cmp r2d, 16 + jge .simd_loop + +.simd_loop_done: + cmp r2d, 4 + jl .less_than_4b + +.loop_4b: + mov r5d, [r1] + bswap r5d + sub r2d, 3 + add r1, 3 + mov r6d, r5d + shr r6d, 26 + movzx r6d, byte [r3 + r6] + mov [r4], r6b + mov r6d, r5d + shr r6d, 20 + and r6d, 63 + movzx r6d, byte [r3 + r6] + mov [r4 + 1], r6b + mov r6d, r5d + shr r6d, 14 + and r6d, 63 + movzx r6d, byte [r3 + r6] + mov [r4 + 2], r6b + shr r5d, 8 + and r5d, 63 + movzx r5d, byte [r3 + r5] + mov [r4 + 3], r5b + add r4, 4 + cmp r2d, 3 + jg .loop_4b + +.less_than_4b: + cmp r2d, 3 + je .tail_3b + cmp r2d, 2 + je .tail_2b + cmp r2d, 1 + je .tail_1b + +.last: + mov byte [r4], 0 + mov rax, r0 + RET + +.tail_1b: + movzx r5d, byte [r1] + mov r6d, r5d + shr r6d, 2 + movzx r6d, byte [r3 + r6] + mov [r4], r6b + shl r5d, 4 + and r5d, 48 + movzx r6d, byte [r3 + r5] + mov [r4 + 1], r6b + mov word [r4 + 2], 15677 ; '==' + add r4, 4 + jmp .last + +.tail_2b: + movzx r5d, byte [r1] + movzx r2d, byte [r1 + 1] + mov r6d, r5d + shr r6d, 2 + movzx r6d, byte [r3 + r6] + mov [r4], r6b + shl r5d, 4 + and r5d, 48 + mov r6d, r2d + shr r6d, 4 + or r6d, r5d + movzx r6d, byte [r3 + r6] + mov [r4 + 1], r6b + and r2d, 15 + movzx r6d, byte [r3 + 4 * r2] + mov [r4 + 2], r6b + mov byte [r4 + 3], '=' + add r4, 4 + jmp .last + +.tail_3b: + movzx r5d, byte [r1] + movzx r2d, byte [r1 + 1] + movzx r1d, byte [r1 + 2] + mov r6d, r5d + shr r6d, 2 + movzx r6d, byte [r3 + r6] + mov [r4], r6b + shl r5d, 4 + and r5d, 48 + mov r6d, r2d + shr r6d, 4 + or r6d, r5d + movzx r6d, byte [r3 + r6] + mov [r4 + 1], r6b + and r2d, 15 + mov r5d, r1d + shr r5d, 6 + lea r6d, [r5 + 4 * r2] + movzx r6d, byte [r3 + r6] + mov [r4 + 2], r6b + and r1d, 63 + movzx r6d, byte [r3 + r1] + mov [r4 + 3], r6b + add r4, 4 + jmp .last +%endmacro + +%if ARCH_X86_64 && HAVE_SSSE3_EXTERNAL +INIT_XMM ssse3 +BASE64_ENCODE +%endif diff --git a/libavutil/x86/base64.h b/libavutil/x86/base64.h new file mode 100644 index 0000000000..0d1b4ba100 --- /dev/null +++ b/libavutil/x86/base64.h @@ -0,0 +1,26 @@ +/* + * This file is part of FFmpeg. + * + * FFmpeg is free software; you can redistribute it and/or + * modify it under the terms of the GNU Lesser General Public + * License as published by the Free Software Foundation; either + * version 2.1 of the License, or (at your option) any later version. + * + * FFmpeg is distributed in the hope that it will be useful, + * but WITHOUT ANY WARRANTY; without even the implied warranty of + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU + * Lesser General Public License for more details. + * + * You should have received a copy of the GNU Lesser General Public + * License along with FFmpeg; if not, write to the Free Software + * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA + */ + +#ifndef AVUTIL_X86_BASE64_H +#define AVUTIL_X86_BASE64_H + +#include <stdint.h> + +char *ff_base64_encode_ssse3(char *out, const uint8_t *in, int in_size, const char *tbl); + +#endif /* AVUTIL_X86_BASE64_H */ diff --git a/tests/checkasm/base64.c b/tests/checkasm/base64.c index dd75aed9e6..91bfd9acaf 100644 --- a/tests/checkasm/base64.c +++ b/tests/checkasm/base64.c @@ -25,6 +25,9 @@ #if ARCH_AARCH64 #include "libavutil/aarch64/cpu.h" #include "libavutil/aarch64/base64.h" +#elif ARCH_X86_64 +#include "libavutil/x86/cpu.h" +#include "libavutil/x86/base64.h" #endif #include <string.h> @@ -36,6 +39,11 @@ static char *(*base64_encode_get_fn(void))(char *, const uint8_t *, int, const c if (have_neon(cpu_flags)) return ff_base64_encode_neon; else +#elif ARCH_X86_64 + int cpu_flags = av_get_cpu_flags(); + if (EXTERNAL_SSSE3(cpu_flags)) + return ff_base64_encode_ssse3; + else #endif return ff_base64_encode_c; } -- 2.52.0 _______________________________________________ ffmpeg-devel mailing list -- [email protected] To unsubscribe send an email to [email protected]
