FFmpeg/libavcodec/mdct15.h

/*
 * Copyright (c) 2017 Rostislav Pehlivanov <atomnuker@gmail.com>
 *
 * This file is part of FFmpeg.
 *
 * FFmpeg is free software; you can redistribute it and/or
 * modify it under the terms of the GNU Lesser General Public
 * License as published by the Free Software Foundation; either
 * version 2.1 of the License, or (at your option) any later version.
 *
 * FFmpeg is distributed in the hope that it will be useful,
 * but WITHOUT ANY WARRANTY; without even the implied warranty of
 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
 * Lesser General Public License for more details.
 *
 * You should have received a copy of the GNU Lesser General Public
 * License along with FFmpeg; if not, write to the Free Software
 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
 */

#ifndef AVCODEC_MDCT15_H
#define AVCODEC_MDCT15_H

#include <stddef.h>

#include "libavutil/mem_internal.h"

#include "fft.h"

typedef struct MDCT15Context {
    int fft_n;
    int len2;
    int len4;
    int inverse;
    int *pfa_prereindex;
    int *pfa_postreindex;

    FFTContext ptwo_fft;
    FFTComplex *tmp;
    FFTComplex *twiddle_exptab;

    DECLARE_ALIGNED(32, FFTComplex, exptab)[64];

    /* 15-point FFT */
    void (*fft15)(FFTComplex *out, FFTComplex *in, FFTComplex *exptab, ptrdiff_t stride);

    /* PFA postrotate and exptab */
    void (*postreindex)(FFTComplex *out, FFTComplex *in, FFTComplex *exp, int *lut, ptrdiff_t len8);

    /* Calculate a full 2N -> N MDCT */
    void (*mdct)(struct MDCT15Context *s, float *dst, const float *src, ptrdiff_t stride);

    /* Calculate the middle half of the iMDCT */
    void (*imdct_half)(struct MDCT15Context *s, float *dst, const float *src,
                       ptrdiff_t stride);
} MDCT15Context;

/* Init an (i)MDCT of the length 2 * 15 * (2^N) */
int ff_mdct15_init(MDCT15Context **ps, int inverse, int N, double scale);
void ff_mdct15_uninit(MDCT15Context **ps);

void ff_mdct15_init_x86(MDCT15Context *s);

#endif /* AVCODEC_MDCT15_H */
aarch64: opus NEON iMDCT and FFT Opus celt decoding 11% faster and the iMDCT over 2.5 times faster on Apple's A7. 2014-04-28 18:56:43 +03:00			`/*`
imdct15: rename to mdct15 and add a forward transform Handles strides (needed for Opus transients), does pre-reindexing and folding without needing a copy. Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-02-01 05:13:06 +02:00			`* Copyright (c) 2017 Rostislav Pehlivanov <atomnuker@gmail.com>`
			`*`
Merge commit 'd3f5b94762fb803c0f3b29f9ad6c5eaa813998ba' * commit 'd3f5b94762fb803c0f3b29f9ad6c5eaa813998ba': aarch64: opus NEON iMDCT and FFT Merged-by: Michael Niedermayer <michaelni@gmx.at> 2014-05-15 22:13:31 +03:00			`* This file is part of FFmpeg.`
aarch64: opus NEON iMDCT and FFT Opus celt decoding 11% faster and the iMDCT over 2.5 times faster on Apple's A7. 2014-04-28 18:56:43 +03:00			`*`
Merge commit 'd3f5b94762fb803c0f3b29f9ad6c5eaa813998ba' * commit 'd3f5b94762fb803c0f3b29f9ad6c5eaa813998ba': aarch64: opus NEON iMDCT and FFT Merged-by: Michael Niedermayer <michaelni@gmx.at> 2014-05-15 22:13:31 +03:00			`* FFmpeg is free software; you can redistribute it and/or`
aarch64: opus NEON iMDCT and FFT Opus celt decoding 11% faster and the iMDCT over 2.5 times faster on Apple's A7. 2014-04-28 18:56:43 +03:00			`* modify it under the terms of the GNU Lesser General Public`
			`* License as published by the Free Software Foundation; either`
			`* version 2.1 of the License, or (at your option) any later version.`
			`*`
Merge commit 'd3f5b94762fb803c0f3b29f9ad6c5eaa813998ba' * commit 'd3f5b94762fb803c0f3b29f9ad6c5eaa813998ba': aarch64: opus NEON iMDCT and FFT Merged-by: Michael Niedermayer <michaelni@gmx.at> 2014-05-15 22:13:31 +03:00			`* FFmpeg is distributed in the hope that it will be useful,`
aarch64: opus NEON iMDCT and FFT Opus celt decoding 11% faster and the iMDCT over 2.5 times faster on Apple's A7. 2014-04-28 18:56:43 +03:00			`* but WITHOUT ANY WARRANTY; without even the implied warranty of`
			`* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU`
			`* Lesser General Public License for more details.`
			`*`
			`* You should have received a copy of the GNU Lesser General Public`
Merge commit 'd3f5b94762fb803c0f3b29f9ad6c5eaa813998ba' * commit 'd3f5b94762fb803c0f3b29f9ad6c5eaa813998ba': aarch64: opus NEON iMDCT and FFT Merged-by: Michael Niedermayer <michaelni@gmx.at> 2014-05-15 22:13:31 +03:00			`* License along with FFmpeg; if not, write to the Free Software`
aarch64: opus NEON iMDCT and FFT Opus celt decoding 11% faster and the iMDCT over 2.5 times faster on Apple's A7. 2014-04-28 18:56:43 +03:00			`* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA`
			`*/`

imdct15: rename to mdct15 and add a forward transform Handles strides (needed for Opus transients), does pre-reindexing and folding without needing a copy. Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-02-01 05:13:06 +02:00			`#ifndef AVCODEC_MDCT15_H`
			`#define AVCODEC_MDCT15_H`
aarch64: opus NEON iMDCT and FFT Opus celt decoding 11% faster and the iMDCT over 2.5 times faster on Apple's A7. 2014-04-28 18:56:43 +03:00
			`#include <stddef.h>`

lavu/mem: move the DECLARE_ALIGNED macro family to mem_internal on next+1 bump They are not properly namespaced and not intended for public use. 2020-05-27 14:54:38 +02:00			`#include "libavutil/mem_internal.h"`

imdct15: replace the FFT with a faster PFA FFT algorithm This commit replaces the current inefficient non-power-of-two FFT with a much faster FFT based on the Prime Factor Algorithm. Although it is already much faster than the old algorithm without SIMD, the new algorithm makes use of the already very throughouly SIMD'd power of two FFT, which improves performance even more across all platforms which we have SIMD support for. Most of the work was done by Peter Barfuss, who passed the code to me to implement into the iMDCT and the current codebase. The code for a 5-point and 15-point FFT was derived from the previous implementation, although it was optimized and simplified, which will make its future SIMD easier. The 15-point FFT is currently using 6% of the current overall decoder overhead. The FFT can now easily be used as a forward transform by simply not multiplying the 5-point FFT's imaginary component by -1 (which comes from the fact that changing the complex exponential's angle by -1 also changes the output by that) and by multiplying the "theta" angle of the main exptab by -1. Hence the deliberately left multiplication by -1 at the end. FATE passes, and performance reports on other platforms/CPUs are welcome. Performance comparisons: iMDCT, PFA: 101127 decicycles in speed, 32765 runs, 3 skips iMDCT, Old: 211022 decicycles in speed, 32768 runs, 0 skips Standalone FFT, 300000 transforms of size 960: PFA Old FFT kiss_fft libfftw3f 3.659695s, 15.726912s, 13.300789s, 1.182222s Being only 3x slower than libfftw3f is a big achievement by itself. There appears to be something capping the performance in the iMDCT side of things, possibly during the pre-stage reindexing. However, it is certainly fast enough for now. Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-01-04 11:23:24 +02:00			`#include "fft.h"`
aarch64: opus NEON iMDCT and FFT Opus celt decoding 11% faster and the iMDCT over 2.5 times faster on Apple's A7. 2014-04-28 18:56:43 +03:00
imdct15: rename to mdct15 and add a forward transform Handles strides (needed for Opus transients), does pre-reindexing and folding without needing a copy. Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-02-01 05:13:06 +02:00			`typedef struct MDCT15Context {`
aarch64: opus NEON iMDCT and FFT Opus celt decoding 11% faster and the iMDCT over 2.5 times faster on Apple's A7. 2014-04-28 18:56:43 +03:00			`int fft_n;`
			`int len2;`
			`int len4;`
imdct15: rename to mdct15 and add a forward transform Handles strides (needed for Opus transients), does pre-reindexing and folding without needing a copy. Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-02-01 05:13:06 +02:00			`int inverse;`
imdct15: replace the FFT with a faster PFA FFT algorithm This commit replaces the current inefficient non-power-of-two FFT with a much faster FFT based on the Prime Factor Algorithm. Although it is already much faster than the old algorithm without SIMD, the new algorithm makes use of the already very throughouly SIMD'd power of two FFT, which improves performance even more across all platforms which we have SIMD support for. Most of the work was done by Peter Barfuss, who passed the code to me to implement into the iMDCT and the current codebase. The code for a 5-point and 15-point FFT was derived from the previous implementation, although it was optimized and simplified, which will make its future SIMD easier. The 15-point FFT is currently using 6% of the current overall decoder overhead. The FFT can now easily be used as a forward transform by simply not multiplying the 5-point FFT's imaginary component by -1 (which comes from the fact that changing the complex exponential's angle by -1 also changes the output by that) and by multiplying the "theta" angle of the main exptab by -1. Hence the deliberately left multiplication by -1 at the end. FATE passes, and performance reports on other platforms/CPUs are welcome. Performance comparisons: iMDCT, PFA: 101127 decicycles in speed, 32765 runs, 3 skips iMDCT, Old: 211022 decicycles in speed, 32768 runs, 0 skips Standalone FFT, 300000 transforms of size 960: PFA Old FFT kiss_fft libfftw3f 3.659695s, 15.726912s, 13.300789s, 1.182222s Being only 3x slower than libfftw3f is a big achievement by itself. There appears to be something capping the performance in the iMDCT side of things, possibly during the pre-stage reindexing. However, it is certainly fast enough for now. Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-01-04 11:23:24 +02:00			`int *pfa_prereindex;`
			`int *pfa_postreindex;`

			`FFTContext ptwo_fft;`
aarch64: opus NEON iMDCT and FFT Opus celt decoding 11% faster and the iMDCT over 2.5 times faster on Apple's A7. 2014-04-28 18:56:43 +03:00			`FFTComplex *tmp;`
			`FFTComplex *twiddle_exptab;`

mdct15: add assembly optimizations for the 15-point FFT c: 1802 decicycles in fft15,16774635 runs, 2581 skips avx: 865 decicycles in fft15,16776378 runs, 838 skips Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-06-18 13:06:30 +02:00			`DECLARE_ALIGNED(32, FFTComplex, exptab)[64];`
aarch64: opus NEON iMDCT and FFT Opus celt decoding 11% faster and the iMDCT over 2.5 times faster on Apple's A7. 2014-04-28 18:56:43 +03:00
mdct15: add assembly optimizations for the 15-point FFT c: 1802 decicycles in fft15,16774635 runs, 2581 skips avx: 865 decicycles in fft15,16776378 runs, 838 skips Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-06-18 13:06:30 +02:00			`/* 15-point FFT */`
			`void (fft15)(FFTComplex out, FFTComplex in, FFTComplex exptab, ptrdiff_t stride);`

mdct15: add inverse transform postrotation SIMD 2.5ms frames: Before (c): 2638 decicycles in postrotate, 2097040 runs, 112 skips After (sse3): 1467 decicycles in postrotate, 2097083 runs, 69 skips After (avx2): 1244 decicycles in postrotate, 2097085 runs, 67 skips 5ms frames: Before (c): 4987 decicycles in postrotate, 1048371 runs, 205 skips After (sse3): 2644 decicycles in postrotate, 1048509 runs, 67 skips After (avx2): 2031 decicycles in postrotate, 1048523 runs, 53 skips 10ms frames: Before (c): 9153 decicycles in postrotate, 523575 runs, 713 skips After (sse3): 5110 decicycles in postrotate, 523726 runs, 562 skips After (avx2): 3738 decicycles in postrotate, 524223 runs, 65 skips 20ms frames: Before (c): 17857 decicycles in postrotate, 261866 runs, 278 skips After (sse3): 10041 decicycles in postrotate, 261746 runs, 398 skips After (avx2): 7050 decicycles in postrotate, 262116 runs, 28 skips Improves total decoding performance for real world content by 9% with avx2. Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-07-29 22:27:01 +02:00			`/* PFA postrotate and exptab */`
			`void (postreindex)(FFTComplex out, FFTComplex in, FFTComplex exp, int *lut, ptrdiff_t len8);`

mdct15: add assembly optimizations for the 15-point FFT c: 1802 decicycles in fft15,16774635 runs, 2581 skips avx: 865 decicycles in fft15,16776378 runs, 838 skips Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-06-18 13:06:30 +02:00			`/* Calculate a full 2N -> N MDCT */`
imdct15: rename to mdct15 and add a forward transform Handles strides (needed for Opus transients), does pre-reindexing and folding without needing a copy. Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-02-01 05:13:06 +02:00			`void (mdct)(struct MDCT15Context s, float dst, const float src, ptrdiff_t stride);`

mdct15: add assembly optimizations for the 15-point FFT c: 1802 decicycles in fft15,16774635 runs, 2581 skips avx: 865 decicycles in fft15,16776378 runs, 838 skips Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-06-18 13:06:30 +02:00			`/* Calculate the middle half of the iMDCT */`
imdct15: rename to mdct15 and add a forward transform Handles strides (needed for Opus transients), does pre-reindexing and folding without needing a copy. Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-02-01 05:13:06 +02:00			`void (imdct_half)(struct MDCT15Context s, float dst, const float src,`
mdct15: remove redundant scale argument to imdct_half The only use of that argument was for Opus downmixing which is very rare and better done after the mdcts. Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-07-11 22:29:22 +02:00			`ptrdiff_t stride);`
imdct15: rename to mdct15 and add a forward transform Handles strides (needed for Opus transients), does pre-reindexing and folding without needing a copy. Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-02-01 05:13:06 +02:00			`} MDCT15Context;`
aarch64: opus NEON iMDCT and FFT Opus celt decoding 11% faster and the iMDCT over 2.5 times faster on Apple's A7. 2014-04-28 18:56:43 +03:00
mdct15: add assembly optimizations for the 15-point FFT c: 1802 decicycles in fft15,16774635 runs, 2581 skips avx: 865 decicycles in fft15,16776378 runs, 838 skips Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-06-18 13:06:30 +02:00			`/* Init an (i)MDCT of the length 2 * 15 * (2^N) */`
imdct15: rename to mdct15 and add a forward transform Handles strides (needed for Opus transients), does pre-reindexing and folding without needing a copy. Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-02-01 05:13:06 +02:00			`int ff_mdct15_init(MDCT15Context **ps, int inverse, int N, double scale);`
			`void ff_mdct15_uninit(MDCT15Context **ps);`
aarch64: opus NEON iMDCT and FFT Opus celt decoding 11% faster and the iMDCT over 2.5 times faster on Apple's A7. 2014-04-28 18:56:43 +03:00
mdct15: add assembly optimizations for the 15-point FFT c: 1802 decicycles in fft15,16774635 runs, 2581 skips avx: 865 decicycles in fft15,16776378 runs, 838 skips Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-06-18 13:06:30 +02:00			`void ff_mdct15_init_x86(MDCT15Context *s);`

imdct15: rename to mdct15 and add a forward transform Handles strides (needed for Opus transients), does pre-reindexing and folding without needing a copy. Signed-off-by: Rostislav Pehlivanov <atomnuker@gmail.com> 2017-02-01 05:13:06 +02:00			`#endif /* AVCODEC_MDCT15_H */`