12 years ago · 6e6e170898
--- a/libavcodec/aacdec.c
+++ b/libavcodec/aacdec.c
@@ -2131,9 +2131,9 @@ static void windowing_and_mdct_ltp(AACContext *ac, float *out,
 
				         ac->fdsp.vector_fmul(in + 448, in + 448, swindow_prev, 128);
			
 
				     }
			
 
				     if (ics->window_sequence[0] != LONG_START_SEQUENCE) {
			
 
				-        ac->dsp.vector_fmul_reverse(in + 1024, in + 1024, lwindow, 1024);
			
 
				+        ac->fdsp.vector_fmul_reverse(in + 1024, in + 1024, lwindow, 1024);
			
 
				     } else {
			
 
				-        ac->dsp.vector_fmul_reverse(in + 1024 + 448, in + 1024 + 448, swindow, 128);
			
 
				+        ac->fdsp.vector_fmul_reverse(in + 1024 + 448, in + 1024 + 448, swindow, 128);
			
 
				         memset(in + 1024 + 576, 0, 448 * sizeof(float));
			
 
				     }
			
 
				     ac->mdct_ltp.mdct_calc(&ac->mdct_ltp, out, in);
			
@@ -2186,17 +2186,17 @@ static void update_ltp(AACContext *ac, SingleChannelElement *sce)
 
				     if (ics->window_sequence[0] == EIGHT_SHORT_SEQUENCE) {
			
 
				         memcpy(saved_ltp,       saved, 512 * sizeof(float));
			
 
				         memset(saved_ltp + 576, 0,     448 * sizeof(float));
			
 
				-        ac->dsp.vector_fmul_reverse(saved_ltp + 448, ac->buf_mdct + 960,     &swindow[64],      64);
			
 
				+        ac->fdsp.vector_fmul_reverse(saved_ltp + 448, ac->buf_mdct + 960,     &swindow[64],      64);
			
 
				         for (i = 0; i < 64; i++)
			
 
				             saved_ltp[i + 512] = ac->buf_mdct[1023 - i] * swindow[63 - i];
			
 
				     } else if (ics->window_sequence[0] == LONG_START_SEQUENCE) {
			
 
				         memcpy(saved_ltp,       ac->buf_mdct + 512, 448 * sizeof(float));
			
 
				         memset(saved_ltp + 576, 0,                  448 * sizeof(float));
			
 
				-        ac->dsp.vector_fmul_reverse(saved_ltp + 448, ac->buf_mdct + 960,     &swindow[64],      64);
			
 
				+        ac->fdsp.vector_fmul_reverse(saved_ltp + 448, ac->buf_mdct + 960,     &swindow[64],      64);
			
 
				         for (i = 0; i < 64; i++)
			
 
				             saved_ltp[i + 512] = ac->buf_mdct[1023 - i] * swindow[63 - i];
			
 
				     } else { // LONG_STOP or ONLY_LONG
			
 
				-        ac->dsp.vector_fmul_reverse(saved_ltp,       ac->buf_mdct + 512,     &lwindow[512],     512);
			
 
				+        ac->fdsp.vector_fmul_reverse(saved_ltp,       ac->buf_mdct + 512,     &lwindow[512],     512);
			
 
				         for (i = 0; i < 512; i++)
			
 
				             saved_ltp[i + 512] = ac->buf_mdct[1023 - i] * lwindow[511 - i];
			
 
				     }
			
--- a/libavcodec/aacenc.c
+++ b/libavcodec/aacenc.c
@@ -183,7 +183,7 @@ static void put_audio_specific_config(AVCodecContext *avctx)
 
				 }
			
 
				 
			
 
				 #define WINDOW_FUNC(type) \
			
 
				-static void apply_ ##type ##_window(DSPContext *dsp, AVFloatDSPContext *fdsp, \
			
 
				+static void apply_ ##type ##_window(AVFloatDSPContext *fdsp, \
			
 
				                                     SingleChannelElement *sce, \
			
 
				                                     const float *audio)
			
 
				 
			
@@ -193,8 +193,8 @@ WINDOW_FUNC(only_long)
 
				     const float *pwindow = sce->ics.use_kb_window[1] ? ff_aac_kbd_long_1024 : ff_sine_1024;
			
 
				     float *out = sce->ret_buf;
			
 
				 
			
 
				-    fdsp->vector_fmul       (out,        audio,        lwindow, 1024);
			
 
				-    dsp->vector_fmul_reverse(out + 1024, audio + 1024, pwindow, 1024);
			
 
				+    fdsp->vector_fmul        (out,        audio,        lwindow, 1024);
			
 
				+    fdsp->vector_fmul_reverse(out + 1024, audio + 1024, pwindow, 1024);
			
 
				 }
			
 
				 
			
 
				 WINDOW_FUNC(long_start)
			
@@ -205,7 +205,7 @@ WINDOW_FUNC(long_start)
 
				 
			
 
				     fdsp->vector_fmul(out, audio, lwindow, 1024);
			
 
				     memcpy(out + 1024, audio + 1024, sizeof(out[0]) * 448);
			
 
				-    dsp->vector_fmul_reverse(out + 1024 + 448, audio + 1024 + 448, swindow, 128);
			
 
				+    fdsp->vector_fmul_reverse(out + 1024 + 448, audio + 1024 + 448, swindow, 128);
			
 
				     memset(out + 1024 + 576, 0, sizeof(out[0]) * 448);
			
 
				 }
			
 
				 
			
@@ -218,7 +218,7 @@ WINDOW_FUNC(long_stop)
 
				     memset(out, 0, sizeof(out[0]) * 448);
			
 
				     fdsp->vector_fmul(out + 448, audio + 448, swindow, 128);
			
 
				     memcpy(out + 576, audio + 576, sizeof(out[0]) * 448);
			
 
				-    dsp->vector_fmul_reverse(out + 1024, audio + 1024, lwindow, 1024);
			
 
				+    fdsp->vector_fmul_reverse(out + 1024, audio + 1024, lwindow, 1024);
			
 
				 }
			
 
				 
			
 
				 WINDOW_FUNC(eight_short)
			
@@ -230,15 +230,15 @@ WINDOW_FUNC(eight_short)
 
				     int w;
			
 
				 
			
 
				     for (w = 0; w < 8; w++) {
			
 
				-        fdsp->vector_fmul       (out, in, w ? pwindow : swindow, 128);
			
 
				+        fdsp->vector_fmul        (out, in, w ? pwindow : swindow, 128);
			
 
				         out += 128;
			
 
				         in  += 128;
			
 
				-        dsp->vector_fmul_reverse(out, in, swindow, 128);
			
 
				+        fdsp->vector_fmul_reverse(out, in, swindow, 128);
			
 
				         out += 128;
			
 
				     }
			
 
				 }
			
 
				 
			
 
				-static void (*const apply_window[4])(DSPContext *dsp, AVFloatDSPContext *fdsp,
			
 
				+static void (*const apply_window[4])(AVFloatDSPContext *fdsp,
			
 
				                                      SingleChannelElement *sce,
			
 
				                                      const float *audio) = {
			
 
				     [ONLY_LONG_SEQUENCE]   = apply_only_long_window,
			
@@ -253,7 +253,7 @@ static void apply_window_and_mdct(AACEncContext *s, SingleChannelElement *sce,
 
				     int i;
			
 
				     float *output = sce->ret_buf;
			
 
				 
			
 
				-    apply_window[sce->ics.window_sequence[0]](&s->dsp, &s->fdsp, sce, audio);
			
 
				+    apply_window[sce->ics.window_sequence[0]](&s->fdsp, sce, audio);
			
 
				 
			
 
				     if (sce->ics.window_sequence[0] != EIGHT_SHORT_SEQUENCE)
			
 
				         s->mdct1024.mdct_calc(&s->mdct1024, sce->coeffs, output);
			
@@ -692,7 +692,6 @@ static av_cold int dsp_init(AVCodecContext *avctx, AACEncContext *s)
 
				 {
			
 
				     int ret = 0;
			
 
				 
			
 
				-    ff_dsputil_init(&s->dsp, avctx);
			
 
				     avpriv_float_dsp_init(&s->fdsp, avctx->flags & CODEC_FLAG_BITEXACT);
			
 
				 
			
 
				     // window init
			
--- a/libavcodec/aacenc.h
+++ b/libavcodec/aacenc.h
@@ -61,7 +61,6 @@ typedef struct AACEncContext {
 
				     PutBitContext pb;
			
 
				     FFTContext mdct1024;                         ///< long (1024 samples) frame transform context
			
 
				     FFTContext mdct128;                          ///< short (128 samples) frame transform context
			
 
				-    DSPContext  dsp;
			
 
				     AVFloatDSPContext fdsp;
			
 
				     float *planar_samples[6];                    ///< saved preprocessed input
			
 
				 
			
--- a/libavcodec/aacsbr.c
+++ b/libavcodec/aacsbr.c
@@ -1156,7 +1156,7 @@ static void sbr_dequant(SpectralBandReplication *sbr, int id_aac)
 
				  * @param   x       pointer to the beginning of the first sample window
			
 
				  * @param   W       array of complex-valued samples split into subbands
			
 
				  */
			
 
				-static void sbr_qmf_analysis(DSPContext *dsp, FFTContext *mdct,
			
 
				+static void sbr_qmf_analysis(AVFloatDSPContext *dsp, FFTContext *mdct,
			
 
				                              SBRDSPContext *sbrdsp, const float *in, float *x,
			
 
				                              float z[320], float W[2][32][32][2], int buf_idx)
			
 
				 {
			
@@ -1668,7 +1668,7 @@ void ff_sbr_apply(AACContext *ac, SpectralBandReplication *sbr, int id_aac,
 
				     }
			
 
				     for (ch = 0; ch < nch; ch++) {
			
 
				         /* decode channel */
			
 
				-        sbr_qmf_analysis(&ac->dsp, &sbr->mdct_ana, &sbr->dsp, ch ? R : L, sbr->data[ch].analysis_filterbank_samples,
			
 
				+        sbr_qmf_analysis(&ac->fdsp, &sbr->mdct_ana, &sbr->dsp, ch ? R : L, sbr->data[ch].analysis_filterbank_samples,
			
 
				                          (float*)sbr->qmf_filter_scratch,
			
 
				                          sbr->data[ch].W, sbr->data[ch].Ypos);
			
 
				         sbr_lf_gen(ac, sbr, sbr->X_low, sbr->data[ch].W, sbr->data[ch].Ypos);
			
--- a/libavcodec/arm/Makefile
+++ b/libavcodec/arm/Makefile
@@ -56,9 +56,6 @@ ARMV6-OBJS                             += arm/dsputil_init_armv6.o      \
 
				 
			
 
				 VFP-OBJS-$(HAVE_ARMV6)                 += arm/fmtconvert_vfp.o
			
 
				 
			
 
				-VFP-OBJS                               += arm/dsputil_vfp.o             \
			
 
				-                                          arm/dsputil_init_vfp.o        \
			
 
				-
			
 
				 NEON-OBJS-$(CONFIG_FFT)                += arm/fft_neon.o                \
			
 
				                                           arm/fft_fixed_neon.o          \
			
 
				 
			
--- a/libavcodec/arm/dsputil_init_arm.c
+++ b/libavcodec/arm/dsputil_init_arm.c
@@ -121,6 +121,5 @@ void ff_dsputil_init_arm(DSPContext* c, AVCodecContext *avctx)
 
				 
			
 
				     if (have_armv5te(cpu_flags)) ff_dsputil_init_armv5te(c, avctx);
			
 
				     if (have_armv6(cpu_flags))   ff_dsputil_init_armv6(c, avctx);
			
 
				-    if (have_vfp(cpu_flags))     ff_dsputil_init_vfp(c, avctx);
			
 
				     if (have_neon(cpu_flags))    ff_dsputil_init_neon(c, avctx);
			
 
				 }
			
--- a/libavcodec/arm/dsputil_init_neon.c
+++ b/libavcodec/arm/dsputil_init_neon.c
@@ -144,8 +144,6 @@ void ff_avg_h264_chroma_mc2_neon(uint8_t *, uint8_t *, int, int, int, int);
 
				 
			
 
				 void ff_butterflies_float_neon(float *v1, float *v2, int len);
			
 
				 float ff_scalarproduct_float_neon(const float *v1, const float *v2, int len);
			
 
				-void ff_vector_fmul_reverse_neon(float *dst, const float *src0,
			
 
				-                                 const float *src1, int len);
			
 
				 
			
 
				 void ff_vector_clipf_neon(float *dst, const float *src, float min, float max,
			
 
				                           int len);
			
@@ -298,7 +296,6 @@ void ff_dsputil_init_neon(DSPContext *c, AVCodecContext *avctx)
 
				 
			
 
				     c->butterflies_float          = ff_butterflies_float_neon;
			
 
				     c->scalarproduct_float        = ff_scalarproduct_float_neon;
			
 
				-    c->vector_fmul_reverse        = ff_vector_fmul_reverse_neon;
			
 
				     c->vector_clipf               = ff_vector_clipf_neon;
			
 
				     c->vector_clip_int32          = ff_vector_clip_int32_neon;
			
 
				 
			
--- a/libavcodec/arm/dsputil_init_vfp.c
+++ b/libavcodec/arm/dsputil_init_vfp.c
@@ -1,30 +0,0 @@
 
				-/*
			
 
				- * Copyright (c) 2008 Siarhei Siamashka <ssvb@users.sourceforge.net>
			
 
				- *
			
 
				- * This file is part of FFmpeg.
			
 
				- *
			
 
				- * FFmpeg is free software; you can redistribute it and/or
			
 
				- * modify it under the terms of the GNU Lesser General Public
			
 
				- * License as published by the Free Software Foundation; either
			
 
				- * version 2.1 of the License, or (at your option) any later version.
			
 
				- *
			
 
				- * FFmpeg is distributed in the hope that it will be useful,
			
 
				- * but WITHOUT ANY WARRANTY; without even the implied warranty of
			
 
				- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
			
 
				- * Lesser General Public License for more details.
			
 
				- *
			
 
				- * You should have received a copy of the GNU Lesser General Public
			
 
				- * License along with FFmpeg; if not, write to the Free Software
			
 
				- * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
			
 
				- */
			
 
				-
			
 
				-#include "libavcodec/dsputil.h"
			
 
				-#include "dsputil_arm.h"
			
 
				-
			
 
				-void ff_vector_fmul_reverse_vfp(float *dst, const float *src0,
			
 
				-                                const float *src1, int len);
			
 
				-
			
 
				-void ff_dsputil_init_vfp(DSPContext* c, AVCodecContext *avctx)
			
 
				-{
			
 
				-    c->vector_fmul_reverse = ff_vector_fmul_reverse_vfp;
			
 
				-}
			
--- a/libavcodec/arm/dsputil_neon.S
+++ b/libavcodec/arm/dsputil_neon.S
@@ -556,30 +556,6 @@ NOVFP   vmov.32         r0,  d0[0]
 
				         bx              lr
			
 
				 endfunc
			
 
				 
			
 
				-function ff_vector_fmul_reverse_neon, export=1
			
 
				-        add             r2,  r2,  r3,  lsl #2
			
 
				-        sub             r2,  r2,  #32
			
 
				-        mov             r12, #-32
			
 
				-        vld1.32         {q0-q1},  [r1,:128]!
			
 
				-        vld1.32         {q2-q3},  [r2,:128], r12
			
 
				-1:      pld             [r1, #32]
			
 
				-        vrev64.32       q3,  q3
			
 
				-        vmul.f32        d16, d0,  d7
			
 
				-        vmul.f32        d17, d1,  d6
			
 
				-        pld             [r2, #-32]
			
 
				-        vrev64.32       q2,  q2
			
 
				-        vmul.f32        d18, d2,  d5
			
 
				-        vmul.f32        d19, d3,  d4
			
 
				-        subs            r3,  r3,  #8
			
 
				-        beq             2f
			
 
				-        vld1.32         {q0-q1},  [r1,:128]!
			
 
				-        vld1.32         {q2-q3},  [r2,:128], r12
			
 
				-        vst1.32         {q8-q9},  [r0,:128]!
			
 
				-        b               1b
			
 
				-2:      vst1.32         {q8-q9},  [r0,:128]!
			
 
				-        bx              lr
			
 
				-endfunc
			
 
				-
			
 
				 function ff_vector_clipf_neon, export=1
			
 
				 VFP     vdup.32         q1,  d0[1]
			
 
				 VFP     vdup.32         q0,  d0[0]
			
--- a/libavcodec/arm/dsputil_vfp.S
+++ b/libavcodec/arm/dsputil_vfp.S
@@ -1,106 +0,0 @@
 
				-/*
			
 
				- * Copyright (c) 2008 Siarhei Siamashka <ssvb@users.sourceforge.net>
			
 
				- *
			
 
				- * This file is part of FFmpeg.
			
 
				- *
			
 
				- * FFmpeg is free software; you can redistribute it and/or
			
 
				- * modify it under the terms of the GNU Lesser General Public
			
 
				- * License as published by the Free Software Foundation; either
			
 
				- * version 2.1 of the License, or (at your option) any later version.
			
 
				- *
			
 
				- * FFmpeg is distributed in the hope that it will be useful,
			
 
				- * but WITHOUT ANY WARRANTY; without even the implied warranty of
			
 
				- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
			
 
				- * Lesser General Public License for more details.
			
 
				- *
			
 
				- * You should have received a copy of the GNU Lesser General Public
			
 
				- * License along with FFmpeg; if not, write to the Free Software
			
 
				- * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
			
 
				- */
			
 
				-
			
 
				-#include "config.h"
			
 
				-#include "libavutil/arm/asm.S"
			
 
				-
			
 
				-/*
			
 
				- * VFP is a floating point coprocessor used in some ARM cores. VFP11 has 1 cycle
			
 
				- * throughput for almost all the instructions (except for double precision
			
 
				- * arithmetics), but rather high latency. Latency is 4 cycles for loads and 8 cycles
			
 
				- * for arithmetic operations. Scheduling code to avoid pipeline stalls is very
			
 
				- * important for performance. One more interesting feature is that VFP has
			
 
				- * independent load/store and arithmetics pipelines, so it is possible to make
			
 
				- * them work simultaneously and get more than 1 operation per cycle. Load/store
			
 
				- * pipeline can process 2 single precision floating point values per cycle and
			
 
				- * supports bulk loads and stores for large sets of registers. Arithmetic operations
			
 
				- * can be done on vectors, which allows to keep the arithmetics pipeline busy,
			
 
				- * while the processor may issue and execute other instructions. Detailed
			
 
				- * optimization manuals can be found at http://www.arm.com
			
 
				- */
			
 
				-
			
 
				-/**
			
 
				- * ARM VFP optimized implementation of 'vector_fmul_reverse_c' function.
			
 
				- * Assume that len is a positive number and is multiple of 8
			
 
				- */
			
 
				-@ void ff_vector_fmul_reverse_vfp(float *dst, const float *src0,
			
 
				-@                                 const float *src1, int len)
			
 
				-function ff_vector_fmul_reverse_vfp, export=1
			
 
				-        vpush           {d8-d15}
			
 
				-        add             r2,  r2,  r3, lsl #2
			
 
				-        vldmdb          r2!, {s0-s3}
			
 
				-        vldmia          r1!, {s8-s11}
			
 
				-        vldmdb          r2!, {s4-s7}
			
 
				-        vldmia          r1!, {s12-s15}
			
 
				-        vmul.f32        s8,  s3,  s8
			
 
				-        vmul.f32        s9,  s2,  s9
			
 
				-        vmul.f32        s10, s1,  s10
			
 
				-        vmul.f32        s11, s0,  s11
			
 
				-1:
			
 
				-        subs            r3,  r3,  #16
			
 
				-        it              ge
			
 
				-        vldmdbge        r2!, {s16-s19}
			
 
				-        vmul.f32        s12, s7,  s12
			
 
				-        it              ge
			
 
				-        vldmiage        r1!, {s24-s27}
			
 
				-        vmul.f32        s13, s6,  s13
			
 
				-        it              ge
			
 
				-        vldmdbge        r2!, {s20-s23}
			
 
				-        vmul.f32        s14, s5,  s14
			
 
				-        it              ge
			
 
				-        vldmiage        r1!, {s28-s31}
			
 
				-        vmul.f32        s15, s4,  s15
			
 
				-        it              ge
			
 
				-        vmulge.f32      s24, s19, s24
			
 
				-        it              gt
			
 
				-        vldmdbgt        r2!, {s0-s3}
			
 
				-        it              ge
			
 
				-        vmulge.f32      s25, s18, s25
			
 
				-        vstmia          r0!, {s8-s13}
			
 
				-        it              ge
			
 
				-        vmulge.f32      s26, s17, s26
			
 
				-        it              gt
			
 
				-        vldmiagt        r1!, {s8-s11}
			
 
				-        itt             ge
			
 
				-        vmulge.f32      s27, s16, s27
			
 
				-        vmulge.f32      s28, s23, s28
			
 
				-        it              gt
			
 
				-        vldmdbgt        r2!, {s4-s7}
			
 
				-        it              ge
			
 
				-        vmulge.f32      s29, s22, s29
			
 
				-        vstmia          r0!, {s14-s15}
			
 
				-        ittt            ge
			
 
				-        vmulge.f32      s30, s21, s30
			
 
				-        vmulge.f32      s31, s20, s31
			
 
				-        vmulge.f32      s8,  s3,  s8
			
 
				-        it              gt
			
 
				-        vldmiagt        r1!, {s12-s15}
			
 
				-        itttt           ge
			
 
				-        vmulge.f32      s9,  s2,  s9
			
 
				-        vmulge.f32      s10, s1,  s10
			
 
				-        vstmiage        r0!, {s24-s27}
			
 
				-        vmulge.f32      s11, s0,  s11
			
 
				-        it              ge
			
 
				-        vstmiage        r0!, {s28-s31}
			
 
				-        bgt             1b
			
 
				-
			
 
				-        vpop            {d8-d15}
			
 
				-        bx              lr
			
 
				-endfunc