diff --git a/CMakeLists.txt b/CMakeLists.txt index aa7b1ae17..37128d7ff 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -576,6 +576,9 @@ if(NOT OPUS_DISABLE_INTRINSICS) add_sources_group(opus celt ${celt_sources_arm_neon_intr}) add_sources_group(opus silk ${silk_sources_arm_neon_intr}) + if (NOT OPUS_FIXED_POINT) + add_sources_group(opus silk ${silk_sources_float_arm_neon_intr}) + endif() if (OPUS_DNN) add_sources_group(opus lpcnet ${dnn_sources_arm_neon}) endif() diff --git a/Makefile.am b/Makefile.am index 1cf73822c..f196862ba 100644 --- a/Makefile.am +++ b/Makefile.am @@ -45,6 +45,9 @@ endif if HAVE_AVX2 SILK_SOURCES += $(SILK_SOURCES_FLOAT_AVX2) endif +if HAVE_ARM_NEON_INTR +SILK_SOURCES += $(SILK_SOURCES_FLOAT_ARM_NEON_INTR) +endif endif if DISABLE_FLOAT_API @@ -508,7 +511,8 @@ if HAVE_ARM_NEON_INTR ARM_NEON_INTR_OBJ = $(CELT_SOURCES_ARM_NEON_INTR:.c=.lo) \ $(SILK_SOURCES_ARM_NEON_INTR:.c=.lo) \ $(DNN_SOURCES_NEON:.c=.lo) \ - $(SILK_SOURCES_FIXED_ARM_NEON_INTR:.c=.lo) + $(SILK_SOURCES_FIXED_ARM_NEON_INTR:.c=.lo) \ + $(SILK_SOURCES_FLOAT_ARM_NEON_INTR:.c=.lo) $(ARM_NEON_INTR_OBJ): CFLAGS += \ $(OPUS_ARM_NEON_INTR_CFLAGS) $(NE10_CFLAGS) endif diff --git a/cmake/OpusSources.cmake b/cmake/OpusSources.cmake index 06cb0d4fa..0c3adabee 100644 --- a/cmake/OpusSources.cmake +++ b/cmake/OpusSources.cmake @@ -15,6 +15,8 @@ get_opus_sources(SILK_SOURCES_FIXED_SSE4_1 silk_sources.mk silk_sources_fixed_sse4_1) get_opus_sources(SILK_SOURCES_AVX2 silk_sources.mk silk_sources_avx2) get_opus_sources(SILK_SOURCES_FLOAT_AVX2 silk_sources.mk silk_sources_float_avx2) +get_opus_sources(SILK_SOURCES_FLOAT_ARM_NEON_INTR silk_sources.mk + silk_sources_float_arm_neon_intr) get_opus_sources(SILK_SOURCES_ARM_RTCD silk_sources.mk silk_sources_arm_rtcd) get_opus_sources(SILK_SOURCES_ARM_NEON_INTR silk_sources.mk silk_sources_arm_neon_intr) diff --git a/silk/arm/SigProc_FLP_arm.h b/silk/arm/SigProc_FLP_arm.h new file mode 100644 index 000000000..3264d3a48 --- /dev/null +++ b/silk/arm/SigProc_FLP_arm.h @@ -0,0 +1,101 @@ +/* Copyright (c) 2026 Xiph.Org Foundation */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifndef SILK_SIGPROC_FLP_ARM_H +#define SILK_SIGPROC_FLP_ARM_H + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + +#ifndef FIXED_POINT + +#if defined(OPUS_ARM_MAY_HAVE_NEON_INTR) + +double silk_inner_product_FLP_neon( + const silk_float *data1, + const silk_float *data2, + opus_int dataSize +); + +#if defined(OPUS_ARM_PRESUME_NEON_INTR) + +#define OVERRIDE_inner_product_FLP +#define silk_inner_product_FLP(data1, data2, dataSize, arch) \ + ((void)arch, silk_inner_product_FLP_neon(data1, data2, dataSize)) + +#elif defined(OPUS_HAVE_RTCD) + +#define OVERRIDE_inner_product_FLP +extern double (*const SILK_INNER_PRODUCT_FLP_IMPL[OPUS_ARCHMASK + 1])( + const silk_float *data1, + const silk_float *data2, + opus_int dataSize +); +#define silk_inner_product_FLP(data1, data2, dataSize, arch) \ + ((void)arch, (*SILK_INNER_PRODUCT_FLP_IMPL[(arch) & OPUS_ARCHMASK])(data1, data2, dataSize)) + +#endif + +void silk_warped_autocorrelation_FLP_neon( + silk_float *corr, + const silk_float *input, + const silk_float warping, + const opus_int length, + const opus_int order +); + +/* silk_warped_autocorrelation_FLP has no arch argument at its call site, so it + cannot be dispatched through an arch-indexed RTCD table. We therefore only + override it on PRESUME-NEON targets (e.g. aarch64, where NEON is baseline); + ARMv7 runtime-detection builds keep the C reference. Adding an arch + parameter to the signature would enable RTCD here too. */ +#if defined(OPUS_ARM_PRESUME_NEON_INTR) + +#define OVERRIDE_warped_autocorrelation_FLP +#define silk_warped_autocorrelation_FLP(corr, input, warping, length, order) \ + silk_warped_autocorrelation_FLP_neon(corr, input, warping, length, order) + +#endif + +/* silk_energy_FLP also takes no arch argument -> PRESUME only. */ +double silk_energy_FLP_neon( + const silk_float *data, + opus_int dataSize +); + +#if defined(OPUS_ARM_PRESUME_NEON_INTR) + +#define OVERRIDE_energy_FLP +#define silk_energy_FLP(data, dataSize) silk_energy_FLP_neon(data, dataSize) + +#endif + +#endif /* OPUS_ARM_MAY_HAVE_NEON_INTR */ + +#endif /* !FIXED_POINT */ + +#endif /* SILK_SIGPROC_FLP_ARM_H */ diff --git a/silk/arm/arm_silk_map.c b/silk/arm/arm_silk_map.c index a91f79b59..97be158fc 100644 --- a/silk/arm/arm_silk_map.c +++ b/silk/arm/arm_silk_map.c @@ -31,6 +31,9 @@ POSSIBILITY OF SUCH DAMAGE. #include "main_FIX.h" #include "NSQ.h" #include "SigProc_FIX.h" +#ifndef FIXED_POINT +# include "SigProc_FLP.h" +#endif #if defined(OPUS_HAVE_RTCD) @@ -125,4 +128,21 @@ void (*const SILK_WARPED_AUTOCORRELATION_FIX_IMPL[OPUS_ARCHMASK + 1])( # endif +# if !defined(FIXED_POINT) && \ + defined(OPUS_ARM_MAY_HAVE_NEON_INTR) && !defined(OPUS_ARM_PRESUME_NEON_INTR) + +double (*const SILK_INNER_PRODUCT_FLP_IMPL[OPUS_ARCHMASK + 1])( + const silk_float *data1, + const silk_float *data2, + opus_int dataSize +) = { + silk_inner_product_FLP_c, /* ARMv4 */ + silk_inner_product_FLP_c, /* EDSP */ + silk_inner_product_FLP_c, /* Media */ + silk_inner_product_FLP_neon, /* Neon */ + silk_inner_product_FLP_neon, /* dotprod */ +}; + +# endif + #endif /* OPUS_HAVE_RTCD */ diff --git a/silk/float/SigProc_FLP.h b/silk/float/SigProc_FLP.h index 783099beb..0928b2fac 100644 --- a/silk/float/SigProc_FLP.h +++ b/silk/float/SigProc_FLP.h @@ -135,11 +135,15 @@ double silk_inner_product_FLP_c( /* sum of squares of a silk_float array, with result as double */ -double silk_energy_FLP( +double silk_energy_FLP_c( const silk_float *data, opus_int dataSize ); +#ifndef OVERRIDE_energy_FLP +#define silk_energy_FLP(data, dataSize) (silk_energy_FLP_c(data, dataSize)) +#endif + /********************************************************************/ /* MACROS */ /********************************************************************/ diff --git a/silk/float/arm/energy_FLP_neon_intr.c b/silk/float/arm/energy_FLP_neon_intr.c new file mode 100644 index 000000000..3428aa2fc --- /dev/null +++ b/silk/float/arm/energy_FLP_neon_intr.c @@ -0,0 +1,73 @@ +/* Copyright (c) 2026 Xiph.Org Foundation */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + +#include +#ifdef OPUS_CHECK_ASM +# include +#endif +#include "SigProc_FLP.h" + +/* NEON implementation of silk_energy_FLP (sum of squares). Like the C + reference (and the inner_product kernel) it accumulates in double: each + f32 is widened to f64 before squaring and summed in two float64x2 lanes, + so the result is within rounding of the scalar reference. */ +double silk_energy_FLP_neon( + const silk_float *data, + opus_int dataSize +) +{ + opus_int i; + float64x2_t acc0 = vdupq_n_f64( 0.0 ); + float64x2_t acc1 = vdupq_n_f64( 0.0 ); + double result; + + for( i = 0; i + 4 <= dataSize; i += 4 ) { + float32x4_t x = vld1q_f32( &data[ i ] ); + float64x2_t lo = vcvt_f64_f32( vget_low_f32( x ) ); + float64x2_t hi = vcvt_f64_f32( vget_high_f32( x ) ); + acc0 = vfmaq_f64( acc0, lo, lo ); + acc1 = vfmaq_f64( acc1, hi, hi ); + } + result = vaddvq_f64( vaddq_f64( acc0, acc1 ) ); + + for( ; i < dataSize; i++ ) { + result += data[ i ] * (double)data[ i ]; + } + +#ifdef OPUS_CHECK_ASM + /* Float NEON kernel: f64 accumulation in a different order than the C + reference, so within-rounding rather than bit-exact (cf. fixed-point). */ + { + double result_c = silk_energy_FLP_c( data, dataSize ); + celt_assert( fabs( result - result_c ) <= 1e-5 * ( fabs( result_c ) + 1e-30 ) ); + } +#endif + return result; +} diff --git a/silk/float/arm/inner_product_FLP_neon_intr.c b/silk/float/arm/inner_product_FLP_neon_intr.c new file mode 100644 index 000000000..d45a3de70 --- /dev/null +++ b/silk/float/arm/inner_product_FLP_neon_intr.c @@ -0,0 +1,79 @@ +/* Copyright (c) 2026 Xiph.Org Foundation */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + +#include +#ifdef OPUS_CHECK_ASM +# include +#endif +#include "SigProc_FLP.h" + +/* NEON implementation of silk_inner_product_FLP. The C reference accumulates + the products in double precision; we widen each float operand to double + before multiplying (matching data1[i]*(double)data2[i]) and accumulate in + two float64x2 lanes, exactly as the x86 AVX2 implementation does. The + result is within rounding of the C reference (well below float precision). */ +double silk_inner_product_FLP_neon( + const silk_float *data1, + const silk_float *data2, + opus_int dataSize +) +{ + opus_int i; + float64x2_t acc0 = vdupq_n_f64( 0.0 ); + float64x2_t acc1 = vdupq_n_f64( 0.0 ); + double result; + + for( i = 0; i + 4 <= dataSize; i += 4 ) { + float32x4_t a = vld1q_f32( &data1[ i ] ); + float32x4_t b = vld1q_f32( &data2[ i ] ); + acc0 = vfmaq_f64( acc0, vcvt_f64_f32( vget_low_f32( a ) ), vcvt_f64_f32( vget_low_f32( b ) ) ); + acc1 = vfmaq_f64( acc1, vcvt_f64_f32( vget_high_f32( a ) ), vcvt_f64_f32( vget_high_f32( b ) ) ); + } + result = vaddvq_f64( vaddq_f64( acc0, acc1 ) ); + + for( ; i < dataSize; i++ ) { + result += data1[ i ] * (double)data2[ i ]; + } + +#ifdef OPUS_CHECK_ASM + /* Float NEON kernels accumulate in double like the C reference but in a + different summation order, so (unlike fixed-point kernels) they are not + bit-exact. The rounding error of an f64 dot product is bounded by + ~N*eps*sum|terms|, so we check against the accumulation magnitude rather + than the result (which can be near zero from cancellation). */ + { + double result_c = silk_inner_product_FLP_c( data1, data2, dataSize ); + double sum_abs = 0; + for( i = 0; i < dataSize; i++ ) sum_abs += fabs( data1[ i ] * (double)data2[ i ] ); + celt_assert( fabs( result - result_c ) <= 1e-6 * sum_abs + 1e-30 ); + } +#endif + return result; +} diff --git a/silk/float/arm/warped_autocorrelation_FLP_neon_intr.c b/silk/float/arm/warped_autocorrelation_FLP_neon_intr.c new file mode 100644 index 000000000..1cc25ec8f --- /dev/null +++ b/silk/float/arm/warped_autocorrelation_FLP_neon_intr.c @@ -0,0 +1,158 @@ +/* Copyright (c) 2026 Xiph.Org Foundation */ +/* + Redistribution and use in source and binary forms, with or without + modification, are permitted provided that the following conditions + are met: + + - Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + + - Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + + THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS + ``AS IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT + LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR + A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE FOUNDATION OR + CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, + EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, + PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR + PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF + LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING + NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS + SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. +*/ + +#ifdef HAVE_CONFIG_H +#include "config.h" +#endif + +#include +#ifdef OPUS_CHECK_ASM +# include +# include "main_FLP.h" +#endif +#include "SigProc_FLP.h" +#include "define.h" + +/* NEON implementation of silk_warped_autocorrelation_FLP. + * + * The reference runs a serial all-pass cascade per sample (loop-carried) and + * accumulates C[i] += state[0] * state[i] interleaved with that cascade. + * The all-pass chain is inherently serial, so it stays scalar in double + * precision -- producing the SAME state[] as the C reference bit-for-bit -- + * while the per-lag correlation accumulation is a SAXPY + * (C[i] += input[n]*state[i], where state[0]==input[n]) that we vectorise. + * + * Design notes / why this layout: + * + * - The accumulation is done INLINE inside the all-pass loop, consuming the + * freshly computed tmp1/tmp2 straight from registers. An earlier version + * instead ran a separate vector pass that re-read the whole state[] array + * from the stack; that second traversal dominates at order 24 + * (MAX_SHAPE_LPC_ORDER == 24) and turned the gain into a ~0.85x regression. + * Folding the SAXPY back in removes the second pass entirely. + * + * - The all-pass advances two taps per iteration (state[i], state[i+1]), so + * we unroll two sections (four lags, i..i+3) per iteration: C[i..i+3] are + * loaded once into two float64x2 accumulators, both 2-wide fma are fused, + * and the pair is written back once. This halves the C[] memory traffic + * versus a one-section-per-iteration loop and shrinks the per-sample branch + * count. The remaining tail (one section when order % 4 == 2) uses the same + * 2-wide step. + * + * Bit-exactness: every C[k] receives exactly one fused C[k] += input[n]*state[k] + * per sample, in the same n-order as the C reference (which compiles to scalar + * fmadd). f32 operands widen to f64 exactly, and the 2-way fma.2d updates C[i], + * C[i+1] on independent lanes, so the result matches silk_warped_autocorrelation + * _FLP_c bit-for-bit. */ +void silk_warped_autocorrelation_FLP_neon( + silk_float *corr, /* O Result [order + 1] */ + const silk_float *input, /* I Input data to correlate */ + const silk_float warping, /* I Warping coefficient */ + const opus_int length, /* I Length of input */ + const opus_int order /* I Correlation order (even) */ +) +{ + opus_int n, i; + double tmp1, tmp2; + double state[ MAX_SHAPE_LPC_ORDER + 1 ] = { 0 }; + double C[ MAX_SHAPE_LPC_ORDER + 1 ] = { 0 }; + const double w = (double)warping; + + silk_assert( ( order & 1 ) == 0 ); + silk_assert( order <= MAX_SHAPE_LPC_ORDER ); + + /* Loop over samples */ + for( n = 0; n < length; n++ ) { + /* state[0] == input[n] for this sample; broadcast once. */ + const double d = (double)input[ n ]; + const float64x2_t sinv = vdupq_n_f64( d ); + tmp1 = d; + + /* Two all-pass sections (four lags) per iteration. */ + for( i = 0; i + 4 <= order; i += 4 ) { + /* Section i: compute tmp2 from state[i..i+1], store state[i]. */ + tmp2 = state[ i ] + w * state[ i + 1 ] - w * tmp1; + state[ i ] = tmp1; + /* Accumulate lag i+0 (new state[i]==tmp1) and lag i+1 (tmp2), + still holding the pre-reassignment tmp1. C[i..i+1] stay in a + register across the next section. */ + float64x2_t sv0 = vsetq_lane_f64( tmp2, vsetq_lane_f64( tmp1, vdupq_n_f64( 0 ), 0 ), 1 ); + float64x2_t cv0 = vld1q_f64( &C[ i ] ); + cv0 = vfmaq_f64( cv0, sv0, sinv ); + + tmp1 = state[ i + 1 ] + w * state[ i + 2 ] - w * tmp2; + state[ i + 1 ] = tmp2; + + /* Section i+2: compute tmp2 from state[i+2..i+3], store state[i+2]. */ + tmp2 = state[ i + 2 ] + w * state[ i + 3 ] - w * tmp1; + state[ i + 2 ] = tmp1; + float64x2_t sv1 = vsetq_lane_f64( tmp2, vsetq_lane_f64( tmp1, vdupq_n_f64( 0 ), 0 ), 1 ); + float64x2_t cv1 = vld1q_f64( &C[ i + 2 ] ); + cv1 = vfmaq_f64( cv1, sv1, sinv ); + + tmp1 = state[ i + 3 ] + w * state[ i + 4 ] - w * tmp2; + state[ i + 3 ] = tmp2; + + /* Write back both 2-lag blocks. */ + vst1q_f64( &C[ i ], cv0 ); + vst1q_f64( &C[ i + 2 ], cv1 ); + } + + /* Remaining sections (used when order % 4 == 2). */ + for( ; i < order; i += 2 ) { + tmp2 = state[ i ] + w * state[ i + 1 ] - w * tmp1; + state[ i ] = tmp1; + float64x2_t sv = vsetq_lane_f64( tmp2, vsetq_lane_f64( tmp1, vdupq_n_f64( 0 ), 0 ), 1 ); + float64x2_t cv = vld1q_f64( &C[ i ] ); + cv = vfmaq_f64( cv, sv, sinv ); + vst1q_f64( &C[ i ], cv ); + tmp1 = state[ i + 1 ] + w * state[ i + 2 ] - w * tmp2; + state[ i + 1 ] = tmp2; + } + + /* Final lag (state[order] == tmp1 from the last section). */ + state[ order ] = tmp1; + C[ order ] += d * tmp1; + } + + /* Copy correlations in silk_float output format */ + for( i = 0; i < order + 1; i++ ) { + corr[ i ] = (silk_float)C[ i ]; + } + +#ifdef OPUS_CHECK_ASM + /* The all-pass state is computed in double identically to the C reference, + and each C[k] accumulates input[n]*state[k] once per sample in f64 -- so + the result matches silk_warped_autocorrelation_FLP_c bit-for-bit. */ + { + silk_float corr_c[ MAX_SHAPE_LPC_ORDER + 1 ]; + silk_warped_autocorrelation_FLP_c( corr_c, input, warping, length, order ); + for( i = 0; i < order + 1; i++ ) { + celt_assert( corr[ i ] == corr_c[ i ] ); + } + } +#endif +} diff --git a/silk/float/energy_FLP.c b/silk/float/energy_FLP.c index 7bc7173c9..f9af5957b 100644 --- a/silk/float/energy_FLP.c +++ b/silk/float/energy_FLP.c @@ -32,7 +32,7 @@ POSSIBILITY OF SUCH DAMAGE. #include "SigProc_FLP.h" /* sum of squares of a silk_float array, with result as double */ -double silk_energy_FLP( +double silk_energy_FLP_c( const silk_float *data, opus_int dataSize ) diff --git a/silk/float/main_FLP.h b/silk/float/main_FLP.h index 8173bb32a..4393ff6e8 100644 --- a/silk/float/main_FLP.h +++ b/silk/float/main_FLP.h @@ -93,13 +93,17 @@ void silk_noise_shape_analysis_FLP( ); /* Autocorrelations for a warped frequency axis */ -void silk_warped_autocorrelation_FLP( +void silk_warped_autocorrelation_FLP_c( silk_float *corr, /* O Result [order + 1] */ const silk_float *input, /* I Input data to correlate */ const silk_float warping, /* I Warping coefficient */ const opus_int length, /* I Length of input */ const opus_int order /* I Correlation order (even) */ ); +#ifndef OVERRIDE_warped_autocorrelation_FLP +#define silk_warped_autocorrelation_FLP(corr, input, warping, length, order) \ + silk_warped_autocorrelation_FLP_c(corr, input, warping, length, order) +#endif /* Calculation of LTP state scaling */ void silk_LTP_scale_ctrl_FLP( diff --git a/silk/float/warped_autocorrelation_FLP.c b/silk/float/warped_autocorrelation_FLP.c index 116dab923..3b37f7784 100644 --- a/silk/float/warped_autocorrelation_FLP.c +++ b/silk/float/warped_autocorrelation_FLP.c @@ -32,7 +32,7 @@ POSSIBILITY OF SUCH DAMAGE. #include "main_FLP.h" /* Autocorrelations for a warped frequency axis */ -void silk_warped_autocorrelation_FLP( +void silk_warped_autocorrelation_FLP_c( silk_float *corr, /* O Result [order + 1] */ const silk_float *input, /* I Input data to correlate */ const silk_float warping, /* I Warping coefficient */ diff --git a/silk/main.h b/silk/main.h index 5ca9aa52a..3db986fc3 100644 --- a/silk/main.h +++ b/silk/main.h @@ -44,6 +44,7 @@ POSSIBILITY OF SUCH DAMAGE. #if (defined(OPUS_ARM_ASM) || defined(OPUS_ARM_MAY_HAVE_NEON_INTR)) #include "arm/NSQ_del_dec_arm.h" +#include "arm/SigProc_FLP_arm.h" #endif /* Convert Left/Right stereo signal to adaptive Mid/Side representation */ diff --git a/silk/meson.build b/silk/meson.build index 35d955784..15f987aed 100644 --- a/silk/meson.build +++ b/silk/meson.build @@ -13,7 +13,7 @@ silk_sources_fixed = sources['SILK_SOURCES_FIXED'] silk_sources_fixed_sse4_1 = sources['SILK_SOURCES_FIXED_SSE4_1'] silk_sources_float_sse4_1 = [] -silk_sources_float_neon_intr = [] +silk_sources_float_neon_intr = sources['SILK_SOURCES_FLOAT_ARM_NEON_INTR'] silk_sources_float_avx2 = sources['SILK_SOURCES_FLOAT_AVX2'] silk_sources_float = sources['SILK_SOURCES_FLOAT'] diff --git a/silk_sources.mk b/silk_sources.mk index 3780b1647..e5ba244d7 100644 --- a/silk_sources.mk +++ b/silk_sources.mk @@ -161,4 +161,9 @@ silk/float/schur_FLP.c \ silk/float/sort_FLP.c SILK_SOURCES_FLOAT_AVX2 = \ -silk/float/x86/inner_product_FLP_avx2.c \ No newline at end of file +silk/float/x86/inner_product_FLP_avx2.c + +SILK_SOURCES_FLOAT_ARM_NEON_INTR = \ +silk/float/arm/inner_product_FLP_neon_intr.c \ +silk/float/arm/warped_autocorrelation_FLP_neon_intr.c \ +silk/float/arm/energy_FLP_neon_intr.c \ No newline at end of file