From bf7d31f14b102a94232112e5688928d5c7dfa70e Mon Sep 17 00:00:00 2001 From: zhangyi101 Date: Mon, 14 Sep 2026 14:32:18 +0800 Subject: [PATCH 1/2] ltpf: add Arm Helium (MVE) SIMD backend Add an M-Profile Vector Extension (MVE) backend for the LTPF pitch-detection hotspot, mirroring the existing NEON and Arm-DSP backends. `dot` / `correlate` use vmlaldavaq_s16 (8-lane int16 -> int64 multiply-accumulate); the integer sum and the (v + 32) >> 6 rounding match the scalar reference exactly, so the output is bit-exact. The header is guarded by `#if (__ARM_FEATURE_MVE & 1)` and is inert on non-MVE targets, so existing scalar / NEON / DSP builds are unchanged. Measured on-device (Cortex-M55 @197 MHz, 16 kHz mono LC3 encode): - LTPF correlate: 15105 -> 8370 cyc/frame (about -44%) - encoder output bit-exact vs the scalar path (FNV-1a checksum identical) Signed-off-by: zhangyi101 --- src/ltpf.c | 1 + src/ltpf_mve.h | 87 ++++++++++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 88 insertions(+) create mode 100644 src/ltpf_mve.h diff --git a/src/ltpf.c b/src/ltpf.c index 08cbae6..5b5fd59 100644 --- a/src/ltpf.c +++ b/src/ltpf.c @@ -19,6 +19,7 @@ #include "ltpf.h" #include "tables.h" +#include "ltpf_mve.h" #include "ltpf_neon.h" #include "ltpf_arm.h" diff --git a/src/ltpf_mve.h b/src/ltpf_mve.h new file mode 100644 index 0000000..2049b50 --- /dev/null +++ b/src/ltpf_mve.h @@ -0,0 +1,87 @@ +/****************************************************************************** + * + * Copyright 2026 Google LLC + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at: + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + ******************************************************************************/ + +/* + * Arm Helium (M-Profile Vector Extension, MVE) backend for the LTPF + * pitch-detection hotspot, targeting Armv8.1-M cores (e.g. Cortex-M55). + * + * `dot()` / `correlate()` accumulate int16 x int16 products into a 64-bit + * integer and apply the rounding `(v + 32) >> 6`. MVE `vmlaldavaq_s16` + * performs an 8-lane int16 x int16 -> int64 multiply-accumulate across the + * vector; the integer sum and the subsequent rounding are therefore + * identical to the scalar reference, i.e. the result is strictly bit-exact. + * + * This mirrors the NEON (`ltpf_neon.h`) and Arm DSP (`ltpf_arm.h`) backends + * and hooks the same `dot` / `correlate` override points. Include this file + * before the NEON/DSP backends in `ltpf.c` so its definitions take + * precedence on MVE targets. + */ + +#if (__ARM_FEATURE_MVE & 1) + +#include + + +/** + * Return dot product of 2 vectors + * a, b, n The 2 vectors of size `n` (> 0 and <= 128) + * return round( sum( a[i] * b[i] ) / 64 ), as float + * + * The size `n` of vectors is a multiple of 16 (so also of 8). + */ +#ifndef dot + +LC3_HOT static inline float mve_dot(const int16_t *a, const int16_t *b, int n) +{ + int64_t v = 0; + + for (int i = 0; i < (n >> 3); i++) { + int16x8_t va = vldrhq_s16(a); a += 8; + int16x8_t vb = vldrhq_s16(b); b += 8; + v = vmlaldavaq_s16(v, va, vb); + } + + int32_t v32 = (v + (1 << 5)) >> 6; + return (float)v32; +} + +#define dot mve_dot +#endif /* dot */ + +/** + * Return vector of correlations + * a, b, n The 2 vectors of size `n` (> 0 and <= 128) + * y, nc Output the correlation vector of size `nc` + * + * Sliding dot product: for each output the `b` window is shifted by one + * sample. Each lag reuses the 8-lane MVE `dot`, matching the scalar + * reference exactly. + */ +#ifndef correlate + +LC3_HOT static void mve_correlate( + const int16_t *a, const int16_t *b, int n, float *y, int nc) +{ + for (const float *ye = y + nc; y < ye; ) + *(y++) = mve_dot(a, b--, n); +} + +#define correlate mve_correlate +#endif /* correlate */ + +#endif /* __ARM_FEATURE_MVE & 1 */ From 1237843b86c772a205ea9e282559dd82a5686f20 Mon Sep 17 00:00:00 2001 From: zhangyi101 Date: Mon, 14 Sep 2026 17:44:15 +0800 Subject: [PATCH 2/2] ltpf: add MVE backend host test Add host coverage for the LTPF Arm Helium (MVE) backend, mirroring the existing `test/arm` and `test/neon` suites. Previously the MVE `dot` / `correlate` paths had no automated coverage: the reference tests run the scalar path on x86, and only ARM and NEON had backend unit tests. `test/mve/mve.h` provides a small host emulation of the two intrinsics used by the backend (`vldrhq_s16`, `vmlaldavaq_s16`), matching the `test/neon/neon.h` and `test/arm/simd32.h` approach, so the backend compiles and runs on the CI host. `test/mve/ltpf_mve.c` compiles `ltpf.c` with `TEST_MVE` defined and checks `mve_dot` / `mve_correlate` against the scalar reference for bit-exactness. `ltpf_mve.h` gains a `TEST_MVE` hook (guard opt-in and the `dot` / `correlate` overrides wrapped in `#ifndef TEST_MVE`) so both the scalar and MVE versions are visible to the test, and so the MVE backend stays inert while the ARM / NEON tests build. Signed-off-by: zhangyi101 --- src/ltpf_mve.h | 11 +++++- test/makefile.mk | 1 + test/mve/ltpf_mve.c | 86 ++++++++++++++++++++++++++++++++++++++++++++ test/mve/makefile.mk | 31 ++++++++++++++++ test/mve/mve.h | 61 +++++++++++++++++++++++++++++++ test/mve/test_mve.c | 32 +++++++++++++++++ 6 files changed, 221 insertions(+), 1 deletion(-) create mode 100644 test/mve/ltpf_mve.c create mode 100644 test/mve/makefile.mk create mode 100644 test/mve/mve.h create mode 100644 test/mve/test_mve.c diff --git a/src/ltpf_mve.h b/src/ltpf_mve.h index 2049b50..24f88dc 100644 --- a/src/ltpf_mve.h +++ b/src/ltpf_mve.h @@ -32,9 +32,12 @@ * precedence on MVE targets. */ -#if (__ARM_FEATURE_MVE & 1) +#if (__ARM_FEATURE_MVE & 1) && !defined(TEST_ARM) && !defined(TEST_NEON) \ + || defined(TEST_MVE) +#ifndef TEST_MVE #include +#endif /* TEST_MVE */ /** @@ -60,7 +63,10 @@ LC3_HOT static inline float mve_dot(const int16_t *a, const int16_t *b, int n) return (float)v32; } +#ifndef TEST_MVE #define dot mve_dot +#endif + #endif /* dot */ /** @@ -81,7 +87,10 @@ LC3_HOT static void mve_correlate( *(y++) = mve_dot(a, b--, n); } +#ifndef TEST_MVE #define correlate mve_correlate +#endif + #endif /* correlate */ #endif /* __ARM_FEATURE_MVE & 1 */ diff --git a/test/makefile.mk b/test/makefile.mk index a7bcb04..d689ccb 100644 --- a/test/makefile.mk +++ b/test/makefile.mk @@ -26,5 +26,6 @@ test-clean: -include $(TEST_DIR)/arm/makefile.mk -include $(TEST_DIR)/neon/makefile.mk +-include $(TEST_DIR)/mve/makefile.mk clean-all: test-clean diff --git a/test/mve/ltpf_mve.c b/test/mve/ltpf_mve.c new file mode 100644 index 0000000..203974c --- /dev/null +++ b/test/mve/ltpf_mve.c @@ -0,0 +1,86 @@ +/****************************************************************************** + * + * Copyright 2026 Google LLC + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at: + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + ******************************************************************************/ + +#include "mve.h" + +#include +#include +#include + +/* -------------------------------------------------------------------------- */ + +#define TEST_MVE +#include + +void lc3_put_bits_generic(lc3_bits_t *a, unsigned b, int c) +{ (void)a, (void)b, (void)c; } + +unsigned lc3_get_bits_generic(struct lc3_bits *a, int b) +{ return (void)a, (void)b, 0; } + +/* -------------------------------------------------------------------------- */ + +static int check_dot() +{ + int16_t x[200]; + for (int i = 0; i < 200; i++) + x[i] = rand() & 0xffff; + + float y = dot(x, x+3, 128); + float y_mve = mve_dot(x, x+3, 128); + if (y != y_mve) + return -1; + + return 0; +} + +static int check_correlate() +{ + int16_t alignas(4) a[500], b[500]; + float y[100], y_mve[100]; + + for (int i = 0; i < 500; i++) { + a[i] = rand() & 0xffff; + b[i] = rand() & 0xffff; + } + + correlate(a, b+200, 128, y, 100); + mve_correlate(a, b+200, 128, y_mve, 100); + if (memcmp(y, y_mve, 100 * sizeof(*y)) != 0) + return -1; + + correlate(a, b+199, 128, y, 99); + mve_correlate(a, b+199, 128, y_mve, 99); + if (memcmp(y, y_mve, 99 * sizeof(*y)) != 0) + return -1; + + return 0; +} + +int check_ltpf(void) +{ + int ret; + + if ((ret = check_dot()) < 0) + return ret; + + if ((ret = check_correlate()) < 0) + return ret; + + return 0; +} diff --git a/test/mve/makefile.mk b/test/mve/makefile.mk new file mode 100644 index 0000000..1bd39e2 --- /dev/null +++ b/test/mve/makefile.mk @@ -0,0 +1,31 @@ +# +# Copyright 2026 Google LLC +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at: +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +test_mve_src += \ + $(TEST_DIR)/mve/test_mve.c \ + $(TEST_DIR)/mve/ltpf_mve.c \ + $(SRC_DIR)/tables.c + +test_mve_include += $(SRC_DIR) +test_mve_ldlibs += m + +$(eval $(call add-bin,test_mve)) + +test_mve: $(test_mve_bin) + @echo " RUN $(notdir $<)" + $(V)$< + +test: test_mve diff --git a/test/mve/mve.h b/test/mve/mve.h new file mode 100644 index 0000000..cc78dfc --- /dev/null +++ b/test/mve/mve.h @@ -0,0 +1,61 @@ +/****************************************************************************** + * + * Copyright 2026 Google LLC + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at: + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + ******************************************************************************/ + +#if (__ARM_FEATURE_MVE & 1) + +#include + +#else + +#include + + +/* ---------------------------------------------------------------------------- + * Integer + * -------------------------------------------------------------------------- */ + +typedef struct { int16_t e[8]; } int16x8_t; + + +/** + * Load + */ + +__attribute__((unused)) +static int16x8_t vldrhq_s16(const int16_t *p) +{ + return (int16x8_t){ { p[0], p[1], p[2], p[3], + p[4], p[5], p[6], p[7] } }; +} + + +/** + * Multiply-accumulate across vector, into a 64-bit accumulator + */ + +__attribute__((unused)) +static int64_t vmlaldavaq_s16(int64_t a, int16x8_t b, int16x8_t c) +{ + for (int i = 0; i < 8; i++) + a += (int32_t)b.e[i] * c.e[i]; + + return a; +} + + +#endif /* __ARM_FEATURE_MVE */ diff --git a/test/mve/test_mve.c b/test/mve/test_mve.c new file mode 100644 index 0000000..eb4ba13 --- /dev/null +++ b/test/mve/test_mve.c @@ -0,0 +1,32 @@ +/****************************************************************************** + * + * Copyright 2026 Google LLC + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at: + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + ******************************************************************************/ + +#include + +int check_ltpf(void); + +int main() +{ + int r, ret = 0; + + printf("Checking LTPF MVE... "); fflush(stdout); + printf("%s\n", (r = check_ltpf()) == 0 ? "OK" : "Failed"); + ret = ret || r; + + return ret; +}