Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions src/ltpf.c
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,7 @@
#include "ltpf.h"
#include "tables.h"

#include "ltpf_mve.h"
#include "ltpf_neon.h"
#include "ltpf_arm.h"

Expand Down
96 changes: 96 additions & 0 deletions src/ltpf_mve.h
Original file line number Diff line number Diff line change
@@ -0,0 +1,96 @@
/******************************************************************************
*
* Copyright 2026 Google LLC
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at:
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*
******************************************************************************/

/*
* Arm Helium (M-Profile Vector Extension, MVE) backend for the LTPF
* pitch-detection hotspot, targeting Armv8.1-M cores (e.g. Cortex-M55).
*
* `dot()` / `correlate()` accumulate int16 x int16 products into a 64-bit
* integer and apply the rounding `(v + 32) >> 6`. MVE `vmlaldavaq_s16`
* performs an 8-lane int16 x int16 -> int64 multiply-accumulate across the
* vector; the integer sum and the subsequent rounding are therefore
* identical to the scalar reference, i.e. the result is strictly bit-exact.
*
* This mirrors the NEON (`ltpf_neon.h`) and Arm DSP (`ltpf_arm.h`) backends
* and hooks the same `dot` / `correlate` override points. Include this file
* before the NEON/DSP backends in `ltpf.c` so its definitions take
* precedence on MVE targets.
*/

#if (__ARM_FEATURE_MVE & 1) && !defined(TEST_ARM) && !defined(TEST_NEON) \
|| defined(TEST_MVE)

#ifndef TEST_MVE
#include <arm_mve.h>
#endif /* TEST_MVE */


/**
* Return dot product of 2 vectors
* a, b, n The 2 vectors of size `n` (> 0 and <= 128)
* return round( sum( a[i] * b[i] ) / 64 ), as float
*
* The size `n` of vectors is a multiple of 16 (so also of 8).
*/
#ifndef dot

LC3_HOT static inline float mve_dot(const int16_t *a, const int16_t *b, int n)
{
int64_t v = 0;

for (int i = 0; i < (n >> 3); i++) {
int16x8_t va = vldrhq_s16(a); a += 8;
int16x8_t vb = vldrhq_s16(b); b += 8;
v = vmlaldavaq_s16(v, va, vb);
}

int32_t v32 = (v + (1 << 5)) >> 6;
return (float)v32;
}

#ifndef TEST_MVE
#define dot mve_dot
#endif

#endif /* dot */

/**
* Return vector of correlations
* a, b, n The 2 vectors of size `n` (> 0 and <= 128)
* y, nc Output the correlation vector of size `nc`
*
* Sliding dot product: for each output the `b` window is shifted by one
* sample. Each lag reuses the 8-lane MVE `dot`, matching the scalar
* reference exactly.
*/
#ifndef correlate

LC3_HOT static void mve_correlate(
const int16_t *a, const int16_t *b, int n, float *y, int nc)
{
for (const float *ye = y + nc; y < ye; )
*(y++) = mve_dot(a, b--, n);
}

#ifndef TEST_MVE
#define correlate mve_correlate
#endif

#endif /* correlate */

#endif /* __ARM_FEATURE_MVE & 1 */
1 change: 1 addition & 0 deletions test/makefile.mk
Original file line number Diff line number Diff line change
Expand Up @@ -26,5 +26,6 @@ test-clean:

-include $(TEST_DIR)/arm/makefile.mk
-include $(TEST_DIR)/neon/makefile.mk
-include $(TEST_DIR)/mve/makefile.mk

clean-all: test-clean
86 changes: 86 additions & 0 deletions test/mve/ltpf_mve.c
Original file line number Diff line number Diff line change
@@ -0,0 +1,86 @@
/******************************************************************************
*
* Copyright 2026 Google LLC
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at:
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*
******************************************************************************/

#include "mve.h"

#include <stdio.h>
#include <stdint.h>
#include <stdlib.h>

/* -------------------------------------------------------------------------- */

#define TEST_MVE
#include <ltpf.c>

void lc3_put_bits_generic(lc3_bits_t *a, unsigned b, int c)
{ (void)a, (void)b, (void)c; }

unsigned lc3_get_bits_generic(struct lc3_bits *a, int b)
{ return (void)a, (void)b, 0; }

/* -------------------------------------------------------------------------- */

static int check_dot()
{
int16_t x[200];
for (int i = 0; i < 200; i++)
x[i] = rand() & 0xffff;

float y = dot(x, x+3, 128);
float y_mve = mve_dot(x, x+3, 128);
if (y != y_mve)
return -1;

return 0;
}

static int check_correlate()
{
int16_t alignas(4) a[500], b[500];
float y[100], y_mve[100];

for (int i = 0; i < 500; i++) {
a[i] = rand() & 0xffff;
b[i] = rand() & 0xffff;
}

correlate(a, b+200, 128, y, 100);
mve_correlate(a, b+200, 128, y_mve, 100);
if (memcmp(y, y_mve, 100 * sizeof(*y)) != 0)
return -1;

correlate(a, b+199, 128, y, 99);
mve_correlate(a, b+199, 128, y_mve, 99);
if (memcmp(y, y_mve, 99 * sizeof(*y)) != 0)
return -1;

return 0;
}

int check_ltpf(void)
{
int ret;

if ((ret = check_dot()) < 0)
return ret;

if ((ret = check_correlate()) < 0)
return ret;

return 0;
}
31 changes: 31 additions & 0 deletions test/mve/makefile.mk
Original file line number Diff line number Diff line change
@@ -0,0 +1,31 @@
#
# Copyright 2026 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at:
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#

test_mve_src += \
$(TEST_DIR)/mve/test_mve.c \
$(TEST_DIR)/mve/ltpf_mve.c \
$(SRC_DIR)/tables.c

test_mve_include += $(SRC_DIR)
test_mve_ldlibs += m

$(eval $(call add-bin,test_mve))

test_mve: $(test_mve_bin)
@echo " RUN $(notdir $<)"
$(V)$<

test: test_mve
61 changes: 61 additions & 0 deletions test/mve/mve.h
Original file line number Diff line number Diff line change
@@ -0,0 +1,61 @@
/******************************************************************************
*
* Copyright 2026 Google LLC
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at:
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*
******************************************************************************/

#if (__ARM_FEATURE_MVE & 1)

#include <arm_mve.h>

#else

#include <stdint.h>


/* ----------------------------------------------------------------------------
* Integer
* -------------------------------------------------------------------------- */

typedef struct { int16_t e[8]; } int16x8_t;


/**
* Load
*/

__attribute__((unused))
static int16x8_t vldrhq_s16(const int16_t *p)
{
return (int16x8_t){ { p[0], p[1], p[2], p[3],
p[4], p[5], p[6], p[7] } };
}


/**
* Multiply-accumulate across vector, into a 64-bit accumulator
*/

__attribute__((unused))
static int64_t vmlaldavaq_s16(int64_t a, int16x8_t b, int16x8_t c)
{
for (int i = 0; i < 8; i++)
a += (int32_t)b.e[i] * c.e[i];

return a;
}


#endif /* __ARM_FEATURE_MVE */
32 changes: 32 additions & 0 deletions test/mve/test_mve.c
Original file line number Diff line number Diff line change
@@ -0,0 +1,32 @@
/******************************************************************************
*
* Copyright 2026 Google LLC
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at:
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*
******************************************************************************/

#include <stdio.h>

int check_ltpf(void);

int main()
{
int r, ret = 0;

printf("Checking LTPF MVE... "); fflush(stdout);
printf("%s\n", (r = check_ltpf()) == 0 ? "OK" : "Failed");
ret = ret || r;

return ret;
}