Add a NEON codepath for panning and width
This commit is contained in:
parent
a2ecaf9a95
commit
3bac650498
12 changed files with 418 additions and 77 deletions
134
benchmarks/BM_pan_arm.cpp
Normal file
134
benchmarks/BM_pan_arm.cpp
Normal file
|
|
@ -0,0 +1,134 @@
|
||||||
|
// SPDX-License-Identifier: BSD-2-Clause
|
||||||
|
|
||||||
|
// This code is part of the sfizz library and is licensed under a BSD 2-clause
|
||||||
|
// license. You should have receive a LICENSE.md file along with the code.
|
||||||
|
// If not, contact the sfizz maintainers at https://github.com/sfztools/sfizz
|
||||||
|
|
||||||
|
#include "Panning.h"
|
||||||
|
#include "simd/Common.h"
|
||||||
|
#include <benchmark/benchmark.h>
|
||||||
|
#include <random>
|
||||||
|
#include <absl/algorithm/container.h>
|
||||||
|
#include "absl/types/span.h"
|
||||||
|
#include <arm_neon.h>
|
||||||
|
|
||||||
|
#include <jsl/allocator>
|
||||||
|
template <class T, std::size_t A = 16>
|
||||||
|
using aligned_vector = std::vector<T, jsl::aligned_allocator<T, A>>;
|
||||||
|
|
||||||
|
// Number of elements in the table, odd for equal volume at center
|
||||||
|
constexpr int panSize = 4095;
|
||||||
|
|
||||||
|
// Table of pan values for the left channel, extra element for safety
|
||||||
|
static const auto panData = []()
|
||||||
|
{
|
||||||
|
std::array<float, panSize + 1> pan;
|
||||||
|
int i = 0;
|
||||||
|
|
||||||
|
for (; i < panSize; ++i)
|
||||||
|
pan[i] = std::cos(i * (piTwo<double>() / (panSize - 1)));
|
||||||
|
|
||||||
|
for (; i < static_cast<int>(pan.size()); ++i)
|
||||||
|
pan[i] = pan[panSize - 1];
|
||||||
|
|
||||||
|
return pan;
|
||||||
|
}();
|
||||||
|
|
||||||
|
float _panLookup(float pan)
|
||||||
|
{
|
||||||
|
// reduce range, round to nearest
|
||||||
|
int index = lroundPositive(pan * (panSize - 1));
|
||||||
|
return panData[index];
|
||||||
|
}
|
||||||
|
|
||||||
|
void panScalar(const float* panEnvelope, float* leftBuffer, float* rightBuffer, unsigned size) noexcept
|
||||||
|
{
|
||||||
|
const auto sentinel = panEnvelope + size;
|
||||||
|
while (panEnvelope < sentinel) {
|
||||||
|
auto p =(*panEnvelope + 1.0f) * 0.5f;
|
||||||
|
p = clamp(p, 0.0f, 1.0f);
|
||||||
|
*leftBuffer *= _panLookup(p);
|
||||||
|
*rightBuffer *= _panLookup(1 - p);
|
||||||
|
incrementAll(panEnvelope, leftBuffer, rightBuffer);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void panSIMD(const float* panEnvelope, float* leftBuffer, float* rightBuffer, unsigned size) noexcept
|
||||||
|
{
|
||||||
|
const auto sentinel = panEnvelope + size;
|
||||||
|
int32_t indices[4];
|
||||||
|
while (panEnvelope < sentinel) {
|
||||||
|
float32x4_t mmPan = vld1q_f32(panEnvelope);
|
||||||
|
mmPan = vaddq_f32(mmPan, vdupq_n_f32(1.0f));
|
||||||
|
mmPan = vmulq_n_f32(mmPan, 0.5f * panSize);
|
||||||
|
mmPan = vaddq_f32(mmPan, vdupq_n_f32(0.5f));
|
||||||
|
mmPan = vminq_f32(mmPan, vdupq_n_f32(panSize));
|
||||||
|
mmPan = vmaxq_f32(mmPan, vdupq_n_f32(0.0f));
|
||||||
|
int32x4_t mmIdx = vcvtq_s32_f32(mmPan);
|
||||||
|
vst1q_s32(indices, mmIdx);
|
||||||
|
|
||||||
|
leftBuffer[0] *= panData[indices[0]];
|
||||||
|
rightBuffer[0] *= panData[panSize - indices[0] - 1];
|
||||||
|
leftBuffer[1] *= panData[indices[1]];
|
||||||
|
rightBuffer[1] *= panData[panSize - indices[1]- 1];
|
||||||
|
leftBuffer[2] *= panData[indices[2]];
|
||||||
|
rightBuffer[2] *= panData[panSize - indices[2]- 1];
|
||||||
|
leftBuffer[3] *= panData[indices[3]];
|
||||||
|
rightBuffer[3] *= panData[panSize - indices[3]- 1];
|
||||||
|
|
||||||
|
incrementAll<4>(panEnvelope, leftBuffer, rightBuffer);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
class PanFixture : public benchmark::Fixture {
|
||||||
|
public:
|
||||||
|
void SetUp(const ::benchmark::State& state) {
|
||||||
|
std::random_device rd { };
|
||||||
|
std::mt19937 gen { rd() };
|
||||||
|
std::uniform_real_distribution<float> dist { -1.0f, 1.0f };
|
||||||
|
pan.resize(state.range(0));
|
||||||
|
right.resize(state.range(0));
|
||||||
|
left.resize(state.range(0));
|
||||||
|
|
||||||
|
if (!willAlign<16>(pan.data(), left.data(), right.data()))
|
||||||
|
std::cout << "Will not align!" << '\n';
|
||||||
|
absl::c_generate(pan, [&]() { return dist(gen); });
|
||||||
|
absl::c_generate(left, [&]() { return dist(gen); });
|
||||||
|
absl::c_generate(right, [&]() { return dist(gen); });
|
||||||
|
}
|
||||||
|
|
||||||
|
void TearDown(const ::benchmark::State& /* state */) {
|
||||||
|
|
||||||
|
}
|
||||||
|
|
||||||
|
aligned_vector<float> pan;
|
||||||
|
aligned_vector<float> right;
|
||||||
|
aligned_vector<float> left;
|
||||||
|
};
|
||||||
|
|
||||||
|
BENCHMARK_DEFINE_F(PanFixture, PanScalar)(benchmark::State& state) {
|
||||||
|
for (auto _ : state)
|
||||||
|
{
|
||||||
|
panScalar(pan.data(), left.data(), right.data(), state.range(0));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
BENCHMARK_DEFINE_F(PanFixture, PanSIMD)(benchmark::State& state) {
|
||||||
|
for (auto _ : state)
|
||||||
|
{
|
||||||
|
panSIMD(pan.data(), left.data(), right.data(), state.range(0));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
BENCHMARK_DEFINE_F(PanFixture, PanSfizz)(benchmark::State& state) {
|
||||||
|
for (auto _ : state)
|
||||||
|
{
|
||||||
|
sfz::pan(pan.data(), left.data(), right.data(), state.range(0));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Register the function as a benchmark
|
||||||
|
BENCHMARK_REGISTER_F(PanFixture, PanScalar)->RangeMultiplier(4)->Range((1 << 4), (1 << 12));
|
||||||
|
BENCHMARK_REGISTER_F(PanFixture, PanSIMD)->RangeMultiplier(4)->Range((1 << 4), (1 << 12));
|
||||||
|
BENCHMARK_REGISTER_F(PanFixture, PanSfizz)->RangeMultiplier(4)->Range((1 << 4), (1 << 12));
|
||||||
|
BENCHMARK_MAIN();
|
||||||
|
|
@ -147,6 +147,12 @@ if (TARGET bm_resample)
|
||||||
add_dependencies(sfizz_benchmarks bm_resample)
|
add_dependencies(sfizz_benchmarks bm_resample)
|
||||||
endif()
|
endif()
|
||||||
|
|
||||||
|
if (SFIZZ_SYSTEM_PROCESSOR MATCHES "armv7l")
|
||||||
|
sfizz_add_benchmark(bm_pan_arm BM_pan_arm.cpp ../src/sfizz/Panning.cpp)
|
||||||
|
target_link_libraries(bm_pan_arm PRIVATE sfizz-jsl)
|
||||||
|
add_dependencies(sfizz_benchmarks bm_pan_arm)
|
||||||
|
endif()
|
||||||
|
|
||||||
configure_file("sample.wav" "${CMAKE_BINARY_DIR}/benchmarks/sample1.wav" COPYONLY)
|
configure_file("sample.wav" "${CMAKE_BINARY_DIR}/benchmarks/sample1.wav" COPYONLY)
|
||||||
configure_file("sample.wav" "${CMAKE_BINARY_DIR}/benchmarks/sample2.wav" COPYONLY)
|
configure_file("sample.wav" "${CMAKE_BINARY_DIR}/benchmarks/sample2.wav" COPYONLY)
|
||||||
configure_file("sample.wav" "${CMAKE_BINARY_DIR}/benchmarks/sample3.wav" COPYONLY)
|
configure_file("sample.wav" "${CMAKE_BINARY_DIR}/benchmarks/sample3.wav" COPYONLY)
|
||||||
|
|
|
||||||
|
|
@ -60,9 +60,9 @@ if (CMAKE_CXX_COMPILER_ID MATCHES "GNU|Clang")
|
||||||
add_compile_options(-Werror=return-type)
|
add_compile_options(-Werror=return-type)
|
||||||
if (SFIZZ_SYSTEM_PROCESSOR MATCHES "^(i.86|x86_64)$")
|
if (SFIZZ_SYSTEM_PROCESSOR MATCHES "^(i.86|x86_64)$")
|
||||||
add_compile_options(-msse2)
|
add_compile_options(-msse2)
|
||||||
elseif (SFIZZ_SYSTEM_PROCESSOR MATCHES "^(armv.*)$")
|
elseif(SFIZZ_SYSTEM_PROCESSOR MATCHES "^(arm.*)$")
|
||||||
add_compile_options(-mfloat-abi=hard)
|
|
||||||
add_compile_options(-mfpu=neon)
|
add_compile_options(-mfpu=neon)
|
||||||
|
add_compile_options(-mfloat-abi=hard)
|
||||||
endif()
|
endif()
|
||||||
elseif (CMAKE_CXX_COMPILER_ID MATCHES "MSVC")
|
elseif (CMAKE_CXX_COMPILER_ID MATCHES "MSVC")
|
||||||
set(CMAKE_CXX_STANDARD 17)
|
set(CMAKE_CXX_STANDARD 17)
|
||||||
|
|
|
||||||
|
|
@ -3,6 +3,7 @@ macro(sfizz_add_simd_sources SOURCES_VAR PREFIX)
|
||||||
|
|
||||||
list (APPEND ${SOURCES_VAR}
|
list (APPEND ${SOURCES_VAR}
|
||||||
${PREFIX}/sfizz/SIMDHelpers.cpp
|
${PREFIX}/sfizz/SIMDHelpers.cpp
|
||||||
|
${PREFIX}/sfizz/simd/HelpersNEON.cpp
|
||||||
${PREFIX}/sfizz/simd/HelpersSSE.cpp
|
${PREFIX}/sfizz/simd/HelpersSSE.cpp
|
||||||
${PREFIX}/sfizz/simd/HelpersAVX.cpp)
|
${PREFIX}/sfizz/simd/HelpersAVX.cpp)
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,11 +1,21 @@
|
||||||
#include "Panning.h"
|
#include "Panning.h"
|
||||||
|
#include "MathHelpers.h"
|
||||||
#include <array>
|
#include <array>
|
||||||
#include <cmath>
|
#include <cmath>
|
||||||
|
|
||||||
|
#if SFIZZ_HAVE_NEON
|
||||||
|
#include <arm_neon.h>
|
||||||
|
#include "simd/Common.h"
|
||||||
|
using Type = float;
|
||||||
|
constexpr unsigned TypeAlignment = 4;
|
||||||
|
constexpr unsigned ByteAlignment = TypeAlignment * sizeof(Type);
|
||||||
|
#endif
|
||||||
|
|
||||||
|
|
||||||
namespace sfz
|
namespace sfz
|
||||||
{
|
{
|
||||||
// Number of elements in the table, odd for equal volume at center
|
|
||||||
constexpr int panSize = 4095;
|
constexpr int panSize { 4095 };
|
||||||
|
|
||||||
// Table of pan values for the left channel, extra element for safety
|
// Table of pan values for the left channel, extra element for safety
|
||||||
static const auto panData = []()
|
static const auto panData = []()
|
||||||
|
|
@ -25,35 +35,134 @@ static const auto panData = []()
|
||||||
float panLookup(float pan)
|
float panLookup(float pan)
|
||||||
{
|
{
|
||||||
// reduce range, round to nearest
|
// reduce range, round to nearest
|
||||||
int index = lroundPositive(pan * (panSize - 1));
|
const int index = lroundPositive(pan * (panSize - 1));
|
||||||
return panData[index];
|
return panData[index];
|
||||||
}
|
}
|
||||||
|
|
||||||
|
inline void tickPan(const float* pan, float* leftBuffer, float* rightBuffer)
|
||||||
|
{
|
||||||
|
auto p = (*pan + 1.0f) * 0.5f;
|
||||||
|
p = clamp(p, 0.0f, 1.0f);
|
||||||
|
*leftBuffer *= panLookup(p);
|
||||||
|
*rightBuffer *= panLookup(1 - p);
|
||||||
|
}
|
||||||
|
|
||||||
void pan(const float* panEnvelope, float* leftBuffer, float* rightBuffer, unsigned size) noexcept
|
void pan(const float* panEnvelope, float* leftBuffer, float* rightBuffer, unsigned size) noexcept
|
||||||
{
|
{
|
||||||
const auto sentinel = panEnvelope + size;
|
const auto sentinel = panEnvelope + size;
|
||||||
|
|
||||||
|
#if SFIZZ_HAVE_NEON
|
||||||
|
const auto firstAligned = prevAligned<ByteAlignment>(panEnvelope + TypeAlignment - 1);
|
||||||
|
|
||||||
|
if (willAlign<ByteAlignment>(panEnvelope, leftBuffer, rightBuffer) && (firstAligned < sentinel)) {
|
||||||
|
while (panEnvelope < firstAligned) {
|
||||||
|
tickPan(panEnvelope, leftBuffer, rightBuffer);
|
||||||
|
incrementAll(panEnvelope, leftBuffer, rightBuffer);
|
||||||
|
}
|
||||||
|
|
||||||
|
uint32_t indices[TypeAlignment];
|
||||||
|
float leftPan[TypeAlignment];
|
||||||
|
float rightPan[TypeAlignment];
|
||||||
|
const auto lastAligned = prevAligned<ByteAlignment>(sentinel);
|
||||||
|
while (panEnvelope < lastAligned) {
|
||||||
|
float32x4_t mmPan = vld1q_f32(panEnvelope);
|
||||||
|
mmPan = vaddq_f32(mmPan, vdupq_n_f32(1.0f));
|
||||||
|
mmPan = vmulq_n_f32(mmPan, 0.5f * panSize);
|
||||||
|
mmPan = vaddq_f32(mmPan, vdupq_n_f32(0.5f));
|
||||||
|
uint32x4_t mmIdx = vcvtq_u32_f32(mmPan);
|
||||||
|
mmIdx = vminq_u32(mmIdx, vdupq_n_u32(panSize - 1));
|
||||||
|
mmIdx = vmaxq_u32(mmIdx, vdupq_n_u32(0));
|
||||||
|
vst1q_u32(indices, mmIdx);
|
||||||
|
|
||||||
|
leftPan[0] = panData[indices[0]];
|
||||||
|
rightPan[0] = panData[panSize - indices[0] - 1];
|
||||||
|
leftPan[1] = panData[indices[1]];
|
||||||
|
rightPan[1] = panData[panSize - indices[1] - 1];
|
||||||
|
leftPan[2] = panData[indices[2]];
|
||||||
|
rightPan[2] = panData[panSize - indices[2] - 1];
|
||||||
|
leftPan[3] = panData[indices[3]];
|
||||||
|
rightPan[3] = panData[panSize - indices[3] - 1];
|
||||||
|
|
||||||
|
vst1q_f32(leftBuffer, vmulq_f32(vld1q_f32(leftBuffer), vld1q_f32(leftPan)));
|
||||||
|
vst1q_f32(rightBuffer, vmulq_f32(vld1q_f32(rightBuffer), vld1q_f32(rightPan)));
|
||||||
|
|
||||||
|
incrementAll<TypeAlignment>(panEnvelope, leftBuffer, rightBuffer);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
while (panEnvelope < sentinel) {
|
while (panEnvelope < sentinel) {
|
||||||
auto p =(*panEnvelope + 1.0f) * 0.5f;
|
tickPan(panEnvelope, leftBuffer, rightBuffer);
|
||||||
p = clamp(p, 0.0f, 1.0f);
|
|
||||||
*leftBuffer *= panLookup(p);
|
|
||||||
*rightBuffer *= panLookup(1 - p);
|
|
||||||
incrementAll(panEnvelope, leftBuffer, rightBuffer);
|
incrementAll(panEnvelope, leftBuffer, rightBuffer);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
}
|
||||||
|
|
||||||
|
inline void tickWidth(const float* width, float* leftBuffer, float* rightBuffer)
|
||||||
|
{
|
||||||
|
float w = (*width + 1.0f) * 0.5f;
|
||||||
|
w = clamp(w, 0.0f, 1.0f);
|
||||||
|
const auto coeff1 = panLookup(w);
|
||||||
|
const auto coeff2 = panLookup(1 - w);
|
||||||
|
const auto l = *leftBuffer;
|
||||||
|
const auto r = *rightBuffer;
|
||||||
|
*leftBuffer = l * coeff2 + r * coeff1;
|
||||||
|
*rightBuffer = l * coeff1 + r * coeff2;
|
||||||
}
|
}
|
||||||
|
|
||||||
void width(const float* widthEnvelope, float* leftBuffer, float* rightBuffer, unsigned size) noexcept
|
void width(const float* widthEnvelope, float* leftBuffer, float* rightBuffer, unsigned size) noexcept
|
||||||
{
|
{
|
||||||
const auto sentinel = widthEnvelope + size;
|
const auto sentinel = widthEnvelope + size;
|
||||||
|
|
||||||
|
#if SFIZZ_HAVE_NEON
|
||||||
|
const auto firstAligned = prevAligned<ByteAlignment>(widthEnvelope + TypeAlignment - 1);
|
||||||
|
|
||||||
|
if (willAlign<ByteAlignment>(widthEnvelope, leftBuffer, rightBuffer) && firstAligned < sentinel) {
|
||||||
|
while (widthEnvelope < firstAligned) {
|
||||||
|
tickWidth(widthEnvelope, leftBuffer, rightBuffer);
|
||||||
|
incrementAll(widthEnvelope, leftBuffer, rightBuffer);
|
||||||
|
}
|
||||||
|
|
||||||
|
uint32_t indices[TypeAlignment];
|
||||||
|
float coeff1[TypeAlignment];
|
||||||
|
float coeff2[TypeAlignment];
|
||||||
|
const auto lastAligned = prevAligned<ByteAlignment>(sentinel);
|
||||||
|
while (widthEnvelope < lastAligned) {
|
||||||
|
float32x4_t mmWidth = vld1q_f32(widthEnvelope);
|
||||||
|
mmWidth = vaddq_f32(mmWidth, vdupq_n_f32(1.0f));
|
||||||
|
mmWidth = vmulq_n_f32(mmWidth, 0.5f * panSize);
|
||||||
|
mmWidth = vaddq_f32(mmWidth, vdupq_n_f32(0.5f));
|
||||||
|
uint32x4_t mmIdx = vcvtq_u32_f32(mmWidth);
|
||||||
|
mmIdx = vminq_u32(mmIdx, vdupq_n_u32(panSize - 1));
|
||||||
|
mmIdx = vmaxq_u32(mmIdx, vdupq_n_u32(0));
|
||||||
|
vst1q_u32(indices, mmIdx);
|
||||||
|
|
||||||
|
coeff1[0] = panData[indices[0]];
|
||||||
|
coeff2[0] = panData[panSize - indices[0] - 1];
|
||||||
|
coeff1[1] = panData[indices[1]];
|
||||||
|
coeff2[1] = panData[panSize - indices[1] - 1];
|
||||||
|
coeff1[2] = panData[indices[2]];
|
||||||
|
coeff2[2] = panData[panSize - indices[2] - 1];
|
||||||
|
coeff1[3] = panData[indices[3]];
|
||||||
|
coeff2[3] = panData[panSize - indices[3] - 1];
|
||||||
|
|
||||||
|
float32x4_t mmCoeff1 = vld1q_f32(coeff1);
|
||||||
|
float32x4_t mmCoeff2 = vld1q_f32(coeff2);
|
||||||
|
float32x4_t mmLeft = vld1q_f32(leftBuffer);
|
||||||
|
float32x4_t mmRight = vld1q_f32(rightBuffer);
|
||||||
|
|
||||||
|
vst1q_f32(leftBuffer, vaddq_f32(vmulq_f32(mmCoeff2, mmLeft), vmulq_f32(mmCoeff1, mmRight)));
|
||||||
|
vst1q_f32(rightBuffer, vaddq_f32(vmulq_f32(mmCoeff1, mmLeft), vmulq_f32(mmCoeff2, mmRight)));
|
||||||
|
|
||||||
|
incrementAll<TypeAlignment>(widthEnvelope, leftBuffer, rightBuffer);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
#endif // SFIZZ_HAVE_NEON
|
||||||
|
|
||||||
while (widthEnvelope < sentinel) {
|
while (widthEnvelope < sentinel) {
|
||||||
float w = (*widthEnvelope + 1.0f) * 0.5f;
|
tickWidth(widthEnvelope, leftBuffer, rightBuffer);
|
||||||
w = clamp(w, 0.0f, 1.0f);
|
|
||||||
const auto coeff1 = panLookup(w);
|
|
||||||
const auto coeff2 = panLookup(1 - w);
|
|
||||||
const auto l = *leftBuffer;
|
|
||||||
const auto r = *rightBuffer;
|
|
||||||
*leftBuffer = l * coeff2 + r * coeff1;
|
|
||||||
*rightBuffer = l * coeff1 + r * coeff2;
|
|
||||||
incrementAll(widthEnvelope, leftBuffer, rightBuffer);
|
incrementAll(widthEnvelope, leftBuffer, rightBuffer);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -6,13 +6,16 @@ namespace sfz
|
||||||
{
|
{
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @brief Lookup a value from the pan table
|
* @brief Lookup a value from the pan table
|
||||||
*
|
* No check is done on the range, needs to be capped
|
||||||
* @param pan
|
* between 0 and panSize.
|
||||||
* @return float
|
*
|
||||||
*/
|
* @param pan
|
||||||
|
* @return float
|
||||||
|
*/
|
||||||
float panLookup(float pan);
|
float panLookup(float pan);
|
||||||
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @brief Pans a mono signal left or right
|
* @brief Pans a mono signal left or right
|
||||||
*
|
*
|
||||||
|
|
|
||||||
|
|
@ -15,8 +15,8 @@
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#include "Width.h"
|
#include "Width.h"
|
||||||
#include "Opcode.h"
|
|
||||||
#include "Panning.h"
|
#include "Panning.h"
|
||||||
|
#include "Opcode.h"
|
||||||
#include "absl/memory/memory.h"
|
#include "absl/memory/memory.h"
|
||||||
|
|
||||||
namespace sfz {
|
namespace sfz {
|
||||||
|
|
|
||||||
|
|
@ -24,7 +24,7 @@ T* prevAligned(const T* ptr)
|
||||||
template<unsigned N, class T>
|
template<unsigned N, class T>
|
||||||
bool unaligned(const T* ptr)
|
bool unaligned(const T* ptr)
|
||||||
{
|
{
|
||||||
return (reinterpret_cast<uintptr_t>(ptr) & ByteAlignmentMask(N) )!= 0;
|
return (reinterpret_cast<uintptr_t>(ptr) & ByteAlignmentMask(N) ) != 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
template<unsigned N, class T, class... Args>
|
template<unsigned N, class T, class... Args>
|
||||||
|
|
@ -32,3 +32,20 @@ bool unaligned(const T* ptr1, Args... rest)
|
||||||
{
|
{
|
||||||
return unaligned<N>(ptr1) || unaligned<N>(rest...);
|
return unaligned<N>(ptr1) || unaligned<N>(rest...);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
template<unsigned N, class T, class... Args>
|
||||||
|
bool willAlign(const T* ptr1, const T* ptr2)
|
||||||
|
{
|
||||||
|
const auto p1 = reinterpret_cast<uintptr_t>(ptr1);
|
||||||
|
const auto p2 = reinterpret_cast<uintptr_t>(ptr2);
|
||||||
|
return (
|
||||||
|
(p1 & ByteAlignmentMask(N)) == (p2 & ByteAlignmentMask(N))
|
||||||
|
&& ((p1 & ByteAlignmentMask(sizeof(T))) == 0)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
template<unsigned N, class T, class... Args>
|
||||||
|
bool willAlign(const T* ptr1, const T* ptr2, Args... rest)
|
||||||
|
{
|
||||||
|
return willAlign<N>(ptr1, ptr2) && willAlign<N>(ptr2, rest...);
|
||||||
|
}
|
||||||
|
|
|
||||||
16
src/sfizz/simd/HelpersNEON.cpp
Normal file
16
src/sfizz/simd/HelpersNEON.cpp
Normal file
|
|
@ -0,0 +1,16 @@
|
||||||
|
// SPDX-License-Identifier: BSD-2-Clause
|
||||||
|
|
||||||
|
// This code is part of the sfizz library and is licensed under a BSD 2-clause
|
||||||
|
// license. You should have receive a LICENSE.md file along with the code.
|
||||||
|
// If not, contact the sfizz maintainers at https://github.com/sfztools/sfizz
|
||||||
|
|
||||||
|
#include "HelpersNEON.h"
|
||||||
|
#include "Common.h"
|
||||||
|
|
||||||
|
#if SFIZZ_HAVE_NEON
|
||||||
|
#include <arm_neon.h>
|
||||||
|
#endif
|
||||||
|
|
||||||
|
using Type = float;
|
||||||
|
constexpr unsigned TypeAlignment = 4;
|
||||||
|
constexpr unsigned ByteAlignment = TypeAlignment * sizeof(Type);
|
||||||
7
src/sfizz/simd/HelpersNEON.h
Normal file
7
src/sfizz/simd/HelpersNEON.h
Normal file
|
|
@ -0,0 +1,7 @@
|
||||||
|
// SPDX-License-Identifier: BSD-2-Clause
|
||||||
|
|
||||||
|
// This code is part of the sfizz library and is licensed under a BSD 2-clause
|
||||||
|
// license. You should have receive a LICENSE.md file along with the code.
|
||||||
|
// If not, contact the sfizz maintainers at https://github.com/sfztools/sfizz
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
@ -46,7 +46,7 @@ set(SFIZZ_TEST_SOURCES
|
||||||
)
|
)
|
||||||
|
|
||||||
add_executable(sfizz_tests ${SFIZZ_TEST_SOURCES})
|
add_executable(sfizz_tests ${SFIZZ_TEST_SOURCES})
|
||||||
target_link_libraries(sfizz_tests PRIVATE sfizz::sfizz)
|
target_link_libraries(sfizz_tests PRIVATE sfizz::sfizz sfizz-jsl)
|
||||||
sfizz_enable_lto_if_needed(sfizz_tests)
|
sfizz_enable_lto_if_needed(sfizz_tests)
|
||||||
sfizz_enable_fast_math(sfizz_tests)
|
sfizz_enable_fast_math(sfizz_tests)
|
||||||
# target_link_libraries(sfizz_tests PRIVATE absl::strings absl::str_format absl::flat_hash_map cnpy absl::span absl::algorithm)
|
# target_link_libraries(sfizz_tests PRIVATE absl::strings absl::str_format absl::flat_hash_map cnpy absl::span absl::algorithm)
|
||||||
|
|
|
||||||
|
|
@ -4,6 +4,7 @@
|
||||||
// license. You should have receive a LICENSE.md file along with the code.
|
// license. You should have receive a LICENSE.md file along with the code.
|
||||||
// If not, contact the sfizz maintainers at https://github.com/sfztools/sfizz
|
// If not, contact the sfizz maintainers at https://github.com/sfztools/sfizz
|
||||||
|
|
||||||
|
#include "sfizz/simd/Common.h"
|
||||||
#include "sfizz/SIMDHelpers.h"
|
#include "sfizz/SIMDHelpers.h"
|
||||||
#include "sfizz/Panning.h"
|
#include "sfizz/Panning.h"
|
||||||
#include "catch2/catch.hpp"
|
#include "catch2/catch.hpp"
|
||||||
|
|
@ -12,8 +13,12 @@
|
||||||
#include <algorithm>
|
#include <algorithm>
|
||||||
#include <array>
|
#include <array>
|
||||||
#include <iostream>
|
#include <iostream>
|
||||||
|
#include <jsl/allocator>
|
||||||
using namespace Catch::literals;
|
using namespace Catch::literals;
|
||||||
|
|
||||||
|
template <class T, std::size_t A = sfz::config::defaultAlignment>
|
||||||
|
using aligned_vector = std::vector<T, jsl::aligned_allocator<T, A>>;
|
||||||
|
|
||||||
constexpr int smallBufferSize { 3 };
|
constexpr int smallBufferSize { 3 };
|
||||||
constexpr int bigBufferSize { 4095 };
|
constexpr int bigBufferSize { 4095 };
|
||||||
constexpr int medBufferSize { 127 };
|
constexpr int medBufferSize { 127 };
|
||||||
|
|
@ -49,6 +54,40 @@ inline bool approxEqual(absl::Span<const Type> lhs, absl::Span<const Type> rhs,
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
TEST_CASE("[Helpers] willAlign, prevAligned and unaligned tests")
|
||||||
|
{
|
||||||
|
aligned_vector<float, 32> array(16);
|
||||||
|
REQUIRE( !unaligned<16>(&array[0]) );
|
||||||
|
REQUIRE( !unaligned<16>(&array[4]) );
|
||||||
|
REQUIRE( !unaligned<32>(&array[8]) );
|
||||||
|
REQUIRE( unaligned<32>(&array[7]) );
|
||||||
|
REQUIRE( unaligned<32>(&array[4]) );
|
||||||
|
REQUIRE( unaligned<16>(&array[3]) );
|
||||||
|
REQUIRE( !unaligned<16>(&array[0], &array[4]) );
|
||||||
|
REQUIRE( !unaligned<16>(&array[0], &array[4], &array[8]) );
|
||||||
|
REQUIRE( unaligned<16>(&array[0], &array[3], &array[8]) );
|
||||||
|
|
||||||
|
REQUIRE( prevAligned<16>(&array[0]) == &array[0] );
|
||||||
|
REQUIRE( prevAligned<16>(&array[1]) == &array[0] );
|
||||||
|
REQUIRE( prevAligned<16>(&array[2]) == &array[0] );
|
||||||
|
REQUIRE( prevAligned<16>(&array[3]) == &array[0] );
|
||||||
|
REQUIRE( prevAligned<16>(&array[4]) == &array[4] );
|
||||||
|
REQUIRE( prevAligned<16>(&array[5]) == &array[4] );
|
||||||
|
REQUIRE( prevAligned<32>(&array[7]) == &array[0] );
|
||||||
|
REQUIRE( prevAligned<32>(&array[8]) == &array[8] );
|
||||||
|
REQUIRE( prevAligned<32>(&array[9]) == &array[8] );
|
||||||
|
|
||||||
|
REQUIRE( willAlign<16>(&array[0], &array[4]) );
|
||||||
|
REQUIRE( willAlign<16>(&array[5], &array[1]) );
|
||||||
|
REQUIRE( !willAlign<16>(&array[2], &array[1]) );
|
||||||
|
REQUIRE( willAlign<32>(&array[9], &array[1]) );
|
||||||
|
REQUIRE( willAlign<32>(&array[8], &array[0]) );
|
||||||
|
|
||||||
|
float* meanPointer = (float*)((uint8_t*)&array[1] + 1);
|
||||||
|
REQUIRE( !willAlign<16>(&array[0], meanPointer) );
|
||||||
|
REQUIRE( !willAlign<16>(&array[4], &array[0], meanPointer) );
|
||||||
|
}
|
||||||
|
|
||||||
TEST_CASE("[Helpers] Interleaved read")
|
TEST_CASE("[Helpers] Interleaved read")
|
||||||
{
|
{
|
||||||
std::array<float, 16> input { 0.0f, 10.0f, 1.0f, 11.0f, 2.0f, 12.0f, 3.0f, 13.0f, 4.0f, 14.0f, 5.0f, 15.0f, 6.0f, 16.0f, 7.0f, 17.0f };
|
std::array<float, 16> input { 0.0f, 10.0f, 1.0f, 11.0f, 2.0f, 12.0f, 3.0f, 13.0f, 4.0f, 14.0f, 5.0f, 15.0f, 6.0f, 16.0f, 7.0f, 17.0f };
|
||||||
|
|
@ -834,62 +873,71 @@ TEST_CASE("[Helpers] Diff (SIMD vs Scalar)")
|
||||||
REQUIRE(approxEqual<float>(outputScalar, outputSIMD));
|
REQUIRE(approxEqual<float>(outputScalar, outputSIMD));
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST_CASE("[Helpers] Pan Scalar")
|
template<unsigned N>
|
||||||
|
void panTest(float leftValue, float rightValue, float panValue, float expectedLeft, float expectedRight)
|
||||||
{
|
{
|
||||||
std::array<float, 1> leftValue { 1.0f };
|
std::vector<float> leftChannel(N);
|
||||||
std::array<float, 1> rightValue { 1.0f };
|
std::vector<float> rightChannel(N);
|
||||||
auto left = absl::MakeSpan(leftValue);
|
std::vector<float> pan(N);
|
||||||
auto right = absl::MakeSpan(rightValue);
|
std::vector<float> expectedLeftChannel(N);
|
||||||
SECTION("Pan = 0")
|
std::vector<float> expectedRightChannel(N);
|
||||||
{
|
std::fill(leftChannel.begin(), leftChannel.end(), leftValue);
|
||||||
std::array<float, 1> pan { 0.0f };
|
std::fill(expectedLeftChannel.begin(), expectedLeftChannel.end(), expectedLeft);
|
||||||
sfz::pan(pan, left, right);
|
std::fill(rightChannel.begin(), rightChannel.end(), rightValue);
|
||||||
REQUIRE(left[0] == Approx(0.70711f).margin(0.001f));
|
std::fill(expectedRightChannel.begin(), expectedRightChannel.end(), expectedRight);
|
||||||
REQUIRE(right[0] == Approx(0.70711f).margin(0.001f));
|
std::fill(pan.begin(), pan.end(), panValue);
|
||||||
}
|
auto left = absl::MakeSpan(leftChannel);
|
||||||
SECTION("Pan = 1")
|
auto right = absl::MakeSpan(rightChannel);
|
||||||
{
|
sfz::pan(pan, left, right);
|
||||||
std::array<float, 1> pan { 1.0f };
|
REQUIRE_THAT( leftChannel, Catch::Approx(expectedLeftChannel).margin(0.001) );
|
||||||
sfz::pan(pan, left, right);
|
REQUIRE_THAT( rightChannel, Catch::Approx(expectedRightChannel).margin(0.001) );
|
||||||
REQUIRE(left[0] == Approx(0.0f).margin(0.001f));
|
|
||||||
REQUIRE(right[0] == Approx(1.0f).margin(0.001f));
|
|
||||||
}
|
|
||||||
SECTION("Pan = -1")
|
|
||||||
{
|
|
||||||
std::array<float, 1> pan { -1.0f };
|
|
||||||
sfz::pan(pan, left, right);
|
|
||||||
REQUIRE(left[0] == Approx(1.0f).margin(0.001f));
|
|
||||||
REQUIRE(right[0] == Approx(0.0f).margin(0.001f));
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST_CASE("[Helpers] Width Scalar")
|
template<unsigned N>
|
||||||
|
void widthTest(float leftValue, float rightValue, float widthValue, float expectedLeft, float expectedRight)
|
||||||
{
|
{
|
||||||
std::array<float, 1> leftValue { 1.0f };
|
std::vector<float> leftChannel(N);
|
||||||
std::array<float, 1> rightValue { 1.0f };
|
std::vector<float> rightChannel(N);
|
||||||
auto left = absl::MakeSpan(leftValue);
|
std::vector<float> width(N);
|
||||||
auto right = absl::MakeSpan(rightValue);
|
std::vector<float> expectedLeftChannel(N);
|
||||||
SECTION("width = 1")
|
std::vector<float> expectedRightChannel(N);
|
||||||
{
|
std::fill(leftChannel.begin(), leftChannel.end(), leftValue);
|
||||||
std::array<float, 1> width { 1.0f };
|
std::fill(expectedLeftChannel.begin(), expectedLeftChannel.end(), expectedLeft);
|
||||||
sfz::width(width, left, right);
|
std::fill(rightChannel.begin(), rightChannel.end(), rightValue);
|
||||||
REQUIRE(left[0] == Approx(1.0f).margin(0.001f));
|
std::fill(expectedRightChannel.begin(), expectedRightChannel.end(), expectedRight);
|
||||||
REQUIRE(right[0] == Approx(1.0f).margin(0.001f));
|
std::fill(width.begin(), width.end(), widthValue);
|
||||||
}
|
auto left = absl::MakeSpan(leftChannel);
|
||||||
SECTION("width = 0")
|
auto right = absl::MakeSpan(rightChannel);
|
||||||
{
|
sfz::width(width, left, right);
|
||||||
std::array<float, 1> width { 0.0f };
|
REQUIRE_THAT( leftChannel, Catch::Approx(expectedLeftChannel).margin(0.001) );
|
||||||
sfz::width(width, left, right);
|
REQUIRE_THAT( rightChannel, Catch::Approx(expectedRightChannel).margin(0.001) );
|
||||||
REQUIRE(left[0] == Approx(1.414f).margin(0.001f));
|
}
|
||||||
REQUIRE(right[0] == Approx(1.414f).margin(0.001f));
|
|
||||||
}
|
TEST_CASE("[Helpers] Pan tests")
|
||||||
SECTION("width = -1")
|
{
|
||||||
{
|
// Testing different sizes to check that SIMD and unrolling works as expected
|
||||||
std::array<float, 1> width { -1.0f };
|
panTest<1>(1.0f, 1.0f, 0.0f, 0.70711f, 0.70711f);
|
||||||
sfz::width(width, left, right);
|
panTest<1>(1.0f, 1.0f, 1.0f, 0.0f, 1.0f);
|
||||||
REQUIRE(left[0] == Approx(1.0f).margin(0.001f));
|
panTest<1>(1.0f, 1.0f, -1.0f, 1.0f, 0.0f);
|
||||||
REQUIRE(right[0] == Approx(1.0f).margin(0.001f));
|
panTest<3>(1.0f, 1.0f, 0.0f, 0.70711f, 0.70711f);
|
||||||
}
|
panTest<3>(1.0f, 1.0f, 1.0f, 0.0f, 1.0f);
|
||||||
|
panTest<3>(1.0f, 1.0f, -1.0f, 1.0f, 0.0f);
|
||||||
|
panTest<10>(1.0f, 1.0f, 0.0f, 0.70711f, 0.70711f);
|
||||||
|
panTest<10>(1.0f, 1.0f, 1.0f, 0.0f, 1.0f);
|
||||||
|
panTest<10>(1.0f, 1.0f, -1.0f, 1.0f, 0.0f);
|
||||||
|
}
|
||||||
|
|
||||||
|
TEST_CASE("[Helpers] Width tests")
|
||||||
|
{
|
||||||
|
widthTest<1>(1.0f, 1.0f, 0.0f, 1.414f, 1.414f);
|
||||||
|
widthTest<1>(1.0f, 1.0f, 1.0f, 1.0f, 1.0f);
|
||||||
|
widthTest<1>(1.0f, 1.0f, -1.0f, 1.0f, 1.0f);
|
||||||
|
widthTest<3>(1.0f, 1.0f, 0.0f, 1.414f, 1.414f);
|
||||||
|
widthTest<3>(1.0f, 1.0f, 1.0f, 1.0f, 1.0f);
|
||||||
|
widthTest<3>(1.0f, 1.0f, -1.0f, 1.0f, 1.0f);
|
||||||
|
widthTest<10>(1.0f, 1.0f, 0.0f, 1.414f, 1.414f);
|
||||||
|
widthTest<10>(1.0f, 1.0f, 1.0f, 1.0f, 1.0f);
|
||||||
|
widthTest<10>(1.0f, 1.0f, -1.0f, 1.0f, 1.0f);
|
||||||
}
|
}
|
||||||
|
|
||||||
TEST_CASE("[Helpers] clampAll")
|
TEST_CASE("[Helpers] clampAll")
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue