Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 19 additions & 2 deletions src/renderer/obr/obr_capi/obr/obr/audio_buffer/simd_macros.h
Original file line number Diff line number Diff line change
Expand Up @@ -13,8 +13,25 @@
#ifndef OBR_AUDIO_BUFFER_SIMD_MACROS_H_
#define OBR_AUDIO_BUFFER_SIMD_MACROS_H_

#if !defined(DISABLE_SIMD) && (defined(__x86_64__) || defined(_M_X64) || \
defined(i386) || defined(_M_IX86))
#if !defined(DISABLE_SIMD) && defined(__wasm_simd128__)
// Wasm SIMD is enabled.
// Define __SSE__ for the xmmintrin header and SIMD_WASM for simd_utils.cc
#define __SSE__

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This define exists only because the #if defined(SIMD_SSE) || defined(SIMD_WASM) blocks in simd_utils.cc call _mm_loadu_ps, _mm_storeu_ps and _mm_shuffle_ps directly. Those blocks split aligned and unaligned cases because SSE has separate instructions for them. v128.load accepts any alignment, so the split buys nothing on Wasm. Every function on the render path (AddPointwise, SubtractPointwise, MultiplyPointwise, ScalarMultiply) already has a generic #else loop that NEON uses today, and MultiplyPointwise, which this PR leaves untouched, runs through that loop on Wasm as-is.

Suggestion: keep the #ifdef SIMD_SSE guards as they were and handle only the three functions that have no #else. None of them is called outside the tests.

  • ReciprocalSqrt and Sqrt: change #elif defined SIMD_NEON to #elif defined(SIMD_NEON) || defined(SIMD_WASM).
  • ApproxComplexMagnitude: add a SIMD_WASM branch that replaces the two _mm_shuffle_ps calls with wasm_i32x4_shuffle(a, b, 0, 2, 4, 6) and wasm_i32x4_shuffle(a, b, 1, 3, 5, 7), or a scalar fallback.

Then this branch needs no xmmintrin.h. SIMD_RECIPROCAL_SQRT becomes wasm_f32x4_div(wasm_f32x4_splat(1.0f), wasm_f32x4_sqrt(a)), which is what Emscripten's _mm_rsqrt_ps expands to, and the v128_t/__m128 mixing that currently relies on clang's lax vector conversions goes away.

As written, the empty define also collides with the __SSE__ 1 that -msse predefines, which triggers -Wmacro-redefined.

#define SIMD_WASM
#include <wasm_simd128.h>
// Needed for _mm_rsqrt_ps
#include <xmmintrin.h>
typedef v128_t SimdVector;
#define SIMD_LENGTH 4
#define SIMD_MULTIPLY(a, b) wasm_f32x4_mul(a, b)
#define SIMD_ADD(a, b) wasm_f32x4_add(a, b)
#define SIMD_SUB(a, b) wasm_f32x4_sub(a, b)
#define SIMD_MULTIPLY_ADD(a, b, c) wasm_f32x4_add(wasm_f32x4_mul(a, b), c)
#define SIMD_SQRT(a) wasm_f32x4_sqrt(a)
#define SIMD_RECIPROCAL_SQRT(a) _mm_rsqrt_ps(a)
#define SIMD_LOAD_ONE_FLOAT(p) wasm_f32x4_splat(p)
#elif !defined(DISABLE_SIMD) && (defined(__x86_64__) || defined(_M_X64) || \
defined(i386) || defined(_M_IX86))
// SSE1 is enabled.
#include <xmmintrin.h>
typedef __m128 SimdVector;
Expand Down
36 changes: 18 additions & 18 deletions src/renderer/obr/obr_capi/obr/obr/audio_buffer/simd_utils.cc
Original file line number Diff line number Diff line change
Expand Up @@ -94,7 +94,7 @@ void AddPointwise(size_t length, const float* input_a, const float* input_b,
const SimdVector* input_b_vector =
reinterpret_cast<const SimdVector*>(input_b);
SimdVector* output_vector = reinterpret_cast<SimdVector*>(output);
#ifdef SIMD_SSE
#if defined(SIMD_SSE) || defined(SIMD_WASM)
const size_t num_chunks = GetNumChunks(length);
const bool inputs_aligned = IsAligned(input_a) && IsAligned(input_b);
const bool output_aligned = IsAligned(output);
Expand Down Expand Up @@ -126,7 +126,7 @@ void AddPointwise(size_t length, const float* input_a, const float* input_b,
for (size_t i = 0; i < GetNumChunks(length); ++i) {
output_vector[i] = SIMD_ADD(input_a_vector[i], input_b_vector[i]);
}
#endif // SIMD_SSE
#endif // defined(SIMD_SSE) || defined(SIMD_WASM)

// Add samples at the end that were missed by the SIMD chunking.
const size_t leftover_samples = GetLeftoverSamples(length);
Expand All @@ -148,7 +148,7 @@ void SubtractPointwise(size_t length, const float* input_a,
reinterpret_cast<const SimdVector*>(input_b);
SimdVector* output_vector = reinterpret_cast<SimdVector*>(output);

#ifdef SIMD_SSE
#if defined(SIMD_SSE) || defined(SIMD_WASM)
const size_t num_chunks = GetNumChunks(length);
const bool inputs_aligned = IsAligned(input_a) && IsAligned(input_b);
const bool output_aligned = IsAligned(output);
Expand Down Expand Up @@ -180,7 +180,7 @@ void SubtractPointwise(size_t length, const float* input_a,
for (size_t i = 0; i < GetNumChunks(length); ++i) {
output_vector[i] = SIMD_SUB(input_b_vector[i], input_a_vector[i]);
}
#endif // SIMD_SSE
#endif // defined(SIMD_SSE) || defined(SIMD_WASM)

// Subtract samples at the end that were missed by the SIMD chunking.
const size_t leftover_samples = GetLeftoverSamples(length);
Expand Down Expand Up @@ -256,7 +256,7 @@ void MultiplyAndAccumulatePointwise(size_t length, const float* input_a,
reinterpret_cast<const SimdVector*>(input_b);
SimdVector* accumulator_vector = reinterpret_cast<SimdVector*>(accumulator);

#ifdef SIMD_SSE
#if defined(SIMD_SSE) || defined(SIMD_WASM)
const size_t num_chunks = GetNumChunks(length);
const bool inputs_aligned = IsAligned(input_a) && IsAligned(input_b);
const bool accumulator_aligned = IsAligned(accumulator);
Expand Down Expand Up @@ -294,7 +294,7 @@ void MultiplyAndAccumulatePointwise(size_t length, const float* input_a,
accumulator_vector[i] = SIMD_MULTIPLY_ADD(
input_a_vector[i], input_b_vector[i], accumulator_vector[i]);
}
#endif // SIMD_SSE
#endif // defined(SIMD_SSE) || defined(SIMD_WASM)

// Apply gain and accumulate to samples at the end that were missed by the
// SIMD chunking.
Expand All @@ -314,7 +314,7 @@ void ScalarMultiply(size_t length, float gain, const float* input,
SimdVector* output_vector = reinterpret_cast<SimdVector*>(output);

const SimdVector gain_vector = SIMD_LOAD_ONE_FLOAT(gain);
#ifdef SIMD_SSE
#if defined(SIMD_SSE) || defined(SIMD_WASM)
const size_t num_chunks = GetNumChunks(length);
const bool input_aligned = IsAligned(input);
const bool output_aligned = IsAligned(output);
Expand Down Expand Up @@ -344,7 +344,7 @@ void ScalarMultiply(size_t length, float gain, const float* input,
for (size_t i = 0; i < GetNumChunks(length); ++i) {
output_vector[i] = SIMD_MULTIPLY(gain_vector, input_vector[i]);
}
#endif // SIMD_SSE
#endif // defined(SIMD_SSE) || defined(SIMD_WASM)

// Apply gain to samples at the end that were missed by the SIMD chunking.
const size_t leftover_samples = GetLeftoverSamples(length);
Expand All @@ -363,7 +363,7 @@ void ScalarMultiplyAndAccumulate(size_t length, float gain, const float* input,
SimdVector* accumulator_vector = reinterpret_cast<SimdVector*>(accumulator);

const SimdVector gain_vector = SIMD_LOAD_ONE_FLOAT(gain);
#ifdef SIMD_SSE
#if defined(SIMD_SSE) || defined(SIMD_WASM)
const size_t num_chunks = GetNumChunks(length);
const bool input_aligned = IsAligned(input);
const bool accumulator_aligned = IsAligned(accumulator);
Expand Down Expand Up @@ -399,7 +399,7 @@ void ScalarMultiplyAndAccumulate(size_t length, float gain, const float* input,
accumulator_vector[i] =
SIMD_MULTIPLY_ADD(gain_vector, input_vector[i], accumulator_vector[i]);
}
#endif // SIMD_SSE
#endif // defined(SIMD_SSE) || defined(SIMD_WASM)

// Apply gain and accumulate to samples at the end that were missed by the
// SIMD chunking.
Expand All @@ -419,7 +419,7 @@ void ReciprocalSqrt(size_t length, const float* input, float* output) {
SimdVector* output_vector = reinterpret_cast<SimdVector*>(output);
#endif // !defined(SIMD_DISABLED)

#ifdef SIMD_SSE
#if defined(SIMD_SSE) || defined(SIMD_WASM)
const size_t num_chunks = GetNumChunks(length);
const bool input_aligned = IsAligned(input);
const bool output_aligned = IsAligned(output);
Expand Down Expand Up @@ -448,7 +448,7 @@ void ReciprocalSqrt(size_t length, const float* input, float* output) {
for (size_t i = 0; i < GetNumChunks(length); ++i) {
output_vector[i] = SIMD_RECIPROCAL_SQRT(input_vector[i]);
}
#endif // SIMD_SSE
#endif // defined(SIMD_SSE) || defined(SIMD_WASM)

// Apply to samples at the end that were missed by the SIMD chunking.
const size_t leftover_samples = GetLeftoverSamples(length);
Expand All @@ -467,7 +467,7 @@ void Sqrt(size_t length, const float* input, float* output) {
SimdVector* output_vector = reinterpret_cast<SimdVector*>(output);
#endif // !defined(SIMD_DISABLED)

#ifdef SIMD_SSE
#if defined(SIMD_SSE) || defined(SIMD_WASM)
const size_t num_chunks = GetNumChunks(length);
const bool input_aligned = IsAligned(input);
const bool output_aligned = IsAligned(output);
Expand Down Expand Up @@ -497,7 +497,7 @@ void Sqrt(size_t length, const float* input, float* output) {
// This should be faster than using a sqrt method : https://goo.gl/XRKwFp
output_vector[i] = SIMD_SQRT(input_vector[i]);
}
#endif // SIMD_SSE
#endif // defined(SIMD_SSE) || defined(SIMD_WASM)

// Apply to samples at the end that were missed by the SIMD chunking.
const size_t leftover_samples = GetLeftoverSamples(length);
Expand All @@ -519,7 +519,7 @@ void ApproxComplexMagnitude(size_t length, const float* input, float* output) {
const bool output_aligned = IsAligned(output);
#endif // !defined(SIMD_DISABLED)

#ifdef SIMD_SSE
#if defined(SIMD_SSE) || defined(SIMD_WASM)
if (input_aligned && output_aligned) {
for (size_t out_index = 0; out_index < num_chunks; ++out_index) {
const size_t first_index = out_index * 2;
Expand Down Expand Up @@ -643,7 +643,7 @@ void ApproxComplexMagnitude(size_t length, const float* input, float* output) {
vst1q_f32(&output[out_index * SIMD_LENGTH], output_temp);
}
}
#endif // SIMD_SSE
#endif // SIMD_SSE || SIMD_WASM

// Apply to samples at the end that were missed by the SIMD chunking.
const size_t leftover_samples = GetLeftoverSamples(length);
Expand Down Expand Up @@ -712,7 +712,7 @@ void MonoFromStereoSimd(size_t length, const float* left, const float* right,
SimdVector* mono_vector = reinterpret_cast<SimdVector*>(mono);

const SimdVector inv_root_two_vec = SIMD_LOAD_ONE_FLOAT(kInverseSqrtTwo);
#ifdef SIMD_SSE
#if defined(SIMD_SSE) || defined(SIMD_WASM)
const size_t num_chunks = GetNumChunks(length);
const bool inputs_aligned = IsAligned(left) && IsAligned(right);
const bool mono_aligned = IsAligned(mono);
Expand Down Expand Up @@ -748,7 +748,7 @@ void MonoFromStereoSimd(size_t length, const float* left, const float* right,
mono_vector[i] = SIMD_MULTIPLY(inv_root_two_vec,
SIMD_ADD(left_vector[i], right_vector[i]));
}
#endif // SIMD_SSE
#endif // defined(SIMD_SSE) || defined(SIMD_WASM)
const size_t leftover_samples = GetLeftoverSamples(length);
// Downmix samples at the end that were missed by the SIMD chunking.
ABSL_DCHECK_GE(length, leftover_samples);
Expand Down
Loading