Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
78 changes: 77 additions & 1 deletion .github/workflows/public.continuous.yml
Original file line number Diff line number Diff line change
Expand Up @@ -113,4 +113,80 @@ jobs:
cmake -D BUILD_TBB_FROM_SOURCE=ON -D EMBREE_EXTRA_OPTIONS="-DBUILD_TESTING=ON -DEMBREE_TUTORIALS=ON -DEMBREE_ISPC_SUPPORT=ON -DEMBREE_TESTING_INTENSITY=3" ../superbuild
cmake --build .
cd embree/build
ctest
ctest

windows-11-arm:
runs-on: windows-11-arm
steps:
- name: Checkout Repository
uses: actions/checkout@v4

- name: Build and Run
shell: pwsh
run: |
$opts = "-DBUILD_TESTING=ON"
$opts += " -DEMBREE_TUTORIALS=ON"
$opts += " -DEMBREE_ISPC_SUPPORT=OFF"
$opts += " -DEMBREE_TASKING_SYSTEM=INTERNAL"
$opts += " -DEMBREE_TESTING_INTENSITY=2"

cmake -B build -G "Visual Studio 17 2022" -A ARM64 $opts.Split(" ") .
cmake --build build --config Release
ctest --test-dir build -C Release --output-on-failure

windows-11:
runs-on: windows-latest
steps:
- name: Checkout Repository
uses: actions/checkout@v4

- name: Build and Run
shell: pwsh
run: |
$opts = "-DBUILD_TESTING=ON"
$opts += " -DEMBREE_TUTORIALS=ON"
$opts += " -DEMBREE_ISPC_SUPPORT=OFF"
$opts += " -DEMBREE_TASKING_SYSTEM=INTERNAL"
$opts += " -DEMBREE_TESTING_INTENSITY=2"

cmake -B build -G "Visual Studio 17 2022" $opts.Split(" ") .
cmake --build build --config Release
ctest --test-dir build -C Release --output-on-failure

macos-26:
runs-on: macos-26
steps:
- name: Install packages (macOS 26)
run: |
brew update
brew install cmake git-lfs freeglut glfw tbb

- name: Checkout Repository
uses: actions/checkout@v4

- name: Build and Run
run: |
mkdir build
cd build
cmake -DBUILD_TESTING=ON -DEMBREE_TUTORIALS=ON -DEMBREE_ISPC_SUPPORT=OFF -DEMBREE_TESTING_INTENSITY=3 ..
make -j$(sysctl -n hw.ncpu)
ctest --test-dir . -C Release --output-on-failure

macos-26-intel:
runs-on: macos-26-intel
steps:
- name: Install packages (macOS 26)
run: |
brew update
brew install cmake git-lfs freeglut glfw tbb

- name: Checkout Repository
uses: actions/checkout@v4

- name: Build and Run
run: |
mkdir build
cd build
cmake -DBUILD_TESTING=ON -DEMBREE_TUTORIALS=ON -DEMBREE_ISPC_SUPPORT=OFF -DEMBREE_TESTING_INTENSITY=3 ..
make -j$(sysctl -n hw.ncpu)
ctest --test-dir . -C Release --output-on-failure
4 changes: 4 additions & 0 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -221,6 +221,10 @@ OPTION(EMBREE_MIN_WIDTH "Enables min-width feature to enlarge curve and point th
IF (APPLE AND CMAKE_SYSTEM_NAME STREQUAL "Darwin" AND (CMAKE_SYSTEM_PROCESSOR STREQUAL "arm64" AND CMAKE_OSX_ARCHITECTURES STREQUAL "") OR ("arm64" IN_LIST CMAKE_OSX_ARCHITECTURES))
MESSAGE(STATUS "Building for Apple silicon")
SET(EMBREE_ARM ON)
# CMAKE_SYSTEM_PROCESSOR is unreliable on windows where it would report AMD64 with cross compilation
ELSEIF(CMAKE_SYSTEM_NAME STREQUAL "Windows" AND CMAKE_GENERATOR_PLATFORM STREQUAL "ARM64")
MESSAGE(STATUS "Building for Windows ARM64 (MSVC)")
SET(EMBREE_ARM ON)
ELSEIF(CMAKE_SYSTEM_PROCESSOR STREQUAL "aarch64" OR CMAKE_SYSTEM_PROCESSOR STREQUAL "ARM64")
MESSAGE(STATUS "Building for AArch64")
SET(EMBREE_ARM ON)
Expand Down
2 changes: 1 addition & 1 deletion common/cmake/check_arm_neon.cpp
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
// Copyright 2009-2020 Intel Corporation
// SPDX-License-Identifier: Apache-2.0

#if !defined(__ARM_NEON)
#if !defined(__ARM_NEON) && !defined(_M_ARM64)
#error "No ARM Neon support"
#endif

Expand Down
23 changes: 17 additions & 6 deletions common/cmake/msvc.cmake
Original file line number Diff line number Diff line change
@@ -1,12 +1,19 @@
## Copyright 2009-2021 Intel Corporation
## SPDX-License-Identifier: Apache-2.0

SET(FLAGS_SSE2 "/D__SSE__ /D__SSE2__")
SET(FLAGS_SSE42 "${FLAGS_SSE2} /D__SSE3__ /D__SSSE3__ /D__SSE4_1__ /D__SSE4_2__")
SET(FLAGS_AVX "${FLAGS_SSE42} /arch:AVX")
SET(FLAGS_AVX2 "${FLAGS_SSE42} /arch:AVX2")
SET(FLAGS_AVX512 "${FLAGS_AVX2} /arch:AVX512")
SET(FLAGS_APX "${FLAGS_AVX512} /arch:AVX10.2 /vlen=512 /feature:APX")
IF (EMBREE_ARM)
SET(FLAGS_SSE2 "/D__SSE__ /D__SSE2__")
SET(FLAGS_SSE42 "/D__SSE4_2__ /D__SSE4_1__")
SET(FLAGS_AVX "/D__AVX__ /D__SSE4_2__ /D__SSE4_1__ /D__BMI__ /D__BMI2__ /D__LZCNT__")
SET(FLAGS_AVX2 "/D__AVX2__ /D__AVX__ /D__SSE4_2__ /D__SSE4_1__ /D__BMI__ /D__BMI2__ /D__LZCNT__")
ELSE()
SET(FLAGS_SSE2 "/D__SSE__ /D__SSE2__")
SET(FLAGS_SSE42 "${FLAGS_SSE2} /D__SSE3__ /D__SSSE3__ /D__SSE4_1__ /D__SSE4_2__")
SET(FLAGS_AVX "${FLAGS_SSE42} /arch:AVX")
SET(FLAGS_AVX2 "${FLAGS_SSE42} /arch:AVX2")
SET(FLAGS_AVX512 "${FLAGS_AVX2} /arch:AVX512")
SET(FLAGS_APX "${FLAGS_AVX512} /arch:AVX10.2 /vlen=512 /feature:APX")
ENDIF()

SET(COMMON_CXX_FLAGS "")
SET(COMMON_CXX_FLAGS "${COMMON_CXX_FLAGS} /EHsc") # catch C++ exceptions only and extern "C" functions never throw a C++ exception
Expand All @@ -18,6 +25,10 @@ IF (EMBREE_STACK_PROTECTOR)
ELSE()
SET(COMMON_CXX_FLAGS "${COMMON_CXX_FLAGS} /GS-") # do not protect against return address overrides
ENDIF()
IF (EMBREE_ARM)
# sse2neon uses the new preprocessor
SET(COMMON_CXX_FLAGS "${COMMON_CXX_FLAGS} /Zc:preprocessor")
ENDIF()
MACRO(DISABLE_STACK_PROTECTOR_FOR_FILE file)
IF (EMBREE_STACK_PROTECTOR)
SET_SOURCE_FILES_PROPERTIES(${file} PROPERTIES COMPILE_FLAGS "/GS-")
Expand Down
4 changes: 2 additions & 2 deletions common/math/bbox.h
Original file line number Diff line number Diff line change
Expand Up @@ -82,7 +82,7 @@ namespace embree
return lower > upper;
}

#if defined(__SSE__) || defined(__ARM_NEON)
#if defined(__SSE__) || defined(__ARM_NEON) || defined(EMBREE_ARM64)
template<> __forceinline bool BBox<Vec3fa>::empty() const {
return !all(le_mask(lower,upper));
}
Expand Down Expand Up @@ -233,7 +233,7 @@ namespace embree
/// SSE / AVX / MIC specializations
////////////////////////////////////////////////////////////////////////////////

#if defined (__SSE__) || defined(__ARM_NEON)
#if defined (__SSE__) || defined(__ARM_NEON) || defined(EMBREE_ARM64)
#include "../simd/sse.h"
#endif

Expand Down
8 changes: 4 additions & 4 deletions common/math/color.h
Original file line number Diff line number Diff line change
Expand Up @@ -160,7 +160,7 @@ namespace embree
}
__forceinline const Color rcp ( const Color& a )
{
#if defined(__aarch64__)
#if defined(EMBREE_ARM64)
__m128 reciprocal = _mm_rcp_ps(a.m128);
reciprocal = vmulq_f32(vrecpsq_f32(a.m128, reciprocal), reciprocal);
reciprocal = vmulq_f32(vrecpsq_f32(a.m128, reciprocal), reciprocal);
Expand All @@ -173,11 +173,11 @@ namespace embree
#endif
return _mm_add_ps(r,_mm_mul_ps(r, _mm_sub_ps(_mm_set1_ps(1.0f), _mm_mul_ps(a, r)))); // computes r + r * (1 - a * r)

#endif //defined(__aarch64__)
#endif //defined(EMBREE_ARM64)
}
__forceinline const Color rsqrt( const Color& a )
{
#if defined(__aarch64__)
#if defined(EMBREE_ARM64)
__m128 r = _mm_rsqrt_ps(a.m128);
r = vmulq_f32(r, vrsqrtsq_f32(vmulq_f32(a.m128, r), r));
r = vmulq_f32(r, vrsqrtsq_f32(vmulq_f32(a.m128, r), r));
Expand All @@ -191,7 +191,7 @@ namespace embree
#endif
return _mm_add_ps(_mm_mul_ps(_mm_set1_ps(1.5f),r), _mm_mul_ps(_mm_mul_ps(_mm_mul_ps(a, _mm_set1_ps(-0.5f)), r), _mm_mul_ps(r, r)));

#endif //defined(__aarch64__)
#endif //defined(EMBREE_ARM64)
}
__forceinline const Color sqrt ( const Color& a ) { return _mm_sqrt_ps(a.m128); }

Expand Down
78 changes: 33 additions & 45 deletions common/math/emath.h
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@
# include "math_sycl.h"
#else

#if defined(__ARM_NEON)
#if defined(__ARM_NEON) || defined(EMBREE_ARM64)
#include "../simd/arm/emulation.h"
#else
#include <emmintrin.h>
Expand Down Expand Up @@ -60,14 +60,13 @@ namespace embree

__forceinline float rcp ( const float x )
{
#if defined(__aarch64__)
#if defined(EMBREE_ARM64)
// Move scalar to vector register and do rcp.
__m128 a;
a[0] = x;
__m128 a = vdupq_n_f32(x);
float32x4_t reciprocal = vrecpeq_f32(a);
reciprocal = vmulq_f32(vrecpsq_f32(a, reciprocal), reciprocal);
reciprocal = vmulq_f32(vrecpsq_f32(a, reciprocal), reciprocal);
return reciprocal[0];
return vgetq_lane_f32(reciprocal, 0);
#else

const __m128 a = _mm_set_ss(x);
Expand All @@ -84,58 +83,51 @@ namespace embree
return _mm_cvtss_f32(_mm_mul_ss(r,_mm_sub_ss(_mm_set_ss(2.0f), _mm_mul_ss(r, a))));
#endif

#endif //defined(__aarch64__)
#endif //defined(EMBREE_ARM64)
}

__forceinline float signmsk ( const float x ) {
#if defined(__aarch64__)
#if defined(EMBREE_ARM64)
// FP and Neon shares same vector register in arm64
__m128 a;
__m128i b;
a[0] = x;
b[0] = 0x80000000;
__m128 a = vdupq_n_f32(x);
__m128i b = vdupq_n_s32(0x80000000);
a = _mm_and_ps(a, vreinterpretq_f32_s32(b));
return a[0];
return vgetq_lane_f32(a, 0);
#else
return _mm_cvtss_f32(_mm_and_ps(_mm_set_ss(x),_mm_castsi128_ps(_mm_set1_epi32(0x80000000))));
#endif
}
__forceinline float xorf( const float x, const float y ) {
#if defined(__aarch64__)
#if defined(EMBREE_ARM64)
// FP and Neon shares same vector register in arm64
__m128 a;
__m128 b;
a[0] = x;
b[0] = y;
__m128 a = vdupq_n_f32(x);
__m128 b = vdupq_n_f32(y);
a = _mm_xor_ps(a, b);
return a[0];
return vgetq_lane_f32(a, 0);
#else
return _mm_cvtss_f32(_mm_xor_ps(_mm_set_ss(x),_mm_set_ss(y)));
#endif
}
__forceinline float andf( const float x, const unsigned y ) {
#if defined(__aarch64__)
#if defined(EMBREE_ARM64)
// FP and Neon shares same vector register in arm64
__m128 a;
__m128i b;
a[0] = x;
b[0] = y;
__m128 a = vdupq_n_f32(x);
__m128i b = vdupq_n_u32(y);
a = _mm_and_ps(a, vreinterpretq_f32_s32(b));
return a[0];
return vgetq_lane_f32(a, 0);
#else
return _mm_cvtss_f32(_mm_and_ps(_mm_set_ss(x),_mm_castsi128_ps(_mm_set1_epi32(y))));
#endif
}
__forceinline float rsqrt( const float x )
{
#if defined(__aarch64__)
#if defined(EMBREE_ARM64)
// FP and Neon shares same vector register in arm64
__m128 a;
a[0] = x;
__m128 a = vdupq_n_f32(x);
__m128 value = _mm_rsqrt_ps(a);
value = vmulq_f32(value, vrsqrtsq_f32(vmulq_f32(a, value), value));
value = vmulq_f32(value, vrsqrtsq_f32(vmulq_f32(a, value), value));
return value[0];
return vgetq_lane_f32(value, 0);
#else

const __m128 a = _mm_set_ss(x);
Expand Down Expand Up @@ -204,15 +196,13 @@ namespace embree
__forceinline double floor( const double x ) { return ::floor (x); }
__forceinline double ceil ( const double x ) { return ::ceil (x); }

#if defined(__aarch64__)
#if defined(EMBREE_ARM64)
__forceinline float mini(float a, float b) {
// FP and Neon shares same vector register in arm64
__m128 x;
__m128 y;
x[0] = a;
y[0] = b;
x = _mm_min_ps(x, y);
return x[0];
// FP and Neon shares same vector register in arm64
__m128 x = vdupq_n_f32(a);
__m128 y = vdupq_n_f32(b);
x = _mm_min_ps(x, y);
return vgetq_lane_f32(x, 0);
}
#elif defined(__SSE4_1__)
__forceinline float mini(float a, float b) {
Expand All @@ -223,15 +213,13 @@ namespace embree
}
#endif

#if defined(__aarch64__)
#if defined(EMBREE_ARM64)
__forceinline float maxi(float a, float b) {
// FP and Neon shares same vector register in arm64
__m128 x;
__m128 y;
x[0] = a;
y[0] = b;
__m128 x = vdupq_n_f32(a);
__m128 y = vdupq_n_f32(b);
x = _mm_max_ps(x, y);
return x[0];
return vgetq_lane_f32(x, 0);
}
#elif defined(__SSE4_1__)
__forceinline float maxi(float a, float b) {
Expand All @@ -250,7 +238,7 @@ namespace embree
__forceinline int64_t min(int64_t a, int64_t b) { return a<b ? a:b; }
__forceinline float min(float a, float b) { return a<b ? a:b; }
__forceinline double min(double a, double b) { return a<b ? a:b; }
#if defined(__64BIT__) || defined(__EMSCRIPTEN__)
#if defined(__64BIT__) || defined(__EMSCRIPTEN__) || (defined(_M_ARM64) && !defined(__clang__))
__forceinline size_t min(size_t a, size_t b) { return a<b ? a:b; }
#endif
#if defined(__EMSCRIPTEN__)
Expand All @@ -270,7 +258,7 @@ namespace embree
__forceinline int64_t max(int64_t a, int64_t b) { return a<b ? b:a; }
__forceinline float max(float a, float b) { return a<b ? b:a; }
__forceinline double max(double a, double b) { return a<b ? b:a; }
#if defined(__64BIT__) || defined(__EMSCRIPTEN__)
#if defined(__64BIT__) || defined(__EMSCRIPTEN__) || (defined(_M_ARM64) && !defined(__clang__))
__forceinline size_t max(size_t a, size_t b) { return a<b ? b:a; }
#endif
#if defined(__EMSCRIPTEN__)
Expand Down Expand Up @@ -423,7 +411,7 @@ __forceinline float nmsub ( const float a, const float b, const float c) { retur
return x | (y << 1) | (z << 2);
}

#if defined(__AVX2__) && !defined(__aarch64__)
#if defined(__AVX2__) && !defined(EMBREE_ARM64)

template<>
__forceinline unsigned int bitInterleave(const unsigned int &xi, const unsigned int& yi, const unsigned int& zi)
Expand Down
4 changes: 2 additions & 2 deletions common/math/vec2.h
Original file line number Diff line number Diff line change
Expand Up @@ -205,7 +205,7 @@ namespace embree

#include "vec2fa.h"

#if defined(__SSE__) || defined(__ARM_NEON)
#if defined(__SSE__) || defined(__ARM_NEON) || defined(EMBREE_ARM64)
#include "../simd/sse.h"
#endif

Expand All @@ -221,7 +221,7 @@ namespace embree
{
template<> __forceinline Vec2<float>::Vec2(const Vec2fa& a) : x(a.x), y(a.y) {}

#if defined(__SSE__) || defined(__ARM_NEON)
#if defined(__SSE__) || defined(__ARM_NEON) || defined(EMBREE_ARM64)
template<> __forceinline Vec2<vfloat4>::Vec2(const Vec2fa& a) : x(a.x), y(a.y) {}
#endif

Expand Down
Loading
Loading