Add Chromium-only Blender WebEngine parity work
This commit is contained in:
296
blender-5.2.0/intern/cycles/kernel/device/oneapi/CMakeLists.txt
Normal file
296
blender-5.2.0/intern/cycles/kernel/device/oneapi/CMakeLists.txt
Normal file
@@ -0,0 +1,296 @@
|
||||
# SPDX-FileCopyrightText: 2011-2026 Blender Foundation
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
set(INC
|
||||
../../..
|
||||
)
|
||||
|
||||
set(INC_SYS
|
||||
|
||||
)
|
||||
|
||||
set(SRC_KERNEL_DEVICE_ONEAPI
|
||||
kernel.cpp
|
||||
)
|
||||
|
||||
set(SRC_KERNEL_DEVICE_ONEAPI_HEADERS
|
||||
compat.h
|
||||
context_begin.h
|
||||
context_end.h
|
||||
context_intersect_begin.h
|
||||
context_intersect_end.h
|
||||
globals.h
|
||||
kernel.h
|
||||
kernel_templates.h
|
||||
../cpu/bvh.h
|
||||
)
|
||||
|
||||
set(LIB
|
||||
|
||||
)
|
||||
|
||||
if(WITH_CYCLES_DEVICE_ONEAPI)
|
||||
if(WITH_CYCLES_ONEAPI_BINARIES)
|
||||
set(cycles_kernel_oneapi_lib_suffix "_aot")
|
||||
else()
|
||||
set(cycles_kernel_oneapi_lib_suffix "_jit")
|
||||
endif()
|
||||
|
||||
if(WIN32)
|
||||
set(cycles_kernel_oneapi_lib ${CMAKE_CURRENT_BINARY_DIR}/cycles_kernel_oneapi${cycles_kernel_oneapi_lib_suffix}.dll)
|
||||
set(cycles_kernel_oneapi_linker_lib ${CMAKE_CURRENT_BINARY_DIR}/cycles_kernel_oneapi${cycles_kernel_oneapi_lib_suffix}.lib)
|
||||
else()
|
||||
set(cycles_kernel_oneapi_lib ${CMAKE_CURRENT_BINARY_DIR}/libcycles_kernel_oneapi${cycles_kernel_oneapi_lib_suffix}.so)
|
||||
endif()
|
||||
|
||||
set(cycles_oneapi_kernel_sources
|
||||
${SRC_KERNEL_DEVICE_ONEAPI}
|
||||
${SRC_KERNEL_DEVICE_ONEAPI_HEADERS}
|
||||
$<TARGET_PROPERTY:cycles_kernel,INTERFACE_SOURCES>
|
||||
)
|
||||
|
||||
set(SYCL_OFFLINE_COMPILER_PARALLEL_JOBS 1 CACHE STRING "Number of parallel compiler instances to use for device binaries compilation (expect ~8GB peak memory usage per instance).")
|
||||
mark_as_advanced(SYCL_OFFLINE_COMPILER_PARALLEL_JOBS)
|
||||
|
||||
if(WITH_CYCLES_ONEAPI_BINARIES)
|
||||
message(STATUS "${SYCL_OFFLINE_COMPILER_PARALLEL_JOBS} instance(s) of oneAPI offline compiler will be used.")
|
||||
endif()
|
||||
set(sycl_compiler_flags
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/${SRC_KERNEL_DEVICE_ONEAPI}
|
||||
-fsycl
|
||||
-fsycl-unnamed-lambda
|
||||
-fdelayed-template-parsing
|
||||
-fsycl-device-code-split=per_kernel
|
||||
-fsycl-max-parallel-link-jobs=${SYCL_OFFLINE_COMPILER_PARALLEL_JOBS}
|
||||
--offload-compress
|
||||
--offload-compression-level=19
|
||||
-shared
|
||||
-DWITH_ONEAPI
|
||||
-O2
|
||||
-ffast-math
|
||||
-D__KERNEL_LOCAL_ATOMIC_SORT__
|
||||
-o"${cycles_kernel_oneapi_lib}"
|
||||
-I"${CMAKE_CURRENT_SOURCE_DIR}/../../.."
|
||||
)
|
||||
# SYCL_CPP_FLAGS is a variable that the user can set to pass extra compiler options.
|
||||
if(DEFINED SYCL_CPP_FLAGS)
|
||||
list(APPEND sycl_compiler_flags ${SYCL_CPP_FLAGS})
|
||||
endif()
|
||||
|
||||
# Set defaults for spir64 and spir64_gen options
|
||||
if(NOT DEFINED CYCLES_ONEAPI_SYCL_OPTIONS_spir64)
|
||||
set(CYCLES_ONEAPI_SYCL_OPTIONS_spir64 "-options '-cl-fast-relaxed-math -ze-intel-enable-auto-large-GRF-mode -ze-opt-regular-grf-kernel integrator_intersect -ze-opt-large-grf-kernel shade_surface -ze-opt-no-local-to-generic'")
|
||||
endif()
|
||||
if(NOT DEFINED CYCLES_ONEAPI_SYCL_OPTIONS_spir64_gen)
|
||||
set(CYCLES_ONEAPI_SYCL_OPTIONS_spir64_gen "${CYCLES_ONEAPI_SYCL_OPTIONS_spir64}" CACHE STRING "Extra build options for spir64_gen target")
|
||||
mark_as_advanced(CYCLES_ONEAPI_SYCL_OPTIONS_spir64_gen)
|
||||
endif()
|
||||
# Enable `zebin`, a graphics binary format with improved compatibility.
|
||||
string(PREPEND CYCLES_ONEAPI_SYCL_OPTIONS_spir64_gen "--format zebin ")
|
||||
|
||||
# Host execution won't use GPU binaries, no need to compile them.
|
||||
if(WITH_CYCLES_ONEAPI_BINARIES)
|
||||
# Add the list of Intel devices to build binaries for.
|
||||
foreach(device ${CYCLES_ONEAPI_INTEL_BINARIES_ARCH})
|
||||
# Run `ocloc` ids to test if the device is supported.
|
||||
execute_process(
|
||||
COMMAND ${OCLOC_ENV_COMMAND} ${OCLOC_BINARY_FULL_FILEPATH} ids ${device}
|
||||
RESULT_VARIABLE oclocids_ret
|
||||
OUTPUT_QUIET
|
||||
ERROR_QUIET
|
||||
)
|
||||
if(NOT oclocids_ret EQUAL 0)
|
||||
list(REMOVE_ITEM CYCLES_ONEAPI_INTEL_BINARIES_ARCH ${device})
|
||||
message(STATUS
|
||||
"Cycles oneAPI: "
|
||||
"binaries for ${device} not supported by Intel Graphics Compiler/ocloc, skipped."
|
||||
)
|
||||
endif()
|
||||
endforeach()
|
||||
list(JOIN CYCLES_ONEAPI_INTEL_BINARIES_ARCH "," gen_devices_string)
|
||||
if("${gen_devices_string}" STREQUAL "")
|
||||
# Don't compile spir64_gen if no device is targeted
|
||||
message(STATUS "Cycles oneAPI: skipping spir64_gen compilation as no devices are targeted.")
|
||||
list(REMOVE_ITEM CYCLES_ONEAPI_SYCL_TARGETS spir64_gen)
|
||||
else()
|
||||
string(PREPEND CYCLES_ONEAPI_SYCL_OPTIONS_spir64_gen "-device ${gen_devices_string} ")
|
||||
endif()
|
||||
else()
|
||||
list(REMOVE_ITEM CYCLES_ONEAPI_SYCL_TARGETS spir64_gen)
|
||||
endif()
|
||||
|
||||
# Iterate over all targets and their options.
|
||||
list(JOIN CYCLES_ONEAPI_SYCL_TARGETS "," targets_string)
|
||||
list(APPEND sycl_compiler_flags -fsycl-targets=${targets_string})
|
||||
foreach(target ${CYCLES_ONEAPI_SYCL_TARGETS})
|
||||
if(DEFINED CYCLES_ONEAPI_SYCL_OPTIONS_${target})
|
||||
list(APPEND sycl_compiler_flags
|
||||
"-Xsycl-target-backend=${target} \"${CYCLES_ONEAPI_SYCL_OPTIONS_${target}}\""
|
||||
)
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
if(WITH_NANOVDB)
|
||||
list(APPEND sycl_compiler_flags
|
||||
-DWITH_NANOVDB)
|
||||
endif()
|
||||
|
||||
if(WITH_CYCLES_EMBREE AND EMBREE_SYCL_SUPPORT)
|
||||
list(APPEND sycl_compiler_flags
|
||||
-DWITH_EMBREE
|
||||
-DWITH_EMBREE_GPU
|
||||
-DEMBREE_MAJOR_VERSION=${EMBREE_MAJOR_VERSION}
|
||||
-I"${EMBREE_INCLUDE_DIRS}")
|
||||
|
||||
if(WIN32)
|
||||
list(APPEND sycl_compiler_flags
|
||||
-ladvapi32.lib
|
||||
)
|
||||
endif()
|
||||
|
||||
set(next_library_mode "")
|
||||
foreach(library ${EMBREE_LIBRARIES})
|
||||
string(TOLOWER "${library}" library_lower)
|
||||
if(("${library_lower}" STREQUAL "optimized") OR
|
||||
("${library_lower}" STREQUAL "debug"))
|
||||
set(next_library_mode "${library_lower}")
|
||||
else()
|
||||
if(next_library_mode STREQUAL "")
|
||||
list(APPEND EMBREE_TBB_LIBRARIES_optimized ${library})
|
||||
list(APPEND EMBREE_TBB_LIBRARIES_debug ${library})
|
||||
else()
|
||||
list(APPEND EMBREE_TBB_LIBRARIES_${next_library_mode} ${library})
|
||||
endif()
|
||||
set(next_library_mode "")
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
foreach(library ${TBB_LIBRARIES})
|
||||
string(TOLOWER "${library}" library_lower)
|
||||
if(("${library_lower}" STREQUAL "optimized") OR
|
||||
("${library_lower}" STREQUAL "debug"))
|
||||
set(next_library_mode "${library_lower}")
|
||||
else()
|
||||
if(next_library_mode STREQUAL "")
|
||||
list(APPEND EMBREE_TBB_LIBRARIES_optimized ${library})
|
||||
list(APPEND EMBREE_TBB_LIBRARIES_debug ${library})
|
||||
else()
|
||||
list(APPEND EMBREE_TBB_LIBRARIES_${next_library_mode} ${library})
|
||||
endif()
|
||||
set(next_library_mode "")
|
||||
endif()
|
||||
endforeach()
|
||||
list(APPEND sycl_compiler_flags
|
||||
"$<$<CONFIG:Release>:${EMBREE_TBB_LIBRARIES_optimized}>"
|
||||
"$<$<CONFIG:RelWithDebInfo>:${EMBREE_TBB_LIBRARIES_optimized}>"
|
||||
"$<$<CONFIG:MinSizeRel>:${EMBREE_TBB_LIBRARIES_optimized}>"
|
||||
"$<$<CONFIG:Debug>:${EMBREE_TBB_LIBRARIES_debug}>"
|
||||
)
|
||||
endif()
|
||||
|
||||
if(WITH_CYCLES_DEBUG)
|
||||
list(APPEND sycl_compiler_flags -DWITH_CYCLES_DEBUG)
|
||||
endif()
|
||||
|
||||
get_filename_component(sycl_compiler_root ${SYCL_COMPILER} DIRECTORY)
|
||||
|
||||
if(WIN32) # Add Windows specific compiler flags.
|
||||
list(APPEND sycl_compiler_flags
|
||||
-fms-extensions
|
||||
-fms-compatibility
|
||||
-D_WINDLL
|
||||
-D_MBCS
|
||||
-DWIN32
|
||||
-D_WINDOWS
|
||||
-D_CRT_NONSTDC_NO_DEPRECATE
|
||||
-D_CRT_SECURE_NO_DEPRECATE
|
||||
-DONEAPI_EXPORT
|
||||
)
|
||||
else() # Add Linux specific compiler flags.
|
||||
list(APPEND sycl_compiler_flags -fPIC)
|
||||
list(APPEND sycl_compiler_flags -fvisibility=hidden)
|
||||
|
||||
# Add $ORIGIN to `cycles_kernel_oneapi.so` RPATH so `libsycl.so` and
|
||||
# `libpi_level_zero.so` can be placed next to it and get found.
|
||||
list(APPEND sycl_compiler_flags -Wl,-rpath,'$$ORIGIN')
|
||||
endif()
|
||||
|
||||
# Create CONFIG specific compiler flags.
|
||||
set(sycl_compiler_flags_Release ${sycl_compiler_flags})
|
||||
set(sycl_compiler_flags_Debug ${sycl_compiler_flags})
|
||||
set(sycl_compiler_flags_RelWithDebInfo ${sycl_compiler_flags})
|
||||
|
||||
list(APPEND sycl_compiler_flags_Release
|
||||
-DNDEBUG
|
||||
)
|
||||
list(APPEND sycl_compiler_flags_RelWithDebInfo
|
||||
-DNDEBUG
|
||||
-g
|
||||
)
|
||||
list(APPEND sycl_compiler_flags_Debug
|
||||
-g
|
||||
)
|
||||
|
||||
if(WIN32)
|
||||
list(APPEND sycl_compiler_flags_Debug
|
||||
-D_DEBUG
|
||||
-nostdlib
|
||||
-Xclang --dependent-lib=msvcrtd
|
||||
)
|
||||
|
||||
list(APPEND sycl_compiler_flags
|
||||
-L"${sycl_compiler_root}/../lib" # To find sycl.lib
|
||||
-L"${sycl_compiler_root}/../compiler/lib/intel64_win" # To find libircmt.lib (when using `icpx`)
|
||||
)
|
||||
|
||||
add_custom_command(
|
||||
OUTPUT ${cycles_kernel_oneapi_lib} ${cycles_kernel_oneapi_linker_lib}
|
||||
COMMAND ${CMAKE_COMMAND} -E env
|
||||
"PATH=${OCLOC_INSTALL_DIR}\;${sycl_compiler_root}"
|
||||
${SYCL_COMPILER}
|
||||
"$<$<CONFIG:Release>:${sycl_compiler_flags_Release}>"
|
||||
"$<$<CONFIG:RelWithDebInfo>:${sycl_compiler_flags_RelWithDebInfo}>"
|
||||
"$<$<CONFIG:Debug>:${sycl_compiler_flags_Debug}>"
|
||||
"$<$<CONFIG:MinSizeRel>:${sycl_compiler_flags_Release}>"
|
||||
COMMAND_EXPAND_LISTS
|
||||
DEPENDS ${cycles_oneapi_kernel_sources} ${SYCL_COMPILER})
|
||||
else()
|
||||
# The following join/replace operations are to prevent cmake from
|
||||
# escaping space chars with backslashes in add_custom_command.
|
||||
list(JOIN sycl_compiler_flags_Release " " sycl_compiler_flags_Release_str)
|
||||
string(REPLACE " " ";" sycl_compiler_flags_Release_str ${sycl_compiler_flags_Release_str})
|
||||
list(JOIN sycl_compiler_flags_RelWithDebInfo " " sycl_compiler_flags_RelWithDebInfo_str)
|
||||
string(REPLACE " " ";" sycl_compiler_flags_RelWithDebInfo_str ${sycl_compiler_flags_RelWithDebInfo_str})
|
||||
list(JOIN sycl_compiler_flags_Debug " " sycl_compiler_flags_Debug_str)
|
||||
string(REPLACE " " ";" sycl_compiler_flags_Debug_str ${sycl_compiler_flags_Debug_str})
|
||||
add_custom_command(
|
||||
OUTPUT ${cycles_kernel_oneapi_lib}
|
||||
COMMAND
|
||||
${CMAKE_COMMAND} -E env
|
||||
"LD_LIBRARY_PATH=${sycl_compiler_root}/../lib:${OCLOC_LD_LIBRARY_PATH}"
|
||||
# `$ENV{PATH}` is for compiler to find `ld`.
|
||||
"PATH=${OCLOC_INSTALL_DIR}/bin:${sycl_compiler_root}:$ENV{PATH}"
|
||||
${SYCL_COMPILER}
|
||||
"$<$<CONFIG:Release>:${sycl_compiler_flags_Release_str}>"
|
||||
"$<$<CONFIG:RelWithDebInfo>:${sycl_compiler_flags_RelWithDebInfo_str}>"
|
||||
"$<$<CONFIG:Debug>:${sycl_compiler_flags_Debug_str}>"
|
||||
"$<$<CONFIG:MinSizeRel>:${sycl_compiler_flags_Release_str}>"
|
||||
COMMAND_EXPAND_LISTS
|
||||
DEPENDS ${cycles_oneapi_kernel_sources} ${SYCL_COMPILER})
|
||||
endif()
|
||||
|
||||
# install dynamic libraries required at runtime
|
||||
delayed_install("" "${cycles_kernel_oneapi_lib}" ${cycles_kernel_runtime_lib_target_path})
|
||||
|
||||
add_custom_target(cycles_kernel_oneapi
|
||||
ALL
|
||||
DEPENDS ${cycles_kernel_oneapi_lib}
|
||||
SOURCES ${SRC_KERNEL_DEVICE_ONEAPI} ${SRC_KERNEL_DEVICE_ONEAPI_HEADERS}
|
||||
)
|
||||
cycles_set_solution_folder(cycles_kernel_oneapi)
|
||||
|
||||
source_group("device\\oneapi" FILES ${SRC_KERNEL_DEVICE_ONEAPI} ${SRC_KERNEL_DEVICE_ONEAPI_HEADERS})
|
||||
|
||||
add_dependencies(cycles_kernel cycles_kernel_oneapi)
|
||||
endif()
|
||||
270
blender-5.2.0/intern/cycles/kernel/device/oneapi/compat.h
Normal file
270
blender-5.2.0/intern/cycles/kernel/device/oneapi/compat.h
Normal file
@@ -0,0 +1,270 @@
|
||||
/* SPDX-FileCopyrightText: 2021-2022 Intel Corporation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
|
||||
#pragma once
|
||||
|
||||
#define __KERNEL_GPU__
|
||||
#define __KERNEL_ONEAPI__
|
||||
#define __KERNEL_64_BIT__
|
||||
|
||||
#ifdef WITH_EMBREE_GPU
|
||||
# define __KERNEL_GPU_RAYTRACING__
|
||||
#endif
|
||||
|
||||
#define CCL_NAMESPACE_BEGIN
|
||||
#define CCL_NAMESPACE_END
|
||||
|
||||
#include <cstdint>
|
||||
#include <math.h>
|
||||
|
||||
#ifndef __NODES_MAX_GROUP__
|
||||
# define __NODES_MAX_GROUP__ NODE_GROUP_LEVEL_MAX
|
||||
#endif
|
||||
#ifndef __NODES_FEATURES__
|
||||
# define __NODES_FEATURES__ NODE_FEATURE_ALL
|
||||
#endif
|
||||
|
||||
/* This one does not have an abstraction.
|
||||
* It's used by other devices directly.
|
||||
*/
|
||||
|
||||
#define __device__
|
||||
|
||||
/* Qualifier wrappers for different names on different devices */
|
||||
|
||||
#define ccl_device inline
|
||||
#define ccl_device_extern extern "C"
|
||||
#define ccl_global
|
||||
#define ccl_always_inline __attribute__((always_inline))
|
||||
#define ccl_device_inline __attribute__((always_inline))
|
||||
#define ccl_noinline __attribute__((noinline))
|
||||
#define ccl_inline_constant const constexpr
|
||||
#define ccl_device_constant static constexpr
|
||||
#define ccl_static_constexpr static constexpr
|
||||
#define ccl_device_forceinline __attribute__((always_inline))
|
||||
#define ccl_device_noinline __attribute__((noinline))
|
||||
#define ccl_device_noinline_cpu ccl_device
|
||||
#define ccl_device_inline_method ccl_device
|
||||
#define ccl_device_template_spec template<> ccl_device_inline
|
||||
#define ccl_restrict __restrict__
|
||||
#define ccl_optional_struct_init
|
||||
#define ccl_private
|
||||
#define ccl_ray_data ccl_private
|
||||
#define ccl_gpu_shared
|
||||
#define ATTR_FALLTHROUGH __attribute__((fallthrough))
|
||||
#define ccl_constant const
|
||||
#define ccl_try_align(...) __attribute__((aligned(__VA_ARGS__)))
|
||||
#define ccl_align(n) __attribute__((aligned(n)))
|
||||
#define kernel_assert(cond)
|
||||
#define ccl_may_alias
|
||||
#define ccl_attr_maybe_unused [[maybe_unused]]
|
||||
|
||||
/* clang-format off */
|
||||
|
||||
/* kernel.h adapters */
|
||||
#define ccl_gpu_kernel(block_num_threads, thread_num_registers)
|
||||
#define ccl_gpu_kernel_threads(block_num_threads)
|
||||
|
||||
# define __ccl_gpu_kernel_signature(name, ...) \
|
||||
void oneapi_kernel_##name(KernelGlobalsGPU *ccl_restrict kg, \
|
||||
size_t kernel_global_size, \
|
||||
size_t kernel_local_size, \
|
||||
sycl::handler &cgh, \
|
||||
__VA_ARGS__) { \
|
||||
(void)(kg); \
|
||||
cgh.parallel_for( \
|
||||
sycl::nd_range<1>(kernel_global_size, kernel_local_size), \
|
||||
[=](sycl::nd_item<1> item) {
|
||||
|
||||
# define ccl_gpu_kernel_signature __ccl_gpu_kernel_signature
|
||||
|
||||
# define ccl_gpu_kernel_postfix \
|
||||
}); \
|
||||
}
|
||||
|
||||
#define ccl_gpu_kernel_call(x) ((ONEAPIKernelContext*)kg)->x
|
||||
#define ccl_gpu_kernel_within_bounds(i, n) ((i) < (n))
|
||||
|
||||
#define ccl_gpu_kernel_lambda(func, ...) \
|
||||
struct KernelLambda \
|
||||
{ \
|
||||
KernelLambda(const ONEAPIKernelContext *_kg) : kg(_kg) {} \
|
||||
ccl_private const ONEAPIKernelContext *kg; \
|
||||
__VA_ARGS__; \
|
||||
int operator()(const int state) const { return (func); } \
|
||||
} ccl_gpu_kernel_lambda_pass((ONEAPIKernelContext *)kg)
|
||||
|
||||
/* GPU thread, block, grid size and index */
|
||||
|
||||
# define ccl_gpu_thread_idx_x (sycl::ext::oneapi::this_work_item::get_nd_item<1>().get_local_id(0))
|
||||
# define ccl_gpu_block_dim_x (sycl::ext::oneapi::this_work_item::get_nd_item<1>().get_local_range(0))
|
||||
# define ccl_gpu_block_idx_x (sycl::ext::oneapi::this_work_item::get_nd_item<1>().get_group(0))
|
||||
# define ccl_gpu_grid_dim_x (sycl::ext::oneapi::this_work_item::get_nd_item<1>().get_group_range(0))
|
||||
# define ccl_gpu_warp_size (sycl::ext::oneapi::this_work_item::get_sub_group().get_local_range()[0])
|
||||
# define ccl_gpu_thread_mask(thread_warp) uint(0xFFFFFFFF >> (ccl_gpu_warp_size - thread_warp))
|
||||
|
||||
# define ccl_gpu_global_id_x() (sycl::ext::oneapi::this_work_item::get_nd_item<1>().get_global_id(0))
|
||||
# define ccl_gpu_global_size_x() (sycl::ext::oneapi::this_work_item::get_nd_item<1>().get_global_range(0))
|
||||
|
||||
/* GPU warp synchronization */
|
||||
# define ccl_gpu_syncthreads() sycl::ext::oneapi::this_work_item::get_nd_item<1>().barrier()
|
||||
# define ccl_gpu_local_syncthreads() sycl::ext::oneapi::this_work_item::get_nd_item<1>().barrier(sycl::access::fence_space::local_space)
|
||||
|
||||
/* A ballot in SYCL is only available as an Intel extension and its DPC++ v6.3 implementation
|
||||
* does not support devices with sub-group sizes above 64. Summing values (of any type) within
|
||||
* sub-groups can be achieved with the SYCL core feature inclusive_scan_over_group, which has
|
||||
* better support on non-Intel devices. */
|
||||
# define ccl_gpu_ballot(predicate) 0; static_assert(false, "Use sycl::inclusive_scan_over_group on oneAPI device instead of ccl_gpu_ballot")
|
||||
|
||||
/* Debug defines */
|
||||
#if defined(__SYCL_DEVICE_ONLY__)
|
||||
# define CCL_ONEAPI_CONSTANT __attribute__((opencl_constant))
|
||||
#else
|
||||
# define CCL_ONEAPI_CONSTANT
|
||||
#endif
|
||||
|
||||
#define sycl_printf(format, ...) { \
|
||||
static const CCL_ONEAPI_CONSTANT char fmt[] = format; \
|
||||
sycl::ext::oneapi::experimental::printf(fmt, __VA_ARGS__ ); \
|
||||
}
|
||||
|
||||
#define sycl_printf_(format) { \
|
||||
static const CCL_ONEAPI_CONSTANT char fmt[] = format; \
|
||||
sycl::ext::oneapi::experimental::printf(fmt); \
|
||||
}
|
||||
|
||||
/* GPU texture objects */
|
||||
|
||||
/* clang-format on */
|
||||
|
||||
/* Types */
|
||||
|
||||
/* It's not possible to use sycl types like sycl::float3, sycl::int3, etc
|
||||
* because these types have different interfaces from blender version. */
|
||||
|
||||
using uchar = unsigned char;
|
||||
using sycl::half;
|
||||
|
||||
/* math functions */
|
||||
ccl_device_forceinline float __uint_as_float(unsigned int x)
|
||||
{
|
||||
return sycl::bit_cast<float>(x);
|
||||
}
|
||||
ccl_device_forceinline unsigned int __float_as_uint(const float x)
|
||||
{
|
||||
return sycl::bit_cast<unsigned int>(x);
|
||||
}
|
||||
ccl_device_forceinline float __int_as_float(const int x)
|
||||
{
|
||||
return sycl::bit_cast<float>(x);
|
||||
}
|
||||
ccl_device_forceinline int __float_as_int(const float x)
|
||||
{
|
||||
return sycl::bit_cast<int>(x);
|
||||
}
|
||||
|
||||
#define fabsf(x) sycl::fabs((x))
|
||||
#define copysignf(x, y) sycl::copysign((x), (y))
|
||||
#define asinf(x) sycl::asin((x))
|
||||
#define acosf(x) sycl::acos((x))
|
||||
#define atanf(x) sycl::atan((x))
|
||||
#define floorf(x) sycl::floor((x))
|
||||
#define ceilf(x) sycl::ceil((x))
|
||||
#define roundf(x) sycl::round((x))
|
||||
#define sinhf(x) sycl::sinh((x))
|
||||
#define coshf(x) sycl::cosh((x))
|
||||
#define tanhf(x) sycl::tanh((x))
|
||||
#define hypotf(x, y) sycl::hypot((x), (y))
|
||||
#define atan2f(x, y) sycl::atan2((x), (y))
|
||||
#define fmaxf(x, y) sycl::fmax((x), (y))
|
||||
#define fminf(x, y) sycl::fmin((x), (y))
|
||||
#define fmodf(x, y) sycl::fmod((x), (y))
|
||||
#define lgammaf(x) sycl::lgamma((x))
|
||||
#define ldexpf(x, y) sycl::ldexp((x), (y))
|
||||
|
||||
#define cosf(x) sycl::native::cos(((float)(x)))
|
||||
#define sinf(x) sycl::native::sin(((float)(x)))
|
||||
#define powf(x, y) sycl::native::powr(((float)(x)), ((float)(y)))
|
||||
#define tanf(x) sycl::native::tan(((float)(x)))
|
||||
#define logf(x) sycl::native::log(((float)(x)))
|
||||
#define expf(x) sycl::native::exp(((float)(x)))
|
||||
#define sqrtf(x) sycl::native::sqrt(((float)(x)))
|
||||
|
||||
#define __forceinline __attribute__((always_inline))
|
||||
|
||||
/* Types */
|
||||
#include "util/half.h"
|
||||
#include "util/types.h"
|
||||
|
||||
static_assert(
|
||||
sizeof(sycl::ext::oneapi::experimental::sampled_image_handle::raw_image_handle_type) ==
|
||||
sizeof(uint64_t));
|
||||
typedef uint64_t ccl_gpu_image_object_2D;
|
||||
typedef uint64_t ccl_gpu_image_object_3D;
|
||||
|
||||
template<typename T>
|
||||
ccl_device_forceinline T ccl_gpu_image_object_read_2D(const ccl_gpu_image_object_2D texobj,
|
||||
const float x,
|
||||
const float y)
|
||||
{
|
||||
/* Generic implementation not possible due to limitation with SYCL bindless sampled images
|
||||
* not being able to read in a format, which is different from the supported data type of
|
||||
* the texture.
|
||||
* But looks it looks like this is not a problem at the moment. */
|
||||
static_assert(false);
|
||||
return T();
|
||||
}
|
||||
|
||||
template<>
|
||||
ccl_device_forceinline float ccl_gpu_image_object_read_2D<float>(
|
||||
const ccl_gpu_image_object_2D texobj, const float x, const float y)
|
||||
{
|
||||
sycl::ext::oneapi::experimental::sampled_image_handle image(
|
||||
(sycl::ext::oneapi::experimental::sampled_image_handle::raw_image_handle_type)texobj);
|
||||
return sycl::ext::oneapi::experimental::sample_image<float>(image, sycl::float2{x, y});
|
||||
}
|
||||
|
||||
template<>
|
||||
ccl_device_forceinline float4 ccl_gpu_image_object_read_2D<float4>(
|
||||
const ccl_gpu_image_object_2D texobj, const float x, const float y)
|
||||
{
|
||||
sycl::ext::oneapi::experimental::sampled_image_handle image(
|
||||
(sycl::ext::oneapi::experimental::sampled_image_handle::raw_image_handle_type)texobj);
|
||||
return sycl::ext::oneapi::experimental::sample_image<float4, sycl::vec<float, 4>>(
|
||||
image, sycl::float2{x, y});
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
ccl_device_forceinline T ccl_gpu_image_object_read_3D(const ccl_gpu_image_object_3D texobj,
|
||||
const float x,
|
||||
const float y,
|
||||
const float z)
|
||||
{
|
||||
/* A generic implementation is not possible due to limitations with SYCL bindless sampled images
|
||||
* not being able to read in a format that is different from the supported data type of
|
||||
* the texture.
|
||||
* However, it looks like this is not a problem at the moment, but I am leaving a static
|
||||
* assert in order to easily detect if it becomes a problem in the future. */
|
||||
static_assert(false);
|
||||
return T();
|
||||
}
|
||||
|
||||
template<>
|
||||
ccl_device_forceinline float ccl_gpu_image_object_read_3D<float>(
|
||||
const ccl_gpu_image_object_3D texobj, const float x, const float y, const float z)
|
||||
{
|
||||
sycl::ext::oneapi::experimental::sampled_image_handle image(
|
||||
(sycl::ext::oneapi::experimental::sampled_image_handle::raw_image_handle_type)texobj);
|
||||
return sycl::ext::oneapi::experimental::sample_image<float>(image, sycl::float3{x, y, z});
|
||||
}
|
||||
|
||||
template<>
|
||||
ccl_device_forceinline float4 ccl_gpu_image_object_read_3D<float4>(
|
||||
const ccl_gpu_image_object_3D texobj, const float x, const float y, const float z)
|
||||
{
|
||||
sycl::ext::oneapi::experimental::sampled_image_handle image(
|
||||
(sycl::ext::oneapi::experimental::sampled_image_handle::raw_image_handle_type)texobj);
|
||||
return sycl::ext::oneapi::experimental::sample_image<float4, sycl::vec<float, 4>>(
|
||||
image, sycl::float3{x, y, z});
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
/* SPDX-FileCopyrightText: 2021-2022 Intel Corporation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
|
||||
#include "kernel/util/nanovdb.h"
|
||||
|
||||
/* clang-format off */
|
||||
struct ONEAPIKernelContext : public KernelGlobalsGPU {
|
||||
public:
|
||||
# include "kernel/device/gpu/image.h"
|
||||
/* clang-format on */
|
||||
@@ -0,0 +1,8 @@
|
||||
/* SPDX-FileCopyrightText: 2021-2022 Intel Corporation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
}
|
||||
; /* end of ONEAPIKernelContext class definition */
|
||||
|
||||
#undef kernel_integrator_state
|
||||
#define kernel_integrator_state (*(kg->integrator_state))
|
||||
@@ -0,0 +1,19 @@
|
||||
/* SPDX-FileCopyrightText: 2023 Intel Corporation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
|
||||
#if defined(WITH_EMBREE_GPU)
|
||||
# undef ccl_gpu_kernel_signature
|
||||
# define ccl_gpu_kernel_signature(name, ...) \
|
||||
void oneapi_kernel_##name(KernelGlobalsGPU *ccl_restrict kg, \
|
||||
size_t kernel_global_size, \
|
||||
size_t kernel_local_size, \
|
||||
sycl::handler &cgh, \
|
||||
__VA_ARGS__) \
|
||||
{ \
|
||||
(void)(kg); \
|
||||
cgh.parallel_for( \
|
||||
sycl::nd_range<1>(kernel_global_size, kernel_local_size), \
|
||||
[=](sycl::nd_item<1> item, sycl::kernel_handler oneapi_kernel_handler) { \
|
||||
((ONEAPIKernelContext*)kg)->kernel_handler = oneapi_kernel_handler;
|
||||
#endif
|
||||
@@ -0,0 +1,8 @@
|
||||
/* SPDX-FileCopyrightText: 2023 Intel Corporation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
|
||||
#if defined(WITH_EMBREE_GPU)
|
||||
# undef ccl_gpu_kernel_signature
|
||||
# define ccl_gpu_kernel_signature __ccl_gpu_kernel_signature
|
||||
#endif
|
||||
47
blender-5.2.0/intern/cycles/kernel/device/oneapi/globals.h
Normal file
47
blender-5.2.0/intern/cycles/kernel/device/oneapi/globals.h
Normal file
@@ -0,0 +1,47 @@
|
||||
/* SPDX-FileCopyrightText: 2021-2022 Intel Corporation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
|
||||
/* Constant Globals */
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "kernel/types.h"
|
||||
|
||||
#include "kernel/integrator/state.h"
|
||||
#include "kernel/util/profiler.h"
|
||||
|
||||
#include "util/color.h"
|
||||
#include "util/types_image.h"
|
||||
|
||||
CCL_NAMESPACE_BEGIN
|
||||
|
||||
/* NOTE(@nsirgien): With SYCL we can't declare __constant__ global variable, which will be
|
||||
* accessible from device code, like it has been done for Cycles CUDA backend. So, the backend will
|
||||
* allocate this "constant" memory regions and store pointers to them in oneAPI context class */
|
||||
|
||||
struct IntegratorStateGPU;
|
||||
struct IntegratorQueueCounter;
|
||||
|
||||
struct KernelGlobalsGPU {
|
||||
|
||||
#define KERNEL_DATA_ARRAY(type, name) const type *__##name = nullptr;
|
||||
#define KERNEL_DATA_ARRAY_WRITABLE(type, name) type *__##name = nullptr;
|
||||
#include "kernel/data_arrays.h"
|
||||
IntegratorStateGPU *integrator_state;
|
||||
const KernelData *__data;
|
||||
sycl::kernel_handler kernel_handler;
|
||||
};
|
||||
|
||||
using KernelGlobals = ccl_global KernelGlobalsGPU *ccl_restrict;
|
||||
|
||||
#define kernel_data (*(__data))
|
||||
#define kernel_integrator_state (*(integrator_state))
|
||||
|
||||
/* data lookup defines */
|
||||
|
||||
#define kernel_data_fetch(name, index) __##name[(index)]
|
||||
#define kernel_data_write(name, index, value) __##name[(index)] = (value)
|
||||
#define kernel_data_array(name) __##name
|
||||
|
||||
CCL_NAMESPACE_END
|
||||
758
blender-5.2.0/intern/cycles/kernel/device/oneapi/kernel.cpp
Normal file
758
blender-5.2.0/intern/cycles/kernel/device/oneapi/kernel.cpp
Normal file
@@ -0,0 +1,758 @@
|
||||
/* SPDX-FileCopyrightText: 2021-2022 Intel Corporation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
|
||||
#ifdef WITH_ONEAPI
|
||||
|
||||
# include "kernel.h"
|
||||
# include <iostream>
|
||||
# include <map>
|
||||
# include <set>
|
||||
|
||||
/* <algorithm> is needed until included upstream in sycl/detail/property_list_base.hpp */
|
||||
# include <algorithm>
|
||||
# include <sycl/sycl.hpp>
|
||||
|
||||
# include "kernel/device/oneapi/compat.h"
|
||||
# include "kernel/device/oneapi/globals.h"
|
||||
# include "kernel/device/oneapi/kernel_templates.h"
|
||||
|
||||
# include "kernel/device/gpu/kernel.h"
|
||||
|
||||
# include "device/kernel.cpp"
|
||||
|
||||
static OneAPIErrorCallback s_error_cb = nullptr;
|
||||
static void *s_error_user_ptr = nullptr;
|
||||
|
||||
# ifdef WITH_EMBREE_GPU
|
||||
static RTCFeatureFlags oneapi_embree_features_from_kernel_features(const uint kernel_features)
|
||||
{
|
||||
unsigned int feature_flags = RTC_FEATURE_FLAG_TRIANGLE | RTC_FEATURE_FLAG_INSTANCE |
|
||||
RTC_FEATURE_FLAG_FILTER_FUNCTION_IN_ARGUMENTS;
|
||||
|
||||
if (kernel_features & KERNEL_FEATURE_HAIR_THICK) {
|
||||
feature_flags |= RTC_FEATURE_FLAG_ROUND_CATMULL_ROM_CURVE |
|
||||
RTC_FEATURE_FLAG_ROUND_LINEAR_CURVE;
|
||||
}
|
||||
if (kernel_features & KERNEL_FEATURE_HAIR) {
|
||||
feature_flags |= RTC_FEATURE_FLAG_FLAT_CATMULL_ROM_CURVE;
|
||||
}
|
||||
if (kernel_features & KERNEL_FEATURE_POINTCLOUD) {
|
||||
feature_flags |= RTC_FEATURE_FLAG_POINT;
|
||||
}
|
||||
if (kernel_features & KERNEL_FEATURE_OBJECT_MOTION) {
|
||||
feature_flags |= RTC_FEATURE_FLAG_MOTION_BLUR;
|
||||
}
|
||||
|
||||
return (RTCFeatureFlags)feature_flags;
|
||||
}
|
||||
# endif
|
||||
|
||||
void oneapi_set_error_cb(OneAPIErrorCallback cb, void *user_ptr)
|
||||
{
|
||||
s_error_cb = cb;
|
||||
s_error_user_ptr = user_ptr;
|
||||
}
|
||||
|
||||
size_t oneapi_suggested_gpu_kernel_size(const DeviceKernel kernel)
|
||||
{
|
||||
/* This defines are available only to the device code, so making this function
|
||||
* seems to be the most reasonable way to provide access to them for the host code. */
|
||||
switch (kernel) {
|
||||
case DEVICE_KERNEL_INTEGRATOR_QUEUED_PATHS_ARRAY:
|
||||
case DEVICE_KERNEL_INTEGRATOR_QUEUED_SHADOW_PATHS_ARRAY:
|
||||
case DEVICE_KERNEL_INTEGRATOR_ACTIVE_PATHS_ARRAY:
|
||||
case DEVICE_KERNEL_INTEGRATOR_TERMINATED_PATHS_ARRAY:
|
||||
case DEVICE_KERNEL_INTEGRATOR_TERMINATED_SHADOW_PATHS_ARRAY:
|
||||
case DEVICE_KERNEL_INTEGRATOR_COMPACT_PATHS_ARRAY:
|
||||
case DEVICE_KERNEL_INTEGRATOR_COMPACT_SHADOW_PATHS_ARRAY:
|
||||
return GPU_PARALLEL_ACTIVE_INDEX_DEFAULT_BLOCK_SIZE;
|
||||
|
||||
case DEVICE_KERNEL_INTEGRATOR_SORTED_PATHS_ARRAY:
|
||||
case DEVICE_KERNEL_INTEGRATOR_COMPACT_STATES:
|
||||
case DEVICE_KERNEL_INTEGRATOR_COMPACT_SHADOW_STATES:
|
||||
return GPU_PARALLEL_SORTED_INDEX_DEFAULT_BLOCK_SIZE;
|
||||
|
||||
case DEVICE_KERNEL_INTEGRATOR_SORT_BUCKET_PASS:
|
||||
case DEVICE_KERNEL_INTEGRATOR_SORT_WRITE_PASS:
|
||||
return GPU_PARALLEL_SORT_BLOCK_SIZE;
|
||||
|
||||
case DEVICE_KERNEL_PREFIX_SUM:
|
||||
return GPU_PARALLEL_PREFIX_SUM_DEFAULT_BLOCK_SIZE;
|
||||
|
||||
default:
|
||||
return (size_t)0;
|
||||
}
|
||||
}
|
||||
|
||||
/* NOTE(@nsirgien): Execution of this simple kernel will check basic functionality like
|
||||
* memory allocations, memory transfers and execution of kernel with USM memory. */
|
||||
bool oneapi_run_test_kernel(SyclQueue *queue_)
|
||||
{
|
||||
assert(queue_);
|
||||
sycl::queue *queue = reinterpret_cast<sycl::queue *>(queue_);
|
||||
const size_t N = 8;
|
||||
const size_t memory_byte_size = sizeof(int) * N;
|
||||
|
||||
bool is_computation_correct = true;
|
||||
try {
|
||||
int *A_host = (int *)sycl::aligned_alloc_host(16, memory_byte_size, *queue);
|
||||
|
||||
for (size_t i = (size_t)0; i < N; i++) {
|
||||
A_host[i] = rand() % 32;
|
||||
}
|
||||
|
||||
int *A_device = (int *)sycl::malloc_device(memory_byte_size, *queue);
|
||||
int *B_device = (int *)sycl::malloc_device(memory_byte_size, *queue);
|
||||
|
||||
queue->memcpy(A_device, A_host, memory_byte_size);
|
||||
queue->wait_and_throw();
|
||||
|
||||
queue->submit([&](sycl::handler &cgh) {
|
||||
cgh.parallel_for(N, [=](sycl::id<1> idx) { B_device[idx] = A_device[idx] + idx.get(0); });
|
||||
});
|
||||
queue->wait_and_throw();
|
||||
|
||||
int *B_host = (int *)sycl::aligned_alloc_host(16, memory_byte_size, *queue);
|
||||
|
||||
queue->memcpy(B_host, B_device, memory_byte_size);
|
||||
queue->wait_and_throw();
|
||||
|
||||
for (size_t i = (size_t)0; i < N; i++) {
|
||||
const int expected_result = i + A_host[i];
|
||||
if (B_host[i] != expected_result) {
|
||||
is_computation_correct = false;
|
||||
if (s_error_cb) {
|
||||
s_error_cb(("Incorrect result in test kernel execution - expected " +
|
||||
std::to_string(expected_result) + ", got " + std::to_string(B_host[i]))
|
||||
.c_str(),
|
||||
s_error_user_ptr);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
sycl::free(A_host, *queue);
|
||||
sycl::free(B_host, *queue);
|
||||
sycl::free(A_device, *queue);
|
||||
sycl::free(B_device, *queue);
|
||||
queue->wait_and_throw();
|
||||
}
|
||||
catch (const sycl::exception &e) {
|
||||
if (s_error_cb) {
|
||||
s_error_cb(e.what(), s_error_user_ptr);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
return is_computation_correct;
|
||||
}
|
||||
|
||||
bool oneapi_zero_memory_on_device(SyclQueue *queue_, void *device_pointer, const size_t num_bytes)
|
||||
{
|
||||
assert(queue_);
|
||||
sycl::queue *queue = reinterpret_cast<sycl::queue *>(queue_);
|
||||
try {
|
||||
queue->memset(device_pointer, 0, num_bytes);
|
||||
queue->wait_and_throw();
|
||||
return true;
|
||||
}
|
||||
catch (const sycl::exception &e) {
|
||||
if (s_error_cb) {
|
||||
s_error_cb(e.what(), s_error_user_ptr);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool oneapi_kernel_is_required_for_features(const std::string &kernel_name,
|
||||
const uint kernel_features)
|
||||
{
|
||||
/* Skip all non-Cycles kernels */
|
||||
if (kernel_name.find("oneapi_kernel_") == std::string::npos) {
|
||||
return false;
|
||||
}
|
||||
|
||||
if ((kernel_features & KERNEL_FEATURE_NODE_RAYTRACE) == 0 &&
|
||||
kernel_name.find(device_kernel_as_string(DEVICE_KERNEL_INTEGRATOR_SHADE_SURFACE_RAYTRACE)) !=
|
||||
std::string::npos)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
if ((kernel_features & KERNEL_FEATURE_MNEE) == 0 &&
|
||||
kernel_name.find(device_kernel_as_string(DEVICE_KERNEL_INTEGRATOR_INTERSECT_MNEE)) !=
|
||||
std::string::npos)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
if ((kernel_features & KERNEL_FEATURE_VOLUME) == 0 &&
|
||||
kernel_name.find(device_kernel_as_string(DEVICE_KERNEL_INTEGRATOR_INTERSECT_VOLUME_STACK)) !=
|
||||
std::string::npos)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
if (((kernel_features & (KERNEL_FEATURE_PATH_TRACING | KERNEL_FEATURE_BAKING)) == 0) &&
|
||||
((kernel_name.find(device_kernel_as_string(DEVICE_KERNEL_INTEGRATOR_INTERSECT_CLOSEST)) !=
|
||||
std::string::npos) ||
|
||||
(kernel_name.find(device_kernel_as_string(DEVICE_KERNEL_INTEGRATOR_INTERSECT_SHADOW)) !=
|
||||
std::string::npos) ||
|
||||
(kernel_name.find(device_kernel_as_string(DEVICE_KERNEL_INTEGRATOR_INTERSECT_SUBSURFACE)) !=
|
||||
std::string::npos) ||
|
||||
(kernel_name.find(device_kernel_as_string(DEVICE_KERNEL_INTEGRATOR_INTERSECT_MNEE)) !=
|
||||
std::string::npos) ||
|
||||
(kernel_name.find(device_kernel_as_string(
|
||||
DEVICE_KERNEL_INTEGRATOR_INTERSECT_DEDICATED_LIGHT)) != std::string::npos)))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
bool oneapi_kernel_is_compatible_with_hardware_raytracing(const std::string &kernel_name)
|
||||
{
|
||||
/* MNEE and Ray-trace kernels work correctly with Hardware Ray-tracing starting with Embree 4.1.
|
||||
*/
|
||||
# if defined(RTC_VERSION) && RTC_VERSION < 40100
|
||||
return (kernel_name.find(device_kernel_as_string(DEVICE_KERNEL_INTEGRATOR_INTERSECT_MNEE)) ==
|
||||
std::string::npos) &&
|
||||
(kernel_name.find(device_kernel_as_string(
|
||||
DEVICE_KERNEL_INTEGRATOR_SHADE_SURFACE_RAYTRACE)) == std::string::npos);
|
||||
# else
|
||||
return true;
|
||||
# endif
|
||||
}
|
||||
|
||||
bool oneapi_kernel_has_intersections(const std::string &kernel_name)
|
||||
{
|
||||
for (int i = 0; i < (int)DEVICE_KERNEL_NUM; i++) {
|
||||
DeviceKernel kernel = (DeviceKernel)i;
|
||||
if (device_kernel_has_intersection(kernel)) {
|
||||
if (kernel_name.find(device_kernel_as_string(kernel)) != std::string::npos) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool oneapi_load_kernels(SyclQueue *queue_,
|
||||
const uint kernel_features,
|
||||
bool use_hardware_raytracing)
|
||||
{
|
||||
assert(queue_);
|
||||
sycl::queue *queue = reinterpret_cast<sycl::queue *>(queue_);
|
||||
|
||||
# ifdef WITH_EMBREE_GPU
|
||||
/* For best performance, we always JIT compile the kernels that are using Embree. */
|
||||
if (use_hardware_raytracing) {
|
||||
try {
|
||||
sycl::kernel_bundle<sycl::bundle_state::input> all_kernels_bundle =
|
||||
sycl::get_kernel_bundle<sycl::bundle_state::input>(queue->get_context(),
|
||||
{queue->get_device()});
|
||||
|
||||
for (const sycl::kernel_id &kernel_id : all_kernels_bundle.get_kernel_ids()) {
|
||||
const std::string &kernel_name = kernel_id.get_name();
|
||||
|
||||
if (!oneapi_kernel_is_required_for_features(kernel_name, kernel_features) ||
|
||||
!(oneapi_kernel_has_intersections(kernel_name) &&
|
||||
oneapi_kernel_is_compatible_with_hardware_raytracing(kernel_name)))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
sycl::kernel_bundle<sycl::bundle_state::input> one_kernel_bundle_input =
|
||||
sycl::get_kernel_bundle<sycl::bundle_state::input>(
|
||||
queue->get_context(), {queue->get_device()}, {kernel_id});
|
||||
|
||||
const RTCFeatureFlags embree_features = oneapi_embree_features_from_kernel_features(
|
||||
kernel_features);
|
||||
one_kernel_bundle_input
|
||||
.set_specialization_constant<ONEAPIKernelContext::oneapi_embree_features>(
|
||||
embree_features);
|
||||
sycl::build(one_kernel_bundle_input);
|
||||
}
|
||||
}
|
||||
catch (const sycl::exception &e) {
|
||||
if (s_error_cb) {
|
||||
s_error_cb(e.what(), s_error_user_ptr);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
# endif
|
||||
|
||||
try {
|
||||
sycl::kernel_bundle<sycl::bundle_state::input> all_kernels_bundle =
|
||||
sycl::get_kernel_bundle<sycl::bundle_state::input>(queue->get_context(),
|
||||
{queue->get_device()});
|
||||
|
||||
for (const sycl::kernel_id &kernel_id : all_kernels_bundle.get_kernel_ids()) {
|
||||
const std::string &kernel_name = kernel_id.get_name();
|
||||
|
||||
/* In case HWRT is on, compilation of kernels using Embree is already handled in previous
|
||||
* block. */
|
||||
if (!oneapi_kernel_is_required_for_features(kernel_name, kernel_features) ||
|
||||
(use_hardware_raytracing && oneapi_kernel_has_intersections(kernel_name) &&
|
||||
oneapi_kernel_is_compatible_with_hardware_raytracing(kernel_name)))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
# ifdef WITH_EMBREE_GPU
|
||||
if (oneapi_kernel_has_intersections(kernel_name)) {
|
||||
sycl::kernel_bundle<sycl::bundle_state::input> one_kernel_bundle_input =
|
||||
sycl::get_kernel_bundle<sycl::bundle_state::input>(
|
||||
queue->get_context(), {queue->get_device()}, {kernel_id});
|
||||
one_kernel_bundle_input
|
||||
.set_specialization_constant<ONEAPIKernelContext::oneapi_embree_features>(
|
||||
RTC_FEATURE_FLAG_NONE);
|
||||
sycl::build(one_kernel_bundle_input);
|
||||
continue;
|
||||
}
|
||||
# endif
|
||||
/* This call will ensure that AoT or cached JIT binaries are available
|
||||
* for execution. It will trigger compilation if it is not already the case. */
|
||||
(void)sycl::get_kernel_bundle<sycl::bundle_state::executable>(
|
||||
queue->get_context(), {queue->get_device()}, {kernel_id});
|
||||
}
|
||||
}
|
||||
catch (const sycl::exception &e) {
|
||||
if (s_error_cb) {
|
||||
s_error_cb(e.what(), s_error_user_ptr);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool oneapi_enqueue_kernel(KernelContext *kernel_context,
|
||||
const int kernel,
|
||||
const size_t global_size,
|
||||
const size_t local_size,
|
||||
const uint kernel_features,
|
||||
bool use_hardware_raytracing,
|
||||
void **args)
|
||||
{
|
||||
bool success = true;
|
||||
::DeviceKernel device_kernel = (::DeviceKernel)kernel;
|
||||
KernelGlobalsGPU *kg = (KernelGlobalsGPU *)kernel_context->kernel_globals;
|
||||
sycl::queue *queue = reinterpret_cast<sycl::queue *>(kernel_context->queue);
|
||||
assert(queue);
|
||||
if (!queue) {
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Let the compiler throw an error if there are any kernels missing in this implementation. */
|
||||
# if defined(_WIN32)
|
||||
# pragma warning(error : 4062)
|
||||
# elif defined(__GNUC__)
|
||||
# pragma GCC diagnostic push
|
||||
# pragma GCC diagnostic error "-Wswitch"
|
||||
# endif
|
||||
|
||||
int max_shaders = 0;
|
||||
|
||||
if (device_kernel == DEVICE_KERNEL_INTEGRATOR_SORT_BUCKET_PASS ||
|
||||
device_kernel == DEVICE_KERNEL_INTEGRATOR_SORT_WRITE_PASS)
|
||||
{
|
||||
max_shaders = (kernel_context->scene_max_shaders);
|
||||
}
|
||||
|
||||
try {
|
||||
queue->submit([&](sycl::handler &cgh) {
|
||||
# ifdef WITH_EMBREE_GPU
|
||||
/* Spec says it has no effect if the called kernel doesn't support the below specialization
|
||||
* constant but it can still trigger a recompilation, so we set it only if needed. */
|
||||
if (device_kernel_has_intersection(device_kernel)) {
|
||||
const RTCFeatureFlags embree_features = use_hardware_raytracing ?
|
||||
oneapi_embree_features_from_kernel_features(
|
||||
kernel_features) :
|
||||
RTC_FEATURE_FLAG_NONE;
|
||||
cgh.set_specialization_constant<ONEAPIKernelContext::oneapi_embree_features>(
|
||||
embree_features);
|
||||
}
|
||||
# else
|
||||
(void)kernel_features;
|
||||
# endif
|
||||
switch (device_kernel) {
|
||||
case DEVICE_KERNEL_INTEGRATOR_RESET: {
|
||||
oneapi_call(kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_reset);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_INIT_FROM_CAMERA: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_init_from_camera);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_INIT_FROM_BAKE: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_init_from_bake);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_INTERSECT_CLOSEST: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_intersect_closest);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_INTERSECT_SHADOW: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_intersect_shadow);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_INTERSECT_SUBSURFACE: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_intersect_subsurface);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_INTERSECT_VOLUME_STACK: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_intersect_volume_stack);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_INTERSECT_DEDICATED_LIGHT: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_intersect_dedicated_light);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_INTERSECT_MNEE: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_intersect_mnee);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_SHADE_BACKGROUND: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_shade_background);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_SHADE_LIGHT_NEE: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_shade_light_nee);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_SHADE_LIGHT_FORWARD: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_shade_light_forward);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_SHADE_SHADOW: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_shade_shadow);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_SHADE_SURFACE: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_shade_surface);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_SHADE_SURFACE_RAYTRACE: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_shade_surface_raytrace);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_SHADE_VOLUME: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_shade_volume);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_SHADE_VOLUME_RAY_MARCHING: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_shade_volume_ray_marching);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_SHADE_DEDICATED_LIGHT: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_shade_dedicated_light);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_QUEUED_PATHS_ARRAY: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_queued_paths_array);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_QUEUED_SHADOW_PATHS_ARRAY: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_queued_shadow_paths_array);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_ACTIVE_PATHS_ARRAY: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_active_paths_array);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_TERMINATED_PATHS_ARRAY: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_terminated_paths_array);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_TERMINATED_SHADOW_PATHS_ARRAY: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_terminated_shadow_paths_array);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_SORTED_PATHS_ARRAY: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_sorted_paths_array);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_SORT_BUCKET_PASS: {
|
||||
sycl::local_accessor<int> local_mem(max_shaders, cgh);
|
||||
oneapi_kernel_integrator_sort_bucket_pass(kg,
|
||||
global_size,
|
||||
local_size,
|
||||
cgh,
|
||||
*(int *)(args[0]),
|
||||
*(int *)(args[1]),
|
||||
*(int *)(args[2]),
|
||||
*(int **)(args[3]),
|
||||
*(int *)(args[4]),
|
||||
local_mem);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_SORT_WRITE_PASS: {
|
||||
sycl::local_accessor<int> local_mem(max_shaders, cgh);
|
||||
oneapi_kernel_integrator_sort_write_pass(kg,
|
||||
global_size,
|
||||
local_size,
|
||||
cgh,
|
||||
*(int *)(args[0]),
|
||||
*(int *)(args[1]),
|
||||
*(int *)(args[2]),
|
||||
*(int **)(args[3]),
|
||||
*(int *)(args[4]),
|
||||
local_mem);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_COMPACT_PATHS_ARRAY: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_compact_paths_array);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_COMPACT_SHADOW_PATHS_ARRAY: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_compact_shadow_paths_array);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_ADAPTIVE_SAMPLING_CONVERGENCE_CHECK: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_adaptive_sampling_convergence_check);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_ADAPTIVE_SAMPLING_CONVERGENCE_FILTER_X: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_adaptive_sampling_filter_x);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_ADAPTIVE_SAMPLING_CONVERGENCE_FILTER_Y: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_adaptive_sampling_filter_y);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_SHADER_EVAL_DISPLACE: {
|
||||
oneapi_call(kg, cgh, global_size, local_size, args, oneapi_kernel_shader_eval_displace);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_SHADER_EVAL_BACKGROUND: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_shader_eval_background);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_SHADER_EVAL_CURVE_SHADOW_TRANSPARENCY: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_shader_eval_curve_shadow_transparency);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_SHADER_EVAL_VOLUME_DENSITY: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_shader_eval_volume_density);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_PREFIX_SUM: {
|
||||
oneapi_call(kg, cgh, global_size, local_size, args, oneapi_kernel_prefix_sum);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_VOLUME_GUIDING_FILTER_X: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_volume_guiding_filter_x);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_VOLUME_GUIDING_FILTER_Y: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_volume_guiding_filter_y);
|
||||
break;
|
||||
}
|
||||
|
||||
/* clang-format off */
|
||||
# define DEVICE_KERNEL_FILM_CONVERT_PARTIAL(VARIANT, variant) \
|
||||
case DEVICE_KERNEL_FILM_CONVERT_##VARIANT: { \
|
||||
oneapi_call(kg, cgh, \
|
||||
global_size, \
|
||||
local_size, \
|
||||
args, \
|
||||
oneapi_kernel_film_convert_##variant); \
|
||||
break; \
|
||||
}
|
||||
|
||||
# define DEVICE_KERNEL_FILM_CONVERT(variant, VARIANT) \
|
||||
DEVICE_KERNEL_FILM_CONVERT_PARTIAL(VARIANT, variant) \
|
||||
DEVICE_KERNEL_FILM_CONVERT_PARTIAL(VARIANT##_HALF_RGBA, variant##_half_rgba)
|
||||
|
||||
DEVICE_KERNEL_FILM_CONVERT(depth, DEPTH);
|
||||
DEVICE_KERNEL_FILM_CONVERT(mist, MIST);
|
||||
DEVICE_KERNEL_FILM_CONVERT(volume_majorant, VOLUME_MAJORANT);
|
||||
DEVICE_KERNEL_FILM_CONVERT(sample_count, SAMPLE_COUNT);
|
||||
DEVICE_KERNEL_FILM_CONVERT(float, FLOAT);
|
||||
DEVICE_KERNEL_FILM_CONVERT(light_path, LIGHT_PATH);
|
||||
DEVICE_KERNEL_FILM_CONVERT(rgbe, RGBE);
|
||||
DEVICE_KERNEL_FILM_CONVERT(float3, FLOAT3);
|
||||
DEVICE_KERNEL_FILM_CONVERT(motion, MOTION);
|
||||
DEVICE_KERNEL_FILM_CONVERT(cryptomatte, CRYPTOMATTE);
|
||||
DEVICE_KERNEL_FILM_CONVERT(shadow_catcher, SHADOW_CATCHER);
|
||||
DEVICE_KERNEL_FILM_CONVERT(shadow_catcher_matte_with_shadow,
|
||||
SHADOW_CATCHER_MATTE_WITH_SHADOW);
|
||||
DEVICE_KERNEL_FILM_CONVERT(combined, COMBINED);
|
||||
DEVICE_KERNEL_FILM_CONVERT(float4, FLOAT4);
|
||||
|
||||
# undef DEVICE_KERNEL_FILM_CONVERT
|
||||
# undef DEVICE_KERNEL_FILM_CONVERT_PARTIAL
|
||||
/* clang-format on */
|
||||
|
||||
case DEVICE_KERNEL_FILTER_GUIDING_PREPROCESS: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_filter_guiding_preprocess);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_FILTER_GUIDING_SET_FAKE_ALBEDO: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_filter_guiding_set_fake_albedo);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_FILTER_COLOR_PREPROCESS: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_filter_color_preprocess);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_FILTER_COLOR_POSTPROCESS: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_filter_color_postprocess);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_FILTER_COLOR_FLIP_Y: {
|
||||
oneapi_call(kg, cgh, global_size, local_size, args, oneapi_kernel_filter_color_flip_y);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_CRYPTOMATTE_POSTPROCESS: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_cryptomatte_postprocess);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_COMPACT_STATES: {
|
||||
oneapi_call(
|
||||
kg, cgh, global_size, local_size, args, oneapi_kernel_integrator_compact_states);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_COMPACT_SHADOW_STATES: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_compact_shadow_states);
|
||||
break;
|
||||
}
|
||||
case DEVICE_KERNEL_INTEGRATOR_SHADOW_CATCHER_COUNT_POSSIBLE_SPLITS: {
|
||||
oneapi_call(kg,
|
||||
cgh,
|
||||
global_size,
|
||||
local_size,
|
||||
args,
|
||||
oneapi_kernel_integrator_shadow_catcher_count_possible_splits);
|
||||
break;
|
||||
}
|
||||
/* Unsupported kernels */
|
||||
case DEVICE_KERNEL_NUM:
|
||||
case DEVICE_KERNEL_INTEGRATOR_MEGAKERNEL:
|
||||
case DEVICE_KERNEL_INTEGRATOR_SHADOW_PATH_MNEE_PENDING:
|
||||
kernel_assert(0);
|
||||
break;
|
||||
}
|
||||
});
|
||||
}
|
||||
catch (const sycl::exception &e) {
|
||||
if (s_error_cb) {
|
||||
s_error_cb(e.what(), s_error_user_ptr);
|
||||
success = false;
|
||||
}
|
||||
}
|
||||
|
||||
# if defined(_WIN32)
|
||||
# pragma warning(default : 4062)
|
||||
# elif defined(__GNUC__)
|
||||
# pragma GCC diagnostic pop
|
||||
# endif
|
||||
return success;
|
||||
}
|
||||
|
||||
#endif /* WITH_ONEAPI */
|
||||
67
blender-5.2.0/intern/cycles/kernel/device/oneapi/kernel.h
Normal file
67
blender-5.2.0/intern/cycles/kernel/device/oneapi/kernel.h
Normal file
@@ -0,0 +1,67 @@
|
||||
/* SPDX-FileCopyrightText: 2021-2022 Intel Corporation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
|
||||
#pragma once
|
||||
|
||||
#ifdef WITH_ONEAPI
|
||||
|
||||
# include <stddef.h>
|
||||
|
||||
/* NOTE(@nsirgien): Should match underlying type in the declaration inside "kernel/types.h"
|
||||
* TODO: use kernel/types.h directly. */
|
||||
enum DeviceKernel : int;
|
||||
|
||||
# ifndef CYCLES_KERNEL_ONEAPI_EXPORT
|
||||
# ifdef _WIN32
|
||||
# if defined(ONEAPI_EXPORT)
|
||||
# define CYCLES_KERNEL_ONEAPI_EXPORT extern __declspec(dllexport)
|
||||
# else
|
||||
# define CYCLES_KERNEL_ONEAPI_EXPORT extern __declspec(dllimport)
|
||||
# endif
|
||||
# else
|
||||
# define CYCLES_KERNEL_ONEAPI_EXPORT extern __attribute__((visibility("default")))
|
||||
# endif
|
||||
# endif
|
||||
|
||||
class SyclQueue;
|
||||
class SyclDevice;
|
||||
|
||||
typedef void (*OneAPIErrorCallback)(const char *error, void *user_ptr);
|
||||
|
||||
struct KernelContext {
|
||||
/* Queue, associated with selected device */
|
||||
SyclQueue *queue = nullptr;
|
||||
/* Pointer to USM device memory with all global/constant allocation on this device */
|
||||
void *kernel_globals = nullptr;
|
||||
/* We needs this additional data for some kernels. */
|
||||
int scene_max_shaders = 0;
|
||||
};
|
||||
|
||||
/* Use extern C linking so that the symbols can be easily load from the dynamic library at runtime.
|
||||
*/
|
||||
# ifdef __cplusplus
|
||||
extern "C" {
|
||||
# endif
|
||||
|
||||
CYCLES_KERNEL_ONEAPI_EXPORT bool oneapi_run_test_kernel(SyclQueue *queue_);
|
||||
CYCLES_KERNEL_ONEAPI_EXPORT bool oneapi_zero_memory_on_device(SyclQueue *queue_,
|
||||
void *device_pointer,
|
||||
size_t num_bytes);
|
||||
CYCLES_KERNEL_ONEAPI_EXPORT void oneapi_set_error_cb(OneAPIErrorCallback cb, void *user_ptr);
|
||||
CYCLES_KERNEL_ONEAPI_EXPORT size_t oneapi_suggested_gpu_kernel_size(const DeviceKernel kernel);
|
||||
CYCLES_KERNEL_ONEAPI_EXPORT bool oneapi_enqueue_kernel(KernelContext *context,
|
||||
const int kernel,
|
||||
const size_t global_size,
|
||||
const size_t local_size,
|
||||
const unsigned int kernel_features,
|
||||
bool use_hardware_raytracing,
|
||||
void **args);
|
||||
CYCLES_KERNEL_ONEAPI_EXPORT bool oneapi_load_kernels(SyclQueue *queue,
|
||||
const unsigned int kernel_features,
|
||||
bool use_hardware_raytracing);
|
||||
# ifdef __cplusplus
|
||||
}
|
||||
|
||||
# endif
|
||||
#endif /* WITH_ONEAPI */
|
||||
@@ -0,0 +1,124 @@
|
||||
/* SPDX-FileCopyrightText: 2021-2022 Intel Corporation
|
||||
*
|
||||
* SPDX-License-Identifier: Apache-2.0 */
|
||||
|
||||
#pragma once
|
||||
|
||||
/* Some macro magic to generate templates for kernel arguments.
|
||||
* The resulting oneapi_call() template allows to call a SYCL/C++ kernel
|
||||
* with typed arguments by only giving it a void `**args` as given by Cycles.
|
||||
* The template will automatically cast from void* to the expected type. */
|
||||
|
||||
/* When expanded by the preprocessor, the generated templates will look like this example: */
|
||||
#if 0
|
||||
template<typename T0, typename T1, typename T2>
|
||||
void oneapi_call(
|
||||
KernelGlobalsGPU *kg,
|
||||
sycl::handler &cgh,
|
||||
const size_t global_size,
|
||||
const size_t local_size,
|
||||
void **args,
|
||||
void (*func)(const KernelGlobalsGPU *, size_t, size_t, sycl::handler &, T0, T1, T2))
|
||||
{
|
||||
func(kg, global_size, local_size, cgh, *(T0 *)(args[0]), *(T1 *)(args[1]), *(T2 *)(args[2]));
|
||||
}
|
||||
#endif
|
||||
|
||||
/* clang-format off */
|
||||
#define ONEAPI_TYP(x) typename T##x
|
||||
#define ONEAPI_CAST(x) *(T##x *)(args[x])
|
||||
#define ONEAPI_T(x) T##x
|
||||
|
||||
#define ONEAPI_GET_NTH_ARG(_1, _2, _3, _4, _5, _6, _7, _8, _9, _10, _11, _12, _13, _14, _15, _16, _17, _18, _19, _20, _21, _22, N, ...) N
|
||||
#define ONEAPI_0(_call, ...)
|
||||
#define ONEAPI_1(_call, x) _call(x)
|
||||
#define ONEAPI_2(_call, x, ...) _call(x), ONEAPI_1(_call, __VA_ARGS__)
|
||||
#define ONEAPI_3(_call, x, ...) _call(x), ONEAPI_2(_call, __VA_ARGS__)
|
||||
#define ONEAPI_4(_call, x, ...) _call(x), ONEAPI_3(_call, __VA_ARGS__)
|
||||
#define ONEAPI_5(_call, x, ...) _call(x), ONEAPI_4(_call, __VA_ARGS__)
|
||||
#define ONEAPI_6(_call, x, ...) _call(x), ONEAPI_5(_call, __VA_ARGS__)
|
||||
#define ONEAPI_7(_call, x, ...) _call(x), ONEAPI_6(_call, __VA_ARGS__)
|
||||
#define ONEAPI_8(_call, x, ...) _call(x), ONEAPI_7(_call, __VA_ARGS__)
|
||||
#define ONEAPI_9(_call, x, ...) _call(x), ONEAPI_8(_call, __VA_ARGS__)
|
||||
#define ONEAPI_10(_call, x, ...) _call(x), ONEAPI_9(_call, __VA_ARGS__)
|
||||
#define ONEAPI_11(_call, x, ...) _call(x), ONEAPI_10(_call, __VA_ARGS__)
|
||||
#define ONEAPI_12(_call, x, ...) _call(x), ONEAPI_11(_call, __VA_ARGS__)
|
||||
#define ONEAPI_13(_call, x, ...) _call(x), ONEAPI_12(_call, __VA_ARGS__)
|
||||
#define ONEAPI_14(_call, x, ...) _call(x), ONEAPI_13(_call, __VA_ARGS__)
|
||||
#define ONEAPI_15(_call, x, ...) _call(x), ONEAPI_14(_call, __VA_ARGS__)
|
||||
#define ONEAPI_16(_call, x, ...) _call(x), ONEAPI_15(_call, __VA_ARGS__)
|
||||
#define ONEAPI_17(_call, x, ...) _call(x), ONEAPI_16(_call, __VA_ARGS__)
|
||||
#define ONEAPI_18(_call, x, ...) _call(x), ONEAPI_17(_call, __VA_ARGS__)
|
||||
#define ONEAPI_19(_call, x, ...) _call(x), ONEAPI_18(_call, __VA_ARGS__)
|
||||
#define ONEAPI_20(_call, x, ...) _call(x), ONEAPI_19(_call, __VA_ARGS__)
|
||||
#define ONEAPI_21(_call, x, ...) _call(x), ONEAPI_20(_call, __VA_ARGS__)
|
||||
|
||||
#define ONEAPI_CALL_FOR(x, ...) \
|
||||
ONEAPI_GET_NTH_ARG("ignored", \
|
||||
##__VA_ARGS__, \
|
||||
ONEAPI_21, \
|
||||
ONEAPI_20, \
|
||||
ONEAPI_19, \
|
||||
ONEAPI_18, \
|
||||
ONEAPI_17, \
|
||||
ONEAPI_16, \
|
||||
ONEAPI_15, \
|
||||
ONEAPI_14, \
|
||||
ONEAPI_13, \
|
||||
ONEAPI_12, \
|
||||
ONEAPI_11, \
|
||||
ONEAPI_10, \
|
||||
ONEAPI_9, \
|
||||
ONEAPI_8, \
|
||||
ONEAPI_7, \
|
||||
ONEAPI_6, \
|
||||
ONEAPI_5, \
|
||||
ONEAPI_4, \
|
||||
ONEAPI_3, \
|
||||
ONEAPI_2, \
|
||||
ONEAPI_1, \
|
||||
ONEAPI_0) \
|
||||
(x, ##__VA_ARGS__)
|
||||
|
||||
/* This template automatically casts entries in the void **args array to the types requested by the kernel func.
|
||||
* Since kernel parameters are passed as void ** to the device, this is the closest that we have to type safety. */
|
||||
#define oneapi_template(...) \
|
||||
template<ONEAPI_CALL_FOR(ONEAPI_TYP, __VA_ARGS__)> \
|
||||
void oneapi_call( \
|
||||
KernelGlobalsGPU *kg, \
|
||||
sycl::handler &cgh, \
|
||||
size_t global_size, \
|
||||
size_t local_size, \
|
||||
void **args, \
|
||||
void (*func)(KernelGlobalsGPU*, size_t, size_t, sycl::handler &, ONEAPI_CALL_FOR(ONEAPI_T, __VA_ARGS__))) \
|
||||
{ \
|
||||
func(kg, \
|
||||
global_size, \
|
||||
local_size, \
|
||||
cgh, \
|
||||
ONEAPI_CALL_FOR(ONEAPI_CAST, __VA_ARGS__)); \
|
||||
}
|
||||
|
||||
oneapi_template(0)
|
||||
oneapi_template(0, 1)
|
||||
oneapi_template(0, 1, 2)
|
||||
oneapi_template(0, 1, 2, 3)
|
||||
oneapi_template(0, 1, 2, 3, 4)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7, 8)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7, 8, 9)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19)
|
||||
oneapi_template(0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20)
|
||||
|
||||
/* clang-format on */
|
||||
Reference in New Issue
Block a user