From 699ee4484fefb6f41bc6b186266d5a9e8f1c03b7 Mon Sep 17 00:00:00 2001 From: Tim Grant Date: Wed, 23 Sep 2026 16:12:42 -0600 Subject: [PATCH 1/5] build(cuda): supply removed CUDA 13 texture header CUDA 13 removed texture_fetch_functions.h, but older Clang CUDA wrappers can still include it while generating device bitcode. Search an OSL compatibility directory first and provide an empty stub because OSL does not use the legacy texture-reference API. Assisted-by: OpenAI Codex / GPT-6 Signed-off-by: Tim Grant --- src/cmake/cuda_compat/texture_fetch_functions.h | 7 +++++++ src/cmake/cuda_macros.cmake | 8 ++++++++ 2 files changed, 15 insertions(+) create mode 100644 src/cmake/cuda_compat/texture_fetch_functions.h diff --git a/src/cmake/cuda_compat/texture_fetch_functions.h b/src/cmake/cuda_compat/texture_fetch_functions.h new file mode 100644 index 0000000000..065cf77f04 --- /dev/null +++ b/src/cmake/cuda_compat/texture_fetch_functions.h @@ -0,0 +1,7 @@ +// Copyright Contributors to the Open Shading Language project. +// SPDX-License-Identifier: BSD-3-Clause + +// CUDA 13 removed this header, but older Clang CUDA wrappers still include it. +// OSL does not use the legacy texture-reference API, so an empty stub is +// sufficient for CUDA bitcode generation. +#pragma once diff --git a/src/cmake/cuda_macros.cmake b/src/cmake/cuda_macros.cmake index e87b359f16..76c3b5377d 100644 --- a/src/cmake/cuda_macros.cmake +++ b/src/cmake/cuda_macros.cmake @@ -148,6 +148,13 @@ function ( MAKE_CUDA_BITCODE src suffix generated_bc extra_clang_args ) set (CUDA_TEXREF_FIX "-D__CLANG_CUDA_TEXTURE_INTRINSICS_H__") endif() + if ("${CUDA_VERSION}" VERSION_GREATER_EQUAL "13.0") + # CUDA 13 removed texture_fetch_functions.h, but older Clang CUDA + # wrappers still include it. Search OSL's compatibility headers first. + set (CUDA_COMPAT_INCLUDE + "-I${PROJECT_SOURCE_DIR}/src/cmake/cuda_compat") + endif () + list (TRANSFORM IMATH_INCLUDES PREPEND -I OUTPUT_VARIABLE ALL_IMATH_INCLUDES) list (TRANSFORM OPENEXR_INCLUDES PREPEND -I @@ -157,6 +164,7 @@ function ( MAKE_CUDA_BITCODE src suffix generated_bc extra_clang_args ) add_custom_command (OUTPUT ${bc_cuda} COMMAND ${LLVM_BC_GENERATOR} + ${CUDA_COMPAT_INCLUDE} "-I${OPTIX_INCLUDES}" "-I${CUDA_INCLUDES}" "-I${CMAKE_CURRENT_SOURCE_DIR}" From 1a92b1ceaccd2e8aab8854da635a173de9394cc2 Mon Sep 17 00:00:00 2001 From: Tim Grant Date: Wed, 23 Sep 2026 16:14:37 -0600 Subject: [PATCH 2/5] build(cuda): support CUDA 13.2 rsqrt with older Clang CUDA 13.2 added _NV_RSQRT_SPECIFIER, but older Clang CUDA wrappers include math_functions.hpp without defining it. Supply a narrow compatibility wrapper that mirrors the definition in newer Clang, including noexcept(true) on glibc 2.42. With the header mismatch handled locally, remove the configuration-time rejection of CUDA 13.2+ with LLVM older than 22.1.2. Assisted-by: OpenAI Codex / GPT-6 Signed-off-by: Tim Grant --- src/cmake/cuda_compat/crt/math_functions.hpp | 21 ++++++++++++++++++++ src/cmake/cuda_macros.cmake | 2 ++ src/cmake/externalpackages.cmake | 10 ---------- 3 files changed, 23 insertions(+), 10 deletions(-) create mode 100644 src/cmake/cuda_compat/crt/math_functions.hpp diff --git a/src/cmake/cuda_compat/crt/math_functions.hpp b/src/cmake/cuda_compat/crt/math_functions.hpp new file mode 100644 index 0000000000..965721fcc1 --- /dev/null +++ b/src/cmake/cuda_compat/crt/math_functions.hpp @@ -0,0 +1,21 @@ +// Copyright Contributors to the Open Shading Language project. +// SPDX-License-Identifier: BSD-3-Clause + +// CUDA 13.2 added _NV_RSQRT_SPECIFIER to math_functions.hpp, but older Clang +// CUDA wrappers include that file without defining the macro first. Match the +// definition used by newer Clang wrappers, including glibc 2.42's exception +// specifier. +// Upstream fix: https://github.com/llvm/llvm-project/pull/185701 +#ifndef _NV_RSQRT_SPECIFIER +# if defined(__GNUC__) && defined(__GLIBC_PREREQ) +# if __GLIBC_PREREQ(2, 42) +# define _NV_RSQRT_SPECIFIER noexcept(true) +# endif +# endif +# ifndef _NV_RSQRT_SPECIFIER +# define _NV_RSQRT_SPECIFIER +# endif +#endif + +// Clang's CUDA wrapper intentionally includes this header more than once. +#include_next diff --git a/src/cmake/cuda_macros.cmake b/src/cmake/cuda_macros.cmake index 76c3b5377d..5bab74c59c 100644 --- a/src/cmake/cuda_macros.cmake +++ b/src/cmake/cuda_macros.cmake @@ -153,6 +153,8 @@ function ( MAKE_CUDA_BITCODE src suffix generated_bc extra_clang_args ) # wrappers still include it. Search OSL's compatibility headers first. set (CUDA_COMPAT_INCLUDE "-I${PROJECT_SOURCE_DIR}/src/cmake/cuda_compat") + list (APPEND exec_headers + "${PROJECT_SOURCE_DIR}/src/cmake/cuda_compat/crt/math_functions.hpp") endif () list (TRANSFORM IMATH_INCLUDES PREPEND -I diff --git a/src/cmake/externalpackages.cmake b/src/cmake/externalpackages.cmake index 9cbbee1d6a..11343a0228 100644 --- a/src/cmake/externalpackages.cmake +++ b/src/cmake/externalpackages.cmake @@ -141,16 +141,6 @@ if (OSL_USE_OPTIX) message (FATAL_ERROR "NVPTX target is not available in the provided LLVM build") endif() - # CUDA 13.2+ requires LLVM 22.1.2+ for the _NV_RSQRT_SPECIFIER fix. - # See: https://github.com/AcademySoftwareFoundation/OpenShadingLanguage/issues/2129 - if (CUDA_VERSION VERSION_GREATER_EQUAL "13.2" AND LLVM_VERSION VERSION_LESS "22.1.2") - message (FATAL_ERROR - "${ColorRed}CUDA ${CUDA_VERSION} requires LLVM 22.1.2 or newer " - "(you have LLVM ${LLVM_VERSION}). CUDA 13.2+ needs the " - "_NV_RSQRT_SPECIFIER fix from llvm-project commit c1c833734cf0, " - "first shipped in LLVM 22.1.2. Either upgrade LLVM or downgrade CUDA.${ColorReset}") - endif () - set (CUDA_LIB_FLAGS "--cuda-path=${CUDA_TOOLKIT_ROOT_DIR}") # If the user wants, try to use static libs here to putting static lib From a3955902099c7d6740e60af03ed705f23c0ef363 Mon Sep 17 00:00:00 2001 From: Tim Grant Date: Wed, 30 Sep 2026 16:17:56 -0600 Subject: [PATCH 3/5] fix(cuda): preload standard headers with noinline macro hidden For non-Windows LLVM versions older than 18, force-include an OSL header that temporarily hides the CUDA __noinline__ macro while loading string and memory. Restore the macro afterward so device code can still use it; standard-library include guards prevent the affected declarations from being parsed again. Track the pre-include header as a bitcode build dependency. Assisted-by: OpenAI Codex / GPT-6 Signed-off-by: Tim Grant --- src/cmake/cuda_compat/cuda_stdlib.h | 14 ++++++++++++++ src/cmake/cuda_macros.cmake | 8 ++++++++ 2 files changed, 22 insertions(+) create mode 100644 src/cmake/cuda_compat/cuda_stdlib.h diff --git a/src/cmake/cuda_compat/cuda_stdlib.h b/src/cmake/cuda_compat/cuda_stdlib.h new file mode 100644 index 0000000000..6fe9a43145 --- /dev/null +++ b/src/cmake/cuda_compat/cuda_stdlib.h @@ -0,0 +1,14 @@ +// Copyright Contributors to the Open Shading Language project. +// SPDX-License-Identifier: BSD-3-Clause + +#pragma once + +// Preload these headers before compiling OSL device code. CUDA's __noinline__ +// macro conflicts with attribute names in libstdc++; hide it during the includes +// and restore it for device code. The headers' include guards prevent reprocessing. +// Related upstream fix: https://github.com/llvm/llvm-project/pull/66138 +#pragma push_macro("__noinline__") +#undef __noinline__ +#include +#include +#pragma pop_macro("__noinline__") diff --git a/src/cmake/cuda_macros.cmake b/src/cmake/cuda_macros.cmake index 5bab74c59c..e8a56f5175 100644 --- a/src/cmake/cuda_macros.cmake +++ b/src/cmake/cuda_macros.cmake @@ -157,6 +157,13 @@ function ( MAKE_CUDA_BITCODE src suffix generated_bc extra_clang_args ) "${PROJECT_SOURCE_DIR}/src/cmake/cuda_compat/crt/math_functions.hpp") endif () + # Load libstdc++ before OSL headers while CUDA's __noinline__ macro is hidden. + if (NOT WIN32 AND LLVM_VERSION VERSION_LESS 18.0) + set (CUDA_STDLIB_HEADER "${PROJECT_SOURCE_DIR}/src/cmake/cuda_compat/cuda_stdlib.h") + set (CLANG_CUDA_COMPAT_FLAGS -include "${CUDA_STDLIB_HEADER}") + list (APPEND exec_headers "${CUDA_STDLIB_HEADER}") + endif () + list (TRANSFORM IMATH_INCLUDES PREPEND -I OUTPUT_VARIABLE ALL_IMATH_INCLUDES) list (TRANSFORM OPENEXR_INCLUDES PREPEND -I @@ -167,6 +174,7 @@ function ( MAKE_CUDA_BITCODE src suffix generated_bc extra_clang_args ) add_custom_command (OUTPUT ${bc_cuda} COMMAND ${LLVM_BC_GENERATOR} ${CUDA_COMPAT_INCLUDE} + ${CLANG_CUDA_COMPAT_FLAGS} "-I${OPTIX_INCLUDES}" "-I${CUDA_INCLUDES}" "-I${CMAKE_CURRENT_SOURCE_DIR}" From 6694900d84aafb36bc53b01ee6b06a10aaaaac36 Mon Sep 17 00:00:00 2001 From: Tim Grant Date: Wed, 23 Sep 2026 16:19:41 -0600 Subject: [PATCH 4/5] build(cuda): expose custom NVCC arguments The cache declaration accidentally used a literal dollar-prefixed variable name while NVCC_COMPILE reads OSL_EXTRA_NVCC_ARGS. Declare the intended cache variable so builds can inject dependency-specific NVCC flags without hard-coding them in OSL. Assisted-by: OpenAI Codex / GPT-6 Signed-off-by: Tim Grant --- src/cmake/cuda_macros.cmake | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/cmake/cuda_macros.cmake b/src/cmake/cuda_macros.cmake index e8a56f5175..1a9f77a2e8 100644 --- a/src/cmake/cuda_macros.cmake +++ b/src/cmake/cuda_macros.cmake @@ -2,7 +2,7 @@ # SPDX-License-Identifier: BSD-3-Clause # https://github.com/AcademySoftwareFoundation/OpenShadingLanguage -set ($OSL_EXTRA_NVCC_ARGS "" CACHE STRING "Custom args passed to nvcc when compiling CUDA code") +set (OSL_EXTRA_NVCC_ARGS "" CACHE STRING "Custom args passed to nvcc when compiling CUDA code") set (CUDA_OPT_FLAG_NVCC "-O3" CACHE STRING "The optimization level to use when compiling CUDA/C++ files with nvcc") set (CUDA_OPT_FLAG_CLANG "-O3" CACHE STRING "The optimization level to use when compiling CUDA/C++ files with clang") From 4724d647b14c0ba325cbd03963cfec3ea02cdcae Mon Sep 17 00:00:00 2001 From: Tim Grant Date: Wed, 7 Oct 2026 14:58:48 -0600 Subject: [PATCH 5/5] clang-format tweak. Signed-off-by: Tim Grant --- src/cmake/cuda_compat/cuda_stdlib.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/cmake/cuda_compat/cuda_stdlib.h b/src/cmake/cuda_compat/cuda_stdlib.h index 6fe9a43145..aa8cd1f7a2 100644 --- a/src/cmake/cuda_compat/cuda_stdlib.h +++ b/src/cmake/cuda_compat/cuda_stdlib.h @@ -9,6 +9,6 @@ // Related upstream fix: https://github.com/llvm/llvm-project/pull/66138 #pragma push_macro("__noinline__") #undef __noinline__ -#include #include +#include #pragma pop_macro("__noinline__")