diff --git a/.github/workflows/build-upload-conda.yml b/.github/workflows/build-upload-conda.yml index 74bbc236c29f..26e3360aa489 100644 --- a/.github/workflows/build-upload-conda.yml +++ b/.github/workflows/build-upload-conda.yml @@ -64,7 +64,7 @@ jobs: # Select the python variant; 3.14t needs the free-threaded ABI. if [[ "${{ matrix.python-version }}" == "3.14t" ]]; then - PY_ARGS=(--python "3.14.* *_cp314t" -m buildscripts/conda-recipes/pyomp/freethreading.yaml) + PY_ARGS=(--python "3.14.* *_cp314t") else PY_ARGS=(--python ${{ matrix.python-version }}) fi diff --git a/.github/workflows/build-upload-wheels.yml b/.github/workflows/build-upload-wheels.yml index 0bec5453fcff..e0a56b05facb 100644 --- a/.github/workflows/build-upload-wheels.yml +++ b/.github/workflows/build-upload-wheels.yml @@ -52,7 +52,7 @@ jobs: # Set LLVM_VERSION for the host to forward to the cibuildwheel # environment. env: - LLVM_VERSION: "20.1.8" + LLVM_VERSION: "22.1.8" run: python -m cibuildwheel --output-dir wheelhouse - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 @@ -69,7 +69,7 @@ jobs: - name: Build sdist env: - LLVM_VERSION: "20.1.8" + LLVM_VERSION: "22.1.8" run: pipx run build --sdist - uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 @@ -85,23 +85,7 @@ jobs: matrix: os: [ubuntu-latest, macos-latest, ubuntu-24.04-arm] python-version: ['3.10', '3.11', '3.12', '3.13', '3.14', '3.14t'] - numba-version: ['0.62.0', '0.62.1', '0.63.0', '0.63.1', '0.64.0', '0.65.0', '0.65.1'] - exclude: - - python-version: '3.14' - numba-version: '0.62.0' - - python-version: '3.14' - numba-version: '0.62.1' - # Numba supports free-threaded Python starting with 0.65. - - python-version: '3.14t' - numba-version: '0.62.0' - - python-version: '3.14t' - numba-version: '0.62.1' - - python-version: '3.14t' - numba-version: '0.63.0' - - python-version: '3.14t' - numba-version: '0.63.1' - - python-version: '3.14t' - numba-version: '0.64.0' + numba-version: ['0.66.0', '0.67.0'] steps: - name: Download built wheels diff --git a/README.md b/README.md index cadef92ae7fe..6c92ea7ebc91 100644 --- a/README.md +++ b/README.md @@ -45,11 +45,12 @@ conda install -c python-for-hpc -c conda-forge pyomp | PyOMP | Numba | | ----- | --------------- | +| 0.6.x | 0.66.x - 0.67.x | | 0.5.x | 0.62.x - 0.65.x | | 0.4.x | 0.61.x | | 0.3.x | 0.57.x - 0.60.x | -Python 3.14t (free-threaded) requires Numba 0.65.x. +Python 3.14t (free-threaded) requires PyOMP 0.5.2 or later and Numba 0.65.x or later. Besides a standard installation, we also provide the following options to quickly try out PyOMP online or through a container. diff --git a/buildscripts/cibuildwheel/setup-miniforge3.sh b/buildscripts/cibuildwheel/setup-miniforge3.sh index 2142c212429f..53707d7d0c19 100644 --- a/buildscripts/cibuildwheel/setup-miniforge3.sh +++ b/buildscripts/cibuildwheel/setup-miniforge3.sh @@ -58,6 +58,7 @@ bash _downloads/miniforge3.sh -b -f -p "_stage/miniforge3" echo "Miniforge installed" source "_stage/miniforge3/bin/activate" base -# Create conda environment with tools and libraries for the LLVM_VERSION. +# Create conda environment with tools and libraries for the LLVM_VERSION. lld +# links the AMDGPU device runtime. echo "Installing llvmdev ${LLVM_VERSION}..." -conda create -n llvmdev-${LLVM_VERSION} --override-channels -c conda-forge -q -y clang=${LLVM_VERSION} clangxx=${LLVM_VERSION} clang-tools=${LLVM_VERSION} llvmdev=${LLVM_VERSION} zstd +conda create -n llvmdev-${LLVM_VERSION} --override-channels -c conda-forge -q -y clang=${LLVM_VERSION} clangxx=${LLVM_VERSION} clang-tools=${LLVM_VERSION} llvmdev=${LLVM_VERSION} lld=${LLVM_VERSION} zstd diff --git a/buildscripts/conda-recipes/pyomp/conda_build_config.yaml b/buildscripts/conda-recipes/pyomp/conda_build_config.yaml index ab3a7bdab64b..7db90af45097 100644 --- a/buildscripts/conda-recipes/pyomp/conda_build_config.yaml +++ b/buildscripts/conda-recipes/pyomp/conda_build_config.yaml @@ -1,8 +1,3 @@ -# --- Free-threading --- -# Default for the is_freethreading selector in meta.yaml; override for 3.14t. -is_freethreading: - - false - # --- Compilers --- c_compiler_version: - 9 # [linux] diff --git a/buildscripts/conda-recipes/pyomp/freethreading.yaml b/buildscripts/conda-recipes/pyomp/freethreading.yaml deleted file mode 100644 index cc62364b2688..000000000000 --- a/buildscripts/conda-recipes/pyomp/freethreading.yaml +++ /dev/null @@ -1,4 +0,0 @@ -# Variant for free-threaded Python (3.14t) builds, passed with -m. Use this -# instead of --variants, which `conda build --output` ignores. -is_freethreading: - - true diff --git a/buildscripts/conda-recipes/pyomp/meta.yaml b/buildscripts/conda-recipes/pyomp/meta.yaml index e7c7030a050e..7f9b94920bd0 100644 --- a/buildscripts/conda-recipes/pyomp/meta.yaml +++ b/buildscripts/conda-recipes/pyomp/meta.yaml @@ -1,5 +1,5 @@ {% set version = environ.get('GIT_DESCRIBE_TAG', '0.0.0').lstrip('v') %} -{% set LLVM_VERSION = environ.get('LLVM_VERSION', '20.1.8') %} +{% set LLVM_VERSION = environ.get('LLVM_VERSION', '22.1.8') %} {% set LLVM_VERSION_MAJOR = LLVM_VERSION.split('.')[0] %} {% set LLVM_VERSION_MINOR = LLVM_VERSION.split('.')[1] %} @@ -45,26 +45,24 @@ requirements: - sysroot_linux-aarch64 # [aarch64] - setuptools - setuptools_scm - - numba >=0.62, <0.66 - - numba >=0.65, <0.66 # [is_freethreading] + - numba >=0.66, <0.68 - clang {{ LLVM_VERSION }} - clangxx {{ LLVM_VERSION }} - clang-tools {{ LLVM_VERSION }} - llvmdev {{ LLVM_VERSION }} + # lld links the AMDGPU device runtime. + - lld {{ LLVM_VERSION }} # [linux] - zlib # require llvm-openmp for the openmp cpu runtime. - - llvm-openmp 21.1.8.* # [osx] - - llvm-openmp {{ LLVM_VERSION }} # [not osx] + - llvm-openmp {{ LLVM_VERSION }} - elfutils # [linux] - libffi # [linux] run: - python - setuptools - - numba >=0.62, <0.66 - - numba >=0.65, <0.66 # [is_freethreading] + - numba >=0.66, <0.68 # require llvm-openmp for the openmp cpu runtime. - - llvm-openmp 21.1.8.* # [osx] - - llvm-openmp {{ LLVM_VERSION }} # [not osx] + - llvm-openmp {{ LLVM_VERSION }} - lark - cffi diff --git a/buildscripts/conda-recipes/pyomp/run_test.sh b/buildscripts/conda-recipes/pyomp/run_test.sh index 64ac06240a74..d2f9d0932799 100644 --- a/buildscripts/conda-recipes/pyomp/run_test.sh +++ b/buildscripts/conda-recipes/pyomp/run_test.sh @@ -12,7 +12,7 @@ export PYTHONFAULTHANDLER=1 # of low accuracy SVML libm replacements in ufunc loops. _NPY_CMD='from numba.misc import numba_sysinfo;\ sysinfo=numba_sysinfo.get_sysinfo();\ - print(sysinfo["NumPy AVX512_SKX detected"] and + print("AVX512_SKX" in sysinfo["NumPy Supported SIMD features"] and sysinfo["NumPy Version"]>="1.22")' NUMPY_DETECTS_AVX512_SKX_NP_GT_122=$(python -c "$_NPY_CMD") echo "NumPy >= 1.22 with AVX512_SKX detected: $NUMPY_DETECTS_AVX512_SKX_NP_GT_122" diff --git a/buildscripts/modal/test_gpu.py b/buildscripts/modal/test_gpu.py index 59f4d919d9b7..571292517e11 100644 --- a/buildscripts/modal/test_gpu.py +++ b/buildscripts/modal/test_gpu.py @@ -8,7 +8,7 @@ PYTHON_VERSIONS = ("3.10", "3.11", "3.12", "3.13", "3.14") -NUMBA_VERSION = "0.65.1" +NUMBA_VERSION = "0.67.0" MINIFORGE_VERSION = "26.3.2-3" MINIFORGE_SHA256 = "848194851a98903134187fbb4ab50efe87b003e0c0f808f97644b7524a62bf2c" WHEEL_DIRECTORY = ( diff --git a/docs/source/installation.rst b/docs/source/installation.rst index 6e0eeba7db19..fb1157a883cd 100644 --- a/docs/source/installation.rst +++ b/docs/source/installation.rst @@ -24,6 +24,8 @@ supported. +--------+---------------------+ | PyOMP | Numba | +========+=====================+ +| 0.6.x | 0.66.x - 0.67.x | ++--------+---------------------+ | 0.5.x | 0.62.x - 0.65.x | +--------+---------------------+ | 0.4.x | 0.61.x | @@ -31,7 +33,7 @@ supported. | 0.3.x | 0.57.x - 0.60.x | +--------+---------------------+ -Python 3.14t (free-threaded) requires Numba 0.65.x. +Python 3.14t (free-threaded) requires PyOMP 0.5.2 or later and Numba 0.65.x or later. Additional options ------------------ diff --git a/docs/source/openmp.rst b/docs/source/openmp.rst index 4dd05db54d07..536a2fba1f90 100644 --- a/docs/source/openmp.rst +++ b/docs/source/openmp.rst @@ -356,7 +356,8 @@ The following table shows tested combinations of PyOMP, Numba, Python, LLVM, and ===================== ==================== ==================== ============ ================================ PyOMP Numba Python LLVM Supported Platforms ===================== ==================== ==================== ============ ================================ - 0.5.x 0.62.x - 0.63.x 3.10 - 3.14 20.x linux-64, osx-arm64, linux-arm64 + 0.6.x 0.66.x - 0.67.x 3.10 - 3.14 22.x linux-64, osx-arm64, linux-arm64 + 0.5.x 0.62.x - 0.65.x 3.10 - 3.14 20.x linux-64, osx-arm64, linux-arm64 0.4.x 0.61.x 3.10 - 3.13 15.x linux-64, osx-arm64, linux-arm64 0.3.x 0.57.x - 0.60.x 3.9 - 3.12 14.x linux-64, osx-arm64, linux-arm64 ===================== ==================== ==================== ============ ================================ @@ -384,7 +385,9 @@ Platform details Notes ^^^^^ -* Python 3.14 free-threaded build (cp314t) is not supported with the current Numba/llvmlite version. -* LLVM version 20.1.8 is used for the current PyOMP 0.5.x releases. +* Python 3.14 free-threaded builds (cp314t) are supported starting with PyOMP + 0.5.2 (Numba 0.65.x). +* LLVM version 22.1.8 is used for the current PyOMP 0.6.x releases, and LLVM + version 20.1.8 for the PyOMP 0.5.x releases. * For GPU offloading support, NVIDIA GPU and NVIDIA driver are required on supported Linux platforms. * AMD GPU support is in active development. diff --git a/pyproject.toml b/pyproject.toml index b8aedd06d6ba..c23e1444c457 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -17,7 +17,7 @@ classifiers = [ "Intended Audience :: Developers", "Topic :: Software Development :: Compilers", ] -dependencies = ["numba>=0.62, <0.66", "lark", "cffi", "setuptools"] +dependencies = ["numba>=0.66, <0.68", "lark", "cffi", "setuptools"] maintainers = [ { name = "Giorgis Georgakoudis", email = "georgakoudis1@llnl.gov" }, ] @@ -55,16 +55,23 @@ USE_CXX11_ABI = "1" PIP_NO_INPUT = "1" [tool.cibuildwheel.linux] +# The manylinux image has no LLVM_VERSION packages, so use the conda-forge +# toolchain. libzstd-devel is needed by the conda LLVM cmake exports. before-all = [ - "yum install -y elfutils-libelf-devel libffi-devel clang-devel-20.1.8 llvm-devel-20.1.8", + "yum install -y elfutils-libelf-devel libffi-devel libzstd-devel", + "bash buildscripts/cibuildwheel/setup-miniforge3.sh", ] [tool.cibuildwheel.linux.environment] ENABLE_BUNDLED_LIBOMP = "1" ENABLE_BUNDLED_LIBOMPTARGET = "1" -LLVM_DIR = "/usr/lib64/cmake/llvm" -CC = "/usr/bin/clang" -CXX = "/usr/bin/clang++" +LLVM_DIR = "${PWD}/_stage/miniforge3/envs/llvmdev-${LLVM_VERSION}/lib/cmake/llvm" +CC = "${PWD}/_stage/miniforge3/envs/llvmdev-${LLVM_VERSION}/bin/clang" +CXX = "${PWD}/_stage/miniforge3/envs/llvmdev-${LLVM_VERSION}/bin/clang++" +# Compile against the image's gcc-toolset libstdc++ instead of the conda one +# that clang selects by default, so the libraries stay manylinux compatible. +CFLAGS = "--gcc-toolchain=/opt/rh/gcc-toolset-14/root/usr" +CXXFLAGS = "--gcc-toolchain=/opt/rh/gcc-toolset-14/root/usr" [tool.cibuildwheel.macos] before-all = ["bash buildscripts/cibuildwheel/setup-miniforge3.sh"] diff --git a/setup.py b/setup.py index 9bb623935421..58ff2130568f 100644 --- a/setup.py +++ b/setup.py @@ -1,5 +1,6 @@ from pathlib import Path import subprocess +import shlex import shutil import tarfile import urllib @@ -133,6 +134,9 @@ def _build_cmake(self, ext: CMakeExtension): include_dir = install_dir / "lib/cmake" if include_dir.exists(): shutil.rmtree(include_dir) + # Keep only the device runtime bitcode, not its static archive. + if ext.name.startswith("devicertl"): + (install_dir / "lib/libompdevice.a").unlink(missing_ok=True) # Remove symlinks in the install directory to avoid copies. for file in install_dir.rglob("*"): if file.is_symlink(): @@ -209,9 +213,11 @@ def _build_loader_wrapper(self): loader_so = target_lib_dir / f"libpyomp_loader{lib_ext}" cc = os.environ.get("CC", "gcc") - # Compile the wrapper. + # Compile the wrapper. Pass CFLAGS so it uses the same toolchain as the + # other libraries, e.g., the manylinux --gcc-toolchain. cmd = [ cc, + *shlex.split(os.environ.get("CFLAGS", "")), "-shared", "-fPIC", str(loader_c), @@ -359,6 +365,8 @@ def _check_true(env_var): cmake_args=[ "-DOPENMP_STANDALONE_BUILD=ON", "-DLLVM_ENABLE_RUNTIMES=offload", + # Tests depend on liboffload, which is not built. + "-DOFFLOAD_INCLUDE_TESTS=OFF", # Avoid conflicts in manylinux builds with packaged clang/llvm # under /usr/include and its gcc-toolset provided header files. "-DCMAKE_NO_SYSTEM_FROM_IMPORTED=ON", @@ -366,6 +374,34 @@ def _check_true(env_var): ) ) + # Build the device runtime bitcode (libomptarget-.bc) for each GPU + # target, compiled by clang from openmp/device. + for triple in ("nvptx64-nvidia-cuda", "amdgcn-amd-amdhsa"): + ext_modules.append( + CMakeExtension( + f"devicertl-{triple}", + setup=PrepareOpenMP, + source_dir=PrepareOpenMP.get_source_dir(), + install_dir="openmp", + cmake_args=[ + "-DOPENMP_STANDALONE_BUILD=ON", + "-DLLVM_ENABLE_RUNTIMES=openmp", + f"-DLLVM_RUNTIMES_TARGET={triple}", + f"-DLLVM_DEFAULT_TARGET_TRIPLE={triple}", + f"-DCMAKE_C_COMPILER_TARGET={triple}", + f"-DCMAKE_CXX_COMPILER_TARGET={triple}", + # The compiler cannot link host test programs for GPU + # targets. + "-DCMAKE_C_COMPILER_WORKS=ON", + "-DCMAKE_CXX_COMPILER_WORKS=ON", + # Ignore host flags from CFLAGS and CXXFLAGS, such as + # conda's -march, which fail for GPU targets. + "-DCMAKE_C_FLAGS=", + "-DCMAKE_CXX_FLAGS=", + ], + ) + ) + setup( ext_modules=ext_modules, diff --git a/src/numba/openmp/libs/openmp/patches/14.0.6/0001-BACKPORT-Fix-for-CUDA-OpenMP-RTL.patch b/src/numba/openmp/libs/openmp/patches/14.0.6/0001-BACKPORT-Fix-for-CUDA-OpenMP-RTL.patch deleted file mode 100644 index baa96cda795e..000000000000 --- a/src/numba/openmp/libs/openmp/patches/14.0.6/0001-BACKPORT-Fix-for-CUDA-OpenMP-RTL.patch +++ /dev/null @@ -1,39 +0,0 @@ -diff --git a/libomptarget/plugins/cuda/src/rtl.cpp b/libomptarget/plugins/cuda/src/rtl.cpp -index 0ca05f0ec3a0..16da3f434bba 100644 ---- a/libomptarget/plugins/cuda/src/rtl.cpp -+++ b/libomptarget/plugins/cuda/src/rtl.cpp -@@ -234,6 +234,7 @@ template class ResourcePoolTy { - std::mutex Mutex; - /// Pool of resources. - std::vector Resources; -+ std::vector Pool; - /// A reference to the corresponding allocator. - AllocatorTy Allocator; - -@@ -243,11 +244,13 @@ template class ResourcePoolTy { - auto CurSize = Resources.size(); - assert(Size > CurSize && "Unexpected smaller size"); - Resources.reserve(Size); -+ Pool.reserve(Size); - for (auto I = CurSize; I < Size; ++I) { - T NewItem; - int Ret = Allocator.create(NewItem); - if (Ret != OFFLOAD_SUCCESS) - return false; -+ Pool.push_back(NewItem); - Resources.push_back(NewItem); - } - return true; -@@ -308,8 +311,9 @@ public: - /// Released all stored resources and clear the pool. - /// Note: This function is not thread safe. Be sure to guard it if necessary. - void clear() noexcept { -- for (auto &R : Resources) -+ for (auto &R : Pool) - (void)Allocator.destroy(R); -+ Pool.clear(); - Resources.clear(); - } - }; --- -2.29.1 diff --git a/src/numba/openmp/libs/openmp/patches/14.0.6/0002-Fix-missing-includes.patch b/src/numba/openmp/libs/openmp/patches/14.0.6/0002-Fix-missing-includes.patch deleted file mode 100644 index 76f3c3105175..000000000000 --- a/src/numba/openmp/libs/openmp/patches/14.0.6/0002-Fix-missing-includes.patch +++ /dev/null @@ -1,12 +0,0 @@ -diff -Naur a/libomptarget/include/Debug.h b/libomptarget/include/Debug.h ---- a/libomptarget/include/Debug.h -+++ b/libomptarget/include/Debug.h -@@ -39,6 +39,8 @@ - - #include - #include -+#include -+#include - - /// 32-Bit field data attributes controlling information presented to the user. - enum OpenMPInfoType : uint32_t { diff --git a/src/numba/openmp/libs/openmp/patches/14.0.6/0003-Link-static-LLVM-libs.patch b/src/numba/openmp/libs/openmp/patches/14.0.6/0003-Link-static-LLVM-libs.patch deleted file mode 100644 index aac8c1b7de2a..000000000000 --- a/src/numba/openmp/libs/openmp/patches/14.0.6/0003-Link-static-LLVM-libs.patch +++ /dev/null @@ -1,13 +0,0 @@ -diff -Naur a/libomptarget/plugins/common/elf_common/CMakeLists.txt b/libomptarget/plugins/common/elf_common/CMakeLists.txt ---- a/libomptarget/plugins/common/elf_common/CMakeLists.txt -+++ b/libomptarget/plugins/common/elf_common/CMakeLists.txt -@@ -16,9 +16,6 @@ - set_property(TARGET elf_common PROPERTY POSITION_INDEPENDENT_CODE ON) - llvm_update_compile_flags(elf_common) - set(LINK_LLVM_LIBS LLVMBinaryFormat LLVMObject LLVMSupport) --if (LLVM_LINK_LLVM_DYLIB) -- set(LINK_LLVM_LIBS LLVM) --endif() - target_link_libraries(elf_common INTERFACE ${LINK_LLVM_LIBS}) - include_directories(${LIBOMPTARGET_LLVM_INCLUDE_DIRS}) - add_dependencies(elf_common ${LINK_LLVM_LIBS}) diff --git a/src/numba/openmp/libs/openmp/patches/15.0.7/0001-Fix-missing-includes.patch b/src/numba/openmp/libs/openmp/patches/15.0.7/0001-Fix-missing-includes.patch deleted file mode 100644 index 86a42aa23c42..000000000000 --- a/src/numba/openmp/libs/openmp/patches/15.0.7/0001-Fix-missing-includes.patch +++ /dev/null @@ -1,14 +0,0 @@ -diff --git a/libomptarget/include/Debug.h b/libomptarget/include/Debug.h -index 8ff4695..d789551 100644 ---- a/libomptarget/include/Debug.h -+++ b/libomptarget/include/Debug.h -@@ -38,7 +38,9 @@ - #define _OMPTARGET_DEBUG_H - - #include -+#include - #include -+#include - - /// 32-Bit field data attributes controlling information presented to the user. - enum OpenMPInfoType : uint32_t { diff --git a/src/numba/openmp/libs/openmp/patches/15.0.7/0002-Link-LLVM-statically.patch b/src/numba/openmp/libs/openmp/patches/15.0.7/0002-Link-LLVM-statically.patch deleted file mode 100644 index 8952859b2e4a..000000000000 --- a/src/numba/openmp/libs/openmp/patches/15.0.7/0002-Link-LLVM-statically.patch +++ /dev/null @@ -1,101 +0,0 @@ -diff --git a/libomptarget/plugins/CMakeLists.txt b/libomptarget/plugins/CMakeLists.txt -index 64c2539..6abc109 100644 ---- a/libomptarget/plugins/CMakeLists.txt -+++ b/libomptarget/plugins/CMakeLists.txt -@@ -31,7 +31,7 @@ if(CMAKE_SYSTEM_PROCESSOR MATCHES "${tmachine}$") - add_definitions("-DTARGET_ELF_ID=${elf_machine_id}") - - add_llvm_library("omptarget.rtl.${tmachine_libname}" -- SHARED -+ SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - - ${CMAKE_CURRENT_SOURCE_DIR}/../generic-elf-64bit/src/rtl.cpp - -@@ -97,4 +97,3 @@ add_subdirectory(remote) - # Make sure the parent scope can see the plugins that will be created. - set(LIBOMPTARGET_SYSTEM_TARGETS "${LIBOMPTARGET_SYSTEM_TARGETS}" PARENT_SCOPE) - set(LIBOMPTARGET_TESTED_PLUGINS "${LIBOMPTARGET_TESTED_PLUGINS}" PARENT_SCOPE) -- -diff --git a/libomptarget/plugins/amdgpu/CMakeLists.txt b/libomptarget/plugins/amdgpu/CMakeLists.txt -index 66bf680..47935e5 100644 ---- a/libomptarget/plugins/amdgpu/CMakeLists.txt -+++ b/libomptarget/plugins/amdgpu/CMakeLists.txt -@@ -66,7 +66,7 @@ else() - set(LDFLAGS_UNDEFINED "-Wl,-z,defs") - endif() - --add_llvm_library(omptarget.rtl.amdgpu SHARED -+add_llvm_library(omptarget.rtl.amdgpu SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - impl/impl.cpp - impl/interop_hsa.cpp - impl/data.cpp -@@ -126,4 +126,3 @@ else() - list(APPEND LIBOMPTARGET_TESTED_PLUGINS "omptarget.rtl.amdgpu") - set(LIBOMPTARGET_TESTED_PLUGINS "${LIBOMPTARGET_TESTED_PLUGINS}" PARENT_SCOPE) - endif() -- -diff --git a/libomptarget/plugins/common/elf_common/CMakeLists.txt b/libomptarget/plugins/common/elf_common/CMakeLists.txt -index 9ea2926..b3fb758 100644 ---- a/libomptarget/plugins/common/elf_common/CMakeLists.txt -+++ b/libomptarget/plugins/common/elf_common/CMakeLists.txt -@@ -16,9 +16,7 @@ add_library(elf_common OBJECT elf_common.cpp) - set_property(TARGET elf_common PROPERTY POSITION_INDEPENDENT_CODE ON) - llvm_update_compile_flags(elf_common) - set(LINK_LLVM_LIBS LLVMBinaryFormat LLVMObject LLVMSupport) --if (LLVM_LINK_LLVM_DYLIB) -- set(LINK_LLVM_LIBS LLVM) --endif() -+# Link LLVM static libraries to avoid dependency on shared LLVM libraries. - target_link_libraries(elf_common INTERFACE ${LINK_LLVM_LIBS}) - add_dependencies(elf_common ${LINK_LLVM_LIBS}) - -diff --git a/libomptarget/plugins/cuda/CMakeLists.txt b/libomptarget/plugins/cuda/CMakeLists.txt -index 46e04c3..825e273 100644 ---- a/libomptarget/plugins/cuda/CMakeLists.txt -+++ b/libomptarget/plugins/cuda/CMakeLists.txt -@@ -40,7 +40,7 @@ endif() - if (LIBOMPTARGET_CAN_LINK_LIBCUDA AND NOT LIBOMPTARGET_FORCE_DLOPEN_LIBCUDA) - libomptarget_say("Building CUDA plugin linked against libcuda") - include_directories(${LIBOMPTARGET_DEP_CUDA_INCLUDE_DIRS}) -- add_llvm_library(omptarget.rtl.cuda SHARED -+ add_llvm_library(omptarget.rtl.cuda SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - - src/rtl.cpp - -@@ -64,7 +64,7 @@ else() - libomptarget_say("Building CUDA plugin for dlopened libcuda") - include_directories(dynamic_cuda) - add_llvm_library(omptarget.rtl.cuda -- SHARED -+ SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - - src/rtl.cpp - dynamic_cuda/cuda.cpp -diff --git a/libomptarget/plugins/ve/CMakeLists.txt b/libomptarget/plugins/ve/CMakeLists.txt -index 5aded32..4a81583 100644 ---- a/libomptarget/plugins/ve/CMakeLists.txt -+++ b/libomptarget/plugins/ve/CMakeLists.txt -@@ -24,7 +24,7 @@ if(${LIBOMPTARGET_DEP_VEO_FOUND}) - add_definitions("-DTARGET_ELF_ID=${elf_machine_id}") - - add_llvm_library("omptarget.rtl.${tmachine_libname}" -- SHARED -+ SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - ${CMAKE_CURRENT_SOURCE_DIR}/src/rtl.cpp - - ADDITIONAL_HEADER_DIRS -diff --git a/libomptarget/src/CMakeLists.txt b/libomptarget/src/CMakeLists.txt -index 071ec61..98b48ac 100644 ---- a/libomptarget/src/CMakeLists.txt -+++ b/libomptarget/src/CMakeLists.txt -@@ -12,8 +12,9 @@ - - libomptarget_say("Building offloading runtime library libomptarget.") - -+# Link LLVM statically to avoid dependency on dynamic libLLVM. - add_llvm_library(omptarget -- SHARED -+ SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - - api.cpp - device.cpp diff --git a/src/numba/openmp/libs/openmp/patches/15.0.7/0003-Disable-opaque-pointers-DeviceRTL-bitcode.patch b/src/numba/openmp/libs/openmp/patches/15.0.7/0003-Disable-opaque-pointers-DeviceRTL-bitcode.patch deleted file mode 100644 index 0f39bf28d3c4..000000000000 --- a/src/numba/openmp/libs/openmp/patches/15.0.7/0003-Disable-opaque-pointers-DeviceRTL-bitcode.patch +++ /dev/null @@ -1,12 +0,0 @@ -diff --git a/libomptarget/DeviceRTL/CMakeLists.txt b/libomptarget/DeviceRTL/CMakeLists.txt -index ce66214..04c9dd5 100644 ---- a/libomptarget/DeviceRTL/CMakeLists.txt -+++ b/libomptarget/DeviceRTL/CMakeLists.txt -@@ -127,6 +127,7 @@ set(bc_flags -c -emit-llvm -std=c++17 -fvisibility=hidden - -fopenmp -fopenmp-cuda-mode - -I${include_directory} - -I${devicertl_base_directory}/../include -+ -Xclang -no-opaque-pointers - ${LIBOMPTARGET_LLVM_INCLUDE_DIRS_DEVICERTL} - ) - diff --git a/src/numba/openmp/libs/openmp/patches/16.0.6/0001-Load-plugins-from-install-directory.patch b/src/numba/openmp/libs/openmp/patches/16.0.6/0001-Load-plugins-from-install-directory.patch deleted file mode 100644 index 2f7446d9674a..000000000000 --- a/src/numba/openmp/libs/openmp/patches/16.0.6/0001-Load-plugins-from-install-directory.patch +++ /dev/null @@ -1,53 +0,0 @@ -diff --git a/libomptarget/CMakeLists.txt b/libomptarget/CMakeLists.txt -index bc6e615..2c41595 100644 ---- a/libomptarget/CMakeLists.txt -+++ b/libomptarget/CMakeLists.txt -@@ -24,6 +24,19 @@ set(CMAKE_ARCHIVE_OUTPUT_DIRECTORY ${LIBOMPTARGET_LIBRARY_DIR}) - set(CMAKE_LIBRARY_OUTPUT_DIRECTORY ${LIBOMPTARGET_LIBRARY_DIR}) - set(CMAKE_RUNTIME_OUTPUT_DIRECTORY ${LIBOMPTARGET_LIBRARY_DIR}) - -+# Define plugin install directory for runtime plugin loading. Prefer -+# standardized install libdir when available, fall back to ${CMAKE_INSTALL_PREFIX}/lib. -+if(NOT DEFINED LIBOMPTARGET_PLUGIN_DIR) -+ if(DEFINED CMAKE_INSTALL_LIBDIR) -+ set(LIBOMPTARGET_PLUGIN_DIR "${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}" CACHE PATH "Path where libomptarget plugins are installed") -+ else() -+ set(LIBOMPTARGET_PLUGIN_DIR "${CMAKE_INSTALL_PREFIX}/lib" CACHE PATH "Path where libomptarget plugins are installed") -+ endif() -+endif() -+ -+# Expose the plugin directory to sources as a compile-time definition. -+add_definitions(-DLIBOMPTARGET_PLUGIN_DIR=\"${LIBOMPTARGET_PLUGIN_DIR}\") -+ - # Message utilities. - include(LibomptargetUtils) - -diff --git a/libomptarget/src/rtl.cpp b/libomptarget/src/rtl.cpp -index 230a829..ff7e704 100644 ---- a/libomptarget/src/rtl.cpp -+++ b/libomptarget/src/rtl.cpp -@@ -118,12 +118,22 @@ void RTLsTy::loadRTLs() { - - bool RTLsTy::attemptLoadRTL(const std::string &RTLName, RTLInfoTy &RTL) { - const char *Name = RTLName.c_str(); -- - DP("Loading library '%s'...\n", Name); - -+ // First, try to load the plugin from the configured plugin directory -+ // (LIBOMPTARGET_PLUGIN_DIR), falling back to the system library lookup. - std::string ErrMsg; -+ std::string PluginPath = std::string(LIBOMPTARGET_PLUGIN_DIR) + "/" + RTLName; - auto DynLibrary = std::make_unique( -- sys::DynamicLibrary::getPermanentLibrary(Name, &ErrMsg)); -+ sys::DynamicLibrary::getPermanentLibrary(PluginPath.c_str(), &ErrMsg)); -+ -+ if (!DynLibrary->isValid()) { -+ DP("Unable to load library from plugin dir: %s\n", ErrMsg.c_str()); -+ // Try default lookup (PATH/LD_LIBRARY_PATH/etc.) -+ ErrMsg.clear(); -+ DynLibrary = std::make_unique( -+ sys::DynamicLibrary::getPermanentLibrary(Name, &ErrMsg)); -+ } - - if (!DynLibrary->isValid()) { - // Library does not exist or cannot be found. diff --git a/src/numba/openmp/libs/openmp/patches/16.0.6/0002-Link-LLVM-statically.patch b/src/numba/openmp/libs/openmp/patches/16.0.6/0002-Link-LLVM-statically.patch deleted file mode 100644 index 782e63c6910e..000000000000 --- a/src/numba/openmp/libs/openmp/patches/16.0.6/0002-Link-LLVM-statically.patch +++ /dev/null @@ -1,218 +0,0 @@ -diff --git a/libomptarget/plugins-nextgen/CMakeLists.txt b/libomptarget/plugins-nextgen/CMakeLists.txt -index 95e359c..8946fe8 100644 ---- a/libomptarget/plugins-nextgen/CMakeLists.txt -+++ b/libomptarget/plugins-nextgen/CMakeLists.txt -@@ -37,7 +37,7 @@ if(CMAKE_SYSTEM_PROCESSOR MATCHES "${tmachine}$") - add_definitions("-DLIBOMPTARGET_NEXTGEN_GENERIC_PLUGIN_TRIPLE=${tmachine}") - - add_llvm_library("omptarget.rtl.${tmachine_libname}.nextgen" -- SHARED -+ SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - - ${CMAKE_CURRENT_SOURCE_DIR}/../generic-elf-64bit/src/rtl.cpp - -diff --git a/libomptarget/plugins-nextgen/amdgpu/CMakeLists.txt b/libomptarget/plugins-nextgen/amdgpu/CMakeLists.txt -index 8f234ee..435a8cd 100644 ---- a/libomptarget/plugins-nextgen/amdgpu/CMakeLists.txt -+++ b/libomptarget/plugins-nextgen/amdgpu/CMakeLists.txt -@@ -66,7 +66,7 @@ else() - set(LDFLAGS_UNDEFINED "-Wl,-z,defs") - endif() - --add_llvm_library(omptarget.rtl.amdgpu.nextgen SHARED -+add_llvm_library(omptarget.rtl.amdgpu.nextgen SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - src/rtl.cpp - ${LIBOMPTARGET_EXTRA_SOURCE} - -diff --git a/libomptarget/plugins-nextgen/common/PluginInterface/CMakeLists.txt b/libomptarget/plugins-nextgen/common/PluginInterface/CMakeLists.txt -index 91d64f4..db16105 100644 ---- a/libomptarget/plugins-nextgen/common/PluginInterface/CMakeLists.txt -+++ b/libomptarget/plugins-nextgen/common/PluginInterface/CMakeLists.txt -@@ -24,36 +24,33 @@ endforeach() - # This is required when using LLVM libraries. - llvm_update_compile_flags(PluginInterface) - --if (LLVM_LINK_LLVM_DYLIB) -- set(llvm_libs LLVM) --else() -- llvm_map_components_to_libnames(llvm_libs -- ${LLVM_TARGETS_TO_BUILD} -- AggressiveInstCombine -- Analysis -- BinaryFormat -- BitReader -- BitWriter -- CodeGen -- Core -- Extensions -- InstCombine -- Instrumentation -- IPO -- IRReader -- Linker -- MC -- Object -- Passes -- Remarks -- ScalarOpts -- Support -- Target -- TargetParser -- TransformUtils -- Vectorize -- ) --endif() -+# Link LLVM libraries statically. -+llvm_map_components_to_libnames(llvm_libs -+ ${LLVM_TARGETS_TO_BUILD} -+ AggressiveInstCombine -+ Analysis -+ BinaryFormat -+ BitReader -+ BitWriter -+ CodeGen -+ Core -+ Extensions -+ InstCombine -+ Instrumentation -+ IPO -+ IRReader -+ Linker -+ MC -+ Object -+ Passes -+ Remarks -+ ScalarOpts -+ Support -+ Target -+ TargetParser -+ TransformUtils -+ Vectorize -+) - - target_link_libraries(PluginInterface - PUBLIC -diff --git a/libomptarget/plugins-nextgen/cuda/CMakeLists.txt b/libomptarget/plugins-nextgen/cuda/CMakeLists.txt -index da19ec3..c2d6279 100644 ---- a/libomptarget/plugins-nextgen/cuda/CMakeLists.txt -+++ b/libomptarget/plugins-nextgen/cuda/CMakeLists.txt -@@ -41,7 +41,7 @@ endif() - if (LIBOMPTARGET_CAN_LINK_LIBCUDA AND NOT LIBOMPTARGET_FORCE_DLOPEN_LIBCUDA) - libomptarget_say("Building CUDA NextGen plugin linked against libcuda") - include_directories(${LIBOMPTARGET_DEP_CUDA_INCLUDE_DIRS}) -- add_llvm_library(omptarget.rtl.cuda.nextgen SHARED -+ add_llvm_library(omptarget.rtl.cuda.nextgen SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - - src/rtl.cpp - -@@ -64,7 +64,7 @@ else() - libomptarget_say("Building CUDA NextGen plugin for dlopened libcuda") - include_directories(../../plugins/cuda/dynamic_cuda) - add_llvm_library(omptarget.rtl.cuda.nextgen -- SHARED -+ SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - - src/rtl.cpp - ../../plugins/cuda/dynamic_cuda/cuda.cpp -diff --git a/libomptarget/plugins/CMakeLists.txt b/libomptarget/plugins/CMakeLists.txt -index 005a372..fef1aec 100644 ---- a/libomptarget/plugins/CMakeLists.txt -+++ b/libomptarget/plugins/CMakeLists.txt -@@ -30,7 +30,7 @@ if(CMAKE_SYSTEM_PROCESSOR MATCHES "${tmachine}$") - add_definitions("-DTARGET_ELF_ID=${elf_machine_id}") - - add_llvm_library("omptarget.rtl.${tmachine_libname}" -- SHARED -+ SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - - ${CMAKE_CURRENT_SOURCE_DIR}/../generic-elf-64bit/src/rtl.cpp - -@@ -90,4 +90,3 @@ add_subdirectory(remote) - # Make sure the parent scope can see the plugins that will be created. - set(LIBOMPTARGET_SYSTEM_TARGETS "${LIBOMPTARGET_SYSTEM_TARGETS}" PARENT_SCOPE) - set(LIBOMPTARGET_TESTED_PLUGINS "${LIBOMPTARGET_TESTED_PLUGINS}" PARENT_SCOPE) -- -diff --git a/libomptarget/plugins/amdgpu/CMakeLists.txt b/libomptarget/plugins/amdgpu/CMakeLists.txt -index 1619f1e..299a25d 100644 ---- a/libomptarget/plugins/amdgpu/CMakeLists.txt -+++ b/libomptarget/plugins/amdgpu/CMakeLists.txt -@@ -61,7 +61,7 @@ else() - set(LDFLAGS_UNDEFINED "-Wl,-z,defs") - endif() - --add_llvm_library(omptarget.rtl.amdgpu SHARED -+add_llvm_library(omptarget.rtl.amdgpu SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - impl/impl.cpp - impl/interop_hsa.cpp - impl/data.cpp -@@ -121,4 +121,3 @@ else() - libomptarget_say("Not generating amdgcn test targets as libhsa is not linkable") - return() - endif() -- -diff --git a/libomptarget/plugins/common/elf_common/CMakeLists.txt b/libomptarget/plugins/common/elf_common/CMakeLists.txt -index 54d28bc..b615359 100644 ---- a/libomptarget/plugins/common/elf_common/CMakeLists.txt -+++ b/libomptarget/plugins/common/elf_common/CMakeLists.txt -@@ -18,11 +18,8 @@ add_library(elf_common OBJECT elf_common.cpp ELFSymbols.cpp) - # This is required when using LLVM libraries. - llvm_update_compile_flags(elf_common) - --if (LLVM_LINK_LLVM_DYLIB) -- set(llvm_libs LLVM) --else() -- llvm_map_components_to_libnames(llvm_libs BinaryFormat Object Support) --endif() -+# Link LLVM libraries statically. -+llvm_map_components_to_libnames(llvm_libs BinaryFormat Object Support) - - target_link_libraries(elf_common PUBLIC ${llvm_libs} ${OPENMP_PTHREAD_LIB}) - -diff --git a/libomptarget/plugins/cuda/CMakeLists.txt b/libomptarget/plugins/cuda/CMakeLists.txt -index 6d0b767..97e4bba 100644 ---- a/libomptarget/plugins/cuda/CMakeLists.txt -+++ b/libomptarget/plugins/cuda/CMakeLists.txt -@@ -37,7 +37,7 @@ endif() - if (LIBOMPTARGET_CAN_LINK_LIBCUDA AND NOT LIBOMPTARGET_FORCE_DLOPEN_LIBCUDA) - libomptarget_say("Building CUDA plugin linked against libcuda") - include_directories(${LIBOMPTARGET_DEP_CUDA_INCLUDE_DIRS}) -- add_llvm_library(omptarget.rtl.cuda SHARED -+ add_llvm_library(omptarget.rtl.cuda SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - - src/rtl.cpp - -@@ -63,7 +63,7 @@ else() - libomptarget_say("Building CUDA plugin for dlopened libcuda") - include_directories(dynamic_cuda) - add_llvm_library(omptarget.rtl.cuda -- SHARED -+ SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - - src/rtl.cpp - dynamic_cuda/cuda.cpp -diff --git a/libomptarget/plugins/ve/CMakeLists.txt b/libomptarget/plugins/ve/CMakeLists.txt -index a949031..318f5e4 100644 ---- a/libomptarget/plugins/ve/CMakeLists.txt -+++ b/libomptarget/plugins/ve/CMakeLists.txt -@@ -24,7 +24,7 @@ if(${LIBOMPTARGET_DEP_VEO_FOUND}) - add_definitions("-DTARGET_ELF_ID=${elf_machine_id}") - - add_llvm_library("omptarget.rtl.${tmachine_libname}" -- SHARED -+ SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - ${CMAKE_CURRENT_SOURCE_DIR}/src/rtl.cpp - - ADDITIONAL_HEADER_DIRS -diff --git a/libomptarget/src/CMakeLists.txt b/libomptarget/src/CMakeLists.txt -index 2a6cd93..1e24f73 100644 ---- a/libomptarget/src/CMakeLists.txt -+++ b/libomptarget/src/CMakeLists.txt -@@ -13,7 +13,7 @@ - libomptarget_say("Building offloading runtime library libomptarget.") - - add_llvm_library(omptarget -- SHARED -+ SHARED DISABLE_LLVM_LINK_LLVM_DYLIB - - api.cpp - device.cpp diff --git a/src/numba/openmp/libs/openmp/patches/20.1.8/0001-Enable-standalone-build.patch b/src/numba/openmp/libs/openmp/patches/20.1.8/0001-Enable-standalone-build.patch deleted file mode 100644 index 9f73e9d3fb0c..000000000000 --- a/src/numba/openmp/libs/openmp/patches/20.1.8/0001-Enable-standalone-build.patch +++ /dev/null @@ -1,13 +0,0 @@ -diff --git a/offload/CMakeLists.txt b/offload/CMakeLists.txt -index f6e894d39..9096d4ed5 100644 ---- a/offload/CMakeLists.txt -+++ b/offload/CMakeLists.txt -@@ -4,7 +4,7 @@ - cmake_minimum_required(VERSION 3.20.0) - set(LLVM_SUBPROJECT_TITLE "liboffload") - --if ("${CMAKE_SOURCE_DIR}" STREQUAL "${CMAKE_CURRENT_SOURCE_DIR}") -+if (OPENMP_STANDALONE_BUILD OR "${CMAKE_SOURCE_DIR}" STREQUAL "${CMAKE_CURRENT_SOURCE_DIR}") - set(OPENMP_STANDALONE_BUILD TRUE) - project(offload C CXX ASM) - else() diff --git a/src/numba/openmp/libs/openmp/patches/20.1.8/0003-Do-not-build-liboffload.patch b/src/numba/openmp/libs/openmp/patches/20.1.8/0003-Do-not-build-liboffload.patch deleted file mode 100644 index bc64017d94cf..000000000000 --- a/src/numba/openmp/libs/openmp/patches/20.1.8/0003-Do-not-build-liboffload.patch +++ /dev/null @@ -1,12 +0,0 @@ -diff --git a/offload/CMakeLists.txt b/offload/CMakeLists.txt -index 9096d4e..a0aff92 100644 ---- a/offload/CMakeLists.txt -+++ b/offload/CMakeLists.txt -@@ -380,7 +380,6 @@ add_subdirectory(tools) - add_subdirectory(src) - - add_subdirectory(tools/offload-tblgen) --add_subdirectory(liboffload) - - # Add tests. - add_subdirectory(test) diff --git a/src/numba/openmp/libs/openmp/patches/20.1.8/0002-Link-statically-LLVM.patch b/src/numba/openmp/libs/openmp/patches/22.1.8/0001-Link-statically-LLVM.patch similarity index 63% rename from src/numba/openmp/libs/openmp/patches/20.1.8/0002-Link-statically-LLVM.patch rename to src/numba/openmp/libs/openmp/patches/22.1.8/0001-Link-statically-LLVM.patch index 43a9aaff0d74..708a11da9bc4 100644 --- a/src/numba/openmp/libs/openmp/patches/20.1.8/0002-Link-statically-LLVM.patch +++ b/src/numba/openmp/libs/openmp/patches/22.1.8/0001-Link-statically-LLVM.patch @@ -1,8 +1,20 @@ +diff --git a/offload/libomptarget/CMakeLists.txt b/offload/libomptarget/CMakeLists.txt +index 93e684e..22b35a2 100644 +--- a/offload/libomptarget/CMakeLists.txt ++++ b/offload/libomptarget/CMakeLists.txt +@@ -8,6 +8,7 @@ endif() + + add_llvm_library(omptarget + SHARED ++ DISABLE_LLVM_LINK_LLVM_DYLIB + + device.cpp + interface.cpp diff --git a/offload/plugins-nextgen/CMakeLists.txt b/offload/plugins-nextgen/CMakeLists.txt -index 9b5b12bea..78dde405b 100644 +index a72befd..c83ebb8 100644 --- a/offload/plugins-nextgen/CMakeLists.txt +++ b/offload/plugins-nextgen/CMakeLists.txt -@@ -3,6 +3,7 @@ set(common_dir ${CMAKE_CURRENT_SOURCE_DIR}/common) +@@ -4,6 +4,7 @@ set(common_bin_dir ${CMAKE_CURRENT_BINARY_DIR}/common) add_subdirectory(common) function(add_target_library target_name lib_name) add_llvm_library(${target_name} STATIC @@ -10,15 +22,3 @@ index 9b5b12bea..78dde405b 100644 LINK_COMPONENTS AggressiveInstCombine Analysis -diff --git a/offload/src/CMakeLists.txt b/offload/src/CMakeLists.txt -index c5f5d902f..ca5135b13 100644 ---- a/offload/src/CMakeLists.txt -+++ b/offload/src/CMakeLists.txt -@@ -8,6 +8,7 @@ endif() - - add_llvm_library(omptarget - SHARED -+ DISABLE_LLVM_LINK_LLVM_DYLIB - - device.cpp - interface.cpp diff --git a/src/numba/openmp/libs/openmp/patches/22.1.8/0002-Do-not-build-liboffload.patch b/src/numba/openmp/libs/openmp/patches/22.1.8/0002-Do-not-build-liboffload.patch new file mode 100644 index 000000000000..2b47e7614712 --- /dev/null +++ b/src/numba/openmp/libs/openmp/patches/22.1.8/0002-Do-not-build-liboffload.patch @@ -0,0 +1,12 @@ +diff --git a/offload/CMakeLists.txt b/offload/CMakeLists.txt +index 81e246d..e94d309 100644 +--- a/offload/CMakeLists.txt ++++ b/offload/CMakeLists.txt +@@ -391,7 +391,6 @@ add_subdirectory(docs) + # Build target agnostic offloading library. + add_subdirectory(libomptarget) + +-add_subdirectory(liboffload) + + # Add tests. + if(OFFLOAD_INCLUDE_TESTS) diff --git a/src/numba/openmp/libs/openmp/patches/20.1.8/0004-Add-getter-for-device-info.patch b/src/numba/openmp/libs/openmp/patches/22.1.8/0003-Add-getter-for-device-info.patch similarity index 53% rename from src/numba/openmp/libs/openmp/patches/20.1.8/0004-Add-getter-for-device-info.patch rename to src/numba/openmp/libs/openmp/patches/22.1.8/0003-Add-getter-for-device-info.patch index 5b6edcd9d598..fa8fe63bec67 100644 --- a/src/numba/openmp/libs/openmp/patches/20.1.8/0004-Add-getter-for-device-info.patch +++ b/src/numba/openmp/libs/openmp/patches/22.1.8/0003-Add-getter-for-device-info.patch @@ -1,5 +1,3 @@ -diff --git a/offload/include/device.h b/offload/include/device.h -index 3132d35..8e70851 100644 --- a/offload/include/device.h +++ b/offload/include/device.h @@ -18,6 +18,7 @@ @@ -10,7 +8,7 @@ index 3132d35..8e70851 100644 #include #include #include -@@ -122,7 +123,7 @@ struct DeviceTy { +@@ -128,7 +129,7 @@ int32_t queryAsync(AsyncInfoTy &AsyncInfo); /// Calls the corresponding print device info function in the plugin. @@ -19,48 +17,75 @@ index 3132d35..8e70851 100644 /// Event related interfaces. /// { -diff --git a/offload/plugins-nextgen/common/include/PluginInterface.h b/offload/plugins-nextgen/common/include/PluginInterface.h -index eb266e8..241d132 100644 --- a/offload/plugins-nextgen/common/include/PluginInterface.h +++ b/offload/plugins-nextgen/common/include/PluginInterface.h -@@ -158,10 +158,13 @@ public: - const std::deque &getQueue() const { return Queue; } +@@ -247,43 +247,46 @@ + } - /// Print all info entries added to the queue. + /// Print all info entries in the tree - void print() const { + void print(llvm::raw_ostream *OS = nullptr) const { - // We print four spances for each level. - constexpr uint64_t IndentSize = 4; - + // Choose output stream: provided one or llvm::outs(). + llvm::raw_ostream &Out = OS ? *OS : llvm::outs(); + - // Find the maximum key length (level + key) to compute the individual - // indentation of each entry. - uint64_t MaxKeySize = 0; -@@ -178,9 +181,9 @@ public: + // Fake an additional indent so that values are offset from the keys +- doPrint(0, maxKeySize(1)); ++ doPrint(0, maxKeySize(1), Out); + } + + private: +- void doPrint(int Level, uint64_t MaxKeySize) const { ++ void doPrint(int Level, uint64_t MaxKeySize, llvm::raw_ostream &Out) const { + if (Key.size()) { + // Compute the indentations for the current entry. + uint64_t KeyIndentSize = Level * IndentSize; uint64_t ValIndentSize = - MaxKeySize - (Entry.Key.size() + KeyIndentSize) + IndentSize; - -- llvm::outs() << std::string(KeyIndentSize, ' ') << Entry.Key -- << std::string(ValIndentSize, ' ') << Entry.Value -- << (Entry.Units.empty() ? "" : " ") << Entry.Units << "\n"; -+ Out << std::string(KeyIndentSize, ' ') << Entry.Key -+ << std::string(ValIndentSize, ' ') << Entry.Value -+ << (Entry.Units.empty() ? "" : " ") << Entry.Units << "\n"; + MaxKeySize - (Key.size() + KeyIndentSize) + IndentSize; + +- llvm::outs() << std::string(KeyIndentSize, ' ') << Key ++ Out << std::string(KeyIndentSize, ' ') << Key + << std::string(ValIndentSize, ' '); + std::visit( +- [](auto &&V) { ++ [&Out](auto &&V) { + using T = std::decay_t; + if constexpr (std::is_same_v) +- llvm::outs() << V; ++ Out << V; + else if constexpr (std::is_same_v) +- llvm::outs() << (V ? "Yes" : "No"); ++ Out << (V ? "Yes" : "No"); + else if constexpr (std::is_same_v) +- llvm::outs() << V; ++ Out << V; + else if constexpr (std::is_same_v) { + // Do nothing + } else + static_assert(false, "doPrint visit not exhaustive"); + }, + Value); +- llvm::outs() << (Units.empty() ? "" : " ") << Units << "\n"; ++ Out << (Units.empty() ? "" : " ") << Units << "\n"; } + + // Print children + if (Children) + for (const auto &Entry : *Children) +- Entry.doPrint(Level + 1, MaxKeySize); ++ Entry.doPrint(Level + 1, MaxKeySize, Out); } - }; -@@ -866,7 +869,7 @@ struct GenericDeviceTy : public DeviceAllocatorTy { - virtual Error syncEventImpl(void *EventPtr) = 0; + + // Recursively calculates the maximum width of each key, including indentation +@@ -999,7 +1002,7 @@ + virtual Expected obtainInfoImpl() = 0; /// Print information about the device. - Error printInfo(); + Error printInfo(llvm::raw_ostream *OS = nullptr); - virtual Error obtainInfoImpl(InfoQueueTy &Info) = 0; - /// Getters of the grid values. -@@ -1295,7 +1298,7 @@ public: + /// Return true if the device has work that is either queued or currently + /// running +@@ -1499,7 +1502,7 @@ int32_t query_async(int32_t DeviceId, __tgt_async_info *AsyncInfoPtr); /// Prints information about the given devices supported by the plugin. @@ -69,29 +94,27 @@ index eb266e8..241d132 100644 /// Creates an event in the given plugin if supported. int32_t create_event(int32_t DeviceId, void **EventPtr); -diff --git a/offload/plugins-nextgen/common/src/PluginInterface.cpp b/offload/plugins-nextgen/common/src/PluginInterface.cpp -index 57672b0..b8c84e6 100644 --- a/offload/plugins-nextgen/common/src/PluginInterface.cpp +++ b/offload/plugins-nextgen/common/src/PluginInterface.cpp -@@ -1546,7 +1546,7 @@ Error GenericDeviceTy::initDeviceInfo(__tgt_device_info *DeviceInfo) { - return initDeviceInfoImpl(DeviceInfo); +@@ -1451,7 +1451,7 @@ + return InfoOrErr; } -Error GenericDeviceTy::printInfo() { +Error GenericDeviceTy::printInfo(llvm::raw_ostream *OS) { - InfoQueueTy InfoQueue; + auto InfoOrErr = obtainInfo(); // Get the vendor-specific info entries describing the device properties. -@@ -1554,7 +1554,7 @@ Error GenericDeviceTy::printInfo() { +@@ -1459,7 +1459,7 @@ return Err; // Print all info entries. -- InfoQueue.print(); -+ InfoQueue.print(OS); +- InfoOrErr->print(); ++ InfoOrErr->print(OS); return Plugin::success(); } -@@ -2036,8 +2036,8 @@ int32_t GenericPluginTy::query_async(int32_t DeviceId, +@@ -1974,8 +1974,8 @@ return OFFLOAD_SUCCESS; } @@ -99,13 +122,11 @@ index 57672b0..b8c84e6 100644 - if (auto Err = getDevice(DeviceId).printInfo()) +void GenericPluginTy::print_device_info(int32_t DeviceId, raw_ostream *OS) { + if (auto Err = getDevice(DeviceId).printInfo(OS)) - REPORT("Failure to print device %d info: %s\n", DeviceId, - toString(std::move(Err)).data()); + REPORT() << "Failure to print device " << DeviceId + << " info: " << toString(std::move(Err)); } -diff --git a/offload/src/device.cpp b/offload/src/device.cpp -index 943c778..e2640f2 100644 ---- a/offload/src/device.cpp -+++ b/offload/src/device.cpp +--- a/offload/libomptarget/device.cpp ++++ b/offload/libomptarget/device.cpp @@ -24,6 +24,7 @@ #include "Shared/EnvironmentVar.h" @@ -114,7 +135,7 @@ index 943c778..e2640f2 100644 #include #include -@@ -221,8 +222,8 @@ int32_t DeviceTy::launchKernel(void *TgtEntryPtr, void **TgtVarsPtr, +@@ -345,8 +346,8 @@ } // Run region on device @@ -125,11 +146,9 @@ index 943c778..e2640f2 100644 return true; } -diff --git a/offload/src/exports b/offload/src/exports -index 2406776..f43f9a5 100644 ---- a/offload/src/exports -+++ b/offload/src/exports -@@ -62,6 +62,7 @@ VERS1.0 { +--- a/offload/libomptarget/exports ++++ b/offload/libomptarget/exports +@@ -66,6 +66,7 @@ llvm_omp_target_unlock_mem; __tgt_set_info_flag; __tgt_print_device_info; @@ -137,10 +156,8 @@ index 2406776..f43f9a5 100644 omp_get_interop_ptr; omp_get_interop_str; omp_get_interop_int; -diff --git a/offload/src/interface.cpp b/offload/src/interface.cpp -index ad84a43..ed25c85 100644 ---- a/offload/src/interface.cpp -+++ b/offload/src/interface.cpp +--- a/offload/libomptarget/interface.cpp ++++ b/offload/libomptarget/interface.cpp @@ -25,6 +25,7 @@ #include "Utils/ExponentialBackoff.h" @@ -149,16 +166,7 @@ index ad84a43..ed25c85 100644 #include #include -@@ -294,7 +295,7 @@ static KernelArgsTy *upgradeKernelArgs(KernelArgsTy *KernelArgs, - - // FIXME: This is a WA to "calibrate" the bad work done in the front end. - // Delete this ugly code after the front end emits proper values. -- auto CorrectMultiDim = [](uint32_t(&Val)[3]) { -+ auto CorrectMultiDim = [](uint32_t (&Val)[3]) { - if (Val[1] == 0) - Val[1] = 1; - if (Val[2] == 0) -@@ -512,6 +513,21 @@ EXTERN int __tgt_print_device_info(int64_t DeviceId) { +@@ -530,6 +531,21 @@ return DeviceOrErr->printDeviceInfo(); } diff --git a/src/numba/openmp/libs/pass/CGIntrinsicsOpenMP.cpp b/src/numba/openmp/libs/pass/CGIntrinsicsOpenMP.cpp index 489ccb0145c3..62934eead366 100644 --- a/src/numba/openmp/libs/pass/CGIntrinsicsOpenMP.cpp +++ b/src/numba/openmp/libs/pass/CGIntrinsicsOpenMP.cpp @@ -73,13 +73,9 @@ CallInst *checkCreateCall(IRBuilderBase &Builder, FunctionCallee &Fn, // We retrieve the type from the DSAValueMap to store the pointee type for // opaque pointer values. Type *getPointeeType(DSAValueMapTy &DSAValueMap, Value *V) { -#if LLVM_VERSION_MAJOR <= 15 - return V->getType()->getPointerElementType(); -#else // assert(V->getType()->isOpaquePointerTy() && "Expected opaque pointer // type"); assert(V->getType()->isPointerTy() && "Expected pointer type"); -#endif if (auto *Alloca = dyn_cast(V)) { return Alloca->getAllocatedType(); @@ -103,10 +99,7 @@ using namespace iomp::helpers; InsertPointTy CGIntrinsicsOpenMP::emitReductionsHost( const OpenMPIRBuilder::LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef ReductionInfos) { -// If targeting the host runtime, use the OpenMP IR builder. -#if LLVM_VERSION_MAJOR <= 16 - return OMPBuilder.createReductions(Loc, AllocaIP, ReductionInfos); -#else + // If targeting the host runtime, use the OpenMP IR builder. // TODO: look into the ByRef parameter. SmallVector IsByRef(ReductionInfos.size(), false); auto IPOrError = @@ -115,8 +108,6 @@ InsertPointTy CGIntrinsicsOpenMP::emitReductionsHost( FATAL_ERROR("Error in createReductions:" + toString(std::move(E))); return *IPOrError; - -#endif } InsertPointTy CGIntrinsicsOpenMP::emitReductionsDevice( @@ -167,11 +158,6 @@ InsertPointTy CGIntrinsicsOpenMP::emitReductionsDevice( assert(RI.Variable->getType()->isPointerTy() && "Expected variables to be pointers"); -#if LLVM_VERSION_MAJOR <= 16 - OMPBuilder.Builder.restoreIP( - RI.AtomicReductionGen(OMPBuilder.Builder.saveIP(), RI.ElementType, - RI.Variable, RI.PrivateVariable)); -#else auto IPOrErr = RI.AtomicReductionGen(OMPBuilder.Builder.saveIP(), RI.ElementType, RI.Variable, RI.PrivateVariable); @@ -179,7 +165,6 @@ InsertPointTy CGIntrinsicsOpenMP::emitReductionsDevice( FATAL_ERROR("Error in AtomicReductionGen: " + toString(std::move(E))); OMPBuilder.Builder.restoreIP(*IPOrErr); -#endif } // Add terminator branch to the continuation block. @@ -234,9 +219,7 @@ OutlinedInfoStruct CGIntrinsicsOpenMP::createOutlinedFunction( /* AssumptionCache */ nullptr, /* AllowVarArgs */ true, /* AllowAlloca */ true, -#if LLVM_VERSION_MAJOR >= 15 /* AllocationBlock */ nullptr, -#endif /* Suffix */ "."); // Find inputs to, outputs from the code region. @@ -538,6 +521,15 @@ OutlinedInfoStruct CGIntrinsicsOpenMP::createOutlinedFunction( } CGIntrinsicsOpenMP::CGIntrinsicsOpenMP(Module &M) : OMPBuilder(M), M(M) { + // OpenMPIRBuilder methods such as createReductions read the config, which + // is unset by default. + bool IsGPU = isOpenMPDeviceRuntime(); + OMPBuilder.Config = OpenMPIRBuilderConfig( + /* IsTargetDevice */ IsGPU, IsGPU, /* OpenMPOffloadMandatory */ false, + /* HasRequiresReverseOffload */ false, + /* HasRequiresUnifiedAddress */ false, + /* HasRequiresUnifiedSharedMemory */ false, + /* HasRequiresDynamicAllocators */ false); OMPBuilder.initialize(); TgtOffloadEntryTy = StructType::create({OMPBuilder.Int8Ptr, @@ -763,7 +755,7 @@ void CGIntrinsicsOpenMP::emitOMPParallelDeviceRuntime( << *OutlinedWrapperFn << "=== End of Dump OutlinedWrapper\n"); - // Setup the call to kmpc_parallel_51 + // Setup the call to kmpc_parallel_60 BBEntry->getTerminator()->eraseFromParent(); OpenMPIRBuilder::LocationDescription Loc( InsertPointTy(BBEntry, BBEntry->end()), DL); @@ -783,7 +775,7 @@ void CGIntrinsicsOpenMP::emitOMPParallelDeviceRuntime( // TODO: Re-think allocas, move to start of caller. If the caller is outlined // in an outer OpenMP region, dot naming ensures captured_var_addrs is a // private value, since it's only used for setting up the call to - // kmpc_parallel_51. + // kmpc_parallel_60. auto PrevIP = OMPBuilder.Builder.saveIP(); InsertPointTy AllocaIP(&Fn->getEntryBlock(), Fn->getEntryBlock().getFirstInsertionPt()); @@ -851,8 +843,8 @@ void CGIntrinsicsOpenMP::emitOMPParallelDeviceRuntime( assert(NumThreads && "Expected non-null NumThreads"); - FunctionCallee KmpcParallel51 = - OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_parallel_51); + FunctionCallee KmpcParallel = + OMPBuilder.getOrCreateRuntimeFunction(M, OMPRTL___kmpc_parallel_60); // Set proc_bind to -1 by default as it is unused. assert(Ident && "Expected non-null Ident"); @@ -880,10 +872,12 @@ void CGIntrinsicsOpenMP::emitOMPParallelDeviceRuntime( OutlinedWrapperFnBitcast, CapturedVarAddrsBitcast, NumCapturedArgs}; + // nt_strict is 0: the strict modifier of num_threads is not supported. + Args.push_back(OMPBuilder.Builder.getInt32(0)); - auto *CallKmpcParallel51 = - checkCreateCall(OMPBuilder.Builder, KmpcParallel51, Args); - assert(CallKmpcParallel51 && + auto *CallKmpcParallel = + checkCreateCall(OMPBuilder.Builder, KmpcParallel, Args); + assert(CallKmpcParallel && "Expected non-null call instr from code generation"); FunctionCallee KmpcFreeShared = @@ -1994,12 +1988,7 @@ void CGIntrinsicsOpenMP::emitOMPSingle(Function *Fn, BasicBlock *BBEntry, OpenMPIRBuilder::LocationDescription Loc( InsertPointTy(BBEntry, BBEntry->end()), DL); -// TODO: handle nowait clause. -#if LLVM_VERSION_MAJOR <= 16 - InsertPointTy AfterIP = OMPBuilder.createSingle( - Loc, BodyGenCB, FiniCB, /* IsNoWait*/ false, /*DidIt*/ nullptr); -#else - + // TODO: handle nowait clause. auto IPOrError = OMPBuilder.createSingle(Loc, BodyGenCB, FiniCB, /* IsNoWait*/ false); if (auto E = IPOrError.takeError()) { @@ -2008,7 +1997,6 @@ void CGIntrinsicsOpenMP::emitOMPSingle(Function *Fn, BasicBlock *BBEntry, } InsertPointTy AfterIP = *IPOrError; -#endif BranchInst::Create(AfterBB, AfterIP.getBlock()); DEBUG_ENABLE(dbgs() << "=== Single Fn\n" << *Fn << "=== End of Single Fn\n"); } @@ -2026,11 +2014,6 @@ void CGIntrinsicsOpenMP::emitOMPCritical(Function *Fn, BasicBlock *BBEntry, OpenMPIRBuilder::LocationDescription Loc( InsertPointTy(BBEntry, BBEntry->end()), DL); -#if LLVM_VERSION_MAJOR <= 16 - InsertPointTy AfterIP = OMPBuilder.createCritical(Loc, BodyGenCB, FiniCB, "", - /*HintInst*/ nullptr); -#else - auto IPOrError = OMPBuilder.createCritical(Loc, BodyGenCB, FiniCB, "", /*HintInst*/ nullptr); if (auto E = IPOrError.takeError()) { @@ -2039,7 +2022,6 @@ void CGIntrinsicsOpenMP::emitOMPCritical(Function *Fn, BasicBlock *BBEntry, } InsertPointTy AfterIP = *IPOrError; -#endif BranchInst::Create(AfterBB, AfterIP.getBlock()); DEBUG_ENABLE(dbgs() << "=== Critical Fn\n" << *Fn << "=== End of Critical Fn\n"); @@ -2315,7 +2297,8 @@ void CGIntrinsicsOpenMP::emitOMPTargetHost( KernelNumTeams, KernelNumThreads, Constant::getNullValue(OMPBuilder.VoidPtr), - /*TargetInfo.NoWait*/ false}; + /*TargetInfo.NoWait*/ false, + omp::OMPDynGroupprivateFallbackType::Abort}; OpenMPIRBuilder::getKernelArgsVector(Args, OMPBuilder.Builder, ArgsVector); assert(TargetInfo.DeviceID && "Expected non-null device id"); @@ -2424,11 +2407,6 @@ void CGIntrinsicsOpenMP::emitOMPTargetDevice(Function *Fn, BasicBlock *EntryBB, bool IsSPMD = (TargetInfo.ExecMode == omp::OMP_TGT_EXEC_MODE_SPMD); if (isOpenMPDeviceRuntime()) { OpenMPIRBuilder::LocationDescription Loc(Builder); -#if LLVM_VERSION_MAJOR <= 15 - auto IP = OMPBuilder.createTargetInit(Loc, IsSPMD, true); -#elif LLVM_VERSION_MAJOR <= 16 - auto IP = OMPBuilder.createTargetInit(Loc, IsSPMD); -#else // Note the default for MaxThreads is 0. OpenMPIRBuilder::TargetKernelDefaultAttrs Attrs{ (IsSPMD ? OMP_TGT_EXEC_MODE_SPMD : OMP_TGT_EXEC_MODE_GENERIC), @@ -2437,7 +2415,6 @@ void CGIntrinsicsOpenMP::emitOMPTargetDevice(Function *Fn, BasicBlock *EntryBB, {0, -1, -1}, 1}; auto IP = OMPBuilder.createTargetInit(Builder, Attrs); -#endif Builder.restoreIP(IP); } @@ -2446,13 +2423,7 @@ void CGIntrinsicsOpenMP::emitOMPTargetDevice(Function *Fn, BasicBlock *EntryBB, if (isOpenMPDeviceRuntime()) { OpenMPIRBuilder::LocationDescription Loc(Builder); -#if LLVM_VERSION_MAJOR <= 15 - OMPBuilder.createTargetDeinit(Loc, /* IsSPMD */ IsSPMD, true); -#elif LLVM_VERSION_MAJOR <= 16 - OMPBuilder.createTargetDeinit(Loc, /* IsSPMD */ IsSPMD); -#else OMPBuilder.createTargetDeinit(Loc); -#endif } Builder.CreateRetVoid(); @@ -2821,11 +2792,7 @@ void CGIntrinsicsOpenMP::emitOMPDistributeParallelFor( BasicBlock *ParAfterBB = ForExitAfter; DEBUG_ENABLE(dbgs() << "ParAfterBB " << ParAfterBB->getName() << "\n"); -#if LLVM_VERSION_MAJOR <= 16 - auto FiniCB = [](auto) {}; -#else auto FiniCB = [](InsertPointTy) { return Error::success(); }; -#endif emitOMPParallel(DSAValueMap, nullptr, DL, Fn, ParEntryBB, ParStartBB, ParEndBB, ParAfterBB, FiniCB, ParRegionInfo); diff --git a/src/numba/openmp/libs/pass/CGIntrinsicsOpenMP.h b/src/numba/openmp/libs/pass/CGIntrinsicsOpenMP.h index 7d36d4b8848c..11b1b8ea8ddc 100644 --- a/src/numba/openmp/libs/pass/CGIntrinsicsOpenMP.h +++ b/src/numba/openmp/libs/pass/CGIntrinsicsOpenMP.h @@ -275,12 +275,6 @@ struct CGReduction { unsigned int Bitwidth = VTy->getScalarSizeInBits(); auto *IntTy = (Bitwidth == 64 ? Type::getInt64Ty(Ctx) : Type::getInt32Ty(Ctx)); -#if LLVM_VERSION_MAJOR <= 15 - auto *IntPtrTy = - (Bitwidth == 64 ? Type::getInt64PtrTy(Ctx) : Type::getInt32PtrTy(Ctx)); -#else - auto *IntPtrTy = PointerType::getUnqual(IntTy); -#endif auto SaveIP = IRB.saveIP(); // TODO: move alloca to function entry point, may be outlined later, e.g., @@ -288,10 +282,8 @@ struct CGReduction { Value *AllocaTemp = IRB.CreateAlloca(IntTy, nullptr, "atomic.alloca.tmp"); IRB.restoreIP(SaveIP); - Value *CastLHS = - IRB.CreateBitCast(LHS, IntPtrTy, LHS->getName() + ".cast.int"); auto *LoadAtomic = - IRB.CreateLoad(IntTy, CastLHS, LHS->getName() + ".load.atomic"); + IRB.CreateLoad(IntTy, LHS, LHS->getName() + ".load.atomic"); LoadAtomic->setAtomic(AtomicOrdering::Monotonic); Value *CastFP = IRB.CreateBitCast(LoadAtomic, VTy, "cast.fp"); @@ -300,7 +292,7 @@ struct CGReduction { IRB.CreateBitCast(RedOp, IntTy, RedOp->getName() + ".cast.int"); auto *CmpXchg = IRB.CreateAtomicCmpXchg( - CastLHS, LoadAtomic, CastFAdd, MaybeAlign(), AtomicOrdering::Monotonic, + LHS, LoadAtomic, CastFAdd, MaybeAlign(), AtomicOrdering::Monotonic, AtomicOrdering::Monotonic); auto *Returned = IRB.CreateExtractValue(CmpXchg, 0); @@ -322,8 +314,8 @@ struct CGReduction { // FAdd = IRB.CreateFAdd(CastLoad, Partial, "retry.add"); RedOp = emitOperation(IRB, CastLoad, Partial); CastFAdd = IRB.CreateBitCast(RedOp, IntTy, RedOp->getName() + ".cast.int"); - CmpXchg = IRB.CreateAtomicCmpXchg(CastLHS, LoadReturned, CastFAdd, - MaybeAlign(), AtomicOrdering::Monotonic, + CmpXchg = IRB.CreateAtomicCmpXchg(LHS, LoadReturned, CastFAdd, MaybeAlign(), + AtomicOrdering::Monotonic, AtomicOrdering::Monotonic); Returned = IRB.CreateExtractValue(CmpXchg, 0); StoreTemp = IRB.CreateStore(Returned, AllocaTemp); @@ -411,19 +403,14 @@ struct CGReduction { } else FATAL_ERROR("Unsupported type to init with identity reduction value"); -#if LLVM_VERSION_MAJOR <= 16 - ReductionInfos.push_back( - {ReductionTy, Orig, Priv, - CGReduction::reductionNonAtomic, - CGReduction::reductionAtomic}); -#else // TODO: Support more evaluation kinds besides scalar. + // DataPtrPtrGen is only used for by-ref reductions, which are unused. ReductionInfos.push_back( {ReductionTy, Orig, Priv, OpenMPIRBuilder::EvalKind::Scalar, CGReduction::reductionNonAtomic, /* ReductionGenClang */ nullptr, - CGReduction::reductionAtomic}); -#endif + CGReduction::reductionAtomic, + /* DataPtrPtrGen */ nullptr}); return Priv; } diff --git a/src/numba/openmp/libs/pass/IntrinsicsOpenMP.cpp b/src/numba/openmp/libs/pass/IntrinsicsOpenMP.cpp index 351e415cf408..d415cbbbf150 100644 --- a/src/numba/openmp/libs/pass/IntrinsicsOpenMP.cpp +++ b/src/numba/openmp/libs/pass/IntrinsicsOpenMP.cpp @@ -33,7 +33,7 @@ #include #include #include -#include +#include #include #include #include @@ -571,17 +571,11 @@ struct IntrinsicsOpenMP { CGStartBB->getTerminator()->setSuccessor(0, StartBB); assert(EndBB != nullptr && "EndBB should not be null"); EndBB->getTerminator()->setSuccessor(0, CGEndBB); -#if LLVM_VERSION_MAJOR > 16 return Error::success(); -#endif }; // Define the default FiniCB lambda. -#if LLVM_VERSION_MAJOR <= 16 - auto FiniCB = [&](InsertPointTy CodeGenIP) {}; -#else auto FiniCB = [&](InsertPointTy) { return Error::success(); }; -#endif // Remove intrinsics of OpenMP tags, first CBExit to also remove use // of CBEntry, then CBEntry.