Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
110 changes: 110 additions & 0 deletions GPU/GPUTracking/Base/metal/CMakeLists.txt
Original file line number Diff line number Diff line change
@@ -0,0 +1,110 @@
# Copyright 2019-2020 CERN and copyright holders of ALICE O2.
# See https://alice-o2.web.cern.ch/copyright for details of the copyright holders.
# All rights not expressly granted are reserved.
#
# This software is distributed under the terms of the GNU General Public
# License v3 (GPL Version 3), copied verbatim in the file "COPYING".
#
# In applying this license CERN does not waive the privileges and immunities
# granted to it by virtue of its status as an Intergovernmental Organization
# or submit itself to any jurisdiction.

set(MODULE GPUTrackingMETAL)
enable_language(ASM)

message(STATUS "Building GPUTracking with Metal support")

# convenience variables
if(ALIGPU_BUILD_TYPE STREQUAL "Standalone")
set(GPUDIR ${CMAKE_SOURCE_DIR}/../)
else()
set(GPUDIR ${CMAKE_SOURCE_DIR}/GPU/GPUTracking)
endif()
set(METAL_SRC ${GPUDIR}/Base/metal/GPUReconstructionMETAL.metal)
set(METAL_BIN ${CMAKE_CURRENT_BINARY_DIR}/GPUReconstructionMetalCode)

# MSL 4.1 is the first version with a generic address space: earlier versions
# reject an unannotated pointer or `this` outright, which GPUCommonDefAPI.h
# relies on for GPUgeneric() and GPUdDefault().
set(METAL_FLAGS -std=metal4.1 ${GPUCA_METAL_DENORMALS_FLAGS})
if(GPUCA_DETERMINISTIC_MODE GREATER_EQUAL ${GPUCA_DETERMINISTIC_MODE_MAP_NO_FAST_MATH})
set(METAL_FLAGS ${METAL_FLAGS} ${GPUCA_METAL_NO_FAST_MATH_FLAGS})
endif()
set(METAL_DEFINES "-D$<JOIN:$<TARGET_PROPERTY:O2::GPUTracking,COMPILE_DEFINITIONS>,$<SEMICOLON>-D>"
"-I$<JOIN:$<FILTER:$<TARGET_PROPERTY:O2::GPUTracking,INCLUDE_DIRECTORIES>,EXCLUDE,^/usr/include/?>,$<SEMICOLON>-I>"
-I${CMAKE_SOURCE_DIR}/Detectors/TRD/base/src
-I${CMAKE_SOURCE_DIR}/Detectors/Base/src
-I${CMAKE_SOURCE_DIR}/DataFormats/Reconstruction/src
)

set(SRCS GPUReconstructionMetal.mm GPUReconstructionMetalKernels.mm)
set(HDRS GPUReconstructionMetal.h GPUReconstructionMetalIncludesHost.h)

if(ALIGPU_BUILD_TYPE STREQUAL "O2")
o2_add_library(${MODULE}
SOURCES ${SRCS}
PUBLIC_LINK_LIBRARIES O2::GPUTracking
TARGETVARNAME targetName)

target_link_libraries(${targetName} PUBLIC ${METAL_FRAMEWORKS})

target_compile_definitions(${targetName} PRIVATE $<TARGET_PROPERTY:O2::GPUTracking,COMPILE_DEFINITIONS>)
# the compile_defitions are not propagated automatically on purpose (they are
# declared PRIVATE) so we are not leaking them outside of the GPU**
# directories
endif()

if(ALIGPU_BUILD_TYPE STREQUAL "Standalone")
add_library(${MODULE} SHARED ${SRCS})
target_link_libraries(${MODULE} GPUTracking)
install(TARGETS ${MODULE})
set(targetName ${MODULE})
endif()

if(METAL_ENABLED) # BUILD Metal source code for runtime compilation target

# executes clang to preprocess
add_custom_command(
OUTPUT ${METAL_BIN}.metal
COMMAND xcrun -sdk macosx metal
-Wno-unused-command-line-argument
${METAL_FLAGS}
${METAL_DEFINES}
-MD -MT ${METAL_BIN}.src -MF ${METAL_BIN}.src.d
-E -P ${METAL_SRC} > ${METAL_BIN}.metal
DEPENDS ${METAL_SRC}
DEPFILE ${METAL_BIN}.src.d
COMMAND_EXPAND_LISTS
COMMENT "Preparing Metal source file for run time compilation ${METAL_BIN}.metal")

# Create the ir
add_custom_command(
OUTPUT ${METAL_BIN}.ir
COMMAND xcrun -sdk macosx metal
-Wno-unused-command-line-argument
-Wno-c++17-extensions
-ferror-limit=10000
${METAL_FLAGS}
${METAL_DEFINES}
${METAL_BIN}.metal
-o ${METAL_BIN}.ir
DEPENDS ${METAL_BIN}.metal
COMMAND_EXPAND_LISTS
COMMENT "Preparing Metal intermediate representation for run time compilation ${METAL_BIN}.ir")

add_custom_target(metal_preprocessed_code ALL DEPENDS ${METAL_BIN}.metal COMMENT "Needed to inject dependency on its creation")
add_custom_target(metal_intermediate_representation ALL DEPENDS ${METAL_BIN}.ir COMMENT "Needed to inject dependency on its creation")

# Pack the compiled library into __DATA,__gpu_resource during final link. This
# way we do not need to create an intermediate object. Compiling the source at
# run time is not an option: the driver's compiler service dies on it.
target_link_options(${targetName}
PRIVATE
"-Wl,-sectcreate,__DATA,__gpu_resource,${METAL_BIN}.ir")
add_dependencies(${targetName} metal_preprocessed_code)
add_dependencies(${targetName} metal_intermediate_representation)
endif()

install(FILES ${HDRS} DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}/GPU)

target_compile_definitions(${targetName} PRIVATE GPUCA_METAL_BUILD_FLAGS=$<JOIN:${METAL_FLAGS},\ > )
77 changes: 77 additions & 0 deletions GPU/GPUTracking/Base/metal/GPUReconstructionMETAL.metal
Original file line number Diff line number Diff line change
@@ -0,0 +1,77 @@
// Copyright 2019-2025 CERN and copyright holders of ALICE O2.
// See https://alice-o2.web.cern.ch/copyright for details of the copyright holders.
// All rights not expressly granted are reserved.
//
// This software is distributed under the terms of the GNU General Public
// License v3 (GPL Version 3), copied verbatim in the file "COPYING".
//
// In applying this license CERN does not waive the privileges and immunities
// granted to it by virtue of its status as an Intergovernmental Organization
// or submit itself to any jurisdiction.

/// \file GPUReconstructionMETAL.metal

#pragma clang diagnostic push
#pragma clang diagnostic ignored "-Wgnu-zero-variadic-macro-arguments"
// clang-format off

// --- Backend selection -------------------------------------------------------
#define GPUCA_GPUTYPE_METAL 1

// --- Metal stdlib ------------------------------------------------------------
#include <metal_stdlib>
// MSL rejects derived classes outside of this pragma, and the kernels are
// class-based throughout. metal_stdlib itself uses it in 46 paired places.
#pragma METAL internals : enable
using namespace metal;

// --- OpenCL compatibility shims ---------------------------------------------

// Address space aliases (match OpenCL vernacular used by the project); constant
// is spelled the same in MSL
#define global device
#define local threadgroup

#ifndef M_PI
#define M_PI 3.1415926535f
#endif

// Disable assertions inside GPU code (same as OpenCL variant)
#ifdef assert
# undef assert
#endif
#define assert(param)

// --- double ------------------------------------------------------------------
// MSL has no double. GPUdoubleBinary64 is IEEE-754 binary64 in software, with the
// same eight bytes in the same order, so the keyword can simply name it and the
// shared code needs no separate spelling. Must come after metal_stdlib, which
// uses the token itself.
#include "GPUCommonDoubleBinary64.h"
#define double o2::gpu::GPUdoubleBinary64
#include "GPUCommonDouble.h"

// --- Project headers ---------------------------------------------------------
#include "GPUCommonDef.h"
#include "GPUCommonTypeTraits.h"
#include "GPUCommonArray.h"

#include "GPUConstantMem.h"
#include "GPUReconstructionIncludesDeviceAll.h"

// --- Kernel list expansion ---------------------------------------------------
#define GPUCA_KRNL(...) GPUCA_KRNLGPU(__VA_ARGS__)

// --- Constant memory + global heap plumbing ---------------------------------
// The heap and the constant memory arrive as buffer(0) and buffer(1). The latter
// is untyped because a buffer of GPUConstantMem, which has base classes, is not
// a valid kernel argument type.
#define GPUCA_CONSMEM_PTR \
device char* gpu_mem [[buffer(0)]], \
device char* pConstantRaw [[buffer(1)]],
#define GPUCA_CONSMEM (*(device GPUConstantMem*)pConstantRaw)

#include "GPUReconstructionKernelList.h"

// clang-format on
#pragma clang diagnostic pop
66 changes: 66 additions & 0 deletions GPU/GPUTracking/Base/metal/GPUReconstructionMetal.h
Original file line number Diff line number Diff line change
@@ -0,0 +1,66 @@
// Copyright 2019-2025 CERN and copyright holders of ALICE O2.
// See https://alice-o2.web.cern.ch/copyright for details of the copyright holders.
// All rights not expressly granted are reserved.
//
// This software is distributed under the terms of the GNU General Public
// License v3 (GPL Version 3), copied verbatim in the file "COPYING".
//
// In applying this license CERN does not waive the privileges and immunities
// granted to it by virtue of its status as an Intergovernmental Organization
// or submit itself to any jurisdiction.

#ifndef GPURECONSTRUCTIONMETAL_H
#define GPURECONSTRUCTIONMETAL_H

#include "GPUReconstructionDeviceBase.h"

extern "C" o2::gpu::GPUReconstruction* GPUReconstruction_Create_METAL(const o2::gpu::GPUSettingsDeviceBackend& cfg);

namespace o2::gpu
{
struct GPUReconstructionMetalInternals;

class GPUReconstructionMetal : public GPUReconstructionProcessing::KernelInterface<GPUReconstructionMetal, GPUReconstructionDeviceBase>
{
public:
GPUReconstructionMetal(const GPUSettingsDeviceBackend& cfg);
~GPUReconstructionMetal() override;

template <class T, int32_t I = 0, typename... Args>
void runKernelBackend(const krnlSetupTime& _xyz, const Args&... args);

protected:
int32_t InitDevice_Runtime() override;
int32_t ExitDevice_Runtime() override;

virtual int32_t GPUChkErrInternal(const int64_t error, const char* file, int32_t line) const override;

void SynchronizeGPU() override;
int32_t GPUDebug(const char* state = "UNKNOWN", int32_t stream = -1, bool force = false) override;
void SynchronizeStream(int32_t stream) override;
void SynchronizeEvents(deviceEvent* evList, int32_t nEvents = 1) override;
void StreamWaitForEvents(int32_t stream, deviceEvent* evList, int32_t nEvents = 1) override;
bool IsEventDone(deviceEvent* evList, int32_t nEvents = 1) override;

size_t WriteToConstantMemory(size_t offset, const void* src, size_t size, int32_t stream = -1, deviceEvent* ev = nullptr) override;
size_t GPUMemCpy(void* dst, const void* src, size_t size, int32_t stream, int32_t toGPU, deviceEvent* ev = nullptr, deviceEvent* evList = nullptr, int32_t nEvents = 1) override;
void ReleaseEvent(deviceEvent ev) override;
void RecordMarker(deviceEvent* ev, int32_t stream) override;

template <class T, int32_t I = 0>
int32_t AddKernel();

GPUReconstructionMetalInternals* mInternals;

template <class S, class T, int32_t I>
S& getKernelObject();

int32_t GetMetalPrograms();

private:
int32_t AddKernels();
};

} // namespace o2::gpu

#endif // GPURECONSTRUCTIONMETAL_H
Loading
Loading