Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions Detectors/Base/src/O2Tessellated.cxx
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@
// Will be deleted once we get this from ROOT.

#include <iostream>
#include <fstream>
#include <sstream>

#include "TGeoManager.h"
Expand Down
1 change: 1 addition & 0 deletions Detectors/MUON/MCH/Evaluation/src/Draw.cxx
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@
#include <TLegend.h>
#include <TStyle.h>
#include <fmt/format.h>
#include <iostream>
#include <limits>
#include <memory>
#include <stdexcept>
Expand Down
4 changes: 2 additions & 2 deletions GPU/GPUTracking/Base/GPUConstantMem.h
Original file line number Diff line number Diff line change
Expand Up @@ -34,7 +34,7 @@
#include "GPUKernelDebugOutput.h"
#endif

#ifdef GPUCA_HAS_ONNX
#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE)
#include "GPUTPCNNClusterizer.h"
#endif

Expand All @@ -56,7 +56,7 @@ struct GPUConstantMem {
#ifdef GPUCA_KERNEL_DEBUGGER_OUTPUT
GPUKernelDebugOutput debugOutput;
#endif
#ifdef GPUCA_HAS_ONNX
#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE)
GPUTPCNNClusterizer tpcNNClusterer[GPUTPCGeometry::NSECTORS];
#endif
template <int32_t I>
Expand Down
2 changes: 1 addition & 1 deletion GPU/GPUTracking/Base/GPUReconstruction.cxx
Original file line number Diff line number Diff line change
Expand Up @@ -97,7 +97,7 @@ GPUReconstruction::GPUReconstruction(const GPUSettingsDeviceBackend& cfg) : mHos
for (uint32_t i = 0; i < NSECTORS; i++) {
processors()->tpcTrackers[i].SetSector(i); // TODO: Move to a better place
processors()->tpcClusterer[i].mISector = i;
#ifdef GPUCA_HAS_ONNX
#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE)
processors()->tpcNNClusterer[i].mISector = i;
#endif
}
Expand Down
3 changes: 3 additions & 0 deletions GPU/GPUTracking/Base/GPUReconstructionCPU.h
Original file line number Diff line number Diff line change
Expand Up @@ -83,6 +83,9 @@ class GPUReconstructionCPU : public GPUReconstructionProcessing::KernelInterface
size_t WriteToConstantMemory(size_t offset, const void* src, size_t size, int32_t stream = -1, deviceEvent* ev = nullptr) override;
virtual size_t TransferMemoryInternal(GPUMemoryResource* res, int32_t stream, deviceEvent* ev, deviceEvent* evList, int32_t nEvents, bool toGPU, const void* src, void* dst);

virtual int32_t GetNativeGPUDevice() const { throw std::runtime_error("Native GPU device is unavailable for this backend"); }
virtual void* GetNativeGPUStream(int32_t) const { throw std::runtime_error("Native GPU streams are unavailable for this backend"); }

// ONNX runtime
virtual void SetONNXGPUStream(Ort::SessionOptions&, int32_t, int32_t*) {}

Expand Down
21 changes: 19 additions & 2 deletions GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu
Original file line number Diff line number Diff line change
Expand Up @@ -635,11 +635,28 @@ void GPUReconstructionCUDA::loadKernelModules(bool perKernel)
} \
}

int32_t GPUReconstructionCUDA::GetNativeGPUDevice() const
{
int device = -1;
if (GPUChkErrInternal(cudaGetDevice(&device), __FILE__, __LINE__)) {
throw std::runtime_error("GPU device query failed");
}
return device;
}

void* GPUReconstructionCUDA::GetNativeGPUStream(int32_t stream) const
{
if (stream < 0 || stream >= mNStreams) {
throw std::out_of_range("GPU stream index out of range");
}
return mInternals->Streams[stream];
}

void GPUReconstructionCUDA::SetONNXGPUStream(Ort::SessionOptions& sessionOptions, int32_t stream, int32_t* deviceId)
{
GPUChkErr(cudaGetDevice(deviceId));

#if !defined(__HIPCC__) && defined(ORT_CUDA_BUILD)
#if defined(GPUCA_HAS_ONNX) && !defined(__HIPCC__) && defined(ORT_CUDA_BUILD)
const OrtApi* api = OrtGetApiBase()->GetApi(ORT_API_VERSION);

#ifdef ORT_TENSORRT_BUILD
Expand All @@ -666,7 +683,7 @@ void GPUReconstructionCUDA::SetONNXGPUStream(Ort::SessionOptions& sessionOptions
ORTCHK(api->SessionOptionsAppendExecutionProvider_CUDA_V2(sessionOptions, cudaOptions));
api->ReleaseCUDAProviderOptions(cudaOptions);

#elif defined(ORT_ROCM_BUILD)
#elif defined(GPUCA_HAS_ONNX) && defined(ORT_ROCM_BUILD)
// const auto& api = Ort::GetApi();
// api.GetCurrentGpuDeviceId(deviceId);
OrtROCMProviderOptions rocmOptions;
Expand Down
2 changes: 2 additions & 0 deletions GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.h
Original file line number Diff line number Diff line change
Expand Up @@ -74,6 +74,8 @@ class GPUReconstructionCUDA : public GPUReconstructionProcessing::KernelInterfac
size_t GPUMemCpy(void* dst, const void* src, size_t size, int32_t stream, int32_t toGPU, deviceEvent* ev = nullptr, deviceEvent* evList = nullptr, int32_t nEvents = 1) override;
void ReleaseEvent(deviceEvent ev) override;
void RecordMarker(deviceEvent* ev, int32_t stream) override;
int32_t GetNativeGPUDevice() const override;
void* GetNativeGPUStream(int32_t stream) const override;
void SetONNXGPUStream(Ort::SessionOptions& session_options, int32_t stream, int32_t* deviceId) override;

void GetITSTraits(std::unique_ptr<o2::its::TrackerTraits<7>>* trackerTraits, std::unique_ptr<o2::its::VertexerTraits<7>>* vertexerTraits, std::unique_ptr<o2::its::TimeFrame<7>>* timeFrame) override;
Expand Down
23 changes: 21 additions & 2 deletions GPU/GPUTracking/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,18 @@
# or submit itself to any jurisdiction.

set(MODULE GPUTracking)
option(GPUCA_BUILD_SOFIE "Enable experimental SOFIE GPU inference" OFF)
option(GPUCA_BUILD_ORT "Enable ONNXRuntime inference in GPUTracking" ON)
if(NOT GPUCA_BUILD_ORT)
set(onnxruntime_FOUND OFF)
endif()
if(GPUCA_BUILD_SOFIE)
find_package(ROOT CONFIG REQUIRED COMPONENTS ROOTTMVASofie ROOTTMVASofieParser)
find_path(GPUCA_SOFIE_GPU_INCLUDE TMVA/RGPUModel.hxx HINTS ${ROOT_INCLUDE_DIRS} NO_DEFAULT_PATH)
if(NOT GPUCA_SOFIE_GPU_INCLUDE)
message(FATAL_ERROR "SOFIE GPU requires the ROOT development package with RGPUModel support")
endif()
endif()

# set(CMAKE_CXX_FLAGS_${CMAKE_BUILD_TYPE_UPPER} "${CMAKE_CXX_FLAGS_${CMAKE_BUILD_TYPE_UPPER}} -O0") # to uncomment if needed, tired of typing this...
# set(GPUCA_BUILD_DEBUG 1)
Expand Down Expand Up @@ -196,7 +208,7 @@ set(SRCS_NO_CINT ${SRCS_NO_CINT}
Refit/GPUTrackingRefitKernel.cxx
Merger/GPUTPCGMO2Output.cxx)

if(onnxruntime_FOUND)
if(onnxruntime_FOUND OR GPUCA_BUILD_SOFIE)
list(APPEND SRCS_NO_CINT TPCClusterFinder/GPUTPCNNClusterizerKernels.cxx TPCClusterFinder/GPUTPCNNClusterizer.cxx TPCClusterFinder/GPUTPCNNClusterizerHost.cxx)
endif()

Expand Down Expand Up @@ -327,6 +339,7 @@ set(HDRS_CINT_DATATYPES ${HDRS_CINT_DATATYPES} ${HDRS_TMP})
unset(HDRS_TMP)

set(INCDIRS
${CMAKE_CURRENT_SOURCE_DIR}/../../Common/ML/include
${ON_THE_FLY_DIR}
${CMAKE_CURRENT_SOURCE_DIR}
${CMAKE_CURRENT_SOURCE_DIR}/Definitions
Expand Down Expand Up @@ -382,7 +395,6 @@ if(ALIGPU_BUILD_TYPE STREQUAL "O2")
O2::TPCFastTransformation
O2::DetectorsRaw
O2::Steer
O2::ML
PUBLIC_INCLUDE_DIRECTORIES ${INCDIRS}
SOURCES ${SRCS} ${SRCS_NO_CINT} ${SRCS_NO_H})

Expand Down Expand Up @@ -454,12 +466,19 @@ if(GPUCA_QA)
endif()

target_link_libraries(${targetName} PRIVATE TBB::tbb)
if(GPUCA_BUILD_SOFIE)
target_compile_definitions(${targetName} PRIVATE GPUCA_HAS_SOFIE=1)
target_link_libraries(${targetName} PRIVATE ROOT::ROOTTMVASofie ROOT::ROOTTMVASofieParser)
endif()

target_compile_options(${targetName} PRIVATE -Wno-instantiation-after-specialization)

if (onnxruntime_FOUND)
target_compile_definitions(${targetName} PRIVATE GPUCA_HAS_ONNX=1)
target_link_libraries(${targetName} PRIVATE onnxruntime::onnxruntime)
if(TARGET O2::ML)
target_link_libraries(${targetName} PRIVATE O2::ML)
endif()
endif()

# Add CMake recipes for GPU Tracking librararies
Expand Down
3 changes: 3 additions & 0 deletions GPU/GPUTracking/Definitions/GPUSettingsList.h
Original file line number Diff line number Diff line change
Expand Up @@ -260,6 +260,9 @@ EndConfig()
// Settings steering the processing of NN Clusterization
BeginSubConfig(GPUSettingsProcessingNNclusterizer, nn, configStandalone.proc, "NN", 0, "Processing settings for neural network clusterizer", proc_nn)
AddOption(applyNNclusterizer, int, 0, "", 0, "(bool, default = 0), if the neural network clusterizer should be used.")
AddOption(mlFramework, std::string, "ORT", "ml-framework", 0, "ML inference backend: ORT or SOFIE")
AddOption(sofieCompiler, std::string, "", "sofie-compiler", 0, "Runtime compiler executable (default: nvcc or hipcc)")
AddOption(sofieArchitecture, std::string, "", "sofie-architecture", 0, "Required SOFIE target architecture, e.g. sm_80 or gfx90a")
AddOption(nnInferenceDevice, std::string, "CPU", "", 0, "(std::string) Specify inference device (cpu (default), rocm, cuda)")
AddOption(nnInferenceDeviceId, unsigned int, 0, "", 0, "(unsigned int) Specify inference device id")
AddOption(nnInferenceAllocateDevMem, int, 0, "", 0, "(bool, default = 0), if the device memory should be allocated for inference")
Expand Down
3 changes: 3 additions & 0 deletions GPU/GPUTracking/Global/GPUChain.h
Original file line number Diff line number Diff line change
Expand Up @@ -87,6 +87,9 @@ class GPUChain
inline GPUParam& param() { return mRec->param(); }
inline const GPUConstantMem* processors() const { return mRec->processors(); }
inline void SynchronizeStream(int32_t stream) { mRec->SynchronizeStream(stream); }
// Borrowed native stream; ownership and synchronization remain with the reconstruction.
inline int32_t GetNativeGPUDevice() const { return mRec->GetNativeGPUDevice(); }
inline void* GetNativeGPUStream(int32_t stream) const { return mRec->GetNativeGPUStream(stream); }
inline void SetONNXGPUStream(Ort::SessionOptions& opt, int32_t stream, int32_t* deviceId) { mRec->SetONNXGPUStream(opt, stream, deviceId); }
inline void SynchronizeEvents(deviceEvent* evList, int32_t nEvents = 1) { mRec->SynchronizeEvents(evList, nEvents); }
inline void SynchronizeEventAndRelease(deviceEvent& ev, bool doGPU = true)
Expand Down
92 changes: 89 additions & 3 deletions GPU/GPUTracking/Global/GPUChainTracking.cxx
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,11 @@
#include <chrono>

#include "GPUChainTracking.h"
#ifdef GPUCA_HAS_SOFIE
#include "GPUTPCNNClusterizerHost.h"
#include "ORTRootSerializer.h"
#include "GPUTPCNNClusterizer.h"
#endif
#include "GPUChainTrackingGetters.inc"
#include "GPUReconstructionIO.h"
#include "GPUChainTrackingDefs.h"
Expand Down Expand Up @@ -68,7 +73,16 @@ GPUChainTracking::GPUChainTracking(GPUReconstruction* rec, uint32_t maxTPCHits,
mFlatObjectsDevice.mChainTracking = this;
}

GPUChainTracking::~GPUChainTracking() = default;
GPUChainTracking::~GPUChainTracking()
{
#ifdef GPUCA_HAS_SOFIE
if (!mSofieApplications.empty()) {
const auto context = GetThreadContext();
SynchronizeGPU();
mSofieApplications.clear();
}
#endif
}

void GPUChainTracking::RegisterPermanentMemoryAndProcessors()
{
Expand Down Expand Up @@ -102,7 +116,7 @@ void GPUChainTracking::RegisterPermanentMemoryAndProcessors()
if (GetRecoSteps() & RecoStep::TPCClusterFinding) {
for (uint32_t i = 0; i < NSECTORS; i++) {
mRec->RegisterGPUProcessor(&processors()->tpcClusterer[i], GetRecoStepsGPU() & RecoStep::TPCClusterFinding);
#ifdef GPUCA_HAS_ONNX
#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE)
mRec->RegisterGPUProcessor(&processors()->tpcNNClusterer[i], GetRecoStepsGPU() & RecoStep::TPCClusterFinding);
#endif
}
Expand Down Expand Up @@ -147,7 +161,7 @@ void GPUChainTracking::RegisterGPUProcessors()
if (GetRecoStepsGPU() & RecoStep::TPCClusterFinding) {
for (uint32_t i = 0; i < NSECTORS; i++) {
mRec->RegisterGPUDeviceProcessor(&processorsShadow()->tpcClusterer[i], &processors()->tpcClusterer[i]);
#ifdef GPUCA_HAS_ONNX
#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE)
mRec->RegisterGPUDeviceProcessor(&processorsShadow()->tpcNNClusterer[i], &processors()->tpcNNClusterer[i]);
#endif
}
Expand Down Expand Up @@ -392,6 +406,7 @@ int32_t GPUChainTracking::Init()
}
}

InitSofieClusterizer(true);
return 0;
}

Expand Down Expand Up @@ -470,6 +485,13 @@ int32_t GPUChainTracking::ForceInitQA()

int32_t GPUChainTracking::Finalize()
{
#ifdef GPUCA_HAS_SOFIE
if (!mSofieApplications.empty()) {
const auto context = GetThreadContext();
SynchronizeGPU();
mSofieApplications.clear();
}
#endif
if (GetProcessingSettings().runQA && GetQA()->IsInitialized() && !(mConfigQA && mConfigQA->shipToQC) && !mQAFromForeignChain) {
GetQA()->UpdateChain(this);
GetQA()->DrawQAHistograms();
Expand Down Expand Up @@ -1006,3 +1028,67 @@ void GPUChainTracking::ApplySyncSettings(GPUSettingsProcessing& proc, GPUSetting
steps.setBits(gpudatatypes::RecoStep::TPCdEdx, dEdxMode == -1 ? !syncMode : (dEdxMode > 0));
}
}

void GPUChainTracking::InitSofieClusterizer(bool deferCCDB)
{
const auto& settings = GetProcessingSettings().nn;
if (settings.mlFramework != "ORT" && settings.mlFramework != "SOFIE") {
throw std::runtime_error("ml-framework must be ORT or SOFIE");
}
if (!settings.applyNNclusterizer) {
return;
}
if (settings.mlFramework == "ORT") {
#ifndef GPUCA_HAS_ONNX
throw std::runtime_error("ORT was requested but GPUCA_BUILD_ORT is disabled or ONNXRuntime is unavailable");
#endif
return;
}
#ifdef GPUCA_HAS_SOFIE
std::array<std::string_view, 3> buffers{};
if (settings.nnLoadFromCCDB) {
for (size_t i = 0; i < buffers.size(); i++) {
const auto* network = processors()->calibObjects.nnClusterizerNetworks[i];
if (network && network->getONNXModelSize()) {
buffers[i] = std::string_view(network->getONNXModel(), network->getONNXModelSize());
}
}
}
const auto* previous = mSofieApplications.empty() ? nullptr : mSofieApplications.front().get();
if (previous && (!settings.nnLoadFromCCDB || previous->hasSofieBuffers(buffers))) {
return;
}
const bool hip = mRec->GetDeviceType() == GPUReconstruction::DeviceType::HIP;
if ((!hip && mRec->GetDeviceType() != GPUReconstruction::DeviceType::CUDA) || !(GetRecoStepsGPU() & RecoStep::TPCClusterFinding)) {
throw std::runtime_error("SOFIE requires TPC cluster finding on CUDA or HIP");
}
const int lanes = GetProcessingSettings().nTPCClustererLanes;
if (lanes < 1 || lanes > 4 || static_cast<uint32_t>(lanes) > mRec->NStreams()) {
throw std::runtime_error("SOFIE clusterizer requires 1..4 lanes and a stream for each lane");
}
if (deferCCDB && settings.nnLoadFromCCDB) {
return;
}
GPUInfo("SOFIE: %s, backend=%s, device=%d, lanes=%d, source=%s, batch capacity=%u",
previous ? "CCDB model change detected; reloading" : "initializing",
hip ? "HIP" : "CUDA", GetNativeGPUDevice(), lanes,
settings.nnLoadFromCCDB ? "CCDB" : "local ONNX files", settings.nnClusterizerBatchedMode);
std::vector<std::unique_ptr<GPUTPCNNClusterizerHost>> applications;
for (int lane = 0; lane < lanes; lane++) {
auto host = std::make_unique<GPUTPCNNClusterizerHost>();
host->initSofie(settings, GetNativeGPUStream(lane), GetNativeGPUDevice(), hip, lane ? applications.front().get() : nullptr, buffers, lane ? nullptr : previous);
GPUTPCNNClusterizer check;
host->initClusterizer(settings, check);
applications.push_back(std::move(host));
}
if (previous) {
SynchronizeGPU();
}
const bool reloaded = previous != nullptr;
mSofieApplications = std::move(applications);
GPUInfo("SOFIE: %s succeeded; inference backend active on device %d with %d lanes",
reloaded ? "CCDB model reload" : "initialization", GetNativeGPUDevice(), lanes);
#else
throw std::runtime_error("SOFIE was requested but GPUCA_BUILD_SOFIE is disabled");
#endif
}
5 changes: 5 additions & 0 deletions GPU/GPUTracking/Global/GPUChainTracking.h
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,7 @@ namespace o2::gpu
{
// class GPUTRDTrackerGPU;
class GPUTPCGPUTracker;
class GPUTPCNNClusterizerHost;
class GPUDisplayInterface;
class GPUQA;
class GPUTPCClusterStatistics;
Expand Down Expand Up @@ -298,6 +299,10 @@ class GPUChainTracking : public GPUChain
int32_t RunChainFinalize();
void OutputSanityCheck();
int32_t RunTPCTrackingSectors_internal();
void InitSofieClusterizer(bool deferCCDB = false);
#ifdef GPUCA_HAS_SOFIE
std::vector<std::unique_ptr<GPUTPCNNClusterizerHost>> mSofieApplications;
#endif
int32_t RunTPCClusterizer_prepare(bool restorePointers, const GPUTPCExtraADC& extraADCs);
#ifndef GPUCA_RUN2
std::pair<uint32_t, uint32_t> RunTPCClusterizer_transferZS(int32_t iSector, const CfFragment& fragment, int32_t lane, const GPUTPCExtraADC& extraADCs);
Expand Down
Loading
Loading