diff --git a/Detectors/Base/src/O2Tessellated.cxx b/Detectors/Base/src/O2Tessellated.cxx index d50b922cd8d25..2393d6f5d8321 100644 --- a/Detectors/Base/src/O2Tessellated.cxx +++ b/Detectors/Base/src/O2Tessellated.cxx @@ -16,6 +16,7 @@ // Will be deleted once we get this from ROOT. #include +#include #include #include "TGeoManager.h" diff --git a/Detectors/MUON/MCH/Evaluation/src/Draw.cxx b/Detectors/MUON/MCH/Evaluation/src/Draw.cxx index 918d89cff0912..3c721b01007d4 100644 --- a/Detectors/MUON/MCH/Evaluation/src/Draw.cxx +++ b/Detectors/MUON/MCH/Evaluation/src/Draw.cxx @@ -18,6 +18,7 @@ #include #include #include +#include #include #include #include diff --git a/GPU/GPUTracking/Base/GPUConstantMem.h b/GPU/GPUTracking/Base/GPUConstantMem.h index 05547262ce100..212f526e71ae7 100644 --- a/GPU/GPUTracking/Base/GPUConstantMem.h +++ b/GPU/GPUTracking/Base/GPUConstantMem.h @@ -34,7 +34,7 @@ #include "GPUKernelDebugOutput.h" #endif -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) #include "GPUTPCNNClusterizer.h" #endif @@ -56,7 +56,7 @@ struct GPUConstantMem { #ifdef GPUCA_KERNEL_DEBUGGER_OUTPUT GPUKernelDebugOutput debugOutput; #endif -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) GPUTPCNNClusterizer tpcNNClusterer[GPUTPCGeometry::NSECTORS]; #endif template diff --git a/GPU/GPUTracking/Base/GPUReconstruction.cxx b/GPU/GPUTracking/Base/GPUReconstruction.cxx index 3e7509d8fad73..cf898c7b0c807 100644 --- a/GPU/GPUTracking/Base/GPUReconstruction.cxx +++ b/GPU/GPUTracking/Base/GPUReconstruction.cxx @@ -97,7 +97,7 @@ GPUReconstruction::GPUReconstruction(const GPUSettingsDeviceBackend& cfg) : mHos for (uint32_t i = 0; i < NSECTORS; i++) { processors()->tpcTrackers[i].SetSector(i); // TODO: Move to a better place processors()->tpcClusterer[i].mISector = i; -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) processors()->tpcNNClusterer[i].mISector = i; #endif } diff --git a/GPU/GPUTracking/Base/GPUReconstructionCPU.h b/GPU/GPUTracking/Base/GPUReconstructionCPU.h index 466ad318bbe3b..59ccfed332f78 100644 --- a/GPU/GPUTracking/Base/GPUReconstructionCPU.h +++ b/GPU/GPUTracking/Base/GPUReconstructionCPU.h @@ -83,6 +83,9 @@ class GPUReconstructionCPU : public GPUReconstructionProcessing::KernelInterface size_t WriteToConstantMemory(size_t offset, const void* src, size_t size, int32_t stream = -1, deviceEvent* ev = nullptr) override; virtual size_t TransferMemoryInternal(GPUMemoryResource* res, int32_t stream, deviceEvent* ev, deviceEvent* evList, int32_t nEvents, bool toGPU, const void* src, void* dst); + virtual int32_t GetNativeGPUDevice() const { throw std::runtime_error("Native GPU device is unavailable for this backend"); } + virtual void* GetNativeGPUStream(int32_t) const { throw std::runtime_error("Native GPU streams are unavailable for this backend"); } + // ONNX runtime virtual void SetONNXGPUStream(Ort::SessionOptions&, int32_t, int32_t*) {} diff --git a/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu b/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu index 63992bed65fc5..34652858e9cd8 100644 --- a/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu +++ b/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu @@ -635,11 +635,28 @@ void GPUReconstructionCUDA::loadKernelModules(bool perKernel) } \ } +int32_t GPUReconstructionCUDA::GetNativeGPUDevice() const +{ + int device = -1; + if (GPUChkErrInternal(cudaGetDevice(&device), __FILE__, __LINE__)) { + throw std::runtime_error("GPU device query failed"); + } + return device; +} + +void* GPUReconstructionCUDA::GetNativeGPUStream(int32_t stream) const +{ + if (stream < 0 || stream >= mNStreams) { + throw std::out_of_range("GPU stream index out of range"); + } + return mInternals->Streams[stream]; +} + void GPUReconstructionCUDA::SetONNXGPUStream(Ort::SessionOptions& sessionOptions, int32_t stream, int32_t* deviceId) { GPUChkErr(cudaGetDevice(deviceId)); -#if !defined(__HIPCC__) && defined(ORT_CUDA_BUILD) +#if defined(GPUCA_HAS_ONNX) && !defined(__HIPCC__) && defined(ORT_CUDA_BUILD) const OrtApi* api = OrtGetApiBase()->GetApi(ORT_API_VERSION); #ifdef ORT_TENSORRT_BUILD @@ -666,7 +683,7 @@ void GPUReconstructionCUDA::SetONNXGPUStream(Ort::SessionOptions& sessionOptions ORTCHK(api->SessionOptionsAppendExecutionProvider_CUDA_V2(sessionOptions, cudaOptions)); api->ReleaseCUDAProviderOptions(cudaOptions); -#elif defined(ORT_ROCM_BUILD) +#elif defined(GPUCA_HAS_ONNX) && defined(ORT_ROCM_BUILD) // const auto& api = Ort::GetApi(); // api.GetCurrentGpuDeviceId(deviceId); OrtROCMProviderOptions rocmOptions; diff --git a/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.h b/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.h index 6f92ca1938e0d..4e5dd48aa0644 100644 --- a/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.h +++ b/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.h @@ -74,6 +74,8 @@ class GPUReconstructionCUDA : public GPUReconstructionProcessing::KernelInterfac size_t GPUMemCpy(void* dst, const void* src, size_t size, int32_t stream, int32_t toGPU, deviceEvent* ev = nullptr, deviceEvent* evList = nullptr, int32_t nEvents = 1) override; void ReleaseEvent(deviceEvent ev) override; void RecordMarker(deviceEvent* ev, int32_t stream) override; + int32_t GetNativeGPUDevice() const override; + void* GetNativeGPUStream(int32_t stream) const override; void SetONNXGPUStream(Ort::SessionOptions& session_options, int32_t stream, int32_t* deviceId) override; void GetITSTraits(std::unique_ptr>* trackerTraits, std::unique_ptr>* vertexerTraits, std::unique_ptr>* timeFrame) override; diff --git a/GPU/GPUTracking/CMakeLists.txt b/GPU/GPUTracking/CMakeLists.txt index cc44b2003a01d..74a5238cfb0ca 100644 --- a/GPU/GPUTracking/CMakeLists.txt +++ b/GPU/GPUTracking/CMakeLists.txt @@ -10,6 +10,18 @@ # or submit itself to any jurisdiction. set(MODULE GPUTracking) +option(GPUCA_BUILD_SOFIE "Enable experimental SOFIE GPU inference" OFF) +option(GPUCA_BUILD_ORT "Enable ONNXRuntime inference in GPUTracking" ON) +if(NOT GPUCA_BUILD_ORT) + set(onnxruntime_FOUND OFF) +endif() +if(GPUCA_BUILD_SOFIE) + find_package(ROOT CONFIG REQUIRED COMPONENTS ROOTTMVASofie ROOTTMVASofieParser) + find_path(GPUCA_SOFIE_GPU_INCLUDE TMVA/RGPUModel.hxx HINTS ${ROOT_INCLUDE_DIRS} NO_DEFAULT_PATH) + if(NOT GPUCA_SOFIE_GPU_INCLUDE) + message(FATAL_ERROR "SOFIE GPU requires the ROOT development package with RGPUModel support") + endif() +endif() # set(CMAKE_CXX_FLAGS_${CMAKE_BUILD_TYPE_UPPER} "${CMAKE_CXX_FLAGS_${CMAKE_BUILD_TYPE_UPPER}} -O0") # to uncomment if needed, tired of typing this... # set(GPUCA_BUILD_DEBUG 1) @@ -196,7 +208,7 @@ set(SRCS_NO_CINT ${SRCS_NO_CINT} Refit/GPUTrackingRefitKernel.cxx Merger/GPUTPCGMO2Output.cxx) -if(onnxruntime_FOUND) +if(onnxruntime_FOUND OR GPUCA_BUILD_SOFIE) list(APPEND SRCS_NO_CINT TPCClusterFinder/GPUTPCNNClusterizerKernels.cxx TPCClusterFinder/GPUTPCNNClusterizer.cxx TPCClusterFinder/GPUTPCNNClusterizerHost.cxx) endif() @@ -327,6 +339,7 @@ set(HDRS_CINT_DATATYPES ${HDRS_CINT_DATATYPES} ${HDRS_TMP}) unset(HDRS_TMP) set(INCDIRS + ${CMAKE_CURRENT_SOURCE_DIR}/../../Common/ML/include ${ON_THE_FLY_DIR} ${CMAKE_CURRENT_SOURCE_DIR} ${CMAKE_CURRENT_SOURCE_DIR}/Definitions @@ -382,7 +395,6 @@ if(ALIGPU_BUILD_TYPE STREQUAL "O2") O2::TPCFastTransformation O2::DetectorsRaw O2::Steer - O2::ML PUBLIC_INCLUDE_DIRECTORIES ${INCDIRS} SOURCES ${SRCS} ${SRCS_NO_CINT} ${SRCS_NO_H}) @@ -454,12 +466,19 @@ if(GPUCA_QA) endif() target_link_libraries(${targetName} PRIVATE TBB::tbb) +if(GPUCA_BUILD_SOFIE) + target_compile_definitions(${targetName} PRIVATE GPUCA_HAS_SOFIE=1) + target_link_libraries(${targetName} PRIVATE ROOT::ROOTTMVASofie ROOT::ROOTTMVASofieParser) +endif() target_compile_options(${targetName} PRIVATE -Wno-instantiation-after-specialization) if (onnxruntime_FOUND) target_compile_definitions(${targetName} PRIVATE GPUCA_HAS_ONNX=1) target_link_libraries(${targetName} PRIVATE onnxruntime::onnxruntime) + if(TARGET O2::ML) + target_link_libraries(${targetName} PRIVATE O2::ML) + endif() endif() # Add CMake recipes for GPU Tracking librararies diff --git a/GPU/GPUTracking/Definitions/GPUSettingsList.h b/GPU/GPUTracking/Definitions/GPUSettingsList.h index 81949f0b8ed2a..ff76a962dd168 100644 --- a/GPU/GPUTracking/Definitions/GPUSettingsList.h +++ b/GPU/GPUTracking/Definitions/GPUSettingsList.h @@ -260,6 +260,9 @@ EndConfig() // Settings steering the processing of NN Clusterization BeginSubConfig(GPUSettingsProcessingNNclusterizer, nn, configStandalone.proc, "NN", 0, "Processing settings for neural network clusterizer", proc_nn) AddOption(applyNNclusterizer, int, 0, "", 0, "(bool, default = 0), if the neural network clusterizer should be used.") +AddOption(mlFramework, std::string, "ORT", "ml-framework", 0, "ML inference backend: ORT or SOFIE") +AddOption(sofieCompiler, std::string, "", "sofie-compiler", 0, "Runtime compiler executable (default: nvcc or hipcc)") +AddOption(sofieArchitecture, std::string, "", "sofie-architecture", 0, "Required SOFIE target architecture, e.g. sm_80 or gfx90a") AddOption(nnInferenceDevice, std::string, "CPU", "", 0, "(std::string) Specify inference device (cpu (default), rocm, cuda)") AddOption(nnInferenceDeviceId, unsigned int, 0, "", 0, "(unsigned int) Specify inference device id") AddOption(nnInferenceAllocateDevMem, int, 0, "", 0, "(bool, default = 0), if the device memory should be allocated for inference") diff --git a/GPU/GPUTracking/Global/GPUChain.h b/GPU/GPUTracking/Global/GPUChain.h index 61107f7893e9c..535289c6bba58 100644 --- a/GPU/GPUTracking/Global/GPUChain.h +++ b/GPU/GPUTracking/Global/GPUChain.h @@ -87,6 +87,9 @@ class GPUChain inline GPUParam& param() { return mRec->param(); } inline const GPUConstantMem* processors() const { return mRec->processors(); } inline void SynchronizeStream(int32_t stream) { mRec->SynchronizeStream(stream); } + // Borrowed native stream; ownership and synchronization remain with the reconstruction. + inline int32_t GetNativeGPUDevice() const { return mRec->GetNativeGPUDevice(); } + inline void* GetNativeGPUStream(int32_t stream) const { return mRec->GetNativeGPUStream(stream); } inline void SetONNXGPUStream(Ort::SessionOptions& opt, int32_t stream, int32_t* deviceId) { mRec->SetONNXGPUStream(opt, stream, deviceId); } inline void SynchronizeEvents(deviceEvent* evList, int32_t nEvents = 1) { mRec->SynchronizeEvents(evList, nEvents); } inline void SynchronizeEventAndRelease(deviceEvent& ev, bool doGPU = true) diff --git a/GPU/GPUTracking/Global/GPUChainTracking.cxx b/GPU/GPUTracking/Global/GPUChainTracking.cxx index eb6d880398eec..bdae01693a29a 100644 --- a/GPU/GPUTracking/Global/GPUChainTracking.cxx +++ b/GPU/GPUTracking/Global/GPUChainTracking.cxx @@ -18,6 +18,11 @@ #include #include "GPUChainTracking.h" +#ifdef GPUCA_HAS_SOFIE +#include "GPUTPCNNClusterizerHost.h" +#include "ORTRootSerializer.h" +#include "GPUTPCNNClusterizer.h" +#endif #include "GPUChainTrackingGetters.inc" #include "GPUReconstructionIO.h" #include "GPUChainTrackingDefs.h" @@ -68,7 +73,16 @@ GPUChainTracking::GPUChainTracking(GPUReconstruction* rec, uint32_t maxTPCHits, mFlatObjectsDevice.mChainTracking = this; } -GPUChainTracking::~GPUChainTracking() = default; +GPUChainTracking::~GPUChainTracking() +{ +#ifdef GPUCA_HAS_SOFIE + if (!mSofieApplications.empty()) { + const auto context = GetThreadContext(); + SynchronizeGPU(); + mSofieApplications.clear(); + } +#endif +} void GPUChainTracking::RegisterPermanentMemoryAndProcessors() { @@ -102,7 +116,7 @@ void GPUChainTracking::RegisterPermanentMemoryAndProcessors() if (GetRecoSteps() & RecoStep::TPCClusterFinding) { for (uint32_t i = 0; i < NSECTORS; i++) { mRec->RegisterGPUProcessor(&processors()->tpcClusterer[i], GetRecoStepsGPU() & RecoStep::TPCClusterFinding); -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) mRec->RegisterGPUProcessor(&processors()->tpcNNClusterer[i], GetRecoStepsGPU() & RecoStep::TPCClusterFinding); #endif } @@ -147,7 +161,7 @@ void GPUChainTracking::RegisterGPUProcessors() if (GetRecoStepsGPU() & RecoStep::TPCClusterFinding) { for (uint32_t i = 0; i < NSECTORS; i++) { mRec->RegisterGPUDeviceProcessor(&processorsShadow()->tpcClusterer[i], &processors()->tpcClusterer[i]); -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) mRec->RegisterGPUDeviceProcessor(&processorsShadow()->tpcNNClusterer[i], &processors()->tpcNNClusterer[i]); #endif } @@ -392,6 +406,7 @@ int32_t GPUChainTracking::Init() } } + InitSofieClusterizer(true); return 0; } @@ -470,6 +485,13 @@ int32_t GPUChainTracking::ForceInitQA() int32_t GPUChainTracking::Finalize() { +#ifdef GPUCA_HAS_SOFIE + if (!mSofieApplications.empty()) { + const auto context = GetThreadContext(); + SynchronizeGPU(); + mSofieApplications.clear(); + } +#endif if (GetProcessingSettings().runQA && GetQA()->IsInitialized() && !(mConfigQA && mConfigQA->shipToQC) && !mQAFromForeignChain) { GetQA()->UpdateChain(this); GetQA()->DrawQAHistograms(); @@ -1006,3 +1028,67 @@ void GPUChainTracking::ApplySyncSettings(GPUSettingsProcessing& proc, GPUSetting steps.setBits(gpudatatypes::RecoStep::TPCdEdx, dEdxMode == -1 ? !syncMode : (dEdxMode > 0)); } } + +void GPUChainTracking::InitSofieClusterizer(bool deferCCDB) +{ + const auto& settings = GetProcessingSettings().nn; + if (settings.mlFramework != "ORT" && settings.mlFramework != "SOFIE") { + throw std::runtime_error("ml-framework must be ORT or SOFIE"); + } + if (!settings.applyNNclusterizer) { + return; + } + if (settings.mlFramework == "ORT") { +#ifndef GPUCA_HAS_ONNX + throw std::runtime_error("ORT was requested but GPUCA_BUILD_ORT is disabled or ONNXRuntime is unavailable"); +#endif + return; + } +#ifdef GPUCA_HAS_SOFIE + std::array buffers{}; + if (settings.nnLoadFromCCDB) { + for (size_t i = 0; i < buffers.size(); i++) { + const auto* network = processors()->calibObjects.nnClusterizerNetworks[i]; + if (network && network->getONNXModelSize()) { + buffers[i] = std::string_view(network->getONNXModel(), network->getONNXModelSize()); + } + } + } + const auto* previous = mSofieApplications.empty() ? nullptr : mSofieApplications.front().get(); + if (previous && (!settings.nnLoadFromCCDB || previous->hasSofieBuffers(buffers))) { + return; + } + const bool hip = mRec->GetDeviceType() == GPUReconstruction::DeviceType::HIP; + if ((!hip && mRec->GetDeviceType() != GPUReconstruction::DeviceType::CUDA) || !(GetRecoStepsGPU() & RecoStep::TPCClusterFinding)) { + throw std::runtime_error("SOFIE requires TPC cluster finding on CUDA or HIP"); + } + const int lanes = GetProcessingSettings().nTPCClustererLanes; + if (lanes < 1 || lanes > 4 || static_cast(lanes) > mRec->NStreams()) { + throw std::runtime_error("SOFIE clusterizer requires 1..4 lanes and a stream for each lane"); + } + if (deferCCDB && settings.nnLoadFromCCDB) { + return; + } + GPUInfo("SOFIE: %s, backend=%s, device=%d, lanes=%d, source=%s, batch capacity=%u", + previous ? "CCDB model change detected; reloading" : "initializing", + hip ? "HIP" : "CUDA", GetNativeGPUDevice(), lanes, + settings.nnLoadFromCCDB ? "CCDB" : "local ONNX files", settings.nnClusterizerBatchedMode); + std::vector> applications; + for (int lane = 0; lane < lanes; lane++) { + auto host = std::make_unique(); + host->initSofie(settings, GetNativeGPUStream(lane), GetNativeGPUDevice(), hip, lane ? applications.front().get() : nullptr, buffers, lane ? nullptr : previous); + GPUTPCNNClusterizer check; + host->initClusterizer(settings, check); + applications.push_back(std::move(host)); + } + if (previous) { + SynchronizeGPU(); + } + const bool reloaded = previous != nullptr; + mSofieApplications = std::move(applications); + GPUInfo("SOFIE: %s succeeded; inference backend active on device %d with %d lanes", + reloaded ? "CCDB model reload" : "initialization", GetNativeGPUDevice(), lanes); +#else + throw std::runtime_error("SOFIE was requested but GPUCA_BUILD_SOFIE is disabled"); +#endif +} diff --git a/GPU/GPUTracking/Global/GPUChainTracking.h b/GPU/GPUTracking/Global/GPUChainTracking.h index 759aaf818028e..5624f9c064126 100644 --- a/GPU/GPUTracking/Global/GPUChainTracking.h +++ b/GPU/GPUTracking/Global/GPUChainTracking.h @@ -56,6 +56,7 @@ namespace o2::gpu { // class GPUTRDTrackerGPU; class GPUTPCGPUTracker; +class GPUTPCNNClusterizerHost; class GPUDisplayInterface; class GPUQA; class GPUTPCClusterStatistics; @@ -298,6 +299,10 @@ class GPUChainTracking : public GPUChain int32_t RunChainFinalize(); void OutputSanityCheck(); int32_t RunTPCTrackingSectors_internal(); + void InitSofieClusterizer(bool deferCCDB = false); +#ifdef GPUCA_HAS_SOFIE + std::vector> mSofieApplications; +#endif int32_t RunTPCClusterizer_prepare(bool restorePointers, const GPUTPCExtraADC& extraADCs); #ifndef GPUCA_RUN2 std::pair RunTPCClusterizer_transferZS(int32_t iSector, const CfFragment& fragment, int32_t lane, const GPUTPCExtraADC& extraADCs); diff --git a/GPU/GPUTracking/Global/GPUChainTrackingClusterizer.cxx b/GPU/GPUTracking/Global/GPUChainTrackingClusterizer.cxx index f884b4fb8a10a..bf52a6b9421b8 100644 --- a/GPU/GPUTracking/Global/GPUChainTrackingClusterizer.cxx +++ b/GPU/GPUTracking/Global/GPUChainTrackingClusterizer.cxx @@ -45,7 +45,7 @@ #include "DataFormatsTPC/Constants.h" #include "TPCBase/RDHUtils.h" -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) #include "GPUTPCNNClusterizerKernels.h" #include "GPUTPCNNClusterizerHost.h" #include "ORTRootSerializer.h" @@ -809,7 +809,7 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) WriteToConstantMemory(RecoStep::TPCClusterFinding, (char*)processors()->tpcClusterer - (char*)processors(), processorsShadow()->tpcClusterer, sizeof(GPUTPCClusterFinder) * NSECTORS, mRec->NStreams() - 1, &mEvents->init); } -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) const GPUSettingsProcessingNNclusterizer& nn_settings = GetProcessingSettings().nn; GPUTPCNNClusterizerHost nnApplications[GetProcessingSettings().nTPCClustererLanes]; @@ -817,9 +817,12 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) HighResTimer* nnTimers[12]; if (nn_settings.applyNNclusterizer) { - int32_t deviceId = -1; + InitSofieClusterizer(); + int32_t deviceId = nn_settings.mlFramework == "SOFIE" ? GetNativeGPUDevice() : -1; int32_t numLanes = GetProcessingSettings().nTPCClustererLanes; +#ifdef GPUCA_HAS_ONNX int32_t maxThreads = mRec->getNKernelHostThreads(true); +#endif // bool recreateMemoryAllocator = false; if (GetProcessingSettings().debugLevel >= 1) { @@ -838,6 +841,18 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) } mRec->runParallelOuterLoop(doGPU, numLanes, [&](uint32_t lane) { + if (nn_settings.mlFramework == "SOFIE") { +#ifdef GPUCA_HAS_SOFIE + if (mSofieApplications.size() != static_cast(numLanes)) { + throw std::runtime_error("SOFIE clusterizer must be initialized before processing"); + } + nnApplications[lane].useSofie(*mSofieApplications[lane]); + return; +#else + throw std::runtime_error("SOFIE was not enabled in this build"); +#endif + } +#ifdef GPUCA_HAS_ONNX nnApplications[lane].init(nn_settings, GetProcessingSettings().deterministicGPUReconstruction); if (nnApplications[lane].mModelsUsed[0]) { SetONNXGPUStream(*(nnApplications[lane].mModelClass).getSessionOptions(), lane, &deviceId); @@ -893,6 +908,7 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) if (nn_settings.nnClusterizerVerbosity > 0) { LOG(info) << "(ORT) Allocated ONNX stream for lane " << lane << " and device " << deviceId; } +#endif }); const int16_t maxFragmentLen = GetProcessingSettings().overrideClusterizerFragmentLen; const uint32_t maxAllowedTimebin = param().par.continuousTracking ? std::max(param().continuousMaxTimeBin, maxFragmentLen) : constants::TPC_MAX_TIME_BIN_TRIGGERED; @@ -1223,11 +1239,13 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) const auto nRegularClusters = clusterer.mPmemory->counters.nClusters; if (nRegularClusters != 0) { if (GetProcessingSettings().nn.applyNNclusterizer) { -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) GPUTPCNNClusterizer& clustererNN = processors()->tpcNNClusterer[lane]; GPUTPCNNClusterizer& clustererNNShadow = doGPU ? processorsShadow()->tpcNNClusterer[lane] : clustererNN; GPUTPCNNClusterizerHost& nnApplication = nnApplications[lane]; + nnApplication.bindSofieWorkspace(clustererNNShadow.mSofieWorkspace, clustererNNShadow.mSofieWorkspaceSize); + // int withMC = (doGPU && propagateMCLabels); if (nn_settings.nnClusterizerApplyCfDeconvolution) { @@ -1271,15 +1289,15 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) if(GetProcessingSettings().debugLevel >= 1 && (doGPU || lane < 4)) { nnTimers[3*lane]->Start(); } if (clustererNNShadow.mNnInferenceInputDType == 0) { if (clustererNNShadow.mNnInferenceOutputDType == 0) { - (nnApplication.mModelClass).inference(clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mModelProbabilities_32); + nnApplication.inference(0, clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mModelProbabilities_32); } else { - (nnApplication.mModelClass).inference(clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mModelProbabilities_16); + nnApplication.inference(0, clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mModelProbabilities_16); } } else if (clustererNNShadow.mNnInferenceInputDType == 1) { if (clustererNNShadow.mNnInferenceOutputDType == 0) { - (nnApplication.mModelClass).inference(clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mModelProbabilities_32); + nnApplication.inference(0, clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mModelProbabilities_32); } else { - (nnApplication.mModelClass).inference(clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mModelProbabilities_16); + nnApplication.inference(0, clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mModelProbabilities_16); } } if(GetProcessingSettings().debugLevel >= 1 && (doGPU || lane < 4)) { nnTimers[3*lane]->Stop(); } // doGPU || lane<4 -> only for GPU or first 4 CPU lanes (to limit number of concurrent timers). At least gives some statistics for CPU time... @@ -1291,31 +1309,31 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) if(GetProcessingSettings().debugLevel >= 1 && (doGPU || lane < 4)) { nnTimers[3*lane + 1]->Start(); } if (clustererNNShadow.mNnInferenceInputDType == 0) { if (clustererNNShadow.mNnInferenceOutputDType == 0) { - (nnApplication.mModelReg1).inference(clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg1_32); + nnApplication.inference(1, clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg1_32); } else { - (nnApplication.mModelReg1).inference(clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg1_16); + nnApplication.inference(1, clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg1_16); } } else { if (clustererNNShadow.mNnInferenceOutputDType == 0) { - (nnApplication.mModelReg1).inference(clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg1_32); + nnApplication.inference(1, clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg1_32); } else { - (nnApplication.mModelReg1).inference(clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg1_16); + nnApplication.inference(1, clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg1_16); } } if(GetProcessingSettings().debugLevel >= 1 && (doGPU || lane < 4)) { nnTimers[3*lane + 1]->Stop(); } - if (nnApplication.mModelClass.getNumOutputNodes()[0][1] > 1 && nnApplication.mModelReg2.isInitialized()) { + if (nnApplication.modelOutputs(0) > 1 && nnApplication.mModelsUsed[2]) { if(GetProcessingSettings().debugLevel >= 1 && (doGPU || lane < 4)) { nnTimers[3*lane + 2]->Start(); } if (clustererNNShadow.mNnInferenceInputDType == 0) { if (clustererNNShadow.mNnInferenceOutputDType == 0) { - (nnApplication.mModelReg2).inference(clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg2_32); + nnApplication.inference(2, clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg2_32); } else { - (nnApplication.mModelReg2).inference(clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg2_16); + nnApplication.inference(2, clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg2_16); } } else if (clustererNNShadow.mNnInferenceInputDType == 1) { if (clustererNNShadow.mNnInferenceOutputDType == 0) { - (nnApplication.mModelReg2).inference(clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg2_32); + nnApplication.inference(2, clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg2_32); } else { - (nnApplication.mModelReg2).inference(clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg2_16); + nnApplication.inference(2, clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg2_16); } } if(GetProcessingSettings().debugLevel >= 1 && (doGPU || lane < 4)) { nnTimers[3*lane + 2]->Stop(); } @@ -1327,14 +1345,14 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) // Publishing kernels for class labels and regression results // In case classification should not be used, this kernel should still be executed to fill the mOutputDataClass array with default values - if (nnApplication.mModelClass.getNumOutputNodes()[0][1] == 1) { + if (nnApplication.modelOutputs(0) == 1) { runKernel({GetGrid(iSize, lane), krnlRunRangeNone}, iSector, clustererNNShadow.mNnInferenceOutputDType, propagateMCLabels, batchStart); // Assigning class labels } else { runKernel({GetGrid(iSize, lane), krnlRunRangeNone}, iSector, clustererNNShadow.mNnInferenceOutputDType, propagateMCLabels, batchStart); // Assigning class labels } if (!clustererNNShadow.mNnClusterizerUseCfRegression) { runKernel({GetGrid(iSize, lane), krnlRunRangeNone}, iSector, clustererNNShadow.mNnInferenceOutputDType, propagateMCLabels, batchStart); // Publishing class 1 regression results - if (nnApplication.mModelClass.getNumOutputNodes()[0][1] > 1 && nnApplication.mModelReg2.isInitialized()) { + if (nnApplication.modelOutputs(0) > 1 && nnApplication.mModelsUsed[2]) { runKernel({GetGrid(iSize, lane), krnlRunRangeNone}, iSector, clustererNNShadow.mNnInferenceOutputDType, propagateMCLabels, batchStart); // Publishing class 2 regression results } } @@ -1485,7 +1503,7 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) } for (int32_t i = 0; i < GetProcessingSettings().nTPCClustererLanes; i++) { #ifdef GPUCA_HAS_ONNX - if (GetProcessingSettings().nn.applyNNclusterizer) { + if (GetProcessingSettings().nn.applyNNclusterizer && GetProcessingSettings().nn.mlFramework == "ORT") { if (GetProcessingSettings().nn.nnClusterizerVerbosity > 0) { LOG(info) << "(ORT) Environment releasing..."; } diff --git a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.cxx b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.cxx index 0e77393be1ce3..96cb234619b02 100644 --- a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.cxx +++ b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.cxx @@ -145,6 +145,9 @@ void* GPUTPCNNClusterizer::setIOPointers(void* mem) } } + if (mSofieWorkspaceSize) { + computePointerWithAlignment<256>(mem, mSofieWorkspace, mSofieWorkspaceSize); + } return mem; } diff --git a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.h b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.h index 7aa23489eb2f4..578a39f1f31b7 100644 --- a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.h +++ b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.h @@ -101,6 +101,8 @@ class GPUTPCNNClusterizer : public GPUProcessor OrtDataType::Float16_t* mOutputDataReg1_16 = nullptr; OrtDataType::Float16_t* mOutputDataReg2_16 = nullptr; + char* mSofieWorkspace = nullptr; + size_t mSofieWorkspaceSize = 0; int16_t mMemoryId = -1; }; // class GPUTPCNNClusterizer diff --git a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx index 96b8a1d7ed2fd..7068cf8dfec2b 100644 --- a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx +++ b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx @@ -12,6 +12,10 @@ /// \file GPUTPCNNClusterizerHost.cxx /// \author Christian Sonnabend +#ifdef GPUCA_HAS_SOFIE +#include "Rtypes.h" +#endif + #include #include "GPUTPCNNClusterizerHost.h" @@ -19,6 +23,7 @@ #include "GPUSettings.h" #include "ML/3rdparty/GPUORTFloat16.h" #include "GPUReconstruction.h" +#include "GPULogging.h" #include "GPUTPCGeometry.h" #include "DataFormatsTPC/Constants.h" #include "clusterFinderDefs.h" @@ -27,10 +32,31 @@ #include #endif +#ifdef GPUCA_HAS_SOFIE +#include +#include +#endif +#include +#include +#include + using namespace o2::gpu; +struct GPUTPCNNClusterizerHost::SofieState { +#ifdef GPUCA_HAS_SOFIE + std::array, 3> models; + std::array, 3> sessions; +#endif + std::array buffers; + size_t workspaceSize = 0; + unsigned int maxBatch = 0; +}; + void GPUTPCNNClusterizerHost::init(const GPUSettingsProcessingNNclusterizer& settings, bool useDeterministicMode) { +#ifndef GPUCA_HAS_ONNX + throw std::runtime_error("ONNXRuntime was not enabled in this build"); +#else std::string class_model_path = settings.nnClassificationPath, reg_model_path = settings.nnRegressionPath; std::vector reg_model_paths_local; std::vector evalMode = o2::utils::Str::tokenize(settings.nnEvalMode, ':'); @@ -83,6 +109,7 @@ void GPUTPCNNClusterizerHost::init(const GPUSettingsProcessingNNclusterizer& set mModelsUsed[2] = true; } } +#endif } void GPUTPCNNClusterizerHost::initClusterizer(const GPUSettingsProcessingNNclusterizer& settings, GPUTPCNNClusterizer& clustererNN, int32_t maxFragmentLen, int32_t maxAllowedTimebin) @@ -134,13 +161,29 @@ void GPUTPCNNClusterizerHost::initClusterizer(const GPUSettingsProcessingNNclust } else { clustererNN.mNnInferenceOutputDType = 1; // Default to float16 } - clustererNN.mNnClusterizerModelClassNumOutputNodes = mModelClass.getNumOutputNodes()[0][1]; +#ifdef GPUCA_HAS_SOFIE + if (mSofie) { + if (settings.nnClusterizerBatchedMode != mSofie->maxBatch) { + throw std::runtime_error("Changing SOFIE batch capacity requires chain reinitialization"); + } + clustererNN.mSofieWorkspaceSize = mSofie->workspaceSize; + for (const auto& model : mSofie->models) { + if (model && ((model->GetPrecision() == TMVA::Experimental::SOFIE::RGPUModel::Precision::Float16) != (clustererNN.mNnInferenceInputDType == 1) || clustererNN.mNnInferenceInputDType != clustererNN.mNnInferenceOutputDType)) { + throw std::runtime_error("Changing SOFIE precision requires chain reinitialization"); + } + if (model && model->InputSize() != static_cast(clustererNN.mNnClusterizerElementSize)) { + throw std::runtime_error("SOFIE input shape does not match the configured clusterizer window"); + } + } + } +#endif + clustererNN.mNnClusterizerModelClassNumOutputNodes = modelOutputs(0); if (!settings.nnClusterizerUseCfRegression) { - if (mModelClass.getNumOutputNodes()[0][1] == 1 || !mModelReg2.isInitialized()) { - clustererNN.mNnClusterizerModelReg1NumOutputNodes = mModelReg1.getNumOutputNodes()[0][1]; + if (modelOutputs(0) == 1 || !mModelsUsed[2]) { + clustererNN.mNnClusterizerModelReg1NumOutputNodes = modelOutputs(1); } else { - clustererNN.mNnClusterizerModelReg1NumOutputNodes = mModelReg1.getNumOutputNodes()[0][1]; - clustererNN.mNnClusterizerModelReg2NumOutputNodes = mModelReg2.getNumOutputNodes()[0][1]; + clustererNN.mNnClusterizerModelReg1NumOutputNodes = modelOutputs(1); + clustererNN.mNnClusterizerModelReg2NumOutputNodes = modelOutputs(2); } } } @@ -178,6 +221,7 @@ void GPUTPCNNClusterizerHost::initClusterizer(const GPUSettingsProcessingNNclust // } // } +#ifdef GPUCA_HAS_ONNX // MockedOrtAllocator implementation to be able to use volatile assignment struct MockedOrtAllocator : OrtAllocator { MockedOrtAllocator(GPUReconstruction* = nullptr, OrtMemoryInfo* = nullptr); @@ -278,3 +322,194 @@ MockedOrtAllocator* GPUTPCNNClusterizerHost::getMockedAllocator() { return mMockedAlloc.get(); } + +#endif + +void GPUTPCNNClusterizerHost::initSofie(const GPUSettingsProcessingNNclusterizer& settings, void* stream, int32_t device, bool hip, const GPUTPCNNClusterizerHost* source, const std::array& buffers, const GPUTPCNNClusterizerHost* previous) +{ +#ifdef GPUCA_HAS_SOFIE + using Model = TMVA::Experimental::SOFIE::RGPUModel; + if (!settings.nnClusterizerBatchedMode || settings.nnClusterizerBatchedMode > static_cast(std::numeric_limits::max())) { + throw std::runtime_error("SOFIE requires a positive batch capacity within int32 range"); + } + auto precision = [](const std::string& value) { + if (value == "FP32" || value == "fp32") { + return Model::Precision::Float32; + } + if (value == "FP16" || value == "fp16") { + return Model::Precision::Float16; + } + throw std::runtime_error("SOFIE precision must be FP32 or FP16"); + }; + const auto inputType = precision(settings.nnInferenceInputDType); + if (inputType != precision(settings.nnInferenceOutputDType)) { + throw std::runtime_error("SOFIE requires matching model, input and output precision"); + } + if (settings.nnInferenceDevice != (hip ? "rocm" : "cuda") && settings.nnInferenceDevice != (hip ? "ROCM" : "CUDA")) { + throw std::runtime_error("SOFIE inference device must match the reconstruction backend (cuda or rocm)"); + } + size_t inputSize = 1; + for (int radius : {settings.nnClusterizerSizeInputRow, settings.nnClusterizerSizeInputPad, settings.nnClusterizerSizeInputTime}) { + if (radius < 0 || radius > 16383 || inputSize > static_cast(std::numeric_limits::max()) / (2 * radius + 1)) { + throw std::runtime_error("SOFIE clusterizer input window is too large or invalid"); + } + inputSize *= 2 * radius + 1; + } + inputSize += settings.nnClusterizerAddIndexData ? 3 : 0; + if (inputSize > static_cast(std::numeric_limits::max()) / settings.nnClusterizerBatchedMode) { + throw std::runtime_error("SOFIE input batch exceeds the clusterizer's index range"); + } + auto state = std::make_shared(); + state->maxBatch = settings.nnClusterizerBatchedMode; + std::array paths{settings.nnClassificationPath, "", ""}; + if (settings.nnLoadFromCCDB) { + auto mode = o2::utils::Str::tokenize(settings.nnEvalMode, ':'); + if (mode.size() != 2 || (mode[0] != "c1" && mode[0] != "c2") || (mode[1] != "r1" && mode[1] != "r2")) { + throw std::runtime_error("SOFIE CCDB loading requires nnEvalMode c1:r1, c1:r2, c2:r1 or c2:r2"); + } + paths[0] = "CCDB classification"; + if (!settings.nnClusterizerUseCfRegression) { + paths[1] = "CCDB regression 1"; + if (mode[1] == "r2") { + paths[2] = "CCDB regression 2"; + } + } + } else if (!settings.nnClusterizerUseCfRegression) { + auto regression = o2::utils::Str::tokenize(settings.nnRegressionPath, ':'); + if (regression.empty() || regression.size() > 2) { + throw std::runtime_error("SOFIE expects one or two regression paths"); + } + for (size_t i = 0; i < regression.size(); i++) { + paths[i + 1] = regression[i]; + } + } + TMVA::Experimental::SOFIE::RModelParser_ONNX parser; + std::unordered_map> compiled; + if (settings.nnLoadFromCCDB && previous && previous->mSofie) { + for (size_t i = 0; i < previous->mSofie->models.size(); i++) { + if (previous->mSofie->models[i] && !previous->mSofie->buffers[i].empty()) { + compiled.emplace(previous->mSofie->buffers[i], previous->mSofie->models[i]); + } + } + } + for (size_t i = 0; i < paths.size(); i++) { + mModelsUsed[i] = !paths[i].empty(); + if (!mModelsUsed[i]) { + continue; + } + if (source) { + state->models[i] = source->mSofie->models[i]; + } else { + if (settings.nnLoadFromCCDB && buffers[i].empty()) { + throw std::runtime_error("Missing or empty ONNX buffer for " + paths[i]); + } + auto& model = compiled[settings.nnLoadFromCCDB ? std::string(buffers[i]) : paths[i]]; + if (settings.nnLoadFromCCDB) { + state->buffers[i] = buffers[i]; + } + if (!model) { + GPUInfo("SOFIE: preparing model %zu (%s), backend=%s, architecture=%s, precision=%s", + i, paths[i].c_str(), hip ? "HIP" : "CUDA", settings.sofieArchitecture.c_str(), + inputType == Model::Precision::Float16 ? "FP16" : "FP32"); + if (settings.nnLoadFromCCDB) { + std::istringstream input(state->buffers[i], std::ios::in | std::ios::binary); + model = std::make_shared(parser.ParseGPU(input)); + } else { + model = std::make_shared(parser.ParseGPU(paths[i])); + } + if (model->GetPrecision() != inputType) { + throw std::runtime_error("SOFIE model precision differs from nnInferenceInputDType: " + paths[i]); + } + if (model->InputSize() != inputSize || model->OutputSize() > static_cast(std::numeric_limits::max()) / settings.nnClusterizerBatchedMode) { + throw std::runtime_error("SOFIE model shape does not match the clusterizer window or index range: " + paths[i]); + } + model->WorkspaceSize(settings.nnClusterizerBatchedMode); + model->Compile({hip ? Model::Backend::HIP : Model::Backend::CUDA, settings.sofieCompiler, settings.sofieArchitecture}); + GPUInfo("SOFIE: model %zu compilation succeeded, input=%zu, output=%zu", + i, model->InputSize(), model->OutputSize()); + } else { + GPUInfo("SOFIE: model %zu reuses an existing compiled program", i); + } + state->models[i] = model; + } + state->sessions[i] = state->models[i]->CreateSession(stream, device, settings.nnClusterizerBatchedMode); + if (state->sessions[i]->WorkspaceSize() > std::numeric_limits::max() - state->workspaceSize) { + throw std::overflow_error("SOFIE combined workspace size overflow"); + } + state->workspaceSize += state->sessions[i]->WorkspaceSize(); + } + mSofie = std::move(state); + mDeviceId = device; +#else + throw std::runtime_error("SOFIE was not enabled in this build"); +#endif +} + +bool GPUTPCNNClusterizerHost::hasSofieBuffers(const std::array& buffers) const +{ + if (!mSofie) { + return false; + } + for (size_t i = 0; i < buffers.size(); i++) { + if (mModelsUsed[i] && buffers[i] != mSofie->buffers[i]) { + return false; + } + } + return true; +} + +void GPUTPCNNClusterizerHost::useSofie(const GPUTPCNNClusterizerHost& source) +{ + mSofie = source.mSofie; + mModelsUsed = source.mModelsUsed; + mDeviceId = source.mDeviceId; +} + +void GPUTPCNNClusterizerHost::bindSofieWorkspace(void* workspace, size_t bytes) +{ +#ifdef GPUCA_HAS_SOFIE + if (!mSofie) { + return; + } + if (!workspace || bytes < mSofie->workspaceSize) { + throw std::runtime_error("O2 supplied insufficient SOFIE workspace"); + } + auto* next = static_cast(workspace); + for (const auto& session : mSofie->sessions) { + if (session) { + session->SetWorkspace(next, session->WorkspaceSize()); + next += session->WorkspaceSize(); + } + } +#endif +} + +void GPUTPCNNClusterizerHost::inferenceSofie(int model, const void* input, size_t batch, void* output) +{ +#ifdef GPUCA_HAS_SOFIE + if (!mSofie || model < 0 || model >= 3 || !mSofie->sessions[model]) { + throw std::runtime_error("SOFIE model is not initialized"); + } + mSofie->sessions[model]->Infer(input, output, batch); +#else + throw std::runtime_error("SOFIE was not enabled in this build"); +#endif +} + +int32_t GPUTPCNNClusterizerHost::modelOutputs(int model) const +{ +#ifdef GPUCA_HAS_SOFIE + if (mSofie) { + return mSofie->models.at(model) ? static_cast(mSofie->models.at(model)->OutputSize()) : 0; + } +#endif +#ifdef GPUCA_HAS_ONNX + return (model == 0 ? mModelClass : model == 1 ? mModelReg1 + : mModelReg2) + .getNumOutputNodes() + .at(0) + .at(1); +#else + throw std::runtime_error("No initialized inference backend"); +#endif +} diff --git a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.h b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.h index 8f8465d5dca34..29a02d3147bff 100644 --- a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.h +++ b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.h @@ -15,10 +15,16 @@ #ifndef O2_GPUTPCNNCLUSTERIZERHOST_H #define O2_GPUTPCNNCLUSTERIZERHOST_H +#include +#include #include #include #include +#include +#include +#ifdef GPUCA_HAS_ONNX #include "ML/OrtInterface.h" +#endif class OrtMemoryInfo; class OrtAllocator; @@ -52,6 +58,29 @@ class GPUTPCNNClusterizerHost void createBoundary(GPUTPCNNClusterizer&); void createIndexLookup(GPUTPCNNClusterizer&); + void initSofie(const GPUSettingsProcessingNNclusterizer&, void* stream, int32_t device, bool hip, const GPUTPCNNClusterizerHost* source = nullptr, const std::array& buffers = {}, const GPUTPCNNClusterizerHost* previous = nullptr); + bool hasSofieBuffers(const std::array& buffers) const; + void useSofie(const GPUTPCNNClusterizerHost&); + void bindSofieWorkspace(void* workspace, size_t bytes); + void inferenceSofie(int model, const void* input, size_t batch, void* output); + int32_t modelOutputs(int model) const; + template + void inference(int model, Input* input, size_t batch, Output* output) + { + if (mSofie) { + inferenceSofie(model, input, batch, output); + return; + } +#ifdef GPUCA_HAS_ONNX + (model == 0 ? mModelClass : model == 1 ? mModelReg1 + : mModelReg2) + .inference(input, batch, output); +#else + throw std::runtime_error("ONNXRuntime was not enabled in this build"); +#endif + } + +#ifdef GPUCA_HAS_ONNX // ONNX void directOrtAllocator(Ort::Env*, Ort::MemoryInfo*, GPUReconstruction*, bool = false); MockedOrtAllocator* getMockedAllocator(); @@ -59,9 +88,14 @@ class GPUTPCNNClusterizerHost std::unordered_map mOrtOptions; o2::ml::OrtModel mModelClass, mModelReg1, mModelReg2; // For splitting clusters + std::shared_ptr mMockedAlloc = nullptr; +#endif std::vector mModelsUsed = {false, false, false}; // 0: class, 1: reg_1, 2: reg_2 int32_t mDeviceId = -1; - std::shared_ptr mMockedAlloc = nullptr; + + private: + struct SofieState; + std::shared_ptr mSofie; }; // class GPUTPCNNClusterizerHost } // namespace o2::gpu diff --git a/GPU/GPUTracking/kernels.cmake b/GPU/GPUTracking/kernels.cmake index ed155788b0bef..19835d4b58ee9 100644 --- a/GPU/GPUTracking/kernels.cmake +++ b/GPU/GPUTracking/kernels.cmake @@ -26,7 +26,7 @@ o2_gpu_kernel_file_list(TPCDECOMPRESSION GPUTPCCompressionTrackModel.cxx ERRORS) o2_gpu_kernel_file_list(TPCCLUSTERFINDER ERRORS ClusterAccumulator.cxx) o2_gpu_kernel_file_list(TRDTRACKER GPUTRDTrack.cxx GPUTRDTracker.cxx GPUTRDTrackletWord.cxx GeometryBase.cxx) o2_gpu_kernel_file_list(GLOBALREFIT TPCMERGER O2PROPAGATOR MATLUT GPUTrackingRefit.cxx) -if(onnxruntime_FOUND) +if(onnxruntime_FOUND OR GPUCA_BUILD_SOFIE) o2_gpu_kernel_file_list(TPCNNCLUSTERFINDER ERRORS ClusterAccumulator.cxx GPUTPCNNClusterizerKernels.cxx) endif() @@ -126,7 +126,7 @@ o2_gpu_add_kernel("GPUTPCCFDecodeZSDenseLink" "GPUTP o2_gpu_add_kernel("GPUTPCCFGather" "=" LB o2::tpc::ClusterNative* dest) o2_gpu_add_kernel("GPUTrackingRefitKernel, mode0asGPU" "= GLOBALREFIT " LB) o2_gpu_add_kernel("GPUTrackingRefitKernel, mode1asTrackParCov" "= GLOBALREFIT " LB) -if(onnxruntime_FOUND) +if(onnxruntime_FOUND OR GPUCA_BUILD_SOFIE) o2_gpu_add_kernel("GPUTPCNNClusterizerKernels, runCfClusterizer" "= TPCNNCLUSTERFINDER" LB uint8_t sector int8_t dtype int8_t withMC uint32_t batchStart) o2_gpu_add_kernel("GPUTPCNNClusterizerKernels, fillInputNNCPU" "= TPCNNCLUSTERFINDER" LB uint8_t sector int8_t dtype int8_t withMC uint32_t batchStart) o2_gpu_add_kernel("GPUTPCNNClusterizerKernels, fillInputNNGPU" "= TPCNNCLUSTERFINDER" LB uint8_t sector int8_t dtype int8_t withMC uint32_t batchStart) diff --git a/GPU/GPUTracking/qa/GPUQA.cxx b/GPU/GPUTracking/qa/GPUQA.cxx index 060766b54f3f4..44c4711b04df3 100644 --- a/GPU/GPUTracking/qa/GPUQA.cxx +++ b/GPU/GPUTracking/qa/GPUQA.cxx @@ -525,6 +525,10 @@ int32_t GPUQA::InitQACreateHistograms() createHist(mClusters[i], name, name, AXIS_BINS[4], binsPt.get()); } + const int32_t maxPads = GPUTPCGeometry::NPads(GPUTPCGeometry::NROWS - 1); + createHist(mLowestRowDifferenceVsPad, "lowest_row_difference_vs_pad", "Lowest cluster row difference;Pad of lowest reconstructed cluster;Lowest reconstructed row - lowest MC row", maxPads, -0.5, maxPads - 0.5, 2 * GPUTPCGeometry::NROWS - 1, -GPUTPCGeometry::NROWS + 0.5, GPUTPCGeometry::NROWS - 0.5); + createHist(mLowestRowTrackMultiplicity, "lowest_row_track_multiplicity", "Track multiplicity for missing inner clusters;Reconstructed tracks with the same MC label;Selected reconstructed tracks", 100, -0.5, 99.5); + createHist(mPadRow[0], "padrow0", "padrow0", GPUTPCGeometry::NROWS - PADROW_CHECK_MINCLS, 0, GPUTPCGeometry::NROWS - 1 - PADROW_CHECK_MINCLS, GPUTPCGeometry::NROWS - PADROW_CHECK_MINCLS, 0, GPUTPCGeometry::NROWS - 1 - PADROW_CHECK_MINCLS); createHist(mPadRow[1], "padrow1", "padrow1", 100.f, -0.2f, 0.2f, GPUTPCGeometry::NROWS - PADROW_CHECK_MINCLS, 0, GPUTPCGeometry::NROWS - 1 - PADROW_CHECK_MINCLS); createHist(mPadRow[2], "padrow2", "padrow2", 100.f, -0.2f, 0.2f, GPUTPCGeometry::NROWS - PADROW_CHECK_MINCLS, 0, GPUTPCGeometry::NROWS - 1 - PADROW_CHECK_MINCLS); @@ -1103,12 +1107,35 @@ void GPUQA::RunQA(bool matchOnly, const std::vector* tracksEx } } if (mTracking->mIOPtrs.nMergedTracks && clNative) { + // Keep these per MC label: the reverse label map retains only one clone. + struct LowestRowInfo { + int32_t row = GPUTPCGeometry::NROWS; + int32_t nTracks = 0; + }; + std::vector> lowestRowInfo(GetNMCCollissions()); + for (uint32_t iCol = 0; iCol < GetNMCCollissions(); iCol++) { + lowestRowInfo[iCol].resize(GetNMCTracks(iCol)); + } + for (uint32_t i = 0; i < nReconstructedTracks; i++) { + const auto& label = mTrackMCLabels[i]; + // Count all reconstructed segments, including fake-flagged label matches, + // without applying the selection cuts to the other segments. + if (mTracking->mIOPtrs.mergedTracks[i].OK() && label.isValid() && !label.isNoise()) { + GetMCTrackObj(lowestRowInfo, label).nTracks++; + } + } std::fill(lowestPadRow.begin(), lowestPadRow.end(), 255); for (uint32_t iSector = 0; iSector < GPUTPCGeometry::NSECTORS; iSector++) { for (uint32_t iRow = 0; iRow < GPUTPCGeometry::NROWS; iRow++) { for (uint32_t iCl = 0; iCl < clNative->nClusters[iSector][iRow]; iCl++) { int32_t i = clNative->clusterOffset[iSector][iRow] + iCl; for (int32_t j = 0; j < GetMCLabelNID(i); j++) { + const mcLabelI_t label = GetMCLabel(i, j); + if (!label.isValid() || label.isNoise()) { + continue; + } + auto& info = GetMCTrackObj(lowestRowInfo, label); + info.row = std::min(info.row, (int32_t)iRow); uint32_t trackId = GetMCTrackObj(mTrackMCLabelsReverse, GetMCLabel(i, j)); if (trackId < lowestPadRow.size() && lowestPadRow[trackId] > iRow) { lowestPadRow[trackId] = iRow; @@ -1119,6 +1146,28 @@ void GPUQA::RunQA(bool matchOnly, const std::vector* tracksEx } for (uint32_t i = 0; i < mTracking->mIOPtrs.nMergedTracks; i++) { const auto& trk = mTracking->mIOPtrs.mergedTracks[i]; + const auto& label = mTrackMCLabels[i]; + if (trk.OK() && trk.NClustersFitted() > 50 && CAMath::Abs(trk.GetParam().GetQPt()) < 1.f && label.isValid() && !label.isNoise() && + (mMCTrackMin == -1 || label.getTrackID() >= mMCTrackMin) && (mMCTrackMax == -1 || label.getTrackID() < mMCTrackMax)) { + const auto& info = GetMCTrackObj(lowestRowInfo, label); + if (info.row < 10) { + const GPUTPCGMMergedTrackHit* lowestCl = nullptr; + for (uint32_t j = 0; j < trk.NClusters(); j++) { + const auto& cl = mTracking->mIOPtrs.mergedTrackHits[trk.FirstClusterRef() + j]; + if (!(cl.state & GPUTPCGMMergedTrackHit::flagReject) && (!lowestCl || cl.row < lowestCl->row)) { + lowestCl = &cl; + } + } + if (lowestCl) { + const float pad = clNative->clustersLinear[lowestCl->num].getPad(); + const int32_t difference = (int32_t)lowestCl->row - info.row; + mLowestRowDifferenceVsPad->Fill(pad, difference); + if (CAMath::Abs(difference) > 5 && pad > 5.f && pad < GPUTPCGeometry::NPads(lowestCl->row) - 5.f) { + mLowestRowTrackMultiplicity->Fill(info.nTracks); + } + } + } + } if (trk.OK() && lowestPadRow[i] != 255 && trk.NClustersFitted() >= PADROW_CHECK_MINCLS && CAMath::Abs(trk.GetParam().GetQPt()) < 1.0) { const auto& lowestCl = mTracking->mIOPtrs.mergedTrackHits[trk.FirstClusterRef()].row < mTracking->mIOPtrs.mergedTrackHits[trk.FirstClusterRef() + trk.NClusters() - 1].row ? mTracking->mIOPtrs.mergedTrackHits[trk.FirstClusterRef()] : mTracking->mIOPtrs.mergedTrackHits[trk.FirstClusterRef() + trk.NClusters() - 1]; const int32_t lowestRow = lowestCl.row; @@ -2912,6 +2961,17 @@ int32_t GPUQA::DrawQAHistograms(TObjArray* qcout) } } + if (mQATasks & taskClusterAttach) { + for (TH1* h : {static_cast(mLowestRowDifferenceVsPad), static_cast(mLowestRowTrackMultiplicity)}) { + if (tout && !mConfig.inputHistogramsOnly) { + h->Write(); + } + if (qcout) { + qcout->Add(h); + } + } + } + // Process cluster count statistics if ((mQATasks & taskClusterCounts) && !mHaveExternalHists && !mConfig.clusterRejectionHistograms && !mConfig.inputHistogramsOnly) { DoClusterCounts(attachClusterCounts); diff --git a/GPU/GPUTracking/qa/GPUQA.h b/GPU/GPUTracking/qa/GPUQA.h index b3c518e9798e8..f257c9a418659 100644 --- a/GPU/GPUTracking/qa/GPUQA.h +++ b/GPU/GPUTracking/qa/GPUQA.h @@ -318,6 +318,9 @@ class GPUQA TPad* mPClRej[3]; TPad* mPClRejP; + TH2F* mLowestRowDifferenceVsPad; + TH1F* mLowestRowTrackMultiplicity; + TH2F* mPadRow[4]; TCanvas* mCPadRow[4]; TPad* mPPadRow[4]; diff --git a/GPU/README.md b/GPU/README.md index 1aa6575e0ae67..9ca58abdf8cf5 100644 --- a/GPU/README.md +++ b/GPU/README.md @@ -10,3 +10,60 @@ This module contains the following submodules: * \subpage refGPUTrackingStandalone * \subpage refGPUTrackingDisplayFilterMacros /doxy --> + +## Experimental SOFIE NN clusterizer + +Build against the ROOT development package containing `TMVA/RGPUModel.hxx`: + +- `GPUCA_BUILD_SOFIE=ON` enables the SOFIE host integration. +- `GPUCA_BUILD_ORT=ON` (default) retains ORT; set it to `OFF` for a SOFIE-only + GPUTracking backend. Other O2 components may still require ONNXRuntime. +- SOFIE is disabled by default. This change does not run a compiler at O2 build + configuration time to generate models. + +Select `ml-framework=SOFIE` (the `mlFramework` field in the NN settings), or +`ORT` to retain the existing path. For SOFIE, also set `sofie-architecture` +(`sofieArchitecture`) to the target GPU architecture, for example `sm_80` for +CUDA or `gfx90a` for HIP. `sofie-compiler` (`sofieCompiler`) optionally supplies +an explicit compiler path; otherwise `nvcc`/`hipcc` is found on PATH at runtime. +The compiler and GPU toolkit must be available on the worker at startup. + +Use GPU TPC cluster finding, `nnInferenceDevice=cuda` or `rocm` as appropriate, +local classification/regression ONNX paths with `nnLoadFromCCDB=0`, or +`nnLoadFromCCDB=1` to parse the ONNX buffers fetched by the existing CCDB workflow. +CCDB model slots follow `nnEvalMode` (classification, regression 1, optional regression 2). Set both +`nnInferenceInputDType` and `nnInferenceOutputDType` to `FP32` or `FP16` to match +the model. Mixed input/output precision, unsupported graphs, CPU SOFIE inference +are rejected explicitly. + +Each distinct local model path is parsed and compiled once during chain initialization. +For CCDB, initialization is deferred until the first event has loaded its calibration +objects, before NN inference. Identical model buffers share a compiled program. +Later events reuse the sessions while the model bytes are unchanged. When CCDB +provides changed bytes, only new model contents are parsed and compiled, sharing +unchanged programs across lanes. All replacement sessions are prepared and +validated before synchronizing GPU work and replacing the old sessions. A failed +reload propagates an error without replacing the active sessions. O2 sizes the +workspace for the new models before inference. +Missing or empty required buffers fail explicitly. No temporary ONNX files are needed. +Each lane then gets a session bound to its existing native stream. O2 allocates +input, output, weights and intermediate storage through its registered clusterizer +memory. When O2 recycles the scratch arena, weights are uploaded again on the +lane's stream; this does not recompile the model. Sessions persist until a model +reload or chain finalization, with GPU synchronization before unloading their libraries. +Changing local model paths, batch capacity, precision or backend requires reinitializing the chain. + +The supplied classification/regression networks have six Gemm layers, five Relu +layers, 246 input values and respectively 1/5 output values. The clusterizer's +input-window and index-data settings must reproduce the training layout; +the default window does not have 246 values. `nnClusterizerBatchedMode` sets +the maximum batch; smaller final batches reuse the same code and workspace. + +Validation to run after building: ROOT's `TestSofieGPU` and CUDA/HIP +`TestSofieGPUDevice`, then compare SOFIE against ORT using both supplied +precisions and multiple batch sizes. Check numerical tolerances and downstream +cluster decisions; the scalar dense kernels are not performance tuned. For CCDB, +check identical bytes in a new buffer, one changed model, changed hidden-layer +sizes, and invalid replacement data. Only changed contents should compile; an +invalid replacement must fail before switching sessions. +No ROOT/O2 build or GPU test was run while implementing this change. diff --git a/GPU/Workflow/src/GPUWorkflowSpec.cxx b/GPU/Workflow/src/GPUWorkflowSpec.cxx index 990873f053fe3..9895ef93c6e54 100644 --- a/GPU/Workflow/src/GPUWorkflowSpec.cxx +++ b/GPU/Workflow/src/GPUWorkflowSpec.cxx @@ -1120,6 +1120,7 @@ void GPURecoWorkflowSpec::doCalibUpdates(o2::framework::ProcessingContext& pc, c needCalibUpdate = true; } if (mSpecConfig.nnLoadFromCCDB) { + needCalibUpdate |= mConfig->configProcessing.nn.mlFramework == "SOFIE"; auto dumpToFile = [](const char* buffer, std::size_t validSize, const std::string& path) { std::ofstream out(path, std::ios::binary | std::ios::trunc); if (!out.is_open()) { diff --git a/dependencies/FindFairRoot.cmake b/dependencies/FindFairRoot.cmake index b48d2cad67c4b..a13378314bf7e 100644 --- a/dependencies/FindFairRoot.cmake +++ b/dependencies/FindFairRoot.cmake @@ -60,6 +60,9 @@ if(NOT TARGET FairRoot::GeoBase) set_target_properties(FairRoot::GeoBase PROPERTIES INTERFACE_INCLUDE_DIRECTORIES ${FairRoot_INC} INTERFACE_LINK_LIBRARIES ${FairRoot_GeoBase}) + if(TARGET ROOT::TGeometry) + target_link_libraries(FairRoot::GeoBase INTERFACE ROOT::TGeometry) + endif() endif() if(NOT TARGET FairRoot::Base)