From 45ea3f97f5461b5dbf1156143dceaa651a7dec56 Mon Sep 17 00:00:00 2001 From: Christian Sonnabend Date: Wed, 7 Oct 2026 15:50:23 +0200 Subject: [PATCH 1/9] Add GPU QA histograms for missing inner clusters and track clones --- GPU/GPUTracking/qa/GPUQA.cxx | 60 ++++++++++++++++++++++++++++++++++++ GPU/GPUTracking/qa/GPUQA.h | 3 ++ 2 files changed, 63 insertions(+) diff --git a/GPU/GPUTracking/qa/GPUQA.cxx b/GPU/GPUTracking/qa/GPUQA.cxx index 060766b54f3f4..44c4711b04df3 100644 --- a/GPU/GPUTracking/qa/GPUQA.cxx +++ b/GPU/GPUTracking/qa/GPUQA.cxx @@ -525,6 +525,10 @@ int32_t GPUQA::InitQACreateHistograms() createHist(mClusters[i], name, name, AXIS_BINS[4], binsPt.get()); } + const int32_t maxPads = GPUTPCGeometry::NPads(GPUTPCGeometry::NROWS - 1); + createHist(mLowestRowDifferenceVsPad, "lowest_row_difference_vs_pad", "Lowest cluster row difference;Pad of lowest reconstructed cluster;Lowest reconstructed row - lowest MC row", maxPads, -0.5, maxPads - 0.5, 2 * GPUTPCGeometry::NROWS - 1, -GPUTPCGeometry::NROWS + 0.5, GPUTPCGeometry::NROWS - 0.5); + createHist(mLowestRowTrackMultiplicity, "lowest_row_track_multiplicity", "Track multiplicity for missing inner clusters;Reconstructed tracks with the same MC label;Selected reconstructed tracks", 100, -0.5, 99.5); + createHist(mPadRow[0], "padrow0", "padrow0", GPUTPCGeometry::NROWS - PADROW_CHECK_MINCLS, 0, GPUTPCGeometry::NROWS - 1 - PADROW_CHECK_MINCLS, GPUTPCGeometry::NROWS - PADROW_CHECK_MINCLS, 0, GPUTPCGeometry::NROWS - 1 - PADROW_CHECK_MINCLS); createHist(mPadRow[1], "padrow1", "padrow1", 100.f, -0.2f, 0.2f, GPUTPCGeometry::NROWS - PADROW_CHECK_MINCLS, 0, GPUTPCGeometry::NROWS - 1 - PADROW_CHECK_MINCLS); createHist(mPadRow[2], "padrow2", "padrow2", 100.f, -0.2f, 0.2f, GPUTPCGeometry::NROWS - PADROW_CHECK_MINCLS, 0, GPUTPCGeometry::NROWS - 1 - PADROW_CHECK_MINCLS); @@ -1103,12 +1107,35 @@ void GPUQA::RunQA(bool matchOnly, const std::vector* tracksEx } } if (mTracking->mIOPtrs.nMergedTracks && clNative) { + // Keep these per MC label: the reverse label map retains only one clone. + struct LowestRowInfo { + int32_t row = GPUTPCGeometry::NROWS; + int32_t nTracks = 0; + }; + std::vector> lowestRowInfo(GetNMCCollissions()); + for (uint32_t iCol = 0; iCol < GetNMCCollissions(); iCol++) { + lowestRowInfo[iCol].resize(GetNMCTracks(iCol)); + } + for (uint32_t i = 0; i < nReconstructedTracks; i++) { + const auto& label = mTrackMCLabels[i]; + // Count all reconstructed segments, including fake-flagged label matches, + // without applying the selection cuts to the other segments. + if (mTracking->mIOPtrs.mergedTracks[i].OK() && label.isValid() && !label.isNoise()) { + GetMCTrackObj(lowestRowInfo, label).nTracks++; + } + } std::fill(lowestPadRow.begin(), lowestPadRow.end(), 255); for (uint32_t iSector = 0; iSector < GPUTPCGeometry::NSECTORS; iSector++) { for (uint32_t iRow = 0; iRow < GPUTPCGeometry::NROWS; iRow++) { for (uint32_t iCl = 0; iCl < clNative->nClusters[iSector][iRow]; iCl++) { int32_t i = clNative->clusterOffset[iSector][iRow] + iCl; for (int32_t j = 0; j < GetMCLabelNID(i); j++) { + const mcLabelI_t label = GetMCLabel(i, j); + if (!label.isValid() || label.isNoise()) { + continue; + } + auto& info = GetMCTrackObj(lowestRowInfo, label); + info.row = std::min(info.row, (int32_t)iRow); uint32_t trackId = GetMCTrackObj(mTrackMCLabelsReverse, GetMCLabel(i, j)); if (trackId < lowestPadRow.size() && lowestPadRow[trackId] > iRow) { lowestPadRow[trackId] = iRow; @@ -1119,6 +1146,28 @@ void GPUQA::RunQA(bool matchOnly, const std::vector* tracksEx } for (uint32_t i = 0; i < mTracking->mIOPtrs.nMergedTracks; i++) { const auto& trk = mTracking->mIOPtrs.mergedTracks[i]; + const auto& label = mTrackMCLabels[i]; + if (trk.OK() && trk.NClustersFitted() > 50 && CAMath::Abs(trk.GetParam().GetQPt()) < 1.f && label.isValid() && !label.isNoise() && + (mMCTrackMin == -1 || label.getTrackID() >= mMCTrackMin) && (mMCTrackMax == -1 || label.getTrackID() < mMCTrackMax)) { + const auto& info = GetMCTrackObj(lowestRowInfo, label); + if (info.row < 10) { + const GPUTPCGMMergedTrackHit* lowestCl = nullptr; + for (uint32_t j = 0; j < trk.NClusters(); j++) { + const auto& cl = mTracking->mIOPtrs.mergedTrackHits[trk.FirstClusterRef() + j]; + if (!(cl.state & GPUTPCGMMergedTrackHit::flagReject) && (!lowestCl || cl.row < lowestCl->row)) { + lowestCl = &cl; + } + } + if (lowestCl) { + const float pad = clNative->clustersLinear[lowestCl->num].getPad(); + const int32_t difference = (int32_t)lowestCl->row - info.row; + mLowestRowDifferenceVsPad->Fill(pad, difference); + if (CAMath::Abs(difference) > 5 && pad > 5.f && pad < GPUTPCGeometry::NPads(lowestCl->row) - 5.f) { + mLowestRowTrackMultiplicity->Fill(info.nTracks); + } + } + } + } if (trk.OK() && lowestPadRow[i] != 255 && trk.NClustersFitted() >= PADROW_CHECK_MINCLS && CAMath::Abs(trk.GetParam().GetQPt()) < 1.0) { const auto& lowestCl = mTracking->mIOPtrs.mergedTrackHits[trk.FirstClusterRef()].row < mTracking->mIOPtrs.mergedTrackHits[trk.FirstClusterRef() + trk.NClusters() - 1].row ? mTracking->mIOPtrs.mergedTrackHits[trk.FirstClusterRef()] : mTracking->mIOPtrs.mergedTrackHits[trk.FirstClusterRef() + trk.NClusters() - 1]; const int32_t lowestRow = lowestCl.row; @@ -2912,6 +2961,17 @@ int32_t GPUQA::DrawQAHistograms(TObjArray* qcout) } } + if (mQATasks & taskClusterAttach) { + for (TH1* h : {static_cast(mLowestRowDifferenceVsPad), static_cast(mLowestRowTrackMultiplicity)}) { + if (tout && !mConfig.inputHistogramsOnly) { + h->Write(); + } + if (qcout) { + qcout->Add(h); + } + } + } + // Process cluster count statistics if ((mQATasks & taskClusterCounts) && !mHaveExternalHists && !mConfig.clusterRejectionHistograms && !mConfig.inputHistogramsOnly) { DoClusterCounts(attachClusterCounts); diff --git a/GPU/GPUTracking/qa/GPUQA.h b/GPU/GPUTracking/qa/GPUQA.h index b3c518e9798e8..f257c9a418659 100644 --- a/GPU/GPUTracking/qa/GPUQA.h +++ b/GPU/GPUTracking/qa/GPUQA.h @@ -318,6 +318,9 @@ class GPUQA TPad* mPClRej[3]; TPad* mPClRejP; + TH2F* mLowestRowDifferenceVsPad; + TH1F* mLowestRowTrackMultiplicity; + TH2F* mPadRow[4]; TCanvas* mCPadRow[4]; TPad* mPPadRow[4]; From 3e6417dd64a8c0f4928fe0d6d90dcd8546915825 Mon Sep 17 00:00:00 2001 From: Christian Sonnabend Date: Wed, 7 Oct 2026 19:35:18 +0200 Subject: [PATCH 2/9] SOFIE implementation --- GPU/GPUTracking/Base/GPUConstantMem.h | 4 +- GPU/GPUTracking/Base/GPUReconstruction.cxx | 2 +- GPU/GPUTracking/Base/GPUReconstructionCPU.h | 3 + .../Base/cuda/GPUReconstructionCUDA.cu | 19 +- .../Base/cuda/GPUReconstructionCUDA.h | 2 + GPU/GPUTracking/CMakeLists.txt | 23 ++- GPU/GPUTracking/Definitions/GPUSettingsList.h | 3 + GPU/GPUTracking/Global/GPUChain.h | 3 + GPU/GPUTracking/Global/GPUChainTracking.cxx | 68 ++++++- GPU/GPUTracking/Global/GPUChainTracking.h | 5 + .../Global/GPUChainTrackingClusterizer.cxx | 57 ++++-- .../TPCClusterFinder/GPUTPCNNClusterizer.cxx | 3 + .../TPCClusterFinder/GPUTPCNNClusterizer.h | 2 + .../GPUTPCNNClusterizerHost.cxx | 191 +++++++++++++++++- .../GPUTPCNNClusterizerHost.h | 33 ++- GPU/GPUTracking/kernels.cmake | 4 +- GPU/README.md | 43 ++++ 17 files changed, 427 insertions(+), 38 deletions(-) diff --git a/GPU/GPUTracking/Base/GPUConstantMem.h b/GPU/GPUTracking/Base/GPUConstantMem.h index 05547262ce100..212f526e71ae7 100644 --- a/GPU/GPUTracking/Base/GPUConstantMem.h +++ b/GPU/GPUTracking/Base/GPUConstantMem.h @@ -34,7 +34,7 @@ #include "GPUKernelDebugOutput.h" #endif -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) #include "GPUTPCNNClusterizer.h" #endif @@ -56,7 +56,7 @@ struct GPUConstantMem { #ifdef GPUCA_KERNEL_DEBUGGER_OUTPUT GPUKernelDebugOutput debugOutput; #endif -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) GPUTPCNNClusterizer tpcNNClusterer[GPUTPCGeometry::NSECTORS]; #endif template diff --git a/GPU/GPUTracking/Base/GPUReconstruction.cxx b/GPU/GPUTracking/Base/GPUReconstruction.cxx index 3e7509d8fad73..cf898c7b0c807 100644 --- a/GPU/GPUTracking/Base/GPUReconstruction.cxx +++ b/GPU/GPUTracking/Base/GPUReconstruction.cxx @@ -97,7 +97,7 @@ GPUReconstruction::GPUReconstruction(const GPUSettingsDeviceBackend& cfg) : mHos for (uint32_t i = 0; i < NSECTORS; i++) { processors()->tpcTrackers[i].SetSector(i); // TODO: Move to a better place processors()->tpcClusterer[i].mISector = i; -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) processors()->tpcNNClusterer[i].mISector = i; #endif } diff --git a/GPU/GPUTracking/Base/GPUReconstructionCPU.h b/GPU/GPUTracking/Base/GPUReconstructionCPU.h index 466ad318bbe3b..59ccfed332f78 100644 --- a/GPU/GPUTracking/Base/GPUReconstructionCPU.h +++ b/GPU/GPUTracking/Base/GPUReconstructionCPU.h @@ -83,6 +83,9 @@ class GPUReconstructionCPU : public GPUReconstructionProcessing::KernelInterface size_t WriteToConstantMemory(size_t offset, const void* src, size_t size, int32_t stream = -1, deviceEvent* ev = nullptr) override; virtual size_t TransferMemoryInternal(GPUMemoryResource* res, int32_t stream, deviceEvent* ev, deviceEvent* evList, int32_t nEvents, bool toGPU, const void* src, void* dst); + virtual int32_t GetNativeGPUDevice() const { throw std::runtime_error("Native GPU device is unavailable for this backend"); } + virtual void* GetNativeGPUStream(int32_t) const { throw std::runtime_error("Native GPU streams are unavailable for this backend"); } + // ONNX runtime virtual void SetONNXGPUStream(Ort::SessionOptions&, int32_t, int32_t*) {} diff --git a/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu b/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu index 63992bed65fc5..fddc61a241c3f 100644 --- a/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu +++ b/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu @@ -635,11 +635,26 @@ void GPUReconstructionCUDA::loadKernelModules(bool perKernel) } \ } +int32_t GPUReconstructionCUDA::GetNativeGPUDevice() const +{ + int device = -1; + GPUChkErr(cudaGetDevice(&device)); + return device; +} + +void* GPUReconstructionCUDA::GetNativeGPUStream(int32_t stream) const +{ + if (stream < 0 || stream >= mNStreams) { + throw std::out_of_range("GPU stream index out of range"); + } + return mInternals->Streams[stream]; +} + void GPUReconstructionCUDA::SetONNXGPUStream(Ort::SessionOptions& sessionOptions, int32_t stream, int32_t* deviceId) { GPUChkErr(cudaGetDevice(deviceId)); -#if !defined(__HIPCC__) && defined(ORT_CUDA_BUILD) +#if defined(GPUCA_HAS_ONNX) && !defined(__HIPCC__) && defined(ORT_CUDA_BUILD) const OrtApi* api = OrtGetApiBase()->GetApi(ORT_API_VERSION); #ifdef ORT_TENSORRT_BUILD @@ -666,7 +681,7 @@ void GPUReconstructionCUDA::SetONNXGPUStream(Ort::SessionOptions& sessionOptions ORTCHK(api->SessionOptionsAppendExecutionProvider_CUDA_V2(sessionOptions, cudaOptions)); api->ReleaseCUDAProviderOptions(cudaOptions); -#elif defined(ORT_ROCM_BUILD) +#elif defined(GPUCA_HAS_ONNX) && defined(ORT_ROCM_BUILD) // const auto& api = Ort::GetApi(); // api.GetCurrentGpuDeviceId(deviceId); OrtROCMProviderOptions rocmOptions; diff --git a/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.h b/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.h index 6f92ca1938e0d..4e5dd48aa0644 100644 --- a/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.h +++ b/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.h @@ -74,6 +74,8 @@ class GPUReconstructionCUDA : public GPUReconstructionProcessing::KernelInterfac size_t GPUMemCpy(void* dst, const void* src, size_t size, int32_t stream, int32_t toGPU, deviceEvent* ev = nullptr, deviceEvent* evList = nullptr, int32_t nEvents = 1) override; void ReleaseEvent(deviceEvent ev) override; void RecordMarker(deviceEvent* ev, int32_t stream) override; + int32_t GetNativeGPUDevice() const override; + void* GetNativeGPUStream(int32_t stream) const override; void SetONNXGPUStream(Ort::SessionOptions& session_options, int32_t stream, int32_t* deviceId) override; void GetITSTraits(std::unique_ptr>* trackerTraits, std::unique_ptr>* vertexerTraits, std::unique_ptr>* timeFrame) override; diff --git a/GPU/GPUTracking/CMakeLists.txt b/GPU/GPUTracking/CMakeLists.txt index cc44b2003a01d..74a5238cfb0ca 100644 --- a/GPU/GPUTracking/CMakeLists.txt +++ b/GPU/GPUTracking/CMakeLists.txt @@ -10,6 +10,18 @@ # or submit itself to any jurisdiction. set(MODULE GPUTracking) +option(GPUCA_BUILD_SOFIE "Enable experimental SOFIE GPU inference" OFF) +option(GPUCA_BUILD_ORT "Enable ONNXRuntime inference in GPUTracking" ON) +if(NOT GPUCA_BUILD_ORT) + set(onnxruntime_FOUND OFF) +endif() +if(GPUCA_BUILD_SOFIE) + find_package(ROOT CONFIG REQUIRED COMPONENTS ROOTTMVASofie ROOTTMVASofieParser) + find_path(GPUCA_SOFIE_GPU_INCLUDE TMVA/RGPUModel.hxx HINTS ${ROOT_INCLUDE_DIRS} NO_DEFAULT_PATH) + if(NOT GPUCA_SOFIE_GPU_INCLUDE) + message(FATAL_ERROR "SOFIE GPU requires the ROOT development package with RGPUModel support") + endif() +endif() # set(CMAKE_CXX_FLAGS_${CMAKE_BUILD_TYPE_UPPER} "${CMAKE_CXX_FLAGS_${CMAKE_BUILD_TYPE_UPPER}} -O0") # to uncomment if needed, tired of typing this... # set(GPUCA_BUILD_DEBUG 1) @@ -196,7 +208,7 @@ set(SRCS_NO_CINT ${SRCS_NO_CINT} Refit/GPUTrackingRefitKernel.cxx Merger/GPUTPCGMO2Output.cxx) -if(onnxruntime_FOUND) +if(onnxruntime_FOUND OR GPUCA_BUILD_SOFIE) list(APPEND SRCS_NO_CINT TPCClusterFinder/GPUTPCNNClusterizerKernels.cxx TPCClusterFinder/GPUTPCNNClusterizer.cxx TPCClusterFinder/GPUTPCNNClusterizerHost.cxx) endif() @@ -327,6 +339,7 @@ set(HDRS_CINT_DATATYPES ${HDRS_CINT_DATATYPES} ${HDRS_TMP}) unset(HDRS_TMP) set(INCDIRS + ${CMAKE_CURRENT_SOURCE_DIR}/../../Common/ML/include ${ON_THE_FLY_DIR} ${CMAKE_CURRENT_SOURCE_DIR} ${CMAKE_CURRENT_SOURCE_DIR}/Definitions @@ -382,7 +395,6 @@ if(ALIGPU_BUILD_TYPE STREQUAL "O2") O2::TPCFastTransformation O2::DetectorsRaw O2::Steer - O2::ML PUBLIC_INCLUDE_DIRECTORIES ${INCDIRS} SOURCES ${SRCS} ${SRCS_NO_CINT} ${SRCS_NO_H}) @@ -454,12 +466,19 @@ if(GPUCA_QA) endif() target_link_libraries(${targetName} PRIVATE TBB::tbb) +if(GPUCA_BUILD_SOFIE) + target_compile_definitions(${targetName} PRIVATE GPUCA_HAS_SOFIE=1) + target_link_libraries(${targetName} PRIVATE ROOT::ROOTTMVASofie ROOT::ROOTTMVASofieParser) +endif() target_compile_options(${targetName} PRIVATE -Wno-instantiation-after-specialization) if (onnxruntime_FOUND) target_compile_definitions(${targetName} PRIVATE GPUCA_HAS_ONNX=1) target_link_libraries(${targetName} PRIVATE onnxruntime::onnxruntime) + if(TARGET O2::ML) + target_link_libraries(${targetName} PRIVATE O2::ML) + endif() endif() # Add CMake recipes for GPU Tracking librararies diff --git a/GPU/GPUTracking/Definitions/GPUSettingsList.h b/GPU/GPUTracking/Definitions/GPUSettingsList.h index 81949f0b8ed2a..ff76a962dd168 100644 --- a/GPU/GPUTracking/Definitions/GPUSettingsList.h +++ b/GPU/GPUTracking/Definitions/GPUSettingsList.h @@ -260,6 +260,9 @@ EndConfig() // Settings steering the processing of NN Clusterization BeginSubConfig(GPUSettingsProcessingNNclusterizer, nn, configStandalone.proc, "NN", 0, "Processing settings for neural network clusterizer", proc_nn) AddOption(applyNNclusterizer, int, 0, "", 0, "(bool, default = 0), if the neural network clusterizer should be used.") +AddOption(mlFramework, std::string, "ORT", "ml-framework", 0, "ML inference backend: ORT or SOFIE") +AddOption(sofieCompiler, std::string, "", "sofie-compiler", 0, "Runtime compiler executable (default: nvcc or hipcc)") +AddOption(sofieArchitecture, std::string, "", "sofie-architecture", 0, "Required SOFIE target architecture, e.g. sm_80 or gfx90a") AddOption(nnInferenceDevice, std::string, "CPU", "", 0, "(std::string) Specify inference device (cpu (default), rocm, cuda)") AddOption(nnInferenceDeviceId, unsigned int, 0, "", 0, "(unsigned int) Specify inference device id") AddOption(nnInferenceAllocateDevMem, int, 0, "", 0, "(bool, default = 0), if the device memory should be allocated for inference") diff --git a/GPU/GPUTracking/Global/GPUChain.h b/GPU/GPUTracking/Global/GPUChain.h index 61107f7893e9c..535289c6bba58 100644 --- a/GPU/GPUTracking/Global/GPUChain.h +++ b/GPU/GPUTracking/Global/GPUChain.h @@ -87,6 +87,9 @@ class GPUChain inline GPUParam& param() { return mRec->param(); } inline const GPUConstantMem* processors() const { return mRec->processors(); } inline void SynchronizeStream(int32_t stream) { mRec->SynchronizeStream(stream); } + // Borrowed native stream; ownership and synchronization remain with the reconstruction. + inline int32_t GetNativeGPUDevice() const { return mRec->GetNativeGPUDevice(); } + inline void* GetNativeGPUStream(int32_t stream) const { return mRec->GetNativeGPUStream(stream); } inline void SetONNXGPUStream(Ort::SessionOptions& opt, int32_t stream, int32_t* deviceId) { mRec->SetONNXGPUStream(opt, stream, deviceId); } inline void SynchronizeEvents(deviceEvent* evList, int32_t nEvents = 1) { mRec->SynchronizeEvents(evList, nEvents); } inline void SynchronizeEventAndRelease(deviceEvent& ev, bool doGPU = true) diff --git a/GPU/GPUTracking/Global/GPUChainTracking.cxx b/GPU/GPUTracking/Global/GPUChainTracking.cxx index eb6d880398eec..d46af23d9a60f 100644 --- a/GPU/GPUTracking/Global/GPUChainTracking.cxx +++ b/GPU/GPUTracking/Global/GPUChainTracking.cxx @@ -18,6 +18,10 @@ #include #include "GPUChainTracking.h" +#ifdef GPUCA_HAS_SOFIE +#include "GPUTPCNNClusterizerHost.h" +#include "GPUTPCNNClusterizer.h" +#endif #include "GPUChainTrackingGetters.inc" #include "GPUReconstructionIO.h" #include "GPUChainTrackingDefs.h" @@ -68,7 +72,16 @@ GPUChainTracking::GPUChainTracking(GPUReconstruction* rec, uint32_t maxTPCHits, mFlatObjectsDevice.mChainTracking = this; } -GPUChainTracking::~GPUChainTracking() = default; +GPUChainTracking::~GPUChainTracking() +{ +#ifdef GPUCA_HAS_SOFIE + if (!mSofieApplications.empty()) { + const auto context = GetThreadContext(); + SynchronizeGPU(); + mSofieApplications.clear(); + } +#endif +} void GPUChainTracking::RegisterPermanentMemoryAndProcessors() { @@ -102,7 +115,7 @@ void GPUChainTracking::RegisterPermanentMemoryAndProcessors() if (GetRecoSteps() & RecoStep::TPCClusterFinding) { for (uint32_t i = 0; i < NSECTORS; i++) { mRec->RegisterGPUProcessor(&processors()->tpcClusterer[i], GetRecoStepsGPU() & RecoStep::TPCClusterFinding); -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) mRec->RegisterGPUProcessor(&processors()->tpcNNClusterer[i], GetRecoStepsGPU() & RecoStep::TPCClusterFinding); #endif } @@ -147,7 +160,7 @@ void GPUChainTracking::RegisterGPUProcessors() if (GetRecoStepsGPU() & RecoStep::TPCClusterFinding) { for (uint32_t i = 0; i < NSECTORS; i++) { mRec->RegisterGPUDeviceProcessor(&processorsShadow()->tpcClusterer[i], &processors()->tpcClusterer[i]); -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) mRec->RegisterGPUDeviceProcessor(&processorsShadow()->tpcNNClusterer[i], &processors()->tpcNNClusterer[i]); #endif } @@ -392,6 +405,7 @@ int32_t GPUChainTracking::Init() } } + InitSofieClusterizer(); return 0; } @@ -470,6 +484,13 @@ int32_t GPUChainTracking::ForceInitQA() int32_t GPUChainTracking::Finalize() { +#ifdef GPUCA_HAS_SOFIE + if (!mSofieApplications.empty()) { + const auto context = GetThreadContext(); + SynchronizeGPU(); + mSofieApplications.clear(); + } +#endif if (GetProcessingSettings().runQA && GetQA()->IsInitialized() && !(mConfigQA && mConfigQA->shipToQC) && !mQAFromForeignChain) { GetQA()->UpdateChain(this); GetQA()->DrawQAHistograms(); @@ -1006,3 +1027,44 @@ void GPUChainTracking::ApplySyncSettings(GPUSettingsProcessing& proc, GPUSetting steps.setBits(gpudatatypes::RecoStep::TPCdEdx, dEdxMode == -1 ? !syncMode : (dEdxMode > 0)); } } + +void GPUChainTracking::InitSofieClusterizer() +{ + const auto& settings = GetProcessingSettings().nn; + if (settings.mlFramework != "ORT" && settings.mlFramework != "SOFIE") { + throw std::runtime_error("ml-framework must be ORT or SOFIE"); + } + if (!settings.applyNNclusterizer) { + return; + } + if (settings.mlFramework == "ORT") { +#ifndef GPUCA_HAS_ONNX + throw std::runtime_error("ORT was requested but GPUCA_BUILD_ORT is disabled or ONNXRuntime is unavailable"); +#endif + return; + } +#ifdef GPUCA_HAS_SOFIE + if (!mSofieApplications.empty()) { + throw std::runtime_error("SOFIE clusterizer is already initialized"); + } + const bool hip = mRec->GetDeviceType() == GPUReconstruction::GetDeviceType("HIP"); + if ((!hip && mRec->GetDeviceType() != GPUReconstruction::GetDeviceType("CUDA")) || !(GetRecoStepsGPU() & RecoStep::TPCClusterFinding)) { + throw std::runtime_error("SOFIE requires TPC cluster finding on CUDA or HIP"); + } + const int lanes = GetProcessingSettings().nTPCClustererLanes; + if (lanes < 1 || lanes > 4 || static_cast(lanes) > mRec->NStreams()) { + throw std::runtime_error("SOFIE clusterizer requires 1..4 lanes and a stream for each lane"); + } + std::vector> applications; + for (int lane = 0; lane < lanes; lane++) { + auto host = std::make_unique(); + host->initSofie(settings, GetNativeGPUStream(lane), GetNativeGPUDevice(), hip, lane ? applications.front().get() : nullptr); + GPUTPCNNClusterizer check; + host->initClusterizer(settings, check); + applications.push_back(std::move(host)); + } + mSofieApplications = std::move(applications); +#else + throw std::runtime_error("SOFIE was requested but GPUCA_BUILD_SOFIE is disabled"); +#endif +} diff --git a/GPU/GPUTracking/Global/GPUChainTracking.h b/GPU/GPUTracking/Global/GPUChainTracking.h index 759aaf818028e..49ad15acf14fb 100644 --- a/GPU/GPUTracking/Global/GPUChainTracking.h +++ b/GPU/GPUTracking/Global/GPUChainTracking.h @@ -56,6 +56,7 @@ namespace o2::gpu { // class GPUTRDTrackerGPU; class GPUTPCGPUTracker; +class GPUTPCNNClusterizerHost; class GPUDisplayInterface; class GPUQA; class GPUTPCClusterStatistics; @@ -298,6 +299,10 @@ class GPUChainTracking : public GPUChain int32_t RunChainFinalize(); void OutputSanityCheck(); int32_t RunTPCTrackingSectors_internal(); + void InitSofieClusterizer(); +#ifdef GPUCA_HAS_SOFIE + std::vector> mSofieApplications; +#endif int32_t RunTPCClusterizer_prepare(bool restorePointers, const GPUTPCExtraADC& extraADCs); #ifndef GPUCA_RUN2 std::pair RunTPCClusterizer_transferZS(int32_t iSector, const CfFragment& fragment, int32_t lane, const GPUTPCExtraADC& extraADCs); diff --git a/GPU/GPUTracking/Global/GPUChainTrackingClusterizer.cxx b/GPU/GPUTracking/Global/GPUChainTrackingClusterizer.cxx index f884b4fb8a10a..304fef6f126c6 100644 --- a/GPU/GPUTracking/Global/GPUChainTrackingClusterizer.cxx +++ b/GPU/GPUTracking/Global/GPUChainTrackingClusterizer.cxx @@ -45,7 +45,7 @@ #include "DataFormatsTPC/Constants.h" #include "TPCBase/RDHUtils.h" -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) #include "GPUTPCNNClusterizerKernels.h" #include "GPUTPCNNClusterizerHost.h" #include "ORTRootSerializer.h" @@ -809,7 +809,7 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) WriteToConstantMemory(RecoStep::TPCClusterFinding, (char*)processors()->tpcClusterer - (char*)processors(), processorsShadow()->tpcClusterer, sizeof(GPUTPCClusterFinder) * NSECTORS, mRec->NStreams() - 1, &mEvents->init); } -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) const GPUSettingsProcessingNNclusterizer& nn_settings = GetProcessingSettings().nn; GPUTPCNNClusterizerHost nnApplications[GetProcessingSettings().nTPCClustererLanes]; @@ -817,9 +817,11 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) HighResTimer* nnTimers[12]; if (nn_settings.applyNNclusterizer) { - int32_t deviceId = -1; + int32_t deviceId = nn_settings.mlFramework == "SOFIE" ? GetNativeGPUDevice() : -1; int32_t numLanes = GetProcessingSettings().nTPCClustererLanes; +#ifdef GPUCA_HAS_ONNX int32_t maxThreads = mRec->getNKernelHostThreads(true); +#endif // bool recreateMemoryAllocator = false; if (GetProcessingSettings().debugLevel >= 1) { @@ -838,6 +840,18 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) } mRec->runParallelOuterLoop(doGPU, numLanes, [&](uint32_t lane) { + if (nn_settings.mlFramework == "SOFIE") { +#ifdef GPUCA_HAS_SOFIE + if (mSofieApplications.size() != static_cast(numLanes)) { + throw std::runtime_error("SOFIE clusterizer must be initialized before processing"); + } + nnApplications[lane].useSofie(*mSofieApplications[lane]); + return; +#else + throw std::runtime_error("SOFIE was not enabled in this build"); +#endif + } +#ifdef GPUCA_HAS_ONNX nnApplications[lane].init(nn_settings, GetProcessingSettings().deterministicGPUReconstruction); if (nnApplications[lane].mModelsUsed[0]) { SetONNXGPUStream(*(nnApplications[lane].mModelClass).getSessionOptions(), lane, &deviceId); @@ -893,6 +907,7 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) if (nn_settings.nnClusterizerVerbosity > 0) { LOG(info) << "(ORT) Allocated ONNX stream for lane " << lane << " and device " << deviceId; } +#endif }); const int16_t maxFragmentLen = GetProcessingSettings().overrideClusterizerFragmentLen; const uint32_t maxAllowedTimebin = param().par.continuousTracking ? std::max(param().continuousMaxTimeBin, maxFragmentLen) : constants::TPC_MAX_TIME_BIN_TRIGGERED; @@ -1223,11 +1238,13 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) const auto nRegularClusters = clusterer.mPmemory->counters.nClusters; if (nRegularClusters != 0) { if (GetProcessingSettings().nn.applyNNclusterizer) { -#ifdef GPUCA_HAS_ONNX +#if defined(GPUCA_HAS_ONNX) || defined(GPUCA_HAS_SOFIE) GPUTPCNNClusterizer& clustererNN = processors()->tpcNNClusterer[lane]; GPUTPCNNClusterizer& clustererNNShadow = doGPU ? processorsShadow()->tpcNNClusterer[lane] : clustererNN; GPUTPCNNClusterizerHost& nnApplication = nnApplications[lane]; + nnApplication.bindSofieWorkspace(clustererNNShadow.mSofieWorkspace, clustererNNShadow.mSofieWorkspaceSize); + // int withMC = (doGPU && propagateMCLabels); if (nn_settings.nnClusterizerApplyCfDeconvolution) { @@ -1271,15 +1288,15 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) if(GetProcessingSettings().debugLevel >= 1 && (doGPU || lane < 4)) { nnTimers[3*lane]->Start(); } if (clustererNNShadow.mNnInferenceInputDType == 0) { if (clustererNNShadow.mNnInferenceOutputDType == 0) { - (nnApplication.mModelClass).inference(clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mModelProbabilities_32); + nnApplication.inference(0, clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mModelProbabilities_32); } else { - (nnApplication.mModelClass).inference(clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mModelProbabilities_16); + nnApplication.inference(0, clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mModelProbabilities_16); } } else if (clustererNNShadow.mNnInferenceInputDType == 1) { if (clustererNNShadow.mNnInferenceOutputDType == 0) { - (nnApplication.mModelClass).inference(clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mModelProbabilities_32); + nnApplication.inference(0, clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mModelProbabilities_32); } else { - (nnApplication.mModelClass).inference(clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mModelProbabilities_16); + nnApplication.inference(0, clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mModelProbabilities_16); } } if(GetProcessingSettings().debugLevel >= 1 && (doGPU || lane < 4)) { nnTimers[3*lane]->Stop(); } // doGPU || lane<4 -> only for GPU or first 4 CPU lanes (to limit number of concurrent timers). At least gives some statistics for CPU time... @@ -1291,31 +1308,31 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) if(GetProcessingSettings().debugLevel >= 1 && (doGPU || lane < 4)) { nnTimers[3*lane + 1]->Start(); } if (clustererNNShadow.mNnInferenceInputDType == 0) { if (clustererNNShadow.mNnInferenceOutputDType == 0) { - (nnApplication.mModelReg1).inference(clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg1_32); + nnApplication.inference(1, clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg1_32); } else { - (nnApplication.mModelReg1).inference(clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg1_16); + nnApplication.inference(1, clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg1_16); } } else { if (clustererNNShadow.mNnInferenceOutputDType == 0) { - (nnApplication.mModelReg1).inference(clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg1_32); + nnApplication.inference(1, clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg1_32); } else { - (nnApplication.mModelReg1).inference(clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg1_16); + nnApplication.inference(1, clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg1_16); } } if(GetProcessingSettings().debugLevel >= 1 && (doGPU || lane < 4)) { nnTimers[3*lane + 1]->Stop(); } - if (nnApplication.mModelClass.getNumOutputNodes()[0][1] > 1 && nnApplication.mModelReg2.isInitialized()) { + if (nnApplication.modelOutputs(0) > 1 && nnApplication.mModelsUsed[2]) { if(GetProcessingSettings().debugLevel >= 1 && (doGPU || lane < 4)) { nnTimers[3*lane + 2]->Start(); } if (clustererNNShadow.mNnInferenceInputDType == 0) { if (clustererNNShadow.mNnInferenceOutputDType == 0) { - (nnApplication.mModelReg2).inference(clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg2_32); + nnApplication.inference(2, clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg2_32); } else { - (nnApplication.mModelReg2).inference(clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg2_16); + nnApplication.inference(2, clustererNNShadow.mInputData_32, iSize, clustererNNShadow.mOutputDataReg2_16); } } else if (clustererNNShadow.mNnInferenceInputDType == 1) { if (clustererNNShadow.mNnInferenceOutputDType == 0) { - (nnApplication.mModelReg2).inference(clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg2_32); + nnApplication.inference(2, clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg2_32); } else { - (nnApplication.mModelReg2).inference(clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg2_16); + nnApplication.inference(2, clustererNNShadow.mInputData_16, iSize, clustererNNShadow.mOutputDataReg2_16); } } if(GetProcessingSettings().debugLevel >= 1 && (doGPU || lane < 4)) { nnTimers[3*lane + 2]->Stop(); } @@ -1327,14 +1344,14 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) // Publishing kernels for class labels and regression results // In case classification should not be used, this kernel should still be executed to fill the mOutputDataClass array with default values - if (nnApplication.mModelClass.getNumOutputNodes()[0][1] == 1) { + if (nnApplication.modelOutputs(0) == 1) { runKernel({GetGrid(iSize, lane), krnlRunRangeNone}, iSector, clustererNNShadow.mNnInferenceOutputDType, propagateMCLabels, batchStart); // Assigning class labels } else { runKernel({GetGrid(iSize, lane), krnlRunRangeNone}, iSector, clustererNNShadow.mNnInferenceOutputDType, propagateMCLabels, batchStart); // Assigning class labels } if (!clustererNNShadow.mNnClusterizerUseCfRegression) { runKernel({GetGrid(iSize, lane), krnlRunRangeNone}, iSector, clustererNNShadow.mNnInferenceOutputDType, propagateMCLabels, batchStart); // Publishing class 1 regression results - if (nnApplication.mModelClass.getNumOutputNodes()[0][1] > 1 && nnApplication.mModelReg2.isInitialized()) { + if (nnApplication.modelOutputs(0) > 1 && nnApplication.mModelsUsed[2]) { runKernel({GetGrid(iSize, lane), krnlRunRangeNone}, iSector, clustererNNShadow.mNnInferenceOutputDType, propagateMCLabels, batchStart); // Publishing class 2 regression results } } @@ -1485,7 +1502,7 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) } for (int32_t i = 0; i < GetProcessingSettings().nTPCClustererLanes; i++) { #ifdef GPUCA_HAS_ONNX - if (GetProcessingSettings().nn.applyNNclusterizer) { + if (GetProcessingSettings().nn.applyNNclusterizer && GetProcessingSettings().nn.mlFramework == "ORT") { if (GetProcessingSettings().nn.nnClusterizerVerbosity > 0) { LOG(info) << "(ORT) Environment releasing..."; } diff --git a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.cxx b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.cxx index 0e77393be1ce3..96cb234619b02 100644 --- a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.cxx +++ b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.cxx @@ -145,6 +145,9 @@ void* GPUTPCNNClusterizer::setIOPointers(void* mem) } } + if (mSofieWorkspaceSize) { + computePointerWithAlignment<256>(mem, mSofieWorkspace, mSofieWorkspaceSize); + } return mem; } diff --git a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.h b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.h index 7aa23489eb2f4..578a39f1f31b7 100644 --- a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.h +++ b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizer.h @@ -101,6 +101,8 @@ class GPUTPCNNClusterizer : public GPUProcessor OrtDataType::Float16_t* mOutputDataReg1_16 = nullptr; OrtDataType::Float16_t* mOutputDataReg2_16 = nullptr; + char* mSofieWorkspace = nullptr; + size_t mSofieWorkspaceSize = 0; int16_t mMemoryId = -1; }; // class GPUTPCNNClusterizer diff --git a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx index 96b8a1d7ed2fd..40eeec2d61f6d 100644 --- a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx +++ b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx @@ -27,10 +27,29 @@ #include #endif +#ifdef GPUCA_HAS_SOFIE +#include +#include +#endif +#include +#include + using namespace o2::gpu; +struct GPUTPCNNClusterizerHost::SofieState { +#ifdef GPUCA_HAS_SOFIE + std::array, 3> models; + std::array, 3> sessions; +#endif + size_t workspaceSize = 0; + unsigned int maxBatch = 0; +}; + void GPUTPCNNClusterizerHost::init(const GPUSettingsProcessingNNclusterizer& settings, bool useDeterministicMode) { +#ifndef GPUCA_HAS_ONNX + throw std::runtime_error("ONNXRuntime was not enabled in this build"); +#else std::string class_model_path = settings.nnClassificationPath, reg_model_path = settings.nnRegressionPath; std::vector reg_model_paths_local; std::vector evalMode = o2::utils::Str::tokenize(settings.nnEvalMode, ':'); @@ -83,6 +102,7 @@ void GPUTPCNNClusterizerHost::init(const GPUSettingsProcessingNNclusterizer& set mModelsUsed[2] = true; } } +#endif } void GPUTPCNNClusterizerHost::initClusterizer(const GPUSettingsProcessingNNclusterizer& settings, GPUTPCNNClusterizer& clustererNN, int32_t maxFragmentLen, int32_t maxAllowedTimebin) @@ -134,13 +154,29 @@ void GPUTPCNNClusterizerHost::initClusterizer(const GPUSettingsProcessingNNclust } else { clustererNN.mNnInferenceOutputDType = 1; // Default to float16 } - clustererNN.mNnClusterizerModelClassNumOutputNodes = mModelClass.getNumOutputNodes()[0][1]; +#ifdef GPUCA_HAS_SOFIE + if (mSofie) { + if (settings.nnClusterizerBatchedMode != mSofie->maxBatch) { + throw std::runtime_error("Changing SOFIE batch capacity requires chain reinitialization"); + } + clustererNN.mSofieWorkspaceSize = mSofie->workspaceSize; + for (const auto& model : mSofie->models) { + if (model && ((model->GetPrecision() == TMVA::Experimental::SOFIE::RGPUModel::Precision::Float16) != (clustererNN.mNnInferenceInputDType == 1) || clustererNN.mNnInferenceInputDType != clustererNN.mNnInferenceOutputDType)) { + throw std::runtime_error("Changing SOFIE precision requires chain reinitialization"); + } + if (model && model->InputSize() != static_cast(clustererNN.mNnClusterizerElementSize)) { + throw std::runtime_error("SOFIE input shape does not match the configured clusterizer window"); + } + } + } +#endif + clustererNN.mNnClusterizerModelClassNumOutputNodes = modelOutputs(0); if (!settings.nnClusterizerUseCfRegression) { - if (mModelClass.getNumOutputNodes()[0][1] == 1 || !mModelReg2.isInitialized()) { - clustererNN.mNnClusterizerModelReg1NumOutputNodes = mModelReg1.getNumOutputNodes()[0][1]; + if (modelOutputs(0) == 1 || !mModelsUsed[2]) { + clustererNN.mNnClusterizerModelReg1NumOutputNodes = modelOutputs(1); } else { - clustererNN.mNnClusterizerModelReg1NumOutputNodes = mModelReg1.getNumOutputNodes()[0][1]; - clustererNN.mNnClusterizerModelReg2NumOutputNodes = mModelReg2.getNumOutputNodes()[0][1]; + clustererNN.mNnClusterizerModelReg1NumOutputNodes = modelOutputs(1); + clustererNN.mNnClusterizerModelReg2NumOutputNodes = modelOutputs(2); } } } @@ -178,6 +214,7 @@ void GPUTPCNNClusterizerHost::initClusterizer(const GPUSettingsProcessingNNclust // } // } +#ifdef GPUCA_HAS_ONNX // MockedOrtAllocator implementation to be able to use volatile assignment struct MockedOrtAllocator : OrtAllocator { MockedOrtAllocator(GPUReconstruction* = nullptr, OrtMemoryInfo* = nullptr); @@ -278,3 +315,147 @@ MockedOrtAllocator* GPUTPCNNClusterizerHost::getMockedAllocator() { return mMockedAlloc.get(); } + +#endif + +void GPUTPCNNClusterizerHost::initSofie(const GPUSettingsProcessingNNclusterizer& settings, void* stream, int32_t device, bool hip, const GPUTPCNNClusterizerHost* source) +{ +#ifdef GPUCA_HAS_SOFIE + using Model = TMVA::Experimental::SOFIE::RGPUModel; + if (settings.nnLoadFromCCDB) { + throw std::runtime_error("SOFIE currently requires local ONNX files; disable nnLoadFromCCDB"); + } + if (!settings.nnClusterizerBatchedMode || settings.nnClusterizerBatchedMode > static_cast(std::numeric_limits::max())) { + throw std::runtime_error("SOFIE requires a positive batch capacity within int32 range"); + } + auto precision = [](const std::string& value) { + if (value == "FP32" || value == "fp32") { + return Model::Precision::Float32; + } + if (value == "FP16" || value == "fp16") { + return Model::Precision::Float16; + } + throw std::runtime_error("SOFIE precision must be FP32 or FP16"); + }; + const auto inputType = precision(settings.nnInferenceInputDType); + if (inputType != precision(settings.nnInferenceOutputDType)) { + throw std::runtime_error("SOFIE requires matching model, input and output precision"); + } + if (settings.nnInferenceDevice != (hip ? "rocm" : "cuda") && settings.nnInferenceDevice != (hip ? "ROCM" : "CUDA")) { + throw std::runtime_error("SOFIE inference device must match the reconstruction backend (cuda or rocm)"); + } + size_t inputSize = 1; + for (int radius : {settings.nnClusterizerSizeInputRow, settings.nnClusterizerSizeInputPad, settings.nnClusterizerSizeInputTime}) { + if (radius < 0 || radius > 16383 || inputSize > static_cast(std::numeric_limits::max()) / (2 * radius + 1)) { + throw std::runtime_error("SOFIE clusterizer input window is too large or invalid"); + } + inputSize *= 2 * radius + 1; + } + inputSize += settings.nnClusterizerAddIndexData ? 3 : 0; + if (inputSize > static_cast(std::numeric_limits::max()) / settings.nnClusterizerBatchedMode) { + throw std::runtime_error("SOFIE input batch exceeds the clusterizer's index range"); + } + auto state = std::make_shared(); + state->maxBatch = settings.nnClusterizerBatchedMode; + std::array paths{settings.nnClassificationPath, "", ""}; + if (!settings.nnClusterizerUseCfRegression) { + auto regression = o2::utils::Str::tokenize(settings.nnRegressionPath, ':'); + if (regression.empty() || regression.size() > 2) { + throw std::runtime_error("SOFIE expects one or two regression paths"); + } + for (size_t i = 0; i < regression.size(); i++) { + paths[i + 1] = regression[i]; + } + } + TMVA::Experimental::SOFIE::RModelParser_ONNX parser; + std::unordered_map> compiled; + for (size_t i = 0; i < paths.size(); i++) { + mModelsUsed[i] = !paths[i].empty(); + if (!mModelsUsed[i]) { + continue; + } + if (source) { + state->models[i] = source->mSofie->models[i]; + } else { + auto& model = compiled[paths[i]]; + if (!model) { + model = std::make_shared(parser.ParseGPU(paths[i])); + if (model->GetPrecision() != inputType) { + throw std::runtime_error("SOFIE model precision differs from nnInferenceInputDType: " + paths[i]); + } + if (model->InputSize() != inputSize || model->OutputSize() > static_cast(std::numeric_limits::max()) / settings.nnClusterizerBatchedMode) { + throw std::runtime_error("SOFIE model shape does not match the clusterizer window or index range: " + paths[i]); + } + model->WorkspaceSize(settings.nnClusterizerBatchedMode); + model->Compile({hip ? Model::Backend::HIP : Model::Backend::CUDA, settings.sofieCompiler, settings.sofieArchitecture}); + } + state->models[i] = model; + } + state->sessions[i] = state->models[i]->CreateSession(stream, device, settings.nnClusterizerBatchedMode); + if (state->sessions[i]->WorkspaceSize() > std::numeric_limits::max() - state->workspaceSize) { + throw std::overflow_error("SOFIE combined workspace size overflow"); + } + state->workspaceSize += state->sessions[i]->WorkspaceSize(); + } + mSofie = std::move(state); + mDeviceId = device; +#else + throw std::runtime_error("SOFIE was not enabled in this build"); +#endif +} + +void GPUTPCNNClusterizerHost::useSofie(const GPUTPCNNClusterizerHost& source) +{ + mSofie = source.mSofie; + mModelsUsed = source.mModelsUsed; + mDeviceId = source.mDeviceId; +} + +void GPUTPCNNClusterizerHost::bindSofieWorkspace(void* workspace, size_t bytes) +{ +#ifdef GPUCA_HAS_SOFIE + if (!mSofie) { + return; + } + if (!workspace || bytes < mSofie->workspaceSize) { + throw std::runtime_error("O2 supplied insufficient SOFIE workspace"); + } + auto* next = static_cast(workspace); + for (const auto& session : mSofie->sessions) { + if (session) { + session->SetWorkspace(next, session->WorkspaceSize()); + next += session->WorkspaceSize(); + } + } +#endif +} + +void GPUTPCNNClusterizerHost::inferenceSofie(int model, const void* input, size_t batch, void* output) +{ +#ifdef GPUCA_HAS_SOFIE + if (!mSofie || model < 0 || model >= 3 || !mSofie->sessions[model]) { + throw std::runtime_error("SOFIE model is not initialized"); + } + mSofie->sessions[model]->Infer(input, output, batch); +#else + throw std::runtime_error("SOFIE was not enabled in this build"); +#endif +} + +int32_t GPUTPCNNClusterizerHost::modelOutputs(int model) const +{ +#ifdef GPUCA_HAS_SOFIE + if (mSofie) { + return mSofie->models.at(model) ? static_cast(mSofie->models.at(model)->OutputSize()) : 0; + } +#endif +#ifdef GPUCA_HAS_ONNX + return (model == 0 ? mModelClass : model == 1 ? mModelReg1 + : mModelReg2) + .getNumOutputNodes() + .at(0) + .at(1); +#else + throw std::runtime_error("No initialized inference backend"); +#endif +} diff --git a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.h b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.h index 8f8465d5dca34..98c6d73246d20 100644 --- a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.h +++ b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.h @@ -18,7 +18,11 @@ #include #include #include +#include +#include +#ifdef GPUCA_HAS_ONNX #include "ML/OrtInterface.h" +#endif class OrtMemoryInfo; class OrtAllocator; @@ -52,6 +56,28 @@ class GPUTPCNNClusterizerHost void createBoundary(GPUTPCNNClusterizer&); void createIndexLookup(GPUTPCNNClusterizer&); + void initSofie(const GPUSettingsProcessingNNclusterizer&, void* stream, int32_t device, bool hip, const GPUTPCNNClusterizerHost* source = nullptr); + void useSofie(const GPUTPCNNClusterizerHost&); + void bindSofieWorkspace(void* workspace, size_t bytes); + void inferenceSofie(int model, const void* input, size_t batch, void* output); + int32_t modelOutputs(int model) const; + template + void inference(int model, Input* input, size_t batch, Output* output) + { + if (mSofie) { + inferenceSofie(model, input, batch, output); + return; + } +#ifdef GPUCA_HAS_ONNX + (model == 0 ? mModelClass : model == 1 ? mModelReg1 + : mModelReg2) + .inference(input, batch, output); +#else + throw std::runtime_error("ONNXRuntime was not enabled in this build"); +#endif + } + +#ifdef GPUCA_HAS_ONNX // ONNX void directOrtAllocator(Ort::Env*, Ort::MemoryInfo*, GPUReconstruction*, bool = false); MockedOrtAllocator* getMockedAllocator(); @@ -59,9 +85,14 @@ class GPUTPCNNClusterizerHost std::unordered_map mOrtOptions; o2::ml::OrtModel mModelClass, mModelReg1, mModelReg2; // For splitting clusters + std::shared_ptr mMockedAlloc = nullptr; +#endif std::vector mModelsUsed = {false, false, false}; // 0: class, 1: reg_1, 2: reg_2 int32_t mDeviceId = -1; - std::shared_ptr mMockedAlloc = nullptr; + + private: + struct SofieState; + std::shared_ptr mSofie; }; // class GPUTPCNNClusterizerHost } // namespace o2::gpu diff --git a/GPU/GPUTracking/kernels.cmake b/GPU/GPUTracking/kernels.cmake index ed155788b0bef..19835d4b58ee9 100644 --- a/GPU/GPUTracking/kernels.cmake +++ b/GPU/GPUTracking/kernels.cmake @@ -26,7 +26,7 @@ o2_gpu_kernel_file_list(TPCDECOMPRESSION GPUTPCCompressionTrackModel.cxx ERRORS) o2_gpu_kernel_file_list(TPCCLUSTERFINDER ERRORS ClusterAccumulator.cxx) o2_gpu_kernel_file_list(TRDTRACKER GPUTRDTrack.cxx GPUTRDTracker.cxx GPUTRDTrackletWord.cxx GeometryBase.cxx) o2_gpu_kernel_file_list(GLOBALREFIT TPCMERGER O2PROPAGATOR MATLUT GPUTrackingRefit.cxx) -if(onnxruntime_FOUND) +if(onnxruntime_FOUND OR GPUCA_BUILD_SOFIE) o2_gpu_kernel_file_list(TPCNNCLUSTERFINDER ERRORS ClusterAccumulator.cxx GPUTPCNNClusterizerKernels.cxx) endif() @@ -126,7 +126,7 @@ o2_gpu_add_kernel("GPUTPCCFDecodeZSDenseLink" "GPUTP o2_gpu_add_kernel("GPUTPCCFGather" "=" LB o2::tpc::ClusterNative* dest) o2_gpu_add_kernel("GPUTrackingRefitKernel, mode0asGPU" "= GLOBALREFIT " LB) o2_gpu_add_kernel("GPUTrackingRefitKernel, mode1asTrackParCov" "= GLOBALREFIT " LB) -if(onnxruntime_FOUND) +if(onnxruntime_FOUND OR GPUCA_BUILD_SOFIE) o2_gpu_add_kernel("GPUTPCNNClusterizerKernels, runCfClusterizer" "= TPCNNCLUSTERFINDER" LB uint8_t sector int8_t dtype int8_t withMC uint32_t batchStart) o2_gpu_add_kernel("GPUTPCNNClusterizerKernels, fillInputNNCPU" "= TPCNNCLUSTERFINDER" LB uint8_t sector int8_t dtype int8_t withMC uint32_t batchStart) o2_gpu_add_kernel("GPUTPCNNClusterizerKernels, fillInputNNGPU" "= TPCNNCLUSTERFINDER" LB uint8_t sector int8_t dtype int8_t withMC uint32_t batchStart) diff --git a/GPU/README.md b/GPU/README.md index 1aa6575e0ae67..836aeffad606b 100644 --- a/GPU/README.md +++ b/GPU/README.md @@ -10,3 +10,46 @@ This module contains the following submodules: * \subpage refGPUTrackingStandalone * \subpage refGPUTrackingDisplayFilterMacros /doxy --> + +## Experimental SOFIE NN clusterizer + +Build against the ROOT development package containing `TMVA/RGPUModel.hxx`: + +- `GPUCA_BUILD_SOFIE=ON` enables the SOFIE host integration. +- `GPUCA_BUILD_ORT=ON` (default) retains ORT; set it to `OFF` for a SOFIE-only + GPUTracking backend. Other O2 components may still require ONNXRuntime. +- SOFIE is disabled by default. This change does not run a compiler at O2 build + configuration time to generate models. + +Select `ml-framework=SOFIE` (the `mlFramework` field in the NN settings), or +`ORT` to retain the existing path. For SOFIE, also set `sofie-architecture` +(`sofieArchitecture`) to the target GPU architecture, for example `sm_80` for +CUDA or `gfx90a` for HIP. `sofie-compiler` (`sofieCompiler`) optionally supplies +an explicit compiler path; otherwise `nvcc`/`hipcc` is found on PATH at runtime. +The compiler and GPU toolkit must be available on the worker at startup. + +Use GPU TPC cluster finding, `nnInferenceDevice=cuda` or `rocm` as appropriate, +`nnLoadFromCCDB=0`, and local classification/regression ONNX paths. Set both +`nnInferenceInputDType` and `nnInferenceOutputDType` to `FP32` or `FP16` to match +the model. Mixed input/output precision, unsupported graphs, CPU SOFIE inference +and CCDB model loading are rejected explicitly. + +Each distinct model path is parsed and compiled once during chain initialization. +Each lane then gets a session bound to its existing native stream. O2 allocates +input, output, weights and intermediate storage through its registered clusterizer +memory. When O2 recycles the scratch arena, weights are uploaded again on the +lane's stream; this does not recompile the model. Sessions persist until chain +finalization, which synchronizes before unloading the generated libraries. +Changing models, batch capacity or backend requires reinitializing the chain. + +The supplied classification/regression networks have six Gemm layers, five Relu +layers, 246 input values and respectively 1/5 output values. The clusterizer's +input-window and index-data settings must reproduce the training layout; +the default window does not have 246 values. `nnClusterizerBatchedMode` sets +the maximum batch; smaller final batches reuse the same code and workspace. + +Validation to run after building: ROOT's `TestSofieGPU` and CUDA/HIP +`TestSofieGPUDevice`, then compare SOFIE against ORT using both supplied +precisions and multiple batch sizes. Check numerical tolerances and downstream +cluster decisions; the scalar dense kernels are not performance tuned. +No ROOT/O2 build or GPU test was run while implementing this change. From 12aa60155e8ac96f973e137ceeedc7eecee94691 Mon Sep 17 00:00:00 2001 From: Christian Sonnabend Date: Wed, 7 Oct 2026 23:28:25 +0200 Subject: [PATCH 3/9] Support CCDB loading and recompiling in case model bytes change --- GPU/GPUTracking/Global/GPUChainTracking.cxx | 27 +++++++-- GPU/GPUTracking/Global/GPUChainTracking.h | 2 +- .../Global/GPUChainTrackingClusterizer.cxx | 1 + .../GPUTPCNNClusterizerHost.cxx | 56 ++++++++++++++++--- .../GPUTPCNNClusterizerHost.h | 5 +- GPU/README.md | 28 +++++++--- GPU/Workflow/src/GPUWorkflowSpec.cxx | 1 + 7 files changed, 99 insertions(+), 21 deletions(-) diff --git a/GPU/GPUTracking/Global/GPUChainTracking.cxx b/GPU/GPUTracking/Global/GPUChainTracking.cxx index d46af23d9a60f..fd3fbf530e770 100644 --- a/GPU/GPUTracking/Global/GPUChainTracking.cxx +++ b/GPU/GPUTracking/Global/GPUChainTracking.cxx @@ -20,6 +20,7 @@ #include "GPUChainTracking.h" #ifdef GPUCA_HAS_SOFIE #include "GPUTPCNNClusterizerHost.h" +#include "ORTRootSerializer.h" #include "GPUTPCNNClusterizer.h" #endif #include "GPUChainTrackingGetters.inc" @@ -405,7 +406,7 @@ int32_t GPUChainTracking::Init() } } - InitSofieClusterizer(); + InitSofieClusterizer(true); return 0; } @@ -1028,7 +1029,7 @@ void GPUChainTracking::ApplySyncSettings(GPUSettingsProcessing& proc, GPUSetting } } -void GPUChainTracking::InitSofieClusterizer() +void GPUChainTracking::InitSofieClusterizer(bool deferCCDB) { const auto& settings = GetProcessingSettings().nn; if (settings.mlFramework != "ORT" && settings.mlFramework != "SOFIE") { @@ -1044,8 +1045,18 @@ void GPUChainTracking::InitSofieClusterizer() return; } #ifdef GPUCA_HAS_SOFIE - if (!mSofieApplications.empty()) { - throw std::runtime_error("SOFIE clusterizer is already initialized"); + std::array buffers{}; + if (settings.nnLoadFromCCDB) { + for (size_t i = 0; i < buffers.size(); i++) { + const auto* network = processors()->calibObjects.nnClusterizerNetworks[i]; + if (network && network->getONNXModelSize()) { + buffers[i] = std::string_view(network->getONNXModel(), network->getONNXModelSize()); + } + } + } + const auto* previous = mSofieApplications.empty() ? nullptr : mSofieApplications.front().get(); + if (previous && (!settings.nnLoadFromCCDB || previous->hasSofieBuffers(buffers))) { + return; } const bool hip = mRec->GetDeviceType() == GPUReconstruction::GetDeviceType("HIP"); if ((!hip && mRec->GetDeviceType() != GPUReconstruction::GetDeviceType("CUDA")) || !(GetRecoStepsGPU() & RecoStep::TPCClusterFinding)) { @@ -1055,14 +1066,20 @@ void GPUChainTracking::InitSofieClusterizer() if (lanes < 1 || lanes > 4 || static_cast(lanes) > mRec->NStreams()) { throw std::runtime_error("SOFIE clusterizer requires 1..4 lanes and a stream for each lane"); } + if (deferCCDB && settings.nnLoadFromCCDB) { + return; + } std::vector> applications; for (int lane = 0; lane < lanes; lane++) { auto host = std::make_unique(); - host->initSofie(settings, GetNativeGPUStream(lane), GetNativeGPUDevice(), hip, lane ? applications.front().get() : nullptr); + host->initSofie(settings, GetNativeGPUStream(lane), GetNativeGPUDevice(), hip, lane ? applications.front().get() : nullptr, buffers, lane ? nullptr : previous); GPUTPCNNClusterizer check; host->initClusterizer(settings, check); applications.push_back(std::move(host)); } + if (previous) { + SynchronizeGPU(); + } mSofieApplications = std::move(applications); #else throw std::runtime_error("SOFIE was requested but GPUCA_BUILD_SOFIE is disabled"); diff --git a/GPU/GPUTracking/Global/GPUChainTracking.h b/GPU/GPUTracking/Global/GPUChainTracking.h index 49ad15acf14fb..5624f9c064126 100644 --- a/GPU/GPUTracking/Global/GPUChainTracking.h +++ b/GPU/GPUTracking/Global/GPUChainTracking.h @@ -299,7 +299,7 @@ class GPUChainTracking : public GPUChain int32_t RunChainFinalize(); void OutputSanityCheck(); int32_t RunTPCTrackingSectors_internal(); - void InitSofieClusterizer(); + void InitSofieClusterizer(bool deferCCDB = false); #ifdef GPUCA_HAS_SOFIE std::vector> mSofieApplications; #endif diff --git a/GPU/GPUTracking/Global/GPUChainTrackingClusterizer.cxx b/GPU/GPUTracking/Global/GPUChainTrackingClusterizer.cxx index 304fef6f126c6..bf52a6b9421b8 100644 --- a/GPU/GPUTracking/Global/GPUChainTrackingClusterizer.cxx +++ b/GPU/GPUTracking/Global/GPUChainTrackingClusterizer.cxx @@ -817,6 +817,7 @@ int32_t GPUChainTracking::RunTPCClusterizer(bool synchronizeOutput) HighResTimer* nnTimers[12]; if (nn_settings.applyNNclusterizer) { + InitSofieClusterizer(); int32_t deviceId = nn_settings.mlFramework == "SOFIE" ? GetNativeGPUDevice() : -1; int32_t numLanes = GetProcessingSettings().nTPCClustererLanes; #ifdef GPUCA_HAS_ONNX diff --git a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx index 40eeec2d61f6d..bffe47402e8f8 100644 --- a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx +++ b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx @@ -33,6 +33,7 @@ #endif #include #include +#include using namespace o2::gpu; @@ -41,6 +42,7 @@ struct GPUTPCNNClusterizerHost::SofieState { std::array, 3> models; std::array, 3> sessions; #endif + std::array buffers; size_t workspaceSize = 0; unsigned int maxBatch = 0; }; @@ -318,13 +320,10 @@ MockedOrtAllocator* GPUTPCNNClusterizerHost::getMockedAllocator() #endif -void GPUTPCNNClusterizerHost::initSofie(const GPUSettingsProcessingNNclusterizer& settings, void* stream, int32_t device, bool hip, const GPUTPCNNClusterizerHost* source) +void GPUTPCNNClusterizerHost::initSofie(const GPUSettingsProcessingNNclusterizer& settings, void* stream, int32_t device, bool hip, const GPUTPCNNClusterizerHost* source, const std::array& buffers, const GPUTPCNNClusterizerHost* previous) { #ifdef GPUCA_HAS_SOFIE using Model = TMVA::Experimental::SOFIE::RGPUModel; - if (settings.nnLoadFromCCDB) { - throw std::runtime_error("SOFIE currently requires local ONNX files; disable nnLoadFromCCDB"); - } if (!settings.nnClusterizerBatchedMode || settings.nnClusterizerBatchedMode > static_cast(std::numeric_limits::max())) { throw std::runtime_error("SOFIE requires a positive batch capacity within int32 range"); } @@ -358,7 +357,19 @@ void GPUTPCNNClusterizerHost::initSofie(const GPUSettingsProcessingNNclusterizer auto state = std::make_shared(); state->maxBatch = settings.nnClusterizerBatchedMode; std::array paths{settings.nnClassificationPath, "", ""}; - if (!settings.nnClusterizerUseCfRegression) { + if (settings.nnLoadFromCCDB) { + auto mode = o2::utils::Str::tokenize(settings.nnEvalMode, ':'); + if (mode.size() != 2 || (mode[0] != "c1" && mode[0] != "c2") || (mode[1] != "r1" && mode[1] != "r2")) { + throw std::runtime_error("SOFIE CCDB loading requires nnEvalMode c1:r1, c1:r2, c2:r1 or c2:r2"); + } + paths[0] = "CCDB classification"; + if (!settings.nnClusterizerUseCfRegression) { + paths[1] = "CCDB regression 1"; + if (mode[1] == "r2") { + paths[2] = "CCDB regression 2"; + } + } + } else if (!settings.nnClusterizerUseCfRegression) { auto regression = o2::utils::Str::tokenize(settings.nnRegressionPath, ':'); if (regression.empty() || regression.size() > 2) { throw std::runtime_error("SOFIE expects one or two regression paths"); @@ -369,6 +380,13 @@ void GPUTPCNNClusterizerHost::initSofie(const GPUSettingsProcessingNNclusterizer } TMVA::Experimental::SOFIE::RModelParser_ONNX parser; std::unordered_map> compiled; + if (settings.nnLoadFromCCDB && previous && previous->mSofie) { + for (size_t i = 0; i < previous->mSofie->models.size(); i++) { + if (previous->mSofie->models[i] && !previous->mSofie->buffers[i].empty()) { + compiled.emplace(previous->mSofie->buffers[i], previous->mSofie->models[i]); + } + } + } for (size_t i = 0; i < paths.size(); i++) { mModelsUsed[i] = !paths[i].empty(); if (!mModelsUsed[i]) { @@ -377,9 +395,20 @@ void GPUTPCNNClusterizerHost::initSofie(const GPUSettingsProcessingNNclusterizer if (source) { state->models[i] = source->mSofie->models[i]; } else { - auto& model = compiled[paths[i]]; + if (settings.nnLoadFromCCDB && buffers[i].empty()) { + throw std::runtime_error("Missing or empty ONNX buffer for " + paths[i]); + } + auto& model = compiled[settings.nnLoadFromCCDB ? std::string(buffers[i]) : paths[i]]; + if (settings.nnLoadFromCCDB) { + state->buffers[i] = buffers[i]; + } if (!model) { - model = std::make_shared(parser.ParseGPU(paths[i])); + if (settings.nnLoadFromCCDB) { + std::istringstream input(state->buffers[i], std::ios::in | std::ios::binary); + model = std::make_shared(parser.ParseGPU(input)); + } else { + model = std::make_shared(parser.ParseGPU(paths[i])); + } if (model->GetPrecision() != inputType) { throw std::runtime_error("SOFIE model precision differs from nnInferenceInputDType: " + paths[i]); } @@ -404,6 +433,19 @@ void GPUTPCNNClusterizerHost::initSofie(const GPUSettingsProcessingNNclusterizer #endif } +bool GPUTPCNNClusterizerHost::hasSofieBuffers(const std::array& buffers) const +{ + if (!mSofie) { + return false; + } + for (size_t i = 0; i < buffers.size(); i++) { + if (mModelsUsed[i] && buffers[i] != mSofie->buffers[i]) { + return false; + } + } + return true; +} + void GPUTPCNNClusterizerHost::useSofie(const GPUTPCNNClusterizerHost& source) { mSofie = source.mSofie; diff --git a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.h b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.h index 98c6d73246d20..29a02d3147bff 100644 --- a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.h +++ b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.h @@ -15,6 +15,8 @@ #ifndef O2_GPUTPCNNCLUSTERIZERHOST_H #define O2_GPUTPCNNCLUSTERIZERHOST_H +#include +#include #include #include #include @@ -56,7 +58,8 @@ class GPUTPCNNClusterizerHost void createBoundary(GPUTPCNNClusterizer&); void createIndexLookup(GPUTPCNNClusterizer&); - void initSofie(const GPUSettingsProcessingNNclusterizer&, void* stream, int32_t device, bool hip, const GPUTPCNNClusterizerHost* source = nullptr); + void initSofie(const GPUSettingsProcessingNNclusterizer&, void* stream, int32_t device, bool hip, const GPUTPCNNClusterizerHost* source = nullptr, const std::array& buffers = {}, const GPUTPCNNClusterizerHost* previous = nullptr); + bool hasSofieBuffers(const std::array& buffers) const; void useSofie(const GPUTPCNNClusterizerHost&); void bindSofieWorkspace(void* workspace, size_t bytes); void inferenceSofie(int model, const void* input, size_t batch, void* output); diff --git a/GPU/README.md b/GPU/README.md index 836aeffad606b..9ca58abdf8cf5 100644 --- a/GPU/README.md +++ b/GPU/README.md @@ -29,18 +29,29 @@ an explicit compiler path; otherwise `nvcc`/`hipcc` is found on PATH at runtime. The compiler and GPU toolkit must be available on the worker at startup. Use GPU TPC cluster finding, `nnInferenceDevice=cuda` or `rocm` as appropriate, -`nnLoadFromCCDB=0`, and local classification/regression ONNX paths. Set both +local classification/regression ONNX paths with `nnLoadFromCCDB=0`, or +`nnLoadFromCCDB=1` to parse the ONNX buffers fetched by the existing CCDB workflow. +CCDB model slots follow `nnEvalMode` (classification, regression 1, optional regression 2). Set both `nnInferenceInputDType` and `nnInferenceOutputDType` to `FP32` or `FP16` to match the model. Mixed input/output precision, unsupported graphs, CPU SOFIE inference -and CCDB model loading are rejected explicitly. +are rejected explicitly. -Each distinct model path is parsed and compiled once during chain initialization. +Each distinct local model path is parsed and compiled once during chain initialization. +For CCDB, initialization is deferred until the first event has loaded its calibration +objects, before NN inference. Identical model buffers share a compiled program. +Later events reuse the sessions while the model bytes are unchanged. When CCDB +provides changed bytes, only new model contents are parsed and compiled, sharing +unchanged programs across lanes. All replacement sessions are prepared and +validated before synchronizing GPU work and replacing the old sessions. A failed +reload propagates an error without replacing the active sessions. O2 sizes the +workspace for the new models before inference. +Missing or empty required buffers fail explicitly. No temporary ONNX files are needed. Each lane then gets a session bound to its existing native stream. O2 allocates input, output, weights and intermediate storage through its registered clusterizer memory. When O2 recycles the scratch arena, weights are uploaded again on the -lane's stream; this does not recompile the model. Sessions persist until chain -finalization, which synchronizes before unloading the generated libraries. -Changing models, batch capacity or backend requires reinitializing the chain. +lane's stream; this does not recompile the model. Sessions persist until a model +reload or chain finalization, with GPU synchronization before unloading their libraries. +Changing local model paths, batch capacity, precision or backend requires reinitializing the chain. The supplied classification/regression networks have six Gemm layers, five Relu layers, 246 input values and respectively 1/5 output values. The clusterizer's @@ -51,5 +62,8 @@ the maximum batch; smaller final batches reuse the same code and workspace. Validation to run after building: ROOT's `TestSofieGPU` and CUDA/HIP `TestSofieGPUDevice`, then compare SOFIE against ORT using both supplied precisions and multiple batch sizes. Check numerical tolerances and downstream -cluster decisions; the scalar dense kernels are not performance tuned. +cluster decisions; the scalar dense kernels are not performance tuned. For CCDB, +check identical bytes in a new buffer, one changed model, changed hidden-layer +sizes, and invalid replacement data. Only changed contents should compile; an +invalid replacement must fail before switching sessions. No ROOT/O2 build or GPU test was run while implementing this change. diff --git a/GPU/Workflow/src/GPUWorkflowSpec.cxx b/GPU/Workflow/src/GPUWorkflowSpec.cxx index 990873f053fe3..9895ef93c6e54 100644 --- a/GPU/Workflow/src/GPUWorkflowSpec.cxx +++ b/GPU/Workflow/src/GPUWorkflowSpec.cxx @@ -1120,6 +1120,7 @@ void GPURecoWorkflowSpec::doCalibUpdates(o2::framework::ProcessingContext& pc, c needCalibUpdate = true; } if (mSpecConfig.nnLoadFromCCDB) { + needCalibUpdate |= mConfig->configProcessing.nn.mlFramework == "SOFIE"; auto dumpToFile = [](const char* buffer, std::size_t validSize, const std::string& path) { std::ofstream out(path, std::ios::binary | std::ios::trunc); if (!out.is_open()) { From bf4cb1d955c8a3b8cb1a00ae565ae496062917b1 Mon Sep 17 00:00:00 2001 From: Christian Sonnabend Date: Wed, 7 Oct 2026 23:43:07 +0200 Subject: [PATCH 4/9] Fix TRotMatrix issue --- dependencies/FindFairRoot.cmake | 3 +++ 1 file changed, 3 insertions(+) diff --git a/dependencies/FindFairRoot.cmake b/dependencies/FindFairRoot.cmake index b48d2cad67c4b..a13378314bf7e 100644 --- a/dependencies/FindFairRoot.cmake +++ b/dependencies/FindFairRoot.cmake @@ -60,6 +60,9 @@ if(NOT TARGET FairRoot::GeoBase) set_target_properties(FairRoot::GeoBase PROPERTIES INTERFACE_INCLUDE_DIRECTORIES ${FairRoot_INC} INTERFACE_LINK_LIBRARIES ${FairRoot_GeoBase}) + if(TARGET ROOT::TGeometry) + target_link_libraries(FairRoot::GeoBase INTERFACE ROOT::TGeometry) + endif() endif() if(NOT TARGET FairRoot::Base) From a804c479bd89832a398719b8f4e6e0d0abb0c79a Mon Sep 17 00:00:00 2001 From: Christian Sonnabend Date: Thu, 8 Oct 2026 00:08:28 +0200 Subject: [PATCH 5/9] Add device query failure --- GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu b/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu index fddc61a241c3f..34652858e9cd8 100644 --- a/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu +++ b/GPU/GPUTracking/Base/cuda/GPUReconstructionCUDA.cu @@ -638,7 +638,9 @@ void GPUReconstructionCUDA::loadKernelModules(bool perKernel) int32_t GPUReconstructionCUDA::GetNativeGPUDevice() const { int device = -1; - GPUChkErr(cudaGetDevice(&device)); + if (GPUChkErrInternal(cudaGetDevice(&device), __FILE__, __LINE__)) { + throw std::runtime_error("GPU device query failed"); + } return device; } From 6b8f72fbcc437a2bd36e3a8e5e6f42575646b523 Mon Sep 17 00:00:00 2001 From: Christian Sonnabend Date: Thu, 8 Oct 2026 00:09:07 +0200 Subject: [PATCH 6/9] Fix build issues --- Detectors/Base/src/O2Tessellated.cxx | 1 + Detectors/MUON/MCH/Evaluation/src/Draw.cxx | 1 + 2 files changed, 2 insertions(+) diff --git a/Detectors/Base/src/O2Tessellated.cxx b/Detectors/Base/src/O2Tessellated.cxx index d50b922cd8d25..2393d6f5d8321 100644 --- a/Detectors/Base/src/O2Tessellated.cxx +++ b/Detectors/Base/src/O2Tessellated.cxx @@ -16,6 +16,7 @@ // Will be deleted once we get this from ROOT. #include +#include #include #include "TGeoManager.h" diff --git a/Detectors/MUON/MCH/Evaluation/src/Draw.cxx b/Detectors/MUON/MCH/Evaluation/src/Draw.cxx index 918d89cff0912..3c721b01007d4 100644 --- a/Detectors/MUON/MCH/Evaluation/src/Draw.cxx +++ b/Detectors/MUON/MCH/Evaluation/src/Draw.cxx @@ -18,6 +18,7 @@ #include #include #include +#include #include #include #include From 82d8eca6fb07bab7e18bba23eb6c86d85c03b11b Mon Sep 17 00:00:00 2001 From: Christian Sonnabend Date: Thu, 8 Oct 2026 10:31:03 +0200 Subject: [PATCH 7/9] Include Rtypes.h --- GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx index bffe47402e8f8..52becb6e7ae29 100644 --- a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx +++ b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx @@ -12,6 +12,10 @@ /// \file GPUTPCNNClusterizerHost.cxx /// \author Christian Sonnabend +#ifdef GPUCA_HAS_SOFIE +#include "Rtypes.h" +#endif + #include #include "GPUTPCNNClusterizerHost.h" From d4a2c019e77f00601fea932574ccaf27f2c56d43 Mon Sep 17 00:00:00 2001 From: Christian Sonnabend Date: Thu, 8 Oct 2026 10:36:34 +0200 Subject: [PATCH 8/9] Fix devicetype for SOFIE --- GPU/GPUTracking/Global/GPUChainTracking.cxx | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/GPU/GPUTracking/Global/GPUChainTracking.cxx b/GPU/GPUTracking/Global/GPUChainTracking.cxx index fd3fbf530e770..2992fbd5b6099 100644 --- a/GPU/GPUTracking/Global/GPUChainTracking.cxx +++ b/GPU/GPUTracking/Global/GPUChainTracking.cxx @@ -1058,8 +1058,8 @@ void GPUChainTracking::InitSofieClusterizer(bool deferCCDB) if (previous && (!settings.nnLoadFromCCDB || previous->hasSofieBuffers(buffers))) { return; } - const bool hip = mRec->GetDeviceType() == GPUReconstruction::GetDeviceType("HIP"); - if ((!hip && mRec->GetDeviceType() != GPUReconstruction::GetDeviceType("CUDA")) || !(GetRecoStepsGPU() & RecoStep::TPCClusterFinding)) { + const bool hip = mRec->GetDeviceType() == GPUReconstruction::DeviceType::HIP; + if ((!hip && mRec->GetDeviceType() != GPUReconstruction::DeviceType::CUDA) || !(GetRecoStepsGPU() & RecoStep::TPCClusterFinding)) { throw std::runtime_error("SOFIE requires TPC cluster finding on CUDA or HIP"); } const int lanes = GetProcessingSettings().nTPCClustererLanes; From a88675ff0c9b5f06972a0e3e5a53f12a70ce7dcf Mon Sep 17 00:00:00 2001 From: Christian Sonnabend Date: Thu, 8 Oct 2026 10:59:20 +0200 Subject: [PATCH 9/9] Add SOFIE logging --- GPU/GPUTracking/Global/GPUChainTracking.cxx | 7 +++++++ .../TPCClusterFinder/GPUTPCNNClusterizerHost.cxx | 8 ++++++++ 2 files changed, 15 insertions(+) diff --git a/GPU/GPUTracking/Global/GPUChainTracking.cxx b/GPU/GPUTracking/Global/GPUChainTracking.cxx index 2992fbd5b6099..bdae01693a29a 100644 --- a/GPU/GPUTracking/Global/GPUChainTracking.cxx +++ b/GPU/GPUTracking/Global/GPUChainTracking.cxx @@ -1069,6 +1069,10 @@ void GPUChainTracking::InitSofieClusterizer(bool deferCCDB) if (deferCCDB && settings.nnLoadFromCCDB) { return; } + GPUInfo("SOFIE: %s, backend=%s, device=%d, lanes=%d, source=%s, batch capacity=%u", + previous ? "CCDB model change detected; reloading" : "initializing", + hip ? "HIP" : "CUDA", GetNativeGPUDevice(), lanes, + settings.nnLoadFromCCDB ? "CCDB" : "local ONNX files", settings.nnClusterizerBatchedMode); std::vector> applications; for (int lane = 0; lane < lanes; lane++) { auto host = std::make_unique(); @@ -1080,7 +1084,10 @@ void GPUChainTracking::InitSofieClusterizer(bool deferCCDB) if (previous) { SynchronizeGPU(); } + const bool reloaded = previous != nullptr; mSofieApplications = std::move(applications); + GPUInfo("SOFIE: %s succeeded; inference backend active on device %d with %d lanes", + reloaded ? "CCDB model reload" : "initialization", GetNativeGPUDevice(), lanes); #else throw std::runtime_error("SOFIE was requested but GPUCA_BUILD_SOFIE is disabled"); #endif diff --git a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx index 52becb6e7ae29..7068cf8dfec2b 100644 --- a/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx +++ b/GPU/GPUTracking/TPCClusterFinder/GPUTPCNNClusterizerHost.cxx @@ -23,6 +23,7 @@ #include "GPUSettings.h" #include "ML/3rdparty/GPUORTFloat16.h" #include "GPUReconstruction.h" +#include "GPULogging.h" #include "GPUTPCGeometry.h" #include "DataFormatsTPC/Constants.h" #include "clusterFinderDefs.h" @@ -407,6 +408,9 @@ void GPUTPCNNClusterizerHost::initSofie(const GPUSettingsProcessingNNclusterizer state->buffers[i] = buffers[i]; } if (!model) { + GPUInfo("SOFIE: preparing model %zu (%s), backend=%s, architecture=%s, precision=%s", + i, paths[i].c_str(), hip ? "HIP" : "CUDA", settings.sofieArchitecture.c_str(), + inputType == Model::Precision::Float16 ? "FP16" : "FP32"); if (settings.nnLoadFromCCDB) { std::istringstream input(state->buffers[i], std::ios::in | std::ios::binary); model = std::make_shared(parser.ParseGPU(input)); @@ -421,6 +425,10 @@ void GPUTPCNNClusterizerHost::initSofie(const GPUSettingsProcessingNNclusterizer } model->WorkspaceSize(settings.nnClusterizerBatchedMode); model->Compile({hip ? Model::Backend::HIP : Model::Backend::CUDA, settings.sofieCompiler, settings.sofieArchitecture}); + GPUInfo("SOFIE: model %zu compilation succeeded, input=%zu, output=%zu", + i, model->InputSize(), model->OutputSize()); + } else { + GPUInfo("SOFIE: model %zu reuses an existing compiled program", i); } state->models[i] = model; }