Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 5 additions & 5 deletions GPU/Common/GPUCommonAlgorithm.h
Original file line number Diff line number Diff line change
Expand Up @@ -32,13 +32,13 @@ class GPUCommonAlgorithm
template <class T>
GPUd() static void sort(T* begin, T* end);
template <class T>
GPUd() static void sortInBlock(T* begin, T* end);
GPUd() static void sortInBlock(T* begin, T* end GPUCA_THREAD_INFO_DECL);
template <class T>
GPUd() static void sortDeviceDynamic(T* begin, T* end);
template <class T, class S>
GPUd() static void sort(T* begin, T* end, const S& comp);
template <class T, class S>
GPUd() static void sortInBlock(T* begin, T* end, const S& comp);
GPUd() static void sortInBlock(T* begin, T* end, const S& comp GPUCA_THREAD_INFO_DECL);
template <class T, class S>
GPUd() static void sortDeviceDynamic(T* begin, T* end, const S& comp);
#if __cplusplus >= 202002L // sortOnDevice takes an auto parameter
Expand Down Expand Up @@ -268,17 +268,17 @@ GPUdi() void GPUCommonAlgorithm::sort(T* begin, T* end, const S& comp)
}

template <class T>
GPUdi() void GPUCommonAlgorithm::sortInBlock(T* begin, T* end)
GPUdi() void GPUCommonAlgorithm::sortInBlock(T* begin, T* end GPUCA_THREAD_INFO_DECL)
{
#ifndef GPUCA_GPUCODE
GPUCommonAlgorithm::sort(begin, end);
#else
GPUCommonAlgorithm::sortInBlock(begin, end, [](auto&& x, auto&& y) { return x < y; });
GPUCommonAlgorithm::sortInBlock(begin, end, [](auto&& x, auto&& y) { return x < y; } GPUCA_THREAD_INFO_PROVIDE);
#endif
}

template <class T, class S>
GPUdi() void GPUCommonAlgorithm::sortInBlock(T* begin, T* end, const S& comp)
GPUdi() void GPUCommonAlgorithm::sortInBlock(T* begin, T* end, const S& comp GPUCA_THREAD_INFO_DECL)
{
#ifndef GPUCA_GPUCODE
GPUCommonAlgorithm::sort(begin, end, comp);
Expand Down
4 changes: 2 additions & 2 deletions GPU/Common/GPUCommonAlgorithmThrust.h
Original file line number Diff line number Diff line change
Expand Up @@ -63,15 +63,15 @@ GPUdi() void GPUCommonAlgorithm::sort(T* begin, T* end, const S& comp)
}

template <class T>
GPUdi() void GPUCommonAlgorithm::sortInBlock(T* begin, T* end) // TODO: Try cub::BlockMergeSort
GPUdi() void GPUCommonAlgorithm::sortInBlock(T* begin, T* end GPUCA_THREAD_INFO_DECL) // TODO: Try cub::BlockMergeSort
{
if (get_local_id(0) == 0) {
sortDeviceDynamic(begin, end);
}
}

template <class T, class S>
GPUdi() void GPUCommonAlgorithm::sortInBlock(T* begin, T* end, const S& comp)
GPUdi() void GPUCommonAlgorithm::sortInBlock(T* begin, T* end, const S& comp GPUCA_THREAD_INFO_DECL)
{
if (get_local_id(0) == 0) {
sortDeviceDynamic(begin, end, comp);
Expand Down
21 changes: 21 additions & 0 deletions GPU/Common/GPUCommonDefAPI.h
Original file line number Diff line number Diff line change
Expand Up @@ -277,6 +277,16 @@
#define get_group_id(dim) (blockIdx.x)
#elif defined(__OPENCL__)
// Using OpenCL defaults
#elif defined(__METAL__)
// MSL has no work-item builtins. They arrive as attributes on the entry point,
// named there after the nBlocks / nThreads / iBlock / iThread that Thread()
// already takes, so these resolve both there and in every function below it.
#define get_global_id(dim) (iBlock * nThreads + iThread)
#define get_global_size(dim) (nBlocks * nThreads)
#define get_num_groups(dim) (nBlocks)
#define get_local_id(dim) (iThread)
#define get_local_size(dim) (nThreads)
#define get_group_id(dim) (iBlock)
#else
#define get_global_id(dim) iBlock
#define get_global_size(dim) nBlocks
Expand All @@ -286,5 +296,16 @@
#define get_group_id(dim) iBlock
#endif

// A device function that uses the helpers above but has none of the indices in
// scope needs them passed in on Metal, where they are ordinary parameters.
// Both expand to nothing everywhere else, and go last in the parameter list.
#ifdef __METAL__
#define GPUCA_THREAD_INFO_DECL , int32_t nBlocks, int32_t nThreads, int32_t iBlock, int32_t iThread
#define GPUCA_THREAD_INFO_PROVIDE , nBlocks, nThreads, iBlock, iThread
#else
#define GPUCA_THREAD_INFO_DECL
#define GPUCA_THREAD_INFO_PROVIDE
#endif

// clang-format on
#endif
11 changes: 10 additions & 1 deletion GPU/GPUTracking/Base/GPUReconstructionKernelMacros.h
Original file line number Diff line number Diff line change
Expand Up @@ -63,8 +63,17 @@
#define GPUCA_ATTRRES(...) GPUCA_M_EXPAND(GPUCA_M_CAT(GPUCA_ATTRRES_, GPUCA_M_FIRST(__VA_ARGS__)))(__VA_ARGS__)

// GPU Kernel entry point
// MSL requires every kernel parameter to carry an attribute, and supplies the
// grid dimensions the same way, so the backend gets to shape both ends of the
// parameter list.
#ifndef GPUCA_KRNL_SECTOR_ARG
#define GPUCA_KRNL_SECTOR_ARG int32_t _iSector_internal

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't understand why you need a special treatment for the sector variable in metal?
The sector variable is a normal variable, which is passed in like any other parameter to function calls.

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

It needs to be bound to a buffer. There is some buffer counting logic which was in a subsequent commit and now sits together with this one.

#endif
#ifndef GPUCA_KRNL_GRID_ARGS
#define GPUCA_KRNL_GRID_ARGS
#endif
#define GPUCA_KRNLGPU_DEF(x_class, x_attributes, x_arguments, ...) \
GPUg() void GPUCA_ATTRRES(GPUCA_M_STRIP(x_attributes)) GPUCA_M_CAT(krnl_, GPUCA_M_KRNL_NAME(x_class))(GPUCA_CONSMEM_PTR int32_t _iSector_internal GPUCA_M_STRIP(x_arguments))
GPUg() void GPUCA_ATTRRES(GPUCA_M_STRIP(x_attributes)) GPUCA_M_CAT(krnl_, GPUCA_M_KRNL_NAME(x_class))(GPUCA_CONSMEM_PTR GPUCA_KRNL_SECTOR_ARG GPUCA_M_STRIP(x_arguments) GPUCA_KRNL_GRID_ARGS)

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Hm, but this means that you pass in local and global id and size as argument to the kernel function.
However, in OpenCL / CUDA / HIP, these varaibles are available everywhere, without being passed in.
I.e., they are also available in subfunctions. And I don't want to pass them in explicitly to each place where they are used. Is this somehow possible with metal?

@ktf ktf Oct 9, 2026 •

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

As far as I understand, no, and it's a limitation / design choice of the internal representation which does not expose any getter for the thread-related indices. They need to be passed as specially marked arguments. I guess the idea is that signatures are more "functional" such a way and there is no hidden state in the functions. This is no different from what happens on the CPU where you expect that 4 of the 6 OpenCL helpers are provided as parameters / available in scope. This does the same for the other two.

I've reordered the series so this migration comes first, ahead of any Metal change — on its own it is backend-neutral: it rewrites 105 uses of the six helpers across 21 files, and only four functions gain an index parameter (sortInBlock, buildCluster, findMinimaAndPeaks, isPeak). That should let you evaluate the impact of the whole change, and then we can decide.

As a side benefit, the index arithmetic no longer assumes one thread per block, so it is correct on the CPU for any nThreads. Parallelism over blocks is already there; this would make it possible to also use the thread dimension within a block on the host, if desired / supported by TBB.


#ifdef GPUCA_KRNL_DEFONLY
#define GPUCA_KRNLGPU(...) GPUCA_KRNLGPU_DEF(__VA_ARGS__);
Expand Down
10 changes: 10 additions & 0 deletions GPU/GPUTracking/Base/metal/GPUReconstructionMETAL.metal
Original file line number Diff line number Diff line change
Expand Up @@ -70,6 +70,16 @@ using namespace metal;
device char* pConstantRaw [[buffer(1)]],
#define GPUCA_CONSMEM (*(device GPUConstantMem*)pConstantRaw)

// Every kernel parameter needs an attribute, so the sector index arrives as a
// buffer rather than by value, and the grid dimensions come in at the end, where
// GPUCommonDefAPI.h's get_group_id() and friends pick them up.
#define GPUCA_KRNL_SECTOR_ARG constant int32_t& _iSector_internal [[buffer(2)]]
#define GPUCA_KRNL_GRID_ARGS \
, uint iBlock [[threadgroup_position_in_grid]] \
, uint iThread [[thread_position_in_threadgroup]] \
, uint nThreads [[threads_per_threadgroup]] \
, uint nBlocks [[threadgroups_per_grid]]

#include "GPUReconstructionKernelList.h"

// clang-format on
Expand Down
10 changes: 5 additions & 5 deletions GPU/GPUTracking/DataCompression/GPUTPCCompressionKernels.cxx
Original file line number Diff line number Diff line change
Expand Up @@ -274,16 +274,16 @@ GPUdii() void GPUTPCCompressionKernels::Thread<GPUTPCCompressionKernels::step1un
static_assert(GPUCA_GET_THREAD_COUNT(GPUCA_LB_GPUTPCCompressionKernels_step1unattached) * 2 <= constants::TPC_COMP_CHUNK_SIZE);
#endif
#ifdef GPUCA_DETERMINISTIC_MODE
CAAlgo::sortInBlock(sortBuffer, sortBuffer + count, GPUTPCCompressionKernels_Compare<GPUSettings::SortZPadTime>(clusters->clusters[iSector][iRow]));
CAAlgo::sortInBlock(sortBuffer, sortBuffer + count, GPUTPCCompressionKernels_Compare<GPUSettings::SortZPadTime>(clusters->clusters[iSector][iRow]) GPUCA_THREAD_INFO_PROVIDE);
#else // GPUCA_DETERMINISTIC_MODE
if (param.rec.tpc.compressionSortOrder == GPUSettings::SortZPadTime) {
CAAlgo::sortInBlock(sortBuffer, sortBuffer + count, GPUTPCCompressionKernels_Compare<GPUSettings::SortZPadTime>(clusters->clusters[iSector][iRow]));
CAAlgo::sortInBlock(sortBuffer, sortBuffer + count, GPUTPCCompressionKernels_Compare<GPUSettings::SortZPadTime>(clusters->clusters[iSector][iRow]) GPUCA_THREAD_INFO_PROVIDE);
} else if (param.rec.tpc.compressionSortOrder == GPUSettings::SortZTimePad) {
CAAlgo::sortInBlock(sortBuffer, sortBuffer + count, GPUTPCCompressionKernels_Compare<GPUSettings::SortZTimePad>(clusters->clusters[iSector][iRow]));
CAAlgo::sortInBlock(sortBuffer, sortBuffer + count, GPUTPCCompressionKernels_Compare<GPUSettings::SortZTimePad>(clusters->clusters[iSector][iRow]) GPUCA_THREAD_INFO_PROVIDE);
} else if (param.rec.tpc.compressionSortOrder == GPUSettings::SortPad) {
CAAlgo::sortInBlock(sortBuffer, sortBuffer + count, GPUTPCCompressionKernels_Compare<GPUSettings::SortPad>(clusters->clusters[iSector][iRow]));
CAAlgo::sortInBlock(sortBuffer, sortBuffer + count, GPUTPCCompressionKernels_Compare<GPUSettings::SortPad>(clusters->clusters[iSector][iRow]) GPUCA_THREAD_INFO_PROVIDE);
} else if (param.rec.tpc.compressionSortOrder == GPUSettings::SortTime) {
CAAlgo::sortInBlock(sortBuffer, sortBuffer + count, GPUTPCCompressionKernels_Compare<GPUSettings::SortTime>(clusters->clusters[iSector][iRow]));
CAAlgo::sortInBlock(sortBuffer, sortBuffer + count, GPUTPCCompressionKernels_Compare<GPUSettings::SortTime>(clusters->clusters[iSector][iRow]) GPUCA_THREAD_INFO_PROVIDE);
}
#endif // GPUCA_DETERMINISTIC_MODE
GPUbarrier();
Expand Down
26 changes: 18 additions & 8 deletions GPU/GPUTracking/Definitions/GPUDef.h
Original file line number Diff line number Diff line change
Expand Up @@ -21,17 +21,27 @@
#include "GPUDefParametersWrapper.h"
#include "GPUCommonRtypes.h"

// Macros for masking ptrs in OpenCL kernel calls as uint64_t (The API only allows us to pass buffer objects)
// Macros for kernel arguments. OpenCL can only pass buffer objects, so pointers
// are masked as uint64_t and cast back inside the kernel. MSL needs an explicit
// buffer index on every parameter, but can bind a pointer directly. The index is
// emitted per argument by o2_gpu_add_kernel; 0, 1 and 2 are taken by gpu_mem,
// the constant memory and the sector index.
#ifdef __OPENCL__
#define GPUPtr1(a, b) uint64_t b
#ifdef __OPENCL__
#define GPUPtr2(a, b) ((__generic a) (a) b)
#else
#define GPUPtr2(a, b) ((__global a) (a) b)
#endif
#define GPUPtr1(idx, a, b) uint64_t b
#define GPUPtr2(a, b) ((__generic a) (a) b)
#define GPUArg1(idx, a, b) a b
#elif defined(__METAL__)
// As for OpenCL, pointers travel as a 64-bit address: a pointer to a derived
// class is not a valid kernel argument type in MSL either.
#define GPUPtr1(idx, a, b) constant uint64_t& b [[buffer(idx)]]
// through device and then to generic: the kernel's own buffers are device
// memory, but the Thread() entry points take the pointer unannotated
#define GPUPtr2(a, b) ((a)((device a)(b)))
#define GPUArg1(idx, a, b) constant a& b [[buffer(idx)]]

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

So in principle this is fine, if it is needed by METAL. In any case it is not exposed to the developed but hidden behind the scenes. However, I see 2 possible issues:

  • You seem to number the buffers just counting up, but it could be that 2 of the pointers point to the same buffer. Would that be a problem?
  • You define it as constant a&, but there is no guarantee that the data passed in here is constant, the functions are allowed to modify it. What does constant mean here?

#else
#define GPUPtr1(a, b) a b
#define GPUPtr1(idx, a, b) a b
#define GPUPtr2(a, b) b
#define GPUArg1(idx, a, b) a b
#endif

#define GPUCA_EVDUMP_FILE "event"
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -773,7 +773,7 @@ GPUd() void GPUTPCCFHIPTailConnector::Thread<0>(int32_t nBlocks, int32_t nThread
} else {
return t1.qMax < t2.qMax;
}
});
} GPUCA_THREAD_INFO_PROVIDE);
if (iThread > 0) {
return;
}
Expand Down
2 changes: 1 addition & 1 deletion GPU/GPUTracking/TPCClusterFinder/GPUTPCCFClusterizer.h
Original file line number Diff line number Diff line change
Expand Up @@ -59,7 +59,7 @@ class GPUTPCCFClusterizer : public GPUKernelTemplate

static GPUd() void computeClustersImpl(int32_t, int32_t, int32_t, int32_t, processorType&, const CfFragment&, GPUSharedMemory&, const CfArray2D<PackedCharge>&, const CfChargePos*, const GPUSettingsRec&, MCLabelAccumulator*, uint32_t, uint32_t, uint32_t*, tpc::ClusterNative*, uint32_t*, int8_t);

static GPUd() void buildCluster(const GPUSettingsRec&, const CfArray2D<PackedCharge>&, CfChargePos, CfChargePos*, PackedCharge*, uint8_t*, ClusterAccumulator*, MCLabelAccumulator*);
static GPUd() void buildCluster(const GPUSettingsRec&, const CfArray2D<PackedCharge>&, CfChargePos, CfChargePos*, PackedCharge*, uint8_t*, ClusterAccumulator*, MCLabelAccumulator* GPUCA_THREAD_INFO_DECL);

static GPUd() uint32_t sortIntoBuckets(processorType&, const tpc::ClusterNative&, uint32_t, uint32_t, uint32_t*, tpc::ClusterNative*);

Expand Down
4 changes: 2 additions & 2 deletions GPU/GPUTracking/TPCClusterFinder/GPUTPCCFClusterizer.inc
Original file line number Diff line number Diff line change
Expand Up @@ -49,7 +49,7 @@ GPUdii() void GPUTPCCFClusterizer::computeClustersImpl(int32_t nBlocks, int32_t
smem.buf,
smem.innerAboveThreshold,
&pc,
labelAcc);
labelAcc GPUCA_THREAD_INFO_PROVIDE);

if (idx >= clusternum) {
return;
Expand Down Expand Up @@ -151,7 +151,7 @@ GPUdii() void GPUTPCCFClusterizer::buildCluster(
PackedCharge* buf,
uint8_t* innerAboveThreshold,
ClusterAccumulator* myCluster,
MCLabelAccumulator* labelAcc)
MCLabelAccumulator* labelAcc GPUCA_THREAD_INFO_DECL)
{
uint16_t ll = get_local_id(0);

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -60,7 +60,7 @@ GPUdii() void GPUTPCCFNoiseSuppression::noiseSuppressionImpl(int32_t nBlocks, in
smem.buf,
&minimas,
&bigger,
&peaksAround);
&peaksAround GPUCA_THREAD_INFO_PROVIDE);

peaksAround &= bigger;

Expand Down Expand Up @@ -173,7 +173,7 @@ GPUd() void GPUTPCCFNoiseSuppression::findMinimaAndPeaks(
PackedCharge* buf,
uint64_t* minimas,
uint64_t* bigger,
uint64_t* peaks)
uint64_t* peaks GPUCA_THREAD_INFO_DECL)
{
uint16_t ll = get_local_id(0);

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -69,7 +69,7 @@ class GPUTPCCFNoiseSuppression : public GPUKernelTemplate

static GPUdi() bool keepPeak(uint64_t, uint64_t);

static GPUd() void findMinimaAndPeaks(const CfArray2D<PackedCharge>&, const CfArray2D<uint8_t>&, const GPUSettingsRec&, float, const CfChargePos&, CfChargePos*, PackedCharge*, uint64_t*, uint64_t*, uint64_t*);
static GPUd() void findMinimaAndPeaks(const CfArray2D<PackedCharge>&, const CfArray2D<uint8_t>&, const GPUSettingsRec&, float, const CfChargePos&, CfChargePos*, PackedCharge*, uint64_t*, uint64_t*, uint64_t* GPUCA_THREAD_INFO_DECL);
};

} // namespace o2::gpu
Expand Down
4 changes: 2 additions & 2 deletions GPU/GPUTracking/TPCClusterFinder/GPUTPCCFPeakFinder.cxx
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@ GPUdii() bool GPUTPCCFPeakFinder::isPeak(
const CfArray2D<PackedCharge>& chargeMap,
const GPUSettingsRec& calib,
CfChargePos* posBcast,
PackedCharge* buf)
PackedCharge* buf GPUCA_THREAD_INFO_DECL)
{
uint16_t ll = get_local_id(0);

Expand Down Expand Up @@ -111,7 +111,7 @@ GPUd() void GPUTPCCFPeakFinder::findPeaksImpl(int32_t nBlocks, int32_t nThreads,
bool hasLostBaseline = pos.valid() ? padHasLostBaseline[pos.gpad] : true;
charge = hasLostBaseline ? 0.f : charge;

uint8_t peak = isPeak(smem, charge, pos, SCRATCH_PAD_SEARCH_N, chargeMap, calib, smem.posBcast, smem.buf);
uint8_t peak = isPeak(smem, charge, pos, SCRATCH_PAD_SEARCH_N, chargeMap, calib, smem.posBcast, smem.buf GPUCA_THREAD_INFO_PROVIDE);

// Exit early if dummy. See comment above.
bool iamDummy = (idx >= digitnum);
Expand Down
2 changes: 1 addition & 1 deletion GPU/GPUTracking/TPCClusterFinder/GPUTPCCFPeakFinder.h
Original file line number Diff line number Diff line change
Expand Up @@ -53,7 +53,7 @@ class GPUTPCCFPeakFinder : public GPUKernelTemplate
private:
static GPUd() void findPeaksImpl(int32_t, int32_t, int32_t, int32_t, GPUSharedMemory&, const CfArray2D<PackedCharge>&, const uint8_t*, const CfChargePos*, tpccf::SizeT, const GPUSettingsRec&, const TPCPadGainCalib&, uint8_t*, CfArray2D<uint8_t>&);

static GPUd() bool isPeak(GPUSharedMemory&, tpccf::Charge, const CfChargePos&, uint16_t, const CfArray2D<PackedCharge>&, const GPUSettingsRec&, CfChargePos*, PackedCharge*);
static GPUd() bool isPeak(GPUSharedMemory&, tpccf::Charge, const CfChargePos&, uint16_t, const CfArray2D<PackedCharge>&, const GPUSettingsRec&, CfChargePos*, PackedCharge* GPUCA_THREAD_INFO_DECL);
};

} // namespace o2::gpu
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -344,7 +344,7 @@ GPUdii() void GPUTPCNNClusterizerKernels::Thread<GPUTPCNNClusterizerKernels::pub
smem.buf,
smem.innerAboveThreshold,
&dummy_pc,
labelAcc);
labelAcc GPUCA_THREAD_INFO_PROVIDE);
}
return;
}
Expand All @@ -364,7 +364,7 @@ GPUdii() void GPUTPCNNClusterizerKernels::Thread<GPUTPCNNClusterizerKernels::pub
smem.buf,
smem.innerAboveThreshold,
&dummy_pc,
labelAcc);
labelAcc GPUCA_THREAD_INFO_PROVIDE);
}
if ((clusterer.mPmemory->fragment).isOverlap(peak.time())) {
if (clusterer.mPclusterPosInRow) {
Expand Down Expand Up @@ -538,7 +538,7 @@ GPUdii() void GPUTPCNNClusterizerKernels::Thread<GPUTPCNNClusterizerKernels::pub
smem.buf,
smem.innerAboveThreshold,
&dummy_pc,
labelAcc);
labelAcc GPUCA_THREAD_INFO_PROVIDE);
}
return;
}
Expand All @@ -558,7 +558,7 @@ GPUdii() void GPUTPCNNClusterizerKernels::Thread<GPUTPCNNClusterizerKernels::pub
smem.buf,
smem.innerAboveThreshold,
&dummy_pc,
labelAcc);
labelAcc GPUCA_THREAD_INFO_PROVIDE);
}
if ((clusterer.mPmemory->fragment).isOverlap(peak.time())) {
if (clusterer.mPclusterPosInRow) {
Expand Down
6 changes: 4 additions & 2 deletions GPU/GPUTracking/cmake/kernel_helpers.cmake
Original file line number Diff line number Diff line change
Expand Up @@ -55,11 +55,13 @@ function(o2_gpu_add_kernel kernel_name kernel_files)
math(EXPR n "${n} - 1")
foreach(i RANGE 3 ${n} 2)
math(EXPR j "${i} + 1")
# buffer indices 0, 1 and 2 are gpu_mem, the constant memory and the sector
math(EXPR TMP_ARG_IDX "3 + (${i} - 3) / 2")
if(${ARGV${i}} MATCHES "\\*$")
string(APPEND OPT1 ",GPUPtr1(${ARGV${i}},${ARGV${j}})")
string(APPEND OPT1 ",GPUPtr1(${TMP_ARG_IDX},${ARGV${i}},${ARGV${j}})")
string(APPEND OPT2 ",GPUPtr2(${ARGV${i}},${ARGV${j}})")
else()
string(APPEND OPT1 ",${ARGV${i}} ${ARGV${j}}")
string(APPEND OPT1 ",GPUArg1(${TMP_ARG_IDX},${ARGV${i}},${ARGV${j}})")
string(APPEND OPT2 ",${ARGV${j}}")
endif()
string(APPEND OPT3 ",${ARGV${i}}")
Expand Down
Loading