diff options
author | Sylvain Jeaugey <sjeaugey@nvidia.com> | 2020-01-17 03:02:42 +0300 |
---|---|---|
committer | Sylvain Jeaugey <sjeaugey@nvidia.com> | 2020-03-21 00:58:36 +0300 |
commit | b221128ecacf4ce1b3054172b9f30163307042c5 (patch) | |
tree | 43aa7da7992fea7ce30b8cc3e6220bc56f93dd16 /src/collectives/device/reduce.h | |
parent | c38f174bd436031dbc79dce19ff969f377976a8a (diff) |
2.6.4-1
Add support for network collectives.
Add support for XML topology dump/injection.
Add text values for GDR and P2P Levels, including "NVL".
Add speed detection for PCI, Infiniband and Ethernet cards.
Add CPU detection for ARM and AMD CPUs.
Add support for adaptive routing on Infiniband.
Change NET plugin API to v3 : merge PCI path and GPU pointer
capability into a single structure and add other properties.
Diffstat (limited to 'src/collectives/device/reduce.h')
-rw-r--r-- | src/collectives/device/reduce.h | 9 |
1 files changed, 9 insertions, 0 deletions
diff --git a/src/collectives/device/reduce.h b/src/collectives/device/reduce.h index 0680abe..e36613f 100644 --- a/src/collectives/device/reduce.h +++ b/src/collectives/device/reduce.h @@ -50,6 +50,9 @@ __device__ void ncclReduceRingKernel(struct CollectiveArgs* args) { template<int UNROLL, class FUNC, typename T> __device__ void ncclReduceTreeKernel(struct CollectiveArgs* args) { } +template<int UNROLL, class FUNC, typename T> +__device__ void ncclReduceCollNetKernel(struct CollectiveArgs* args) { } + template<int UNUSED, class FUNC, typename T> __device__ void ncclReduceRingLLKernel(struct CollectiveArgs* args) { const int tid = threadIdx.x; @@ -94,6 +97,9 @@ __device__ void ncclReduceRingLLKernel(struct CollectiveArgs* args) { template<int UNUSED, class FUNC, typename T> __device__ void ncclReduceTreeLLKernel(struct CollectiveArgs* args) { } +template<int UNUSED, class FUNC, typename T> +__device__ void ncclReduceCollNetLLKernel(struct CollectiveArgs* args) { } + #include "prims_ll128.h" template<int UNUSED, class FUNC, typename T> __device__ void ncclReduceRingLL128Kernel(struct CollectiveArgs* args) { @@ -138,3 +144,6 @@ __device__ void ncclReduceRingLL128Kernel(struct CollectiveArgs* args) { template<int UNUSED, class FUNC, typename T> __device__ void ncclReduceTreeLL128Kernel(struct CollectiveArgs* args) { } + +template<int UNUSED, class FUNC, typename T> +__device__ void ncclReduceCollNetLL128Kernel(struct CollectiveArgs* args) { } |