From b221128ecacf4ce1b3054172b9f30163307042c5 Mon Sep 17 00:00:00 2001 From: Sylvain Jeaugey Date: Thu, 16 Jan 2020 16:02:42 -0800 Subject: 2.6.4-1 Add support for network collectives. Add support for XML topology dump/injection. Add text values for GDR and P2P Levels, including "NVL". Add speed detection for PCI, Infiniband and Ethernet cards. Add CPU detection for ARM and AMD CPUs. Add support for adaptive routing on Infiniband. Change NET plugin API to v3 : merge PCI path and GPU pointer capability into a single structure and add other properties. --- src/collectives/device/reduce.h | 9 +++++++++ 1 file changed, 9 insertions(+) (limited to 'src/collectives/device/reduce.h') diff --git a/src/collectives/device/reduce.h b/src/collectives/device/reduce.h index 0680abe..e36613f 100644 --- a/src/collectives/device/reduce.h +++ b/src/collectives/device/reduce.h @@ -50,6 +50,9 @@ __device__ void ncclReduceRingKernel(struct CollectiveArgs* args) { template __device__ void ncclReduceTreeKernel(struct CollectiveArgs* args) { } +template +__device__ void ncclReduceCollNetKernel(struct CollectiveArgs* args) { } + template __device__ void ncclReduceRingLLKernel(struct CollectiveArgs* args) { const int tid = threadIdx.x; @@ -94,6 +97,9 @@ __device__ void ncclReduceRingLLKernel(struct CollectiveArgs* args) { template __device__ void ncclReduceTreeLLKernel(struct CollectiveArgs* args) { } +template +__device__ void ncclReduceCollNetLLKernel(struct CollectiveArgs* args) { } + #include "prims_ll128.h" template __device__ void ncclReduceRingLL128Kernel(struct CollectiveArgs* args) { @@ -138,3 +144,6 @@ __device__ void ncclReduceRingLL128Kernel(struct CollectiveArgs* args) { template __device__ void ncclReduceTreeLL128Kernel(struct CollectiveArgs* args) { } + +template +__device__ void ncclReduceCollNetLL128Kernel(struct CollectiveArgs* args) { } -- cgit v1.2.3