/usr/local/lib64/python3.6/site-packages/torch/include/torch/csrc/cuda
Edit: /usr/local/lib64/python3.6/site-packages/torch/include/torch/csrc/cuda/comm.h (1525B)
#pragma once
#include
#include
#include
#include
#include
#include
#include
namespace torch { namespace cuda {
using tensor_list2d = std::vector>;
TORCH_CUDA_CU_API std::vector& broadcast_out(
const at::Tensor& tensor,
std::vector& out_tensors);
TORCH_CUDA_CU_API std::vector broadcast(
const at::Tensor& tensor,
at::IntArrayRef devices);
TORCH_CUDA_CU_API tensor_list2d broadcast_coalesced(
at::TensorList tensors,
at::IntArrayRef devices,
size_t buffer_size);
TORCH_CUDA_CU_API std::vector& scatter_out(
const at::Tensor& tensor,
std::vector& out_tensors,
int64_t dim = 0,
const c10::optional>>&
streams = c10::nullopt);
TORCH_CUDA_CU_API std::vector scatter(
const at::Tensor& tensor,
at::IntArrayRef devices,
const c10::optional>& chunk_sizes = c10::nullopt,
int64_t dim = 0,
const c10::optional>>&
streams = c10::nullopt);
TORCH_CUDA_CU_API at::Tensor& gather_out(
at::TensorList tensors,
at::Tensor& out_tensor,
int64_t dim);
TORCH_CUDA_CU_API at::Tensor gather(
at::TensorList tensors,
int64_t dim,
c10::optional destination_index);
}}