remove warning in torch2.5+cu124
This commit is contained in:
parent
500ed28a7f
commit
0530ad113e
@ -1,23 +1,23 @@
|
|||||||
#include <torch/extension.h>
|
#include <torch/extension.h>
|
||||||
|
|
||||||
void dct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf);
|
void dct2_fft2_forward_cuda(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf);
|
||||||
void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf);
|
void idct2_fft2_forward_cuda(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf);
|
||||||
void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf);
|
void idct_idxst_forward_cuda(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf);
|
||||||
void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf);
|
void idxst_idct_forward_cuda(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf);
|
||||||
|
|
||||||
void dct2_fft2_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
|
void dct2_fft2_forward(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
|
||||||
dct2_fft2_forward_cuda(x, expkM, expkN, out, buf);
|
dct2_fft2_forward_cuda(x, expkM, expkN, out, buf);
|
||||||
}
|
}
|
||||||
|
|
||||||
void idct2_fft2_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
|
void idct2_fft2_forward(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
|
||||||
idct2_fft2_forward_cuda(x, expkM, expkN, out, buf);
|
idct2_fft2_forward_cuda(x, expkM, expkN, out, buf);
|
||||||
}
|
}
|
||||||
|
|
||||||
void idct_idxst_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
|
void idct_idxst_forward(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
|
||||||
idct_idxst_forward_cuda(x, expkM, expkN, out, buf);
|
idct_idxst_forward_cuda(x, expkM, expkN, out, buf);
|
||||||
}
|
}
|
||||||
|
|
||||||
void idxst_idct_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
|
void idxst_idct_forward(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
|
||||||
idxst_idct_forward_cuda(x, expkM, expkN, out, buf);
|
idxst_idct_forward_cuda(x, expkM, expkN, out, buf);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@ -9,7 +9,7 @@
|
|||||||
* except tiny modifications on preprocessing and postprocessing
|
* except tiny modifications on preprocessing and postprocessing
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#include <ATen/cuda/CUDAContext.h>
|
#include <c10/cuda/CUDAStream.h>
|
||||||
#include <float.h>
|
#include <float.h>
|
||||||
#include <math.h>
|
#include <math.h>
|
||||||
#include <torch/extension.h>
|
#include <torch/extension.h>
|
||||||
@ -120,7 +120,7 @@ template <typename T>
|
|||||||
void dct2dPreprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
|
void dct2dPreprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
|
||||||
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
|
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
|
||||||
dim3 blockSize(TPB, TPB, 1);
|
dim3 blockSize(TPB, TPB, 1);
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
dct2dPreprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2);
|
dct2dPreprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2);
|
||||||
}
|
}
|
||||||
|
|
||||||
@ -208,7 +208,7 @@ void dct2dPostprocessCudaLauncher(
|
|||||||
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
|
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
|
||||||
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
|
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
|
||||||
dim3 blockSize(TPB, TPB, 1);
|
dim3 blockSize(TPB, TPB, 1);
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
dct2dPostprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>((ComplexType<T> *)x,
|
dct2dPostprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>((ComplexType<T> *)x,
|
||||||
y,
|
y,
|
||||||
M,
|
M,
|
||||||
@ -333,7 +333,7 @@ void idct2_fft2PreprocessCudaLauncher(
|
|||||||
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
|
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
|
||||||
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
|
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
|
||||||
dim3 blockSize(TPB, TPB, 1);
|
dim3 blockSize(TPB, TPB, 1);
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
idct2_fft2Preprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>(
|
idct2_fft2Preprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>(
|
||||||
x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN);
|
x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN);
|
||||||
}
|
}
|
||||||
@ -369,7 +369,7 @@ template <typename T>
|
|||||||
void idct2_fft2PostprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
|
void idct2_fft2PostprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
|
||||||
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
|
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
|
||||||
dim3 blockSize(TPB, TPB, 1);
|
dim3 blockSize(TPB, TPB, 1);
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
idct2_fft2Postprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N);
|
idct2_fft2Postprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N);
|
||||||
}
|
}
|
||||||
|
|
||||||
@ -483,7 +483,7 @@ void idct_idxstPreprocessCudaLauncher(
|
|||||||
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
|
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
|
||||||
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
|
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
|
||||||
dim3 blockSize(TPB, TPB, 1);
|
dim3 blockSize(TPB, TPB, 1);
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
idct_idxstPreprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>(
|
idct_idxstPreprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>(
|
||||||
x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN);
|
x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN);
|
||||||
}
|
}
|
||||||
@ -527,7 +527,7 @@ template <typename T>
|
|||||||
void idct_idxstPostprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
|
void idct_idxstPostprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
|
||||||
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
|
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
|
||||||
dim3 blockSize(TPB, TPB, 1);
|
dim3 blockSize(TPB, TPB, 1);
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
idct_idxstPostprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N);
|
idct_idxstPostprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N);
|
||||||
}
|
}
|
||||||
|
|
||||||
@ -645,7 +645,7 @@ void idxst_idctPreprocessCudaLauncher(
|
|||||||
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
|
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
|
||||||
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
|
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
|
||||||
dim3 blockSize(TPB, TPB, 1);
|
dim3 blockSize(TPB, TPB, 1);
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
idxst_idctPreprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>(
|
idxst_idctPreprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>(
|
||||||
x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN);
|
x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN);
|
||||||
}
|
}
|
||||||
@ -689,7 +689,7 @@ template <typename T>
|
|||||||
void idxst_idctPostprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
|
void idxst_idctPostprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
|
||||||
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
|
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
|
||||||
dim3 blockSize(TPB, TPB, 1);
|
dim3 blockSize(TPB, TPB, 1);
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
idxst_idctPostprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N);
|
idxst_idctPostprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N);
|
||||||
}
|
}
|
||||||
|
|
||||||
@ -701,7 +701,8 @@ void idxst_idctPostprocessCudaLauncher(const T *x, T *y, const int M, const int
|
|||||||
CHECK_CUDA(x); \
|
CHECK_CUDA(x); \
|
||||||
CHECK_CONTIGUOUS(x)
|
CHECK_CONTIGUOUS(x)
|
||||||
|
|
||||||
void dct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
|
void dct2_fft2_forward_cuda(
|
||||||
|
torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
|
||||||
cudaSetDevice(x.get_device());
|
cudaSetDevice(x.get_device());
|
||||||
CHECK_INPUT(x);
|
CHECK_INPUT(x);
|
||||||
CHECK_INPUT(expkM);
|
CHECK_INPUT(expkM);
|
||||||
@ -714,13 +715,14 @@ void dct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at
|
|||||||
|
|
||||||
dct2dPreprocessCudaLauncher<float>(x.data_ptr<float>(), out.data_ptr<float>(), M, N);
|
dct2dPreprocessCudaLauncher<float>(x.data_ptr<float>(), out.data_ptr<float>(), M, N);
|
||||||
|
|
||||||
buf = at::view_as_real(at::fft_rfft2(out, c10::nullopt, {-2, -1}, "backward")).contiguous();
|
buf = torch::view_as_real(torch::fft_rfft2(out, c10::nullopt, {-2, -1}, "backward")).contiguous();
|
||||||
|
|
||||||
dct2dPostprocessCudaLauncher<float>(
|
dct2dPostprocessCudaLauncher<float>(
|
||||||
buf.data_ptr<float>(), out.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
|
buf.data_ptr<float>(), out.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
|
||||||
}
|
}
|
||||||
|
|
||||||
void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
|
void idct2_fft2_forward_cuda(
|
||||||
|
torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
|
||||||
cudaSetDevice(x.get_device());
|
cudaSetDevice(x.get_device());
|
||||||
CHECK_INPUT(x);
|
CHECK_INPUT(x);
|
||||||
CHECK_INPUT(expkM);
|
CHECK_INPUT(expkM);
|
||||||
@ -734,12 +736,13 @@ void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a
|
|||||||
idct2_fft2PreprocessCudaLauncher<float>(
|
idct2_fft2PreprocessCudaLauncher<float>(
|
||||||
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
|
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
|
||||||
|
|
||||||
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
|
auto y = torch::fft_irfft2(torch::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
|
||||||
|
|
||||||
idct2_fft2PostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
|
idct2_fft2PostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
|
||||||
}
|
}
|
||||||
|
|
||||||
void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
|
void idct_idxst_forward_cuda(
|
||||||
|
torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
|
||||||
cudaSetDevice(x.get_device());
|
cudaSetDevice(x.get_device());
|
||||||
CHECK_INPUT(x);
|
CHECK_INPUT(x);
|
||||||
CHECK_INPUT(expkM);
|
CHECK_INPUT(expkM);
|
||||||
@ -753,12 +756,13 @@ void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a
|
|||||||
idct_idxstPreprocessCudaLauncher<float>(
|
idct_idxstPreprocessCudaLauncher<float>(
|
||||||
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
|
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
|
||||||
|
|
||||||
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
|
auto y = torch::fft_irfft2(torch::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
|
||||||
|
|
||||||
idct_idxstPostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
|
idct_idxstPostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
|
||||||
}
|
}
|
||||||
|
|
||||||
void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
|
void idxst_idct_forward_cuda(
|
||||||
|
torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
|
||||||
cudaSetDevice(x.get_device());
|
cudaSetDevice(x.get_device());
|
||||||
CHECK_INPUT(x);
|
CHECK_INPUT(x);
|
||||||
CHECK_INPUT(expkM);
|
CHECK_INPUT(expkM);
|
||||||
@ -772,7 +776,7 @@ void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a
|
|||||||
idxst_idctPreprocessCudaLauncher<float>(
|
idxst_idctPreprocessCudaLauncher<float>(
|
||||||
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
|
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
|
||||||
|
|
||||||
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
|
auto y = torch::fft_irfft2(torch::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
|
||||||
|
|
||||||
idxst_idctPostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
|
idxst_idctPostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
|
||||||
}
|
}
|
||||||
@ -1,4 +1,4 @@
|
|||||||
#include <ATen/cuda/CUDAContext.h>
|
#include <c10/cuda/CUDAStream.h>
|
||||||
#include <cuda.h>
|
#include <cuda.h>
|
||||||
#include <cuda_runtime.h>
|
#include <cuda_runtime.h>
|
||||||
#include <torch/extension.h>
|
#include <torch/extension.h>
|
||||||
@ -252,7 +252,7 @@ torch::Tensor density_map_cuda_normalize_node(torch::Tensor node_pos,
|
|||||||
int num_bin_y,
|
int num_bin_y,
|
||||||
int num_nodes) {
|
int num_nodes) {
|
||||||
cudaSetDevice(node_pos.get_device());
|
cudaSetDevice(node_pos.get_device());
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
|
|
||||||
const int threads = 128;
|
const int threads = 128;
|
||||||
const int blocks = (num_nodes + threads - 1) / threads;
|
const int blocks = (num_nodes + threads - 1) / threads;
|
||||||
@ -279,7 +279,7 @@ torch::Tensor density_map_cuda_forward(torch::Tensor normalize_node_info,
|
|||||||
int num_nodes,
|
int num_nodes,
|
||||||
bool deterministic) {
|
bool deterministic) {
|
||||||
cudaSetDevice(normalize_node_info.get_device());
|
cudaSetDevice(normalize_node_info.get_device());
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
|
|
||||||
int thread_count = 64;
|
int thread_count = 64;
|
||||||
dim3 blockSize(2, 2, thread_count);
|
dim3 blockSize(2, 2, thread_count);
|
||||||
@ -362,7 +362,7 @@ torch::Tensor density_map_cuda_backward(torch::Tensor normalize_node_info,
|
|||||||
int num_nodes,
|
int num_nodes,
|
||||||
bool deterministic) {
|
bool deterministic) {
|
||||||
cudaSetDevice(normalize_node_info.get_device());
|
cudaSetDevice(normalize_node_info.get_device());
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
|
|
||||||
if (deterministic) {
|
if (deterministic) {
|
||||||
int threads = 64;
|
int threads = 64;
|
||||||
|
|||||||
@ -1,4 +1,4 @@
|
|||||||
#include <ATen/cuda/CUDAContext.h>
|
#include <c10/cuda/CUDAStream.h>
|
||||||
#include <cuda.h>
|
#include <cuda.h>
|
||||||
#include <cuda_runtime.h>
|
#include <cuda_runtime.h>
|
||||||
#include <torch/extension.h>
|
#include <torch/extension.h>
|
||||||
@ -226,7 +226,7 @@ torch::Tensor density_map_cuda_forward_naive(torch::Tensor node_pos,
|
|||||||
bool clamp_node,
|
bool clamp_node,
|
||||||
bool deterministic) {
|
bool deterministic) {
|
||||||
cudaSetDevice(node_pos.get_device());
|
cudaSetDevice(node_pos.get_device());
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
|
|
||||||
const int threads = 64;
|
const int threads = 64;
|
||||||
const int blocks = (num_nodes + threads - 1) / threads;
|
const int blocks = (num_nodes + threads - 1) / threads;
|
||||||
@ -298,7 +298,7 @@ torch::Tensor density_map_cuda_backward(torch::Tensor node_pos,
|
|||||||
bool clamp_node,
|
bool clamp_node,
|
||||||
bool deterministic) {
|
bool deterministic) {
|
||||||
cudaSetDevice(node_pos.get_device());
|
cudaSetDevice(node_pos.get_device());
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
|
|
||||||
const int threads = 64;
|
const int threads = 64;
|
||||||
const int blocks = (num_nodes + threads - 1) / threads;
|
const int blocks = (num_nodes + threads - 1) / threads;
|
||||||
|
|||||||
@ -1,4 +1,4 @@
|
|||||||
#include <ATen/cuda/CUDAContext.h>
|
#include <c10/cuda/CUDAStream.h>
|
||||||
#include <cuda.h>
|
#include <cuda.h>
|
||||||
#include <cuda_runtime.h>
|
#include <cuda_runtime.h>
|
||||||
#include <torch/extension.h>
|
#include <torch/extension.h>
|
||||||
@ -51,7 +51,7 @@ __global__ void node_pos_to_pin_pos_cuda_kernel(
|
|||||||
|
|
||||||
torch::Tensor hpwl_cuda(torch::Tensor pos, torch::Tensor hyperedge_list, torch::Tensor hyperedge_list_end) {
|
torch::Tensor hpwl_cuda(torch::Tensor pos, torch::Tensor hyperedge_list, torch::Tensor hyperedge_list_end) {
|
||||||
cudaSetDevice(pos.get_device());
|
cudaSetDevice(pos.get_device());
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
|
|
||||||
const auto num_nets = hyperedge_list_end.size(0);
|
const auto num_nets = hyperedge_list_end.size(0);
|
||||||
const int num_channels = 2;
|
const int num_channels = 2;
|
||||||
@ -74,7 +74,7 @@ torch::Tensor node_pos_to_pin_pos_cuda(torch::Tensor node_pos,
|
|||||||
torch::Tensor pin_id2node_id,
|
torch::Tensor pin_id2node_id,
|
||||||
torch::Tensor pin_rel_cpos) {
|
torch::Tensor pin_rel_cpos) {
|
||||||
cudaSetDevice(node_pos.get_device());
|
cudaSetDevice(node_pos.get_device());
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
// pin_pos == node_pos + pin_rel_cpos
|
// pin_pos == node_pos + pin_rel_cpos
|
||||||
|
|
||||||
const auto num_pins = pin_id2node_id.size(0);
|
const auto num_pins = pin_id2node_id.size(0);
|
||||||
|
|||||||
@ -1,4 +1,4 @@
|
|||||||
#include <ATen/cuda/CUDAContext.h>
|
#include <c10/cuda/CUDAStream.h>
|
||||||
#include <cuda.h>
|
#include <cuda.h>
|
||||||
#include <cuda_runtime.h>
|
#include <cuda_runtime.h>
|
||||||
#include <torch/extension.h>
|
#include <torch/extension.h>
|
||||||
@ -153,7 +153,7 @@ void calc_node_grad_cuda(torch::Tensor node_grad,
|
|||||||
int num_nodes,
|
int num_nodes,
|
||||||
bool deterministic) {
|
bool deterministic) {
|
||||||
if (deterministic) {
|
if (deterministic) {
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
const int threads = 128;
|
const int threads = 128;
|
||||||
const int blocks = (num_nodes * 2 + threads - 1) / threads;
|
const int blocks = (num_nodes * 2 + threads - 1) / threads;
|
||||||
calc_node_grad_deterministic_cuda_kernel<<<blocks, threads, 0, stream>>>(
|
calc_node_grad_deterministic_cuda_kernel<<<blocks, threads, 0, stream>>>(
|
||||||
@ -177,7 +177,7 @@ torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos,
|
|||||||
torch::Tensor net_mask,
|
torch::Tensor net_mask,
|
||||||
torch::Tensor hpwl_scale) {
|
torch::Tensor hpwl_scale) {
|
||||||
cudaSetDevice(node_pos.get_device());
|
cudaSetDevice(node_pos.get_device());
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
|
|
||||||
const auto num_nodes = node_pos.size(0);
|
const auto num_nodes = node_pos.size(0);
|
||||||
const auto num_pins = pin_id2node_id.size(0);
|
const auto num_pins = pin_id2node_id.size(0);
|
||||||
@ -224,7 +224,7 @@ std::vector<torch::Tensor> wa_wirelength_masked_scale_hpwl_cuda(torch::Tensor no
|
|||||||
float gamma,
|
float gamma,
|
||||||
bool deterministic) {
|
bool deterministic) {
|
||||||
cudaSetDevice(node_pos.get_device());
|
cudaSetDevice(node_pos.get_device());
|
||||||
auto stream = at::cuda::getCurrentCUDAStream();
|
auto stream = c10::cuda::getCurrentCUDAStream();
|
||||||
|
|
||||||
const auto num_nodes = node_pos.size(0);
|
const auto num_nodes = node_pos.size(0);
|
||||||
const auto num_pins = pin_id2node_id.size(0);
|
const auto num_pins = pin_id2node_id.size(0);
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user