remove warning in torch2.5+cu124

This commit is contained in:
liulixinkerry 2024-11-23 14:40:55 +08:00
parent 500ed28a7f
commit 0530ad113e
6 changed files with 43 additions and 39 deletions

View File

@ -1,23 +1,23 @@
#include <torch/extension.h> #include <torch/extension.h>
void dct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf); void dct2_fft2_forward_cuda(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf);
void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf); void idct2_fft2_forward_cuda(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf);
void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf); void idct_idxst_forward_cuda(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf);
void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf); void idxst_idct_forward_cuda(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf);
void dct2_fft2_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { void dct2_fft2_forward(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
dct2_fft2_forward_cuda(x, expkM, expkN, out, buf); dct2_fft2_forward_cuda(x, expkM, expkN, out, buf);
} }
void idct2_fft2_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { void idct2_fft2_forward(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
idct2_fft2_forward_cuda(x, expkM, expkN, out, buf); idct2_fft2_forward_cuda(x, expkM, expkN, out, buf);
} }
void idct_idxst_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { void idct_idxst_forward(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
idct_idxst_forward_cuda(x, expkM, expkN, out, buf); idct_idxst_forward_cuda(x, expkM, expkN, out, buf);
} }
void idxst_idct_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { void idxst_idct_forward(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
idxst_idct_forward_cuda(x, expkM, expkN, out, buf); idxst_idct_forward_cuda(x, expkM, expkN, out, buf);
} }

View File

@ -9,7 +9,7 @@
* except tiny modifications on preprocessing and postprocessing * except tiny modifications on preprocessing and postprocessing
*/ */
#include <ATen/cuda/CUDAContext.h> #include <c10/cuda/CUDAStream.h>
#include <float.h> #include <float.h>
#include <math.h> #include <math.h>
#include <torch/extension.h> #include <torch/extension.h>
@ -120,7 +120,7 @@ template <typename T>
void dct2dPreprocessCudaLauncher(const T *x, T *y, const int M, const int N) { void dct2dPreprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1); dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1); dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
dct2dPreprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2); dct2dPreprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2);
} }
@ -208,7 +208,7 @@ void dct2dPostprocessCudaLauncher(
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) { const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1); dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1); dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
dct2dPostprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>((ComplexType<T> *)x, dct2dPostprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>((ComplexType<T> *)x,
y, y,
M, M,
@ -333,7 +333,7 @@ void idct2_fft2PreprocessCudaLauncher(
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) { const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1); dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1); dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
idct2_fft2Preprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>( idct2_fft2Preprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>(
x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN); x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN);
} }
@ -369,7 +369,7 @@ template <typename T>
void idct2_fft2PostprocessCudaLauncher(const T *x, T *y, const int M, const int N) { void idct2_fft2PostprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1); dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1); dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
idct2_fft2Postprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N); idct2_fft2Postprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N);
} }
@ -483,7 +483,7 @@ void idct_idxstPreprocessCudaLauncher(
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) { const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1); dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1); dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
idct_idxstPreprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>( idct_idxstPreprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>(
x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN); x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN);
} }
@ -527,7 +527,7 @@ template <typename T>
void idct_idxstPostprocessCudaLauncher(const T *x, T *y, const int M, const int N) { void idct_idxstPostprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1); dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1); dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
idct_idxstPostprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N); idct_idxstPostprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N);
} }
@ -645,7 +645,7 @@ void idxst_idctPreprocessCudaLauncher(
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) { const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1); dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1); dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
idxst_idctPreprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>( idxst_idctPreprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>(
x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN); x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN);
} }
@ -689,7 +689,7 @@ template <typename T>
void idxst_idctPostprocessCudaLauncher(const T *x, T *y, const int M, const int N) { void idxst_idctPostprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1); dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1); dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
idxst_idctPostprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N); idxst_idctPostprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N);
} }
@ -701,7 +701,8 @@ void idxst_idctPostprocessCudaLauncher(const T *x, T *y, const int M, const int
CHECK_CUDA(x); \ CHECK_CUDA(x); \
CHECK_CONTIGUOUS(x) CHECK_CONTIGUOUS(x)
void dct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { void dct2_fft2_forward_cuda(
torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
cudaSetDevice(x.get_device()); cudaSetDevice(x.get_device());
CHECK_INPUT(x); CHECK_INPUT(x);
CHECK_INPUT(expkM); CHECK_INPUT(expkM);
@ -714,13 +715,14 @@ void dct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at
dct2dPreprocessCudaLauncher<float>(x.data_ptr<float>(), out.data_ptr<float>(), M, N); dct2dPreprocessCudaLauncher<float>(x.data_ptr<float>(), out.data_ptr<float>(), M, N);
buf = at::view_as_real(at::fft_rfft2(out, c10::nullopt, {-2, -1}, "backward")).contiguous(); buf = torch::view_as_real(torch::fft_rfft2(out, c10::nullopt, {-2, -1}, "backward")).contiguous();
dct2dPostprocessCudaLauncher<float>( dct2dPostprocessCudaLauncher<float>(
buf.data_ptr<float>(), out.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>()); buf.data_ptr<float>(), out.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
} }
void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { void idct2_fft2_forward_cuda(
torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
cudaSetDevice(x.get_device()); cudaSetDevice(x.get_device());
CHECK_INPUT(x); CHECK_INPUT(x);
CHECK_INPUT(expkM); CHECK_INPUT(expkM);
@ -734,12 +736,13 @@ void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a
idct2_fft2PreprocessCudaLauncher<float>( idct2_fft2PreprocessCudaLauncher<float>(
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>()); x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous(); auto y = torch::fft_irfft2(torch::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
idct2_fft2PostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N); idct2_fft2PostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
} }
void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { void idct_idxst_forward_cuda(
torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
cudaSetDevice(x.get_device()); cudaSetDevice(x.get_device());
CHECK_INPUT(x); CHECK_INPUT(x);
CHECK_INPUT(expkM); CHECK_INPUT(expkM);
@ -753,12 +756,13 @@ void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a
idct_idxstPreprocessCudaLauncher<float>( idct_idxstPreprocessCudaLauncher<float>(
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>()); x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous(); auto y = torch::fft_irfft2(torch::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
idct_idxstPostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N); idct_idxstPostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
} }
void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { void idxst_idct_forward_cuda(
torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) {
cudaSetDevice(x.get_device()); cudaSetDevice(x.get_device());
CHECK_INPUT(x); CHECK_INPUT(x);
CHECK_INPUT(expkM); CHECK_INPUT(expkM);
@ -772,7 +776,7 @@ void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a
idxst_idctPreprocessCudaLauncher<float>( idxst_idctPreprocessCudaLauncher<float>(
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>()); x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous(); auto y = torch::fft_irfft2(torch::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
idxst_idctPostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N); idxst_idctPostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
} }

View File

@ -1,4 +1,4 @@
#include <ATen/cuda/CUDAContext.h> #include <c10/cuda/CUDAStream.h>
#include <cuda.h> #include <cuda.h>
#include <cuda_runtime.h> #include <cuda_runtime.h>
#include <torch/extension.h> #include <torch/extension.h>
@ -252,7 +252,7 @@ torch::Tensor density_map_cuda_normalize_node(torch::Tensor node_pos,
int num_bin_y, int num_bin_y,
int num_nodes) { int num_nodes) {
cudaSetDevice(node_pos.get_device()); cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
const int threads = 128; const int threads = 128;
const int blocks = (num_nodes + threads - 1) / threads; const int blocks = (num_nodes + threads - 1) / threads;
@ -279,7 +279,7 @@ torch::Tensor density_map_cuda_forward(torch::Tensor normalize_node_info,
int num_nodes, int num_nodes,
bool deterministic) { bool deterministic) {
cudaSetDevice(normalize_node_info.get_device()); cudaSetDevice(normalize_node_info.get_device());
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
int thread_count = 64; int thread_count = 64;
dim3 blockSize(2, 2, thread_count); dim3 blockSize(2, 2, thread_count);
@ -362,7 +362,7 @@ torch::Tensor density_map_cuda_backward(torch::Tensor normalize_node_info,
int num_nodes, int num_nodes,
bool deterministic) { bool deterministic) {
cudaSetDevice(normalize_node_info.get_device()); cudaSetDevice(normalize_node_info.get_device());
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
if (deterministic) { if (deterministic) {
int threads = 64; int threads = 64;

View File

@ -1,4 +1,4 @@
#include <ATen/cuda/CUDAContext.h> #include <c10/cuda/CUDAStream.h>
#include <cuda.h> #include <cuda.h>
#include <cuda_runtime.h> #include <cuda_runtime.h>
#include <torch/extension.h> #include <torch/extension.h>
@ -226,7 +226,7 @@ torch::Tensor density_map_cuda_forward_naive(torch::Tensor node_pos,
bool clamp_node, bool clamp_node,
bool deterministic) { bool deterministic) {
cudaSetDevice(node_pos.get_device()); cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
const int threads = 64; const int threads = 64;
const int blocks = (num_nodes + threads - 1) / threads; const int blocks = (num_nodes + threads - 1) / threads;
@ -298,7 +298,7 @@ torch::Tensor density_map_cuda_backward(torch::Tensor node_pos,
bool clamp_node, bool clamp_node,
bool deterministic) { bool deterministic) {
cudaSetDevice(node_pos.get_device()); cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
const int threads = 64; const int threads = 64;
const int blocks = (num_nodes + threads - 1) / threads; const int blocks = (num_nodes + threads - 1) / threads;

View File

@ -1,4 +1,4 @@
#include <ATen/cuda/CUDAContext.h> #include <c10/cuda/CUDAStream.h>
#include <cuda.h> #include <cuda.h>
#include <cuda_runtime.h> #include <cuda_runtime.h>
#include <torch/extension.h> #include <torch/extension.h>
@ -51,7 +51,7 @@ __global__ void node_pos_to_pin_pos_cuda_kernel(
torch::Tensor hpwl_cuda(torch::Tensor pos, torch::Tensor hyperedge_list, torch::Tensor hyperedge_list_end) { torch::Tensor hpwl_cuda(torch::Tensor pos, torch::Tensor hyperedge_list, torch::Tensor hyperedge_list_end) {
cudaSetDevice(pos.get_device()); cudaSetDevice(pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
const auto num_nets = hyperedge_list_end.size(0); const auto num_nets = hyperedge_list_end.size(0);
const int num_channels = 2; const int num_channels = 2;
@ -74,7 +74,7 @@ torch::Tensor node_pos_to_pin_pos_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id, torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos) { torch::Tensor pin_rel_cpos) {
cudaSetDevice(node_pos.get_device()); cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
// pin_pos == node_pos + pin_rel_cpos // pin_pos == node_pos + pin_rel_cpos
const auto num_pins = pin_id2node_id.size(0); const auto num_pins = pin_id2node_id.size(0);

View File

@ -1,4 +1,4 @@
#include <ATen/cuda/CUDAContext.h> #include <c10/cuda/CUDAStream.h>
#include <cuda.h> #include <cuda.h>
#include <cuda_runtime.h> #include <cuda_runtime.h>
#include <torch/extension.h> #include <torch/extension.h>
@ -153,7 +153,7 @@ void calc_node_grad_cuda(torch::Tensor node_grad,
int num_nodes, int num_nodes,
bool deterministic) { bool deterministic) {
if (deterministic) { if (deterministic) {
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
const int threads = 128; const int threads = 128;
const int blocks = (num_nodes * 2 + threads - 1) / threads; const int blocks = (num_nodes * 2 + threads - 1) / threads;
calc_node_grad_deterministic_cuda_kernel<<<blocks, threads, 0, stream>>>( calc_node_grad_deterministic_cuda_kernel<<<blocks, threads, 0, stream>>>(
@ -177,7 +177,7 @@ torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos,
torch::Tensor net_mask, torch::Tensor net_mask,
torch::Tensor hpwl_scale) { torch::Tensor hpwl_scale) {
cudaSetDevice(node_pos.get_device()); cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
const auto num_nodes = node_pos.size(0); const auto num_nodes = node_pos.size(0);
const auto num_pins = pin_id2node_id.size(0); const auto num_pins = pin_id2node_id.size(0);
@ -224,7 +224,7 @@ std::vector<torch::Tensor> wa_wirelength_masked_scale_hpwl_cuda(torch::Tensor no
float gamma, float gamma,
bool deterministic) { bool deterministic) {
cudaSetDevice(node_pos.get_device()); cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream(); auto stream = c10::cuda::getCurrentCUDAStream();
const auto num_nodes = node_pos.size(0); const auto num_nodes = node_pos.size(0);
const auto num_pins = pin_id2node_id.size(0); const auto num_pins = pin_id2node_id.size(0);