From 0530ad113e4f17436926e60d2a17cdf5a9c55296 Mon Sep 17 00:00:00 2001 From: liulixinkerry Date: Sat, 23 Nov 2024 14:40:55 +0800 Subject: [PATCH] remove warning in torch2.5+cu124 --- cpp_to_py/dct_cuda/dct_cuda.cpp | 16 ++++---- cpp_to_py/dct_cuda/dct_cuda_kernel.cu | 38 ++++++++++--------- .../density_map_cuda_kernel.cu | 8 ++-- .../density_map_cuda_naive_kernel.cu | 6 +-- cpp_to_py/hpwl_cuda/hpwl_cuda_kernel.cu | 6 +-- .../wa_wirelength_hpwl_cuda_kernel.cu | 8 ++-- 6 files changed, 43 insertions(+), 39 deletions(-) diff --git a/cpp_to_py/dct_cuda/dct_cuda.cpp b/cpp_to_py/dct_cuda/dct_cuda.cpp index 558c28f..89ece7a 100644 --- a/cpp_to_py/dct_cuda/dct_cuda.cpp +++ b/cpp_to_py/dct_cuda/dct_cuda.cpp @@ -1,23 +1,23 @@ #include -void dct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf); -void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf); -void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf); -void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf); +void dct2_fft2_forward_cuda(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf); +void idct2_fft2_forward_cuda(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf); +void idct_idxst_forward_cuda(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf); +void idxst_idct_forward_cuda(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf); -void dct2_fft2_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { +void dct2_fft2_forward(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) { dct2_fft2_forward_cuda(x, expkM, expkN, out, buf); } -void idct2_fft2_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { +void idct2_fft2_forward(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) { idct2_fft2_forward_cuda(x, expkM, expkN, out, buf); } -void idct_idxst_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { +void idct_idxst_forward(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) { idct_idxst_forward_cuda(x, expkM, expkN, out, buf); } -void idxst_idct_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { +void idxst_idct_forward(torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) { idxst_idct_forward_cuda(x, expkM, expkN, out, buf); } diff --git a/cpp_to_py/dct_cuda/dct_cuda_kernel.cu b/cpp_to_py/dct_cuda/dct_cuda_kernel.cu index 46ed417..901aa04 100644 --- a/cpp_to_py/dct_cuda/dct_cuda_kernel.cu +++ b/cpp_to_py/dct_cuda/dct_cuda_kernel.cu @@ -9,7 +9,7 @@ * except tiny modifications on preprocessing and postprocessing */ -#include +#include #include #include #include @@ -120,7 +120,7 @@ template void dct2dPreprocessCudaLauncher(const T *x, T *y, const int M, const int N) { dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1); dim3 blockSize(TPB, TPB, 1); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); dct2dPreprocess<<>>(x, y, M, N, N / 2); } @@ -208,7 +208,7 @@ void dct2dPostprocessCudaLauncher( const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) { dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1); dim3 blockSize(TPB, TPB, 1); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); dct2dPostprocess><<>>((ComplexType *)x, y, M, @@ -333,7 +333,7 @@ void idct2_fft2PreprocessCudaLauncher( const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) { dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1); dim3 blockSize(TPB, TPB, 1); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); idct2_fft2Preprocess><<>>( x, (ComplexType *)y, M, N, M / 2, N / 2, (ComplexType *)expkM, (ComplexType *)expkN); } @@ -369,7 +369,7 @@ template void idct2_fft2PostprocessCudaLauncher(const T *x, T *y, const int M, const int N) { dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1); dim3 blockSize(TPB, TPB, 1); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); idct2_fft2Postprocess<<>>(x, y, M, N, N / 2, M * N); } @@ -483,7 +483,7 @@ void idct_idxstPreprocessCudaLauncher( const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) { dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1); dim3 blockSize(TPB, TPB, 1); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); idct_idxstPreprocess><<>>( x, (ComplexType *)y, M, N, M / 2, N / 2, (ComplexType *)expkM, (ComplexType *)expkN); } @@ -527,7 +527,7 @@ template void idct_idxstPostprocessCudaLauncher(const T *x, T *y, const int M, const int N) { dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1); dim3 blockSize(TPB, TPB, 1); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); idct_idxstPostprocess<<>>(x, y, M, N, N / 2, M * N); } @@ -645,7 +645,7 @@ void idxst_idctPreprocessCudaLauncher( const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) { dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1); dim3 blockSize(TPB, TPB, 1); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); idxst_idctPreprocess><<>>( x, (ComplexType *)y, M, N, M / 2, N / 2, (ComplexType *)expkM, (ComplexType *)expkN); } @@ -689,7 +689,7 @@ template void idxst_idctPostprocessCudaLauncher(const T *x, T *y, const int M, const int N) { dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1); dim3 blockSize(TPB, TPB, 1); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); idxst_idctPostprocess<<>>(x, y, M, N, N / 2, M * N); } @@ -701,7 +701,8 @@ void idxst_idctPostprocessCudaLauncher(const T *x, T *y, const int M, const int CHECK_CUDA(x); \ CHECK_CONTIGUOUS(x) -void dct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { +void dct2_fft2_forward_cuda( + torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) { cudaSetDevice(x.get_device()); CHECK_INPUT(x); CHECK_INPUT(expkM); @@ -714,13 +715,14 @@ void dct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at dct2dPreprocessCudaLauncher(x.data_ptr(), out.data_ptr(), M, N); - buf = at::view_as_real(at::fft_rfft2(out, c10::nullopt, {-2, -1}, "backward")).contiguous(); + buf = torch::view_as_real(torch::fft_rfft2(out, c10::nullopt, {-2, -1}, "backward")).contiguous(); dct2dPostprocessCudaLauncher( buf.data_ptr(), out.data_ptr(), M, N, expkM.data_ptr(), expkN.data_ptr()); } -void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { +void idct2_fft2_forward_cuda( + torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) { cudaSetDevice(x.get_device()); CHECK_INPUT(x); CHECK_INPUT(expkM); @@ -734,12 +736,13 @@ void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a idct2_fft2PreprocessCudaLauncher( x.data_ptr(), buf.data_ptr(), M, N, expkM.data_ptr(), expkN.data_ptr()); - auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous(); + auto y = torch::fft_irfft2(torch::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous(); idct2_fft2PostprocessCudaLauncher(y.data_ptr(), out.data_ptr(), M, N); } -void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { +void idct_idxst_forward_cuda( + torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) { cudaSetDevice(x.get_device()); CHECK_INPUT(x); CHECK_INPUT(expkM); @@ -753,12 +756,13 @@ void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a idct_idxstPreprocessCudaLauncher( x.data_ptr(), buf.data_ptr(), M, N, expkM.data_ptr(), expkN.data_ptr()); - auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous(); + auto y = torch::fft_irfft2(torch::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous(); idct_idxstPostprocessCudaLauncher(y.data_ptr(), out.data_ptr(), M, N); } -void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) { +void idxst_idct_forward_cuda( + torch::Tensor x, torch::Tensor expkM, torch::Tensor expkN, torch::Tensor out, torch::Tensor buf) { cudaSetDevice(x.get_device()); CHECK_INPUT(x); CHECK_INPUT(expkM); @@ -772,7 +776,7 @@ void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a idxst_idctPreprocessCudaLauncher( x.data_ptr(), buf.data_ptr(), M, N, expkM.data_ptr(), expkN.data_ptr()); - auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous(); + auto y = torch::fft_irfft2(torch::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous(); idxst_idctPostprocessCudaLauncher(y.data_ptr(), out.data_ptr(), M, N); } \ No newline at end of file diff --git a/cpp_to_py/density_map_cuda/density_map_cuda_kernel.cu b/cpp_to_py/density_map_cuda/density_map_cuda_kernel.cu index 6a09f86..d079968 100644 --- a/cpp_to_py/density_map_cuda/density_map_cuda_kernel.cu +++ b/cpp_to_py/density_map_cuda/density_map_cuda_kernel.cu @@ -1,4 +1,4 @@ -#include +#include #include #include #include @@ -252,7 +252,7 @@ torch::Tensor density_map_cuda_normalize_node(torch::Tensor node_pos, int num_bin_y, int num_nodes) { cudaSetDevice(node_pos.get_device()); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); const int threads = 128; const int blocks = (num_nodes + threads - 1) / threads; @@ -279,7 +279,7 @@ torch::Tensor density_map_cuda_forward(torch::Tensor normalize_node_info, int num_nodes, bool deterministic) { cudaSetDevice(normalize_node_info.get_device()); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); int thread_count = 64; dim3 blockSize(2, 2, thread_count); @@ -362,7 +362,7 @@ torch::Tensor density_map_cuda_backward(torch::Tensor normalize_node_info, int num_nodes, bool deterministic) { cudaSetDevice(normalize_node_info.get_device()); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); if (deterministic) { int threads = 64; diff --git a/cpp_to_py/density_map_cuda/density_map_cuda_naive_kernel.cu b/cpp_to_py/density_map_cuda/density_map_cuda_naive_kernel.cu index b60c4bf..49713a1 100644 --- a/cpp_to_py/density_map_cuda/density_map_cuda_naive_kernel.cu +++ b/cpp_to_py/density_map_cuda/density_map_cuda_naive_kernel.cu @@ -1,4 +1,4 @@ -#include +#include #include #include #include @@ -226,7 +226,7 @@ torch::Tensor density_map_cuda_forward_naive(torch::Tensor node_pos, bool clamp_node, bool deterministic) { cudaSetDevice(node_pos.get_device()); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); const int threads = 64; const int blocks = (num_nodes + threads - 1) / threads; @@ -298,7 +298,7 @@ torch::Tensor density_map_cuda_backward(torch::Tensor node_pos, bool clamp_node, bool deterministic) { cudaSetDevice(node_pos.get_device()); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); const int threads = 64; const int blocks = (num_nodes + threads - 1) / threads; diff --git a/cpp_to_py/hpwl_cuda/hpwl_cuda_kernel.cu b/cpp_to_py/hpwl_cuda/hpwl_cuda_kernel.cu index 813b71f..5ffad16 100644 --- a/cpp_to_py/hpwl_cuda/hpwl_cuda_kernel.cu +++ b/cpp_to_py/hpwl_cuda/hpwl_cuda_kernel.cu @@ -1,4 +1,4 @@ -#include +#include #include #include #include @@ -51,7 +51,7 @@ __global__ void node_pos_to_pin_pos_cuda_kernel( torch::Tensor hpwl_cuda(torch::Tensor pos, torch::Tensor hyperedge_list, torch::Tensor hyperedge_list_end) { cudaSetDevice(pos.get_device()); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); const auto num_nets = hyperedge_list_end.size(0); const int num_channels = 2; @@ -74,7 +74,7 @@ torch::Tensor node_pos_to_pin_pos_cuda(torch::Tensor node_pos, torch::Tensor pin_id2node_id, torch::Tensor pin_rel_cpos) { cudaSetDevice(node_pos.get_device()); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); // pin_pos == node_pos + pin_rel_cpos const auto num_pins = pin_id2node_id.size(0); diff --git a/cpp_to_py/wa_wirelength_hpwl_cuda/wa_wirelength_hpwl_cuda_kernel.cu b/cpp_to_py/wa_wirelength_hpwl_cuda/wa_wirelength_hpwl_cuda_kernel.cu index 2a605ce..bd9949d 100644 --- a/cpp_to_py/wa_wirelength_hpwl_cuda/wa_wirelength_hpwl_cuda_kernel.cu +++ b/cpp_to_py/wa_wirelength_hpwl_cuda/wa_wirelength_hpwl_cuda_kernel.cu @@ -1,4 +1,4 @@ -#include +#include #include #include #include @@ -153,7 +153,7 @@ void calc_node_grad_cuda(torch::Tensor node_grad, int num_nodes, bool deterministic) { if (deterministic) { - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); const int threads = 128; const int blocks = (num_nodes * 2 + threads - 1) / threads; calc_node_grad_deterministic_cuda_kernel<<>>( @@ -177,7 +177,7 @@ torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos, torch::Tensor net_mask, torch::Tensor hpwl_scale) { cudaSetDevice(node_pos.get_device()); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); const auto num_nodes = node_pos.size(0); const auto num_pins = pin_id2node_id.size(0); @@ -224,7 +224,7 @@ std::vector wa_wirelength_masked_scale_hpwl_cuda(torch::Tensor no float gamma, bool deterministic) { cudaSetDevice(node_pos.get_device()); - auto stream = at::cuda::getCurrentCUDAStream(); + auto stream = c10::cuda::getCurrentCUDAStream(); const auto num_nodes = node_pos.size(0); const auto num_pins = pin_id2node_id.size(0);