initial commit

This commit is contained in:
liulixinkerry 2022-10-20 22:46:17 +08:00
commit 386bfe5566
194 changed files with 978275 additions and 0 deletions

15
.gitignore vendored Normal file
View File

@ -0,0 +1,15 @@
*.png
*.pdf
.vscode/
__pycache__/
*.so
build/
*.log
*.npy
*.csv
*.pt
*.pkl
*.json
result*
data/cad
misc

3
.gitmodules vendored Normal file
View File

@ -0,0 +1,3 @@
[submodule "thirdparty/pybind11"]
path = thirdparty/pybind11
url = https://github.com/pybind/pybind11.git

156
CMakeLists.txt Normal file
View File

@ -0,0 +1,156 @@
cmake_minimum_required(VERSION 3.12)
project(xplace C CXX)
if(NOT CMAKE_BUILD_TYPE)
set(CMAKE_BUILD_TYPE Release)
endif()
message(STATUS "CMAKE_BUILD_TYPE: ${CMAKE_BUILD_TYPE}")
set(CMAKE_CXX_STANDARD 17)
set(CMAKE_CXX_STANDARD_REQUIRED ON)
message(STATUS PROJECT_SOURCE_DIR=${PROJECT_SOURCE_DIR})
set(PATH_THIRDPARTY_ROOT ${PROJECT_SOURCE_DIR}/thirdparty)
set(XPLACE_LIB_DIR ${PROJECT_SOURCE_DIR}/cpp_to_py/cpybin)
set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} "${CMAKE_CURRENT_SOURCE_DIR}/cmake")
set(CMAKE_INSTALL_RPATH ${XPLACE_LIB_DIR})
# set _GLIBCXX_USE_CXX11_ABI for torch
if(NOT CMAKE_CXX_ABI)
set(CMAKE_CXX_ABI 0 CACHE STRING
"Choose the value for _GLIBCXX_USE_CXX11_ABI, options are: 0|1."
FORCE)
endif(NOT CMAKE_CXX_ABI)
message(STATUS "CMAKE_CXX_ABI: _GLIBCXX_USE_CXX11_ABI=${CMAKE_CXX_ABI}")
add_definitions(-D_GLIBCXX_USE_CXX11_ABI=${CMAKE_CXX_ABI})
# PyBind11
add_subdirectory(${PATH_THIRDPARTY_ROOT}/pybind11)
message(STATUS "PYTHON_INCLUDE_DIRS: ${PYTHON_INCLUDE_DIRS}")
# Flute
add_subdirectory(${PATH_THIRDPARTY_ROOT}/flute)
message(STATUS "FLUTE_INCLUDE_DIR: ${FLUTE_INCLUDE_DIR}")
# Cairo
find_package(Cairo)
message(STATUS "CAIRO_INCLUDE_DIRS: ${CAIRO_INCLUDE_DIRS}")
message(STATUS "CAIRO_LIBRARIES: ${CAIRO_LIBRARIES}")
# Get Torch info
execute_process(COMMAND ${PYTHON_EXECUTABLE} -c
"import torch; print(torch.__path__[0]); print(int(torch.cuda.is_available())); print(torch.__version__);"
OUTPUT_VARIABLE TORCH_OUTPUT OUTPUT_STRIP_TRAILING_WHITESPACE)
string(REPLACE "\n" ";" TORCH_OUTPUT_LIST ${TORCH_OUTPUT})
list(GET TORCH_OUTPUT_LIST 0 TORCH_INSTALL_PREFIX)
list(GET TORCH_OUTPUT_LIST 1 TORCH_ENABLE_CUDA)
list(GET TORCH_OUTPUT_LIST 2 TORCH_VERSION)
string(REPLACE "." ";" TORCH_VERSION_LIST ${TORCH_VERSION})
list(GET TORCH_VERSION_LIST 0 TORCH_MAJOR_VERSION)
list(GET TORCH_VERSION_LIST 1 TORCH_MINOR_VERSION)
message(STATUS TORCH_INSTALL_PREFIX=${TORCH_INSTALL_PREFIX})
message(STATUS TORCH_VERSION=${TORCH_MAJOR_VERSION}.${TORCH_MINOR_VERSION})
# find CUDA
if (TORCH_ENABLE_CUDA)
find_package(CUDA 11.0)
if (NOT CUDA_FOUND)
set(TORCH_ENABLE_CUDA 0 CACHE BOOL "Whether enable CUDA" FORCE)
message(FATAL_ERROR "Xplace only supports CUDA mode, CMake will exit." )
endif(NOT CUDA_FOUND)
endif()
message(STATUS TORCH_ENABLE_CUDA=${TORCH_ENABLE_CUDA})
# set cuda arch and nvcc flags
if (CUDA_FOUND)
if (NOT CUDA_ARCH_LIST)
set(CUDA_ARCH_LIST 7.0 7.5 8.0 8.6)
endif(NOT CUDA_ARCH_LIST)
# for cuda_add_library
cuda_select_nvcc_arch_flags(CUDA_ARCH_FLAGS ${CUDA_ARCH_LIST})
message(STATUS "CUDA_ARCH_FLAGS: ${CUDA_ARCH_FLAGS}")
# set nvcc flags
set(CMAKE_CUDA17_EXTENSION_COMPILE_OPTION "-std=c++17")
list(APPEND CUDA_NVCC_FLAGS ${CUDA_ARCH_FLAGS} --compiler-options;-fPIC;-std=c++17)
list(APPEND CUDA_NVCC_FLAGS ${CUDA_ARCH_FLAGS} --extended-lambda)
list(APPEND TORCH_NVCC_FLAGS -D__CUDA_NO_HALF_OPERATORS__;-D__CUDA_NO_HALF_CONVERSIONS__;-D__CUDA_NO_BFLOAT16_CONVERSIONS__;-D__CUDA_NO_HALF2_OPERATORS__;--expt-relaxed-constexpr)
list(APPEND CUDA_NVCC_FLAGS ${TORCH_NVCC_FLAGS})
message(STATUS "CUDA_NVCC_FLAGS: ${CUDA_NVCC_FLAGS}")
endif(CUDA_FOUND)
# set torch lib
add_library(torch STATIC IMPORTED)
find_library(TORCH_PYTHON_LIBRARY torch_python PATHS "${TORCH_INSTALL_PREFIX}/lib" REQUIRED)
find_library(TORCH_LIBRARY torch PATHS "${TORCH_INSTALL_PREFIX}/lib" REQUIRED)
find_library(C10_LIBRARY c10 PATHS "${TORCH_INSTALL_PREFIX}/lib" REQUIRED)
find_library(C10_CUDA_LIBRARY c10_cuda PATHS "${TORCH_INSTALL_PREFIX}/lib")
find_library(TORCH_CPU_LIBRARY torch_cpu PATHS "${TORCH_INSTALL_PREFIX}/lib" REQUIRED)
find_library(TORCH_CUDA_LIBRARY torch_cuda PATHS "${TORCH_INSTALL_PREFIX}/lib")
set(LINK_LIBS ${C10_LIBRARY} ${TORCH_CPU_LIBRARY})
if (TORCH_ENABLE_CUDA)
set(LINK_LIBS ${LINK_LIBS}
${C10_CUDA_LIBRARY}
${TORCH_CUDA_LIBRARY})
endif()
# set torch include
if (EXISTS ${TORCH_INSTALL_PREFIX}/include)
set(TORCH_HEADER_PREFIX ${TORCH_INSTALL_PREFIX}/include)
endif()
set(TORCH_INCLUDE_DIRS
${PYTHON_INCLUDE_DIRS} ${TORCH_HEADER_PREFIX} ${TORCH_HEADER_PREFIX}/torch/csrc/api/include)
message(STATUS TORCH_INCLUDE_DIRS=${TORCH_INCLUDE_DIRS})
# set torch target
# adapt from https://github.com/limbo018/DREAMPlace/blob/master/cmake/TorchExtension.cmake
set_target_properties(torch PROPERTIES
IMPORTED_LOCATION "${TORCH_LIBRARY}"
INTERFACE_INCLUDE_DIRECTORIES "${TORCH_INCLUDE_DIRS}"
INTERFACE_LINK_LIBRARIES "${LINK_LIBS}"
INTERFACE_COMPILE_OPTIONS "-D_GLIBCXX_USE_CXX11_ABI=${CMAKE_CXX_ABI}"
)
function(add_pytorch_extension target_name)
set(multiValueArgs EXTRA_INCLUDE_DIRS EXTRA_LINK_LIBRARIES EXTRA_DEFINITIONS)
cmake_parse_arguments(ARG "" "" "${multiValueArgs}" ${ARGN})
if (TORCH_ENABLE_CUDA)
set(CUDA_SRCS "${ARG_UNPARSED_ARGUMENTS}")
list(FILTER CUDA_SRCS INCLUDE REGEX ".*cu$")
if (CUDA_SRCS)
cuda_add_library(${target_name}_cuda_tmp STATIC ${CUDA_SRCS})
target_include_directories(${target_name}_cuda_tmp PRIVATE ${ARG_EXTRA_INCLUDE_DIRS} ${TORCH_INCLUDE_DIRS})
target_link_libraries(${target_name}_cuda_tmp ${ARG_EXTRA_LINK_LIBRARIES} ${TORCH_LIBRARY})
target_compile_definitions(${target_name}_cuda_tmp PRIVATE
TORCH_EXTENSION_NAME=${target_name}
TORCH_MAJOR_VERSION=${TORCH_MAJOR_VERSION}
TORCH_MINOR_VERSION=${TORCH_MINOR_VERSION}
ENABLE_CUDA=${TORCH_ENABLE_CUDA}
${ARG_EXTRA_DEFINITIONS})
set_target_properties(${target_name}_cuda_tmp PROPERTIES
POSITION_INDEPENDENT_CODE ON
CXX_VISIBILITY_PRESET "hidden"
CUDA_VISIBILITY_PRESET "hidden"
)
endif()
endif()
list(FILTER ARG_UNPARSED_ARGUMENTS EXCLUDE REGEX ".*cu$")
pybind11_add_module(${target_name} MODULE ${ARG_UNPARSED_ARGUMENTS})
target_include_directories(${target_name} PRIVATE ${ARG_EXTRA_INCLUDE_DIRS} ${TORCH_INCLUDE_DIRS})
if (TORCH_ENABLE_CUDA AND CUDA_SRCS)
target_link_libraries(${target_name} PRIVATE ${target_name}_cuda_tmp ${ARG_EXTRA_LINK_LIBRARIES} torch ${TORCH_PYTHON_LIBRARY})
else()
target_link_libraries(${target_name} PRIVATE ${ARG_EXTRA_LINK_LIBRARIES} torch ${TORCH_PYTHON_LIBRARY})
endif()
target_compile_definitions(${target_name} PRIVATE
TORCH_EXTENSION_NAME=${target_name}
TORCH_MAJOR_VERSION=${TORCH_MAJOR_VERSION}
TORCH_MINOR_VERSION=${TORCH_MINOR_VERSION}
ENABLE_CUDA=${TORCH_ENABLE_CUDA}
${ARG_EXTRA_DEFINITIONS})
endfunction()
# add targets
add_subdirectory(cpp_to_py)

41
LICENSE Normal file
View File

@ -0,0 +1,41 @@
READ THIS LICENSE AGREEMENT CAREFULLY BEFORE USING THIS PRODUCT. BY USING THIS
PRODUCT YOU INDICATE YOUR ACCEPTANCE OF THE TERMS OF THE FOLLOWING AGREEMENT.
THESE TERMS APPLY TO YOU AND ANY SUBSEQUENT LICENSEE OF THIS PRODUCT.
License Agreement for Xplace
Copyright (c) 2022, The Chinese University of Hong Kong
All rights reserved.
CU-SD LICENSE (adapted from the original BSD license) Redistribution of the any
code, with or without modification, are permitted provided that the conditions
below are met.
1. Redistributions of source code must retain the above copyright notice, this
list of conditions and the following disclaimer.
2. Redistributions in binary form must reproduce the above copyright notice,
this list of conditions and the following disclaimer in the documentation
and/or other materials provided with the distribution.
3. Neither the name nor trademark of the copyright holder or the author may be
used to endorse or promote products derived from this software without
specific prior written permission.
4. Users are entirely responsible, to the exclusion of the author, for
compliance with (a) regulations set by owners or administrators of employed
equipment, (b) licensing terms of any other software, and (c) local,
national, and international regulations regarding use, including those
regarding import, export, and use of encryption software.
THIS FREE SOFTWARE IS PROVIDED BY THE AUTHOR "AS IS" AND ANY EXPRESS OR IMPLIED
WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF
MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT
SHALL THE AUTHOR OR ANY CONTRIBUTOR BE LIABLE FOR ANY DIRECT, INDIRECT,
INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
LIMITED TO, EFFECTS OF UNAUTHORIZED OR MALICIOUS NETWORK ACCESS; PROCUREMENT OF
SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
POSSIBILITY OF SUCH DAMAGE.

153
README.md Normal file
View File

@ -0,0 +1,153 @@
# Xplace
Xplace is a fast and extensible GPU accelerated global placement framework developed by the research team supervised by Prof. Evangeline F. Y. Young at The Chinese University of Hong Kong (CUHK). It achieves around 3x speedup per GP iteration compared to the state-of-the-art global placer DREAMPlace and shows high extensiblity.
As shown in the following figure, Xplace framework is built on top of PyTorch and consists of serveral independent modules. One can easily extend Xplace by applying new scheduling techniques, new gradient functions, new placement metrics and so on.
<div align="center">
<img src="assets/xplace_overview.png" width="300"/>
</div>
More details are in the following paper:
Lixin Liu, Bangqi Fu, Martin D. F. Wong, and Evangeline F. Y. Young. "[Xplace: an extremely fast and extensible global placement framework](https://doi.org/10.1145/3489517.3530485)". In Proceedings of the 59th ACM/IEEE Design Automation Conference (DAC '22). Association for Computing Machinery, New York, NY, USA, 1309–1314.
(For the Xplace-NN, please refer to branch [neural](https://github.com/cuhk-eda/Xplace/tree/neural))
## Requirements
- CMake >= 3.12
- GCC >= 7.5.0
- Boost >= 1.56.0
- CUDA >= 11.0
- Python >= 3.8
- PyTorch >= 1.10.1
- Cairo
## Setup
1. Clone the Xplace repository. We'll call the directory that you cloned Xplace as `$XPLACE_HOME`.
```console
git clone --recursive https://github.com/cuhk-eda/Xplace
```
2. Build the shared libraries used in Xplace.
```console
cd $XPLACE_HOME
mkdir build && cd build
cmake -DPYTHON_EXECUTABLE=$(which python) ..
make -j40 && make install
```
## Get started
- To run GP only flow for all the designs in ISPD2005 dataset:
```console
python main.py --dataset_root your_path --dataset ispd2005 --run_all True --load_from_raw True --write_placement True
```
- To run GP + DP flow for `adaptec1` in ISPD2005 dataset:
```console
python main.py --dataset_root your_path --dataset ispd2005 --design_name adaptec1 --load_from_raw True --write_placement True --detail_placement True
```
**Note**: For ISPD2005 dataset, [NTUplace3](http://eda.ee.ntu.edu.tw/research.htm) is used as the detailed placement engine. For ISPD2015 dataset, please run GP only flow and launch [ABCDPlace](https://github.com/limbo018/DREAMPlace) to perform detailed placement.
- Each run will generate serveral output files in `./result/exp_id`. These files can provide valuable information for parameter tuning.
```
In ./result/exp_id
- eval # parameter curves and the visualization of placement
- log # log and statistics
- output # global placement solution files
```
## Parameters
Please refer to `main.py`.
## Load design from preprocessed `pt` file (Optional)
The following script will dump the parsed design into a single torch `pt` file so Xplace can load the design from the `pt` file instead of parsing the input file from scratch.
```console
cd $XPLACE_HOME
python utils/convert_design_to_torch_data.py --dataset_root your_path --dataset ispd2005
```
Preprocessed data is saved in `./data/cad`.
When developing a new global placement technique in Xplace, we highly suggest using the `pt` mode to save the parser time. (set `--load_from_raw False`)
```console
python main.py --dataset ispd2005 --run_all True --load_from_raw False
```
**Note**: Please remember to use the raw mode (set `--load_from_raw True`) when running detailed placement or measuring the total running time.
## Xplace Placement Results
Benchmark | Placement Solutions
|:---:|:---:|
ISPD2005 | [Google Drive](https://drive.google.com/drive/folders/1fUzkT9ymV3n0XxfWXA0mR3WQX55hR1PB?usp=sharing)
ISPD2015 (w/o fence) | [Google Drive](https://drive.google.com/drive/folders/1UsKQ1FQ4fFi4pdJ0VoCoCCjLakhoS20Q?usp=sharing)
## Citation
If you find **Xplace** useful in your research, please consider to cite:
```bibtex
@inproceedings{liu2022xplace,
author={Liu, Lixin and Fu, Bangqi and Wong, Martin D. F. and Young, Evangeline F. Y.},
booktitle={Proceedings of the 59th ACM/IEEE Design Automation Conference},
title={Xplace: An Extremely Fast and Extensible Global Placement Framework},
year={2022},
}
```
Thanks the authors of [ePlace](https://dl.acm.org/doi/10.1145/2699873), [RePlAce](https://github.com/The-OpenROAD-Project/RePlAce), and [DREAMPlace](https://github.com/limbo018/DREAMPlace) for their great work.
```bibtex
@article{lu2015eplace,
author={Lu, Jingwei and Chen, Pengwen and Chang, Chin-Chih and Sha, Lu and Huang, Dennis Jen-Hsin and Teng, Chin-Chi and Cheng, Chung-Kuan},
journal={ACM Trans. Des. Autom. Electron. Syst.},
title={ePlace: Electrostatics-Based Placement Using Fast Fourier Transform and Nesterov's Method},
year={2015},
}
@article{cheng2019replace,
author={Cheng, Chung-Kuan and Kahng, Andrew B. and Kang, Ilgweon and Wang, Lutong},
journal={IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems},
title={RePlAce: Advancing Solution Quality and Routability Validation in Global Placement},
year={2019},
}
@article{lin2021dreamplace,
author={Lin, Yibo and Jiang, Zixuan and Gu, Jiaqi and Li, Wuxi and Dhar, Shounak and Ren, Haoxing and Khailany, Brucek and Pan, David Z.},
journal={IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems},
title={DREAMPlace: Deep Learning Toolkit-Enabled GPU Acceleration for Modern VLSI Placement},
year={2021},
}
```
## Contact
[Lixin Liu](https://liulixinkerry.github.io/) (lxliu@cse.cuhk.edu.hk)
and Bangqi Fu (bqfu21@cse.cuhk.edu.hk)
## License
READ THIS LICENSE AGREEMENT CAREFULLY BEFORE USING THIS PRODUCT. BY USING THIS PRODUCT YOU INDICATE YOUR ACCEPTANCE OF THE TERMS OF THE FOLLOWING AGREEMENT. THESE TERMS APPLY TO YOU AND ANY SUBSEQUENT LICENSEE OF THIS PRODUCT.
License Agreement for Xplace
Copyright (c) 2022 by The Chinese University of Hong Kong
All rights reserved
CU-SD LICENSE (adapted from the original BSD license) Redistribution of the any code, with or without modification, are permitted provided that the conditions below are met.
Redistributions of source code must retain the above copyright notice, this list of conditions and the following disclaimer.
Redistributions in binary form must reproduce the above copyright notice, this list of conditions and the following disclaimer in the documentation and/or other materials provided with the distribution.
Neither the name nor trademark of the copyright holder or the author may be used to endorse or promote products derived from this software without specific prior written permission.
Users are entirely responsible, to the exclusion of the author, for compliance with (a) regulations set by owners or administrators of employed equipment, (b) licensing terms of any other software, and (c) local, national, and international regulations regarding use, including those regarding import, export, and use of encryption software.
THIS FREE SOFTWARE IS PROVIDED BY THE AUTHOR "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHOR OR ANY CONTRIBUTOR BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, EFFECTS OF UNAUTHORIZED OR MALICIOUS NETWORK ACCESS; PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.

BIN
assets/xplace_overview.png Normal file

Binary file not shown.

After

Width:  |  Height:  |  Size: 319 KiB

81
cmake/FindCairo.cmake Normal file
View File

@ -0,0 +1,81 @@
# - Try to find Cairo
# Once done, this will define
#
# CAIRO_FOUND - system has Cairo
# CAIRO_INCLUDE_DIRS - the Cairo include directories
# CAIRO_LIBRARIES - link these to use Cairo
#
# Copyright (C) 2012 Raphael Kubo da Costa <rakuco@webkit.org>
#
# Redistribution and use in source and binary forms, with or without
# modification, are permitted provided that the following conditions
# are met:
# 1. Redistributions of source code must retain the above copyright
# notice, this list of conditions and the following disclaimer.
# 2. Redistributions in binary form must reproduce the above copyright
# notice, this list of conditions and the following disclaimer in the
# documentation and/or other materials provided with the distribution.
#
# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDER AND ITS CONTRIBUTORS ``AS
# IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO,
# THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
# PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR ITS
# CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
# EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
# PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
# OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY,
# WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR
# OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF
# ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
FIND_PACKAGE(PkgConfig)
PKG_CHECK_MODULES(PC_CAIRO cairo) # FIXME: After we require CMake 2.8.2 we can pass QUIET to this call.
FIND_PATH(CAIRO_INCLUDE_DIRS
NAMES cairo.h
HINTS ${PC_CAIRO_INCLUDEDIR}
${PC_CAIRO_INCLUDE_DIRS}
PATH_SUFFIXES cairo
)
FIND_LIBRARY(CAIRO_LIBRARIES
NAMES cairo
HINTS ${PC_CAIRO_LIBDIR}
${PC_CAIRO_LIBRARY_DIRS}
)
IF (CAIRO_INCLUDE_DIRS)
IF (EXISTS "${CAIRO_INCLUDE_DIRS}/cairo-version.h")
FILE(READ "${CAIRO_INCLUDE_DIRS}/cairo-version.h" CAIRO_VERSION_CONTENT)
STRING(REGEX MATCH "#define +CAIRO_VERSION_MAJOR +([0-9]+)" _dummy "${CAIRO_VERSION_CONTENT}")
SET(CAIRO_VERSION_MAJOR "${CMAKE_MATCH_1}")
STRING(REGEX MATCH "#define +CAIRO_VERSION_MINOR +([0-9]+)" _dummy "${CAIRO_VERSION_CONTENT}")
SET(CAIRO_VERSION_MINOR "${CMAKE_MATCH_1}")
STRING(REGEX MATCH "#define +CAIRO_VERSION_MICRO +([0-9]+)" _dummy "${CAIRO_VERSION_CONTENT}")
SET(CAIRO_VERSION_MICRO "${CMAKE_MATCH_1}")
SET(CAIRO_VERSION "${CAIRO_VERSION_MAJOR}.${CAIRO_VERSION_MINOR}.${CAIRO_VERSION_MICRO}")
ENDIF ()
ENDIF ()
# FIXME: Should not be needed anymore once we start depending on CMake 2.8.3
SET(VERSION_OK TRUE)
IF (Cairo_FIND_VERSION)
IF (Cairo_FIND_VERSION_EXACT)
IF ("${Cairo_FIND_VERSION}" VERSION_EQUAL "${CAIRO_VERSION}")
# FIXME: Use IF (NOT ...) with CMake 2.8.2+ to get rid of the ELSE block
ELSE ()
SET(VERSION_OK FALSE)
ENDIF ()
ELSE ()
IF ("${Cairo_FIND_VERSION}" VERSION_GREATER "${CAIRO_VERSION}")
SET(VERSION_OK FALSE)
ENDIF ()
ENDIF ()
ENDIF ()
INCLUDE(FindPackageHandleStandardArgs)
FIND_PACKAGE_HANDLE_STANDARD_ARGS(Cairo DEFAULT_MSG CAIRO_INCLUDE_DIRS CAIRO_LIBRARIES VERSION_OK)

7
cpp_to_py/.clang-format Normal file
View File

@ -0,0 +1,7 @@
BasedOnStyle: Google
IndentWidth: 4
AccessModifierOffset: -4
BinPackArguments: false
BinPackParameters: false
ColumnLimit: 120
SortIncludes: true

9
cpp_to_py/CMakeLists.txt Normal file
View File

@ -0,0 +1,9 @@
add_subdirectory(common)
add_subdirectory(dct_cuda)
add_subdirectory(density_map_cuda)
add_subdirectory(draw_placement)
add_subdirectory(flute_cpp)
add_subdirectory(hpwl_cuda)
add_subdirectory(io_parser)
add_subdirectory(wa_wirelength_hpwl_cuda)
add_subdirectory(node_pos_to_pin_pos_cuda)

3
cpp_to_py/README.md Normal file
View File

@ -0,0 +1,3 @@
# Add new module
To add new module, please modify `__init__.py` and `CMakeLists.txt`.

22
cpp_to_py/__init__.py Normal file
View File

@ -0,0 +1,22 @@
import torch
__all__ = [
"dct_cuda",
"flute_cpp",
"hpwl_cuda",
"io_parser",
"density_map_cuda",
"draw_placement",
"node_pos_to_pin_pos_cuda",
"wa_wirelength_hpwl_cuda",
]
from .cpybin import (
dct_cuda,
flute_cpp,
hpwl_cuda,
io_parser,
density_map_cuda,
draw_placement,
node_pos_to_pin_pos_cuda,
wa_wirelength_hpwl_cuda,
)

View File

@ -0,0 +1,15 @@
file(GLOB_RECURSE SRC_FILES_XPLACE_COMMON ${CMAKE_CURRENT_SOURCE_DIR}/*.cpp)
find_library(LIBDEF def ${PATH_THIRDPARTY_ROOT}/lefdef/def58/lib)
find_library(LIBLEF lef ${PATH_THIRDPARTY_ROOT}/lefdef/lef58/lib)
add_library(xplace_common SHARED ${SRC_FILES_XPLACE_COMMON})
target_include_directories(
xplace_common PRIVATE ${PROJECT_SOURCE_DIR}/cpp_to_py ${PATH_THIRDPARTY_ROOT}/lefdef ${TORCH_INCLUDE_DIRS})
target_link_libraries(
xplace_common PRIVATE torch ${TORCH_PYTHON_LIBRARY} ${LIBDEF} ${LIBLEF})
target_compile_options(xplace_common PRIVATE -fPIC)
install(TARGETS
xplace_common
DESTINATION ${XPLACE_LIB_DIR})

49
cpp_to_py/common/common.h Normal file
View File

@ -0,0 +1,49 @@
#pragma once
// STL libraries
#include <fstream>
#include <iostream>
#include <sstream>
#include <iomanip>
#include <random>
#include <memory>
#include <algorithm>
#include <list>
#include <map>
#include <set>
#include <queue>
#include <stack>
#include <string>
#include <unordered_map>
#include <unordered_set>
#include <vector>
#include <cassert>
#include <cfloat>
#include <climits>
#include <cmath>
#include <cstdio>
#include <cstdlib>
#include <csignal>
#include <cstring>
#include <thread>
// Torch library
#include <torch/extension.h>
// utils
#include "common/utils/utils.h"
// Pybind11
#include <pybind11/pybind11.h>
#include <pybind11/stl.h>
#include <pybind11/stl_bind.h>
#include <pybind11/numpy.h>
using namespace std;
using utils::assert_msg;
using utils::print;
using utils::printlog;

View File

@ -0,0 +1,16 @@
#pragma once
namespace db {
class BsRouteInfo {
// specially restore ICCAD2012/DAC2012 bookshelf routing info
public:
BsRouteInfo() {}
bool hasInfo = false;
double blockagePorosity = 0.0;
vector<int> capV; // numTracksV = capV[i] / layerPitch[i]
vector<int> capH;
vector<int> viaSpace;
private:
};
} // namespace db

View File

@ -0,0 +1,130 @@
#include "Database.h"
using namespace db;
/***** Cell *****/
Cell::~Cell() {
for (Pin* pin : _pins) {
delete pin;
}
_pins.clear();
}
Pin* Cell::pin(const string& name) const {
for (Pin* pin : _pins) {
if (pin->type->name() == name) {
return pin;
}
}
return nullptr;
}
void Cell::ctype(CellType* t) {
if (!t) {
return;
}
if (_type) {
printlog(LOG_ERROR, "type of cell %s already set", _name.c_str());
return;
}
_type = t;
++(_type->usedCount);
_pins.resize(_type->pins.size(), nullptr);
for (unsigned i = 0; i != _pins.size(); ++i) {
_pins[i] = new Pin(this, i);
}
}
int Cell::lx() const { return _lx; }
int Cell::ly() const { return _ly; }
bool Cell::flipX() const { return _flipX; }
bool Cell::flipY() const { return _flipY; }
int Cell::orient() const {
if (!flipX() && !flipY()) {
return 0; // N
} else if (flipX() && flipY()) {
return 2; // S
} else if (flipX() && !flipY()) {
return 4; // FN
} else if (!flipX() && flipY()) {
return 6; // FS
}
return 0;
}
bool Cell::placed() const { return (lx() != INT_MIN) && (ly() != INT_MIN); }
// int Cell::siteWidth() const { return width() / database.siteW; }
// int Cell::siteHeight() const { return height() / database.siteH; }
void Cell::place(int x, int y) {
if (_fixed) {
printlog(LOG_WARN, "moving fixed cell %s to (%d,%d)", _name.c_str(), x, y);
}
_lx = x;
_ly = y;
}
void Cell::place(int x, int y, bool flipX, bool flipY) {
if (_fixed) {
printlog(LOG_WARN, "moving fixed cell %s to (%d,%d)", _name.c_str(), x, y);
}
_lx = x;
_ly = y;
_flipX = flipX;
_flipY = flipY;
}
void Cell::unplace() {
if (_fixed) {
printlog(LOG_WARN, "unplace fixed cell %s", _name.c_str());
}
_lx = _ly = INT_MIN;
_flipX = _flipY = false;
}
/***** Cell Type *****/
CellType::~CellType() {
for (PinType* pin : pins) {
delete pin;
}
}
PinType* CellType::addPin(const string& name, const char direction, const char type) {
PinType* newpintype = new PinType(name, direction, type);
pins.push_back(newpintype);
return newpintype;
}
PinType* CellType::getPin(string& name) {
for (int i = 0; i < (int)pins.size(); i++) {
if (pins[i]->name() == name) {
return pins[i];
}
}
return nullptr;
}
void CellType::setOrigin(int x, int y) {
_originX = x;
_originY = y;
}
bool CellType::operator==(const CellType& r) const {
if (width != r.width || height != r.height) {
return false;
} else if (_originX != r.originX() || _originY != r.originY() || _symmetry != r.symmetry() ||
pins.size() != r.pins.size()) {
return false;
} else if (edgetypeL != r.edgetypeL || edgetypeR != r.edgetypeR) {
return false;
} else {
// return PinType::comparePin(pins, r.pins);
for (unsigned i = 0; i != pins.size(); ++i) {
if (*pins[i] != *r.pins[i]) {
return false;
}
}
}
return true;
}

129
cpp_to_py/common/db/Cell.h Normal file
View File

@ -0,0 +1,129 @@
#pragma once
namespace db {
class CellType {
friend class Database;
private:
int _originX = 0;
int _originY = 0;
// X=1 Y=2 R90=4 (can be combined)
char _symmetry = 0;
string _siteName = "";
char _botPower = 'x';
char _topPower = 'x';
vector<Geometry> _obs;
// _nonRegularRects.size() > 0 implies that this cell is a fixed cell and
// its shape is a polygon. Each rectangle is also appended into
// Databased.placeBlockages during parsing the polygon-shape cell.
// NOTE: We only support this feature for ICCAD/DAC 2012 benchmarks.
// When processing GPDatabase, we will set this kind of cells' width and
// height as 0 and use placeement blockages to represent their shapes.
vector<Rectangle> _nonRegularRects;
int _libcell = -1;
public:
std::string name = "";
char cls = 'x';
bool stdcell = false;
int width = 0;
int height = 0;
vector<PinType*> pins;
int edgetypeL = 0;
int edgetypeR = 0;
int usedCount = 0;
CellType(const string& name, int libcell) : _libcell(libcell), name(name) {}
~CellType();
PinType* addPin(const string& name, const char direction, const char type);
void addPin(PinType& pintype);
template <class... Args>
void addObs(Args&&... args) {
_obs.emplace_back(args...);
}
template <class... Args>
void addNonRegularRects(Args&&... args) {
_nonRegularRects.emplace_back(args...);
}
PinType* getPin(string& name);
int originX() const { return _originX; }
int originY() const { return _originY; }
char symmetry() const { return _symmetry; }
char botPower() const { return _botPower; }
char topPower() const { return _topPower; }
const std::vector<Geometry>& obs() const { return _obs; }
const std::vector<Rectangle>& nonRegularRects() const { return _nonRegularRects; }
int libcell() const { return _libcell; }
void setOrigin(int x, int y);
void setXSymmetry() { _symmetry &= 1; }
void setYSymmetry() { _symmetry &= 2; }
void set90Symmetry() { _symmetry &= 4; }
void siteName(const std::string& name) { _siteName = name; }
bool operator==(const CellType& r) const;
bool operator!=(const CellType& r) const { return !(*this == r); }
};
class Cell {
private:
string _name = "";
int _spaceL = 0;
int _spaceR = 0;
int _spaceB = 0;
int _spaceT = 0;
bool _fixed = false;
CellType* _type = nullptr;
std::vector<Pin*> _pins;
int _lx = INT_MIN;
int _ly = INT_MIN;
bool _flipX = false;
bool _flipY = false;
public:
bool highlighted = false;
Region* region = nullptr;
int gpdb_id = -1;
bool is_connected = false;
Cell(const string& name = "", CellType* t = nullptr) : _name(name) { ctype(t); }
~Cell();
const std::string& name() const { return _name; }
Pin* pin(const std::string& name) const;
Pin* pin(unsigned i) const { return _pins[i]; }
CellType* ctype() const { return _type; }
void ctype(CellType* t);
int lx() const;
int ly() const;
int hx() const { return lx() + width(); }
int hy() const { return ly() + height(); }
int cx() const { return lx() + width() / 2; }
int cy() const { return ly() + height() / 2; }
bool flipX() const;
bool flipY() const;
int orient() const;
int width() const { return _type->width + _spaceL + _spaceR; }
int height() const { return _type->height + _spaceB + _spaceT; }
// int siteWidth() const;
// int siteHeight() const;
bool fixed() const { return _fixed; }
void fixed(bool fix) { _fixed = fix; }
bool placed() const;
void place(int x, int y);
void place(int x, int y, bool flipX, bool flipY);
void unplace();
unsigned numPins() const { return _pins.size(); }
friend ostream& operator<<(ostream& os, const Cell& c) {
return os << c._name << "\t(" << c.lx() << ", " << c.ly() << ')';
}
};
} // namespace db

File diff suppressed because it is too large Load Diff

View File

@ -0,0 +1,252 @@
#pragma once
#include "common/common.h"
#include "Setting.h"
namespace db {
class Rectangle;
class Geometry;
class GeoMap;
class Cell;
class CellType;
class Pin;
class PinType;
class IOPin;
class CellPin;
class Net;
class Row;
class RowSegment;
class Track;
class Layer;
class Via;
class ViaRule;
class ViaType;
class Region;
class NDR;
class SNet;
class Site;
class PowerNet;
class EdgeTypes;
class GCellGrid;
class BsRouteInfo;
} // namespace db
#include "Cell.h"
#include "DesignRule.h"
#include "Geometry.h"
#include "Layer.h"
#include "SiteMap.h"
#include "Net.h"
#include "Pin.h"
#include "Region.h"
#include "GCellGrid.h"
#include "Row.h"
#include "Site.h"
#include "SNet.h"
#include "Via.h"
#include "BsRouteInfo.h"
namespace db {
#define CLEAR_POINTER_LIST(list) \
{ \
for (auto obj : list) { \
delete obj; \
} \
list.clear(); \
}
#define CLEAR_POINTER_MAP(map) \
{ \
for (auto p : map) { \
delete p.second; \
} \
map.clear(); \
}
class Database {
public:
enum IssueType {
E_ROW_EXCEED_DIE,
E_OVERLAP_ROWS,
W_NON_UNIFORM_SITE_WIDTH,
W_NON_HORIZONTAL_ROW,
E_MULTIPLE_NET_DRIVING_PIN,
E_NO_NET_DRIVING_PIN
};
unordered_map<string, CellType*> name_celltypes;
unordered_map<string, Cell*> name_cells;
unordered_map<string, Net*> name_nets;
unordered_map<string, IOPin*> name_iopins;
unordered_map<string, ViaType*> name_viatypes;
vector<Layer> layers;
vector<Site> sites;
vector<ViaType*> viatypes;
vector<CellType*> celltypes;
vector<Cell*> cells;
vector<IOPin*> iopins;
vector<Net*> nets;
vector<Row*> rows;
vector<Region*> regions;
map<string, NDR*> ndrs;
vector<SNet*> snets;
vector<Track*> tracks;
vector<Geometry> routeBlockages;
vector<Rectangle> placeBlockages;
PowerNet powerNet;
private:
static const size_t _bufferCapacity = 128 * 1024;
size_t _bufferSize = 0;
char* _buffer = nullptr;
public:
unsigned siteW = 0;
int siteH = 0;
unsigned nSitesX = 0;
unsigned nSitesY = 0;
SiteMap siteMap;
GCellGrid gcellgrid;
BsRouteInfo bsRouteInfo;
EdgeTypes edgetypes;
int dieLX, dieLY, dieHX, dieHY;
int coreLX, coreLY, coreHX, coreHY;
double maxDensity = 0;
double maxDisp = 0;
int LefConvertFactor;
double DBU_Micron;
double version;
string designName;
vector<IssueType> dbIssues;
public:
Database();
~Database();
void clear();
void clearTechnology();
inline void clearLibrary() { CLEAR_POINTER_LIST(celltypes); }
void clearDesign();
Layer& addLayer(const string& name, const char type = 'x');
Site& addSite(const string& name, const string& siteClassName, const int w, const int h);
ViaType* addViaType(const string& name, bool isDef);
inline ViaType* addViaType(const string& name) { return addViaType(name, false); }
CellType* addCellType(const string& name, unsigned libcell);
void reserveCells(const size_t n) { cells.reserve(n); }
Cell* addCell(const string& name, CellType* type = nullptr);
IOPin* addIOPin(const string& name = "", const string& netName = "", const char direction = 'x');
void reserveNets(const size_t n) { nets.reserve(n); }
Net* addNet(const string& name = "", const NDR* ndr = nullptr);
Row* addRow(const string& name,
const string& macro,
const int x,
const int y,
const unsigned xNum = 0,
const unsigned yNum = 0,
const bool flip = false,
const unsigned xStep = 0,
const unsigned yStep = 0);
Track* addTrack(char direction, double start, double num, double step);
Region* addRegion(const string& name = "", const char type = 'x');
NDR* addNDR(const string& name, const bool hardSpacing);
void reserveSNets(const size_t n) { snets.reserve(n); }
SNet* addSNet(const string& name);
Layer* getRLayer(const int index);
const Layer* getCLayer(const unsigned index) const;
Layer* getLayer(const string& name);
CellType* getCellType(const string& name);
Cell* getCell(const string& name);
Net* getNet(const string& name);
Region* getRegion(const string& name);
Region* getRegion(const unsigned char id);
NDR* getNDR(const string& name) const;
IOPin* getIOPin(const string& name) const;
ViaType* getViaType(const string& name) const;
SNet* getSNet(const string& name);
unsigned getNumRLayers() const;
unsigned getNumCLayers() const;
inline unsigned getNumLayers() const { return layers.size(); }
inline unsigned getNumCells() const { return cells.size(); }
inline unsigned getNumNets() const { return nets.size(); }
inline unsigned getNumRegions() const { return regions.size(); }
inline unsigned getNumIOPins() const { return iopins.size(); }
inline unsigned getNumCellTypes() const { return celltypes.size(); }
inline int getCellTypeSpace(const CellType* L, const CellType* R) const {
return edgetypes.getEdgeSpace(L->edgetypeR, R->edgetypeL);
}
inline int getCellTypeSpace(const Cell* L, const Cell* R) const { return getCellTypeSpace(L->ctype(), R->ctype()); }
int getContainedSites(
const int lx, const int ly, const int hx, const int hy, int& slx, int& sly, int& shx, int& shy) const;
int getOverlappedSites(
const int lx, const int ly, const int hx, const int hy, int& slx, int& sly, int& shx, int& shy) const;
long long getHPWL();
long long getCellArea(Region* region = nullptr) const;
long long getFreeArea(Region* region = nullptr) const;
bool placed();
bool globalRouted();
bool detailedRouted();
void errorCheck(bool autoFix = true);
void checkPlaceError();
void checkDRCError();
void load();
void setup(); // call after read
void reset();
void save(const std::string& given_prefix);
/* defined in io/file_lefdef_db.cpp */
public:
bool readLEF(const std::string& file);
bool readDEF(const std::string& file);
bool readDEFPG(const string& file);
bool writeDEF(const std::string& file);
bool writeICCAD2017(const string& inputDef, const string& outputDef);
bool writeICCAD2017(const string& outputDef);
bool writeComponents(ofstream& ofs);
bool writeBuffer(ofstream& ofs, const string& line);
void writeBufferFlush(ofstream& ofs);
bool readBSAux(const std::string& auxFile, const std::string& plFile);
bool readBSNodes(const std::string& file);
bool readBSNets(const std::string& file);
bool readBSScl(const std::string& file);
bool readBSRoute(const std::string& file);
bool readBSShapes(const std::string& file);
bool readBSWts(const std::string& file);
bool readBSPl(const std::string& file);
bool writeBSPl(const std::string& file);
bool readVerilog(const std::string& file);
bool readLiberty(const std::string& file);
bool readConstraints(const std::string& file);
bool readSize(const std::string& file);
private:
void SetupLayers();
void SetupCellLibrary();
void SetupFloorplan();
void SetupRegions();
void SetupSiteMap();
void SetupRows();
void SetupRowSegments();
};
} // namespace db

View File

@ -0,0 +1,30 @@
#include "Database.h"
using namespace db;
/***** EdgeTypes *****/
int EdgeTypes::getEdgeType(const string& name) const {
unsigned numTypes = types.size();
for (unsigned i = 0; i != numTypes; ++i) {
if (types[i] == name) {
return i;
}
}
return -1;
}
int EdgeTypes::getEdgeSpace(const int edge1, const int edge2) const {
if (!distTable.size()) {
return 0;
}
#ifdef DEBUG
if (edge1 < 0 || edge1 >= (int)types.size()) {
printlog(LOG_ERROR, "invalid edge ID: %d", edge1);
}
if (edge2 < 0 || edge2 >= (int)types.size()) {
printlog(LOG_ERROR, "invalid edge ID: %d", edge2);
}
#endif
return distTable[edge1][edge2];
}

View File

@ -0,0 +1,41 @@
#pragma once
namespace db {
class WireRule {
private:
const Layer* _layer;
public:
const unsigned width;
const unsigned space;
WireRule(const Layer* layer, const unsigned width, const unsigned space)
: _layer(layer), width(width), space(space) {}
const Layer* layer() const { return _layer; }
};
class NDR {
private:
const string _name;
bool _hardSpacing = false;
public:
vector<WireRule> rules;
vector<ViaType*> vias;
NDR(const string& name, const bool hardSpacing) : _name(name), _hardSpacing(hardSpacing) {}
const string& name() const { return _name; }
bool hardSpacing() const { return _hardSpacing; }
};
class EdgeTypes {
public:
vector<string> types = {"default"};
vector<vector<int>> distTable = {{0}};
int getEdgeType(const string& name) const;
int getEdgeSpace(const int edge1, const int edge2) const;
};
} // namespace db

View File

@ -0,0 +1,17 @@
#pragma once
namespace db {
class GCellGrid {
public:
std::vector<int> startX, numX, stepX;
std::vector<int> startY, numY, stepY;
// int nx = 0;
// int ny = 0;
// std::vector<int> gcellX;
// std::vector<int> gcellY;
};
} // namespace db

View File

@ -0,0 +1,122 @@
#include "Database.h"
using namespace db;
/***** Rectangle *****/
Rectangle& Rectangle::operator+=(const Rectangle& geo) {
lx = min(lx, geo.lx);
ly = min(ly, geo.ly);
hx = max(hx, geo.hx);
hy = max(hy, geo.hy);
return *this;
}
void Rectangle::sliceH(vector<Rectangle>& rects) {
// sort all bottom and top bounds
vector<int> ys;
for (const Rectangle& rect : rects) {
ys.push_back(rect.ly);
ys.push_back(rect.hy);
}
sort(ys.begin(), ys.end());
// remove duplicated values
ys.erase(unique(ys.begin(), ys.end()), ys.end());
// cut each rect with the y values
vector<Rectangle> slices;
for (const Rectangle& rect : rects) {
vector<int>::iterator yi = lower_bound(ys.begin(), ys.end(), rect.ly);
vector<int>::iterator ye = upper_bound(yi, ys.end(), rect.hy);
int lasty = *(yi++);
for (; yi != ye; yi++) {
slices.push_back(Rectangle(rect.lx, lasty, rect.hx, *yi));
lasty = *yi;
}
}
// merge overlapping rect
sort(slices.begin(), slices.end(), [](const Rectangle& a, const Rectangle& b) -> bool {
return a.ly == b.ly ? a.lx < b.lx : a.ly < b.ly;
});
for (int i = 1; i < (int)slices.size(); ++i) {
Rectangle& L = slices[i - 1];
Rectangle& R = slices[i];
if (L.ly != R.ly || L.hy != R.hy) {
continue;
}
if (L.hx >= R.lx) {
R.lx = min(L.lx, R.lx);
R.hx = max(L.hx, R.hx);
L.hx = L.lx;
}
}
// remove empty rects
vector<Rectangle>::iterator sEnd = remove_if(slices.begin(), slices.end(), Rectangle::IsInvalid);
if (sEnd != slices.end()) {
slices.resize(distance(slices.begin(), sEnd));
}
rects.swap(slices);
}
void Rectangle::sliceV(vector<Rectangle>& rects) {
// sort all left and right bounds
vector<int> xs;
for (const Rectangle& rect : rects) {
xs.push_back(rect.lx);
xs.push_back(rect.hx);
}
sort(xs.begin(), xs.end());
// remove duplicated values
xs.erase(unique(xs.begin(), xs.end()), xs.end());
// cut each rect with the y values
vector<Rectangle> slices;
for (const Rectangle& rect : rects) {
vector<int>::iterator xi = lower_bound(xs.begin(), xs.end(), rect.lx);
vector<int>::iterator xe = upper_bound(xi, xs.end(), rect.hx);
int lastx = *(xi++);
for (; xi != xe; xi++) {
slices.push_back(Rectangle(lastx, rect.ly, *xi, rect.hy));
lastx = *xi;
}
}
// merge overlapping rect
sort(slices.begin(), slices.end(), [](const Rectangle& a, const Rectangle& b) -> bool {
return a.lx == b.lx ? a.ly < b.ly : a.lx < b.lx;
});
for (int i = 1; i < (int)slices.size(); ++i) {
Rectangle& L = slices[i - 1];
Rectangle& R = slices[i];
if (L.lx != R.lx || L.hx != R.hx) {
continue;
}
if (L.hy >= R.ly) {
R.ly = min(L.ly, R.ly);
R.hy = max(L.hy, R.hy);
L.hy = L.ly;
}
}
// remove empty rects
vector<Rectangle>::iterator sEnd = remove_if(slices.begin(), slices.end(), Rectangle::IsInvalid);
if (sEnd != slices.end()) {
slices.resize(distance(slices.begin(), sEnd));
}
rects.swap(slices);
}
/***** Geometry *****/
bool Geometry::operator==(const Geometry& rhs) const { return layer == rhs.layer && Rectangle::operator==(rhs); };
/***** GeoMap *****/
void GeoMap::emplace(const int k, const Geometry& shape) {
if (_map.find(k) == _map.end()) {
_map.emplace(k, shape);
} else {
_map.at(k) += shape;
}
Rectangle::operator+=(shape);
}

View File

@ -0,0 +1,72 @@
#pragma once
namespace db {
class Rectangle {
public:
int lx = INT_MAX;
int ly = INT_MAX;
int hx = INT_MIN;
int hy = INT_MIN;
Rectangle(int lx = INT_MAX, int ly = INT_MAX, int hx = INT_MIN, int hy = INT_MIN)
: lx(lx), ly(ly), hx(hx), hy(hy) {}
int w() const { return hx - lx; }
int h() const { return hy - ly; }
int cx() const { return (hx + lx) / 2; }
int cy() const { return (hy + ly) / 2; }
bool operator<(const Rectangle& geo) const { return (ly == geo.ly) ? (lx < geo.lx) : (ly < geo.ly); }
bool operator==(const Rectangle& geo) const { return lx == geo.lx && ly == geo.ly && hx == geo.hx && hy == geo.hy; }
Rectangle& operator+=(const Rectangle& geo);
static bool CompareXInc(const Rectangle& a, const Rectangle& b) {
return (a.lx == b.lx) ? (a.hx < b.hx) : (a.lx < b.lx);
}
static bool CompareXDec(const Rectangle& a, const Rectangle& b) {
return (a.hx == b.hx) ? (a.lx > b.lx) : (a.hx > b.hx);
}
static bool CompareYInc(const Rectangle& a, const Rectangle& b) {
return (a.ly == b.ly) ? (a.hy < b.hy) : (a.ly < b.ly);
}
static bool CompareYDec(const Rectangle& a, const Rectangle& b) {
return (a.hy == b.hy) ? (a.ly > b.ly) : (a.hy > b.hy);
}
static bool IsInvalid(const Rectangle& geo) { return (geo.lx >= geo.hx) || (geo.ly >= geo.hy); }
static void sliceH(vector<Rectangle>& rects);
static void sliceV(vector<Rectangle>& rects);
friend ostream& operator<<(ostream& os, const Rectangle& r) {
return os << '(' << r.lx << ", " << r.ly << ")\t(" << r.hx << ", " << r.hy << ')';
}
};
class Geometry : public Rectangle {
public:
const Layer& layer;
Geometry(const Layer& layer, const int lx, const int ly, const int hx, const int hy)
: Rectangle(lx, ly, hx, hy), layer(layer) {}
Geometry(const Geometry& geom) : Rectangle(geom), layer(geom.layer) {}
inline bool operator==(const Geometry& rhs) const;
};
class GeoMap : public Rectangle {
private:
map<int, Geometry> _map;
public:
bool empty() const noexcept { return _map.empty(); }
size_t size() const noexcept { return _map.size(); }
const Geometry& front() const { return _map.begin()->second; }
const Geometry& front2() const { return (++_map.begin())->second; }
const Geometry& back() const { return _map.rbegin()->second; }
map<int, Geometry>::const_iterator begin() const noexcept { return _map.begin(); }
map<int, Geometry>::const_iterator end() const noexcept { return _map.end(); }
map<int, Geometry>::const_iterator find(const int k) const { return _map.find(k); }
const Geometry& at(const int k) const { return _map.at(k); }
void emplace(const int k, const Geometry& shape);
};
} // namespace db

View File

@ -0,0 +1,16 @@
#include "Database.h"
using namespace db;
/***** Track *****/
char Track::macro() const {
switch (direction) {
case 'h':
return 'Y';
case 'v':
return 'X';
default:
printlog(LOG_ERROR, "track direction not recognized: %c", direction);
return '\0';
}
}

View File

@ -0,0 +1,88 @@
#pragma once
namespace db {
class Track {
// contain DEF track info
// support multi layer track DEF definition
// For example:
// TRACKS Y 0 DO 2752 STEP 200 LAYER metal1 metal3 metal5 ;
// Variable will be:
// layers_ = {metal1, metal3, metal5}
// direction = 'h', start = 0, num = 2752, step = 200
protected:
vector<string> layers_;
public:
char direction = 'x'; // 'x' for None, 'v' for "X", 'h' for "Y"
int start = INT_MAX;
unsigned num = 0;
unsigned step = 0;
Track(const char direction = 'x', const int start = INT_MAX, const unsigned num = 0, const unsigned step = 0)
: direction(direction), start(start), num(num), step(step) {}
void addLayer(const string& layer) { layers_.push_back(layer); }
const vector<string>& getLayer() const { return layers_; }
const int getFirstTrackLoc() const { return start; }
const int getLastTrackLoc() const { return start + (num - 1) * step; }
const int getTrackLoc(unsigned trackIndex) const { return start + trackIndex * step; }
const int getPitch() const { return step; }
char macro() const;
unsigned numLayers() const { return layers_.size(); }
const string& layer(unsigned index) const { return layers_[index]; }
};
class Layer {
friend class Database;
private:
string _name = "";
//'r' for route or 'c' for cut
char _type = 'x';
Layer* _below = nullptr;
Layer* _above = nullptr;
public:
char direction = 'x';
// int index;
// index at route layers 'M1' = 0
int rIndex = -1;
// index at cut layers 'M12' = 0
int cIndex = -1;
// for route layer
int pitch = -1;
int offset = -1;
int width = -1;
int area = -1;
int minWidth = -1;
int maxWidth = -1;
int spacing = -1; // minSpacing
std::tuple<int, int, int> maxEOLSpace = {-1, -1, -1}; // spacing, width, within
std::tuple<int, int, int, int, int> maxEOLSpaceParallelEdge = {
-1, -1, -1, -1, -1}; // spacing, width, within, parSpace, parWithin
// ParallelRunLength Spacing
vector<int> parLength;
vector<int> parWidth;
vector<vector<int>> parWidthSpace; // 2D table: (parWidth, parLength) -> widthSpacing
Track track;
Track nonPreferDirTrack;
Layer(const string& name = "", const char type = 'x') : _name(name), _type(type) {}
const string& name() const { return _name; }
bool isRouteLayer() const { return _type == 'r'; }
bool isCutLayer() const { return _type == 'c'; }
Layer* getLayerBelow() const { return _below; }
Layer* getLayerAbove() const { return _above; }
bool operator==(const Layer& rhs) const { return rIndex == rhs.rIndex && cIndex == rhs.cIndex; }
bool operator!=(const Layer& rhs) const { return !(*this == rhs); }
};
} // namespace db

142
cpp_to_py/common/db/Net.cpp Normal file
View File

@ -0,0 +1,142 @@
#include "Database.h"
using namespace db;
/***** NetRouteNode *****/
NetRouteNode::NetRouteNode(const Layer* layer, const int x, const int y, const int z)
: _layer(layer), x(x), y(y), z(z) {}
/***** NetRouteSegment *****/
NetRouteSegment::NetRouteSegment(const unsigned fromi, const unsigned toi, const char dir, const unsigned len)
: fromNode(fromi), toNode(toi) {
if (len) {
path.emplace_back(dir, len);
}
}
long long NetRouteSegment::length() const {
long long len = 0;
for (const auto& [dir, plen] : path) {
if (dir != 'U' && dir != 'D') {
len += plen;
}
}
return len;
}
/***** NetRouting *****/
void NetRouting::addWire(const Layer* layer,
const int fromx,
const int fromy,
const int fromz,
const int tox,
const int toy,
const int toz) {
NetRouteNode fromNode(layer, fromx, fromy, fromz);
NetRouteNode toNode(layer, tox, toy, toz);
int fromi = -1;
int toi = -1;
for (unsigned i = 0; i != nodes.size(); ++i) {
if (nodes[i] == fromNode) {
fromi = i;
}
if (nodes[i] == toNode) {
toi = i;
}
}
if (fromi < 0) {
fromi = nodes.size();
nodes.push_back(fromNode);
}
if (toi < 0) {
toi = nodes.size();
nodes.push_back(toNode);
}
if (fromi == toi) {
return;
}
char dir = '\0';
unsigned len = 0;
if (fromx < tox) {
dir = 'E';
len = tox - fromx;
} else if (fromx > tox) {
dir = 'W';
len = fromx - tox;
} else if (fromy < toy) {
dir = 'N';
len = toy - fromy;
} else {
dir = 'S';
len = fromy - toy;
}
segments.emplace_back(fromi, toi, dir, len);
}
void NetRouting::clear() {
segments.clear();
nodes.clear();
}
long long NetRouting::length() const {
long long len = 0;
for (const NetRouteSegment& seg : segments) {
len += seg.length();
}
return len;
}
/***** Net *****/
void Net::addPin(Pin* pin) {
// if (pin->type->direction() == 'o' && pins.size()) {
// Pin* firstPin = pins[0];
// pins[0] = pin;
// pins.push_back(firstPin);
// } else {
// pins.push_back(pin);
// }
// NOTE: we don't swap the output pin to the vector head now
pins.push_back(pin);
}
/***** PowerNet *****/
void PowerNet::addRail(SNet* snet, int lx, int hx, int y) {
map<int, SNet*>::iterator rail = rails.find(y);
if (rail != rails.end() && rail->second != snet) {
printlog(LOG_ERROR, "rail %s already exists at y=%d , new rail %s is from %d to %d", rail->second->name.c_str(),
y, snet->name.c_str(),
lx,
hx);
return;
}
rails.emplace(y, snet);
}
bool PowerNet::getRowPower(int ly, int hy, char& topPower, char& botPower) {
bool valid = true;
topPower = 'x';
botPower = 'x';
map<int, SNet*>::iterator topRail = rails.find(hy);
if (topRail != rails.end()) {
topPower = topRail->second->type;
} else {
valid = false;
}
map<int, SNet*>::iterator botRail = rails.find(ly);
if (botRail != rails.end()) {
botPower = botRail->second->type;
} else {
valid = false;
}
return valid;
}

106
cpp_to_py/common/db/Net.h Normal file
View File

@ -0,0 +1,106 @@
#pragma once
namespace db {
class NetRouteNode {
private:
const Layer* _layer = nullptr;
public:
int x = 0;
int y = 0;
int z = 0;
Pin* pin;
NetRouteNode(const Layer* layer = nullptr, const int x = 0, const int y = 0, const int z = 0);
inline bool operator==(const NetRouteNode& r) const {
return _layer == r._layer && x == r.x && y == r.y && z == r.z;
}
};
class NetRouteSegment {
public:
int z;
unsigned fromNode;
unsigned toNode;
// path = [<direction,len>]
// direction : N,S,E,W,U,D
vector<pair<char, int>> path;
NetRouteSegment(const unsigned fromi = 0, const unsigned toi = 0, const char dir = '\0', const unsigned len = 0);
long long length() const;
};
class NetRouting {
public:
vector<NetRouteNode> nodes;
vector<NetRouteSegment> segments;
void addWire(const Layer* layer,
const int fromx,
const int fromy,
const int fromz,
const int tox,
const int toy,
const int toz);
void clear();
long long length() const;
};
class Net {
private:
NetRouting _routing;
bool gRouted = false;
bool dRouted = false;
public:
const string name = "";
std::vector<Pin*> pins;
const NDR* ndr = nullptr;
int gpdb_id = -1;
Net(const string& name, const NDR* ndr = nullptr) : name(name), ndr(ndr) {}
Net(const Net& net) : _routing(net._routing), name(net.name), pins(net.pins), ndr(net.ndr) {}
bool globalRouted() const { return gRouted; }
bool detailedRouted() const { return dRouted; }
unsigned numPins() const { return pins.size(); }
void resetRouting() { _routing.clear(); }
void addPin(Pin* pin);
void addWire(const Layer* layer,
const int fromx,
const int fromy,
const int fromz,
const int tox,
const int toy,
const int toz) {
_routing.addWire(layer, fromx, fromy, fromz, tox, toy, toz);
}
};
/*only support horizontal power rails*/
class PowerRail {
public:
SNet* snet;
int lx;
int hx;
PowerRail(SNet* sn, int l, int h) {
snet = sn;
lx = l;
hx = h;
}
};
class PowerNet {
private:
std::map<int, SNet*> rails;
public:
void addRail(SNet* snet, int lx, int hx, int y);
bool getRowPower(int ly, int hy, char& topPower, char& botPower);
};
} // namespace db

162
cpp_to_py/common/db/Pin.cpp Normal file
View File

@ -0,0 +1,162 @@
#include "Database.h"
using namespace db;
/***** PinSTA *****/
PinSTA::PinSTA() {
type = 'x';
capacitance = -1.0;
for (int i = 0; i < 4; i++) {
aat[i] = 0.0;
rat[i] = 0.0;
slack[i] = 0.0;
nCriticalPaths[i] = 0;
}
}
/***** IOPin *****/
IOPin::IOPin(const string& name, const string& netName, const char direction) : _netName(netName), name(name) {
type = new PinType(name, direction, 's');
pin = new Pin(this);
}
IOPin::~IOPin() {
delete pin;
delete type;
}
void IOPin::getBounds(int& lx, int& ly, int& hx, int& hy, int& rIndex) const {
type->getBounds(lx, ly, hx, hy);
lx += x;
ly += y;
hx += x;
hy += y;
if (type->shapes.size()) {
rIndex = type->shapes[0].layer.rIndex;
} else {
rIndex = -1;
}
}
/***** Pin *****/
void Pin::getPinCenter(int& x, int& y) {
int lx, ly, hx, hy;
type->getBounds(lx, ly, hx, hy);
if (cell) {
x = cell->lx() + (lx + hx) / 2;
y = cell->ly() + (ly + hy) / 2;
} else if (iopin) {
x = iopin->x + (lx + hx) / 2;
y = iopin->y + (ly + hy) / 2;
} else {
printlog(LOG_ERROR, "invalid pin %s:%d", __FILE__, __LINE__);
x = INT_MIN;
y = INT_MIN;
}
}
void Pin::getPinBounds(int& lx, int& ly, int& hx, int& hy, int& rIndex) {
if (cell) {
type->getBounds(lx, ly, hx, hy);
lx += cell->lx();
ly += cell->ly();
hx += cell->lx();
hy += cell->ly();
rIndex = type->shapes[0].layer.rIndex;
} else if (iopin) {
iopin->getBounds(lx, ly, hx, hy, rIndex);
} else {
lx = INT_MAX;
ly = INT_MAX;
hx = INT_MIN;
hy = INT_MIN;
rIndex = -1;
}
}
utils::BoxT<double> Pin::getPinParentBBox() const {
double x, y, w, h;
if (this->cell != nullptr) { // cell pin
x = this->cell->lx();
y = this->cell->ly();
w = this->cell->width();
h = this->cell->height();
} else { // iopin
x = this->iopin->lx();
y = this->iopin->ly();
w = this->iopin->width();
h = this->iopin->height();
}
return utils::BoxT<double>(x, y, x + w, y + h);
}
utils::PointT<int> Pin::getPinParentCenter() const {
int cx, cy;
if (this->cell != nullptr) { // cell pin
cx = this->cell->cx();
cy = this->cell->cy();
} else { // iopin
cx = this->iopin->cx();
cy = this->iopin->cy();
}
return utils::PointT<int>(cx, cy);
}
/***** Pin Type *****/
void PinType::addShape(const Layer& layer, const int lx, const int ly, const int hx, const int hy) {
boundLX = min(boundLX, lx);
boundLY = min(boundLY, ly);
boundHX = max(boundHX, hx);
boundHY = max(boundHY, hy);
shapes.emplace_back(layer, lx, ly, hx, hy);
}
void PinType::getBounds(int& lx, int& ly, int& hx, int& hy) const {
lx = boundLX;
ly = boundLY;
hx = boundHX;
hy = boundHY;
}
bool PinType::operator<(const PinType& r) const {
if (boundLX < r.boundLX) {
return true;
}
if (boundLX > r.boundLX) {
return false;
}
if (boundLY < r.boundLY) {
return true;
}
if (boundLY > r.boundLY) {
return false;
}
if (boundHX < r.boundHX) {
return true;
}
if (boundHX > r.boundHX) {
return false;
}
if (boundHY < r.boundHY) {
return true;
} else if (boundHY == r.boundHY) {
if (shapes[0].layer.rIndex < r.shapes[0].layer.rIndex) {
return true;
} else {
return false;
}
} else {
return false;
}
}
bool PinType::comparePin(vector<PinType*> pins1, vector<PinType*> pins2) {
std::sort(pins1.begin(), pins1.end(), [](PinType* lhs, PinType* rhs) { return (*lhs < *rhs); });
std::sort(pins2.begin(), pins2.end(), [](PinType* lhs, PinType* rhs) { return (*lhs < *rhs); });
for (unsigned j = 0; j != pins1.size(); ++j) {
if (pins1[j]->_type != 'p' && pins1[j]->_type != 'g' && *pins1[j] != *pins2[j]) {
return false;
}
}
return true;
}

109
cpp_to_py/common/db/Pin.h Normal file
View File

@ -0,0 +1,109 @@
#pragma once
namespace db {
class PinType {
private:
string _name = "";
// i: input, o:output
char _direction = 'x';
// s: signal, c: clk, p: power, g: ground
char _type = 's';
public:
vector<Geometry> shapes;
int boundLX = INT_MAX;
int boundLY = INT_MAX;
int boundHX = INT_MIN;
int boundHY = INT_MIN;
PinType(const string& name, const char direction, const char type)
: _name(name), _direction(direction), _type(type) {}
const string& name() const { return _name; }
char direction() const { return _direction; }
char type() const { return _type; }
void direction(const char c) { _direction = c; }
void addShape(const Layer& layer, const int lx, const int ly, const int hx, const int hy);
void addShape(const Layer& layer, const int lx, const int ly) { addShape(layer, lx, ly, lx + 1, ly + 1); }
unsigned getW() const { return boundHX - boundLX; }
unsigned getH() const { return boundHY - boundLY; }
void getBounds(int& lx, int& ly, int& hx, int& hy) const;
bool operator==(const PinType& r) const {
return _type == r._type && boundLX == r.boundLX && boundLY == r.boundLY && boundHX == r.boundHX &&
boundHY == r.boundHY && shapes[0].layer.rIndex == r.shapes[0].layer.rIndex;
}
bool operator!=(const PinType& r) const { return !(*this == r); }
bool operator<(const PinType& r) const;
bool operator>(const PinType& r) const { return r < *this; }
bool operator>=(const PinType& r) const { return !(*this < r); }
bool operator<=(const PinType& r) const { return !(*this > r); }
static bool comparePin(vector<PinType*> pins1, vector<PinType*> pins2);
};
class IOPin {
protected:
string _netName = "";
public:
string name = "";
int x = INT_MIN;
int y = INT_MIN;
int _orient = 0; // 0:N, 1:W, 2:S, 3:E, 4:FN, 5:FW, 6:FS, 7:FE
PinType* type;
Pin* pin;
int gpdb_id = -1;
bool is_connected = false;
IOPin(const string& name = "", const string& netName = "", const char direction = 'x');
~IOPin();
const string& netName() const { return _netName; }
const int width() const { return this->type->getW(); }
const int height() const { return this->type->getH(); }
const int lx() const { return x; }
const int ly() const { return y; }
const int hx() const { return x + width(); }
const int hy() const { return y + height(); }
const int cx() const { return x + width() / 2; }
const int cy() const { return y + height() / 2; }
void getBounds(int& lx, int& ly, int& hx, int& hy, int& rIndex) const;
int orient() { return _orient; }
};
class PinSTA {
public:
char type; //'b': timing begin, 'e': timing end, 'i': intermediate
double capacitance;
double aat[4];
double rat[4];
double slack[4];
int nCriticalPaths[4];
PinSTA();
};
class Pin {
public:
Cell* cell = nullptr;
IOPin* iopin = nullptr;
Net* net = nullptr;
const PinType* type = nullptr;
int gpdb_id = -1;
int parentCellPinId = -1;
bool is_connected = false;
PinSTA* staInfo = nullptr;
Pin(const PinType* type = nullptr) : type(type) {}
Pin(Cell* cell, int i) : cell(cell), type(cell->ctype()->pins[i]), parentCellPinId(i) {}
Pin(IOPin* iopin) : iopin(iopin), type(iopin->type) {}
Pin(const Pin& pin) : cell(pin.cell), net(pin.net), type(pin.type) {}
void getPinCenter(int& x, int& y);
void getPinBounds(int& lx, int& ly, int& hx, int& hy, int& rIndex);
utils::BoxT<double> getPinParentBBox() const;
utils::PointT<int> getPinParentCenter() const;
}; // namespace db
} // namespace db

View File

@ -0,0 +1,12 @@
#include "Database.h"
using namespace db;
/***** Region *****/
void Region::addRect(const int xl, const int yl, const int xh, const int yh) {
rects.emplace_back(xl, yl, xh, yh);
lx = min(lx, xl);
ly = min(ly, yl);
hx = max(hx, xh);
hy = max(hy, yh);
}

View File

@ -0,0 +1,30 @@
#pragma once
namespace db {
class Region : public Rectangle {
private:
string _name = "";
// 'f' for fence, 'g' for guide
char _type = 'x';
public:
static const unsigned char InvalidRegion = 0xff;
unsigned char id = InvalidRegion;
double density = 0;
vector<string> members;
vector<Rectangle> rects;
Region(const string& name = "", const char type = 'x')
: Rectangle(INT_MAX, INT_MAX, INT_MIN, INT_MIN), _name(name), _type(type) {}
inline const string& name() const { return _name; }
inline char type() const { return _type; }
void addRect(const int xl, const int yl, const int xh, const int yh);
void resetRects() { sliceH(rects); }
};
} // namespace db

80
cpp_to_py/common/db/Row.h Normal file
View File

@ -0,0 +1,80 @@
#pragma once
namespace db {
class Row {
friend class Database;
private:
string _name = "";
string _macro = "";
int _x = 0;
int _y = 0;
unsigned _xNum = 0;
unsigned _yNum = 0;
bool _flip = false;
unsigned _xStep = 0;
unsigned _yStep = 0;
char _topPower = 'x';
char _botPower = 'x';
public:
std::vector<RowSegment> segments;
Row(const string& name, const string& macro, const int x, const int y, const unsigned xNum = 0, const unsigned yNum = 0, const bool flip = false, const unsigned xStep = 0, const unsigned yStep = 0)
: _name(name)
, _macro(macro)
, _x(x)
, _y(y)
, _xNum(xNum)
, _yNum(yNum)
, _flip(flip)
, _xStep(xStep)
, _yStep(yStep) {}
// return the left-most site x-index
int getSiteL(int dbLX, int siteW) const { return (_x - dbLX) / siteW; }
// return the right-most site x-index + 1
int getSiteR(int dbLX, int siteW) const { return getSiteL(dbLX, siteW) + _xNum; }
// return the bottom-most site y-index
int getSiteB(int dbLY, int siteH) const { return (_y - dbLY) / siteH; }
// return the top-most site y-index + 1
int getSiteT(int dbLY, int siteH) const { return getSiteB(dbLY, siteH) + _yNum; }
const string& name() const { return _name; }
const string& macro() const { return _macro; }
int x() const { return _x; }
int y() const { return _y; }
unsigned xNum() const { return _xNum; }
unsigned yNum() const { return _yNum; }
bool flip() const { return _flip; }
unsigned xStep() const { return _xStep; }
unsigned yStep() const { return _yStep; }
char topPower() const { return _topPower; }
char botPower() const { return _botPower; }
void x(const int value) { _x = value; }
void y(const int value) { _y = value; }
void xNum(const unsigned value) { _xNum = value; }
void yNum(const unsigned value) { _yNum = value; }
void flip(const bool value) { _flip = value; }
void xStep(const unsigned value) { _xStep = value; }
void yStep(const unsigned value) { _yStep = value; }
unsigned width() const { return _xStep * _xNum; }
bool isPowerValid() const { return (_topPower != _botPower) && (_topPower != 'x') && (_botPower != 'x'); }
void shiftX(const int value) { _x += value; }
void shiftY(const int value) { _y += value; }
void shrinkXNum(const int value) { _xNum -= value; }
void shrinkYNum(const int value) { _yNum -= value; }
};
class RowSegment {
public:
int x = 0;
int w = 0;
Region* region = nullptr;
};
} // namespace db

View File

@ -0,0 +1,25 @@
#pragma once
namespace db {
class SNet {
public:
string name;
vector<Geometry> shapes;
vector<Via> vias;
char type = 'x';
SNet(const string& name) : name(name) {}
template <class... Args>
void addShape(Args&&... args) {
shapes.emplace_back(args...);
}
template <class... Args>
void addVia(Args&&... args) {
vias.emplace_back(args...);
}
};
}

View File

@ -0,0 +1,23 @@
#include "Setting.h"
namespace db {
void Setting::reset() {
Format = "";
BookshelfVariety = "";
BookshelfAux = "";
BookshelfPl = "";
DefFile = "";
LefFile = "";
LefCell = "";
LefTech = "";
Constraints = "";
Verilog = "";
OutputFile = "";
liteMode = true;
random_place = false;
}
Setting setting;
} // namespace db

View File

@ -0,0 +1,41 @@
#pragma once
#include "common/common.h"
namespace db {
class Setting {
public:
// 1. SystemSetting
int numThreads = 1;
// 2. DBSetting
bool EdgeSpacing = true;
bool EnableFence = true;
bool EnablePG = true;
bool EnableIOPin = true;
bool liteMode = true;
bool random_place = false;
// 3. IOSetting
std::string Format = "";
std::string BookshelfVariety = "";
std::string BookshelfAux = "";
std::string BookshelfPl = "";
std::string DefFile = "";
std::string LefFile = "";
std::string LefCell = "";
std::string LefTech = "";
std::string Constraints = "";
std::string Verilog = "";
std::string Size = "";
std::string OutputFile = "";
void reset();
};
extern Setting setting;
} // namespace db

View File

@ -0,0 +1,24 @@
#pragma once
namespace db {
class Site {
private:
string _name = "";
string _siteClassName = "";
int _width = 0;
int _height = 0;
public:
Site(const string& name, const string& siteClassName, int width = 0, int height = 0)
: _name(name), _siteClassName(siteClassName), _width(width), _height(height) {}
const string& name() const { return _name; }
const string& siteClassName() const { return _siteClassName; }
int width() const { return _width; }
int height() const { return _height; }
void name(const string& value) { _name = value; }
void siteClassName(const string& value) { _siteClassName = value; }
void width(const int value) { _width = value; }
void height(const int value) { _height = value; }
};
} // namespace db

View File

@ -0,0 +1,78 @@
#include "Database.h"
using namespace db;
/***** SiteMap *****/
void SiteMap::initSiteMap(unsigned nx, unsigned ny) {
this->nx = nx;
this->ny = ny;
sites.resize(nx, vector<unsigned char>(ny));
regions.resize(nx, vector<unsigned char>(ny));
}
void SiteMap::getSiteBound(int x, int y, int &lx, int &ly, int &hx, int &hy) const {
lx = siteL + x * siteStepX;
ly = siteB + y * siteStepY;
hx = lx + siteStepX;
hy = ly + siteStepY;
}
void SiteMap::blockRegion(const unsigned x, const unsigned y) { regions[x][y] = Region::InvalidRegion; }
void SiteMap::setSites(int lx, int ly, int hx, int hy, unsigned char property, bool isContained) {
int slx = 0;
int sly = 0;
int shx = 0;
int shy = 0;
if (isContained) {
slx = binContainedL(lx, siteL, siteR, siteStepX);
sly = binContainedL(ly, siteB, siteT, siteStepY);
shx = binContainedR(hx, siteL, siteR, siteStepX);
shy = binContainedR(hy, siteB, siteT, siteStepY);
} else {
slx = binOverlappedL(lx, siteL, siteR, siteStepX);
sly = binOverlappedL(ly, siteB, siteT, siteStepY);
shx = binOverlappedR(hx, siteL, siteR, siteStepX);
shy = binOverlappedR(hy, siteB, siteT, siteStepY);
}
for (int x = slx; x <= shx; ++x) {
for (int y = sly; y <= shy; ++y) {
setSiteMap(x, y, property);
}
}
}
void SiteMap::unsetSites(int lx, int ly, int hx, int hy, unsigned char property) {
const int slx = binOverlappedL(lx, siteL, siteR, siteStepX);
const int sly = binOverlappedL(ly, siteB, siteT, siteStepY);
const int shx = binOverlappedR(hx, siteL, siteR, siteStepX);
const int shy = binOverlappedR(hy, siteB, siteT, siteStepY);
for (int x = slx; x <= shx; ++x) {
for (int y = sly; y <= shy; ++y) {
unsetSiteMap(x, y, property);
}
}
}
void SiteMap::blockRegion(int lx, int ly, int hx, int hy) {
int slx = binOverlappedL(lx, siteL, siteR, siteStepX);
int sly = binOverlappedL(ly, siteB, siteT, siteStepY);
int shx = binOverlappedR(hx, siteL, siteR, siteStepX);
int shy = binOverlappedR(hy, siteB, siteT, siteStepY);
for (int x = slx; x <= shx; x++) {
for (int y = sly; y <= shy; y++) {
setRegion(x, y, Region::InvalidRegion);
}
}
}
void SiteMap::setRegion(int lx, int ly, int hx, int hy, unsigned char region) {
int slx = binContainedL(lx, siteL, siteR, siteStepX);
int sly = binContainedL(ly, siteB, siteT, siteStepY);
int shx = binContainedR(hx, siteL, siteR, siteStepX);
int shy = binContainedR(hy, siteB, siteT, siteStepY);
for (int x = slx; x <= shx; x++) {
for (int y = sly; y <= shy; y++) {
setRegion(x, y, region);
}
}
}

View File

@ -0,0 +1,53 @@
#pragma once
namespace db {
class SiteMap {
private:
unsigned nx = 0;
unsigned ny = 0;
vector<vector<unsigned char>> sites;
vector<vector<unsigned char>> regions;
public:
static const char SiteBlocked = 1; // nothing can be placed in the site
static const char SiteM2Blocked = 2; // any part of it is blocked by M2 metal
static const char SiteM3Blocked = 4; // any part of it is blocked by M3 metal
static const char SiteM2BlockedIOPin = 8; // any part of it is blocked by M2 metal
int siteL, siteR;
int siteB, siteT;
int siteStepX, siteStepY;
int siteNX, siteNY;
unsigned long long nSites = 0;
unsigned long long nPlaceable = 0;
vector<long long> nRegionSites;
void initSiteMap(unsigned nx, unsigned ny);
void setSiteMap(const unsigned x, const unsigned y, const unsigned char property) { setBit(sites[x][y], property); }
inline void unsetSiteMap(const unsigned x, const unsigned y, const unsigned char property) {
unsetBit(sites[x][y], property);
}
inline bool getSiteMap(const unsigned x, const unsigned y, const unsigned char property) const {
return (getBit(sites[x][y], property) == property);
}
unsigned char getSiteMap(int x, int y) const { return sites[x][y]; }
void getSiteBound(int x, int y, int& lx, int& ly, int& hx, int& hy) const;
void blockRegion(const unsigned x, const unsigned y);
inline void setRegion(const unsigned x, const unsigned y, unsigned char region) { regions[x][y] = region; }
inline unsigned char getRegion(const unsigned x, const unsigned y) const { return regions[x][y]; }
void setSites(const int lx,
const int ly,
const int hx,
const int hy,
const unsigned char property,
const bool isContained = false);
void unsetSites(int lx, int ly, int hx, int hy, unsigned char property);
void blockRegion(int lx, int ly, int hx, int hy);
void setRegion(int lx, int ly, int hx, int hy, unsigned char region);
};
} // namespace db

59
cpp_to_py/common/db/Via.h Normal file
View File

@ -0,0 +1,59 @@
#pragma once
#include <set>
namespace db {
class Via {
public:
int x;
int y;
ViaType* type;
Via(ViaType* type, int x, int y) : x(x), y(y), type(type) {}
};
class ViaRule {
public:
~ViaRule() {
botLayer = nullptr;
cutLayer = nullptr;
topLayer = nullptr;
}
bool hasViaRule = false;
string name = "";
std::pair<int, int> cutSize = {-1, -1}; // (X, Y)
std::pair<int, int> cutSpacing = {-1, -1}; // (X, Y)
std::pair<int, int> botEnclosure = {-1, -1}; // (X, Y)
std::pair<int, int> topEnclosure = {-1, -1}; // (X, Y)
int numCutRows = -1;
int numCutCols = -1;
std::pair<int, int> originOffset = {-1, -1}; // (X, Y)
std::pair<int, int> botOffset = {-1, -1}; // (X, Y)
std::pair<int, int> topOffset = {-1, -1}; // (X, Y)
const Layer* botLayer = nullptr; // Route Layer
const Layer* cutLayer = nullptr; // Cut Layer
const Layer* topLayer = nullptr; // Route Layer
};
class ViaType {
private:
bool isDef_ = false;
public:
string name = "";
set<Geometry> rects;
ViaRule rule;
ViaType(const string& name = "", const bool isDef = false) : isDef_(isDef), name(name) {}
template <class... Args>
void addRect(Args&&... args) {
rects.emplace(args...);
}
void isDef(bool b) { isDef_ = b; }
bool isDef() const { return isDef_; }
};
} // namespace db

File diff suppressed because it is too large Load Diff

View File

@ -0,0 +1,88 @@
#include "common/db/Database.h"
using namespace db;
bool Database::readConstraints(const std::string& file) {
string buffer;
ifstream ifs(file.c_str());
if (!ifs.good()) {
printlog(LOG_ERROR, "cannot open constraint file: %s", file.c_str());
return false;
}
while (ifs >> buffer) {
unsigned equal = buffer.find("=");
string key = buffer.substr(0, equal);
if (key == "maximum_utilization") {
if (!maxDensity) {
unsigned unit = buffer.find("%");
string value = buffer.substr(equal + 1, unit - equal - 1);
maxDensity = atof(value.c_str()) / 100.0;
} else {
printlog(LOG_WARN, "use input max util %f", maxDensity);
}
} else if (key == "maximum_movement") {
if (!maxDisp) {
unsigned unit = buffer.find("rows");
string value = buffer.substr(equal + 1, unit - equal - 1);
maxDisp = atof(value.c_str());
} else {
printlog(LOG_WARN, "use input max disp %f", maxDisp);
}
}
}
ifs.close();
return true;
}
bool Database::readSize(const std::string& file) {
ifstream fs(file.c_str());
if (!fs.good()) {
printlog(LOG_ERROR, "cannot open size file: %s", file.c_str());
return false;
}
if (siteW == 0 && rows.size() > 0) {
siteW = rows[0]->xStep();
}
if (siteH == 0 && celltypes.size() > 0) {
siteH = INT_MAX;
for (CellType* celltype : celltypes) {
if (!celltype->stdcell) {
continue;
}
siteH = std::min(siteH, celltype->height);
}
}
if (siteW == 0 || siteH == 0) {
printlog(LOG_INFO, "not enough information to retrieve placement site size");
return false;
}
#ifndef NDEBUG
printlog(LOG_INFO, "reading %s", file.c_str());
#endif
std::set<CellType*> sized;
do {
string name;
int w, h;
fs >> name >> w >> h;
if (name != "") {
Cell* cell = getCell(name);
if (cell == NULL) {
printlog(LOG_ERROR, "cell not found : %s", name.c_str());
break;
} else {
CellType* celltype = cell->ctype();
if (sized.find(celltype) == sized.end()) {
celltype->width = w * siteW;
celltype->height = h * siteH;
sized.insert(celltype);
}
}
}
} while (!fs.eof());
fs.close();
return true;
}

File diff suppressed because it is too large Load Diff

View File

@ -0,0 +1,161 @@
#include "common/db/Database.h"
using namespace db;
bool isVerilogSymbol(unsigned char c) {
static char symbols[256] = {0};
static bool inited = false;
if (!inited) {
symbols[(int)'('] = 1;
symbols[(int)')'] = 1;
symbols[(int)','] = 1;
symbols[(int)'.'] = 1;
symbols[(int)':'] = 1;
symbols[(int)';'] = 1;
// symbols[(int)'/'] = 1;
symbols[(int)'#'] = 1;
symbols[(int)'['] = 1;
symbols[(int)']'] = 1;
symbols[(int)'{'] = 1;
symbols[(int)'}'] = 1;
symbols[(int)'*'] = 1;
symbols[(int)'\"'] = 1;
symbols[(int)'\\'] = 1;
symbols[(int)' '] = 2;
symbols[(int)'\t'] = 2;
symbols[(int)'\n'] = 2;
symbols[(int)'\r'] = 2;
inited = true;
}
return symbols[(int)c] != 0;
}
bool readVerilogLine(istream &is, vector<string> &tokens) {
tokens.clear();
string line;
while (is && tokens.empty()) {
// read next line in
getline(is, line);
char token[1024] = {0};
int lineLen = (int)line.size();
int tokenLen = 0;
for (int i = 0; i < lineLen; i++) {
char c = line[i];
if (isVerilogSymbol(c)) {
if (tokenLen > 0) {
token[tokenLen] = (char)0;
tokens.push_back(string(token));
token[0] = (char)0;
tokenLen = 0;
}
if (c == ';') {
tokens.push_back(";");
}
} else {
token[tokenLen++] = c;
if (tokenLen > 1024) {
// TODO: unhandled error
tokens.clear();
return false;
}
}
}
// line finished, something else in token
if (tokenLen > 0) {
token[tokenLen] = (char)0;
tokens.push_back(string(token));
tokenLen = 0;
}
}
return !tokens.empty();
}
bool Database::readVerilog(const std::string &file) {
ifstream fs(file.c_str());
if (!fs.good()) {
printlog(LOG_ERROR, "cannot open verilog file: %s", file.c_str());
return false;
}
vector<string> tokens;
const int StatusNone = 0;
const int StatusModule = 1;
const int StatusInput = 2;
const int StatusOutput = 3;
const int StatusWire = 4;
const int StatusGate = 5;
int status = StatusNone;
bool finished = true;
// loop every line in
while (readVerilogLine(fs, tokens)) {
if (tokens[0] == "//") {
continue;
}
if (finished) {
if (tokens[0] == "endmodule") {
status = StatusNone;
tokens.erase(tokens.begin());
finished = true;
} else if (tokens[0] == "module") {
status = StatusModule;
tokens.erase(tokens.begin());
finished = false;
} else if (tokens[0] == "input") {
status = StatusInput;
tokens.erase(tokens.begin());
finished = false;
} else if (tokens[0] == "output") {
status = StatusOutput;
tokens.erase(tokens.begin());
finished = false;
} else if (tokens[0] == "wire") {
status = StatusWire;
tokens.erase(tokens.begin());
finished = false;
} else {
status = StatusGate;
finished = false;
}
}
if (tokens.back() == ";") {
tokens.pop_back();
finished = true;
}
if (status == StatusModule) {
// ignore
} else if (status == StatusInput || status == StatusOutput) {
for (unsigned i = 0; i != tokens.size(); ++i) {
string pinName(tokens[i]);
IOPin *iopin = getIOPin(pinName);
if (!iopin) {
printlog(LOG_ERROR, "io pin not found: %s", pinName.c_str());
}
Net *net = addNet(pinName);
net->addPin(iopin->pin);
}
} else if (status == StatusWire) {
for (int i = 0; i < (int)tokens.size(); i++) {
string netName(tokens[i]);
this->addNet(netName);
}
} else if (status == StatusGate) {
string cellName(tokens[1]);
Cell *cell = this->getCell(cellName);
int nPins = (tokens.size() - 2) / 2;
for (int i = 0; i < nPins; i++) {
string pinName(tokens[2 + i * 2]);
string netName(tokens[3 + i * 2]);
Pin *pin = cell->pin(pinName);
Net *net = this->getNet(netName);
net->addPin(pin);
}
}
}
fs.close();
return true;
}

File diff suppressed because it is too large Load Diff

View File

@ -0,0 +1,400 @@
//
// Some class templates for geometry primitives (point, interval, box)
//
#pragma once
#include <cassert>
#include <cmath>
#include <iostream>
#include <limits>
#include <vector>
#include <algorithm>
namespace utils {
// Point template
template <typename T>
class PointT {
public:
T x, y;
PointT(T xx = std::numeric_limits<T>::has_infinity ? std::numeric_limits<T>::infinity()
: std::numeric_limits<T>::max(),
T yy = std::numeric_limits<T>::has_infinity ? std::numeric_limits<T>::infinity()
: std::numeric_limits<T>::max())
: x(xx), y(yy) {}
bool IsValid() { return *this != PointT(); }
// Operators
const T& operator[](const unsigned d) const {
assert(d == 0 || d == 1);
return (d == 0 ? x : y);
}
T& operator[](const unsigned d) {
assert(d == 0 || d == 1);
return (d == 0 ? x : y);
}
PointT operator+(const PointT& rhs) { return PointT(x + rhs.x, y + rhs.y); }
PointT operator/(T divisor) { return PointT(x / divisor, y / divisor); }
PointT& operator+=(const PointT& rhs) {
x += rhs.x;
y += rhs.y;
return *this;
}
PointT& operator-=(const PointT& rhs) {
x -= rhs.x;
y -= rhs.y;
return *this;
}
bool operator==(const PointT& rhs) const { return x == rhs.x && y == rhs.y; }
bool operator!=(const PointT& rhs) const { return !(*this == rhs); }
friend inline std::ostream& operator<<(std::ostream& os, const PointT& pt) {
os << "(" << pt.x << ", " << pt.y << ")";
return os;
}
};
// L-1 (Manhattan) distance between points
template <typename T>
inline T Dist(const PointT<T>& pt1, const PointT<T>& pt2) {
return std::abs(pt1.x - pt2.x) + std::abs(pt1.y - pt2.y);
}
// L-2 (Euclidean) distance between points
template <typename T>
inline double L2Dist(const PointT<T>& pt1, const PointT<T>& pt2) {
return std::sqrt(std::pow(pt1.x - pt2.x, 2) + std::pow(pt1.y - pt2.y, 2));
}
// L-inf distance between points
template <typename T>
inline T LInfDist(const PointT<T>& pt1, const PointT<T>& pt2) {
return std::max(std::abs(pt1.x - pt2.x), std::abs(pt1.y - pt2.y));
}
// Interval template
template <typename T>
class IntervalT {
public:
T low, high;
template <typename... Args>
IntervalT(Args... params) {
Set(params...);
}
// Setters
void Set() {
low = std::numeric_limits<T>::has_infinity ? std::numeric_limits<T>::infinity() : std::numeric_limits<T>::max();
high = std::numeric_limits<T>::has_infinity ? -std::numeric_limits<T>::infinity()
: std::numeric_limits<T>::lowest();
}
void Set(T val) {
low = val;
high = val;
}
void Set(T lo, T hi) {
low = lo;
high = hi;
}
// Getters
T center() const { return (high + low) / 2; }
T range() const { return high - low; }
// Update
// Update() is always safe, FastUpdate() assumes existing values
void Update(T newVal) {
if (newVal < low) low = newVal;
if (newVal > high) high = newVal;
}
void FastUpdate(T newVal) {
if (newVal < low)
low = newVal;
else if (newVal > high)
high = newVal;
}
// Two types of intervals: 1. normal, 2. degenerated (i.e., point)
// is valid interval (i.e., valid closed interval)
bool IsValid() const { return low <= high; }
// is strictly valid interval (excluding degenerated ones, i.e., valid open interval)
bool IsStrictValid() const { return low < high; }
// Geometric Query/Update
// interval/range of union (not union of intervals)
IntervalT UnionWith(const IntervalT& rhs) const {
if (!IsValid())
return rhs;
else if (!rhs.IsValid())
return *this;
else
return IntervalT(std::min(low, rhs.low), std::max(high, rhs.high));
}
// may return an invalid interval (as empty intersection)
IntervalT IntersectWith(const IntervalT& rhs) const {
return IntervalT(std::max(low, rhs.low), std::min(high, rhs.high));
}
bool HasIntersectWith(const IntervalT& rhs) const { return IntersectWith(rhs).IsValid(); }
bool HasStrictIntersectWith(const IntervalT& rhs) const { return IntersectWith(rhs).IsStrictValid(); }
// Parallel run length between intervals
T ParaRunLength(const IntervalT& rhs) const { return IntersectWith(rhs).range(); }
// contain a val
bool Contain(int val) const { return val >= low && val <= high; }
bool StrictlyContain(int val) const { return val > low && val < high; }
// get nearest point(s) to val (assume valid intervals)
T GetNearestPointTo(T val) const {
if (val <= low) {
return low;
} else if (val >= high) {
return high;
} else {
return val;
}
}
IntervalT GetNearestPointsTo(IntervalT val) const {
if (val.high <= low) {
return {low};
} else if (val.low >= high) {
return {high};
} else {
return IntersectWith(val);
}
}
void ShiftBy(const T& rhs) {
low += rhs;
high += rhs;
}
// Operators
bool operator==(const IntervalT& rhs) const {
return (!IsValid() && !rhs.IsValid()) || (low == rhs.low && high == rhs.high);
}
bool operator!=(const IntervalT& rhs) const { return !(*this == rhs); }
friend inline std::ostream& operator<<(std::ostream& os, const IntervalT<T>& interval) {
os << "(" << interval.low << ", " << interval.high << ")";
return os;
}
};
// Distance between intervals/points (assume valid intervals)
template <typename T>
inline T Dist(const IntervalT<T>& intvl, const T val) {
return std::abs(intvl.GetNearestPointTo(val) - val);
}
template <typename T>
inline T Dist(const IntervalT<T>& int1, const IntervalT<T>& int2) {
if (int1.high <= int2.low) {
return int2.low - int1.high;
} else if (int1.low >= int2.high) {
return int1.low - int2.high;
} else {
return 0;
}
}
// Box template
template <typename T>
class BoxT {
public:
IntervalT<T> x, y;
template <typename... Args>
BoxT(Args... params) {
Set(params...);
}
// Setters
T& lx() { return x.low; }
T& ly() { return y.low; }
T& hy() { return y.high; }
T& hx() { return x.high; }
IntervalT<T>& operator[](unsigned i) {
assert(i == 0 || i == 1);
return (i == 0) ? x : y;
}
void Set() {
x.Set();
y.Set();
}
void Set(T xVal, T yVal) {
x.Set(xVal);
y.Set(yVal);
}
void Set(const PointT<T>& pt) { Set(pt.x, pt.y); }
void Set(T lx, T ly, T hx, T hy) {
x.Set(lx, hx);
y.Set(ly, hy);
}
void Set(const IntervalT<T>& xRange, const IntervalT<T>& yRange) {
x = xRange;
y = yRange;
}
void Set(const PointT<T>& low, const PointT<T>& high) { Set(low.x, low.y, high.x, high.y); }
void Set(const BoxT<T>& box) { Set(box.x, box.y); }
// Two types of boxes: normal & degenerated (line or point)
// is valid box
bool IsValid() const { return x.IsValid() && y.IsValid(); }
// is strictly valid box (excluding degenerated ones)
bool IsStrictValid() const { return x.IsStrictValid() && y.IsStrictValid(); } // tighter
// Getters
T lx() const { return x.low; }
T ly() const { return y.low; }
T hy() const { return y.high; }
T hx() const { return x.high; }
T cx() const { return x.center(); }
T cy() const { return y.center(); }
T width() const { return x.range(); }
T height() const { return y.range(); }
T hp() const { return width() + height(); } // half perimeter
T area() const { return width() * height(); }
const IntervalT<T>& operator[](unsigned i) const {
assert(i == 0 || i == 1);
return (i == 0) ? x : y;
}
// Update() is always safe, FastUpdate() assumes existing values
void Update(T xVal, T yVal) {
x.Update(xVal);
y.Update(yVal);
}
void FastUpdate(T xVal, T yVal) {
x.FastUpdate(xVal);
y.FastUpdate(yVal);
}
void Update(const PointT<T>& pt) { Update(pt.x, pt.y); }
void FastUpdate(const PointT<T>& pt) { FastUpdate(pt.x, pt.y); }
// Geometric Query/Update
BoxT UnionWith(const BoxT& rhs) const { return {x.UnionWith(rhs.x), y.UnionWith(rhs.y)}; }
BoxT IntersectWith(const BoxT& rhs) const { return {x.IntersectWith(rhs.x), y.IntersectWith(rhs.y)}; }
bool HasIntersectWith(const BoxT& rhs) const { return IntersectWith(rhs).IsValid(); }
bool HasStrictIntersectWith(const BoxT& rhs) const { return IntersectWith(rhs).IsStrictValid(); } // tighter
bool Contain(const PointT<T>& pt) const { return x.Contain(pt.x) && y.Contain(pt.y); }
bool StrictlyContain(const PointT<T>& pt) const { return x.StrictlyContain(pt.x) && y.StrictlyContain(pt.y); }
PointT<T> GetNearestPointTo(const PointT<T>& pt) { return {x.GetNearestPointTo(pt.x), y.GetNearestPointTo(pt.y)}; }
BoxT GetNearestPointsTo(BoxT val) const { return {x.GetNearestPointsTo(val.x), y.GetNearestPointsTo(val.y)}; }
void ShiftBy(const PointT<T>& rhs) {
x.ShiftBy(rhs.x);
y.ShiftBy(rhs.y);
}
bool operator==(const BoxT& rhs) const { return (x == rhs.x) && (y == rhs.y); }
bool operator!=(const BoxT& rhs) const { return !(*this == rhs); }
friend inline std::ostream& operator<<(std::ostream& os, const BoxT<T>& box) {
os << "[x: " << box.x << ", y: " << box.y << "]";
return os;
}
};
// L-1 (Manhattan) distance between boxes/points (assume valid boxes)
template <typename T>
inline T Dist(const BoxT<T>& box, const PointT<T>& point) {
return Dist(box.x, point.x) + Dist(box.y, point.y);
}
template <typename T>
inline T Dist(const BoxT<T>& box1, const BoxT<T>& box2) {
return Dist(box1.x, box2.x) + Dist(box1.y, box2.y);
}
// L-2 (Euclidean) distance between boxes
template <typename T>
inline double L2Dist(const BoxT<T>& box1, const BoxT<T>& box2) {
return std::sqrt(std::pow(Dist(box1.x, box2.x), 2) + std::pow(Dist(box1.y, box2.y), 2));
}
// L-Inf (max) distance between boxes
template <typename T>
inline T LInfDist(const BoxT<T>& box1, const BoxT<T>& box2) {
return std::max(Dist(box1.x, box2.x), Dist(box1.y, box2.y));
}
// Parallel run length between boxes
template <typename T>
inline T ParaRunLength(const BoxT<T>& box1, const BoxT<T>& box2) {
return std::max(box1.x.ParaRunLength(box2.x), box1.y.ParaRunLength(box2.y));
}
// Merge/stitch overlapped rectangles along mergeDir
// mergeDir: 0 for x/vertical, 1 for y/horizontal
// use BoxT instead of T & BoxT<T> to make it more general
template <typename BoxT>
void MergeRects(std::vector<BoxT>& boxes, int mergeDir) {
int boundaryDir = 1 - mergeDir;
std::sort(boxes.begin(), boxes.end(), [&](const BoxT& lhs, const BoxT& rhs) {
return lhs[boundaryDir].low < rhs[boundaryDir].low ||
(lhs[boundaryDir].low == rhs[boundaryDir].low && lhs[mergeDir].low < rhs[mergeDir].low);
});
std::vector<BoxT> mergedBoxes;
mergedBoxes.push_back(boxes.front());
for (int i = 1; i < boxes.size(); ++i) {
auto& lastBox = mergedBoxes.back();
auto& slicedBox = boxes[i];
if (slicedBox[boundaryDir] == lastBox[boundaryDir] &&
slicedBox[mergeDir].low <= lastBox[mergeDir].high) { // aligned and intersected
lastBox[mergeDir] = lastBox[mergeDir].UnionWith(slicedBox[mergeDir]);
} else { // neither misaligned not seperated
mergedBoxes.push_back(slicedBox);
}
}
boxes = move(mergedBoxes);
}
// Slice polygons along sliceDir
// sliceDir: 0 for x/vertical, 1 for y/horizontal
// assume no degenerated case
template <typename T>
void SlicePolygons(std::vector<BoxT<T>>& boxes, int sliceDir) {
// Line sweep in sweepDir = 1 - sliceDir
// Suppose sliceDir = y and sweepDir = x (sweep from left to right)
// Not scalable impl (brute force interval query) but fast for small case
if (boxes.size() <= 1) return;
// sort slice lines in sweepDir
int sweepDir = 1 - sliceDir;
std::vector<T> locs;
for (const auto& box : boxes) {
locs.push_back(box[sweepDir].low);
locs.push_back(box[sweepDir].high);
}
std::sort(locs.begin(), locs.end());
locs.erase(std::unique(locs.begin(), locs.end()), locs.end());
// slice each box
std::vector<BoxT<T>> slicedBoxes;
for (const auto& box : boxes) {
BoxT<T> slicedBox = box;
auto itLoc = std::lower_bound(locs.begin(), locs.end(), box[sweepDir].low);
auto itEnd = std::upper_bound(itLoc, locs.end(), box[sweepDir].high);
while ((itLoc + 1) != itEnd) {
slicedBox[sweepDir].Set(*itLoc, *(itLoc + 1));
slicedBoxes.push_back(slicedBox);
++itLoc;
}
}
boxes = move(slicedBoxes);
// merge overlapped boxes along slice dir
MergeRects(boxes, sliceDir);
// stitch boxes along sweep dir
MergeRects(boxes, sweepDir);
}
template <typename T>
class SegmentT : public BoxT<T> {
public:
using BoxT<T>::BoxT;
T length() const { return BoxT<T>::hp(); }
bool IsRectilinear() const { return BoxT<T>::x() == 0 || BoxT<T>::y() == 0; }
};
} // namespace utils

View File

@ -0,0 +1,150 @@
#include "log.h"
#include "log_level.h"
#include <iomanip>
#include <sstream>
#if defined(__unix__)
#include <sys/resource.h>
#include <unistd.h>
#elif defined(_WIN32)
#include <windows.h>
#include <psapi.h>
#endif
namespace utils {
timer::timer() { start(); }
void timer::start() { _start = clock::now(); }
double timer::elapsed() const { return std::chrono::duration<double>(clock::now() - _start).count(); }
std::string timer::get_time_stamp() {
std::stringstream ss;
ss << "[" << std::setprecision(3) << std::setw(8) << std::fixed << this->elapsed() << "] ";
return ss.str();
}
timer tstamp;
bool verbose_parser_log = false;
PrintfLogger logger;
double mem_use::get_current() {
#if defined(__unix__)
long rss = 0L;
FILE* fp = NULL;
if ((fp = fopen("/proc/self/statm", "r")) == NULL) {
return 0.0; /* Can't open? */
}
if (fscanf(fp, "%*s%ld", &rss) != 1) {
fclose(fp);
return 0.0; /* Can't read? */
}
fclose(fp);
return rss * sysconf(_SC_PAGESIZE) / 1048576.0;
#elif defined(_WIN32)
PROCESS_MEMORY_COUNTERS info;
GetProcessMemoryInfo(GetCurrentProcess(), &info, sizeof(info));
return info.WorkingSetSize / 1048576.0;
#else
return 0.0; // unknown
#endif
}
double mem_use::get_peak() {
#if defined(__unix__)
struct rusage rusage;
getrusage(RUSAGE_SELF, &rusage);
return rusage.ru_maxrss / 1024.0;
#elif defined(_WIN32)
PROCESS_MEMORY_COUNTERS info;
GetProcessMemoryInfo(GetCurrentProcess(), &info, sizeof(info));
return info.PeakWorkingSetSize / 1048576.0;
#else
return 0.0; // unknown
#endif
}
void printlog(int log_level, const char* format, ...) {
if (!verbose_parser_log) {
return;
}
if (log_level >= GLOBAL_LOG_LEVEL) {
std::string curr_log = tstamp.get_time_stamp();
if (log_level > LOG_INFO) {
curr_log += log_level_ANSI_color(log_level);
}
std::cout << curr_log;
va_list ap;
va_start(ap, format);
vfprintf(stdout, format, ap);
printf("\n");
fflush(stdout);
}
}
void assert_msg(bool condition, const char* format, ...) {
if (!condition) {
std::cerr << "Assertion failure: ";
va_list ap;
va_start(ap, format);
vfprintf(stdout, format, ap);
std::cerr << std::endl;
std::abort();
}
return;
}
std::string log_level_ANSI_color(int& log_level) {
std::string color_string;
switch (log_level) {
case LOG_NOTICE:
color_string = "\033[1;34mNotice\033[0m: ";
break;
case LOG_WARN:
color_string = "\033[1;93mWarning\033[0m: ";
break;
case LOG_ERROR:
color_string = "\033[1;31mError\033[0m: ";
break;
case LOG_FATAL:
color_string = "\033[1;41;97m F A T A L \033[0m: ";
break;
case LOG_OK:
color_string = "\033[1;32mOK\033[0m: ";
break;
default:
break;
}
return color_string;
}
// void PrintfLogger::setup_logger(argparse::ArgumentParser parser) {
// std::filesystem::path current_dir(std::filesystem::current_path());
// // std::filesystem::path result_dir(parser.get<std::string>("result_dir"));
// std::filesystem::path result_dir(st::setting.result_dir);
// std::filesystem::path exp_id(parser.get<std::string>("exp_id"));
// std::filesystem::path log_dir(parser.get<std::string>("log_dir"));
// std::filesystem::path log_name(parser.get<std::string>("log_name"));
// std::filesystem::path res_root = current_dir / result_dir / exp_id;
// std::filesystem::path log_root = res_root / log_dir;
// // std::filesystem::path log_file_path = log_root / log_name;
// std::string log_file_path = (log_root / log_name).string();
// if (!std::filesystem::exists(log_root)) {
// std::filesystem::create_directories(log_root);
// }
// if (write_log) {
// f = fopen(log_file_path.c_str(), "w");
// this->warning("Logging into file would affect the elapsed time. Disable it before submission.");
// if (f == NULL) {
// this->error("Cannot open logfile {}", log_file_path);
// exit(1);
// }
// }
// }
} // namespace utils

View File

@ -0,0 +1,144 @@
//
// Some logging utilities
// 1. "log() << ..." will show a time stamp first
// 2. "print(a, b, c)" is python-like print for any a, b, c that has operator<< overloaded. For example,
// int a = 10;
// double b = 3.14;
// std::string c = "Gengjie";
// print(a, b, c);
// This code piece will show "10 3.14 Gengjie".
// 3. "assert_msg(condition, format, ...)" is python-like assert
//
#pragma once
#include <chrono>
#include <cstdarg>
#include <fstream>
#include <iostream>
#include <memory>
#include <string>
#include "log_level.h"
// #include "argparse.hpp" // need C++20
namespace utils {
extern bool verbose_parser_log;
// 1. Timer
class timer {
using clock = std::chrono::high_resolution_clock;
private:
clock::time_point _start;
public:
timer();
void start();
double elapsed() const; // seconds
std::string get_time_stamp();
};
extern timer tstamp;
// 2. Memory
class mem_use {
public:
static double get_current(); // MB
static double get_peak(); // MB
};
// 3. Easy print
// print(a, b, c)
inline void print() { std::cout << std::endl; }
template <typename T, typename... TAIL>
void print(const T& t, TAIL... tail) {
std::cout << t << ' ';
print(tail...);
}
// "printlog(LOG_LEVEL, a, b, c...)" puts a time stamp in beginning
// try to make code compatible with old printlog(int level, const char *format, ...)
void printlog(int level, const char* format, ...);
void assert_msg(bool condition, const char* format, ...);
std::string log_level_ANSI_color(int& log_level);
// 4. PrintfLogger format and write to file
// Python kind logger, support printf kind format (can be replaced by fmt library)
class PrintfLogger {
static constexpr bool write_log = false;
FILE* f;
public:
// void setup_logger(argparse::ArgumentParser parser);
~PrintfLogger() {
if (f != NULL) fclose(f);
}
template <typename... Args>
void log(int log_level, const char* format, Args&&... args) {
if (!verbose_parser_log) {
return;
}
if (log_level >= GLOBAL_LOG_LEVEL) {
std::string curr_log = tstamp.get_time_stamp();
if (log_level > LOG_INFO) {
curr_log += log_level_ANSI_color(log_level);
}
std::cout << curr_log;
printf(format, args...);
printf("\n");
fflush(stdout);
if (write_log) {
fprintf(f, "%s", curr_log.c_str());
fprintf(f, format, args...);
fprintf(f, "\n");
fflush(f);
}
}
}
template <typename... Args>
void debug(const char* format, Args&&... args) {
log(LOG_DEBUG, format, args...);
};
template <typename... Args>
void verbose(const char* format, Args&&... args) {
log(LOG_VERBOSE, format, args...);
};
template <typename... Args>
void info(const char* format, Args&&... args) {
log(LOG_INFO, format, args...);
};
template <typename... Args>
void notice(const char* format, Args&&... args) {
log(LOG_NOTICE, format, args...);
};
template <typename... Args>
void warning(const char* format, Args&&... args) {
log(LOG_WARN, format, args...);
};
template <typename... Args>
void error(const char* format, Args&&... args) {
log(LOG_ERROR, format, args...);
};
template <typename... Args>
void fatal(const char* format, Args&&... args) {
log(LOG_FATAL, format, args...);
};
template <typename... Args>
void ok(const char* format, Args&&... args) {
log(LOG_OK, format, args...);
};
};
extern PrintfLogger logger;
} // namespace utils

View File

@ -0,0 +1,21 @@
#pragma once
#include <iostream>
namespace utils::log_level {
///////////////LOGGING FUNCTIONS///////////////////////
//--log type
inline constexpr int LOG_DEBUG = 0; // 0
inline constexpr int LOG_VERBOSE = 1; // 1
inline constexpr int LOG_INFO = 2; // 2
inline constexpr int LOG_NOTICE = 3; // 3
inline constexpr int LOG_WARN = 4; // 4
inline constexpr int LOG_ERROR = 5; // 5
inline constexpr int LOG_FATAL = 6; // 6
inline constexpr int LOG_OK = 7; // 7
inline int GLOBAL_LOG_LEVEL = LOG_INFO; // change verbose level in Setting.h
} // namespace utils::log_level
using namespace utils::log_level;

File diff suppressed because it is too large Load Diff

View File

@ -0,0 +1,94 @@
#pragma once
#include "geo.h"
#include "log.h"
#include "robin_hood.h"
#include <algorithm>
#include <climits>
#include <cstdarg>
#include <string>
//#define isNaN(x) ((x)!=(x))
namespace utils {
template <typename T>
inline void minmax(const T v1, const T v2, T &minv, T &maxv) {
if (v1 < v2) {
minv = v1;
maxv = v2;
} else {
minv = v2;
maxv = v1;
}
}
template <typename T>
inline void bounds(const T v, T &minv, T &maxv) {
if (v < minv) {
minv = v;
} else if (v > maxv) {
maxv = v;
}
}
} // namespace utils
// function for contain / overlap
inline int binContainedL(int lx, int blx, int bhx, int binw) { return (std::max(blx, lx) - blx + binw - 1) / binw; }
inline int binContainedR(int hx, int blx, int bhx, int binw) { return (std::min(bhx, hx) - blx) / binw - 1; }
inline int binOverlappedL(int lx, int blx, int bhx, int binw) { return (std::max(blx, lx) - blx) / binw; }
inline int binOverlappedR(int hx, int blx, int bhx, int binw) {
return (std::min(bhx, hx) - blx + binw - 1) / binw - 1;
}
////////////INTEGER PACKING/////////////////
inline long long packInt(const int x, const int y) { return ((long)x) << 32 | ((long)y); }
inline void unpackInt(int &x, int &y, long long i) {
x = (int)(i >> 32);
y = (int)(i & 0xffffffff);
}
inline long long packCoor(int x, int y) {
int ix = x + (INT_MAX >> 2);
int iy = y + (INT_MAX >> 2);
return packInt(ix, iy);
}
inline void unpackCoor(int &x, int &y, long long i) {
unpackInt(x, y, i);
x -= (INT_MAX >> 2);
y -= (INT_MAX >> 2);
}
////////////FLOATING-POINT RANDOM NUMBER//////////////
inline int getrand(int lo, int hi) { return (rand() % (hi - lo + 1)) + lo; }
inline double getrand(double lo, double hi) { return (((double)rand() / (double)RAND_MAX) * (hi - lo) + lo); }
//////////////BIT MANIPULATION///////////////////////
template <typename T>
inline void setBit(T &val, T bit) {
val |= bit;
}
template <typename T>
inline void unsetBit(T &val, T bit) {
val &= (~bit);
}
template <typename T>
inline void toggleBit(T &val, T bit) {
val ^= bit;
}
template <typename T>
inline bool isSetBit(T val, T bit) {
return (val & bit) > 0;
}
template <typename T>
inline T getBit(T val, T bit) {
return val & bit;
}
template <typename T>
inline T rect_overlap_area(T alx, T aly, T ahx, T ahy, T blx, T bly, T bhx, T bhy){
if(alx>=bhx || ahx<=blx || aly>=bhy || ahy<=bly){
return 0.0;
}
return (std::min(ahx,bhx) - std::max(alx,blx)) * (std::min(ahy,bhy) - std::max(aly,bly));
}

View File

@ -0,0 +1 @@
from . import *

View File

@ -0,0 +1,9 @@
set(TARGET_NAME dct_cuda)
add_pytorch_extension(${TARGET_NAME}
${TARGET_NAME}.cpp
${TARGET_NAME}_kernel.cu)
install(TARGETS
${TARGET_NAME}
DESTINATION ${XPLACE_LIB_DIR})

View File

@ -0,0 +1,29 @@
#include <torch/extension.h>
void dct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf);
void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf);
void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf);
void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf);
void dct2_fft2_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
dct2_fft2_forward_cuda(x, expkM, expkN, out, buf);
}
void idct2_fft2_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
idct2_fft2_forward_cuda(x, expkM, expkN, out, buf);
}
void idct_idxst_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
idct_idxst_forward_cuda(x, expkM, expkN, out, buf);
}
void idxst_idct_forward(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
idxst_idct_forward_cuda(x, expkM, expkN, out, buf);
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
m.def("dct2_fft2", &dct2_fft2_forward, "DCT2 FFT2D (CUDA)");
m.def("idct2_fft2", &idct2_fft2_forward, "IDCT2 FFT2D (CUDA)");
m.def("idct_idxst", &idct_idxst_forward, "IDCT IDXST FFT2D (CUDA)");
m.def("idxst_idct", &idxst_idct_forward, "IDXST IDCT FFT2D (CUDA)");
}

View File

@ -0,0 +1,802 @@
/**
* @file dct2_fft2_cuda_kernel.cu
* @author Zixuan Jiang, Jiaqi Gu
* @date Apr 2019
* @brief Refernece: Byeong Lee, "A new algorithm to compute the discrete cosine Transform,"
* in IEEE Transactions on Acoustics, Speech, and Signal Processing, vol. 32, no. 6, pp. 1243-1245, December 1984.
* The preprocess and postprocess of 2d dct and 2d idct are discussed in the original paper.
* idct(idxst(x)) and idxst(idct(x)) are similar to the idct2d(x),
* except tiny modifications on preprocessing and postprocessing
*/
#include <float.h>
#include <math.h>
#include <torch/extension.h>
#include <ATen/cuda/CUDAContext.h>
#include "cuda_runtime.h"
// ComplexType
template <typename T>
struct ComplexType {
T x;
T y;
__host__ __device__ ComplexType() {
x = 0;
y = 0;
}
__host__ __device__ ComplexType(T real, T imag) {
x = real;
y = imag;
}
__host__ __device__ ~ComplexType() {}
};
template <typename T>
inline __host__ __device__ ComplexType<T> complexMul(const ComplexType<T> &x, const ComplexType<T> &y) {
ComplexType<T> res;
res.x = x.x * y.x - x.y * y.y;
res.y = x.x * y.y + x.y * y.x;
return res;
}
template <typename T>
inline __host__ __device__ T RealPartOfMul(const ComplexType<T> &x, const ComplexType<T> &y) {
return x.x * y.x - x.y * y.y;
}
template <typename T>
inline __host__ __device__ T ImaginaryPartOfMul(const ComplexType<T> &x, const ComplexType<T> &y) {
return x.x * y.y + x.y * y.x;
}
template <typename T>
inline __host__ __device__ ComplexType<T> complexAdd(const ComplexType<T> &x, const ComplexType<T> &y) {
ComplexType<T> res;
res.x = x.x + y.x;
res.y = x.y + y.y;
return res;
}
template <typename T>
inline __host__ __device__ ComplexType<T> complexSubtract(const ComplexType<T> &x, const ComplexType<T> &y) {
ComplexType<T> res;
res.x = x.x - y.x;
res.y = x.y - y.y;
return res;
}
template <typename T>
inline __host__ __device__ ComplexType<T> complexConj(const ComplexType<T> &x) {
ComplexType<T> res;
res.x = x.x;
res.y = -x.y;
return res;
}
template <typename T>
inline __host__ __device__ ComplexType<T> complexMulConj(const ComplexType<T> &x, const ComplexType<T> &y) {
ComplexType<T> res;
res.x = x.x * y.x - x.y * y.y;
res.y = -(x.x * y.y + x.y * y.x);
return res;
}
#define TPB (16)
inline __device__ int INDEX(const int hid, const int wid, const int N) { return (hid * N + wid); }
// dct2_fft2
template <typename T>
__global__ void dct2dPreprocess(const T *x, T *y, const int M, const int N, const int halfN) {
const int wid = blockDim.x * blockIdx.x + threadIdx.x;
const int hid = blockDim.y * blockIdx.y + threadIdx.y;
if (hid < M && wid < N) {
int index;
int cond = (((hid & 1) == 0) << 1) | ((wid & 1) == 0);
switch (cond) {
case 0:
index = INDEX(2 * M - (hid + 1), N - (wid + 1) / 2, halfN);
break;
case 1:
index = INDEX(2 * M - (hid + 1), wid / 2, halfN);
break;
case 2:
index = INDEX(hid, N - (wid + 1) / 2, halfN);
break;
case 3:
index = INDEX(hid, wid / 2, halfN);
break;
default:
break;
}
y[index] = x[INDEX(hid, wid, N)];
}
}
template <typename T>
void dct2dPreprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream();
dct2dPreprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2);
}
template <typename T, typename TComplex>
__global__ void __launch_bounds__(TPB *TPB, 4) dct2dPostprocess(const TComplex *V,
T *y,
const int M,
const int N,
const int halfM,
const int halfN,
const T two_over_MN,
const T four_over_MN,
const TComplex *__restrict__ expkM,
const TComplex *__restrict__ expkN) {
const int wid = blockDim.x * blockIdx.x + threadIdx.x;
const int hid = blockDim.y * blockIdx.y + threadIdx.y;
if (hid < halfM && wid < halfN) {
int cond = ((hid != 0) << 1) | (wid != 0);
switch (cond) {
case 0: {
y[0] = V[0].x * four_over_MN;
y[halfN] = RealPartOfMul(expkN[halfN], V[halfN]) * four_over_MN;
y[INDEX(halfM, 0, N)] = expkM[halfM].x * V[INDEX(halfM, 0, halfN + 1)].x * four_over_MN;
y[INDEX(halfM, halfN, N)] =
expkM[halfM].x * RealPartOfMul(expkN[halfN], V[INDEX(halfM, halfN, halfN + 1)]) * four_over_MN;
break;
}
case 1: {
ComplexType<T> tmp;
tmp = V[wid];
y[wid] = RealPartOfMul(expkN[wid], tmp) * four_over_MN;
y[N - wid] = -ImaginaryPartOfMul(expkN[wid], tmp) * four_over_MN;
tmp = V[INDEX(halfM, wid, halfN + 1)];
y[INDEX(halfM, wid, N)] = expkM[halfM].x * RealPartOfMul(expkN[wid], tmp) * four_over_MN;
y[INDEX(halfM, N - wid, N)] = -expkM[halfM].x * ImaginaryPartOfMul(expkN[wid], tmp) * four_over_MN;
break;
}
case 2: {
ComplexType<T> tmp1, tmp2, tmp_up, tmp_down;
tmp1 = V[INDEX(hid, 0, halfN + 1)];
tmp2 = V[INDEX(M - hid, 0, halfN + 1)];
tmp_up.x = expkM[hid].x * (tmp1.x + tmp2.x) + expkM[hid].y * (tmp2.y - tmp1.y);
tmp_down.x = -expkM[hid].y * (tmp1.x + tmp2.x) + expkM[hid].x * (tmp2.y - tmp1.y);
y[INDEX(hid, 0, N)] = tmp_up.x * two_over_MN;
y[INDEX(M - hid, 0, N)] = tmp_down.x * two_over_MN;
tmp1 = complexAdd(V[INDEX(hid, halfN, halfN + 1)], V[INDEX(M - hid, halfN, halfN + 1)]);
tmp2 = complexSubtract(V[INDEX(hid, halfN, halfN + 1)], V[INDEX(M - hid, halfN, halfN + 1)]);
tmp_up.x = expkM[hid].x * tmp1.x - expkM[hid].y * tmp2.y;
tmp_up.y = expkM[hid].x * tmp1.y + expkM[hid].y * tmp2.x;
tmp_down.x = -expkM[hid].y * tmp1.x - expkM[hid].x * tmp2.y;
tmp_down.y = -expkM[hid].y * tmp1.y + expkM[hid].x * tmp2.x;
y[INDEX(hid, halfN, N)] = RealPartOfMul(expkN[halfN], tmp_up) * two_over_MN;
y[INDEX(M - hid, halfN, N)] = RealPartOfMul(expkN[halfN], tmp_down) * two_over_MN;
break;
}
case 3: {
ComplexType<T> tmp1, tmp2, tmp_up, tmp_down;
tmp1 = complexAdd(V[INDEX(hid, wid, halfN + 1)], V[INDEX(M - hid, wid, halfN + 1)]);
tmp2 = complexSubtract(V[INDEX(hid, wid, halfN + 1)], V[INDEX(M - hid, wid, halfN + 1)]);
tmp_up.x = expkM[hid].x * tmp1.x - expkM[hid].y * tmp2.y;
tmp_up.y = expkM[hid].x * tmp1.y + expkM[hid].y * tmp2.x;
tmp_down.x = -expkM[hid].y * tmp1.x - expkM[hid].x * tmp2.y;
tmp_down.y = -expkM[hid].y * tmp1.y + expkM[hid].x * tmp2.x;
y[INDEX(hid, wid, N)] = RealPartOfMul(expkN[wid], tmp_up) * two_over_MN;
y[INDEX(M - hid, wid, N)] = RealPartOfMul(expkN[wid], tmp_down) * two_over_MN;
y[INDEX(hid, N - wid, N)] = -ImaginaryPartOfMul(expkN[wid], tmp_up) * two_over_MN;
y[INDEX(M - hid, N - wid, N)] = -ImaginaryPartOfMul(expkN[wid], tmp_down) * two_over_MN;
break;
}
default:
break;
}
}
}
template <typename T>
void dct2dPostprocessCudaLauncher(
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream();
dct2dPostprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>((ComplexType<T> *)x,
y,
M,
N,
M / 2,
N / 2,
(T)(2. / (M * N)),
(T)(4. / (M * N)),
(ComplexType<T> *)expkM,
(ComplexType<T> *)expkN);
}
// idct2_fft2
template <typename T, typename TComplex>
__global__ void __launch_bounds__(TPB *TPB, 4) idct2_fft2Preprocess(const T *input,
TComplex *output,
const int M,
const int N,
const int halfM,
const int halfN,
const TComplex *__restrict__ expkM,
const TComplex *__restrict__ expkN) {
const int wid = blockDim.x * blockIdx.x + threadIdx.x;
const int hid = blockDim.y * blockIdx.y + threadIdx.y;
if (hid < halfM && wid < halfN) {
int cond = ((hid != 0) << 1) | (wid != 0);
switch (cond) {
case 0: {
T tmp1;
TComplex tmp_up;
output[0].x = input[0];
output[0].y = 0;
tmp1 = input[halfN];
tmp_up.x = tmp1;
tmp_up.y = tmp1;
output[halfN] = complexConj(complexMul(expkN[halfN], tmp_up));
tmp1 = input[INDEX(halfM, 0, N)];
tmp_up.x = tmp1;
tmp_up.y = tmp1;
output[INDEX(halfM, 0, halfN + 1)] = complexConj(complexMul(expkM[halfM], tmp_up));
tmp1 = input[INDEX(halfM, halfN, N)];
tmp_up.x = 0;
tmp_up.y = 2 * tmp1;
output[INDEX(halfM, halfN, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[halfM], expkN[halfN]), tmp_up));
break;
}
case 1: {
TComplex tmp_up;
tmp_up.x = input[wid];
tmp_up.y = input[N - wid];
output[wid] = complexConj(complexMul(expkN[wid], tmp_up));
T tmp1 = input[INDEX(halfM, wid, N)];
T tmp2 = input[INDEX(halfM, N - wid, N)];
tmp_up.x = tmp1 - tmp2;
tmp_up.y = tmp1 + tmp2;
output[INDEX(halfM, wid, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[halfM], expkN[wid]), tmp_up));
break;
}
case 2: {
T tmp1, tmp3;
TComplex tmp_up, tmp_down;
tmp1 = input[INDEX(hid, 0, N)];
tmp3 = input[INDEX(M - hid, 0, N)];
tmp_up.x = tmp1;
tmp_up.y = tmp3;
tmp_down.x = tmp3;
tmp_down.y = tmp1;
output[INDEX(hid, 0, halfN + 1)] = complexConj(complexMul(expkM[hid], tmp_up));
output[INDEX(M - hid, 0, halfN + 1)] = complexConj(complexMul(expkM[M - hid], tmp_down));
tmp1 = input[INDEX(hid, halfN, N)];
tmp3 = input[INDEX(M - hid, halfN, N)];
tmp_up.x = tmp1 - tmp3;
tmp_up.y = tmp3 + tmp1;
tmp_down.x = tmp3 - tmp1;
tmp_down.y = tmp1 + tmp3;
output[INDEX(hid, halfN, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[hid], expkN[halfN]), tmp_up));
output[INDEX(M - hid, halfN, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[M - hid], expkN[halfN]), tmp_down));
break;
}
case 3: {
T tmp1 = input[INDEX(hid, wid, N)];
T tmp2 = input[INDEX(hid, N - wid, N)];
T tmp3 = input[INDEX(M - hid, wid, N)];
T tmp4 = input[INDEX(M - hid, N - wid, N)];
TComplex tmp_up, tmp_down;
tmp_up.x = tmp1 - tmp4;
tmp_up.y = tmp3 + tmp2;
tmp_down.x = tmp3 - tmp2;
tmp_down.y = tmp1 + tmp4;
output[INDEX(hid, wid, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[hid], expkN[wid]), tmp_up));
output[INDEX(M - hid, wid, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[M - hid], expkN[wid]), tmp_down));
break;
}
default:
break;
}
}
}
template <typename T>
void idct2_fft2PreprocessCudaLauncher(
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream();
idct2_fft2Preprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>(
x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN);
}
template <typename T>
__global__ void idct2_fft2Postprocess(const T *x, T *y, const int M, const int N, const int halfN, const int MN) {
const int wid = blockDim.x * blockIdx.x + threadIdx.x;
const int hid = blockDim.y * blockIdx.y + threadIdx.y;
if (hid < M && wid < N) {
int cond = ((hid < M / 2) << 1) | (wid < N / 2);
int index;
switch (cond) {
case 0:
index = INDEX(((M - hid) << 1) - 1, ((N - wid) << 1) - 1, N);
break;
case 1:
index = INDEX(((M - hid) << 1) - 1, wid << 1, N);
break;
case 2:
index = INDEX(hid << 1, ((N - wid) << 1) - 1, N);
break;
case 3:
index = INDEX(hid << 1, wid << 1, N);
break;
default:
break;
}
y[index] = x[INDEX(hid, wid, N)] * MN;
}
}
template <typename T>
void idct2_fft2PostprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream();
idct2_fft2Postprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N);
}
// idct_idxst
// Adpated from idct2d_preprocess(). The only change is the reordered input
// if (wid != 0)
// new_input[hid][wid] = input[hid][N - wid];
// else
// new_input[hid][0] = 0
template <typename T, typename TComplex>
__global__ void __launch_bounds__(TPB *TPB, 4) idct_idxstPreprocess(const T *input,
TComplex *output,
const int M,
const int N,
const int halfM,
const int halfN,
const TComplex *__restrict__ expkM,
const TComplex *__restrict__ expkN) {
const int wid = blockDim.x * blockIdx.x + threadIdx.x;
const int hid = blockDim.y * blockIdx.y + threadIdx.y;
if (hid < halfM && wid < halfN) {
int cond = ((hid != 0) << 1) | (wid != 0);
switch (cond) {
case 0: {
T tmp1;
TComplex tmp_up;
output[0].x = 0;
output[0].y = 0;
tmp1 = input[halfN];
tmp_up.x = tmp1;
tmp_up.y = tmp1;
output[halfN] = complexConj(complexMul(expkN[halfN], tmp_up));
output[INDEX(halfM, 0, halfN + 1)].x = 0;
output[INDEX(halfM, 0, halfN + 1)].y = 0;
tmp1 = input[INDEX(halfM, halfN, N)];
tmp_up.x = 0;
tmp_up.y = 2 * tmp1;
output[INDEX(halfM, halfN, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[halfM], expkN[halfN]), tmp_up));
break;
}
case 1: {
TComplex tmp_up;
tmp_up.x = input[N - wid];
tmp_up.y = input[wid];
output[wid] = complexConj(complexMul(expkN[wid], tmp_up));
T tmp1 = input[INDEX(halfM, N - wid, N)];
T tmp2 = input[INDEX(halfM, wid, N)];
tmp_up.x = tmp1 - tmp2;
tmp_up.y = tmp1 + tmp2;
output[INDEX(halfM, wid, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[halfM], expkN[wid]), tmp_up));
break;
}
case 2: {
T tmp1, tmp3;
TComplex tmp_up, tmp_down;
output[INDEX(hid, 0, halfN + 1)].x = 0;
output[INDEX(hid, 0, halfN + 1)].y = 0;
output[INDEX(M - hid, 0, halfN + 1)].x = 0;
output[INDEX(M - hid, 0, halfN + 1)].y = 0;
tmp1 = input[INDEX(hid, halfN, N)];
tmp3 = input[INDEX(M - hid, halfN, N)];
tmp_up.x = tmp1 - tmp3;
tmp_up.y = tmp3 + tmp1;
tmp_down.x = tmp3 - tmp1;
tmp_down.y = tmp1 + tmp3;
output[INDEX(hid, halfN, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[hid], expkN[halfN]), tmp_up));
output[INDEX(M - hid, halfN, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[M - hid], expkN[halfN]), tmp_down));
break;
}
case 3: {
T tmp1 = input[INDEX(hid, N - wid, N)];
T tmp2 = input[INDEX(hid, wid, N)];
T tmp3 = input[INDEX(M - hid, N - wid, N)];
T tmp4 = input[INDEX(M - hid, wid, N)];
TComplex tmp_up, tmp_down;
tmp_up.x = tmp1 - tmp4;
tmp_up.y = tmp3 + tmp2;
tmp_down.x = tmp3 - tmp2;
tmp_down.y = tmp1 + tmp4;
output[INDEX(hid, wid, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[hid], expkN[wid]), tmp_up));
output[INDEX(M - hid, wid, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[M - hid], expkN[wid]), tmp_down));
break;
}
default:
break;
}
}
}
template <typename T>
void idct_idxstPreprocessCudaLauncher(
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream();
idct_idxstPreprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>(
x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN);
}
// Adpated from idct2d_postprocess() with changes on sign and scale
// if (wid % 2 == 1)
// new_output[hid][wid] = -output[hid][wid];
// else
// new_output[hid][wid] = output[hid][wid];
template <typename T>
__global__ void idct_idxstPostprocess(const T *x, T *y, const int M, const int N, const int halfN, const int MN) {
const int wid = blockDim.x * blockIdx.x + threadIdx.x;
const int hid = blockDim.y * blockIdx.y + threadIdx.y;
if (hid < M && wid < N) {
int cond = ((hid < M / 2) << 1) | (wid < N / 2);
int index;
switch (cond) {
case 0:
index = INDEX(((M - hid) << 1) - 1, ((N - wid) << 1) - 1, N);
y[index] = -x[INDEX(hid, wid, N)] * MN;
break;
case 1:
index = INDEX(((M - hid) << 1) - 1, wid << 1, N);
y[index] = x[INDEX(hid, wid, N)] * MN;
break;
case 2:
index = INDEX(hid << 1, ((N - wid) << 1) - 1, N);
y[index] = -x[INDEX(hid, wid, N)] * MN;
break;
case 3:
index = INDEX(hid << 1, wid << 1, N);
y[index] = x[INDEX(hid, wid, N)] * MN;
break;
default:
break;
}
}
}
template <typename T>
void idct_idxstPostprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream();
idct_idxstPostprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N);
}
// idxst_idct
// Adpated from idct2d_preprocess(). The only change is the reordered input
// if (hid != 0)
// new_input[hid][wid] = input[M - hid][wid];
// else
// new_input[0][wid] = 0
template <typename T, typename TComplex>
__global__ void __launch_bounds__(TPB *TPB, 4) idxst_idctPreprocess(const T *input,
TComplex *output,
const int M,
const int N,
const int halfM,
const int halfN,
const TComplex *__restrict__ expkM,
const TComplex *__restrict__ expkN) {
const int wid = blockDim.x * blockIdx.x + threadIdx.x;
const int hid = blockDim.y * blockIdx.y + threadIdx.y;
if (hid < halfM && wid < halfN) {
int cond = ((hid != 0) << 1) | (wid != 0);
switch (cond) {
case 0: {
T tmp1;
TComplex tmp_up;
output[0].x = 0;
output[0].y = 0;
output[halfN].x = 0;
output[halfN].y = 0;
tmp1 = input[INDEX(halfM, 0, N)];
tmp_up.x = tmp1;
tmp_up.y = tmp1;
output[INDEX(halfM, 0, halfN + 1)] = complexConj(complexMul(expkM[halfM], tmp_up));
tmp1 = input[INDEX(halfM, halfN, N)];
tmp_up.x = 0;
tmp_up.y = 2 * tmp1;
output[INDEX(halfM, halfN, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[halfM], expkN[halfN]), tmp_up));
break;
}
case 1: {
output[wid].x = 0;
output[wid].y = 0;
TComplex tmp_up;
T tmp1 = input[INDEX(halfM, wid, N)];
T tmp2 = input[INDEX(halfM, N - wid, N)];
tmp_up.x = tmp1 - tmp2;
tmp_up.y = tmp1 + tmp2;
output[INDEX(halfM, wid, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[halfM], expkN[wid]), tmp_up));
break;
}
case 2: {
T tmp1, tmp3;
TComplex tmp_up, tmp_down;
tmp1 = input[INDEX(M - hid, 0, N)];
tmp3 = input[INDEX(hid, 0, N)];
tmp_up.x = tmp1;
tmp_up.y = tmp3;
tmp_down.x = tmp3;
tmp_down.y = tmp1;
output[INDEX(hid, 0, halfN + 1)] = complexConj(complexMul(expkM[hid], tmp_up));
output[INDEX(M - hid, 0, halfN + 1)] = complexConj(complexMul(expkM[M - hid], tmp_down));
tmp1 = input[INDEX(M - hid, halfN, N)];
tmp3 = input[INDEX(hid, halfN, N)];
tmp_up.x = tmp1 - tmp3;
tmp_up.y = tmp3 + tmp1;
tmp_down.x = tmp3 - tmp1;
tmp_down.y = tmp1 + tmp3;
output[INDEX(hid, halfN, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[hid], expkN[halfN]), tmp_up));
output[INDEX(M - hid, halfN, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[M - hid], expkN[halfN]), tmp_down));
break;
}
case 3: {
T tmp1 = input[INDEX(M - hid, wid, N)];
T tmp2 = input[INDEX(M - hid, N - wid, N)];
T tmp3 = input[INDEX(hid, wid, N)];
T tmp4 = input[INDEX(hid, N - wid, N)];
TComplex tmp_up, tmp_down;
tmp_up.x = tmp1 - tmp4;
tmp_up.y = tmp3 + tmp2;
tmp_down.x = tmp3 - tmp2;
tmp_down.y = tmp1 + tmp4;
output[INDEX(hid, wid, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[hid], expkN[wid]), tmp_up));
output[INDEX(M - hid, wid, halfN + 1)] =
complexConj(complexMul(complexMul(expkM[M - hid], expkN[wid]), tmp_down));
break;
}
default:
break;
}
}
}
template <typename T>
void idxst_idctPreprocessCudaLauncher(
const T *x, T *y, const int M, const int N, const T *__restrict__ expkM, const T *__restrict__ expkN) {
dim3 gridSize((N / 2 + TPB - 1) / TPB, (M / 2 + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream();
idxst_idctPreprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>(
x, (ComplexType<T> *)y, M, N, M / 2, N / 2, (ComplexType<T> *)expkM, (ComplexType<T> *)expkN);
}
// Adpated from idct2d_postprocess() with changes on sign and scale
// if (hid % 2 == 1)
// new_output[hid][wid] = -output[hid][wid];
// else
// new_output[hid][wid] = output[hid][wid];
template <typename T>
__global__ void idxst_idctPostprocess(const T *x, T *y, const int M, const int N, const int halfN, const int MN) {
const int wid = blockDim.x * blockIdx.x + threadIdx.x;
const int hid = blockDim.y * blockIdx.y + threadIdx.y;
if (hid < M && wid < N) {
int cond = ((hid < M / 2) << 1) | (wid < N / 2);
int index;
switch (cond) {
case 0:
index = INDEX(((M - hid) << 1) - 1, ((N - wid) << 1) - 1, N);
y[index] = -x[INDEX(hid, wid, N)] * MN;
break;
case 1:
index = INDEX(((M - hid) << 1) - 1, wid << 1, N);
y[index] = -x[INDEX(hid, wid, N)] * MN;
break;
case 2:
index = INDEX(hid << 1, ((N - wid) << 1) - 1, N);
y[index] = x[INDEX(hid, wid, N)] * MN;
break;
case 3:
index = INDEX(hid << 1, wid << 1, N);
y[index] = x[INDEX(hid, wid, N)] * MN;
break;
default:
break;
}
}
}
template <typename T>
void idxst_idctPostprocessCudaLauncher(const T *x, T *y, const int M, const int N) {
dim3 gridSize((N + TPB - 1) / TPB, (M + TPB - 1) / TPB, 1);
dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream();
idxst_idctPostprocess<T><<<gridSize, blockSize, 0, stream>>>(x, y, M, N, N / 2, M * N);
}
// launch
#define CHECK_CUDA(x) TORCH_CHECK(x.device().is_cuda(), #x " must be a CUDA tensor")
#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
#define CHECK_INPUT(x) \
CHECK_CUDA(x); \
CHECK_CONTIGUOUS(x)
void dct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
cudaSetDevice(x.get_device());
CHECK_INPUT(x);
CHECK_INPUT(expkM);
CHECK_INPUT(expkN);
CHECK_INPUT(out);
CHECK_INPUT(buf);
auto N = x.size(-1);
auto M = x.numel() / N;
AT_DISPATCH_ALL_TYPES(x.scalar_type(), "dct2_fft2_forward_cuda", [&] {
dct2dPreprocessCudaLauncher<scalar_t>(x.data_ptr<scalar_t>(), out.data_ptr<scalar_t>(), M, N);
buf = at::view_as_real(at::fft_rfft2(out, c10::nullopt, {-2, -1}, "backward")).contiguous();
dct2dPostprocessCudaLauncher<scalar_t>(buf.data_ptr<scalar_t>(),
out.data_ptr<scalar_t>(),
M,
N,
expkM.data_ptr<scalar_t>(),
expkN.data_ptr<scalar_t>());
});
}
void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
cudaSetDevice(x.get_device());
CHECK_INPUT(x);
CHECK_INPUT(expkM);
CHECK_INPUT(expkN);
CHECK_INPUT(out);
CHECK_INPUT(buf);
auto N = x.size(-1);
auto M = x.numel() / N;
AT_DISPATCH_ALL_TYPES(x.scalar_type(), "idct2_fft2_forward_cuda", [&] {
idct2_fft2PreprocessCudaLauncher<scalar_t>(x.data_ptr<scalar_t>(),
buf.data_ptr<scalar_t>(),
M,
N,
expkM.data_ptr<scalar_t>(),
expkN.data_ptr<scalar_t>());
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
idct2_fft2PostprocessCudaLauncher<scalar_t>(y.data_ptr<scalar_t>(), out.data_ptr<scalar_t>(), M, N);
});
}
void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
cudaSetDevice(x.get_device());
CHECK_INPUT(x);
CHECK_INPUT(expkM);
CHECK_INPUT(expkN);
CHECK_INPUT(out);
CHECK_INPUT(buf);
auto N = x.size(-1);
auto M = x.numel() / N;
AT_DISPATCH_ALL_TYPES(x.scalar_type(), "idct_idxst_forward_cuda", [&] {
idct_idxstPreprocessCudaLauncher<scalar_t>(x.data_ptr<scalar_t>(),
buf.data_ptr<scalar_t>(),
M,
N,
expkM.data_ptr<scalar_t>(),
expkN.data_ptr<scalar_t>());
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
idct_idxstPostprocessCudaLauncher<scalar_t>(y.data_ptr<scalar_t>(), out.data_ptr<scalar_t>(), M, N);
});
}
void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
cudaSetDevice(x.get_device());
CHECK_INPUT(x);
CHECK_INPUT(expkM);
CHECK_INPUT(expkN);
CHECK_INPUT(out);
CHECK_INPUT(buf);
auto N = x.size(-1);
auto M = x.numel() / N;
AT_DISPATCH_ALL_TYPES(x.scalar_type(), "idxst_idct_forward_cuda", [&] {
idxst_idctPreprocessCudaLauncher<scalar_t>(x.data_ptr<scalar_t>(),
buf.data_ptr<scalar_t>(),
M,
N,
expkM.data_ptr<scalar_t>(),
expkN.data_ptr<scalar_t>());
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
idxst_idctPostprocessCudaLauncher<scalar_t>(y.data_ptr<scalar_t>(), out.data_ptr<scalar_t>(), M, N);
});
}

View File

@ -0,0 +1,10 @@
set(TARGET_NAME density_map_cuda)
add_pytorch_extension(${TARGET_NAME}
${TARGET_NAME}.cpp
${TARGET_NAME}_kernel.cu
${TARGET_NAME}_naive_kernel.cu)
install(TARGETS
${TARGET_NAME}
DESTINATION ${XPLACE_LIB_DIR})

View File

@ -0,0 +1,141 @@
#include <torch/extension.h>
torch::Tensor density_map_cuda_normalize_node(torch::Tensor node_pos,
torch::Tensor node_size,
torch::Tensor node_weight,
torch::Tensor expand_ratio,
torch::Tensor unit_len,
torch::Tensor normalize_node_info,
int num_bin_x,
int num_bin_y,
int num_nodes);
torch::Tensor density_map_cuda_forward(torch::Tensor normalize_node_info,
torch::Tensor sorted_node_map,
torch::Tensor aux_mat,
int num_bin_x,
int num_bin_y,
int num_nodes);
torch::Tensor density_map_cuda_backward(torch::Tensor normalize_node_info,
torch::Tensor grad_mat,
torch::Tensor sorted_node_map,
torch::Tensor node_grad,
float grad_weight,
int num_bin_x,
int num_bin_y,
int num_nodes);
torch::Tensor density_map_cuda_forward_naive(torch::Tensor node_pos,
torch::Tensor node_size,
torch::Tensor node_weight,
torch::Tensor unit_len,
torch::Tensor aux_mat,
int num_bin_x,
int num_bin_y,
int num_nodes,
float min_node_w,
float min_node_h,
float margin,
bool clamp_node);
#define CHECK_CUDA(x) TORCH_CHECK(x.device().is_cuda(), #x " must be a CUDA tensor")
#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
#define CHECK_INPUT(x) \
CHECK_CUDA(x); \
CHECK_CONTIGUOUS(x)
torch::Tensor density_map_normalize_node(torch::Tensor node_pos,
torch::Tensor node_size,
torch::Tensor node_weight,
torch::Tensor expand_ratio,
torch::Tensor unit_len,
torch::Tensor normalize_node_info,
int num_bin_x,
int num_bin_y,
int num_nodes) {
CHECK_INPUT(node_pos);
CHECK_INPUT(node_size);
CHECK_INPUT(node_weight);
CHECK_INPUT(expand_ratio);
CHECK_INPUT(unit_len);
CHECK_INPUT(normalize_node_info);
return density_map_cuda_normalize_node(node_pos,
node_size,
node_weight,
expand_ratio,
unit_len,
normalize_node_info,
num_bin_x,
num_bin_y,
num_nodes);
}
torch::Tensor density_map_forward(torch::Tensor normalize_node_info,
torch::Tensor sorted_node_map,
torch::Tensor aux_mat,
int num_bin_x,
int num_bin_y,
int num_nodes) {
CHECK_INPUT(normalize_node_info);
CHECK_INPUT(sorted_node_map);
CHECK_INPUT(aux_mat);
return density_map_cuda_forward(normalize_node_info, sorted_node_map, aux_mat, num_bin_x, num_bin_y, num_nodes);
}
torch::Tensor density_map_backward(torch::Tensor normalize_node_info,
torch::Tensor grad_mat,
torch::Tensor sorted_node_map,
torch::Tensor node_grad,
float grad_weight,
int num_bin_x,
int num_bin_y,
int num_nodes) {
CHECK_INPUT(normalize_node_info);
CHECK_INPUT(grad_mat);
CHECK_INPUT(sorted_node_map);
CHECK_INPUT(node_grad);
return density_map_cuda_backward(
normalize_node_info, grad_mat, sorted_node_map, node_grad, grad_weight, num_bin_x, num_bin_y, num_nodes);
}
torch::Tensor density_map_forward_naive(torch::Tensor node_pos,
torch::Tensor node_size,
torch::Tensor node_weight,
torch::Tensor unit_len,
torch::Tensor aux_mat,
int num_bin_x,
int num_bin_y,
int num_nodes,
float min_node_w,
float min_node_h,
float margin,
bool clamp_node) {
CHECK_INPUT(node_pos);
CHECK_INPUT(node_size);
CHECK_INPUT(node_weight);
CHECK_INPUT(unit_len);
CHECK_INPUT(aux_mat);
return density_map_cuda_forward_naive(node_pos,
node_size,
node_weight,
unit_len,
aux_mat,
num_bin_x,
num_bin_y,
num_nodes,
min_node_w,
min_node_h,
margin,
clamp_node);
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
m.def("pre_normalize", &density_map_normalize_node, "normalize bin size to 1");
m.def("forward", &density_map_forward, "get density map from node information");
m.def("forward_naive", &density_map_cuda_forward_naive, "calculate density map");
m.def("backward", &density_map_backward, "calculate density gradient of each node");
}

View File

@ -0,0 +1,232 @@
#include <cuda.h>
#include <cuda_runtime.h>
#include <torch/extension.h>
#include <THC/THCAtomics.cuh>
#include <vector>
template <typename scalar_t>
__device__ scalar_t overlap(scalar_t x_l, scalar_t x_h, scalar_t bin_x_l) {
// bin_x_h == bin_x_l + 1
return min(x_h, bin_x_l + 1) - max(x_l, bin_x_l);
}
template <typename scalar_t>
__global__ void density_map_cuda_normalize_node_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_size,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> node_weight,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> expand_ratio,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> unit_len,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> normalize_node_info,
int num_bin_x,
int num_bin_y,
int num_nodes) {
const int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_nodes) {
normalize_node_info[i][0] = (node_pos[i][0] - node_size[i][0] / 2) / unit_len[0]; // x_l
normalize_node_info[i][1] = (node_pos[i][0] + node_size[i][0] / 2) / unit_len[0]; // x_h
normalize_node_info[i][2] = (node_pos[i][1] - node_size[i][1] / 2) / unit_len[1]; // y_l
normalize_node_info[i][3] = (node_pos[i][1] + node_size[i][1] / 2) / unit_len[1]; // y_h
normalize_node_info[i][4] = node_weight[i] * expand_ratio[i]; // weight
if (normalize_node_info[i][1] - normalize_node_info[i][0] < 0 ||
normalize_node_info[i][3] - normalize_node_info[i][2] < 0) {
normalize_node_info[i][4] = -normalize_node_info[i][4]; // we should ignore node whose weight <= 0
}
}
}
template <typename scalar_t>
__global__ void __launch_bounds__(256, 4) density_map_cuda_forward_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> normalize_node_info,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> sorted_node_map,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> aux_mat,
int num_nodes,
int num_bin_x,
int num_bin_y) {
const int index = blockIdx.x * blockDim.z + threadIdx.z;
if (index < num_nodes) {
const int i = (sorted_node_map[index] >= 0) ? sorted_node_map[index] : index;
const scalar_t weight = normalize_node_info[i][4];
if (weight > 0) {
const scalar_t x_l = normalize_node_info[i][0];
const scalar_t x_h = normalize_node_info[i][1];
const scalar_t y_l = normalize_node_info[i][2];
const scalar_t y_h = normalize_node_info[i][3];
int x_lf = lround(floor(x_l));
int x_hf = lround(floor(x_h));
int y_lf = lround(floor(y_l));
int y_hf = lround(floor(y_h));
x_lf = max(x_lf, 0);
x_hf = min(x_hf, num_bin_x - 1);
y_lf = max(y_lf, 0);
y_hf = min(y_hf, num_bin_y - 1);
for (int j = x_lf + threadIdx.y; j < x_hf + 1; j += blockDim.y) {
scalar_t bin_x_l = static_cast<scalar_t>(j);
scalar_t overlap_x = overlap(x_l, x_h, bin_x_l);
for (int k = y_lf + threadIdx.x; k < y_hf + 1; k += blockDim.x) {
scalar_t bin_y_l = static_cast<scalar_t>(k);
scalar_t overlap_y = overlap(y_l, y_h, bin_y_l);
scalar_t overlap_area = overlap_x * overlap_y;
gpuAtomicAdd(&aux_mat[j][k], weight * overlap_area);
}
}
}
}
}
template <typename scalar_t>
__global__ void __launch_bounds__(256, 4) density_map_cuda_backward_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> normalize_node_info,
const scalar_t *grad_mat,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> sorted_node_map,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_grad,
float grad_weight,
int num_bin_x,
int num_bin_y,
int num_nodes) {
const int index = blockIdx.x * blockDim.z + threadIdx.z;
if (index < num_nodes) {
const int i = (sorted_node_map[index] >= 0) ? sorted_node_map[index] : index;
const scalar_t weight = normalize_node_info[i][4];
if (weight > 0) {
const scalar_t x_l = normalize_node_info[i][0];
const scalar_t x_h = normalize_node_info[i][1];
const scalar_t y_l = normalize_node_info[i][2];
const scalar_t y_h = normalize_node_info[i][3];
int x_lf = lround(floor(x_l));
int x_hf = lround(floor(x_h));
int y_lf = lround(floor(y_l));
int y_hf = lround(floor(y_h));
x_lf = max(x_lf, 0);
x_hf = min(x_hf, num_bin_x - 1);
y_lf = max(y_lf, 0);
y_hf = min(y_hf, num_bin_y - 1);
extern __shared__ unsigned char grad_xy[];
scalar_t *grad_x = (scalar_t *)grad_xy;
scalar_t *grad_y = grad_x + blockDim.z;
if (threadIdx.x == 0 && threadIdx.y == 0) {
grad_x[threadIdx.z] = grad_y[threadIdx.z] = 0;
}
__syncthreads();
scalar_t part_grad_x = 0;
scalar_t part_grad_y = 0;
for (int j = x_lf + threadIdx.y; j < x_hf + 1; j += blockDim.y) {
scalar_t bin_x_l = static_cast<scalar_t>(j);
scalar_t overlap_x = overlap(x_l, x_h, bin_x_l);
for (int k = y_lf + threadIdx.x; k < y_hf + 1; k += blockDim.x) {
scalar_t bin_y_l = static_cast<scalar_t>(k);
scalar_t overlap_y = overlap(y_l, y_h, bin_y_l);
scalar_t overlap_area = overlap_x * overlap_y;
scalar_t tmp_x = grad_mat[0 * num_bin_x * num_bin_y + j * num_bin_y + k];
scalar_t tmp_y = grad_mat[1 * num_bin_x * num_bin_y + j * num_bin_y + k];
// part_grad_x += overlap_area * grad_mat[0][j][k];
// part_grad_y += overlap_area * grad_mat[1][j][k];
part_grad_x += overlap_area * tmp_x;
part_grad_y += overlap_area * tmp_y;
}
}
gpuAtomicAdd(&grad_x[threadIdx.z], part_grad_x);
gpuAtomicAdd(&grad_y[threadIdx.z], part_grad_y);
__syncthreads();
if (threadIdx.x == 0 && threadIdx.y == 0) {
node_grad[i][0] = grad_weight * weight * grad_x[threadIdx.z];
node_grad[i][1] = grad_weight * weight * grad_y[threadIdx.z];
}
}
}
}
torch::Tensor density_map_cuda_normalize_node(torch::Tensor node_pos,
torch::Tensor node_size,
torch::Tensor node_weight,
torch::Tensor expand_ratio,
torch::Tensor unit_len,
torch::Tensor normalize_node_info,
int num_bin_x,
int num_bin_y,
int num_nodes) {
cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
const int threads = 128;
const int blocks = (num_nodes + threads - 1) / threads;
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "density_map_cuda_normalize_node", ([&] {
density_map_cuda_normalize_node_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
node_size.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
node_weight.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
expand_ratio.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
unit_len.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
normalize_node_info.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_bin_x,
num_bin_y,
num_nodes);
}));
return normalize_node_info;
}
torch::Tensor density_map_cuda_forward(torch::Tensor normalize_node_info,
torch::Tensor sorted_node_map,
torch::Tensor aux_mat,
int num_bin_x,
int num_bin_y,
int num_nodes) {
cudaSetDevice(normalize_node_info.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
int thread_count = 64;
dim3 blockSize(2, 2, thread_count);
int block_count = (num_nodes - 1 + thread_count) / thread_count;
AT_DISPATCH_ALL_TYPES(normalize_node_info.scalar_type(), "density_map_cuda_forward", ([&] {
density_map_cuda_forward_kernel<scalar_t><<<block_count, blockSize, 0, stream>>>(
normalize_node_info.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
aux_mat.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_nodes,
num_bin_x,
num_bin_y);
}));
return aux_mat;
}
torch::Tensor density_map_cuda_backward(torch::Tensor normalize_node_info,
torch::Tensor grad_mat,
torch::Tensor sorted_node_map,
torch::Tensor node_grad,
float grad_weight,
int num_bin_x,
int num_bin_y,
int num_nodes) {
cudaSetDevice(normalize_node_info.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
int thread_count = 64;
dim3 blockSize(2, 2, thread_count);
int block_count = (num_nodes - 1 + thread_count) / thread_count;
AT_DISPATCH_ALL_TYPES(normalize_node_info.scalar_type(), "density_map_cuda_backward", ([&] {
size_t shared_mem_size = sizeof(scalar_t) * thread_count * 2;
density_map_cuda_backward_kernel<scalar_t>
<<<block_count, blockSize, shared_mem_size, stream>>>(
normalize_node_info.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
grad_mat.data_ptr<scalar_t>(),
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
node_grad.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
grad_weight,
num_bin_x,
num_bin_y,
num_nodes);
}));
return node_grad;
}

View File

@ -0,0 +1,214 @@
#include <cuda.h>
#include <cuda_runtime.h>
#include <torch/extension.h>
#include <THC/THCAtomics.cuh>
#include <vector>
template <typename scalar_t>
__global__ void density_map_cuda_forward_naive_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_size,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> node_weight,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> unit_len,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> aux_mat,
int num_bin_x,
int num_bin_y,
int num_nodes,
float min_node_w,
float min_node_h,
float margin,
bool clamp_node) {
const int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_nodes) {
scalar_t node_w = node_size[i][0];
scalar_t node_h = node_size[i][1];
scalar_t ratio = 1.0;
if (clamp_node) {
const scalar_t node_area = node_w * node_h;
node_w = max(node_w, static_cast<scalar_t>(min_node_w));
node_h = max(node_h, static_cast<scalar_t>(min_node_h));
ratio = node_area / (node_w * node_h);
}
const scalar_t mgn = static_cast<scalar_t>(margin);
const scalar_t num_bin_x_minus_mgn = static_cast<scalar_t>(num_bin_x) - mgn;
const scalar_t num_bin_y_minus_mgn = static_cast<scalar_t>(num_bin_y) - mgn;
const scalar_t small_mgn = static_cast<scalar_t>(margin * 0.1);
scalar_t x_l = max((node_pos[i][0] - node_w / 2) / unit_len[0], mgn);
scalar_t x_h = min((node_pos[i][0] + node_w / 2) / unit_len[0], num_bin_x_minus_mgn);
scalar_t y_l = max((node_pos[i][1] - node_h / 2) / unit_len[1], mgn);
scalar_t y_h = min((node_pos[i][1] + node_h / 2) / unit_len[1], num_bin_y_minus_mgn);
x_l = min(x_l, num_bin_x_minus_mgn);
x_h = max(x_h, mgn);
y_l = min(y_l, num_bin_y_minus_mgn);
y_h = max(y_h, mgn);
if (x_h - x_l < small_mgn || y_h - y_l < small_mgn) {
return;
}
const scalar_t p_node_wght = node_weight[i] * ratio;
const int x_lf = lround(floor(x_l));
const int x_hf = lround(floor(x_h));
const int y_lf = lround(floor(y_l));
const int y_hf = lround(floor(y_h));
for (int j = x_lf; j < x_hf + 1; j++) {
const scalar_t bin_x_l = j;
const scalar_t bin_x_h = j + 1;
scalar_t overlap_x = min(x_h, bin_x_h) - max(x_l, bin_x_l);
for (int k = y_lf; k < y_hf + 1; k++) {
const scalar_t bin_y_l = k;
const scalar_t bin_y_h = k + 1;
scalar_t overlap_y = min(y_h, bin_y_h) - max(y_l, bin_y_l);
scalar_t overlap_area = overlap_x * overlap_y;
gpuAtomicAdd(&aux_mat[j][k], p_node_wght * overlap_area);
}
}
}
}
template <typename scalar_t>
__global__ void density_map_cuda_backward_naive_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_size,
const torch::PackedTensorAccessor32<scalar_t, 3, torch::RestrictPtrTraits> grad_mat,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> unit_len,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> node_weight,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_grad,
float grad_weight,
int num_bin_x,
int num_bin_y,
int num_nodes,
float min_node_w,
float min_node_h,
float margin,
bool clamp_node) {
const int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_nodes) {
scalar_t node_w = node_size[i][0];
scalar_t node_h = node_size[i][1];
scalar_t ratio = 1.0;
if (clamp_node) {
const scalar_t node_area = node_w * node_h;
node_w = max(node_w, static_cast<scalar_t>(min_node_w));
node_h = max(node_h, static_cast<scalar_t>(min_node_h));
ratio = node_area / (node_w * node_h);
}
const scalar_t mgn = static_cast<scalar_t>(margin);
const scalar_t num_bin_x_minus_mgn = static_cast<scalar_t>(num_bin_x) - mgn;
const scalar_t num_bin_y_minus_mgn = static_cast<scalar_t>(num_bin_y) - mgn;
const scalar_t small_mgn = static_cast<scalar_t>(margin * 0.1);
scalar_t x_l = max((node_pos[i][0] - node_w / 2) / unit_len[0], mgn);
scalar_t x_h = min((node_pos[i][0] + node_w / 2) / unit_len[0], num_bin_x_minus_mgn);
scalar_t y_l = max((node_pos[i][1] - node_h / 2) / unit_len[1], mgn);
scalar_t y_h = min((node_pos[i][1] + node_h / 2) / unit_len[1], num_bin_y_minus_mgn);
x_l = min(x_l, num_bin_x_minus_mgn);
x_h = max(x_h, mgn);
y_l = min(y_l, num_bin_y_minus_mgn);
y_h = max(y_h, mgn);
if (x_h - x_l < small_mgn || y_h - y_l < small_mgn) {
return;
}
const int x_lf = lround(floor(x_l));
const int x_hf = lround(floor(x_h));
const int y_lf = lround(floor(y_l));
const int y_hf = lround(floor(y_h));
scalar_t gradX = 0;
scalar_t gradY = 0;
for (int j = x_lf; j < x_hf + 1; j++) {
const scalar_t bin_x_l = j;
const scalar_t bin_x_h = j + 1;
scalar_t overlap_x = min(x_h, bin_x_h) - max(x_l, bin_x_l);
for (int k = y_lf; k < y_hf + 1; k++) {
const scalar_t bin_y_l = k;
const scalar_t bin_y_h = k + 1;
scalar_t overlap_y = min(y_h, bin_y_h) - max(y_l, bin_y_l);
scalar_t overlap_area = overlap_x * overlap_y;
gradX += grad_mat[0][j][k] * overlap_area;
gradY += grad_mat[1][j][k] * overlap_area;
}
}
node_grad[i][0] = grad_weight * ratio * node_weight[i] * gradX;
node_grad[i][1] = grad_weight * ratio * node_weight[i] * gradY;
}
}
torch::Tensor density_map_cuda_forward_naive(torch::Tensor node_pos,
torch::Tensor node_size,
torch::Tensor node_weight,
torch::Tensor unit_len,
torch::Tensor aux_mat,
int num_bin_x,
int num_bin_y,
int num_nodes,
float min_node_w,
float min_node_h,
float margin,
bool clamp_node) {
cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
const int threads = 64;
const int blocks = (num_nodes + threads - 1) / threads;
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "density_map_cuda_forward_naive", ([&] {
density_map_cuda_forward_naive_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
node_size.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
node_weight.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
unit_len.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
aux_mat.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_bin_x,
num_bin_y,
num_nodes,
min_node_w,
min_node_h,
margin,
clamp_node);
}));
return aux_mat;
}
torch::Tensor density_map_cuda_backward(torch::Tensor node_pos,
torch::Tensor node_size,
torch::Tensor grad_mat,
torch::Tensor node_weight,
torch::Tensor unit_len,
torch::Tensor node_grad,
float grad_weight,
int num_bin_x,
int num_bin_y,
int num_nodes,
float min_node_w,
float min_node_h,
float margin,
bool clamp_node) {
cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
const int threads = 64;
const int blocks = (num_nodes + threads - 1) / threads;
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "density_map_cuda_backward_naive", ([&] {
density_map_cuda_backward_naive_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
node_size.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
grad_mat.packed_accessor32<scalar_t, 3, torch::RestrictPtrTraits>(),
unit_len.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
node_weight.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
node_grad.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
grad_weight,
num_bin_x,
num_bin_y,
num_nodes,
min_node_w,
min_node_h,
margin,
clamp_node);
}));
return node_grad;
}

View File

@ -0,0 +1,15 @@
set(TARGET_NAME draw_placement)
pybind11_add_module(${TARGET_NAME} MODULE ${TARGET_NAME}.cpp Drawer.cpp)
target_include_directories(${TARGET_NAME} PRIVATE ${TORCH_INCLUDE_DIRS} ${CAIRO_INCLUDE_DIRS})
target_link_libraries(
${TARGET_NAME} PRIVATE torch ${TORCH_PYTHON_LIBRARY} ${CAIRO_LIBRARIES})
target_compile_definitions(${TARGET_NAME} PRIVATE
TORCH_EXTENSION_NAME=${TARGET_NAME}
TORCH_MAJOR_VERSION=${TORCH_MAJOR_VERSION}
TORCH_MINOR_VERSION=${TORCH_MINOR_VERSION}
ENABLE_CUDA=0)
install(TARGETS
${TARGET_NAME}
DESTINATION ${XPLACE_LIB_DIR})

View File

@ -0,0 +1,233 @@
// Created by Lixin LIU on 2021/09/14
// Adapt from https://github.com/limbo018/DREAMPlace/blob/master/dreamplace/ops/draw_place/src/PlaceDrawer.h
#include "Drawer.h"
Drawer::Drawer(const std::vector<std::tuple<std::string, double, double, double, double>>& ele_type_to_rgba_vec,
const std::string& filename_,
double width_,
double height_,
const std::vector<std::string>& contents_) {
filename = filename_;
std::string format_ = filename_.substr(filename_.size() - 4);
if (format_ == ".png") {
format = "png";
} else if (format_ == ".pdf") {
format = "pdf";
} else if (format_ == ".svg") {
format = "svg";
} else if (format_ == ".eps") {
format = "eps";
}
width = width_;
height = height_;
for (auto content_ : contents_) {
contents.emplace(content_);
}
// Customized value in type_to_rgba
for (auto [ele_type, r, g, b, a] : ele_type_to_rgba_vec) {
auto it = type_to_rgba.find(ele_type);
if (it == type_to_rgba.end()) {
type_to_rgba.emplace(ele_type, std::make_tuple(r, g, b, a));
}
}
// Default value in type_to_rgba
std::string ele_type;
ele_type = "Background";
if (type_to_rgba.find(ele_type) == type_to_rgba.end()) {
type_to_rgba.emplace(ele_type, std::make_tuple(1.0, 1.0, 1.0, 1.0));
}
ele_type = "Die";
if (type_to_rgba.find(ele_type) == type_to_rgba.end()) {
type_to_rgba.emplace(ele_type, std::make_tuple(1.0, 1.0, 1.0, 1.0));
}
ele_type = "Bin";
if (type_to_rgba.find(ele_type) == type_to_rgba.end()) {
type_to_rgba.emplace(ele_type, std::make_tuple(0.3, 0.3, 0.3, 1.0));
}
ele_type = "Grid";
if (type_to_rgba.find(ele_type) == type_to_rgba.end()) {
type_to_rgba.emplace(ele_type, std::make_tuple(0.3, 0.3, 0.3, 1.0));
}
ele_type = "Mov";
if (type_to_rgba.find(ele_type) == type_to_rgba.end()) {
type_to_rgba.emplace(ele_type, std::make_tuple(0.475, 0.706, 0.718, 0.8));
}
ele_type = "Fix";
if (type_to_rgba.find(ele_type) == type_to_rgba.end()) {
type_to_rgba.emplace(ele_type, std::make_tuple(0.878, 0.365, 0.365, 0.8));
}
ele_type = "IOPin";
if (type_to_rgba.find(ele_type) == type_to_rgba.end()) {
type_to_rgba.emplace(ele_type, std::make_tuple(0.878, 0.365, 0.365, 0.8));
}
ele_type = "Blkg";
if (type_to_rgba.find(ele_type) == type_to_rgba.end()) {
type_to_rgba.emplace(ele_type, std::make_tuple(0.878, 0.365, 0.365, 0.8));
}
ele_type = "doubleIOPin";
if (type_to_rgba.find(ele_type) == type_to_rgba.end()) {
type_to_rgba.emplace(ele_type, std::make_tuple(0.878, 0.365, 0.365, 0.8));
}
ele_type = "doubleFix";
if (type_to_rgba.find(ele_type) == type_to_rgba.end()) {
type_to_rgba.emplace(ele_type, std::make_tuple(0.878, 0.365, 0.365, 0.8));
}
ele_type = "doubleMov";
if (type_to_rgba.find(ele_type) == type_to_rgba.end()) {
type_to_rgba.emplace(ele_type, std::make_tuple(0.082, 0.176, 0.208, 0.8));
}
ele_type = "Filler";
if (type_to_rgba.find(ele_type) == type_to_rgba.end()) {
type_to_rgba.emplace(ele_type, std::make_tuple(0.082, 0.176, 0.208, 0.8));
}
}
std::tuple<double, double, double, double> Drawer::get_colors(const std::string& ele_type) {
auto it = type_to_rgba.find(ele_type);
if (it != type_to_rgba.end()) {
return it->second;
}
return std::make_tuple(0.0, 0.0, 0.0, 1.0);
}
bool Drawer::run(const std::vector<double>& node_pos_x, // after die scale
const std::vector<double>& node_pos_y, // after die scale
const std::vector<double>& node_size_x, // after die scale
const std::vector<double>& node_size_y, // after die scale
const std::vector<std::string>& node_name,
const std::tuple<double, double, double, double>& die_info, // after die scale
const std::tuple<double, double>& site_info,
const std::tuple<double, double>& bin_size_info, // after die scale
std::vector<std::tuple<index_type, index_type, std::string>> node_types_indices) {
// init cairo
cairo_surface_t* cs;
if (format == "png") {
cs = cairo_image_surface_create(CAIRO_FORMAT_ARGB32, width, height);
} else if (format == "pdf") {
cs = cairo_pdf_surface_create(filename.c_str(), width, height);
} else if (format == "svg") {
cs = cairo_svg_surface_create(filename.c_str(), width, height);
} else if (format == "eps") {
cs = cairo_ps_surface_create(filename.c_str(), width, height);
} else {
std::cout << "Unsupported file format " << format << ". Please use png/pdf/svg/eps instead." << std::endl;
return false;
}
auto [die_lx, die_hx, die_ly, die_hy] = die_info;
auto [site_width, site_height] = site_info;
auto [bin_width, bin_height] = bin_size_info;
double w_ratio = width / (die_hx - die_lx);
double h_ratio = height / (die_hy - die_ly);
cairo_t* c;
cairo_text_extents_t extents;
cairo_matrix_t font_reflection_matrix;
char buf[32];
c = cairo_create(cs);
cairo_save(c);
cairo_translate(c, 0 - die_lx * w_ratio, height + die_ly * h_ratio);
cairo_scale(c, w_ratio, -h_ratio);
auto set_rgba = [&](std::string type) {
auto [r, g, b, a] = get_colors(type);
cairo_set_source_rgba(c, r, g, b, a);
};
double lineWidth = 0.0001 * std::min(die_hx - die_lx, die_hy - die_ly);
// die
cairo_rectangle(c, die_lx, die_ly, (die_hx - die_lx), (die_hy - die_ly));
set_rgba("Die");
cairo_fill(c);
// boundary
cairo_rectangle(c, die_lx, die_ly, (die_hx - die_lx), (die_hy - die_ly));
cairo_set_line_width(c, lineWidth);
cairo_set_source_rgb(c, 0.1, 0.1, 0.1);
cairo_stroke(c);
// bin / grid
cairo_set_line_width(c, lineWidth);
set_rgba("Bin");
for (double bx = die_lx; bx < die_hx; bx += bin_width) {
cairo_move_to(c, bx, die_ly);
cairo_line_to(c, bx, die_hy);
cairo_stroke(c);
}
for (double by = die_ly; by < die_hy; by += bin_height) {
cairo_move_to(c, die_lx, by);
cairo_line_to(c, die_hx, by);
cairo_stroke(c);
}
cairo_set_line_width(c, lineWidth);
cairo_select_font_face(c, "Sans", CAIRO_FONT_SLANT_NORMAL, CAIRO_FONT_WEIGHT_NORMAL);
if (contents.find("Nodes") != contents.end()) {
int num_nodes = node_pos_x.size();
if (std::get<1>(node_types_indices.back()) <= num_nodes) {
// enable filler
index_type last_idx = std::get<1>(node_types_indices.back());
auto tmp = std::make_tuple(last_idx, num_nodes, "Filler");
node_types_indices.emplace_back(tmp);
}
// draw mov last
std::rotate(node_types_indices.begin(), node_types_indices.begin() + 1, node_types_indices.end());
bool draw_node_text = false;
if (contents.find("NodesText") != contents.end()) {
draw_node_text = true;
}
bool draw_node_bd = false;
if (contents.find("NodesBoundary") != contents.end()) {
draw_node_bd = true;
}
for (auto [start_idx, end_idx, node_type] : node_types_indices) {
// std::cout << node_type << std::endl;
for (index_type i = start_idx; i < end_idx; i++) {
if (i >= num_nodes) {
break;
}
double node_lx = node_pos_x[i] - node_size_x[i] / 2;
double node_ly = node_pos_y[i] - node_size_y[i] / 2;
cairo_rectangle(c, node_lx, node_ly, node_size_x[i], node_size_y[i]);
set_rgba(node_type);
cairo_fill(c);
if (draw_node_bd) {
cairo_rectangle(c, node_lx, node_ly, node_size_x[i], node_size_y[i]);
cairo_set_source_rgb(c, 0.1, 0.1, 0.1);
cairo_stroke(c);
}
if (draw_node_text) {
cairo_matrix_t font_reflection_matrix;
sprintf(buf, "%s", node_name[i].c_str());
double rotate = 0;
double font_size = node_size_y[i] / 5;
if (node_size_x[i] < node_size_y[i]) {
rotate = 3.1415926 / 2;
font_size = node_size_x[i] / 5;
}
cairo_set_font_size(c, font_size);
cairo_set_source_rgb(c, 0.2, 0.2, 0.2);
cairo_get_font_matrix(c, &font_reflection_matrix);
font_reflection_matrix.yy = font_reflection_matrix.yy * -1 * w_ratio / h_ratio;
cairo_matrix_rotate(&font_reflection_matrix, rotate);
cairo_set_font_matrix(c, &font_reflection_matrix);
cairo_text_extents(c, buf, &extents);
cairo_move_to(c,
(node_lx + node_size_x[i] / 2) - (extents.width / 2 + extents.x_bearing),
(node_ly + node_size_y[i] / 2) - (extents.height / 2 + extents.y_bearing));
cairo_show_text(c, buf);
}
}
}
}
cairo_restore(c);
cairo_show_page(c);
cairo_destroy(c);
// destory cairo
cairo_surface_flush(cs);
if (format == "png") cairo_surface_write_to_png(cs, filename.c_str());
cairo_surface_destroy(cs);
return true;
}

View File

@ -0,0 +1,27 @@
#include "global.h"
class Drawer {
private:
std::unordered_map<std::string, std::tuple<double, double, double, double>> type_to_rgba;
std::string format = "";
std::string filename = "";
double width = 0.0;
double height = 0.0;
std::unordered_set<std::string> contents;
public:
Drawer(const std::vector<std::tuple<std::string, double, double, double, double>>& ele_type_to_rgba_vec,
const std::string& filename_,
double width_,
double height_,
const std::vector<std::string>& contents_);
std::tuple<double, double, double, double> get_colors(const std::string& ele_type);
bool run(const std::vector<double>& node_pos_x,
const std::vector<double>& node_pos_y,
const std::vector<double>& node_size_x,
const std::vector<double>& node_size_y,
const std::vector<std::string>& node_name,
const std::tuple<double, double, double, double>& die_info,
const std::tuple<double, double>& site_info,
const std::tuple<double, double>& bin_size_info,
std::vector<std::tuple<index_type, index_type, std::string>> node_types_indices);
};

View File

@ -0,0 +1,32 @@
#include "Drawer.h"
bool DrawGlobalPlacement(
const std::vector<double>& node_pos_x,
const std::vector<double>& node_pos_y,
const std::vector<double>& node_size_x,
const std::vector<double>& node_size_y,
const std::vector<std::string>& node_name,
const std::tuple<double, double, double, double>& die_info,
const std::tuple<double, double>& site_info,
const std::tuple<double, double>& bin_size_info,
const std::vector<std::tuple<index_type, index_type, std::string>>& node_types_indices,
const std::vector<std::tuple<std::string, double, double, double, double>>& ele_type_to_rgba_vec,
const std::string& filename,
double width,
double height,
const std::vector<std::string>& draw_contents) {
Drawer drawer(ele_type_to_rgba_vec, filename, width, height, draw_contents);
bool status = drawer.run(node_pos_x,
node_pos_y,
node_size_x,
node_size_y,
node_name,
die_info,
site_info,
bin_size_info,
node_types_indices);
return status;
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) { m.def("draw", &DrawGlobalPlacement, "Draw placement results"); }

View File

@ -0,0 +1,33 @@
#pragma once
// STL libraries
#include <fstream>
#include <iostream>
#include <sstream>
#include <iomanip>
#include <numeric>
#include <stack>
#include <string>
#include <unordered_map>
#include <unordered_set>
#include <vector>
// Torch library
#include <torch/extension.h>
// Pybind11
#include <pybind11/pybind11.h>
#include <pybind11/stl.h>
#include <pybind11/stl_bind.h>
#include <pybind11/numpy.h>
namespace py = pybind11;
// Cairo
#include <cairo-pdf.h>
#include <cairo-ps.h>
#include <cairo-svg.h>
#include <cairo.h>
using index_type = int64_t;

View File

@ -0,0 +1,15 @@
set(TARGET_NAME flute_cpp)
pybind11_add_module(${TARGET_NAME} MODULE ${TARGET_NAME}.cpp)
target_include_directories(${TARGET_NAME} PRIVATE ${TORCH_INCLUDE_DIRS} ${FLUTE_INCLUDE_DIR})
target_link_libraries(
${TARGET_NAME} PRIVATE torch ${TORCH_PYTHON_LIBRARY} flute)
target_compile_definitions(${TARGET_NAME} PRIVATE
TORCH_EXTENSION_NAME=${TARGET_NAME}
TORCH_MAJOR_VERSION=${TORCH_MAJOR_VERSION}
TORCH_MINOR_VERSION=${TORCH_MINOR_VERSION}
ENABLE_CUDA=0)
install(TARGETS
${TARGET_NAME}
DESTINATION ${XPLACE_LIB_DIR})

View File

@ -0,0 +1,95 @@
#include <torch/extension.h>
#include <mutex>
#include <thread>
#include <vector>
#include "flute.h"
// mt flute, DTYPE is int
void runJobsMT(int numJobs, int num_threads, const std::function<void(int)> &handle) {
int numThreads = std::min(numJobs, num_threads);
if (numThreads <= 1) {
for (int i = 0; i < numJobs; ++i) {
handle(i);
}
} else {
int globalJobIdx = 0;
std::mutex mtx;
auto thread_func = [&](int threadIdx) {
int jobIdx;
while (true) {
mtx.lock();
jobIdx = globalJobIdx++;
mtx.unlock();
if (jobIdx >= numJobs) {
break;
}
handle(jobIdx);
}
};
std::thread threads[numThreads];
for (int i = 0; i < numThreads; i++) {
threads[i] = std::thread(thread_func, i);
}
for (int i = 0; i < numThreads; i++) {
threads[i].join();
}
}
}
int FluteRSMTWL(const std::vector<int> &xsvec, const std::vector<int> &ysvec) {
int degree = xsvec.size();
int rsmt_wl = 0;
if (degree > 1) {
Flute::DTYPE xs[5 * degree];
Flute::DTYPE ys[5 * degree];
for (int pt_cnt = 0; pt_cnt < degree; pt_cnt++) {
xs[pt_cnt] = xsvec[pt_cnt];
ys[pt_cnt] = ysvec[pt_cnt];
}
rsmt_wl = Flute::flute_wl(degree, xs, ys, FLUTE_ACCURACY);
}
return rsmt_wl;
}
std::vector<int> MultiNetsFluteRSMTWL(const std::vector<int> &pos_x,
const std::vector<int> &pos_y,
const std::vector<int64_t> &hyperedge_list,
const std::vector<int64_t> &hyperedge_list_end,
const int num_threads) {
int num_hyperedges = hyperedge_list_end.size();
std::vector<int> nets_rmst(num_hyperedges, 0.0);
std::function<void(int)> singleNetFluteWL = [&](int i) {
if (i >= num_hyperedges) return;
int rsmt_wl = 0.0;
int64_t start_idx = 0;
if (i != 0) {
start_idx = hyperedge_list_end[i - 1];
}
int64_t end_idx = hyperedge_list_end[i];
int64_t degree = end_idx - start_idx;
if (degree > 1) {
Flute::DTYPE xs[5 * degree];
Flute::DTYPE ys[5 * degree];
int64_t pt_cnt = 0;
for (int64_t idx = start_idx; idx < end_idx; idx++) {
int64_t pos_idx = hyperedge_list[idx];
xs[pt_cnt] = pos_x[pos_idx];
ys[pt_cnt] = pos_y[pos_idx];
pt_cnt++;
}
rsmt_wl = Flute::flute_wl(degree, xs, ys, FLUTE_ACCURACY);
}
nets_rmst[i] = rsmt_wl;
};
runJobsMT(num_hyperedges, num_threads, singleNetFluteWL);
return nets_rmst;
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
m.def("flute_rsmt_wl", &FluteRSMTWL, "Get Flute wirelength");
m.def("flute_rsmt_wl_mt", &MultiNetsFluteRSMTWL, "Get Multi Nets Flute wirelength (MT Version)");
m.def("read_lut", &Flute::readLUT, "Read Flute LUT");
}

View File

@ -0,0 +1,9 @@
set(TARGET_NAME hpwl_cuda)
add_pytorch_extension(${TARGET_NAME}
${TARGET_NAME}.cpp
${TARGET_NAME}_kernel.cu)
install(TARGETS
${TARGET_NAME}
DESTINATION ${XPLACE_LIB_DIR})

View File

@ -0,0 +1,19 @@
#include <torch/extension.h>
torch::Tensor hpwl_cuda(torch::Tensor pos, torch::Tensor hyperedge_list, torch::Tensor hyperedge_list_end);
#define CHECK_CUDA(x) TORCH_CHECK(x.device().is_cuda(), #x " must be a CUDA tensor")
#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
#define CHECK_INPUT(x) \
CHECK_CUDA(x); \
CHECK_CONTIGUOUS(x)
torch::Tensor hpwl(torch::Tensor pos, torch::Tensor hyperedge_list, torch::Tensor hyperedge_list_end) {
CHECK_INPUT(pos);
CHECK_INPUT(hyperedge_list);
CHECK_INPUT(hyperedge_list_end);
return hpwl_cuda(pos, hyperedge_list, hyperedge_list_end);
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) { m.def("hpwl", &hpwl, "get HPWL wirelength from hyperedge"); }

View File

@ -0,0 +1,60 @@
#include <ATen/cuda/CUDAContext.h>
#include <cuda.h>
#include <cuda_runtime.h>
#include <torch/extension.h>
#include <vector>
template <typename scalar_t>
__global__ void hpwl_cuda_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> pin_pos,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list_end,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> partial_hpwl,
int num_nets) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int i = index >> 1; // pin index
if (i < num_nets) {
const int c = index & 1; // channel index
int64_t start_idx = 0;
if (i != 0) {
start_idx = hyperedge_list_end[i - 1];
}
int64_t end_idx = hyperedge_list_end[i];
partial_hpwl[i][c] = 0;
if (end_idx != start_idx) {
int64_t pin_id = hyperedge_list[start_idx];
scalar_t x_min = pin_pos[pin_id][c];
scalar_t x_max = pin_pos[pin_id][c];
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
scalar_t xx = pin_pos[hyperedge_list[idx]][c];
x_min = min(xx, x_min);
x_max = max(xx, x_max);
}
partial_hpwl[i][c] = abs(x_max - x_min);
}
}
}
torch::Tensor hpwl_cuda(torch::Tensor pos, torch::Tensor hyperedge_list, torch::Tensor hyperedge_list_end) {
cudaSetDevice(pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
const auto num_nets = hyperedge_list_end.size(0);
const int num_channels = 2;
auto partial_hpwl = torch::zeros({num_nets, num_channels}, torch::dtype(pos.dtype()).device(pos.device()));
const int threads = 64;
const int blocks = (num_nets * 2 + threads - 1) / threads;
AT_DISPATCH_ALL_TYPES(pos.scalar_type(), "hpwl_cuda", ([&] {
hpwl_cuda_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
hyperedge_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
hyperedge_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
partial_hpwl.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_nets);
}));
return partial_hpwl;
}

View File

@ -0,0 +1,72 @@
#include "BindHelper.h"
namespace Xplace {
void bindGPDatabase(pybind11::module& m) {
pybind11::bind_vector<std::vector<bool>>(m, "BoolList");
pybind11::bind_vector<std::vector<gp::coord_type>>(m, "CoordList");
pybind11::bind_vector<std::vector<gp::index_type>>(m, "IndexList");
pybind11::class_<gp::Basic>(m, "Basic")
.def(pybind11::init<>())
.def("id", &gp::Basic::getId)
.def("name", &gp::Basic::getName)
.def("__repr__", &gp::Basic::getName);
pybind11::class_<gp::GPNode, gp::Basic>(m, "Node")
.def(pybind11::init<>())
.def("lx", &gp::GPNode::getLx)
.def("ly", &gp::GPNode::getLy)
.def("hx", &gp::GPNode::getHx)
.def("hy", &gp::GPNode::getHy)
.def("width", &gp::GPNode::getWidth)
.def("height", &gp::GPNode::getHeight)
.def("orient", &gp::GPNode::getOrient)
.def("node_type", &gp::GPNode::getNodeType)
.def("pinIds", &gp::GPNode::pins);
pybind11::bind_vector<std::vector<gp::GPNode>>(m, "NodeList");
pybind11::class_<gp::GPPin, gp::Basic>(m, "Pin")
.def(pybind11::init<>())
.def("rel_lx", &gp::GPPin::getRelLx)
.def("rel_ly", &gp::GPPin::getRelLy)
.def("width", &gp::GPPin::getWidth)
.def("height", &gp::GPPin::getHeight)
.def("direction", &gp::GPPin::getDirection)
.def("pin_type", &gp::GPPin::getType)
.def("nodeId", &gp::GPPin::getParNodeId)
.def("netId", &gp::GPPin::getParNetId);
pybind11::bind_vector<std::vector<gp::GPPin>>(m, "PinList");
pybind11::class_<gp::GPNet, gp::Basic>(m, "Net").def(pybind11::init<>()).def("pinIds", &gp::GPNet::pins);
pybind11::bind_vector<std::vector<gp::GPNet>>(m, "NetList");
pybind11::class_<gp::GPDatabase, std::shared_ptr<gp::GPDatabase>>(m, "GPDatabase")
.def(pybind11::init<std::shared_ptr<db::Database>>())
.def("setup", &gp::GPDatabase::setup)
.def("reset", &gp::GPDatabase::reset)
.def("nodes", &gp::GPDatabase::getNodes) // NOTE: using py::return_value_policy::reference is dangerous
.def("pins", &gp::GPDatabase::getPins)
.def("nets", &gp::GPDatabase::getNets)
.def("dieInfo", &gp::GPDatabase::getDieInfo, py::return_value_policy::copy) // dieLX, dieHX, dieLY, dieHY
.def("coreInfo", &gp::GPDatabase::getCoreInfo, py::return_value_policy::copy) // coreLX, coreHX, coreLY, coreHY
.def("siteWidth", &gp::GPDatabase::getSiteWidth, py::return_value_policy::copy)
.def("siteHeight", &gp::GPDatabase::getSiteHeight, py::return_value_policy::copy)
.def("m1direction", &gp::GPDatabase::getM1Direction, py::return_value_policy::copy)
.def("node_type_indices", &gp::GPDatabase::getNodeTypeIndices, py::return_value_policy::copy)
.def("node_id2node_name", &gp::GPDatabase::getNodeId2NodeName, py::return_value_policy::copy)
.def("node_lpos_tensor", &gp::GPDatabase::getNodeLPosTensor, py::return_value_policy::move)
.def("node_cpos_tensor", &gp::GPDatabase::getNodeCPosTensor, py::return_value_policy::move)
.def("node_size_tensor", &gp::GPDatabase::getNodeSizeTensor, py::return_value_policy::move)
.def("pin_rel_lpos_tensor", &gp::GPDatabase::getPinRelLPosTensor, py::return_value_policy::move)
.def("pin_rel_cpos_tensor", &gp::GPDatabase::getPinRelCPosTensor, py::return_value_policy::move)
.def("pin_size_tensor", &gp::GPDatabase::getPinSizeTensor, py::return_value_policy::move)
.def("pin_id2node_id_tensor", &gp::GPDatabase::getPinId2NodeIdTensor, py::return_value_policy::move)
.def("pin_id2net_id_tensor", &gp::GPDatabase::getPinId2NetIdTensor, py::return_value_policy::move)
.def("hyperedge_info_tensor", &gp::GPDatabase::getHyperedgeInfoTensor, py::return_value_policy::move)
.def("region_info_tensor", &gp::GPDatabase::getRegionInfoTensor, py::return_value_policy::move)
.def("apply_node_pos", &gp::GPDatabase::applyNodePos)
.def("write_placement", &gp::GPDatabase::writePlacement);
}
} // namespace Xplace

View File

@ -0,0 +1,12 @@
#pragma once
#include "common/common.h"
#include "common/db/Database.h"
#include "io_parser/gp/GPDatabase.h"
namespace py = pybind11;
namespace Xplace {
void bindGPDatabase(pybind11::module& m);
} // namespace Xplace

View File

@ -0,0 +1,17 @@
file(GLOB_RECURSE SRC_FILES_IO_PARSER ${CMAKE_CURRENT_SOURCE_DIR}/*.cpp)
pybind11_add_module(io_parser SHARED ${SRC_FILES_IO_PARSER})
target_include_directories(
io_parser PUBLIC ${PROJECT_SOURCE_DIR}/cpp_to_py ${TORCH_INCLUDE_DIRS})
target_link_libraries(
io_parser PRIVATE torch ${TORCH_PYTHON_LIBRARY} xplace_common)
target_compile_definitions(io_parser PRIVATE
TORCH_EXTENSION_NAME=io_parser
TORCH_MAJOR_VERSION=${TORCH_MAJOR_VERSION}
TORCH_MINOR_VERSION=${TORCH_MINOR_VERSION}
ENABLE_CUDA=0)
target_compile_options(io_parser PRIVATE -fPIC)
install(TARGETS
io_parser
DESTINATION ${XPLACE_LIB_DIR})

View File

@ -0,0 +1,102 @@
#include "BindHelper.h"
namespace Xplace {
bool loadParams(const py::dict& kwargs) {
db::setting.reset();
// ----- design related options -----
if (kwargs.contains("bookshelf_variety")) {
db::setting.BookshelfVariety = kwargs["bookshelf_variety"].cast<std::string>();
}
if (kwargs.contains("aux")) {
db::setting.BookshelfAux = kwargs["aux"].cast<std::string>();
}
if (kwargs.contains("pl")) {
db::setting.BookshelfPl = kwargs["pl"].cast<std::string>();
}
if (kwargs.contains("def")) {
db::setting.DefFile = kwargs["def"].cast<std::string>();
}
if (kwargs.contains("lef")) {
db::setting.LefFile = kwargs["lef"].cast<std::string>();
} else if (kwargs.contains("cell_lef") && kwargs.contains("tech_lef")) {
db::setting.LefCell = kwargs["cell_lef"].cast<std::string>();
db::setting.LefTech = kwargs["tech_lef"].cast<std::string>();
}
if (kwargs.contains("constraints")) {
db::setting.Constraints = kwargs["constraints"].cast<std::string>();
}
if (kwargs.contains("output")) {
db::setting.OutputFile = kwargs["output"].cast<std::string>();
}
if (db::setting.BookshelfAux == "" && db::setting.DefFile == "") {
printlog(LOG_ERROR, "design is not found");
return false;
}
// verilog is unused now
// if (kwargs.contains("verilog")) {
// db::setting.Verilog = kwargs["verilog"].cast<std::string>();
// }
// ----- other options -----
// db loading mode
if (kwargs.contains("lite_mode")) {
db::setting.liteMode = kwargs["lite_mode"].cast<bool>();
}
// enable random place or not
if (kwargs.contains("random_place")) {
db::setting.random_place = kwargs["random_place"].cast<bool>();
}
// verbose on/off in parser, default is verbose off(false)
if (kwargs.contains("verbose_parser_log")) {
utils::verbose_parser_log = kwargs["verbose_parser_log"].cast<bool>();
} else {
utils::verbose_parser_log = false;
}
if (kwargs.contains("num_threads")) {
db::setting.numThreads = kwargs["num_threads"].cast<int>();
}
return true;
}
std::tuple<std::shared_ptr<db::Database>, std::shared_ptr<gp::GPDatabase>> start_all(const py::dict& kwargs) {
bool load_status = loadParams(kwargs);
if (!load_status) {
throw std::invalid_argument("Received invalid params. Please check!");
}
auto rawdb_ptr = std::make_shared<db::Database>();
rawdb_ptr->load();
rawdb_ptr->setup();
auto gpdb_ptr = std::make_shared<gp::GPDatabase>(rawdb_ptr);
gpdb_ptr->setup();
return {rawdb_ptr, gpdb_ptr};
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
pybind11::class_<db::Database, std::shared_ptr<db::Database>>(m, "Database")
.def(pybind11::init<>())
.def("load", &db::Database::load)
.def("setup", &db::Database::setup)
.def("reset", &db::Database::reset);
bindGPDatabase(m);
m.def("load_params", &loadParams, "Parse input args to DB and return graph information");
m.def("create_database", []() { return std::make_shared<db::Database>(); });
m.def("create_gpdatabase", [](std::shared_ptr<db::Database> db) { return std::make_shared<gp::GPDatabase>(db); });
m.def("start", &start_all, "Parse input args to DB and return pointer from raw db and gp db");
}
} // namespace Xplace

View File

@ -0,0 +1,550 @@
#include "GPDatabase.h"
namespace gp {
GPDatabase::~GPDatabase() { printlog(LOG_INFO, "destruct gpdb"); }
void GPDatabase::addCellNode(index_type cell_id, std::string& node_type) {
auto cell = database.cells[cell_id];
nodes.emplace_back(GPNode());
GPNode& node = nodes.back();
node.setId(nodes.size() - 1);
node.setName(cell->name());
node.setLx(cell->lx());
node.setLy(cell->ly());
node.setWidth(cell->width());
node.setHeight(cell->height());
node.setOrient(cell->orient());
node.setNodeType(node_type);
node.setOriDBId(cell_id);
if (node_type == "Mov" || node_type == "FloatMov") {
node.setRegionId(static_cast<index_type>(cell->region->id));
regions[node.getRegionId()].addNode(node.getId());
}
if (cell->fixed() && cell->ctype()->nonRegularRects().size() > 0) {
node.setIsPolygonShape(true);
}
cell->gpdb_id = nodes.size() - 1;
}
void GPDatabase::addIOPinNode(index_type iopin_id, std::string& node_type) {
auto iopin = database.iopins[iopin_id];
nodes.emplace_back(GPNode());
GPNode& node = nodes.back();
node.setId(nodes.size() - 1);
node.setName(iopin->name);
node.setLx(iopin->x);
node.setLy(iopin->y);
node.setWidth(iopin->width());
node.setHeight(iopin->height());
node.setOrient(iopin->orient());
node.setNodeType(node_type);
node.setOriDBId(iopin_id);
iopin->gpdb_id = nodes.size() - 1;
}
void GPDatabase::addBlockageNode(index_type blkg_id, std::string& node_type) {
auto& blockage = database.placeBlockages[blkg_id];
std::string blockage_name = "Blockage_" + std::to_string(blkg_id);
nodes.emplace_back(GPNode());
GPNode& node = nodes.back();
node.setId(nodes.size() - 1);
node.setName(blockage_name);
node.setLx(blockage.lx);
node.setLy(blockage.ly);
node.setWidth(blockage.w());
node.setHeight(blockage.h());
node.setOrient(-1); // No orientation for placement blockage
node.setNodeType(node_type);
node.setOriDBId(blkg_id);
}
void GPDatabase::addNet(index_type dbnet_id) {
auto dbnet = database.nets[dbnet_id];
nets.emplace_back(GPNet());
GPNet& net = nets.back();
net.setId(nets.size() - 1);
net.setName(dbnet->name);
net.setOriDBId(dbnet_id);
dbnet->gpdb_id = nets.size() - 1;
for (auto dbpin : dbnet->pins) {
assert_msg(dbpin->is_connected, "Pin is not connected!");
if (dbpin->iopin != nullptr) {
auto& node = nodes.at(dbpin->iopin->gpdb_id);
addPin(dbpin, dbpin->iopin->type, node, net, true);
} else {
auto& node = nodes.at(dbpin->cell->gpdb_id);
addPin(dbpin, dbpin->type, node, net, false);
}
}
}
void GPDatabase::addPin(db::Pin* dbpin, const db::PinType* pintype, GPNode& node, GPNet& net, bool isIOPin) {
pins.emplace_back(GPPin());
GPPin& pin = pins.back();
pin.setId(pins.size() - 1);
pin.setRelLx(pintype->boundLX);
pin.setRelLy(pintype->boundLY);
pin.setWidth(pintype->getW());
pin.setHeight(pintype->getH());
pin.setDirection(pintype->direction());
pin.setType(pintype->type());
pin.setParNodeId(node.getId());
pin.setParNetId(net.getId());
pin.setOriDBInfo({node.getOriDBId(), isIOPin ? -1 : dbpin->parentCellPinId, net.getOriDBId()});
node.addPin(pin.getId());
net.addPin(pin.getId());
dbpin->gpdb_id = pins.size() - 1;
}
void GPDatabase::addRegion(index_type dbregion_id) {
auto dbregion = database.regions[dbregion_id];
regions.emplace_back(GPRegion());
GPRegion& region = regions.back();
region.setId(regions.size() - 1);
region.setName(dbregion->name());
region.setOriDBId(dbregion_id);
assert_msg(static_cast<index_type>(dbregion->id) == dbregion_id,
"Set Region ID incorrectly! Please check origin DB");
assert_msg(dbregion_id == region.getId(), "Set Region ID incorrectly! Please check GPDatabase");
region.setType(dbregion->type());
for (auto& rect : dbregion->rects) {
box_type box(rect.lx, rect.ly, rect.hx, rect.hy);
region.addBox(box);
}
}
point_type GPDatabase::getAbsolutePinPos(const GPPin& pin) const {
coord_type absX, absY;
auto& parNode = this->nodes.at(pin.getParNodeId());
absX = pin.getRelLx() + parNode.getLx();
absY = pin.getRelLy() + parNode.getLy();
return std::make_pair(absX, absY);
}
void GPDatabase::setupNum() {
dieInfo = std::make_tuple(database.dieLX, database.dieHX, database.dieLY, database.dieHY);
coreInfo = std::make_tuple(database.coreLX, database.coreHX, database.coreLY, database.coreHY);
siteW = static_cast<int>(database.siteW);
siteH = database.siteH;
num_nodes = database.cells.size() + database.iopins.size() + database.placeBlockages.size();
num_nets = database.nets.size();
num_pins = 0;
for (auto& dbnet : database.nets) {
num_pins += dbnet->pins.size();
}
num_regions = database.regions.size();
nodes.reserve(num_nodes);
pins.reserve(num_pins);
nets.reserve(num_nets);
pin_id2node_id.reserve(num_pins);
pin_id2net_id.reserve(num_pins);
}
void GPDatabase::setupNodes() {
// preprocess nodes in database
std::vector<index_type> all_mov_ids;
all_mov_ids.reserve(database.cells.size());
std::vector<index_type> all_fix_ids;
all_fix_ids.reserve(database.cells.size());
std::vector<index_type> all_iopin_ids;
all_iopin_ids.reserve(database.iopins.size());
std::vector<index_type> all_float_iopin_ids;
all_float_iopin_ids.reserve(database.iopins.size());
std::vector<index_type> all_float_fix_ids;
all_float_fix_ids.reserve(database.cells.size());
std::vector<index_type> all_float_mov_ids;
all_float_mov_ids.reserve(database.cells.size());
for (index_type i = 0; i < static_cast<index_type>(database.cells.size()); i++) {
auto cell = database.cells[i];
if (cell->is_connected) {
if (!cell->fixed()) {
all_mov_ids.emplace_back(i);
} else {
all_fix_ids.emplace_back(i);
}
} else {
if (!cell->fixed()) {
all_float_mov_ids.emplace_back(i);
} else {
all_float_fix_ids.emplace_back(i);
}
}
}
for (index_type i = 0; i < static_cast<index_type>(database.iopins.size()); i++) {
auto iopin = database.iopins[i];
if (iopin->is_connected) {
all_iopin_ids.emplace_back(i);
} else {
all_float_iopin_ids.emplace_back(i);
}
}
std::string cur_node_type = "Mov";
index_type stard_idx = nodes.size();
for (index_type cell_id : all_mov_ids) {
addCellNode(cell_id, cur_node_type);
}
index_type end_idx = nodes.size();
node_types_indices.emplace_back(std::make_tuple(stard_idx, end_idx, cur_node_type));
cur_node_type = "FloatMov";
stard_idx = nodes.size();
for (index_type cell_id : all_float_mov_ids) {
addCellNode(cell_id, cur_node_type);
}
end_idx = nodes.size();
node_types_indices.emplace_back(std::make_tuple(stard_idx, end_idx, cur_node_type));
cur_node_type = "Fix";
stard_idx = nodes.size();
for (index_type cell_id : all_fix_ids) {
addCellNode(cell_id, cur_node_type);
}
end_idx = nodes.size();
node_types_indices.emplace_back(std::make_tuple(stard_idx, end_idx, cur_node_type));
cur_node_type = "IOPin";
stard_idx = nodes.size();
for (index_type iopin_id : all_iopin_ids) {
addIOPinNode(iopin_id, cur_node_type);
}
end_idx = nodes.size();
node_types_indices.emplace_back(std::make_tuple(stard_idx, end_idx, cur_node_type));
cur_node_type = "Blkg";
stard_idx = nodes.size();
for (index_type blkg_id = 0; blkg_id < static_cast<index_type>(database.placeBlockages.size()); blkg_id++) {
addBlockageNode(blkg_id, cur_node_type);
}
end_idx = nodes.size();
node_types_indices.emplace_back(std::make_tuple(stard_idx, end_idx, cur_node_type));
cur_node_type = "FloatIOPin";
stard_idx = nodes.size();
for (index_type iopin_id : all_float_iopin_ids) {
addIOPinNode(iopin_id, cur_node_type);
}
end_idx = nodes.size();
node_types_indices.emplace_back(std::make_tuple(stard_idx, end_idx, cur_node_type));
cur_node_type = "FloatFix";
stard_idx = nodes.size();
for (index_type cell_id : all_float_fix_ids) {
addCellNode(cell_id, cur_node_type);
}
end_idx = nodes.size();
node_types_indices.emplace_back(std::make_tuple(stard_idx, end_idx, cur_node_type));
}
void GPDatabase::setupNets() {
// setup nets and their pins
for (index_type dbnet_id = 0; dbnet_id < static_cast<index_type>(database.nets.size()); dbnet_id++) {
addNet(dbnet_id);
}
}
void GPDatabase::setupRegions() {
// setup regions
// Make sure this step done before setupNodes
for (index_type dbregion_id = 0; dbregion_id < static_cast<index_type>(database.regions.size()); dbregion_id++) {
addRegion(dbregion_id);
}
}
void GPDatabase::setupIndexMap() {
for (auto& pin : pins) {
const GPNode& node = nodes.at(pin.getParNodeId());
const GPNet& net = nets.at(pin.getParNetId());
pin_id2node_id.emplace_back(node.getId());
pin_id2net_id.emplace_back(net.getId());
}
for (auto& node: nodes) {
node_id2node_name.emplace_back(node.getName());
}
}
void GPDatabase::setupCheckVar() {
assert_msg(nodes.size() == num_nodes, "Nodes size is (%d), it should be (%d)", nodes.size(), num_nodes);
assert_msg(nets.size() == num_nets, "Nets size is (%d), it should be (%d)", nets.size(), num_nets);
assert_msg(pins.size() == num_pins, "Pins size is (%d), it should be (%d)", pins.size(), num_pins);
printlog(LOG_VERBOSE, "Finish checking");
// check whether nodes and pins are placed out of boundary or not
// TODO change log_debug to log_warn
// utils::BoxT<coord_type> die_box(database.dieLX, database.dieLY, database.dieHX, database.dieHY);
// for (auto& node : nodes) {
// utils::BoxT<coord_type> node_box(node.getLx(), node.getLy(), node.getHx(), node.getHy());
// auto union_box = node_box.UnionWith(die_box);
// if (union_box != die_box) {
// log(LOG_WARN) << "Node " << node.getName() << "'s (lx,ly,hx,hy) = (" << node.getLx() << ", " <<
// node.getLy()
// << ", " << node.getHx() << ", " << node.getHy() << ")"
// << " is placed outside die boundary" << std::endl;
// }
// }
// for (auto& pin : pins_abs_info) {
// int index = &pin - &pins_abs_info[0];
// utils::BoxT<coord_type> pin_box(pin[0], pin[1], pin[0] + pin[2], pin[1] + pin[3]); // lx, ly, hx, hy
// auto union_box = pin_box.UnionWith(die_box);
// if (union_box != die_box) {
// log(LOG_WARN) << "Node(" << nodes[pins.at(index).getParNodeId()].getName() << ")'s pin on (lx,ly,hx,hy) =
// ("
// << pin[0] << ", " << pin[1] << ", " << pin[0] + pin[2] << ", " << pin[1] + pin[3] << ")"
// << " is outside die boundary" << std::endl;
// }
// }
}
bool GPDatabase::setup() {
if (db::setting.random_place) {
setup_random_place();
printlog(LOG_INFO, "random place db done");
}
setupNum();
setupRegions();
setupNodes();
setupNets();
setupIndexMap();
setupCheckVar();
printlog(LOG_INFO, "Finish initializing global placement database");
return true;
}
bool GPDatabase::reset() {
num_nodes = 0;
num_pins = 0;
num_nets = 0;
nodes.clear();
pins.clear();
nets.clear();
node_types_indices.clear();
pin_id2node_id.clear();
pin_id2net_id.clear();
nodes_id2pins_id.clear();
nets_id2pins_id.clear();
return true;
}
void GPDatabase::setup_random_place() {
int max_pin_width = std::numeric_limits<int>::min();
int max_pin_height = std::numeric_limits<int>::min();
int max_cell_width = std::numeric_limits<int>::min();
int max_cell_height = std::numeric_limits<int>::min();
for (auto cell : database.cells) {
if (!cell->fixed()) {
max_cell_width = std::max(max_cell_width, cell->ctype()->width);
max_cell_height = std::max(max_cell_height, cell->ctype()->height);
for (auto pintype : cell->ctype()->pins) {
max_pin_width = std::max(max_pin_width, int(pintype->getW()));
max_pin_height = std::max(max_pin_height, int(pintype->getH()));
}
}
}
int dieLXLegalPlace = database.dieLX + max_pin_width + max_cell_width + 1;
int dieHXLegalPlace = database.dieHX - max_pin_width - max_cell_width - 1;
int dieLYLegalPlace = database.dieLY + max_pin_height + max_cell_height + 1;
int dieHYLegalPlace = database.dieHY - max_pin_height - max_cell_height - 1;
// Non-deterministic
// std::random_device rdx;
// std::random_device rdy;
// std::mt19937 genX(rdx());
// std::mt19937 genY(rdy());
std::mt19937 genX(123);
std::mt19937 genY(567);
std::uniform_int_distribution<> xDis(dieLXLegalPlace, dieHXLegalPlace);
std::uniform_int_distribution<> yDis(dieLYLegalPlace, dieHYLegalPlace);
for (auto cell : database.cells) {
if (!cell->fixed()) {
int x = xDis(genX);
int y = yDis(genY);
cell->place(x, y);
}
}
}
// Torch Related
torch::Tensor GPDatabase::getNodeLPosTensor() {
torch::Tensor node_lpos = torch::zeros({num_nodes, 2});
auto node_lpos_a = node_lpos.accessor<coord_type, 2>();
for (auto& node : nodes) {
node_lpos_a[node.getId()][0] = node.getLx();
node_lpos_a[node.getId()][1] = node.getLy();
}
return node_lpos;
}
torch::Tensor GPDatabase::getNodeCPosTensor() {
torch::Tensor node_cpos = torch::zeros({num_nodes, 2});
auto node_cpos_a = node_cpos.accessor<coord_type, 2>();
for (auto& node : nodes) {
node_cpos_a[node.getId()][0] = node.getLx() + node.getWidth() / 2;
node_cpos_a[node.getId()][1] = node.getLy() + node.getHeight() / 2;
}
return node_cpos;
}
torch::Tensor GPDatabase::getNodeSizeTensor() {
torch::Tensor node_size = torch::zeros({num_nodes, 2});
auto node_size_a = node_size.accessor<coord_type, 2>();
for (auto& node : nodes) {
if (!node.getIsPolygonShape()) {
node_size_a[node.getId()][0] = node.getWidth();
node_size_a[node.getId()][1] = node.getHeight();
} else {
// ICCAD/DAC 2012 contain fixed polygon-shape nodes. We consider them as placement
// blockages instead of fixed nodes. So we set their width and height as zeros to
// avoid duplicated density calculation.
}
}
return node_size;
}
torch::Tensor GPDatabase::getPinRelLPosTensor() {
// pin_pos == (node_pos - node_size / 2) + (pin_rel_lpos + pin_size / 2)
torch::Tensor pin_rel_lpos = torch::zeros({num_pins, 2});
auto pin_rel_lpos_a = pin_rel_lpos.accessor<coord_type, 2>();
for (auto& pin : pins) {
pin_rel_lpos_a[pin.getId()][0] = pin.getRelLx();
pin_rel_lpos_a[pin.getId()][1] = pin.getRelLy();
}
return pin_rel_lpos;
}
torch::Tensor GPDatabase::getPinRelCPosTensor() {
// pin_pos == node_pos + pin_rel_cpos
torch::Tensor pin_rel_cpos = torch::zeros({num_pins, 2});
auto pin_rel_cpos_a = pin_rel_cpos.accessor<coord_type, 2>();
for (auto& pin : pins) {
auto& node = nodes[pin.getParNodeId()];
pin_rel_cpos_a[pin.getId()][0] = pin.getRelLx() + pin.getWidth() / 2 - node.getWidth() / 2;
pin_rel_cpos_a[pin.getId()][1] = pin.getRelLy() + pin.getHeight() / 2 - node.getHeight() / 2;
}
return pin_rel_cpos;
}
torch::Tensor GPDatabase::getPinSizeTensor() {
torch::Tensor pin_size = torch::zeros({num_pins, 2});
auto pin_size_a = pin_size.accessor<coord_type, 2>();
for (auto& pin : pins) {
pin_size_a[pin.getId()][0] = pin.getWidth();
pin_size_a[pin.getId()][1] = pin.getHeight();
}
return pin_size;
}
torch::Tensor GPDatabase::getPinId2NodeIdTensor() {
auto options = torch::TensorOptions().dtype(torch::kInt64);
return torch::from_blob(pin_id2node_id.data(), {num_pins}, options).clone();
}
torch::Tensor GPDatabase::getPinId2NetIdTensor() {
auto options = torch::TensorOptions().dtype(torch::kInt64);
return torch::from_blob(pin_id2net_id.data(), {num_pins}, options).clone();
}
std::vector<torch::Tensor> GPDatabase::getHyperedgeInfoTensor() {
auto options = torch::TensorOptions().dtype(torch::kInt64);
torch::Tensor hyperedge_index_helper = torch::zeros({num_pins}, options);
torch::Tensor hyperedge_list = torch::zeros({num_pins}, options);
torch::Tensor hyperedge_list_end = torch::zeros({num_nets}, options);
auto hyperedge_index_helper_a = hyperedge_index_helper.accessor<index_type, 1>();
auto hyperedge_list_a = hyperedge_list.accessor<index_type, 1>();
auto hyperedge_list_end_a = hyperedge_list_end.accessor<index_type, 1>();
index_type ptr = 0;
index_type last_idx = 0;
for (auto& net : nets) {
index_type net_id = net.getId();
for (auto pin_id : net.pins()) {
hyperedge_list_a[ptr] = pin_id;
hyperedge_index_helper_a[ptr] = net_id;
ptr += 1;
}
last_idx += net.pins().size();
hyperedge_list_end_a[net_id] = last_idx;
}
auto hyperedge_index = torch::cat({hyperedge_list.unsqueeze(0), hyperedge_index_helper.unsqueeze(0)}, 0);
auto new_order_idx = torch::argsort(hyperedge_index.index({0}), 0);
hyperedge_index = hyperedge_index.index({torch::indexing::Slice(), new_order_idx});
return {hyperedge_index, hyperedge_list, hyperedge_list_end};
}
std::vector<torch::Tensor> GPDatabase::getRegionInfoTensor() {
auto options_int = torch::TensorOptions().dtype(torch::kInt64);
unsigned num_boxes = 0;
for (auto& region : regions) {
num_boxes += region.boxes().size();
}
torch::Tensor node_id2region_id = torch::zeros({num_nodes}, options_int);
torch::Tensor region_boxes = torch::zeros({num_boxes, 4});
torch::Tensor region_boxes_end = torch::zeros({num_regions}, options_int);
auto node_id2region_id_a = node_id2region_id.accessor<index_type, 1>();
auto region_boxes_a = region_boxes.accessor<coord_type, 2>();
auto region_boxes_end_a = region_boxes_end.accessor<index_type, 1>();
for (auto& node : nodes) {
index_type node_id = node.getId();
index_type region_id = node.getRegionId();
node_id2region_id_a[node_id] = region_id;
}
index_type ptr = 0;
index_type last_idx = 0;
for (auto& region : regions) {
index_type region_id = region.getId();
for (auto& box : region.boxes()) {
region_boxes_a[ptr][0] = box.lx();
region_boxes_a[ptr][1] = box.hx();
region_boxes_a[ptr][2] = box.ly();
region_boxes_a[ptr][3] = box.hy();
ptr += 1;
}
last_idx += region.boxes().size();
region_boxes_end_a[region_id] = last_idx;
}
return {node_id2region_id, region_boxes, region_boxes_end};
}
void GPDatabase::applyNodePos(torch::Tensor node_cpos) {
for (auto& node : nodes) {
if (node.getNodeType() != "Mov") {
continue;
}
node.setLx(node_cpos[node.getId()][0].item<coord_type>() - node.getWidth() / 2);
node.setLy(node_cpos[node.getId()][1].item<coord_type>() - node.getHeight() / 2);
auto cell = database.cells[node.getOriDBId()];
cell->place(static_cast<int>(node.getLx()), static_cast<int>(node.getLy()));
}
}
void GPDatabase::writePlacement(const std::string& given_prefix) { database.save(given_prefix); }
} // namespace gp

View File

@ -0,0 +1,225 @@
#pragma once
#include "common/common.h"
#include "common/db/Database.h"
namespace gp {
using index_type = int64_t;
using coord_type = float;
using orient_type = int; // 0:N, 1:W, 2:S, 3:E, 4:FN, 5:FW, 6:FS, 7:FE, -1:NONE
using point_type = std::pair<coord_type, coord_type>;
using box_type = utils::BoxT<coord_type>;
class Basic {
public:
index_type getId() const { return id; }
void setId(const index_type& i) { id = i; }
std::string getName() const { return name; }
void setName(const std::string& name_str) { name = name_str; }
protected:
index_type id = std::numeric_limits<index_type>::max();
std::string name = "";
};
class GPNode : public Basic {
public:
void setLx(const coord_type& lx_) { lx = lx_; }
const coord_type& getLx() const { return lx; }
void setLy(const coord_type& ly_) { ly = ly_; }
const coord_type& getLy() const { return ly; }
coord_type getHx() const { return getLx() + getWidth(); }
coord_type getHy() const { return getLy() + getHeight(); }
void setWidth(const coord_type& width_) { width = width_; }
const coord_type& getWidth() const { return width; }
void setHeight(const coord_type& height_) { height = height_; }
const coord_type& getHeight() const { return height; }
void setOrient(const orient_type& orient_) { orient = orient_; }
const orient_type& getOrient() const { return orient; }
void setNodeType(const std::string& node_type_) { node_type = node_type_; }
const std::string& getNodeType() const { return node_type; }
void setOriDBId(const index_type& ori_db_id_) { ori_db_id = ori_db_id_; }
const index_type& getOriDBId() const { return ori_db_id; }
void setRegionId(const index_type& region_id_) { region_id = region_id_; }
const index_type& getRegionId() const { return region_id; }
const std::vector<index_type>& pins() const { return pins_id; }
void addPin(index_type pin_id) { pins_id.emplace_back(pin_id); }
void setIsPolygonShape(bool isPolygonShape_) { isPolygonShape = isPolygonShape_; }
const bool getIsPolygonShape() const { return isPolygonShape; }
protected:
coord_type lx = std::numeric_limits<coord_type>::max();
coord_type ly = std::numeric_limits<coord_type>::max();
coord_type width = 0;
coord_type height = 0;
orient_type orient = -1;
std::string node_type = ""; // Mov, FloatMov, Fix, IOPin, Blkg, FloatIOPin, FloatFix
index_type ori_db_id = std::numeric_limits<index_type>::max();
index_type region_id = -1; // no region, we should ignore fixed nodes' fence region
std::vector<index_type> pins_id;
bool isPolygonShape = false;
};
class GPPin : public Basic {
public:
void setRelLx(const coord_type& rel_lx_) { rel_lx = rel_lx_; }
const coord_type& getRelLx() const { return rel_lx; }
void setRelLy(const coord_type& rel_ly_) { rel_ly = rel_ly_; }
const coord_type& getRelLy() const { return rel_ly; }
void setWidth(const coord_type& width_) { width = width_; }
const coord_type& getWidth() const { return width; }
void setHeight(const coord_type& height_) { height = height_; }
const coord_type& getHeight() const { return height; }
coord_type getRelHx() const { return getRelLx() + getWidth(); }
coord_type getRelHy() const { return getRelLy() + getHeight(); }
void setDirection(const char& direction_) { direction = direction_; }
const char& getDirection() const { return direction; }
void setType(const char& type_) { type = type_; }
const char& getType() const { return type; }
void setParNodeId(const index_type& parent_node_id_) { parent_node_id = parent_node_id_; }
const index_type& getParNodeId() const { return parent_node_id; }
void setParNetId(const index_type& parent_net_id_) { parent_net_id = parent_net_id_; }
const index_type& getParNetId() const { return parent_net_id; }
void setOriDBInfo(const std::tuple<index_type, index_type, index_type>& ori_db_info_) {
ori_db_info = ori_db_info_;
}
const std::tuple<index_type, index_type, index_type>& getOriDBInfo() const { return ori_db_info; }
protected:
coord_type rel_lx = std::numeric_limits<coord_type>::max(); // relative position from node_lx to pin_lx
coord_type rel_ly = std::numeric_limits<coord_type>::max(); // relative position from node_ly to pin_ly
coord_type width = 0;
coord_type height = 0;
// i: input, o:output
char direction = 'x';
// s: signal, c: clk, p: power, g: ground
char type = 's';
index_type parent_node_id = std::numeric_limits<index_type>::max();
index_type parent_net_id = std::numeric_limits<index_type>::max();
// iopin: ori_db_info == {ori_db_parent_iopin_id, -1, ori_db_parent_net_id}
// cell pin: ori_db_info == {ori_db_parent_cell_id, ori_db_parent_cell_pin_id, ori_db_parent_net_id}
std::tuple<index_type, index_type, index_type> ori_db_info = {-1, -1, -1};
};
class GPNet : public Basic {
public:
void setOriDBId(const index_type& ori_db_id_) { ori_db_id = ori_db_id_; }
const index_type& getOriDBId() const { return ori_db_id; }
const std::vector<index_type>& pins() const { return pins_id; }
void addPin(index_type pin_id) { pins_id.emplace_back(pin_id); }
protected:
index_type ori_db_id = std::numeric_limits<index_type>::max();
std::vector<index_type> pins_id;
};
// TODO: fence region and its mapping to nodes
class GPRegion : public Basic {
public:
void setType(const char& type_) { type = type_; }
const char& getType() const { return type; }
void setOriDBId(const index_type& ori_db_id_) { ori_db_id = ori_db_id_; }
const index_type& getOriDBId() const { return ori_db_id; }
const std::vector<index_type>& nodes() const { return nodes_id; }
void addNode(index_type node_id) { nodes_id.emplace_back(node_id); }
const std::vector<box_type>& boxes() const { return _boxes; }
void addBox(box_type box) { _boxes.emplace_back(box); }
protected:
// f: fence, g: guide
char type = 'f';
index_type ori_db_id = std::numeric_limits<index_type>::max();
std::vector<index_type> nodes_id;
std::vector<box_type> _boxes;
};
class GPDatabase {
protected:
db::Database& database;
std::tuple<int, int, int, int> dieInfo; // dieLX, dieHX, dieLY, dieHY
std::tuple<int, int, int, int> coreInfo; // coreLX, coreHX, coreLY, coreHY
int siteW;
int siteH;
unsigned int num_nodes;
unsigned int num_pins;
unsigned int num_nets;
unsigned int num_regions;
std::vector<GPNode> nodes; // store all nodes Mov + FloatMov + Fix + IOPin + Blkg + FloatIOPin + FloatFix
std::vector<GPPin> pins; // store all pins
std::vector<GPNet> nets; // store all nets
std::vector<GPRegion> regions; // store all regions
std::vector<std::tuple<index_type, index_type, std::string>> node_types_indices; // (start_idx, end_idx, type)
std::vector<std::string> node_id2node_name;
std::vector<index_type> pin_id2node_id; // pin_id to node_id mapping
std::vector<index_type> pin_id2net_id; // pin_id to net_id mapping
std::vector<std::vector<index_type>> nodes_id2pins_id;
std::vector<std::vector<index_type>> nets_id2pins_id;
public:
GPDatabase(std::shared_ptr<db::Database> database_) : database(*database_) {}
~GPDatabase();
point_type getAbsolutePinPos(const GPPin& pin) const;
void addCellNode(index_type cell_id, std::string& node_type);
void addIOPinNode(index_type iopin_id, std::string& node_type);
void addBlockageNode(index_type blkg_id, std::string& node_type);
void addNet(index_type dbnet_id);
void addPin(db::Pin* dbpin, const db::PinType* pintype, GPNode& node, GPNet& net, bool isIOPin);
void addRegion(index_type dbregion_id);
void setupNum();
void setupRegions();
void setupNodes();
void setupNets();
void setupIndexMap();
void setupCheckVar();
bool setup();
bool reset();
void setup_random_place();
const std::vector<GPNode>& getNodes() const { return nodes; }
const std::vector<GPNet>& getNets() const { return nets; }
const std::vector<GPPin>& getPins() const { return pins; }
const std::vector<std::tuple<index_type, index_type, std::string>>& getNodeTypeIndices() const {
return node_types_indices;
}
const std::vector<std::string>& getNodeId2NodeName() const { return node_id2node_name; }
const std::tuple<int, int, int, int>& getDieInfo() const { return dieInfo; }
const std::tuple<int, int, int, int>& getCoreInfo() const { return coreInfo; }
const int getSiteWidth() const { return siteW; }
const int getSiteHeight() const { return siteH; }
const int getM1Direction() const { return database.getRLayer(0)->direction == 'v' ? 1 : 0; }
// Torch related
torch::Tensor getNodeLPosTensor();
torch::Tensor getNodeCPosTensor();
torch::Tensor getNodeSizeTensor();
torch::Tensor getPinRelLPosTensor(); // pin_lx - node_lx
torch::Tensor getPinRelCPosTensor(); // pin_cx - node_cx
torch::Tensor getPinSizeTensor();
torch::Tensor getPinId2NodeIdTensor();
torch::Tensor getPinId2NetIdTensor();
std::vector<torch::Tensor> getHyperedgeInfoTensor(); // hyperedge_index, hyperedge_list, hyperedge_list_end
std::vector<torch::Tensor> getRegionInfoTensor(); // node_id2region_id, region_boxes, region_boxes_end
void applyNodePos(torch::Tensor node_cpos);
// Write Placement
void writePlacement(const std::string& given_prefix = "");
};
} // namespace gp

View File

@ -0,0 +1,9 @@
set(TARGET_NAME node_pos_to_pin_pos_cuda)
add_pytorch_extension(${TARGET_NAME}
${TARGET_NAME}.cpp
${TARGET_NAME}_kernel.cu)
install(TARGETS
${TARGET_NAME}
DESTINATION ${XPLACE_LIB_DIR})

View File

@ -0,0 +1,25 @@
#include <torch/extension.h>
torch::Tensor node_pos_to_pin_pos_cuda_forward(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos);
#define CHECK_CUDA(x) TORCH_CHECK(x.device().is_cuda(), #x " must be a CUDA tensor")
#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
#define CHECK_INPUT(x) \
CHECK_CUDA(x); \
CHECK_CONTIGUOUS(x)
torch::Tensor node_pos_to_pin_pos_forward(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos) {
CHECK_INPUT(node_pos);
CHECK_INPUT(pin_id2node_id);
CHECK_INPUT(pin_rel_cpos);
return node_pos_to_pin_pos_cuda_forward(node_pos, pin_id2node_id, pin_rel_cpos);
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
m.def("forward", &node_pos_to_pin_pos_forward, "get pin pos from node pos");
}

View File

@ -0,0 +1,48 @@
#include <torch/extension.h>
#include <cuda.h>
#include <cuda_runtime.h>
#include <vector>
#include <ATen/cuda/CUDAContext.h>
template <typename scalar_t>
__global__ void node_pos_to_pin_pos_cuda_forward_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> pin_id2node_id,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> pin_pos,
int num_pins) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int i = index >> 1; // pin index
if (i < num_pins) {
const int c = index & 1; // channel index
int64_t node_id = pin_id2node_id[i];
pin_pos[i][c] += node_pos[node_id][c];
// scalar_t result = pin_pos[i][c] + node_pos[node_id][c];
// pin_pos[i][c] = result;
// gpuAtomicAdd(&pin_pos[i][c], node_pos[node_id][c]);
}
}
torch::Tensor node_pos_to_pin_pos_cuda_forward(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos) {
cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
// pin_pos == node_pos + pin_rel_cpos
const auto num_pins = pin_id2node_id.size(0);
auto pin_pos = pin_rel_cpos.clone(); // pin
const int threads = 64;
const int blocks = (num_pins * 2 + threads - 1) / threads;
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "node_pos_to_pin_pos_cuda_forward", ([&] {
node_pos_to_pin_pos_cuda_forward_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
pin_id2node_id.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
pin_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_pins);
}));
return pin_pos;
}

View File

@ -0,0 +1,9 @@
set(TARGET_NAME wa_wirelength_hpwl_cuda)
add_pytorch_extension(${TARGET_NAME}
${TARGET_NAME}.cpp
${TARGET_NAME}_kernel.cu)
install(TARGETS
${TARGET_NAME}
DESTINATION ${XPLACE_LIB_DIR})

View File

@ -0,0 +1,123 @@
#include <torch/extension.h>
torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
torch::Tensor hpwl_scale);
std::vector<torch::Tensor> wa_wirelength_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
float gamma);
std::vector<torch::Tensor> wa_wirelength_hpwl_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
float gamma);
std::vector<torch::Tensor> wa_wirelength_masked_scale_hpwl_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
torch::Tensor hpwl_scale,
float gamma);
#define CHECK_CUDA(x) TORCH_CHECK(x.device().is_cuda(), #x " must be a CUDA tensor")
#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
#define CHECK_INPUT(x) \
CHECK_CUDA(x); \
CHECK_CONTIGUOUS(x)
torch::Tensor masked_scale_hpwl_sum(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
torch::Tensor hpwl_scale) {
CHECK_INPUT(node_pos);
CHECK_INPUT(pin_id2node_id);
CHECK_INPUT(pin_rel_cpos);
CHECK_INPUT(hyperedge_list);
CHECK_INPUT(hyperedge_list_end);
CHECK_INPUT(net_mask);
CHECK_INPUT(hpwl_scale);
return masked_scale_hpwl_sum_cuda(
node_pos, pin_id2node_id, pin_rel_cpos, hyperedge_list, hyperedge_list_end, net_mask, hpwl_scale);
}
std::vector<torch::Tensor> wa_wirelength(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
float gamma) {
CHECK_INPUT(node_pos);
CHECK_INPUT(pin_id2node_id);
CHECK_INPUT(pin_rel_cpos);
CHECK_INPUT(hyperedge_list);
CHECK_INPUT(hyperedge_list_end);
CHECK_INPUT(net_mask);
return wa_wirelength_cuda(
node_pos, pin_id2node_id, pin_rel_cpos, hyperedge_list, hyperedge_list_end, net_mask, gamma);
}
std::vector<torch::Tensor> wa_wirelength_hpwl(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
float gamma) {
CHECK_INPUT(node_pos);
CHECK_INPUT(pin_id2node_id);
CHECK_INPUT(pin_rel_cpos);
CHECK_INPUT(hyperedge_list);
CHECK_INPUT(hyperedge_list_end);
CHECK_INPUT(net_mask);
return wa_wirelength_hpwl_cuda(
node_pos, pin_id2node_id, pin_rel_cpos, hyperedge_list, hyperedge_list_end, net_mask, gamma);
}
std::vector<torch::Tensor> wa_wirelength_masked_scale_hpwl(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
torch::Tensor hpwl_scale,
float gamma) {
CHECK_INPUT(node_pos);
CHECK_INPUT(pin_id2node_id);
CHECK_INPUT(pin_rel_cpos);
CHECK_INPUT(hyperedge_list);
CHECK_INPUT(hyperedge_list_end);
CHECK_INPUT(net_mask);
CHECK_INPUT(hpwl_scale);
return wa_wirelength_masked_scale_hpwl_cuda(
node_pos, pin_id2node_id, pin_rel_cpos, hyperedge_list, hyperedge_list_end, net_mask, hpwl_scale, gamma);
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
m.def("masked_scale_hpwl_sum", &masked_scale_hpwl_sum, "calculate the sum of scaled HPWL");
m.def("merged_forward_backward", &wa_wirelength, "calculate WA wirelength and pin grad");
m.def("merged_forward_backward_with_hpwl", &wa_wirelength_hpwl, "calculate WA wirelength, pin grad and hpwl");
m.def("merged_forward_backward_with_masked_scale_hpwl",
&wa_wirelength_masked_scale_hpwl,
"calculate WA wirelength, pin grad and the scaled hpwl");
}

View File

@ -0,0 +1,467 @@
#include <cuda.h>
#include <cuda_runtime.h>
#include <torch/extension.h>
#include <THC/THCAtomics.cuh>
#include <vector>
template <typename scalar_t>
__global__ void node_pos_to_pin_pos_cuda_forward_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> pin_id2node_id,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> pin_pos,
int num_pins) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int i = index >> 1; // pin index
if (i < num_pins) {
const int c = index & 1; // channel index
int64_t node_id = pin_id2node_id[i];
pin_pos[i][c] += node_pos[node_id][c];
}
}
template <typename scalar_t>
__global__ void masked_scale_hpwl_cuda_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> pin_pos,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list_end,
const torch::PackedTensorAccessor32<bool, 1, torch::RestrictPtrTraits> net_mask,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> hpwl_scale,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> partial_hpwl,
int num_nets) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int i = index >> 1; // net index
if (i < num_nets && net_mask[i]) {
const int c = index & 1; // channel index
int64_t start_idx = 0;
if (i != 0) {
start_idx = hyperedge_list_end[i - 1];
}
int64_t end_idx = hyperedge_list_end[i];
if (end_idx != start_idx) {
int64_t pin_id = hyperedge_list[start_idx];
scalar_t x_min = pin_pos[pin_id][c];
scalar_t x_max = pin_pos[pin_id][c];
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
scalar_t xx = pin_pos[hyperedge_list[idx]][c];
x_min = min(xx, x_min);
x_max = max(xx, x_max);
}
partial_hpwl[i][c] = round(abs(x_max - x_min) * hpwl_scale[c]);
}
}
}
template <typename scalar_t>
__global__ void wa_wirelength_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> pin_pos,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list_end,
const torch::PackedTensorAccessor32<bool, 1, torch::RestrictPtrTraits> net_mask,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> partial_wa_wl,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> pin_grad,
int num_nets,
float inv_gamma) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int i = index >> 1; // net index
if (i < num_nets && net_mask[i]) {
const int c = index & 1; // channel index
int64_t start_idx = 0;
if (i != 0) {
start_idx = hyperedge_list_end[i - 1];
}
int64_t end_idx = hyperedge_list_end[i];
if (end_idx != start_idx) {
int64_t pin_id = hyperedge_list[start_idx];
scalar_t x_min = pin_pos[pin_id][c];
scalar_t x_max = pin_pos[pin_id][c];
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
scalar_t xx = pin_pos[hyperedge_list[idx]][c];
x_min = min(xx, x_min);
x_max = max(xx, x_max);
}
scalar_t xexp_x_sum = 0;
scalar_t xexp_nx_sum = 0;
scalar_t exp_x_sum = 0;
scalar_t exp_nx_sum = 0;
for (int64_t idx = start_idx; idx < end_idx; idx++) {
scalar_t xx = pin_pos[hyperedge_list[idx]][c];
scalar_t exp_x = exp((xx - x_max) * inv_gamma);
scalar_t exp_nx = exp((x_min - xx) * inv_gamma);
xexp_x_sum += xx * exp_x;
xexp_nx_sum += xx * exp_nx;
exp_x_sum += exp_x;
exp_nx_sum += exp_nx;
}
scalar_t wl = xexp_x_sum / exp_x_sum - xexp_nx_sum / exp_nx_sum;
partial_wa_wl[i][c] = wl;
scalar_t b_x = inv_gamma / (exp_x_sum);
scalar_t a_x = (1.0 - b_x * xexp_x_sum) / exp_x_sum;
scalar_t b_nx = -inv_gamma / (exp_nx_sum);
scalar_t a_nx = (1.0 - b_nx * xexp_nx_sum) / exp_nx_sum;
for (int64_t idx = start_idx; idx < end_idx; idx++) {
scalar_t xx = pin_pos[hyperedge_list[idx]][c];
scalar_t exp_x = exp((xx - x_max) * inv_gamma);
scalar_t exp_nx = exp((x_min - xx) * inv_gamma);
pin_grad[hyperedge_list[idx]][c] = (a_x + b_x * xx) * exp_x - (a_nx + b_nx * xx) * exp_nx;
}
}
}
}
template <typename scalar_t>
__global__ void wa_wirelength_hpwl_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> pin_pos,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list_end,
const torch::PackedTensorAccessor32<bool, 1, torch::RestrictPtrTraits> net_mask,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> partial_wa_wl,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> partial_hpwl,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> pin_grad,
int num_nets,
float inv_gamma) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int i = index >> 1; // net index
if (i < num_nets && net_mask[i]) {
const int c = index & 1; // channel index
int64_t start_idx = 0;
if (i != 0) {
start_idx = hyperedge_list_end[i - 1];
}
int64_t end_idx = hyperedge_list_end[i];
if (end_idx != start_idx) {
int64_t pin_id = hyperedge_list[start_idx];
scalar_t x_min = pin_pos[pin_id][c];
scalar_t x_max = pin_pos[pin_id][c];
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
scalar_t xx = pin_pos[hyperedge_list[idx]][c];
x_min = min(xx, x_min);
x_max = max(xx, x_max);
}
partial_hpwl[i][c] = abs(x_max - x_min);
scalar_t xexp_x_sum = 0;
scalar_t xexp_nx_sum = 0;
scalar_t exp_x_sum = 0;
scalar_t exp_nx_sum = 0;
for (int64_t idx = start_idx; idx < end_idx; idx++) {
scalar_t xx = pin_pos[hyperedge_list[idx]][c];
scalar_t exp_x = exp((xx - x_max) * inv_gamma);
scalar_t exp_nx = exp((x_min - xx) * inv_gamma);
xexp_x_sum += xx * exp_x;
xexp_nx_sum += xx * exp_nx;
exp_x_sum += exp_x;
exp_nx_sum += exp_nx;
}
scalar_t wl = xexp_x_sum / exp_x_sum - xexp_nx_sum / exp_nx_sum;
partial_wa_wl[i][c] = wl;
scalar_t b_x = inv_gamma / (exp_x_sum);
scalar_t a_x = (1.0 - b_x * xexp_x_sum) / exp_x_sum;
scalar_t b_nx = -inv_gamma / (exp_nx_sum);
scalar_t a_nx = (1.0 - b_nx * xexp_nx_sum) / exp_nx_sum;
for (int64_t idx = start_idx; idx < end_idx; idx++) {
scalar_t xx = pin_pos[hyperedge_list[idx]][c];
scalar_t exp_x = exp((xx - x_max) * inv_gamma);
scalar_t exp_nx = exp((x_min - xx) * inv_gamma);
pin_grad[hyperedge_list[idx]][c] = (a_x + b_x * xx) * exp_x - (a_nx + b_nx * xx) * exp_nx;
}
}
}
}
template <typename scalar_t>
__global__ void wa_wirelength_masked_scale_hpwl_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> pin_pos,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list_end,
const torch::PackedTensorAccessor32<bool, 1, torch::RestrictPtrTraits> net_mask,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> hpwl_scale,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> partial_wa_wl,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> partial_hpwl,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> pin_grad,
int num_nets,
float inv_gamma) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int i = index >> 1; // net index
if (i < num_nets && net_mask[i]) {
const int c = index & 1; // channel index
int64_t start_idx = 0;
if (i != 0) {
start_idx = hyperedge_list_end[i - 1];
}
int64_t end_idx = hyperedge_list_end[i];
if (end_idx != start_idx) {
int64_t pin_id = hyperedge_list[start_idx];
scalar_t x_min = pin_pos[pin_id][c];
scalar_t x_max = pin_pos[pin_id][c];
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
scalar_t cur_x = pin_pos[hyperedge_list[idx]][c];
x_min = min(cur_x, x_min);
x_max = max(cur_x, x_max);
}
partial_hpwl[i][c] = round((x_max - x_min) * hpwl_scale[c]);
// partial_hpwl[i][c] = round(abs(x_max - x_min) * hpwl_scale[c]);
scalar_t sum_x_exp_x = 0;
scalar_t sum_x_exp_nx = 0;
scalar_t sum_exp_x = 0;
scalar_t sum_exp_nx = 0;
for (int64_t idx = start_idx; idx < end_idx; idx++) {
scalar_t cur_x = pin_pos[hyperedge_list[idx]][c];
scalar_t recenter_exp_x = exp((cur_x - x_max) * inv_gamma);
scalar_t recenter_exp_nx = exp((x_min - cur_x) * inv_gamma);
sum_x_exp_x += cur_x * recenter_exp_x;
sum_x_exp_nx += cur_x * recenter_exp_nx;
sum_exp_x += recenter_exp_x;
sum_exp_nx += recenter_exp_nx;
}
scalar_t inv_sum_exp_x = 1 / sum_exp_x;
scalar_t inv_sum_exp_nx = 1 / sum_exp_nx;
scalar_t s_x = sum_x_exp_x * inv_sum_exp_x;
scalar_t ns_nx = sum_x_exp_nx * inv_sum_exp_nx;
partial_wa_wl[i][c] = s_x - ns_nx;
scalar_t x_coeff = inv_gamma * inv_sum_exp_x;
scalar_t nx_coeff = -inv_gamma * inv_sum_exp_nx;
scalar_t grad_const = (1 - inv_gamma * s_x) * inv_sum_exp_x;
scalar_t grad_nconst = (1 + inv_gamma * ns_nx) * inv_sum_exp_nx;
for (int64_t idx = start_idx; idx < end_idx; idx++) {
int64_t pin_id = hyperedge_list[idx];
scalar_t cur_x = pin_pos[pin_id][c];
scalar_t recenter_exp_x = exp((cur_x - x_max) * inv_gamma);
scalar_t recenter_exp_nx = exp((x_min - cur_x) * inv_gamma);
pin_grad[pin_id][c] = (grad_const + x_coeff * cur_x) * recenter_exp_x -
(grad_nconst + nx_coeff * cur_x) * recenter_exp_nx;
}
}
}
}
std::vector<torch::Tensor> wa_wirelength_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
float gamma) {
cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
const auto num_nodes = node_pos.size(0);
const auto num_pins = pin_id2node_id.size(0);
const auto num_nets = hyperedge_list_end.size(0);
const auto num_channels = 2; // x, y
auto pin_pos = pin_rel_cpos.clone(); // pin
auto partial_wa_wl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
auto pin_grad = torch::zeros({num_pins, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
const int threads = 128;
const int blocks = (num_pins * 2 + threads - 1) / threads;
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "node_pos_to_pin_pos_cuda_forward", ([&] {
node_pos_to_pin_pos_cuda_forward_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
pin_id2node_id.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
pin_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_pins);
}));
const int threads2 = 128;
const int blocks2 = (num_nets * 2 + threads2 - 1) / threads2;
float inv_gamma = 1 / gamma;
AT_DISPATCH_ALL_TYPES(pin_pos.scalar_type(), "wa_wirelength", ([&] {
wa_wirelength_kernel<scalar_t><<<blocks2, threads2, 0, stream>>>(
pin_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
hyperedge_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
hyperedge_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
net_mask.packed_accessor32<bool, 1, torch::RestrictPtrTraits>(),
partial_wa_wl.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
pin_grad.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_nets,
inv_gamma);
}));
auto node_grad = torch::zeros({num_nodes, num_channels}, torch::dtype(pin_grad.dtype()).device(pin_grad.device()));
const auto pin_id2node_id_view = pin_id2node_id.unsqueeze(1).expand({-1, 2});
node_grad.scatter_add_(0, pin_id2node_id_view, pin_grad);
return {partial_wa_wl, node_grad};
}
torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
torch::Tensor hpwl_scale) {
cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
const auto num_nodes = node_pos.size(0);
const auto num_pins = pin_id2node_id.size(0);
const auto num_nets = hyperedge_list_end.size(0);
const auto num_channels = 2; // x, y
auto pin_pos = pin_rel_cpos.clone(); // pin
auto partial_hpwl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
const int threads = 128;
const int blocks = (num_pins * 2 + threads - 1) / threads;
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "node_pos_to_pin_pos_cuda_forward", ([&] {
node_pos_to_pin_pos_cuda_forward_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
pin_id2node_id.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
pin_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_pins);
}));
const int threads2 = 128;
const int blocks2 = (num_nets * 2 + threads2 - 1) / threads2;
AT_DISPATCH_ALL_TYPES(pin_pos.scalar_type(), "masked_scale_hpwl_cuda", ([&] {
masked_scale_hpwl_cuda_kernel<scalar_t><<<blocks2, threads2, 0, stream>>>(
pin_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
hyperedge_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
hyperedge_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
net_mask.packed_accessor32<bool, 1, torch::RestrictPtrTraits>(),
hpwl_scale.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
partial_hpwl.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_nets);
}));
const auto total_hpwl = partial_hpwl.sum();
return total_hpwl;
}
std::vector<torch::Tensor> wa_wirelength_hpwl_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
float gamma) {
cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
const auto num_nodes = node_pos.size(0);
const auto num_pins = pin_id2node_id.size(0);
const auto num_nets = hyperedge_list_end.size(0);
const auto num_channels = 2; // x, y
auto pin_pos = pin_rel_cpos.clone(); // pin
auto partial_wa_wl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
auto partial_hpwl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
auto pin_grad = torch::zeros({num_pins, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
const int threads = 128;
const int blocks = (num_pins * 2 + threads - 1) / threads;
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "node_pos_to_pin_pos_cuda_forward", ([&] {
node_pos_to_pin_pos_cuda_forward_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
pin_id2node_id.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
pin_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_pins);
}));
const int threads2 = 128;
const int blocks2 = (num_nets * 2 + threads2 - 1) / threads2;
float inv_gamma = 1 / gamma;
AT_DISPATCH_ALL_TYPES(pin_pos.scalar_type(), "wa_wirelength_hpwl", ([&] {
wa_wirelength_hpwl_kernel<scalar_t><<<blocks2, threads2, 0, stream>>>(
pin_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
hyperedge_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
hyperedge_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
net_mask.packed_accessor32<bool, 1, torch::RestrictPtrTraits>(),
partial_wa_wl.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
partial_hpwl.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
pin_grad.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_nets,
inv_gamma);
}));
auto node_grad = torch::zeros({num_nodes, num_channels}, torch::dtype(pin_grad.dtype()).device(pin_grad.device()));
const auto pin_id2node_id_view = pin_id2node_id.unsqueeze(1).expand({-1, 2});
node_grad.scatter_add_(0, pin_id2node_id_view, pin_grad);
return {partial_wa_wl, node_grad, partial_hpwl};
}
std::vector<torch::Tensor> wa_wirelength_masked_scale_hpwl_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
torch::Tensor hpwl_scale,
float gamma) {
cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
const auto num_nodes = node_pos.size(0);
const auto num_pins = pin_id2node_id.size(0);
const auto num_nets = hyperedge_list_end.size(0);
const auto num_channels = 2; // x, y
auto pin_pos = pin_rel_cpos.clone(); // pin
auto partial_wa_wl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
auto partial_hpwl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
auto pin_grad = torch::zeros({num_pins, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
const int threads = 128;
const int blocks = (num_pins * 2 + threads - 1) / threads;
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "node_pos_to_pin_pos_cuda_forward", ([&] {
node_pos_to_pin_pos_cuda_forward_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
pin_id2node_id.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
pin_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_pins);
}));
const int threads2 = 128;
const int blocks2 = (num_nets * 2 + threads2 - 1) / threads2;
float inv_gamma = 1 / gamma;
AT_DISPATCH_ALL_TYPES(pin_pos.scalar_type(), "wa_wirelength_masked_scale_hpwl", ([&] {
wa_wirelength_masked_scale_hpwl_kernel<scalar_t><<<blocks2, threads2, 0, stream>>>(
pin_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
hyperedge_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
hyperedge_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
net_mask.packed_accessor32<bool, 1, torch::RestrictPtrTraits>(),
hpwl_scale.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
partial_wa_wl.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
partial_hpwl.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
pin_grad.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_nets,
inv_gamma);
}));
auto node_grad = torch::zeros({num_nodes, num_channels}, torch::dtype(pin_grad.dtype()).device(pin_grad.device()));
const auto pin_id2node_id_view = pin_id2node_id.unsqueeze(1).expand({-1, 2});
node_grad.scatter_add_(0, pin_id2node_id_view, pin_grad);
return {partial_wa_wl, node_grad, partial_hpwl};
}

72
main.py Normal file
View File

@ -0,0 +1,72 @@
from utils import *
from src import run_placement_main, Flute
def get_option():
parser = argparse.ArgumentParser('Xplace')
# general setting
parser.add_argument('--dataset_root', type=str, default='/data/ssd/lxliu/design', help='the parent folder of dataset')
parser.add_argument('--dataset', type=str, default='ispd2005', help='dataset name')
parser.add_argument('--design_name', type=str, default='adaptec1', help='design name')
parser.add_argument('--load_from_raw', type=str2bool, default=False, help='If True, parse and load from benchmark files. If False, load from pt')
parser.add_argument('--run_all', type=str2bool, default=False, help='If True, run all designs in the given dataset. If False, run the given design_name only.')
parser.add_argument('--seed', type=int, default=0, help='seed to initialize all the random modules')
parser.add_argument('--gpu', type=int, default=1, help='gpu id')
parser.add_argument('--num_threads', type=int, default=20, help='threads')
# optimization params
parser.add_argument('--lr', type=float, default=0.01, help='learning rate')
parser.add_argument('--inner_iter', type=int, default=10000, help='#inner iters')
parser.add_argument('--wa_coeff', type=float, default=4.0, help='wa coeff')
parser.add_argument('--num_bin_x', type=int, default=512, help='#binX for density function')
parser.add_argument('--num_bin_y', type=int, default=512, help='#binY for density function')
parser.add_argument('--threshold', type=float, default=4.0, help='normalized node area threshold for using naive mode')
parser.add_argument('--density_weight', type=float, default=8e-5, help='the weight of density loss')
parser.add_argument('--density_weight_coef', type=float, default=1.05, help='the ratio of density_weight')
parser.add_argument('--use_init_density_weight', type=str2bool, default=True, help='enable dynamic initialization of density_weight')
parser.add_argument('--target_density', type=float, default=1.0, help='placement target density')
parser.add_argument('--use_filler', type=str2bool, default=True, help='placement filler')
parser.add_argument('--noise_ratio', type=float, default=0.025, help='noise ratio for initialization')
parser.add_argument('--ignore_net_degree', type=int, default=100, help='threshold of net degree to ignore in wirelength calculation')
parser.add_argument('--scale_design', type=str2bool, default=True, help='normalize die area')
parser.add_argument('--use_eplace_nesterov', type=str2bool, default=True, help='enable eplace nesterov optimizer')
parser.add_argument('--clamp_node', type=str2bool, default=True, help='enable eplace node clamp trick')
parser.add_argument('--use_precond', type=str2bool, default=True, help='apply precond')
parser.add_argument('--stop_overflow', type=float, default=0.07, help='stop overflow in scheduler')
parser.add_argument('--enable_skip_update', type=str2bool, default=True, help='enable skip update')
parser.add_argument("--loss_type", type=str, default="direct", help="loss type")
# logging and saver
parser.add_argument('--log_freq', type=int, default=50)
parser.add_argument('--result_dir', type=str, default='result', help='output root directory')
parser.add_argument('--exp_id', type=str, default='', help='experiment id')
parser.add_argument('--log_dir', type=str, default='log', help='log directory')
parser.add_argument('--log_name', type=str, default='test.log', help='log file name')
parser.add_argument('--eval_dir', type=str, default='eval', help='visualization directory')
# placement related
parser.add_argument('--draw_placement', type=str2bool, default=False, help='draw placement')
parser.add_argument('--write_placement', type=str2bool, default=False, help='write placement result')
parser.add_argument('--output_dir', type=str, default="output", help='output directory')
parser.add_argument('--output_prefix', type=str, default="placement", help='prefix of placement output file')
parser.add_argument('--detail_placement', type=str2bool, default=False, help='perform dp')
parser.add_argument('--dp_engine', type=str, default="ntuplace3", help='choose dp engine')
args = parser.parse_args()
args.exp_id = datetime.datetime.now().strftime('%Y-%m-%d-%H:%M:%S') + args.exp_id
if args.dataset == "ispd2015_2":
args.dataset = "ispd2015_without_fence"
return args
def main():
args = get_option()
logger = setup_logger(args, sys.argv)
set_random_seed(args)
Flute.register(args.num_threads)
run_placement_main(args, logger)
if __name__ == "__main__":
main()

10
src/__init__.py Normal file
View File

@ -0,0 +1,10 @@
from .core import *
from .calculator import *
from .database import *
from .evaluator import *
from .initializer import *
from .nesterov_optimizer import NesterovOptimizer
from .param_scheduler import ParamScheduler
from .run_placement_adam import run_placement_main_adam
from .run_placement_nesterov import run_placement_main_nesterov
from .run_placement import run_placement_main

126
src/calculator.py Normal file
View File

@ -0,0 +1,126 @@
import torch
from .param_scheduler import ParamScheduler
from .core import merged_wl_loss_grad, WAWirelengthLoss, WAWirelengthLossAndHPWL
def calc_loss(wl_loss, density_loss, ps, args):
if args.loss_type == "weighted_sum":
loss = (wl_loss + ps.density_weight * density_loss) / (1 + ps.density_weight)
elif args.loss_type == "direct":
loss = wl_loss + ps.density_weight * density_loss
else:
raise NotImplementedError("Loss type not defined")
return loss
def apply_precond(mov_node_pos: torch.Tensor, ps: ParamScheduler, args):
if not args.use_precond:
return
mov_node_pos.grad /= ps.precond_weight
return mov_node_pos.grad
# For nesterov
def calc_obj_and_grad(
mov_node_pos,
constraint_fn=None,
mov_node_size=None,
init_density_map=None,
density_map_layer=None,
conn_fix_node_pos=None,
ps=None,
data=None,
args=None,
merged_forward_backward=True,
):
mov_lhs, mov_rhs = data.movable_index
mov_node_pos = constraint_fn(mov_node_pos)
conn_node_pos = mov_node_pos[mov_lhs:mov_rhs, ...]
conn_node_pos = torch.cat([conn_node_pos, conn_fix_node_pos], dim=0)
if merged_forward_backward:
if mov_node_pos.grad is not None:
mov_node_pos.grad.zero_()
else:
mov_node_pos.grad = torch.zeros_like(mov_node_pos).detach()
wl_loss, conn_node_grad_by_wl = merged_wl_loss_grad(
conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos,
data.hyperedge_list, data.hyperedge_list_end, data.net_mask,
data.hpwl_scale, ps.wa_coeff
)
mov_node_pos.grad[mov_lhs:mov_rhs] = conn_node_grad_by_wl[mov_lhs:mov_rhs]
if ps.enable_sample_force:
if ps.iter > 3 and ps.iter % 20 == 0:
# ps.iter > 3 for warmup
density_loss, _, node_grad_by_density = density_map_layer.merged_density_loss_grad(
mov_node_pos, mov_node_size, init_density_map, calc_overflow=False
)
ps.force_ratio = (
ps.density_weight * node_grad_by_density[mov_lhs:mov_rhs].norm(p=1) /
conn_node_grad_by_wl[mov_lhs:mov_rhs].norm(p=1)
).clamp_(max=10)
mov_node_pos.grad += node_grad_by_density * ps.density_weight
else:
density_loss = 0.0
if (ps.iter > 3 and ps.recorder.force_ratio[-1] > 1e-2) or ps.iter > 100:
# no longer enable sampling back
ps.enable_sample_force = False
else:
density_loss, _, node_grad_by_density = density_map_layer.merged_density_loss_grad(
mov_node_pos, mov_node_size, init_density_map, calc_overflow=False
)
mov_node_pos.grad += node_grad_by_density * ps.density_weight
grad = apply_precond(mov_node_pos, ps, args)
loss = wl_loss + ps.density_weight * density_loss
else:
if mov_node_pos.grad is not None:
mov_node_pos.grad.zero_()
else:
mov_node_pos.grad = torch.zeros_like(mov_node_pos).detach()
wl_loss = WAWirelengthLoss.apply(
conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos,
data.hyperedge_list, data.hyperedge_list_end, data.net_mask, ps.wa_coeff
)
density_loss, _ = density_map_layer(
mov_node_pos, mov_node_size, init_density_map, calc_overflow=False
)
loss = calc_loss(wl_loss, density_loss, ps, args)
loss.backward()
grad = apply_precond(mov_node_pos, ps, args)
return loss, grad
# For Adam
def calc_grad(
optimizer: torch.optim.Optimizer, mov_node_pos: torch.Tensor, wl_loss, density_loss
):
optimizer.zero_grad()
wl_loss.backward(retain_graph=True)
wl_grad = mov_node_pos.grad.detach().clone()
optimizer.zero_grad()
density_loss.backward(retain_graph=True)
density_grad = mov_node_pos.grad.detach().clone()
optimizer.zero_grad()
return wl_grad, density_grad
def fast_optimization(
mov_node_pos, trunc_node_pos_fn, mov_lhs, mov_rhs, conn_fix_node_pos,
density_map_layer, mov_node_size, init_density_map, ps, data, args
):
mov_node_pos = trunc_node_pos_fn(mov_node_pos)
conn_node_pos = mov_node_pos[mov_lhs:mov_rhs, ...]
conn_node_pos = torch.cat(
[conn_node_pos, conn_fix_node_pos], dim=0
)
wl_loss, hpwl = WAWirelengthLossAndHPWL.apply(
conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos,
data.hyperedge_list, data.hyperedge_list_end, data.net_mask,
ps.wa_coeff, data.hpwl_scale
)
density_loss, overflow = density_map_layer(
mov_node_pos, mov_node_size, init_density_map
)
loss = calc_loss(wl_loss, density_loss, ps, args)
loss.backward()
apply_precond(mov_node_pos, ps, args)
# calculate objective (hpwl, overflow)
return hpwl.detach(), overflow.detach(), mov_node_pos

5
src/core/__init__.py Normal file
View File

@ -0,0 +1,5 @@
from .flute import Flute, get_flute_wl
from .hpwl import HPWL, get_hpwl
from .electronic_density_layer import ElectronicDensityLayer
from .node_pos_to_pin_pos import NodePosToPinPosFunction
from .wa_wirelength_hpwl import WAWirelengthLossAndHPWL, WAWirelengthLoss, masked_scale_hpwl, merged_wl_loss_grad

109
src/core/dct2_fft2.py Normal file
View File

@ -0,0 +1,109 @@
# The implementation of DCT2 mainly follows DREAMPlace. We rewrite the code
# structure and use DCT2FFT2Cache to cache the intermediate results.
import numpy as np
import torch
from cpp_to_py import dct_cuda
class DCT2FFT2Cache:
def __init__(self) -> None:
self.expks_cache = {}
self.dct2_cache = {}
self.idct2_cache = {}
self.idct_idxst_cache = {}
self.idxst_idct_cache = {}
def reset(self):
self.expks_cache = {}
self.dct2_cache = {}
self.idct2_cache = {}
self.idct_idxst_cache = {}
self.idxst_idct_cache = {}
dct2_fft2_cache = DCT2FFT2Cache()
def precompute_expk(N, dtype, device):
# Compute exp(-j*pi*u/(2N)) = cos(pi*u/(2N)) - j * sin(pi*u/(2N))
pik_by_2N = torch.arange(N, dtype=dtype, device=device)
pik_by_2N.mul_(np.pi / (2 * N))
# cos, -sin
expk = torch.stack([pik_by_2N.cos(), -pik_by_2N.sin()], dim=-1)
return expk.contiguous()
def dct2(x):
if x.shape not in dct2_fft2_cache.expks_cache.keys():
expkM = precompute_expk(x.size(-2), dtype=x.dtype, device=x.device)
expkN = precompute_expk(x.size(-1), dtype=x.dtype, device=x.device)
dct2_fft2_cache.expks_cache[x.shape] = (expkM, expkN)
expkM, expkN = dct2_fft2_cache.expks_cache[x.shape]
if x.shape not in dct2_fft2_cache.dct2_cache.keys():
out = torch.empty(x.size(-2), x.size(-1), dtype=x.dtype, device=x.device)
buf = torch.empty(
x.size(-2), x.size(-1) // 2 + 1, 2, dtype=x.dtype, device=x.device
)
dct2_fft2_cache.dct2_cache[x.shape] = (out, buf)
out, buf = dct2_fft2_cache.dct2_cache[x.shape]
dct_cuda.dct2_fft2(x, expkM, expkN, out, buf)
return out
def idct2(x):
if x.shape not in dct2_fft2_cache.expks_cache.keys():
expkM = precompute_expk(x.size(-2), dtype=x.dtype, device=x.device)
expkN = precompute_expk(x.size(-1), dtype=x.dtype, device=x.device)
dct2_fft2_cache.expks_cache[x.shape] = (expkM, expkN)
expkM, expkN = dct2_fft2_cache.expks_cache[x.shape]
if x.shape not in dct2_fft2_cache.idct2_cache.keys():
out = torch.empty(x.size(-2), x.size(-1), dtype=x.dtype, device=x.device)
buf = torch.empty(
x.size(-2), x.size(-1) // 2 + 1, 2, dtype=x.dtype, device=x.device
)
dct2_fft2_cache.idct2_cache[x.shape] = (out, buf)
out, buf = dct2_fft2_cache.idct2_cache[x.shape]
dct_cuda.idct2_fft2(x, expkM, expkN, out, buf)
return out
def idct_idxst(x):
if x.shape not in dct2_fft2_cache.expks_cache.keys():
expkM = precompute_expk(x.size(-2), dtype=x.dtype, device=x.device)
expkN = precompute_expk(x.size(-1), dtype=x.dtype, device=x.device)
dct2_fft2_cache.expks_cache[x.shape] = (expkM, expkN)
expkM, expkN = dct2_fft2_cache.expks_cache[x.shape]
if x.shape not in dct2_fft2_cache.idct_idxst_cache.keys():
out = torch.empty(x.size(-2), x.size(-1), dtype=x.dtype, device=x.device)
buf = torch.empty(
x.size(-2), x.size(-1) // 2 + 1, 2, dtype=x.dtype, device=x.device
)
dct2_fft2_cache.idct_idxst_cache[x.shape] = (out, buf)
out, buf = dct2_fft2_cache.idct_idxst_cache[x.shape]
dct_cuda.idct_idxst(x, expkM, expkN, out, buf)
return out
def idxst_idct(x):
if x.shape not in dct2_fft2_cache.expks_cache.keys():
expkM = precompute_expk(x.size(-2), dtype=x.dtype, device=x.device)
expkN = precompute_expk(x.size(-1), dtype=x.dtype, device=x.device)
dct2_fft2_cache.expks_cache[x.shape] = (expkM, expkN)
expkM, expkN = dct2_fft2_cache.expks_cache[x.shape]
if x.shape not in dct2_fft2_cache.idxst_idct_cache.keys():
out = torch.empty(x.size(-2), x.size(-1), dtype=x.dtype, device=x.device)
buf = torch.empty(
x.size(-2), x.size(-1) // 2 + 1, 2, dtype=x.dtype, device=x.device
)
dct2_fft2_cache.idxst_idct_cache[x.shape] = (out, buf)
out, buf = dct2_fft2_cache.idxst_idct_cache[x.shape]
dct_cuda.idxst_idct(x, expkM, expkN, out, buf)
return out

View File

@ -0,0 +1,358 @@
import torch
from cpp_to_py import density_map_cuda
import numpy as np
import math
from .dct2_fft2 import dct2, idct2, idxst_idct, idct_idxst, dct2_fft2_cache
from .torch_dct import torch_dct_idct
class ElectronicDensityFunction(torch.autograd.Function):
@staticmethod
def forward(
ctx,
node_pos: torch.Tensor,
node_size: torch.Tensor,
node_weight: torch.Tensor,
expand_ratio: torch.Tensor,
unit_len: torch.Tensor,
init_density_map: torch.Tensor,
num_bin_x: int,
num_bin_y: int,
num_nodes: int,
fft_scale: tuple,
overflow_helper: tuple,
sorted_maps: tuple,
calc_overflow: bool,
):
ctx.constant_var = (num_bin_x, num_bin_y, num_nodes)
mov_lhs, mov_rhs, overflow_fn = overflow_helper
mov_sorted_map, mov_conn_sorted_map, filler_sorted_map = sorted_maps
# 1) Compute Density Map
normalize_node_info = node_size.new_empty((num_nodes, 5)) # x_l, x_h, y_l, y_h, weight
normalize_node_info = density_map_cuda.pre_normalize(
node_pos, node_size, node_weight, expand_ratio, unit_len, normalize_node_info,
num_bin_x, num_bin_y, num_nodes,
)
if calc_overflow:
aux_mat = init_density_map.clone()
mov_density_map = density_map_cuda.forward(
normalize_node_info[mov_lhs:mov_rhs], mov_conn_sorted_map, aux_mat,
num_bin_x, num_bin_y, mov_rhs - mov_lhs
)
overflow = overflow_fn(mov_density_map)
aux_mat2 = torch.zeros_like(init_density_map)
if filler_sorted_map is not None:
filler_density_map = density_map_cuda.forward(
normalize_node_info[mov_rhs:], filler_sorted_map, aux_mat2,
num_bin_x, num_bin_y, num_nodes - (mov_rhs - mov_lhs)
)
density_map = mov_density_map + filler_density_map
else:
density_map = mov_density_map
else:
overflow = node_size.new_empty(0)
aux_mat = init_density_map.clone()
density_map = density_map_cuda.forward(
normalize_node_info, mov_sorted_map, aux_mat,
num_bin_x, num_bin_y, num_nodes
)
# density_map -= density_map.mean() # we don't need this one anymore
if True:
potential_scale, potential_coeff, force_x_scale, force_y_scale, force_x_coeff, force_y_coeff = fft_scale
fft_coeff = dct2(density_map)
force_x_map = idxst_idct(fft_coeff * force_x_scale)
force_y_map = idct_idxst(fft_coeff * force_y_scale)
potential_map = idct2(fft_coeff * potential_scale)
grad_mat = torch.vstack(
(force_x_map.unsqueeze(0), force_y_map.unsqueeze(0))
).contiguous() # 2 x M x N
else:
# NOTE: in some of cases, torch version is faster
grad_mat, potential_map = torch_dct_idct(density_map, fft_scale)
energy = (potential_map * density_map).sum()
ctx.save_for_backward(normalize_node_info, mov_sorted_map, grad_mat)
return energy, overflow
@staticmethod
def backward(ctx, energy_grad_out, overflow_grad_out):
normalize_node_info, mov_sorted_map, grad_mat = ctx.saved_tensors
num_bin_x, num_bin_y, num_nodes = ctx.constant_var
grad_mat = grad_mat * energy_grad_out
grad_weight = -1.0 # Gradient descent
node_grad = normalize_node_info.new_zeros((num_nodes, 2))
node_grad = density_map_cuda.backward(
normalize_node_info, grad_mat, mov_sorted_map, node_grad,
grad_weight, num_bin_x, num_bin_y, num_nodes
)
return (node_grad,) + (None,) * 12
def merged_density_loss_grad_main(
node_pos: torch.Tensor,
node_size: torch.Tensor,
node_weight: torch.Tensor,
expand_ratio: torch.Tensor,
unit_len: torch.Tensor,
init_density_map: torch.Tensor,
num_bin_x: int,
num_bin_y: int,
num_nodes: int,
fft_scale: tuple,
overflow_helper: tuple,
sorted_maps: tuple,
calc_overflow: bool,
cache_mov_density_map: bool = True,
):
mov_lhs, mov_rhs, overflow_fn = overflow_helper
mov_sorted_map, mov_conn_sorted_map, filler_sorted_map = sorted_maps
# 1) Compute Density Map
# TODO: Due to atomic add, density map calculation is non-deterministic
# FIXME: replace original floating point based density map calculation to integer based
normalize_node_info = node_size.new_empty((num_nodes, 5)) # x_l, x_h, y_l, y_h, weight
normalize_node_info = density_map_cuda.pre_normalize(
node_pos, node_size, node_weight, expand_ratio, unit_len, normalize_node_info,
num_bin_x, num_bin_y, num_nodes
)
mov_density_map = init_density_map.clone()
mov_density_map = density_map_cuda.forward(
normalize_node_info[mov_lhs:mov_rhs], mov_conn_sorted_map, mov_density_map,
num_bin_x, num_bin_y, mov_rhs - mov_lhs
)
if filler_sorted_map is not None:
filler_density_map = torch.zeros_like(init_density_map)
filler_density_map = density_map_cuda.forward(
normalize_node_info[mov_rhs:], filler_sorted_map, filler_density_map,
num_bin_x, num_bin_y, num_nodes - (mov_rhs - mov_lhs)
)
density_map = filler_density_map
density_map.add_(mov_density_map)
else:
density_map = mov_density_map
if calc_overflow:
overflow = overflow_fn(mov_density_map)
else:
overflow = None
potential_scale, _, force_x_scale, force_y_scale, _, _ = fft_scale
fft_coeff = dct2(density_map)
force_x_map = idxst_idct(fft_coeff * force_x_scale)
force_y_map = idct_idxst(fft_coeff * force_y_scale)
potential_map = idct2(fft_coeff * potential_scale)
grad_mat = torch.vstack(
(force_x_map.unsqueeze(0), force_y_map.unsqueeze(0))
).contiguous() # 2 x M x N
energy = (potential_map * density_map).sum()
grad_weight = -1.0 # Gradient descent
node_grad = normalize_node_info.new_zeros((num_nodes, 2))
node_grad = density_map_cuda.backward(
normalize_node_info, grad_mat, mov_sorted_map, node_grad,
grad_weight, num_bin_x, num_bin_y, num_nodes
)
if not cache_mov_density_map:
mov_density_map = None
return energy, overflow, node_grad, mov_density_map
class ElectronicDensityLayer(torch.nn.Module):
def __init__(
self,
unit_len=None,
inference_mode=True,
num_bin_x=100,
num_bin_y=100,
device=torch.device("cuda:0"),
overflow_helper=None,
expand_ratio=None,
sorted_maps=None,
scale_w_k=True,
):
super(ElectronicDensityLayer, self).__init__()
self.num_bin_x = num_bin_x
self.num_bin_y = num_bin_y
assert overflow_helper is not None
self.overflow_helper = overflow_helper
self.expand_ratio = expand_ratio
self.sorted_maps = sorted_maps
self.inference_mode = inference_mode
self.has_setup_inference_var = False
if unit_len is None:
self.unit_len = torch.tensor([1.0 / num_bin_x, 1.0 / num_bin_y], device=device)
else:
self.unit_len = unit_len
self.min_node_w = self.unit_len[0].item() * math.sqrt(2)
self.min_node_h = self.unit_len[1].item() * math.sqrt(2)
# Pre-compute some constants in FFT computation
w_j = torch.arange(num_bin_x, device=device).float().mul(2 * np.pi / num_bin_x).reshape(num_bin_x, 1)
w_k = torch.arange(num_bin_y, device=device).float().mul(2 * np.pi / num_bin_y).reshape(1, num_bin_y)
# scale_w_k because the aspect ratio of a bin may not be 1
# NOTE: we will not scale down w_k in NN since it may distrub the training
if scale_w_k:
w_k.mul_(self.unit_len[0] / self.unit_len[1])
wj2_plus_wk2 = w_j.pow(2) + w_k.pow(2)
wj2_plus_wk2[0, 0] = 1.0
potential_scale = 1.0 / wj2_plus_wk2
potential_scale[0, 0] = 0.0
force_x_scale = w_j * potential_scale * 0.5
force_y_scale = w_k * potential_scale * 0.5
force_x_coeff = ((-1.0) ** torch.arange(num_bin_x, device=device)).unsqueeze(1)
force_y_coeff = ((-1.0) ** torch.arange(num_bin_y, device=device)).unsqueeze(0)
dct_scalar = 1 / (num_bin_x * num_bin_y)
idct_scalar = num_bin_x * num_bin_y * 4
potential_coeff = 1.0
# potential_scale *= dct_scalar
# force_x_scale *= dct_scalar
# force_y_scale *= dct_scalar
# force_x_coeff *= idct_scalar
# force_y_coeff *= idct_scalar
# potential_coeff = idct_scalar
self.fft_scale = (potential_scale, potential_coeff, force_x_scale, force_y_scale, force_x_coeff, force_y_coeff)
# Cache
self.use_cache_mov_density_map = True
self.mov_density_map = None
# reset global var
dct2_fft2_cache.reset()
def preprocess_inference_var(self, node_pos, node_size, node_weight):
# precompute all common variables for faster inferencing
assert self.inference_mode
assert not self.has_setup_inference_var
if node_weight is None:
# generate all ones node_weight if not given node_weight
node_weight = torch.ones(
node_size.shape[0],
device=node_size.get_device(),
dtype=node_size.dtype,
).detach()
self.cache_node_weight = node_weight
self.has_setup_inference_var = True
def get_cache_var(self, node_pos, node_size, node_weight):
if self.inference_mode:
if not self.has_setup_inference_var:
with torch.no_grad():
self.preprocess_inference_var(
node_pos, node_size, node_weight
)
node_weight = self.cache_node_weight
if node_weight is None:
node_weight = node_pos.new_ones(node_pos.shape[0]).detach()
return node_weight
def get_density_map_naive(
self,
node_pos,
node_size,
init_density_map=None,
):
node_weight = node_size.new_ones(node_pos.shape[0])
if init_density_map is None:
node_pos.new_zeros(self.num_bin_x, self.num_bin_y)
aux_mat = init_density_map.clone()
num_nodes = node_pos.shape[0]
density_map = density_map_cuda.forward_naive(
node_pos,
node_size,
node_weight,
self.unit_len,
aux_mat,
self.num_bin_x,
self.num_bin_y,
num_nodes,
-1.0,
-1.0,
1e-4,
False,
)
return density_map
def direct_calc_overflow(
self,
node_pos,
node_size,
init_density_map,
node_weight=None,
):
mov_lhs, mov_rhs, overflow_fn = self.overflow_helper
if self.use_cache_mov_density_map and self.mov_density_map is not None:
mov_density_map = self.mov_density_map
self.mov_density_map = None
else:
_, mov_conn_sorted_map, _ = self.sorted_maps
num_mov_nodes = mov_rhs - mov_lhs
node_weight = self.get_cache_var(node_pos, node_size, node_weight)
normalize_node_info = node_size.new_zeros((num_mov_nodes, 5)) # x_l, x_h, y_l, y_h, weight
normalize_node_info = density_map_cuda.pre_normalize(
node_pos[mov_lhs:mov_rhs], node_size[mov_lhs:mov_rhs], node_weight[mov_lhs:mov_rhs],
self.expand_ratio[mov_lhs:mov_rhs], self.unit_len, normalize_node_info,
self.num_bin_x, self.num_bin_y, num_mov_nodes
)
aux_mat = init_density_map.clone()
mov_density_map = density_map_cuda.forward(
normalize_node_info, mov_conn_sorted_map, aux_mat,
self.num_bin_x, self.num_bin_y, num_mov_nodes
)
overflow = overflow_fn(mov_density_map)
return overflow
def forward(
self,
node_pos,
node_size,
init_density_map,
node_weight=None,
calc_overflow=True
):
node_weight = self.get_cache_var(
node_pos, node_size, node_weight
)
num_nodes = node_pos.shape[0]
energy, overflow = ElectronicDensityFunction.apply(
node_pos, node_size, node_weight, self.expand_ratio, self.unit_len,
init_density_map, self.num_bin_x, self.num_bin_y, num_nodes,
self.fft_scale, self.overflow_helper, self.sorted_maps, calc_overflow
)
return energy, overflow
def merged_density_loss_grad(
self,
node_pos,
node_size,
init_density_map,
node_weight=None,
calc_overflow=True
):
node_weight = self.get_cache_var(
node_pos, node_size, node_weight
)
num_nodes = node_pos.shape[0]
energy, overflow, node_grad, mov_density_map = merged_density_loss_grad_main(
node_pos, node_size, node_weight, self.expand_ratio, self.unit_len,
init_density_map, self.num_bin_x, self.num_bin_y, num_nodes,
self.fft_scale, self.overflow_helper, self.sorted_maps, calc_overflow,
cache_mov_density_map=self.use_cache_mov_density_map,
)
if self.use_cache_mov_density_map:
self.mov_density_map = mov_density_map
return energy, overflow, node_grad

60
src/core/flute.py Normal file
View File

@ -0,0 +1,60 @@
from cpp_to_py import flute_cpp
import torch
import numpy as np
class Flute(object):
num_threads = 1
@staticmethod
def register(num_threads_=1, POWVFILE="thirdparty/flute/POWV9.dat", POSTFILE="thirdparty/flute/POST9.dat"):
Flute.read_lut(POWVFILE, POSTFILE)
Flute.num_threads = num_threads_
@staticmethod
def read_lut(POWVFILE, POSTFILE):
flute_cpp.read_lut(POWVFILE, POSTFILE) # only need to execute read_lut once
@staticmethod
def flute_wl(xs: list, ys: list):
assert len(xs) == len(ys)
if len(xs) > 150:
raise NotImplementedError("net size is too big, flute only supports 150-pin net")
rsmt_wl = flute_cpp.flute_rsmt_wl(xs, ys)
return rsmt_wl
@staticmethod
def flute_wl_tensor(pos):
if not isinstance(pos, torch.Tensor):
raise NotImplementedError("flute_wl_tensor() only supports torch tensor")
assert pos.shape[1] == 2 # 2D wirelength
if pos.shape[0] > 150:
raise NotImplementedError("net size is too big, flute only supports 150-pin net")
xs_ys = pos.t().tolist()
return flute_cpp.flute_rsmt_wl(xs_ys[0], xs_ys[1])
@staticmethod
def flute_wl_ndarray(pos):
if not isinstance(pos, np.ndarray):
raise NotImplementedError("flute_wl_ndarray() only supports torch ndarray")
assert pos.shape[1] == 2 # 2D wirelength
if pos.shape[0] > 150:
raise NotImplementedError("net size is too big, flute only supports 150-pin net")
xs_ys = pos.T.tolist()
return flute_cpp.flute_rsmt_wl(xs_ys[0], xs_ys[1])
@staticmethod
def flute_wl_mt(pos, hyperedge_list, hyperedge_list_end):
pos_t = pos.t().contiguous().tolist()
hyperedge_list_ = hyperedge_list.tolist()
hyperedge_list_end_ = hyperedge_list_end.tolist()
rsmt = flute_cpp.flute_rsmt_wl_mt(pos_t[0], pos_t[1],
hyperedge_list_, hyperedge_list_end_, Flute.num_threads)
rsmt = torch.FloatTensor(rsmt).to(pos.get_device())
return rsmt
def get_flute_wl(batch, pos):
raise NotImplementedError("Not tested. Please make sure its behavior is correct. (incl. die_scale and site_width)")
# CPU only
y = Flute.flute_wl_mt(pos, batch.hyperedge_list, batch.hyperedge_list_end)
return y.unsqueeze(1)

15
src/core/hpwl.py Normal file
View File

@ -0,0 +1,15 @@
from cpp_to_py import hpwl_cuda
import torch
class HPWL(object):
@staticmethod
def hpwl(pos: torch.Tensor, hyperedge_list, hyperedge_list_end):
hpwl = hpwl_cuda.hpwl(pos, hyperedge_list, hyperedge_list_end)
return hpwl
def get_hpwl(batch, pos):
# CUDA only
y = HPWL.hpwl(pos, batch.hyperedge_list, batch.hyperedge_list_end)
return (
torch.round(y * (batch.die_scale / batch.site_width)).sum(axis=1).unsqueeze(1)
)

View File

@ -0,0 +1,23 @@
import torch
from cpp_to_py import node_pos_to_pin_pos_cuda
class NodePosToPinPosFunction(torch.autograd.Function):
@staticmethod
def forward(
ctx,
node_pos: torch.Tensor,
pin_id2node_id: torch.Tensor,
pin_rel_cpos: torch.Tensor,
):
ctx.num_nodes = node_pos.shape[0]
ctx.save_for_backward(pin_id2node_id)
pin_pos = node_pos_to_pin_pos_cuda.forward(node_pos, pin_id2node_id, pin_rel_cpos)
return pin_pos
@staticmethod
def backward(ctx, pos_grad: torch.Tensor):
pin_id2node_id = ctx.saved_tensors[0]
node_grad = torch.zeros((ctx.num_nodes, 2), dtype=pos_grad.dtype, device=pin_id2node_id.device)
# TODO / NOTE: may cause non-deterministic
node_grad.scatter_add_(0, pin_id2node_id.unsqueeze(1).expand(-1,2), pos_grad)
return node_grad, None, None

232
src/core/torch_dct.py Normal file
View File

@ -0,0 +1,232 @@
import torch
import numpy as np
import torch.fft
def torch_dct_idct(density_map: torch.Tensor, fft_scale):
potential_scale, potential_coeff, force_x_scale, force_y_scale, force_x_coeff, force_y_coeff = fft_scale
fft_coeff = dct_2d(density_map) # Real number, M x N
fft_coeff = fft_coeff * 4 # to align with cuda dct implementation
potential_map = idct_2d(fft_coeff * potential_scale).real * potential_coeff # M x N
force_x_map = compute_electronic_force(fft_coeff, force_x_scale, force_x_coeff, dim=0)
force_y_map = compute_electronic_force(fft_coeff, force_y_scale, force_y_coeff, dim=1)
grad_mat = torch.vstack(
(force_x_map.unsqueeze(0), force_y_map.unsqueeze(0))
) # 2 x M x N
grad_mat = grad_mat.contiguous()
return grad_mat, potential_map
class FFTBasisCache:
def __init__(self) -> None:
self.dct = {}
self.idct = {}
fft_basis_cache = FFTBasisCache()
class DCTmtxCache:
def __init__(self) -> None:
self.dctA = {}
self.idctA = {}
dct_matrix = DCTmtxCache()
def compute_electronic_force(x: torch.Tensor, scale, coeff, dim=0):
# E_x: dim == 0, E_y: dim == 1
assert len(x.shape) == 2
assert dim in [0, 1]
x = x * scale
if dim == 0:
x = torch.cat((x[:1, :] * 0, x[1:, :].flip([0])), dim=0)
else:
x = torch.cat((x[:, :1] * 0, x[:, 1:].flip([1])), dim=1)
x = idct_2d(x).real
x = x * coeff
return x
# https://github.com/zh217/torch-dct/blob/master/torch_dct/_dct.py
def dct(x, norm=None):
"""
Discrete Cosine Transform, Type II (a.k.a. the DCT)
For the meaning of the parameter `norm`, see:
https://docs.scipy.org/doc/scipy-0.14.0/reference/generated/scipy.fftpack.dct.html
:param x: the input signal
:param norm: the normalization, None or 'ortho'
:return: the DCT-II of the signal over the last dimension
"""
x_shape = x.shape
N = x_shape[-1]
x = x.contiguous().view(-1, N)
v = torch.cat([x[:, ::2], x[:, 1::2].flip([1])], dim=1)
Vc = torch.fft.fft(v, dim=1)
if N not in fft_basis_cache.dct.keys():
k = -torch.arange(N, device=x.device)[None, :] * np.pi / (2 * N)
fft_basis_cache.dct[N] = torch.cos(k) + 1j * torch.sin(k)
V = Vc * fft_basis_cache.dct[N]
if norm == "ortho":
V[:, 0] /= np.sqrt(N) * 2
V[:, 1:] /= np.sqrt(N / 2) * 2
V = 2 * V.real.view(*x_shape)
return V
def idct(X, norm=None):
"""
The inverse to DCT-II, which is a scaled Discrete Cosine Transform, Type III
Our definition of idct is that idct(dct(x)) == x
For the meaning of the parameter `norm`, see:
https://docs.scipy.org/doc/scipy-0.14.0/reference/generated/scipy.fftpack.dct.html
:param X: the input signal
:param norm: the normalization, None or 'ortho'
:return: the inverse DCT-II of the signal over the last dimension
"""
x_shape = X.shape
N = x_shape[-1]
X_v = X.contiguous().view(-1, N) / 2
if norm == "ortho":
X_v[:, 0] *= np.sqrt(N) * 2
X_v[:, 1:] *= np.sqrt(N / 2) * 2
if N not in fft_basis_cache.idct.keys():
k = torch.arange(N, device=X.device)[None, :] * np.pi / (2 * N)
fft_basis_cache.idct[N] = torch.cos(k) + 1j * torch.sin(k)
V_t = X_v + 1j * torch.cat([X_v[:, :1] * 0, -X_v.flip([1])[:, :-1]], dim=1)
V = V_t * fft_basis_cache.idct[N]
v = torch.fft.ifft(V, dim=1)
x = v.new_zeros(v.shape)
x[:, ::2] += v[:, : N - (N // 2)]
x[:, 1::2] += v.flip([1])[:, : N // 2]
return x.view(*x_shape)
def dct_2d(x, norm=None):
"""
2-dimentional Discrete Cosine Transform, Type II (a.k.a. the DCT)
For the meaning of the parameter `norm`, see:
https://docs.scipy.org/doc/scipy-0.14.0/reference/generated/scipy.fftpack.dct.html
:param x: the input signal
:param norm: the normalization, None or 'ortho'
:return: the DCT-II of the signal over the last 2 dimensions
"""
X1 = dct(x, norm=norm)
X2 = dct(X1.transpose(-1, -2), norm=norm)
return X2.transpose(-1, -2)
def idct_2d(X, norm=None):
"""
The inverse to 2D DCT-II, which is a scaled Discrete Cosine Transform, Type III
Our definition of idct is that idct_2d(dct_2d(x)) == x
For the meaning of the parameter `norm`, see:
https://docs.scipy.org/doc/scipy-0.14.0/reference/generated/scipy.fftpack.dct.html
:param X: the input signal
:param norm: the normalization, None or 'ortho'
:return: the DCT-II of the signal over the last 2 dimensions
"""
x1 = idct(X, norm=norm)
x2 = idct(x1.transpose(-1, -2), norm=norm)
return x2.transpose(-1, -2)
class LinearDCT(torch.nn.Linear):
"""Implement any DCT as a linear layer; in practice this executes around
50x faster on GPU. Unfortunately, the DCT matrix is stored, which will
increase memory usage.
:param in_features: size of expected input
:param type: which dct function in this file to use"""
def __init__(
self,
in_features,
device=torch.device("cuda:0"),
type="dct",
norm=None,
bias=False,
):
self.type = type
self.device = device
self.N = in_features
self.norm = norm
super(LinearDCT, self).__init__(in_features, in_features, bias=bias)
def reset_parameters(self):
# initialise using dct function
I = torch.eye(self.N, device=self.device)
if self.type == "dct":
self.weight.data = dct(I, norm=self.norm).data.t()
elif self.type == "idct":
self.weight.data = idct(I, norm=self.norm).data.t()
self.weight.requires_grad = False # don't learn this!
def apply_linear_layer_2d(x, linear_dct0, linear_dct1):
"""Can be used with a LinearDCT layer to do a 2D DCT.
:param x: the input signal
:param linear_layer: any PyTorch Linear layer
:return: result of linear layer applied to last 2 dimensions
"""
with torch.no_grad():
X1 = linear_dct1(x)
X2 = linear_dct0(X1.transpose(-1, -2))
return X2.transpose(-1, -2)
def apply_linear_weight_2d(x, W1, W2, scale=None):
# Fastest one but has floating point precision error
if scale is not None:
x = x * scale
X1 = x @ W2
X2 = X1.transpose(-1, -2) @ W1
return X2.transpose(-1, -2)
def get_linear_weight_2d(shape, device, type="dct"):
assert len(shape) == 2
n1, n2 = shape
if type == "dct":
fft_func = dct
elif type == "idct":
fft_func = idct
I1 = torch.eye(n1, device=device)
W1 = fft_func(I1).data
I2 = torch.eye(n2, device=device)
W2 = fft_func(I2).data
return W1, W2
# LinearDCT
def L_dct(x):
tensor_shape = x.shape
if tensor_shape not in dct_matrix.dctA.keys():
layer0, layer1 = get_linear_weight_2d(tensor_shape, x.device, type="dct")
dct_matrix.dctA[tensor_shape] = [layer0, layer1]
layer0, layer1 = dct_matrix.dctA[tensor_shape]
return apply_linear_weight_2d(x, layer0, layer1)
def L_idct(x):
tensor_shape = x.shape
if tensor_shape not in dct_matrix.idctA.keys():
layer0, layer1 = get_linear_weight_2d(tensor_shape, x.device, type="idct")
dct_matrix.idctA[tensor_shape] = [layer0, layer1]
layer0, layer1 = dct_matrix.idctA[tensor_shape]
x = x.to(torch.cfloat)
return apply_linear_weight_2d(x, layer0, layer1)

View File

@ -0,0 +1,127 @@
import torch
from cpp_to_py import wa_wirelength_hpwl_cuda
class HPWLCache:
def __init__(self) -> None:
self.masked_scale_partial_hpwl = None
hpwl_cache = HPWLCache()
class WAWirelengthLossAndHPWL(torch.autograd.Function):
@staticmethod
def forward(
ctx,
node_pos,
pin_id2node_id,
pin_rel_cpos,
hyperedge_list,
hyperedge_list_end,
net_mask,
gamma,
hpwl_scale,
):
(
partial_wa_wl,
node_grad,
partial_hpwl,
) = wa_wirelength_hpwl_cuda.merged_forward_backward_with_hpwl(
node_pos,
pin_id2node_id,
pin_rel_cpos,
hyperedge_list,
hyperedge_list_end,
net_mask,
gamma,
)
sum_hpwl = torch.round(partial_hpwl * hpwl_scale).sum()
ctx.save_for_backward(node_grad)
return torch.sum(partial_wa_wl), sum_hpwl
@staticmethod
def backward(ctx, wa_grad_out, hpwl_grad_out):
node_grad = ctx.saved_tensors[0]
return (node_grad * wa_grad_out,) + (None,) * 7
class WAWirelengthLoss(torch.autograd.Function):
@staticmethod
def forward(
ctx,
node_pos,
pin_id2node_id,
pin_rel_cpos,
hyperedge_list,
hyperedge_list_end,
net_mask,
gamma,
):
partial_wa_wl, node_grad = wa_wirelength_hpwl_cuda.merged_forward_backward(
node_pos,
pin_id2node_id,
pin_rel_cpos,
hyperedge_list,
hyperedge_list_end,
net_mask,
gamma,
)
ctx.save_for_backward(node_grad)
return torch.sum(partial_wa_wl)
@staticmethod
def backward(ctx, wa_grad_out):
node_grad = ctx.saved_tensors[0]
return (node_grad * wa_grad_out,) + (None,) * 6
def merged_wl_loss_grad(
node_pos,
pin_id2node_id,
pin_rel_cpos,
hyperedge_list,
hyperedge_list_end,
net_mask,
hpwl_scale,
gamma,
cache_hpwl=True,
):
(
partial_wa_wl,
node_grad,
partial_hpwl,
) = wa_wirelength_hpwl_cuda.merged_forward_backward_with_masked_scale_hpwl(
node_pos,
pin_id2node_id,
pin_rel_cpos,
hyperedge_list,
hyperedge_list_end,
net_mask,
hpwl_scale,
gamma,
)
if cache_hpwl:
hpwl_cache.masked_scale_partial_hpwl = partial_hpwl
return torch.sum(partial_wa_wl), node_grad
def masked_scale_hpwl(
node_pos,
pin_id2node_id,
pin_rel_cpos,
hyperedge_list,
hyperedge_list_end,
net_mask,
hpwl_scale,
):
if hpwl_cache.masked_scale_partial_hpwl is None:
return wa_wirelength_hpwl_cuda.masked_scale_hpwl_sum(
node_pos,
pin_id2node_id,
pin_rel_cpos,
hyperedge_list,
hyperedge_list_end,
net_mask,
hpwl_scale,
)
else:
return torch.sum(hpwl_cache.masked_scale_partial_hpwl)

812
src/database.py Normal file
View File

@ -0,0 +1,812 @@
"""
Adapted from torch_geometric/data/data.py
"""
import torch
import collections
import copy
import math
from utils import *
def get_dataset(args, logger):
with open("./data/cad/%s/datalist.csv" % (args.dataset), "r") as f:
all_files = f.readlines()
all_files = [line[:-1] for line in all_files]
for cur_file in all_files:
yield cur_file
def load_dataset(args, logger, placement=None):
rawdb, gpdb = None, None
if args.load_from_raw:
logger.info("loading from original benchmark...")
params = get_single_design_params(
args.dataset_root, args.dataset, args.design_name, placement
)
parser = IOParser()
rawdb, gpdb = parser.read(
params, verbose_log=False, lite_mode=True, random_place=False, num_threads=args.num_threads
)
design_info = parser.preprocess_design_info(gpdb)
else:
logger.info("loading from pt benchmark...")
design_pt_path = "./data/cad/%s/%s.pt" % (args.dataset, args.design_name)
design_info = torch.load(design_pt_path)
gpdb = None
data = PlaceData(args, logger, **design_info)
return data, rawdb, gpdb
def size_repr(key, item, indent=0):
indent_str = " " * indent
if torch.is_tensor(item) and item.dim() == 0:
out = item.item()
elif torch.is_tensor(item):
out = str(list(item.size()))
elif isinstance(item, list) or isinstance(item, tuple):
out = str([len(item)])
elif isinstance(item, dict):
lines = [indent_str + size_repr(k, v, 2) for k, v in item.items()]
out = "{\n" + ",\n".join(lines) + "\n" + indent_str + "}"
elif isinstance(item, str):
out = f'"{item}"'
else:
out = str(item)
return f"{indent_str}{key}={out}"
class PlaceData(object):
def __init__(
self,
args,
logger,
node_pos=None,
node_size=None,
pin_rel_cpos=None,
pin_size=None,
pin_id2node_id=None,
hyperedge_index=None,
hyperedge_list=None,
hyperedge_list_end=None,
node_id2region_id=None,
region_boxes=None,
region_boxes_end=None,
dataset_path=None,
benchmark=None,
die_info=None,
site_info=None,
node_type_indices=None,
node_id2node_name=None,
movable_index=None,
connected_index=None,
fixed_index=None,
**kwargs,
):
self.die_info = die_info # lx, hx, ly, hy
self.die_ur = None
self.die_ll = None
self.node_pos = node_pos
self.node_size = node_size
self.pin_rel_cpos = pin_rel_cpos
self.pin_size = pin_size
self.pin_id2node_id = pin_id2node_id
self.hyperedge_index = hyperedge_index
self.hyperedge_list = hyperedge_list
self.hyperedge_list_end = hyperedge_list_end
self.pin_id2net_id = hyperedge_index[1]
self.node_id2region_id = node_id2region_id
self.region_boxes = region_boxes
self.region_boxes_end = region_boxes_end
dataset_format = ""
if "aux" in dataset_path.keys():
dataset_format = "bookshelf"
elif "def" in dataset_path.keys():
dataset_format = "lefdef"
self.__dataset_format__ = dataset_format
self.__dataset_path__ = dataset_path
self.__design_name__ = benchmark + "/" + dataset_path["design_name"]
self.__node_id2node_name__ = node_id2node_name
# NOTE: we set float movable node as connected node for convenience purposes
self.__node_type_indices__ = node_type_indices
self.__movable_index__ = movable_index
self.__movable_connected_index__ = (
movable_index[0],
self.node_type_indices[0][1],
)
self.__connected_index__ = connected_index
self.__fixed_index__ = fixed_index
self.__fixed_connected_index__ = (self.fixed_index[0], self.connected_index[1])
self.__fixed_unconnected_index__ = (
self.connected_index[1],
self.fixed_index[1],
)
self.__site_width__ = site_info[0]
self.__site_height__ = site_info[1]
self.__ori_die_lx__ = die_info[0].item()
self.__ori_die_hx__ = die_info[1].item()
self.__ori_die_ly__ = die_info[2].item()
self.__ori_die_hy__ = die_info[3].item()
self.__num_nodes__ = node_pos.shape[0]
self.__num_pins__ = pin_id2node_id.shape[0]
self.__num_nets__ = hyperedge_list_end.shape[0]
self.__num_bin_x__ = args.num_bin_x
self.__num_bin_y__ = args.num_bin_y
self.__clamp_node__ = args.clamp_node
# fence region
self.__num_regions__ = 1
self.__enable_fence__ = False
# Extra variable to handle corner cases
self.fix_node_in_bd_mask = None
self.dummy_macro_pos = None
self.dummy_macro_size = None
# filler
self.filler_size = None
self.__logger__ = logger
self.__args__ = args
for key, item in kwargs.items():
self[key] = item
@property
def dataset_format(self):
if hasattr(self, "__dataset_format__"):
return self.__dataset_format__
@property
def dataset_path(self):
if hasattr(self, "__dataset_path__"):
return self.__dataset_path__
@property
def design_name(self):
if hasattr(self, "__design_name__"):
return self.__design_name__
@property
def node_id2node_name(self):
if hasattr(self, "__node_id2node_name__"):
return self.__node_id2node_name__
@property
def node_type_indices(self):
if hasattr(self, "__node_type_indices__"):
return self.__node_type_indices__
@property
def movable_index(self):
if hasattr(self, "__movable_index__"):
return self.__movable_index__
@property
def movable_connected_index(self):
if hasattr(self, "__movable_connected_index__"):
return self.__movable_connected_index__
@property
def connected_index(self):
if hasattr(self, "__connected_index__"):
return self.__connected_index__
@property
def fixed_index(self):
if hasattr(self, "__fixed_index__"):
return self.__fixed_index__
@property
def fixed_connected_index(self):
if hasattr(self, "__fixed_connected_index__"):
return self.__fixed_connected_index__
@property
def fixed_unconnected_index(self):
if hasattr(self, "__fixed_unconnected_index__"):
return self.__fixed_unconnected_index__
@property
def site_width(self):
if hasattr(self, "__site_width__"):
return self.__site_width__
@property
def site_height(self):
if hasattr(self, "__site_height__"):
return self.__site_height__
@property
def ori_die_lx(self):
if hasattr(self, "__ori_die_lx__"):
return self.__ori_die_lx__
@property
def ori_die_hx(self):
if hasattr(self, "__ori_die_hx__"):
return self.__ori_die_hx__
@property
def ori_die_ly(self):
if hasattr(self, "__ori_die_ly__"):
return self.__ori_die_ly__
@property
def ori_die_hy(self):
if hasattr(self, "__ori_die_hy__"):
return self.__ori_die_hy__
@property
def die_shift(self):
if hasattr(self, "__die_shift__"):
return self.__die_shift__
@property
def die_scale(self):
if hasattr(self, "__die_scale__"):
return self.__die_scale__
@property
def num_nodes(self):
if hasattr(self, "__num_nodes__"):
return self.__num_nodes__
@property
def num_pins(self):
if hasattr(self, "__num_pins__"):
return self.__num_pins__
@property
def num_nets(self):
if hasattr(self, "__num_nets__"):
return self.__num_nets__
@property
def num_fillers(self):
if hasattr(self, "__num_fillers__"):
return self.__num_fillers__
@property
def num_bin_x(self):
if hasattr(self, "__num_bin_x__"):
return self.__num_bin_x__
@property
def num_bin_y(self):
if hasattr(self, "__num_bin_y__"):
return self.__num_bin_y__
@property
def clamp_node(self):
if hasattr(self, "__clamp_node__"):
return self.__clamp_node__
@property
def enable_fence(self):
if hasattr(self, "__enable_fence__"):
return self.__enable_fence__
@property
def num_regions(self):
if hasattr(self, "__num_regions__"):
return self.__num_regions__
@property
def total_mov_area_without_filler(self):
if hasattr(self, "__total_mov_area_without_filler__"):
return self.__total_mov_area_without_filler__
@property
def bin_area(self):
if hasattr(self, "__bin_area__"):
return self.__bin_area__
@classmethod
def from_dict(cls, dictionary):
r"""Creates a data object from a python dictionary."""
data = cls()
for key, item in dictionary.items():
data[key] = item
return data
def to_dict(self):
return {key: item for key, item in self}
def to_namedtuple(self):
keys = self.keys
DataTuple = collections.namedtuple("DataTuple", keys)
return DataTuple(*[self[key] for key in keys])
def __getitem__(self, key):
r"""Gets the data of the attribute :obj:`key`."""
return getattr(self, key, None)
def __setitem__(self, key, value):
"""Sets the attribute :obj:`key` to :obj:`value`."""
setattr(self, key, value)
def __delitem__(self, key):
r"""Delete the data of the attribute :obj:`key`."""
return delattr(self, key)
@property
def keys(self):
r"""Returns all names of graph attributes."""
keys = [key for key in self.__dict__.keys() if self[key] is not None]
keys = [key for key in keys if key[:2] != "__" and key[-2:] != "__"]
return keys
def __len__(self):
r"""Returns the number of all present attributes."""
return len(self.keys)
def __contains__(self, key):
r"""Returns :obj:`True`, if the attribute :obj:`key` is present in the
data."""
return key in self.keys
def __iter__(self):
r"""Iterates over all present attributes in the data, yielding their
attribute names and content."""
for key in sorted(self.keys):
yield key, self[key]
def __call__(self, *keys):
r"""Iterates over all attributes :obj:`*keys` in the data, yielding
their attribute names and content.
If :obj:`*keys` is not given this method will iterative over all
present attributes."""
for key in sorted(self.keys) if not keys else keys:
if key in self:
yield key, self[key]
def __apply__(self, item, func):
if torch.is_tensor(item):
return func(item)
elif isinstance(item, (tuple, list)):
return [self.__apply__(v, func) for v in item]
elif isinstance(item, dict):
return {k: self.__apply__(v, func) for k, v in item.items()}
else:
return item
def apply(self, func, *keys):
r"""Applies the function :obj:`func` to all tensor attributes
:obj:`*keys`. If :obj:`*keys` is not given, :obj:`func` is applied to
all present attributes.
"""
for key, item in self(*keys):
self[key] = self.__apply__(item, func)
return self
def contiguous(self, *keys):
r"""Ensures a contiguous memory layout for all attributes :obj:`*keys`.
If :obj:`*keys` is not given, all present attributes are ensured to
have a contiguous memory layout."""
return self.apply(lambda x: x.contiguous(), *keys)
def to(self, device, *keys, **kwargs):
r"""Performs tensor dtype and/or device conversion to all attributes
:obj:`*keys`.
If :obj:`*keys` is not given, the conversion is applied to all present
attributes."""
return self.apply(lambda x: x.to(device, **kwargs), *keys)
def cpu(self, *keys):
r"""Copies all attributes :obj:`*keys` to CPU memory.
If :obj:`*keys` is not given, the conversion is applied to all present
attributes."""
return self.apply(lambda x: x.cpu(), *keys)
def cuda(self, device=None, non_blocking=False, *keys):
r"""Copies all attributes :obj:`*keys` to CUDA memory.
If :obj:`*keys` is not given, the conversion is applied to all present
attributes."""
return self.apply(
lambda x: x.cuda(device=device, non_blocking=non_blocking), *keys
)
def clone(self):
return self.__class__.from_dict(
{
k: v.clone() if torch.is_tensor(v) else copy.deepcopy(v)
for k, v in self.__dict__.items()
}
)
def pin_memory(self, *keys):
r"""Copies all attributes :obj:`*keys` to pinned memory.
If :obj:`*keys` is not given, the conversion is applied to all present
attributes."""
return self.apply(lambda x: x.pin_memory(), *keys)
def record_stream(self, stream: torch.cuda.Stream, *keys):
r"""Ensures that the tensor memory is not reused for another tensor
until all current work queued on :obj:`stream` has been completed.
If :obj:`*keys` is not given, this will be ensured for all present
attributes."""
def _record_stream(x):
x.record_stream(stream)
return x
return self.apply(_record_stream, *keys)
def __repr__(self):
cls = str(self.__class__.__name__)
has_dict = any([isinstance(item, dict) for _, item in self])
if not has_dict:
info = [size_repr(key, item) for key, item in self]
return "{}({}, {})".format(cls, self.design_name, ", ".join(info))
else:
info = [size_repr(key, item, indent=2) for key, item in self]
return "{}({}, \n{}\n)".format(cls, self.design_name, ",\n".join(info))
def backup_ori_var(self):
# backup original position and size
self.__ori_die_info__ = self.die_info.clone().cpu().numpy()
self.__ori_node_pos__ = self.node_pos.clone().cpu().numpy()
self.__ori_node_size__ = self.node_size.clone().cpu().numpy()
self.__ori_pin_rel_cpos__ = self.pin_rel_cpos.clone().cpu().numpy()
self.__ori_pin_size__ = self.pin_size.clone().cpu().numpy()
self.__ori_region_boxes__ = self.region_boxes.clone().cpu().numpy()
dtype, device = self.die_info.dtype, self.die_info.device
self.__die_shift__ = torch.tensor([0.0, 0.0], dtype=dtype, device=device)
self.__die_scale__ = torch.tensor([1.0, 1.0], dtype=dtype, device=device)
return self
def preshift(self):
# shift die info to (0.0, hx, 0.0, hy)
die_lx, _, die_ly, _ = self.die_info.tolist()
die_shift = torch.tensor(
[die_lx, die_ly], dtype=self.die_info.dtype, device=self.die_info.device,
)
self.die_info = (self.die_info.reshape(2, 2).t() - die_shift).t().reshape(-1)
self.region_boxes = (
(self.region_boxes.reshape(-1, 2, 2).permute(0, 2, 1) - die_shift)
.permute(0, 2, 1)
.reshape(-1, 4)
)
self.node_pos -= die_shift
self.__die_shift__ += die_shift
return self
def prescale_by_site_width(self):
# inplace scaling
self.die_info /= self.site_width
self.region_boxes /= self.site_width
self.node_pos /= self.site_width
self.node_size /= self.site_width
self.pin_rel_cpos /= self.site_width
self.pin_size /= self.site_width
self.__die_scale__ *= self.site_width
return self
def prescale(self):
# scale die info to (0.0, 1.0, 0.0, 1.0)
die_lx, die_hx, die_ly, die_hy = self.die_info.tolist()
die_scale = torch.tensor(
[die_hx - die_lx, die_hy - die_ly],
dtype=self.die_info.dtype,
device=self.die_info.device,
)
self.node_pos /= die_scale
self.node_size /= die_scale
self.pin_rel_cpos /= die_scale
self.pin_size /= die_scale
self.die_info = (self.die_info.reshape(2, 2).t() / die_scale).t().reshape(-1)
self.region_boxes = (
(self.region_boxes.reshape(-1, 2, 2).permute(0, 2, 1) / die_scale)
.permute(0, 2, 1)
.reshape(-1, 4)
)
self.__die_scale__ *= die_scale
return self
def pre_compute_var(self):
args = self.__args__
device = self.node_size.get_device()
# die related
lx, hx, ly, hy = self.die_info.tolist()
self.unit_len = torch.tensor(
[(hx - lx) / self.num_bin_x, (hy - ly) / self.num_bin_y], device=device
)
self.die_ur = self.die_info.reshape(2, 2).t()[1].clone()
self.die_ll = self.die_info.reshape(2, 2).t()[0].clone()
self.hpwl_scale = self.die_scale / self.site_width
# node related
self.node_area = torch.prod(self.node_size, 1).unsqueeze(1)
self.node_to_num_pins = torch.zeros(self.num_nodes, device=device)
v = torch.ones(self.pin_id2node_id.shape[0], device=device)
self.node_to_num_pins.scatter_add_(0, self.pin_id2node_id, v)
self.node_to_num_pins.unsqueeze_(1)
# net related
start_idx = self.hyperedge_list_end.roll(1)
start_idx[0] = 0
self.net_to_num_pins = self.hyperedge_list_end - start_idx
self.net_mask = torch.logical_and(
self.net_to_num_pins <= args.ignore_net_degree, self.net_to_num_pins >= 2
) # 0: ignore, 1: consider in wirelength calculation
# obj related
mov_lhs, mov_rhs = self.movable_index
mov_cell_area = torch.prod(self.node_size[mov_lhs:mov_rhs, ...], 1)
self.__total_mov_area_without_filler__ = torch.sum(mov_cell_area).item()
self.__bin_area__ = torch.prod(self.unit_len).item()
return self
def init_fence_region(self):
offset = self.region_boxes_end.diff().tolist()
regions = torch.split(self.region_boxes[1:], offset)
self.regions = (
self.region_boxes[0],
*regions,
) # first region is the default region (core area)
self.__num_regions__ = len(self.regions)
self.__enable_fence__ = len(self.regions) > 1
return self
def compute_filler(self, args, logger):
if self.enable_fence:
return self.compute_filler_with_fence(args, logger)
else:
return self.compute_filler_without_fence(args, logger)
def compute_filler_with_fence(self, args, logger):
raise NotImplementedError("We haven't yet supported fence region.")
def compute_filler_without_fence(self, args, logger):
self.__num_fillers__ = 0
if args.use_filler:
mov_lhs, mov_rhs = self.movable_index
mov_node_size = self.node_size[mov_lhs:mov_rhs, ...]
die_area = torch.prod(self.die_ur - self.die_ll)
# init_density_map already multiplies with args.target_density,
# we need to divide it back
ori_dmap = (self.init_density_map / args.target_density).sum()
# init_density_map are all normalized to (0.0, 1.0)
fixed_node_area = ori_dmap * self.bin_area
placeable_area = die_area - fixed_node_area
if True:
mov_cell_area = torch.prod(mov_node_size, 1)
num_movable_nodes = mov_rhs - mov_lhs
mov_node_xsize_order = torch.argsort(mov_node_size[:, 0])
filler_size_x = torch.mean(
mov_node_size[:, 0][
mov_node_xsize_order[
int(num_movable_nodes * 0.05) : int(
num_movable_nodes * 0.95
)
]
]
)
filler_size_y = self.site_height / self.die_scale[1]
total_filler_area = max(
args.target_density * placeable_area - torch.sum(mov_cell_area),
0.0,
)
single_filler_size = torch.tensor(
[filler_size_x, filler_size_y],
device=mov_node_size.device,
dtype=mov_node_size.dtype,
)
self.__num_fillers__ = int(
torch.round(total_filler_area / (filler_size_x * filler_size_y))
)
else:
# Original implementation
mov_cell_area = torch.prod(mov_node_size, 1)
total_filler_area = max(
args.target_density * placeable_area
- torch.sum(mov_cell_area).item(),
0.0,
)
single_filler_area = torch.mean(mov_cell_area)
single_filler_size = single_filler_area.sqrt().repeat(2)
self.__num_fillers__ = int(total_filler_area / single_filler_area)
if self.num_fillers > 0:
self.filler_size = single_filler_size.repeat(self.num_fillers, 1)
logger.info(
"#Fillers: %d Filler size: (%.4e, %.4e)"
% (
self.num_fillers,
single_filler_size[0].item(),
single_filler_size[1].item(),
)
)
else:
logger.warning(
"num_fillers[%d] is smaller or equal to 0. Please make sure target_density[%.2f]"
" is larger than movable cell utilization[%.2f]. use_filler is disable."
% (
self.num_fillers,
args.target_density,
torch.sum(torch.prod(mov_node_size, 1)),
)
)
args.use_filler = False
return self
def compute_precond_var(self):
mov_lhs, mov_rhs = self.movable_index
self.mov_node_area = self.node_area[mov_lhs:mov_rhs]
self.mov_node_to_num_pins = self.node_to_num_pins[mov_lhs:mov_rhs]
if self.filler_size is not None:
num_fillers = self.filler_size.shape[0]
filler_area = torch.prod(self.filler_size, 1).unsqueeze(1)
self.mov_node_area = torch.cat((self.mov_node_area, filler_area), dim=0)
filler_to_num_pins = self.mov_node_to_num_pins.new_zeros((num_fillers, 1))
assert filler_to_num_pins.shape == filler_area.shape
self.mov_node_to_num_pins = torch.cat(
(self.mov_node_to_num_pins, filler_to_num_pins), dim=0
)
return self
def compute_sorted_node_map(self):
_, mov_sorted_map = torch.sort(self.mov_node_area.flatten(), descending=True)
mov_sorted_map = mov_sorted_map.contiguous()
mov_conn_sorted_map = mov_sorted_map
filler_sorted_map = None
if self.filler_size is not None:
mov_lhs, mov_rhs = self.movable_index
_, mov_conn_sorted_map = torch.sort(
self.mov_node_area[mov_lhs:mov_rhs].flatten(), descending=True
)
_, filler_sorted_map = torch.sort(
self.mov_node_area[mov_rhs:].flatten(), descending=True
)
mov_conn_sorted_map = mov_conn_sorted_map.contiguous()
filler_sorted_map = filler_sorted_map.contiguous()
self.sorted_maps = (mov_sorted_map, mov_conn_sorted_map, filler_sorted_map)
def logging_statistics(self):
args = self.__args__
logger = self.__logger__
content = "\n===================\n"
content += "#nodes = %d, #nets = %d, #pins = %d\n" % (
self.num_nodes,
self.num_nets,
self.num_pins,
)
num_conmov_nodes = self.node_type_indices[0][1] - self.node_type_indices[0][0]
num_fltmov_nodes = self.node_type_indices[1][1] - self.node_type_indices[1][0]
num_confix_nodes = self.node_type_indices[2][1] - self.node_type_indices[2][0]
num_fltfix_nodes = self.node_type_indices[6][1] - self.node_type_indices[6][0]
num_coniopin = self.node_type_indices[3][1] - self.node_type_indices[3][0]
num_fltiopin = self.node_type_indices[5][1] - self.node_type_indices[5][0]
num_blkg = self.node_type_indices[4][1] - self.node_type_indices[4][0]
content += "#Mov = %d, #Fix = %d, #IOPin = %d, #Blkg = %d\n" % (
num_conmov_nodes + num_fltmov_nodes,
num_confix_nodes + num_fltfix_nodes,
num_coniopin + num_fltiopin,
num_blkg,
)
content += (
"#ConnMov = %d, #FloatMov = %d, #ConnFix = %d, #FloatFix = %d, #ConnIOPin = %d, #FloatIOPin = %d\n"
% (
num_conmov_nodes,
num_fltmov_nodes,
num_confix_nodes,
num_fltfix_nodes,
num_coniopin,
num_fltiopin,
)
)
content += "Core Info " + str(self.die_info.tolist()) + "\n"
content += "Site Width = %d, Row Height = %d\n" % (
self.site_width,
self.site_height,
)
content += "#Bins = (%d, %d), UnitLen = (%.5f, %.5f)\n" % (
self.num_bin_x,
self.num_bin_y,
self.unit_len[0],
self.unit_len[1],
)
content += "target density = %.2f\n" % (args.target_density)
content += "==================="
logger.info(content)
return self
def preprocess(self):
args = self.__args__
self.backup_ori_var()
self.preshift()
self.prescale_by_site_width()
if args.scale_design:
self.prescale()
self.pre_compute_var()
self.init_fence_region()
self.logging_statistics()
return self
def init_filler(self):
self.compute_filler(self.__args__, self.__logger__)
self.compute_precond_var()
self.compute_sorted_node_map()
return self
def get_mov_node_info(self, init_method="randn_center"):
args = self.__args__
mov_lhs, mov_rhs = self.movable_index
mov_node_pos = self.node_pos[mov_lhs:mov_rhs, ...]
mov_node_size = self.node_size[mov_lhs:mov_rhs, ...]
if init_method == "randn_center":
scale = (self.die_ur - self.die_ll) * 0.001
loc = (self.die_ur + self.die_ll) * 0.5
mov_node_pos = torch.randn_like(mov_node_pos) * scale + loc
if self.num_fillers > 0:
if self.enable_fence:
raise NotImplementedError("We haven't yet supported fence region.")
else:
filler_pos = torch.rand(
(self.num_fillers, 2),
dtype=mov_node_size.dtype,
device=mov_node_size.device,
)
scale = self.die_ur - self.die_ll
shift = self.die_ll
filler_pos = filler_pos * scale + shift
mov_node_pos = torch.cat([mov_node_pos, filler_pos], dim=0)
mov_node_size = torch.cat([mov_node_size, self.filler_size], dim=0)
if args.noise_ratio > 0:
noise = torch.rand_like(mov_node_pos)
noise.sub_(0.5).mul_(mov_node_size).mul_(args.noise_ratio)
mov_node_pos += noise
expand_ratio = mov_node_pos.new_ones((mov_node_pos.shape[0]))
if self.clamp_node:
mov_node_area = torch.prod(mov_node_size, 1)
clamp_mov_node_size = mov_node_size.clamp(min=self.unit_len * math.sqrt(2))
clamp_mov_node_area = torch.prod(clamp_mov_node_size, 1)
# update
expand_ratio = mov_node_area / clamp_mov_node_area
mov_node_size = clamp_mov_node_size
return mov_node_pos, mov_node_size, expand_ratio
def write_pl(self, node_pos, gp_prefix):
# support floating point based .pl file in global placement output
pl_file = gp_prefix + ".pl"
content = "UCLA pl 1.0\n"
# use float here
exact_node_pos = node_pos * self.die_scale + self.die_shift
exact_node_size = torch.round(self.node_size * self.die_scale)
tmp = (exact_node_size.div(self.site_width, rounding_mode="floor") == 1).bool()
is_terminal_ni = torch.logical_and(tmp[:, 0], tmp[:, 1]).cpu()
exact_node_lpos = (
(exact_node_pos - exact_node_size / 2).div_(self.site_width).cpu()
)
for i in range(self.num_nodes):
content += "\n%s %g %g : %s" % (
self.node_id2node_name[i],
exact_node_lpos[i, 0],
exact_node_lpos[i, 1],
"N"
)
if i >= self.fixed_index[0]:
if is_terminal_ni[i]:
content += " /FIXED_NI"
else:
content += " /FIXED"
with open(pl_file, "w") as f:
f.write(content)

55
src/evaluator.py Normal file
View File

@ -0,0 +1,55 @@
import torch
from .database import PlaceData
from .core import NodePosToPinPosFunction, get_hpwl, masked_scale_hpwl
def get_obj_value(pin_pos, density_map, data, args):
with torch.no_grad():
hpwl = torch.sum(get_hpwl(data, pin_pos.detach()))
overflow_sum = ((density_map - args.target_density) * data.bin_area).clamp_(min=0.0).sum()
overflow = overflow_sum / data.total_mov_area_without_filler
return hpwl, overflow
def evaluate_placement(node_pos, density_map_layer, init_density_map, data: PlaceData, args):
# NOTE: since some nets are masked in WAWirelengthLossAndHPWL, hpwl
# from WAWirelengthLossAndHPWL may underestimate, this function return the
# exact value of hpwl
# Original overflow calculation uses the clamp node size (expand ratio),
# this function uses the exact node size to evaluate the overflow
mov_lhs, mov_rhs = data.movable_index
fix_lhs, fix_rhs = data.fixed_connected_index
conn_node_pos = torch.cat([
node_pos[mov_lhs:mov_rhs], node_pos[fix_lhs:fix_rhs]
], dim=0)
pin_pos = NodePosToPinPosFunction.apply(
conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos
)
density_map = density_map_layer.get_density_map_naive(
node_pos[mov_lhs:mov_rhs], data.node_size[mov_lhs:mov_rhs], init_density_map
)
hpwl, overflow = get_obj_value(pin_pos, density_map, data, args)
return hpwl, overflow
def fast_evaluator(
mov_node_pos,
constraint_fn=None,
mov_node_size=None,
init_density_map=None,
density_map_layer=None,
conn_fix_node_pos=None,
ps=None,
data=None,
args=None,
):
mov_lhs, mov_rhs = data.movable_index
mov_node_pos = constraint_fn(mov_node_pos)
conn_node_pos = mov_node_pos[mov_lhs:mov_rhs, ...]
conn_node_pos = torch.cat([conn_node_pos, conn_fix_node_pos], dim=0)
masked_hpwl = masked_scale_hpwl(
conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos,
data.hyperedge_list, data.hyperedge_list_end, data.net_mask, data.hpwl_scale
)
overflow = density_map_layer.direct_calc_overflow(
mov_node_pos, mov_node_size, init_density_map
)
return masked_hpwl, overflow

74
src/initializer.py Normal file
View File

@ -0,0 +1,74 @@
import torch
from .database import PlaceData
from cpp_to_py import density_map_cuda
from .core import WAWirelengthLossAndHPWL
from .calculator import calc_grad
def get_init_density_map(data: PlaceData, args, logger):
lhs, rhs = data.fixed_index
device = data.node_size.get_device()
dtype = data.node_size.dtype
zeros_density_map = torch.zeros(
(data.num_bin_x, data.num_bin_y), device=device, dtype=dtype,
)
if lhs == rhs:
return zeros_density_map
# get fix nodes which are located inside die
node_pos = data.node_pos[lhs:rhs]
node_size = data.node_size[lhs:rhs]
# if data.fix_node_in_bd_mask is not None:
# node_pos = node_pos[data.fix_node_in_bd_mask]
# node_size = node_size[data.fix_node_in_bd_mask]
# if data.dummy_macro_pos is not None and data.dummy_macro_size is not None:
# node_pos = torch.cat([node_pos, data.dummy_macro_pos], dim=0)
# node_size = torch.cat([node_size, data.dummy_macro_size], dim=0)
# if node_size.shape[0] == 0:
# return zeros_density_map
node_weight = node_size.new_ones(node_size.shape[0])
init_density_map = density_map_cuda.forward_naive(
node_pos, node_size, node_weight, data.unit_len, zeros_density_map,
data.num_bin_x, data.num_bin_y, node_pos.shape[0], -1.0, -1.0, 1e-4, False
)
init_density_map = init_density_map.contiguous()
if (init_density_map > 1).sum() > 0:
logger.warning("Some bins in init_density_map are overflow. Clamp them.")
if (init_density_map < 0).sum() > 0:
logger.error("init_density_map has negative value. Please check.")
init_density_map.clamp_(min=0.0, max=1.0).mul_(args.target_density)
data.init_density_map = init_density_map
return data.init_density_map
def init_params(
mov_node_pos, trunc_node_pos_fn, mov_lhs, mov_rhs, conn_fix_node_pos,
density_map_layer, mov_node_size, init_density_map, optimizer, ps, data, args
):
mov_node_pos = trunc_node_pos_fn(mov_node_pos)
conn_node_pos = mov_node_pos[mov_lhs:mov_rhs, ...]
conn_node_pos = torch.cat(
[conn_node_pos, conn_fix_node_pos], dim=0
)
wl_loss, hpwl = WAWirelengthLossAndHPWL.apply(
conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos,
data.hyperedge_list, data.hyperedge_list_end, data.net_mask,
ps.wa_coeff, data.hpwl_scale
)
density_loss, overflow = density_map_layer(
mov_node_pos, mov_node_size, init_density_map
)
wl_grad, density_grad = calc_grad(
optimizer, mov_node_pos, wl_loss, density_loss
)
init_density_weight = (wl_grad.norm(p=1) / density_grad.norm(p=1)).detach()
# init_density_weight = (wl_grad.norm(p=1) / grad_mat.norm(p=1)).detach()
ps.set_init_param(init_density_weight, data, density_loss)
# Nesterove learning rate initialization
def estimate_initial_learning_rate(obj_and_grad_fn, constraint_fn, x_k, lr):
x_k = constraint_fn(x_k).clone().detach().requires_grad_(True)
obj_k, g_k = obj_and_grad_fn(x_k)
x_k_1 = (constraint_fn(x_k - lr * g_k)).clone().detach().requires_grad_(True)
obj_k_1, g_k_1 = obj_and_grad_fn(x_k_1)
return (x_k - x_k_1).norm(p=2) / (g_k - g_k_1).norm(p=2)

128
src/nesterov_optimizer.py Normal file
View File

@ -0,0 +1,128 @@
# We accelerate the nesterov optimizer in DREAMPlace by removing some
# unused var and pre-computing step size
import torch
from torch.optim.optimizer import Optimizer, required
import numpy as np
import numba
import logging
numba_logger = logging.getLogger('numba')
numba_logger.setLevel(logging.WARNING)
@numba.jit(numba.float32[:](numba.int32), nopython=True, nogil=True, cache=True)
def calc_nesterov_step_size32(N):
a_k = np.empty(N + 1, dtype=np.float32)
a_k[0] = 1.0
for i in range(1, len(a_k)):
a_k[i] = (1 + np.sqrt(4 * np.power(a_k[i-1], 2) + 1)) / 2
a_kp1 = np.roll(a_k, -1)[:-1]
a_k = a_k[:-1]
coef = (a_k - 1) / a_kp1
return coef
class NesterovOptimizer(Optimizer):
"""
@brief Follow the Nesterov's implementation of e-place algorithm 2
http://cseweb.ucsd.edu/~jlu/papers/eplace-todaes14/paper.pdf
"""
def __init__(self, params, lr=required):
"""
@brief initialization
@param params variable to optimize
@param lr learning rate
"""
if lr is not required and lr < 0.0:
raise ValueError("Invalid learning rate: {}".format(lr))
# u_k is major solution
# v_k is reference solution
# obj_k is the objective at v_k
# alpha_k is the step size
# v_k_1 is previous reference solution
# g_k_1 is gradient to v_k_1
# obj_k_1 is the objective at v_k_1
defaults = dict(lr=lr, u_k=[], v_k=[], g_k=[], obj_k=[], alpha_k=[], v_kp1 = [])
super().__init__(params, defaults)
self.max_cache_steps = 10000
self.steps = 0
coef = calc_nesterov_step_size32(self.max_cache_steps)
self.last_coef = coef[-1].item()
self.coef = torch.from_numpy(coef)
if len(self.param_groups) != 1:
raise ValueError("Only parameters with single tensor is supported")
def __setstate__(self, state):
super().__setstate__(state)
def step(self, closure):
"""
@brief Performs a single optimization step.
@param closure A callable closure function that reevaluates the model and returns the loss.
"""
for group in self.param_groups:
for i, p in enumerate(group['params']):
if p.grad is None:
continue
if not group['u_k']:
group['u_k'].append(p.data.clone())
# directly use p as v_k to save memory
group['v_k'].append(p)
obj, grad = closure(group['v_k'][i])
group['g_k'].append(grad.data.clone()) # must clone
group['obj_k'].append(obj.data.clone())
u_k = group['u_k'][i]
v_k = group['v_k'][i]
g_k = group['g_k'][i]
obj_k = group['obj_k'][i]
if not group['alpha_k']:
# init alpha_k
self.coef = self.coef.to(g_k.device)
v_k_1 = (group['v_k'][i] - group['lr'] * g_k).detach().requires_grad_(True)
obj_k_1, g_k_1 = closure(v_k_1)
group['alpha_k'].append((v_k-v_k_1).norm(p=2) / (g_k-g_k_1).norm(p=2))
alpha_k = group['alpha_k'][i]
if not group['v_kp1']:
group['v_kp1'].append(torch.zeros_like(v_k, requires_grad=True))
v_kp1 = group['v_kp1'][i]
if self.steps < self.max_cache_steps:
coef = self.coef[self.steps]
else:
coef = self.last_coef
alpha_kp1 = 0
backtrack_cnt = 0
max_backtrack_cnt = 10
while True:
u_kp1 = v_k - alpha_k * g_k
v_kp1.data.copy_(u_kp1 + coef * (u_kp1 - u_k))
f_kp1, g_kp1 = closure(v_kp1)
alpha_kp1 = (v_kp1.data-v_k.data).norm(2) / (g_kp1.data-g_k.data).norm(2)
backtrack_cnt += 1
if alpha_kp1 > 0.95 * alpha_k or backtrack_cnt >= max_backtrack_cnt:
break
alpha_k.data.copy_(alpha_kp1.data)
obj = obj_k.data.clone()
u_k.data.copy_(u_kp1.data)
v_k.data.copy_(v_kp1.data)
g_k.data.copy_(g_kp1.data)
obj_k.data.copy_(f_kp1.data)
alpha_k.data.copy_(alpha_kp1.data)
self.steps += 1
# although the solution should be u_k
# we need the gradient of v_k
# the update of density weight also requires v_k
# I do not know how to copy u_k back to p when exit yet
#p.data.copy_(v_k.data)
return obj

385
src/param_scheduler.py Normal file
View File

@ -0,0 +1,385 @@
from .database import PlaceData
import torch
import numpy as np
import matplotlib.pyplot as plt
import os
class MetricRecorder:
def __init__(self, **kwargs) -> None:
for key, item in kwargs.items():
if not isinstance(item, list):
raise TypeError("%s is not a list for key %s" % (item, key))
self[key] = item
def push(self, **kwargs) -> None:
for key, item in kwargs.items():
if type(item) == torch.Tensor and item.dim() == 0:
item = item.item()
elif np.issubdtype(type(item), np.floating):
item = float(item)
elif np.issubdtype(type(item), np.integer):
item = int(item)
if not type(item) == int and not type(item) == float:
raise TypeError(
"item %s type(%s) is not a number for key %s"
% (item, type(item), key)
)
self[key].append(item)
def visualize(self, prefix):
for key, value in self:
x = list(range(len(value)))
plt.plot(x, value, label=key)
plt.legend()
plt.savefig(prefix + "%s.png" % key)
plt.close()
def __getitem__(self, key):
return getattr(self, key, None)
def __setitem__(self, key, value):
setattr(self, key, value)
def __delitem__(self, key):
return delattr(self, key)
@property
def keys(self):
keys = [key for key in self.__dict__.keys() if self[key] is not None]
return keys
def __len__(self):
r"""Returns the number of all present attributes."""
return len(self.keys)
def __contains__(self, key):
r"""Returns :obj:`True`, if the attribute :obj:`key` is present in the
data."""
return key in self.keys
def __iter__(self):
r"""Iterates over all present attributes in the data, yielding their
attribute names and content."""
for key in sorted(self.keys):
yield key, self[key]
class ParamScheduler:
def __init__(self, data: PlaceData, args, logger) -> None:
self.__logger__ = logger
self.data = data
self.iter = 0
# metrics
self.metrics = [
"hpwl",
"overflow",
"mu",
"wa_coeff",
"density_weight",
"precond_coef",
"weighted_weight",
"force_ratio",
]
self.recorder = MetricRecorder(**{m: [] for m in self.metrics})
# best solution
# main solution
self.best_sol: torch.Tensor = None
self.best_metric = {"overflow": float("inf"), "hpwl": float("inf")}
# aux solution
self.best_sol_aux: torch.Tensor = None
self.best_metric_aux = {"overflow": float("inf"), "hpwl": float("inf")}
# rollback solution
self.best_sol_rollback: torch.Tensor = None
self.best_metric_rollback = {"overflow": float("inf"), "hpwl": float("inf")}
# params
self.precond_coef = 1.0
self.precond_weight = None
self.density_weight = args.density_weight
self.density_weight_coef = args.density_weight_coef
self.wa_coeff = args.wa_coeff
self.base_gamma = args.wa_coeff * torch.sum(data.unit_len).item()
self.wa_coeff = 10 * self.base_gamma
self.use_precond = args.use_precond
self.mu = 1.0
self.max_life = 30
self.life = self.max_life
self.stop_overflow = args.stop_overflow
self.skip_update = False if args.enable_skip_update else None
self.enable_fence = data.enable_fence
# skip density force
self.enable_sample_force = True
self.force_ratio = 0.0
def set_init_param(self, init_density_weight, data: PlaceData, init_density_loss):
# init_density_weight
self.density_weight = self.density_weight * init_density_weight
self.update_precond_weight(data)
def push_metric(self, hpwl, overflow):
metrics_dict = {
"hpwl": hpwl,
"overflow": overflow,
"mu": self.mu,
"wa_coeff": self.wa_coeff,
"density_weight": self.density_weight,
"precond_coef": self.precond_coef,
"weighted_weight": self.weighted_weight,
"force_ratio": self.force_ratio,
}
self.recorder.push(**metrics_dict)
def step(self, hpwl, overflow, node_pos, data):
self.update_precond_weight(data)
self.push_metric(hpwl, overflow)
self.update_best_sol(node_pos)
if self.skip_update is not None:
if self.weighted_weight > 0.5 and self.weighted_weight < 0.99:
self.skip_update = (self.iter % 3 != 0)
elif self.iter < 50:
# slow down the param update of early stage
self.skip_update = (self.iter % 3 != 0)
else:
self.skip_update = False
self.step_density_weight()
self.step_wa_coeff()
self.step_precond_coef()
self.iter += 1
def step_density_weight(self):
if self.iter < 1:
return
if self.skip_update is not None:
if self.skip_update:
return
delta_hpwl = self.recorder.hpwl[-1] - self.recorder.hpwl[-2]
if delta_hpwl < 0:
self.mu = 1.05 * np.maximum(np.power(0.9999, float(self.iter)), 0.98)
else:
self.mu = 1.05 * np.clip(np.power(1.05, -delta_hpwl / 350000), 0.95, 1.05)
self.density_weight *= self.mu
def step_wa_coeff(self):
if self.iter < 1:
return
if self.skip_update is not None:
if self.skip_update:
return
coef = np.power(10, (self.recorder.overflow[-1] - 0.1) * 20 / 9 - 1)
self.wa_coeff = coef * self.base_gamma
def step_precond_coef(self):
if not self.use_precond:
return
if self.recorder.overflow[self.iter] < 0.3 and self.precond_coef < 1024:
if self.iter % 20 == 0:
self.precond_coef *= 2
def update_precond_weight(self, data: PlaceData):
if not self.use_precond:
return
alpha_1 = data.mov_node_to_num_pins
alpha_2 = self.precond_coef * self.density_weight * data.mov_node_area
self.precond_weight = (
alpha_1 + alpha_2
).clamp_(min=1.0)
a2_norm = alpha_2.norm(p=1)
self.weighted_weight = a2_norm / (alpha_1.norm(p=1) + a2_norm)
def update_best_sol(self, sol: torch.Tensor) -> None:
update_flag = False
hpwl, overflow = self.recorder.hpwl[-1], self.recorder.overflow[-1]
if self.iter < 50:
return update_flag
if overflow < self.stop_overflow:
self.life -= 1
if self.life == self.max_life - 1:
# release memory of rollback solution
self.best_sol_rollback = None
self.best_metric_rollback = {
"overflow": float("inf"),
"hpwl": float("inf"),
}
torch.cuda.empty_cache()
if (
overflow < self.stop_overflow * 5
and overflow >= self.stop_overflow
and self.life == self.max_life
):
if (
hpwl < self.best_metric_rollback["hpwl"] * 1.01
and overflow < self.best_metric_rollback["overflow"]
):
if self.best_sol_rollback is None:
self.best_sol_rollback = sol.detach().clone()
else:
self.best_sol_rollback.data.copy_(sol.data)
self.best_metric_rollback["hpwl"] = hpwl
self.best_metric_rollback["overflow"] = overflow
update_flag = True
if (
overflow < self.stop_overflow
and hpwl < self.best_metric_aux["hpwl"] * 1.005
and overflow < self.best_metric_aux["overflow"]
):
if self.best_sol_aux is None:
self.best_sol_aux = sol.detach().clone()
else:
self.best_sol_aux.data.copy_(sol.data)
self.best_metric_aux["hpwl"] = hpwl
self.best_metric_aux["overflow"] = overflow
update_flag = True
if overflow < self.stop_overflow and hpwl < self.best_metric["hpwl"]:
if self.best_sol is None:
self.best_sol = sol.detach().clone()
else:
self.best_sol.data.copy_(sol.data)
self.best_metric["hpwl"] = hpwl
self.best_metric["overflow"] = overflow
update_flag = True
return update_flag
def need_to_early_stop(self):
if self.iter < 100:
return False
ptr = self.iter - 1
if not self.enable_fence and self.check_divergence(
window=3, threshold=0.01 * self.recorder.overflow[ptr]
):
# dead earlier
self.life -= 6
if (
self.recorder.overflow[ptr] < self.stop_overflow * 5
and self.recorder.overflow[ptr] >= self.stop_overflow
):
if self.check_plateau(self.recorder.overflow, window=50, threshold=0.05):
# kill the program since it has converged
self.__logger__.warning(
"Large plateau detected. Kill the optimization process."
)
self.life -= self.max_life
if self.life <= 0:
return True
if (
self.recorder.overflow[ptr] > self.recorder.overflow[ptr - 1]
and self.recorder.hpwl[ptr] > self.best_metric["hpwl"] * 2
):
return True
return False
def check_plateau(self, x, window=10, threshold=0.001):
if len(x) < window:
return False
x = x[-window:]
return (np.max(x) - np.min(x)) / np.mean(x) < threshold
def check_divergence(self, window=50, threshold=0.05):
logger = self.__logger__
if self.best_metric["hpwl"] == float("inf"):
return False
if self.iter <= window:
return False
x = np.array(self.recorder.hpwl[-window:], dtype=np.float32)
wl_mean = np.mean(x).item()
wl_ratio = (wl_mean - self.best_metric["hpwl"]) / self.best_metric["hpwl"]
if wl_ratio > threshold * 1.2:
y = np.array(self.recorder.overflow[-window:], dtype=np.float32)
overflow_mean = np.mean(y).item()
overflow_diff = np.sum(np.maximum(0, np.sign(y[1:] - y[:-1]))) / len(y[1:])
overflow_range = np.max(y) - np.min(y)
overflow_ratio = (
overflow_mean - max(self.stop_overflow, self.best_metric["overflow"])
) / self.best_metric["overflow"]
if overflow_ratio > threshold:
logger.warning(
f"Divergence detected: overflow increases too much than best overflow ({overflow_ratio:.4f} > {threshold:.4f})"
)
return True
elif overflow_range / overflow_mean < threshold:
logger.warning(
f"Divergence detected: overflow plateau ({overflow_range/overflow_mean:.4f} < {threshold:.4f})"
)
return True
elif overflow_diff > 0.6:
logger.warning(
f"Divergence detected: overflow fluctuate too frequently ({overflow_diff:.2f} > 0.6)"
)
return True
else:
return False
else:
return False
def get_best_solution(self):
best_sol = None
best_hpwl = None
best_overflow = None
solution_type = 0
logger = self.__logger__
if self.best_sol_rollback is not None:
best_sol = self.best_sol_rollback.data
best_hpwl = self.best_metric_rollback["hpwl"]
best_overflow = self.best_metric_rollback["overflow"]
solution_type = 3
elif self.best_sol is None and self.best_sol_aux is None:
solution_type = 0
elif self.best_sol_aux is None:
best_sol = self.best_sol.data
best_hpwl = self.best_metric["hpwl"]
best_overflow = self.best_metric["overflow"]
solution_type = 1
elif self.best_sol is None:
best_sol = self.best_sol_aux.data
best_hpwl = self.best_metric_aux["hpwl"]
best_overflow = self.best_metric_aux["overflow"]
solution_type = 2
else:
if (
self.best_metric_aux["hpwl"] < self.best_metric["hpwl"] * 1.005
and self.best_metric_aux["overflow"] * 1.1
< self.best_metric["overflow"]
):
best_sol = self.best_sol_aux.data
best_hpwl = self.best_metric_aux["hpwl"]
best_overflow = self.best_metric_aux["overflow"]
solution_type = 2
else:
best_sol = self.best_sol.data
best_hpwl = self.best_metric["hpwl"]
best_overflow = self.best_metric["overflow"]
solution_type = 1
if solution_type == 0:
logger.info("Cannot find best solution. Use the last solution.")
elif solution_type == 1:
logger.info(
"Find best solution (type %d HPWL driven) masked_hpwl: %.4E overflow: %.4f"
% (solution_type, best_hpwl, best_overflow)
)
elif solution_type == 2:
logger.info(
"Find best solution (type %d OVFL driven) masked_hpwl: %.4E overflow: %.4f"
% (solution_type, best_hpwl, best_overflow)
)
elif solution_type == 3:
logger.info(
"Cannot find best solution. Use roll back solution (type %d) masked_hpwl: %.4E overflow: %.4f"
% (solution_type, best_hpwl, best_overflow)
)
else:
raise NotImplementedError("Unknown solution type")
return best_sol, best_hpwl, best_overflow
def visualize(self, args, logger):
file_prefix = "%s/%s_ms_" % (args.dataset, args.design_name)
res_root = os.path.join(args.result_dir, args.exp_id)
prefix = os.path.join(res_root, args.eval_dir, file_prefix)
if not os.path.exists(os.path.dirname(prefix)):
os.makedirs(os.path.dirname(prefix))
self.recorder.visualize(prefix)

37
src/run_placement.py Normal file
View File

@ -0,0 +1,37 @@
from utils import *
from src import run_placement_main_nesterov, run_placement_main_adam
def run_placement_single(args, logger):
logger.info("=================")
logger.info("Start place %s/%s" % (args.dataset , args.design_name))
set_random_seed(args)
setup_dataset_args(args)
if args.use_eplace_nesterov:
res = run_placement_main_nesterov(args, logger)
else:
res = run_placement_main_adam(args, logger)
return res
def run_placement_all(args, logger):
logger.info("Run all designs in dataset %s." % args.dataset)
df = pd.DataFrame(columns=["design", "dp_hpwl", "gp_hpwl", "top5overflow", "overflow", "gp_time", "dp_time", "gp_per_iter"])
mul_params = sorted(
get_multiple_design_params(args.dataset_root, args.dataset), key=lambda params: params["design_name"]
)
for i, params in enumerate(mul_params):
cur_args = copy.deepcopy(args)
cur_args.design_name = params["design_name"]
result = run_placement_single(cur_args, logger)
df.loc[i] = [cur_args.design_name, *result]
csv_path = os.path.join(args.result_dir, args.exp_id, args.log_dir, "run_all.csv")
df.to_csv(csv_path)
print(df)
def run_placement_main(args, logger):
if args.run_all:
run_placement_all(args, logger)
else:
run_placement_single(args, logger)

267
src/run_placement_adam.py Normal file
View File

@ -0,0 +1,267 @@
from utils import *
from src import *
import torch.optim
def run_placement_main_adam(args, logger):
data, rawdb, gpdb = load_dataset(args, logger)
device = torch.device(
"cuda:{}".format(args.gpu) if torch.cuda.is_available() else "cpu"
)
data = data.to(device)
gp_start_time = time.time()
logger.info("start gp")
data = data.preprocess()
logger.info(data)
logger.info(data.node_type_indices)
init_density_map = get_init_density_map(data, args, logger)
data.init_filler()
mov_lhs, mov_rhs = data.movable_index
mov_node_pos, mov_node_size, expand_ratio = data.get_mov_node_info()
mov_node_pos = mov_node_pos.requires_grad_(True)
node_pos_lb = mov_node_size / 2 + data.die_ll + 1e-4
node_pos_ub = data.die_ur - mov_node_size / 2 + data.die_ll - 1e-4
def trunc_node_pos_fn(x):
x.data.clamp_(min=node_pos_lb, max=node_pos_ub)
return x
conn_fix_node_pos = data.node_pos.new_empty(0, 2)
if data.fixed_connected_index[0] < data.fixed_connected_index[1]:
lhs, rhs = data.fixed_connected_index
conn_fix_node_pos = data.node_pos[lhs:rhs, ...]
conn_fix_node_pos = conn_fix_node_pos.detach()
def overflow_fn(mov_density_map):
overflow_sum = ((mov_density_map - args.target_density) * data.bin_area).clamp_(min=0.0).sum()
return overflow_sum / data.total_mov_area_without_filler
overflow_helper = (mov_lhs, mov_rhs, overflow_fn)
# optimizer = torch.optim.SGD(
# [mov_node_pos],
# lr=args.lr,
# momentum=0.9,
# nesterov=True,
# )
optimizer = torch.optim.Adam(
[mov_node_pos],
lr=args.lr,
)
ps = ParamScheduler(data, args, logger)
density_map_layer = ElectronicDensityLayer(
unit_len=data.unit_len,
num_bin_x=data.num_bin_x,
num_bin_y=data.num_bin_y,
device=device,
overflow_helper=overflow_helper,
sorted_maps=data.sorted_maps,
expand_ratio=expand_ratio,
).to(device)
init_params(mov_node_pos, trunc_node_pos_fn, mov_lhs, mov_rhs, conn_fix_node_pos,
density_map_layer, mov_node_size, init_density_map, optimizer, ps, data, args)
# fix_lhs, fix_rhs = data.fixed_index
# info = (0, 0, data.design_name + "_fix")
# fix_node_pos = data.node_pos[fix_lhs:fix_rhs, ...]
# fix_node_size = data.node_size[fix_lhs:fix_rhs, ...]
# draw_fig_with_cairo(
# None, None, fix_node_pos, fix_node_size, None, None, data, info, args
# )
# def trace_handler(prof):
# print(prof.key_averages().table(
# sort_by="self_cuda_time_total", row_limit=-1))
# prof.export_chrome_trace("test_trace_" + str(prof.step_num) + ".json")
# with torch.profiler.profile(
# activities=[
# torch.profiler.ProfilerActivity.CPU,
# torch.profiler.ProfilerActivity.CUDA,
# ], schedule=torch.profiler.schedule(
# wait=2,
# warmup=2,
# active=2),
# on_trace_ready=trace_handler
# ) as p:
# for iter in range(6):
# if mov_node_pos.grad is not None:
# mov_node_pos.grad.zero_()
# hpwl, overflow, mov_node_pos = fast_optimization(
# mov_node_pos, trunc_node_pos_fn, mov_lhs, mov_rhs, conn_fix_node_pos,
# density_map_layer, mov_node_size, init_density_map, ps, data, args
# )
# optimizer.step()
# ps.step(hpwl, overflow, mov_node_pos, data)
# p.step()
# exit(0)
for iteration in range(args.inner_iter):
optimizer.zero_grad()
hpwl, overflow, mov_node_pos = fast_optimization(
mov_node_pos, trunc_node_pos_fn, mov_lhs, mov_rhs, conn_fix_node_pos,
density_map_layer, mov_node_size, init_density_map, ps, data, args
)
optimizer.step()
# update parameters
ps.step(hpwl, overflow, mov_node_pos, data)
if ps.need_to_early_stop():
break
if iteration % args.log_freq == 0 or iteration == args.inner_iter - 1:
log_str = (
"iter: %d | masked_hpwl: %.2E overflow: %.4f "
"density_weight: %.4E wa_coeff: %.4E"
% (
iteration,
hpwl.item(),
overflow.item(),
ps.density_weight,
ps.wa_coeff,
)
)
logger.info(log_str)
if args.draw_placement:
info = (iteration, hpwl, data.design_name)
# draw_fig_new(conn_node_pos.detach(), info, args)
node_pos_to_draw = mov_node_pos[mov_lhs:mov_rhs, ...].clone()
node_size_to_draw = mov_node_size[mov_lhs:mov_rhs, ...].clone()
node_pos_to_draw = torch.cat(
[node_pos_to_draw, data.node_pos[mov_rhs:, ...].clone()], dim=0
)
node_size_to_draw = torch.cat(
[node_size_to_draw, data.node_size[mov_rhs:, ...].clone()], dim=0
)
if args.use_filler:
node_pos_to_draw = torch.cat(
[node_pos_to_draw, mov_node_pos[mov_rhs:, ...].clone()], dim=0
)
node_size_to_draw = torch.cat(
[node_size_to_draw, mov_node_size[mov_rhs:, ...].clone()], dim=0
)
draw_fig_with_cairo_cpp(
node_pos_to_draw, node_size_to_draw, data, info, args
)
# Save best solution
best_res = ps.get_best_solution()
if best_res[0] is not None:
best_sol, hpwl, overflow = best_res
mov_node_pos.data.copy_(best_sol)
node_pos = mov_node_pos[mov_lhs:mov_rhs]
node_pos = torch.cat([node_pos, data.node_pos[mov_rhs:]], dim=0)
gp_end_time = time.time()
gp_time = gp_end_time - gp_start_time
logger.info("GP Stop! #Iters %d masked_hpwl: %.4E overflow: %.4f GP Time: %.4fs perIterTime: %.6fs" %
(iteration, hpwl, overflow, gp_time, gp_time / (iteration + 1))
)
# Eval
hpwl, overflow = evaluate_placement(
node_pos, density_map_layer, init_density_map, data, args
)
hpwl, overflow = hpwl.item(), overflow.item()
info = (iteration + 1, hpwl, data.design_name)
draw_fig_with_cairo_cpp(node_pos, data.node_size, data, info, args)
logger.info("After GP, best solution eval, exact HPWL: %.4E exact Overflow: %.4f" % (hpwl, overflow))
ps.visualize(args, logger)
gp_hpwl = hpwl
iteration += 1 # increase 1 For DP drawing
# Write placement
if args.write_placement and args.load_from_raw:
res_root = os.path.join(args.result_dir, args.exp_id)
gp_prefix = os.path.join(res_root, args.output_dir, "%s_%s_gp" %(args.output_prefix, args.design_name))
if not os.path.exists(os.path.dirname(gp_prefix)):
os.makedirs(os.path.dirname(gp_prefix))
exact_node_pos = torch.round(node_pos * data.die_scale + data.die_shift).cpu()
gpdb.apply_node_pos(exact_node_pos)
gpdb.write_placement(gp_prefix)
logger.info("Write global placement in %s" % gp_prefix)
dp_start_time = time.time()
dp_end_time = None
dp_hpwl = -1
top5overflow = -1
if args.detail_placement and args.load_from_raw:
# TODO: too ugly...
post_fix = None
if data.dataset_format == "lefdef":
post_fix = ".def"
elif data.dataset_format == "bookshelf":
post_fix = ".pl"
gp_out_file = gp_prefix + post_fix
if args.dp_engine == "ntuplace3":
dp_out_file = gp_out_file.replace("_gp%s" % post_fix, "")
dp_engine = "./thirdparty/placers/ntuplace3/ntuplace3"
aux_input = data.dataset_path["aux"]
target_density_cmd = ""
if args.target_density < 1.0:
target_density_cmd = " -util %f" % (args.target_density)
cmd = "%s -aux %s -loadpl %s %s -out %s -noglobal" % (
dp_engine, aux_input, gp_out_file, target_density_cmd, dp_out_file)
logger.info(cmd)
# os.system(cmd)
output = os.popen(cmd).read()
dp_hpwl = float(output.split("========\n HPWL=")[1].split("Time")[0].strip())
dp_out_file = dp_out_file + ".ntup%s" % post_fix
elif args.dp_engine == "rippledp":
dp_out_file = gp_out_file.replace("_gp", "")
dp_engine = "./thirdparty/placers/ripple/bin/placer"
aux_input = data.dataset_path["aux"]
MLLMaxDensity = int(round(args.target_density * 1000.0))
cmd = "%s -flow dac2016 -bookshelf ispd2005 -aux %s -pl %s -MLLMaxDensity %s -cpu %s -output %s" % (
dp_engine, aux_input, gp_out_file, MLLMaxDensity, args.num_threads, dp_out_file)
os.system(cmd)
elif args.dp_engine == 'ntuplace_4dr':
dp_out_file = gp_out_file.replace(".gp.def", "")
dp_engine = "./thirdparty/placers/ntuplace4dr/ntuplace4dr_binary/placer"
cmd = dp_engine
tech_lef = data.dataset_path["tech_lef"]
cell_lef = data.dataset_path["cell_lef"]
cmd += " -tech_lef %s" % tech_lef
cmd += " -cell_lef %s" % cell_lef
benchmark_dir = os.path.dirname(tech_lef)
cmd += " -floorplan_def %s" % (gp_out_file)
cmd += " -out ntuplace_4dr_out"
cmd += " -placement_constraints %s/placement.constraints" % (benchmark_dir)
cmd += " -noglobal; "
cmd += "mv ntuplace_4dr_out.fence.plt %s.fence.plt ; " % (dp_out_file)
cmd += "mv ntuplace_4dr_out.init.plt %s.init.plt ; " % (dp_out_file)
cmd += "mv ntuplace_4dr_out %s.ntup.def ; " % (dp_out_file)
cmd += "mv ntuplace_4dr_out.ntup.overflow.plt %s.ntup.overflow.plt ; " % (dp_out_file)
cmd += "mv ntuplace_4dr_out.ntup.plt %s.ntup.plt ; " % (dp_out_file)
if os.path.exists("%s/dat" % (os.path.dirname(dp_out_file))):
cmd += "rm -r %s/dat ; " % (os.path.dirname(dp_out_file))
cmd += "mv dat %s/ ; " % (os.path.dirname(dp_out_file))
logger.info("%s" % (cmd))
output = os.popen(cmd).read()
# consider site_width
dp_hpwl = float(output.split("=======\n HPWL=")[1].split("Time")[0].strip())
top5overflow = float(output.split("[CONG] Top 5 Overflow")[1].split("\n")[0].strip())
else:
raise NotImplementedError("DP Engine %s unsupported" % args.dp_engine)
logger.info("External detailed placement takes %.2f seconds" %
(time.time() - dp_start_time))
logger.info("Write detail placement in %s" % dp_out_file)
dp_end_time = time.time()
del gpdb, rawdb
# logger.info("Evaluating detail placement result...")
# data, rawdb, gpdb = load_dataset(args, logger, dp_out_file)
# data = data.to(device).preprocess()
# hpwl, overflow = evaluate_placement(
# data.node_pos, density_map_layer, init_density_map, data, args
# )
# hpwl, overflow = hpwl.item(), overflow.item()
# info = (iteration + 1, hpwl, data.design_name)
# draw_fig_with_cairo_cpp(data.node_pos, data.node_size, data, info, args)
# logger.info("After DP, HPWL: %.4E Overflow: %.4f" % (hpwl, overflow))
gp_time = gp_end_time - gp_start_time
dp_time = dp_end_time - dp_start_time if dp_end_time is not None else 0.0
logger.info("GP Time: %.4f DP Time: %.4f" % (gp_time, dp_time))
return dp_hpwl, gp_hpwl, top5overflow, overflow, gp_time, dp_time

View File

@ -0,0 +1,308 @@
from utils import *
from src import *
from functools import partial
def run_placement_main_nesterov(args, logger):
data, rawdb, gpdb = load_dataset(args, logger)
device = torch.device(
"cuda:{}".format(args.gpu) if torch.cuda.is_available() else "cpu"
)
assert args.use_eplace_nesterov
logger.info("Use Nesterov optimizer!")
if args.scale_design:
logger.warning("Eplace's nesterov optimizer cannot support normalized die. Disable scale_design.")
args.scale_design = False
data = data.to(device)
data = data.preprocess()
logger.info(data)
logger.info(data.node_type_indices)
# args.num_bin_x = args.num_bin_y = 2 ** math.ceil(math.log2(max(data.die_info).item() // 25))
init_density_map = get_init_density_map(data, args, logger)
data.init_filler()
mov_lhs, mov_rhs = data.movable_index
mov_node_pos, mov_node_size, expand_ratio = data.get_mov_node_info()
mov_node_pos = mov_node_pos.requires_grad_(True)
node_pos_lb = mov_node_size / 2 + data.die_ll + 1e-4
node_pos_ub = data.die_ur - mov_node_size / 2 + data.die_ll - 1e-4
def trunc_node_pos_fn(x):
x.data.clamp_(min=node_pos_lb, max=node_pos_ub)
return x
conn_fix_node_pos = data.node_pos.new_empty(0, 2)
if data.fixed_connected_index[0] < data.fixed_connected_index[1]:
lhs, rhs = data.fixed_connected_index
conn_fix_node_pos = data.node_pos[lhs:rhs, ...]
conn_fix_node_pos = conn_fix_node_pos.detach()
def overflow_fn(mov_density_map):
overflow_sum = ((mov_density_map - args.target_density) * data.bin_area).clamp_(min=0.0).sum()
return overflow_sum / data.total_mov_area_without_filler
overflow_helper = (mov_lhs, mov_rhs, overflow_fn)
ps = ParamScheduler(data, args, logger)
density_map_layer = ElectronicDensityLayer(
unit_len=data.unit_len,
num_bin_x=data.num_bin_x,
num_bin_y=data.num_bin_y,
device=device,
overflow_helper=overflow_helper,
sorted_maps=data.sorted_maps,
expand_ratio=expand_ratio,
).to(device)
# fix_lhs, fix_rhs = data.fixed_index
# info = (0, 0, data.design_name + "_fix")
# fix_node_pos = data.node_pos[fix_lhs:fix_rhs, ...]
# fix_node_size = data.node_size[fix_lhs:fix_rhs, ...]
# draw_fig_with_cairo(
# None, None, fix_node_pos, fix_node_size, None, None, data, info, args
# )
obj_and_grad_fn = partial(
calc_obj_and_grad,
constraint_fn=trunc_node_pos_fn,
mov_node_size=mov_node_size,
init_density_map=init_density_map,
density_map_layer=density_map_layer,
conn_fix_node_pos=conn_fix_node_pos,
ps=ps,
data=data,
args=args,
)
evaluator_fn = partial(
fast_evaluator,
constraint_fn=trunc_node_pos_fn,
mov_node_size=mov_node_size,
init_density_map=init_density_map,
density_map_layer=density_map_layer,
conn_fix_node_pos=conn_fix_node_pos,
ps=ps,
data=data,
args=args,
)
optimizer = NesterovOptimizer(
[mov_node_pos],
lr=0,
)
# initialization
init_params(
mov_node_pos, trunc_node_pos_fn, mov_lhs, mov_rhs, conn_fix_node_pos,
density_map_layer, mov_node_size, init_density_map, optimizer, ps, data, args
)
# init learnig rate
init_lr = estimate_initial_learning_rate(obj_and_grad_fn, trunc_node_pos_fn, mov_node_pos, args.lr)
for param_group in optimizer.param_groups:
param_group["lr"] = init_lr.item()
torch.cuda.synchronize()
gp_start_time = time.time()
logger.info("start gp")
# def trace_handler(prof):
# print(prof.key_averages().table(
# sort_by="self_cuda_time_total", row_limit=-1))
# prof.export_chrome_trace("test_trace_" + str(prof.step_num) + ".json")
# with torch.profiler.profile(
# activities=[
# torch.profiler.ProfilerActivity.CPU,
# torch.profiler.ProfilerActivity.CUDA,
# ], schedule=torch.profiler.schedule(
# wait=2,
# warmup=2,
# active=2),
# on_trace_ready=trace_handler
# ) as p:
# for iter in range(6):
# # optimizer.zero_grad()
# obj = optimizer.step(obj_and_grad_fn)
# hpwl, overflow = evaluator_fn(mov_node_pos)
# # update parameters
# ps.step(hpwl, overflow, mov_node_pos, data)
# if ps.need_to_early_stop():
# break
# p.step()
# exit(0)
for iteration in range(args.inner_iter):
# optimizer.zero_grad() # zero grad inside obj_and_grad_fn
obj = optimizer.step(obj_and_grad_fn)
hpwl, overflow = evaluator_fn(mov_node_pos)
# update parameters
ps.step(hpwl, overflow, mov_node_pos, data)
if ps.need_to_early_stop():
break
if iteration % args.log_freq == 0 or iteration == args.inner_iter - 1:
log_str = (
"iter: %d | masked_hpwl: %.2E overflow: %.4f obj: %.4E "
"density_weight: %.4E wa_coeff: %.4E"
% (
iteration,
hpwl,
overflow,
obj,
ps.density_weight,
ps.wa_coeff,
)
)
logger.info(log_str)
if args.draw_placement:
info = (iteration, hpwl, data.design_name)
node_pos_to_draw = mov_node_pos[mov_lhs:mov_rhs, ...].clone()
node_size_to_draw = mov_node_size[mov_lhs:mov_rhs, ...].clone()
node_pos_to_draw = torch.cat(
[node_pos_to_draw, data.node_pos[mov_rhs:, ...].clone()], dim=0
)
node_size_to_draw = torch.cat(
[node_size_to_draw, data.node_size[mov_rhs:, ...].clone()], dim=0
)
if args.use_filler:
node_pos_to_draw = torch.cat(
[node_pos_to_draw, mov_node_pos[mov_rhs:, ...].clone()], dim=0
)
node_size_to_draw = torch.cat(
[node_size_to_draw, mov_node_size[mov_rhs:, ...].clone()], dim=0
)
draw_fig_with_cairo_cpp(
node_pos_to_draw, node_size_to_draw, data, info, args
)
# Save best solution
best_res = ps.get_best_solution()
if best_res[0] is not None:
best_sol, hpwl, overflow = best_res
mov_node_pos.data.copy_(best_sol)
node_pos = mov_node_pos[mov_lhs:mov_rhs]
node_pos = torch.cat([node_pos, data.node_pos[mov_rhs:]], dim=0)
torch.cuda.synchronize()
gp_end_time = time.time()
gp_time = gp_end_time - gp_start_time
gp_per_iter = gp_time / (iteration + 1)
logger.info("GP Stop! #Iters %d masked_hpwl: %.4E overflow: %.4f GP Time: %.4fs perIterTime: %.6fs" %
(iteration, hpwl, overflow, gp_time, gp_time / (iteration + 1))
)
# Eval
hpwl, overflow = evaluate_placement(
node_pos, density_map_layer, init_density_map, data, args
)
hpwl, overflow = hpwl.item(), overflow.item()
info = (iteration + 1, hpwl, data.design_name)
if args.draw_placement:
draw_fig_with_cairo_cpp(node_pos, data.node_size, data, info, args)
logger.info("After GP, best solution eval, exact HPWL: %.4E exact Overflow: %.4f" % (hpwl, overflow))
ps.visualize(args, logger)
gp_hpwl = hpwl
iteration += 1 # increase 1 For DP drawing
# Write placement
if args.write_placement and args.load_from_raw:
res_root = os.path.join(args.result_dir, args.exp_id)
gp_prefix = os.path.join(res_root, args.output_dir, "%s_%s_gp" %(args.output_prefix, args.design_name))
if not os.path.exists(os.path.dirname(gp_prefix)):
os.makedirs(os.path.dirname(gp_prefix))
start_write_time = time.time()
if data.dataset_format == "lefdef":
exact_node_pos = torch.round(node_pos * data.die_scale + data.die_shift).cpu()
gpdb.apply_node_pos(exact_node_pos)
gpdb.write_placement(gp_prefix)
elif data.dataset_format == "bookshelf":
logger.info("Use python to generate .pl file")
data.write_pl(node_pos, gp_prefix)
else:
raise NotImplementedError("Dataset format %s unsupported" % data.dataset_format)
logger.info("Write global placement in %s. Time: %.4f" % (gp_prefix, time.time() - start_write_time))
dp_start_time = None
dp_end_time = None
dp_hpwl = -1
top5overflow = -1
if args.detail_placement and args.load_from_raw:
# TODO: too ugly...
post_fix = None
if data.dataset_format == "lefdef":
post_fix = ".def"
elif data.dataset_format == "bookshelf":
post_fix = ".pl"
gp_out_file = gp_prefix + post_fix
if args.dp_engine == "ntuplace3":
dp_out_file = gp_out_file.replace("_gp%s" % post_fix, "")
dp_engine = "./thirdparty/placers/ntuplace3/ntuplace3"
aux_input = data.dataset_path["aux"]
target_density_cmd = ""
if args.target_density < 1.0:
target_density_cmd = " -util %f" % (args.target_density)
cmd = "%s -aux %s -loadpl %s %s -out %s -noglobal" % (
dp_engine, aux_input, gp_out_file, target_density_cmd, dp_out_file)
logger.info(cmd)
# os.system(cmd)
dp_start_time = time.time()
output = os.popen(cmd).read()
dp_end_time = time.time()
dp_hpwl = float(output.split("========\n HPWL=")[1].split("Time")[0].strip())
dp_out_file = dp_out_file + ".ntup%s" % post_fix
elif args.dp_engine == "rippledp":
dp_out_file = gp_out_file.replace("_gp", "")
dp_engine = "./thirdparty/placers/ripple/bin/placer"
aux_input = data.dataset_path["aux"]
MLLMaxDensity = int(round(args.target_density * 1000.0))
cmd = "%s -flow dac2016 -bookshelf ispd2005 -aux %s -pl %s -MLLMaxDensity %s -cpu %s -output %s" % (
dp_engine, aux_input, gp_out_file, MLLMaxDensity, args.num_threads, dp_out_file)
dp_start_time = time.time()
os.system(cmd)
dp_end_time = time.time()
elif args.dp_engine == 'ntuplace_4dr':
dp_out_file = gp_out_file.replace(".gp.def", "")
dp_engine = "./thirdparty/placers/ntuplace4dr/ntuplace4dr_binary/placer"
cmd = dp_engine
tech_lef = data.dataset_path["tech_lef"]
cell_lef = data.dataset_path["cell_lef"]
cmd += " -tech_lef %s" % tech_lef
cmd += " -cell_lef %s" % cell_lef
benchmark_dir = os.path.dirname(tech_lef)
cmd += " -floorplan_def %s" % (gp_out_file)
cmd += " -out ntuplace_4dr_out"
cmd += " -placement_constraints %s/placement.constraints" % (benchmark_dir)
cmd += " -noglobal; "
cmd += "mv ntuplace_4dr_out.fence.plt %s.fence.plt ; " % (dp_out_file)
cmd += "mv ntuplace_4dr_out.init.plt %s.init.plt ; " % (dp_out_file)
cmd += "mv ntuplace_4dr_out %s.ntup.def ; " % (dp_out_file)
cmd += "mv ntuplace_4dr_out.ntup.overflow.plt %s.ntup.overflow.plt ; " % (dp_out_file)
cmd += "mv ntuplace_4dr_out.ntup.plt %s.ntup.plt ; " % (dp_out_file)
if os.path.exists("%s/dat" % (os.path.dirname(dp_out_file))):
cmd += "rm -r %s/dat ; " % (os.path.dirname(dp_out_file))
cmd += "mv dat %s/ ; " % (os.path.dirname(dp_out_file))
logger.info("%s" % (cmd))
dp_start_time = time.time()
output = os.popen(cmd).read()
dp_end_time = time.time()
# consider site_width
dp_hpwl = float(output.split("=======\n HPWL=")[1].split("Time")[0].strip())
top5overflow = float(output.split("[CONG] Top 5 Overflow")[1].split("\n")[0].strip())
else:
raise NotImplementedError("DP Engine %s unsupported" % args.dp_engine)
logger.info("External detailed placement takes %.2f seconds" %
(dp_end_time - dp_start_time))
logger.info("After DP, HPWL: %.4E" % dp_hpwl)
logger.info("Write detail placement in %s" % dp_out_file)
del gpdb, rawdb
# logger.info("Evaluating detail placement result...")
# data, rawdb, gpdb = load_dataset(args, logger, dp_out_file)
# data = data.to(device).preprocess()
# hpwl, overflow = evaluate_placement(
# data.node_pos, density_map_layer, init_density_map, data, args
# )
# hpwl, overflow = hpwl.item(), overflow.item()
# info = (iteration + 1, hpwl, data.design_name)
# draw_fig_with_cairo_cpp(data.node_pos, data.node_size, data, info, args)
# logger.info("After DP, HPWL: %.4E Overflow: %.4f" % (hpwl, overflow))
gp_time = gp_end_time - gp_start_time
dp_time = dp_end_time - dp_start_time if dp_end_time is not None else 0.0
logger.info("GP Time: %.4f DP Time: %.4f" % (gp_time, dp_time))
return dp_hpwl, gp_hpwl, top5overflow, overflow, gp_time, dp_time, gp_per_iter

49
thirdparty/flute/CMakeLists.txt vendored Normal file
View File

@ -0,0 +1,49 @@
################################################################################
## BSD 3-Clause License
##
## Copyright (c) 2018, Iowa State University All rights reserved.
##
## Redistribution and use in source and binary forms, with or without
## modification, are permitted provided that the following conditions are met:
##
## * Redistributions of source code must retain the above copyright notice,
## this list of conditions and the following disclaimer.
##
## * Redistributions in binary form must reproduce the above copyright notice,
## this list of conditions and the following disclaimer in the documentation
## and/or other materials provided with the distribution.
##
## * Neither the name of the copyright holder nor the names of its contributors
## may be used to endorse or promote products derived from this software
## without specific prior written permission.
##
## THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
## AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
## IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
## DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
## FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
## DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
## SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
## CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
## OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
## USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
################################################################################
CMAKE_MINIMUM_REQUIRED (VERSION 3.1)
PROJECT(flute)
SET(CMAKE_CXX_STANDARD 11)
set(FLUTE_INCLUDE_DIR "${CMAKE_CURRENT_LIST_DIR}" CACHE INTERNAL "Directory where flute.h is located")
FILE(GLOB_RECURSE SRC_FILES_FLUTE ${CMAKE_CURRENT_SOURCE_DIR}/*.cpp)
ADD_LIBRARY(flute
SHARED
${SRC_FILES_FLUTE}
)
TARGET_INCLUDE_DIRECTORIES(flute PUBLIC ${FLUTE_INCLUDE_DIR})
TARGET_COMPILE_OPTIONS(flute PRIVATE -fPIC)
INSTALL(TARGETS flute DESTINATION ${XPLACE_LIB_DIR})

Some files were not shown because too many files have changed in this diff Show More