update to Xplace 2.0
This commit is contained in:
parent
fa107b8924
commit
81257ecf08
1
.gitignore
vendored
1
.gitignore
vendored
@ -12,5 +12,6 @@ build/
|
||||
*.json
|
||||
result*
|
||||
data/cad
|
||||
data/raw
|
||||
misc
|
||||
.venv/
|
||||
38
BENCHMARK.md
Normal file
38
BENCHMARK.md
Normal file
@ -0,0 +1,38 @@
|
||||
# Experimental Results (Last Updated on April 2023)
|
||||
## Detailed Routing Performance of Xplace-Route on ISPD 2015
|
||||
Xplace-Route: Routability GP + DP Flow:
|
||||
```bash
|
||||
python main.py --dataset ispd2015_fix --run_all True --load_from_raw True --detail_placement True --use_cell_inflate True
|
||||
```
|
||||
and use Innovus® to detailedly route the placement solution.
|
||||
<div align="center">
|
||||
<img src="img/exp_dr.png">
|
||||
</div>
|
||||
|
||||
## Performance of Xplace on ISPD 2005
|
||||
1. Xplace GP + DP Deterministic Flow:
|
||||
```bash
|
||||
python main.py --dataset ispd2005 --run_all True --load_from_raw True --detail_placement True
|
||||
```
|
||||
2. Xplace GP + DP Non-deterministic Flow:
|
||||
```bash
|
||||
python main.py --dataset ispd2005 --run_all True --load_from_raw True --detail_placement True --deterministic False
|
||||
```
|
||||
<div align="center">
|
||||
<img src="img/exp_ispd2005.png">
|
||||
</div>
|
||||
|
||||
|
||||
## Performance of Xplace on ISPD 2015
|
||||
1. Xplace GP + DP Deterministic Flow:
|
||||
```bash
|
||||
python main.py --dataset ispd2015_fix --run_all True --load_from_raw True --detail_placement True
|
||||
```
|
||||
2. Xplace GP + DP Non-deterministic Flow:
|
||||
```bash
|
||||
python main.py --dataset ispd2015_fix --run_all True --load_from_raw True --detail_placement True --deterministic False
|
||||
```
|
||||
|
||||
<div align="center">
|
||||
<img src="img/exp_ispd2015.png">
|
||||
</div>
|
||||
@ -9,6 +9,7 @@ message(STATUS "CMAKE_BUILD_TYPE: ${CMAKE_BUILD_TYPE}")
|
||||
|
||||
set(CMAKE_CXX_STANDARD 17)
|
||||
set(CMAKE_CXX_STANDARD_REQUIRED ON)
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
|
||||
message(STATUS PROJECT_SOURCE_DIR=${PROJECT_SOURCE_DIR})
|
||||
set(PATH_THIRDPARTY_ROOT ${PROJECT_SOURCE_DIR}/thirdparty)
|
||||
@ -34,6 +35,14 @@ message(STATUS "PYTHON_INCLUDE_DIRS: ${PYTHON_INCLUDE_DIRS}")
|
||||
add_subdirectory(${PATH_THIRDPARTY_ROOT}/flute)
|
||||
message(STATUS "FLUTE_INCLUDE_DIR: ${FLUTE_INCLUDE_DIR}")
|
||||
|
||||
# Lemon
|
||||
set(LEMON_INCLUDE_DIR "${PATH_THIRDPARTY_ROOT}/lemon/include")
|
||||
set(LEMON_INCLUDE_DIRS "${LEMON_INCLUDE_DIR}")
|
||||
find_library(LEMON_LIBRARY emon ${PATH_THIRDPARTY_ROOT}/lemon/lib)
|
||||
set(LEMON_LIBRARIES "${LEMON_LIBRARY}")
|
||||
message(STATUS "LEMON_INCLUDE_DIRS: ${LEMON_INCLUDE_DIRS}")
|
||||
message(STATUS "LEMON_LIBRARIES: ${LEMON_LIBRARIES}")
|
||||
|
||||
# Cairo
|
||||
find_package(Cairo)
|
||||
message(STATUS "CAIRO_INCLUDE_DIRS: ${CAIRO_INCLUDE_DIRS}")
|
||||
@ -48,14 +57,14 @@ list(GET TORCH_OUTPUT_LIST 0 TORCH_INSTALL_PREFIX)
|
||||
list(GET TORCH_OUTPUT_LIST 1 TORCH_ENABLE_CUDA)
|
||||
list(GET TORCH_OUTPUT_LIST 2 TORCH_VERSION)
|
||||
string(REPLACE "." ";" TORCH_VERSION_LIST ${TORCH_VERSION})
|
||||
list(GET TORCH_VERSION_LIST 0 TORCH_MAJOR_VERSION)
|
||||
list(GET TORCH_VERSION_LIST 1 TORCH_MINOR_VERSION)
|
||||
list(GET TORCH_VERSION_LIST 0 TORCH_VERSION_MAJOR)
|
||||
list(GET TORCH_VERSION_LIST 1 TORCH_VERSION_MINOR)
|
||||
message(STATUS TORCH_INSTALL_PREFIX=${TORCH_INSTALL_PREFIX})
|
||||
message(STATUS TORCH_VERSION=${TORCH_MAJOR_VERSION}.${TORCH_MINOR_VERSION})
|
||||
message(STATUS TORCH_VERSION=${TORCH_VERSION_MAJOR}.${TORCH_VERSION_MINOR})
|
||||
|
||||
# find CUDA
|
||||
if (TORCH_ENABLE_CUDA)
|
||||
find_package(CUDA 11.0)
|
||||
find_package(CUDA 11.4)
|
||||
if (NOT CUDA_FOUND)
|
||||
set(TORCH_ENABLE_CUDA 0 CACHE BOOL "Whether enable CUDA" FORCE)
|
||||
message(FATAL_ERROR "Xplace only supports CUDA mode, CMake will exit." )
|
||||
@ -66,16 +75,17 @@ message(STATUS TORCH_ENABLE_CUDA=${TORCH_ENABLE_CUDA})
|
||||
# set cuda arch and nvcc flags
|
||||
if (CUDA_FOUND)
|
||||
if (NOT CUDA_ARCH_LIST)
|
||||
set(CUDA_ARCH_LIST 7.0 7.5 8.0 8.6)
|
||||
set(CUDA_ARCH_LIST 8.0 8.6)
|
||||
endif(NOT CUDA_ARCH_LIST)
|
||||
# for cuda_add_library
|
||||
cuda_select_nvcc_arch_flags(CUDA_ARCH_FLAGS ${CUDA_ARCH_LIST})
|
||||
message(STATUS "CUDA_ARCH_FLAGS: ${CUDA_ARCH_FLAGS}")
|
||||
# set nvcc flags
|
||||
set(CUDA_USE_STATIC_CUDA_RUNTIME OFF)
|
||||
set(CMAKE_CUDA17_EXTENSION_COMPILE_OPTION "-std=c++17")
|
||||
list(APPEND CUDA_NVCC_FLAGS ${CUDA_ARCH_FLAGS} --compiler-options;-fPIC;-std=c++17)
|
||||
list(APPEND CUDA_NVCC_FLAGS ${CUDA_ARCH_FLAGS} --extended-lambda)
|
||||
list(APPEND TORCH_NVCC_FLAGS -D__CUDA_NO_HALF_OPERATORS__;-D__CUDA_NO_HALF_CONVERSIONS__;-D__CUDA_NO_BFLOAT16_CONVERSIONS__;-D__CUDA_NO_HALF2_OPERATORS__;--expt-relaxed-constexpr)
|
||||
list(APPEND TORCH_NVCC_FLAGS --expt-relaxed-constexpr)
|
||||
list(APPEND CUDA_NVCC_FLAGS ${TORCH_NVCC_FLAGS})
|
||||
message(STATUS "CUDA_NVCC_FLAGS: ${CUDA_NVCC_FLAGS}")
|
||||
endif(CUDA_FOUND)
|
||||
@ -125,8 +135,8 @@ function(add_pytorch_extension target_name)
|
||||
target_link_libraries(${target_name}_cuda_tmp ${ARG_EXTRA_LINK_LIBRARIES} ${TORCH_LIBRARY})
|
||||
target_compile_definitions(${target_name}_cuda_tmp PRIVATE
|
||||
TORCH_EXTENSION_NAME=${target_name}
|
||||
TORCH_MAJOR_VERSION=${TORCH_MAJOR_VERSION}
|
||||
TORCH_MINOR_VERSION=${TORCH_MINOR_VERSION}
|
||||
TORCH_VERSION_MAJOR=${TORCH_VERSION_MAJOR}
|
||||
TORCH_VERSION_MINOR=${TORCH_VERSION_MINOR}
|
||||
ENABLE_CUDA=${TORCH_ENABLE_CUDA}
|
||||
${ARG_EXTRA_DEFINITIONS})
|
||||
set_target_properties(${target_name}_cuda_tmp PROPERTIES
|
||||
@ -146,8 +156,8 @@ function(add_pytorch_extension target_name)
|
||||
endif()
|
||||
target_compile_definitions(${target_name} PRIVATE
|
||||
TORCH_EXTENSION_NAME=${target_name}
|
||||
TORCH_MAJOR_VERSION=${TORCH_MAJOR_VERSION}
|
||||
TORCH_MINOR_VERSION=${TORCH_MINOR_VERSION}
|
||||
TORCH_VERSION_MAJOR=${TORCH_VERSION_MAJOR}
|
||||
TORCH_VERSION_MINOR=${TORCH_VERSION_MINOR}
|
||||
ENABLE_CUDA=${TORCH_ENABLE_CUDA}
|
||||
${ARG_EXTRA_DEFINITIONS})
|
||||
endfunction()
|
||||
|
||||
2
LICENSE
2
LICENSE
@ -24,4 +24,4 @@ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
100
README.md
100
README.md
@ -1,12 +1,24 @@
|
||||
# Xplace
|
||||
# Xplace: An Extremely Fast and Extensible Global Placement Framework
|
||||
|
||||
Xplace is a fast and extensible GPU accelerated global placement framework developed by the research team supervised by Prof. Evangeline F. Y. Young at The Chinese University of Hong Kong (CUHK). It achieves around 3x speedup per GP iteration compared to the state-of-the-art global placer DREAMPlace and shows high extensiblity.
|
||||
## News 🚀
|
||||
We are happy to announce that Xplace 2.0 is now released. Comparing to [Xplace 1.0](https://dl.acm.org/doi/abs/10.1145/3489517.3530485), this version supports the following new features:
|
||||
|
||||
- Support deterministic mode with only 5~25% extra GP runtime overhead.
|
||||
- Implement an extremly fast GPU-accelerated detailed-routability-driven placement algorithm.
|
||||
- Integrate with a GPU-accelerated detailed placer and a GPU-accelerated global router.
|
||||
- Provide benchmark download and preprocess scripts, and three routability evalution scripts.
|
||||
- Code refactoring.
|
||||
|
||||
😄 Detailed Experimental results of Xplace 2.0 are given in [BENCHMARK.md](BENCHMARK.md).
|
||||
|
||||
## About
|
||||
Xplace is a fast and extensible GPU accelerated global placement framework developed by the research team supervised by Prof. Evangeline F. Y. Young at The Chinese University of Hong Kong (CUHK). It achieves around 3x speedup per GP iteration compared to DREAMPlace and shows high extensiblity.
|
||||
|
||||
|
||||
As shown in the following figure, Xplace framework is built on top of PyTorch and consists of serveral independent modules. One can easily extend Xplace by applying new scheduling techniques, new gradient functions, new placement metrics and so on.
|
||||
|
||||
<div align="center">
|
||||
<img src="assets/xplace_overview.png" width="300"/>
|
||||
<img src="img/xplace_overview.png" width="300"/>
|
||||
</div>
|
||||
|
||||
More details are in the following paper:
|
||||
@ -16,41 +28,61 @@ Lixin Liu, Bangqi Fu, Martin D. F. Wong, and Evangeline F. Y. Young. "[Xplace: a
|
||||
(For the Xplace-NN, please refer to branch [neural](https://github.com/cuhk-eda/Xplace/tree/neural))
|
||||
|
||||
## Requirements
|
||||
- CMake >= 3.12
|
||||
- GCC >= 7.5.0
|
||||
- Boost >= 1.56.0
|
||||
- CUDA >= 11.0
|
||||
- Python >= 3.8
|
||||
- PyTorch >= 1.10.1
|
||||
- Cairo
|
||||
|
||||
- [CMake](https://cmake.org/) >= 3.12
|
||||
- [GCC](https://gcc.gnu.org/) >= 7.5.0
|
||||
- [Boost](https://www.boost.org/) >= 1.56.0
|
||||
- [CUDA](https://developer.nvidia.com/cuda-toolkit) >= 11.3
|
||||
- [Python](https://www.python.org/) >= 3.8
|
||||
- [PyTorch](https://pytorch.org/) >= 1.12.0
|
||||
- [Cairo](https://www.cairographics.org/)
|
||||
- [Innovus®](https://www.cadence.com/content/cadence-www/global/en_US/home/tools/digital-design-and-signoff/soc-implementation-and-floorplanning/innovus-implementation-system.html) (version 20.14, optional, for detailed routing and design rule checking)
|
||||
|
||||
## Setup
|
||||
1. Clone the Xplace repository. We'll call the directory that you cloned Xplace as `$XPLACE_HOME`.
|
||||
```console
|
||||
```bash
|
||||
git clone --recursive https://github.com/cuhk-eda/Xplace
|
||||
```
|
||||
2. Build the shared libraries used in Xplace.
|
||||
```console
|
||||
```bash
|
||||
cd $XPLACE_HOME
|
||||
mkdir build && cd build
|
||||
cmake -DPYTHON_EXECUTABLE=$(which python) ..
|
||||
make -j40 && make install
|
||||
```
|
||||
|
||||
## Prepare Data
|
||||
The following script will automatically download `ispd2005`, `ispd2015` and `iccad2019` in `./data/raw`. It also preprocesses `ispd2015` benchmark to fix some errors reported by Innovus.
|
||||
```bash
|
||||
cd $XPLACE_HOME/data
|
||||
./download_data.sh
|
||||
```
|
||||
|
||||
## Get started
|
||||
- To run GP + DP flow for ISPD2005 dataset:
|
||||
```bash
|
||||
# only run adaptec1
|
||||
python main.py --dataset ispd2005 --design_name adaptec1 --load_from_raw True --detail_placement True
|
||||
|
||||
- To run GP only flow for all the designs in ISPD2005 dataset:
|
||||
```console
|
||||
python main.py --dataset_root your_path --dataset ispd2005 --run_all True --load_from_raw True --write_placement True
|
||||
# run all the designs in ispd2005
|
||||
python main.py --dataset ispd2005 --run_all True --load_from_raw True --detail_placement True
|
||||
```
|
||||
|
||||
- To run GP + DP flow for `adaptec1` in ISPD2005 dataset:
|
||||
```console
|
||||
python main.py --dataset_root your_path --dataset ispd2005 --design_name adaptec1 --load_from_raw True --write_placement True --detail_placement True
|
||||
- To run GP + DP flow for ISPD2015 dataset:
|
||||
```bash
|
||||
# only run mgc_fft_1
|
||||
python main.py --dataset ispd2015_fix --design_name mgc_fft_1 --load_from_raw True --detail_placement True
|
||||
|
||||
# run all the designs in ispd2015
|
||||
python main.py --dataset ispd2015_fix --run_all True --load_from_raw True --detail_placement True
|
||||
```
|
||||
|
||||
**Note**: For ISPD2005 dataset, [NTUplace3](http://eda.ee.ntu.edu.tw/research.htm) is used as the detailed placement engine. For ISPD2015 dataset, please run GP only flow and launch [ABCDPlace](https://github.com/limbo018/DREAMPlace) to perform detailed placement.
|
||||
- To run Routability GP + DP flow for ISPD2015 dataset:
|
||||
```bash
|
||||
# run all the designs in ispd2015 with routability optimization
|
||||
python main.py --dataset ispd2015_fix --run_all True --load_from_raw True --detail_placement True --use_cell_inflate True
|
||||
```
|
||||
|
||||
**NOTE**: we defaultly enable the deterministic mode. If you don't need determinism and want to run placement in an extremely fast mode, please try to set `--deterministic False` in the python arguments.
|
||||
|
||||
- Each run will generate serveral output files in `./result/exp_id`. These files can provide valuable information for parameter tuning.
|
||||
```
|
||||
@ -67,26 +99,34 @@ Please refer to `main.py`.
|
||||
## Load design from preprocessed `pt` file (Optional)
|
||||
The following script will dump the parsed design into a single torch `pt` file so Xplace can load the design from the `pt` file instead of parsing the input file from scratch.
|
||||
|
||||
```console
|
||||
cd $XPLACE_HOME
|
||||
python utils/convert_design_to_torch_data.py --dataset_root your_path --dataset ispd2005
|
||||
```bash
|
||||
cd $XPLACE_HOME/data
|
||||
python utils/convert_design_to_torch_data.py --dataset ispd2005
|
||||
python utils/convert_design_to_torch_data.py --dataset ispd2015_fix
|
||||
python utils/convert_design_to_torch_data.py --dataset iccad2019
|
||||
```
|
||||
Preprocessed data is saved in `./data/cad`.
|
||||
|
||||
When developing a new global placement technique in Xplace, we highly suggest using the `pt` mode to save the parser time. (set `--load_from_raw False`)
|
||||
|
||||
```console
|
||||
```bash
|
||||
python main.py --dataset ispd2005 --run_all True --load_from_raw False
|
||||
```
|
||||
|
||||
**Note**: Please remember to use the raw mode (set `--load_from_raw True`) when running detailed placement or measuring the total running time.
|
||||
**Note**:
|
||||
1. Please remember to use the raw mode (set `--load_from_raw True`) when measuring the total running time.
|
||||
2. We currently not support `pt` mode in routability-driven mode.
|
||||
|
||||
## Xplace Placement Results
|
||||
## Evaluate the Routability of Xplace's Solution
|
||||
We provide three ways to evaluate the routability:
|
||||
|
||||
1. Set `--final_route_eval True` in python arguments to invoke the internal global router [GGR](https://dl.acm.org/doi/10.1145/3508352.3549474) to evaluate the placement solution. The evaluation metrics are reported in the log and recorded in `./result/exp_id/log/route.csv`. Besies, the route guide file is written in `./result/exp_id/output/design_name.guide` and
|
||||
More details about using GGR in Xplace can be found in [cpp_to_py/gpugr](cpp_to_py/gpugr).
|
||||
|
||||
2. Use [CU-GR](https://github.com/cuhk-eda/cu-gr) to global route the placement solution. refer to [tool/cugr_ispd2015_fix](tool/cugr_ispd2015_fix) for more instructions.
|
||||
|
||||
3. (Optional). If Innovus® has been properly installed in your OS, you may try to use Innovus® to detailedly route the placement solution. Please refer to [tool/innovus_ispd2015_fix](tool/innovus_ispd2015_fix) for more instructions.
|
||||
|
||||
Benchmark | Placement Solutions
|
||||
|:---:|:---:|
|
||||
ISPD2005 | [Google Drive](https://drive.google.com/drive/folders/1fUzkT9ymV3n0XxfWXA0mR3WQX55hR1PB?usp=sharing)
|
||||
ISPD2015 (w/o fence) | [Google Drive](https://drive.google.com/drive/folders/1UsKQ1FQ4fFi4pdJ0VoCoCCjLakhoS20Q?usp=sharing)
|
||||
|
||||
## Citation
|
||||
If you find **Xplace** useful in your research, please consider to cite:
|
||||
|
||||
Binary file not shown.
|
Before Width: | Height: | Size: 319 KiB |
@ -3,7 +3,9 @@ add_subdirectory(dct_cuda)
|
||||
add_subdirectory(density_map_cuda)
|
||||
add_subdirectory(draw_placement)
|
||||
add_subdirectory(flute_cpp)
|
||||
add_subdirectory(gpudp)
|
||||
add_subdirectory(gpugr)
|
||||
add_subdirectory(hpwl_cuda)
|
||||
add_subdirectory(io_parser)
|
||||
add_subdirectory(routedp)
|
||||
add_subdirectory(wa_wirelength_hpwl_cuda)
|
||||
add_subdirectory(node_pos_to_pin_pos_cuda)
|
||||
|
||||
@ -1,3 +1,3 @@
|
||||
# Add new module
|
||||
|
||||
To add new module, please modify `__init__.py` and `CMakeLists.txt`.
|
||||
To add new module, please modify `cpp_to_py/__init__.py` and `cpp_to_py/CMakeLists.txt`.
|
||||
@ -6,8 +6,10 @@ __all__ = [
|
||||
"io_parser",
|
||||
"density_map_cuda",
|
||||
"draw_placement",
|
||||
"node_pos_to_pin_pos_cuda",
|
||||
"wa_wirelength_hpwl_cuda",
|
||||
"gpugr",
|
||||
"gpudp",
|
||||
"routedp",
|
||||
]
|
||||
from .cpybin import (
|
||||
dct_cuda,
|
||||
@ -16,7 +18,9 @@ from .cpybin import (
|
||||
io_parser,
|
||||
density_map_cuda,
|
||||
draw_placement,
|
||||
node_pos_to_pin_pos_cuda,
|
||||
wa_wirelength_hpwl_cuda,
|
||||
gpugr,
|
||||
gpudp,
|
||||
routedp,
|
||||
)
|
||||
|
||||
|
||||
@ -45,5 +45,4 @@
|
||||
using namespace std;
|
||||
|
||||
using utils::assert_msg;
|
||||
using utils::print;
|
||||
using utils::printlog;
|
||||
using utils::logger;
|
||||
@ -24,7 +24,7 @@ void Cell::ctype(CellType* t) {
|
||||
return;
|
||||
}
|
||||
if (_type) {
|
||||
printlog(LOG_ERROR, "type of cell %s already set", _name.c_str());
|
||||
logger.error("type of cell %s already set", _name.c_str());
|
||||
return;
|
||||
}
|
||||
_type = t;
|
||||
@ -58,15 +58,45 @@ bool Cell::placed() const { return (lx() != INT_MIN) && (ly() != INT_MIN); }
|
||||
|
||||
void Cell::place(int x, int y) {
|
||||
if (_fixed) {
|
||||
printlog(LOG_WARN, "moving fixed cell %s to (%d,%d)", _name.c_str(), x, y);
|
||||
logger.warning("moving fixed cell %s to (%d,%d)", _name.c_str(), x, y);
|
||||
}
|
||||
_lx = x;
|
||||
_ly = y;
|
||||
}
|
||||
|
||||
void Cell::place(int x, int y, int orient) {
|
||||
if (_fixed) {
|
||||
logger.warning("moving fixed cell %s to (%d,%d)", _name.c_str(), x, y);
|
||||
}
|
||||
_lx = x;
|
||||
_ly = y;
|
||||
switch (orient) {
|
||||
case 0:
|
||||
_flipX = false;
|
||||
_flipY = false;
|
||||
break;
|
||||
case 2:
|
||||
_flipX = true;
|
||||
_flipY = true;
|
||||
break;
|
||||
case 4:
|
||||
_flipX = true;
|
||||
_flipY = false;
|
||||
break;
|
||||
case 6:
|
||||
_flipX = false;
|
||||
_flipY = true;
|
||||
break;
|
||||
default:
|
||||
_flipX = false;
|
||||
_flipY = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
void Cell::place(int x, int y, bool flipX, bool flipY) {
|
||||
if (_fixed) {
|
||||
printlog(LOG_WARN, "moving fixed cell %s to (%d,%d)", _name.c_str(), x, y);
|
||||
logger.warning("moving fixed cell %s to (%d,%d)", _name.c_str(), x, y);
|
||||
}
|
||||
_lx = x;
|
||||
_ly = y;
|
||||
@ -76,7 +106,7 @@ void Cell::place(int x, int y, bool flipX, bool flipY) {
|
||||
|
||||
void Cell::unplace() {
|
||||
if (_fixed) {
|
||||
printlog(LOG_WARN, "unplace fixed cell %s", _name.c_str());
|
||||
logger.warning("unplace fixed cell %s", _name.c_str());
|
||||
}
|
||||
_lx = _ly = INT_MIN;
|
||||
_flipX = _flipY = false;
|
||||
|
||||
@ -118,6 +118,7 @@ public:
|
||||
void fixed(bool fix) { _fixed = fix; }
|
||||
bool placed() const;
|
||||
void place(int x, int y);
|
||||
void place(int x, int y, int orient);
|
||||
void place(int x, int y, bool flipX, bool flipY);
|
||||
void unplace();
|
||||
unsigned numPins() const { return _pins.size(); }
|
||||
|
||||
@ -14,7 +14,7 @@ Database::~Database() {
|
||||
clear();
|
||||
// for regions.push_back(new Region("default"));
|
||||
CLEAR_POINTER_LIST(regions);
|
||||
printlog(LOG_INFO, "destruct rawdb");
|
||||
logger.info("destruct rawdb");
|
||||
}
|
||||
|
||||
void Database::load() {
|
||||
@ -52,7 +52,7 @@ void Database::load() {
|
||||
// if (setting.Verilog != "") {
|
||||
// readVerilog(setting.Verilog);
|
||||
// }
|
||||
printlog(LOG_INFO, "Finish loading rawdb");
|
||||
logger.info("Finish loading rawdb");
|
||||
}
|
||||
|
||||
void Database::reset() {
|
||||
@ -244,20 +244,24 @@ void Database::SetupFloorplan() {
|
||||
for (Site& site : sites) {
|
||||
if (site.siteClassName() == "CORE") {
|
||||
if (siteW != (unsigned)site.width()) {
|
||||
printlog(LOG_WARN,
|
||||
"siteW %d in DEF is inconsistent with siteW %d in LEF.",
|
||||
static_cast<int>(siteW),
|
||||
static_cast<int>(site.width()));
|
||||
logger.warning("siteW %d in DEF is inconsistent with siteW %d in LEF.",
|
||||
static_cast<int>(siteW),
|
||||
static_cast<int>(site.width()));
|
||||
}
|
||||
if (siteH != site.height()) {
|
||||
printlog(LOG_WARN,
|
||||
"siteH %d in DEF is inconsistent with siteH %d in LEF.",
|
||||
static_cast<int>(siteH),
|
||||
static_cast<int>(site.height()));
|
||||
logger.warning("siteH %d in DEF is inconsistent with siteH %d in LEF.",
|
||||
static_cast<int>(siteH),
|
||||
static_cast<int>(site.height()));
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
nSitesX = (coreHX - coreLX) / siteW;
|
||||
nSitesY = (coreHY - coreLY) / siteH;
|
||||
if (!maxDisp) {
|
||||
maxDisp = nSitesX;
|
||||
}
|
||||
}
|
||||
|
||||
void Database::SetupRegions() {
|
||||
@ -294,8 +298,7 @@ void Database::SetupRegions() {
|
||||
// member group name is a particular cell name
|
||||
Cell* cell = getCell(member);
|
||||
if (!cell) {
|
||||
printlog(
|
||||
LOG_ERROR, "cell name (%s) not found for group (%s)", member.c_str(), region->name().c_str());
|
||||
logger.error("cell name (%s) not found for group (%s)", member.c_str(), region->name().c_str());
|
||||
}
|
||||
cell->region = region;
|
||||
}
|
||||
@ -311,8 +314,6 @@ void Database::SetupRegions() {
|
||||
|
||||
void Database::SetupSiteMap() {
|
||||
// set up site map
|
||||
nSitesX = (coreHX - coreLX) / siteW;
|
||||
nSitesY = (coreHY - coreLY) / siteH;
|
||||
siteMap.siteL = coreLX;
|
||||
siteMap.siteR = coreHX;
|
||||
siteMap.siteB = coreLY;
|
||||
@ -323,17 +324,13 @@ void Database::SetupSiteMap() {
|
||||
siteMap.siteNY = nSitesY;
|
||||
siteMap.initSiteMap(nSitesX, nSitesY);
|
||||
|
||||
if (!maxDisp) {
|
||||
maxDisp = nSitesX;
|
||||
}
|
||||
|
||||
// mark site partially overlapped by fence
|
||||
int nRegions = regions.size();
|
||||
for (int i = 1; i < nRegions; i++) {
|
||||
// skipped the default region
|
||||
Region* region = regions[i];
|
||||
|
||||
printlog(LOG_VERBOSE, "region : %s", region->name().c_str());
|
||||
logger.verbose("region : %s", region->name().c_str());
|
||||
// partially overlap at left/right
|
||||
vector<Rectangle> hSlices = region->rects;
|
||||
Rectangle::sliceH(hSlices);
|
||||
@ -478,17 +475,14 @@ void Database::SetupSiteMap() {
|
||||
}
|
||||
}
|
||||
|
||||
printlog(LOG_VERBOSE, "core area: %ld", siteMap.nSites);
|
||||
printlog(LOG_VERBOSE,
|
||||
"placeable: %ld (%lf%%)",
|
||||
siteMap.nPlaceable,
|
||||
(double)siteMap.nPlaceable / (double)siteMap.nSites * 100.0);
|
||||
logger.verbose("core area: %ld", siteMap.nSites);
|
||||
logger.verbose(
|
||||
"placeable: %ld (%lf%%)", siteMap.nPlaceable, (double)siteMap.nPlaceable / (double)siteMap.nSites * 100.0);
|
||||
for (int i = 0; i < (int)regions.size(); i++) {
|
||||
printlog(LOG_VERBOSE,
|
||||
"region %d : %ld (%lf%%)",
|
||||
i,
|
||||
siteMap.nRegionSites[i],
|
||||
(double)siteMap.nRegionSites[i] / (double)siteMap.nPlaceable);
|
||||
logger.verbose("region %d : %ld (%lf%%)",
|
||||
i,
|
||||
siteMap.nRegionSites[i],
|
||||
(double)siteMap.nRegionSites[i] / (double)siteMap.nPlaceable);
|
||||
}
|
||||
}
|
||||
|
||||
@ -502,17 +496,17 @@ void Database::SetupRows() {
|
||||
if (flip[y] == 0) {
|
||||
flip[y] = isFlip;
|
||||
} else if (flip[y] != isFlip) {
|
||||
printlog(LOG_ERROR, "row flip conflict %d : %d", y, isFlip);
|
||||
logger.error("row flip conflict %d : %d", y, isFlip);
|
||||
flipCheckPass = false;
|
||||
}
|
||||
}
|
||||
|
||||
if (!flipCheckPass) {
|
||||
printlog(LOG_ERROR, "row flip checking fail");
|
||||
logger.error("row flip checking fail");
|
||||
}
|
||||
|
||||
if (rows.size() != nSitesY) {
|
||||
printlog(LOG_ERROR, "resize rows %d->%d", (int)rows.size(), nSitesY);
|
||||
logger.error("resize rows %d->%d", (int)rows.size(), nSitesY);
|
||||
for (Row*& row : rows) {
|
||||
delete row;
|
||||
row = nullptr;
|
||||
@ -544,27 +538,24 @@ void Database::SetupRows() {
|
||||
if (!powerNet.getRowPower(ly, hy, row->_topPower, row->_botPower)) {
|
||||
if (topNormal && row->topPower() == 'x') {
|
||||
if (y + 1 == nSitesY) {
|
||||
printlog(LOG_WARN, "Top power rail of the row at y=%d is not connected to power rail", row->y());
|
||||
logger.warning("Top power rail of the row at y=%d is not connected to power rail", row->y());
|
||||
} else {
|
||||
printlog(LOG_ERROR, "Top power rail of the row at y=%d is not connected to power rail", row->y());
|
||||
logger.error("Top power rail of the row at y=%d is not connected to power rail", row->y());
|
||||
topNormal = false;
|
||||
}
|
||||
}
|
||||
if (botNormal && row->botPower() == 'x') {
|
||||
if (y) {
|
||||
printlog(
|
||||
LOG_ERROR, "Bottom power rail of the row at y=%d is not connected to power rail", row->y());
|
||||
logger.error("Bottom power rail of the row at y=%d is not connected to power rail", row->y());
|
||||
botNormal = false;
|
||||
} else {
|
||||
printlog(LOG_WARN, "Bottom power rail of the row at y=%d is not connected to power rail", row->y());
|
||||
logger.warning("Bottom power rail of the row at y=%d is not connected to power rail", row->y());
|
||||
}
|
||||
}
|
||||
}
|
||||
if (shrNormal && row->topPower() == row->botPower()) {
|
||||
printlog(LOG_ERROR,
|
||||
"Top and Bottom power rail of the row at y=%d share the same power %c",
|
||||
row->y(),
|
||||
row->topPower());
|
||||
logger.error(
|
||||
"Top and Bottom power rail of the row at y=%d share the same power %c", row->y(), row->topPower());
|
||||
shrNormal = false;
|
||||
}
|
||||
}
|
||||
@ -628,7 +619,7 @@ void Database::setup() {
|
||||
SetupRows();
|
||||
SetupRowSegments();
|
||||
}
|
||||
printlog(LOG_INFO, "Finish setting up rawdb");
|
||||
logger.info("Finish setting up rawdb");
|
||||
}
|
||||
|
||||
Layer& Database::addLayer(const string& name, const char type) {
|
||||
@ -654,7 +645,7 @@ Layer& Database::addLayer(const string& name, const char type) {
|
||||
Site& Database::addSite(const string& name, const string& siteClassName, const int w, const int h) {
|
||||
for (unsigned i = 0; i < sites.size(); i++) {
|
||||
if (name == sites[i].name()) {
|
||||
printlog(LOG_WARN, "site re-defined: %s", name.c_str());
|
||||
logger.warning("site re-defined: %s", name.c_str());
|
||||
return sites[i];
|
||||
}
|
||||
}
|
||||
@ -665,7 +656,7 @@ Site& Database::addSite(const string& name, const string& siteClassName, const i
|
||||
ViaType* Database::addViaType(const string& name, bool isDef) {
|
||||
ViaType* viatype = getViaType(name);
|
||||
if (viatype) {
|
||||
printlog(LOG_WARN, "via type re-defined: %s", name.c_str());
|
||||
logger.warning("via type re-defined: %s", name.c_str());
|
||||
return viatype;
|
||||
}
|
||||
viatype = new ViaType(name, isDef);
|
||||
@ -677,7 +668,7 @@ ViaType* Database::addViaType(const string& name, bool isDef) {
|
||||
CellType* Database::addCellType(const string& name, unsigned libcell) {
|
||||
CellType* celltype = getCellType(name);
|
||||
if (celltype) {
|
||||
printlog(LOG_WARN, "cell type re-defined: %s", name.c_str());
|
||||
logger.warning("cell type re-defined: %s", name.c_str());
|
||||
return celltype;
|
||||
}
|
||||
celltype = new CellType(name, libcell);
|
||||
@ -689,7 +680,7 @@ CellType* Database::addCellType(const string& name, unsigned libcell) {
|
||||
Cell* Database::addCell(const string& name, CellType* type) {
|
||||
Cell* cell = getCell(name);
|
||||
if (cell) {
|
||||
printlog(LOG_WARN, "cell re-defined: %s", name.c_str());
|
||||
logger.warning("cell re-defined: %s", name.c_str());
|
||||
if (!cell->ctype()) {
|
||||
cell->ctype(type);
|
||||
}
|
||||
@ -704,7 +695,7 @@ Cell* Database::addCell(const string& name, CellType* type) {
|
||||
IOPin* Database::addIOPin(const string& name, const string& netName, const char direction) {
|
||||
IOPin* iopin = getIOPin(name);
|
||||
if (iopin) {
|
||||
printlog(LOG_WARN, "IO pin re-defined: %s", name.c_str());
|
||||
logger.warning("IO pin re-defined: %s", name.c_str());
|
||||
return iopin;
|
||||
}
|
||||
iopin = new IOPin(name, netName, direction);
|
||||
@ -716,7 +707,7 @@ IOPin* Database::addIOPin(const string& name, const string& netName, const char
|
||||
Net* Database::addNet(const string& name, const NDR* ndr) {
|
||||
Net* net = getNet(name);
|
||||
if (net) {
|
||||
printlog(LOG_WARN, "Net re-defined: %s", name.c_str());
|
||||
logger.warning("Net re-defined: %s", name.c_str());
|
||||
return net;
|
||||
}
|
||||
net = new Net(name, ndr);
|
||||
@ -748,7 +739,7 @@ Track* Database::addTrack(char direction, double start, double num, double step)
|
||||
Region* Database::addRegion(const string& name, const char type) {
|
||||
Region* region = getRegion(name);
|
||||
if (region) {
|
||||
printlog(LOG_WARN, "Region re-defined: %s", name.c_str());
|
||||
logger.warning("Region re-defined: %s", name.c_str());
|
||||
return region;
|
||||
}
|
||||
region = new Region(name, type);
|
||||
@ -759,7 +750,7 @@ Region* Database::addRegion(const string& name, const char type) {
|
||||
NDR* Database::addNDR(const string& name, const bool hardSpacing) {
|
||||
NDR* ndr = getNDR(name);
|
||||
if (ndr) {
|
||||
printlog(LOG_WARN, "NDR re-defined: %s", name.c_str());
|
||||
logger.warning("NDR re-defined: %s", name.c_str());
|
||||
return ndr;
|
||||
}
|
||||
ndr = new NDR(name, hardSpacing);
|
||||
@ -856,7 +847,7 @@ const Layer* Database::getCLayer(const unsigned index) const {
|
||||
|
||||
/* get cell type by name */
|
||||
CellType* Database::getCellType(const string& name) {
|
||||
unordered_map<string, CellType*>::iterator mi = name_celltypes.find(name);
|
||||
robin_hood::unordered_map<string, CellType*>::iterator mi = name_celltypes.find(name);
|
||||
if (mi == name_celltypes.end()) {
|
||||
return nullptr;
|
||||
}
|
||||
@ -864,7 +855,7 @@ CellType* Database::getCellType(const string& name) {
|
||||
}
|
||||
|
||||
Cell* Database::getCell(const string& name) {
|
||||
unordered_map<string, Cell*>::iterator mi = name_cells.find(name);
|
||||
robin_hood::unordered_map<string, Cell*>::iterator mi = name_cells.find(name);
|
||||
if (mi == name_cells.end()) {
|
||||
return nullptr;
|
||||
}
|
||||
@ -872,7 +863,7 @@ Cell* Database::getCell(const string& name) {
|
||||
}
|
||||
|
||||
Net* Database::getNet(const string& name) {
|
||||
unordered_map<string, Net*>::iterator mi = name_nets.find(name);
|
||||
robin_hood::unordered_map<string, Net*>::iterator mi = name_nets.find(name);
|
||||
if (mi == name_nets.end()) {
|
||||
return nullptr;
|
||||
}
|
||||
@ -904,7 +895,7 @@ NDR* Database::getNDR(const string& name) const {
|
||||
}
|
||||
|
||||
IOPin* Database::getIOPin(const string& name) const {
|
||||
unordered_map<string, IOPin*>::const_iterator mi = name_iopins.find(name);
|
||||
robin_hood::unordered_map<string, IOPin*>::const_iterator mi = name_iopins.find(name);
|
||||
if (mi == name_iopins.end()) {
|
||||
return nullptr;
|
||||
}
|
||||
@ -912,7 +903,7 @@ IOPin* Database::getIOPin(const string& name) const {
|
||||
}
|
||||
|
||||
ViaType* Database::getViaType(const string& name) const {
|
||||
unordered_map<string, ViaType*>::const_iterator mi = name_viatypes.find(name);
|
||||
robin_hood::unordered_map<string, ViaType*>::const_iterator mi = name_viatypes.find(name);
|
||||
if (mi == name_viatypes.end()) {
|
||||
return nullptr;
|
||||
}
|
||||
@ -1031,19 +1022,19 @@ void Database::errorCheck(bool autoFix) {
|
||||
for (int i = 0; i < (int)dbIssues.size(); i++) {
|
||||
switch (dbIssues[i]) {
|
||||
case E_ROW_EXCEED_DIE:
|
||||
printlog(LOG_WARN, "row is placed out of die area");
|
||||
logger.warning("row is placed out of die area");
|
||||
break;
|
||||
case W_NON_UNIFORM_SITE_WIDTH:
|
||||
printlog(LOG_WARN, "non uniform site width detected");
|
||||
logger.warning("non uniform site width detected");
|
||||
break;
|
||||
case W_NON_HORIZONTAL_ROW:
|
||||
printlog(LOG_WARN, "non horizontal row detected");
|
||||
logger.warning("non horizontal row detected");
|
||||
break;
|
||||
case E_NO_NET_DRIVING_PIN:
|
||||
printlog(LOG_WARN, "missing net driving pin");
|
||||
logger.warning("missing net driving pin");
|
||||
break;
|
||||
case E_MULTIPLE_NET_DRIVING_PIN:
|
||||
printlog(LOG_WARN, "multiple net driving pin");
|
||||
logger.warning("multiple net driving pin");
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
@ -1052,7 +1043,7 @@ void Database::errorCheck(bool autoFix) {
|
||||
}
|
||||
|
||||
void Database::checkPlaceError() {
|
||||
printlog(LOG_INFO, "starting checking...");
|
||||
logger.info("starting checking...");
|
||||
int nError = 0;
|
||||
vector<Cell*> cells = this->cells;
|
||||
sort(cells.begin(), cells.end(), [](const Cell* a, const Cell* b) { return a->lx() < b->lx(); });
|
||||
@ -1075,11 +1066,11 @@ void Database::checkPlaceError() {
|
||||
}
|
||||
}
|
||||
|
||||
printlog(LOG_INFO, "#overlap=%d", nError);
|
||||
logger.info("#overlap=%d", nError);
|
||||
}
|
||||
|
||||
void Database::checkDRCError() {
|
||||
printlog(LOG_INFO, "starting checking...");
|
||||
logger.info("starting checking...");
|
||||
vector<int> nOverlapErrors(3);
|
||||
vector<int> nSpacingErrors(3);
|
||||
|
||||
@ -1154,7 +1145,7 @@ void Database::checkDRCError() {
|
||||
}
|
||||
|
||||
for (unsigned i = 0; i != 3; ++i) {
|
||||
printlog(LOG_INFO, "m%d = %u", i + 1, metals[i].size());
|
||||
logger.info("m%d = %u", i + 1, metals[i].size());
|
||||
sort(metals[i].begin(), metals[i].end(), [](const Metal& a, const Metal& b) {
|
||||
return (a.rect.lx == b.rect.lx) ? (a.rect.ly < b.rect.ly) : (a.rect.lx < b.rect.lx);
|
||||
});
|
||||
@ -1217,7 +1208,7 @@ void Database::checkDRCError() {
|
||||
*/
|
||||
|
||||
for (unsigned i = 0; i != 3; ++i) {
|
||||
printlog(LOG_INFO, "#M%u overlaps = %d", i + 1, nOverlapErrors[i]);
|
||||
printlog(LOG_INFO, "#M%u spacings = %d", i + 1, nSpacingErrors[i]);
|
||||
logger.info("#M%u overlaps = %d", i + 1, nOverlapErrors[i]);
|
||||
logger.info("#M%u spacings = %d", i + 1, nSpacingErrors[i]);
|
||||
}
|
||||
}
|
||||
|
||||
@ -75,11 +75,11 @@ public:
|
||||
E_NO_NET_DRIVING_PIN
|
||||
};
|
||||
|
||||
unordered_map<string, CellType*> name_celltypes;
|
||||
unordered_map<string, Cell*> name_cells;
|
||||
unordered_map<string, Net*> name_nets;
|
||||
unordered_map<string, IOPin*> name_iopins;
|
||||
unordered_map<string, ViaType*> name_viatypes;
|
||||
robin_hood::unordered_map<string, CellType*> name_celltypes;
|
||||
robin_hood::unordered_map<string, Cell*> name_cells;
|
||||
robin_hood::unordered_map<string, Net*> name_nets;
|
||||
robin_hood::unordered_map<string, IOPin*> name_iopins;
|
||||
robin_hood::unordered_map<string, ViaType*> name_viatypes;
|
||||
|
||||
vector<Layer> layers;
|
||||
vector<Site> sites;
|
||||
|
||||
@ -20,10 +20,10 @@ int EdgeTypes::getEdgeSpace(const int edge1, const int edge2) const {
|
||||
|
||||
#ifdef DEBUG
|
||||
if (edge1 < 0 || edge1 >= (int)types.size()) {
|
||||
printlog(LOG_ERROR, "invalid edge ID: %d", edge1);
|
||||
logger.error("invalid edge ID: %d", edge1);
|
||||
}
|
||||
if (edge2 < 0 || edge2 >= (int)types.size()) {
|
||||
printlog(LOG_ERROR, "invalid edge ID: %d", edge2);
|
||||
logger.error("invalid edge ID: %d", edge2);
|
||||
}
|
||||
#endif
|
||||
return distTable[edge1][edge2];
|
||||
|
||||
@ -10,7 +10,7 @@ char Track::macro() const {
|
||||
case 'v':
|
||||
return 'X';
|
||||
default:
|
||||
printlog(LOG_ERROR, "track direction not recognized: %c", direction);
|
||||
logger.error("track direction not recognized: %c", direction);
|
||||
return '\0';
|
||||
}
|
||||
}
|
||||
|
||||
@ -24,10 +24,10 @@ public:
|
||||
void addLayer(const string& layer) { layers_.push_back(layer); }
|
||||
|
||||
const vector<string>& getLayer() const { return layers_; }
|
||||
const int getFirstTrackLoc() const { return start; }
|
||||
const int getLastTrackLoc() const { return start + (num - 1) * step; }
|
||||
const int getTrackLoc(unsigned trackIndex) const { return start + trackIndex * step; }
|
||||
const int getPitch() const { return step; }
|
||||
int getFirstTrackLoc() const { return start; }
|
||||
int getLastTrackLoc() const { return start + (num - 1) * step; }
|
||||
int getTrackLoc(unsigned trackIndex) const { return start + trackIndex * step; }
|
||||
int getPitch() const { return step; }
|
||||
|
||||
char macro() const;
|
||||
unsigned numLayers() const { return layers_.size(); }
|
||||
|
||||
@ -111,7 +111,7 @@ void Net::addPin(Pin* pin) {
|
||||
void PowerNet::addRail(SNet* snet, int lx, int hx, int y) {
|
||||
map<int, SNet*>::iterator rail = rails.find(y);
|
||||
if (rail != rails.end() && rail->second != snet) {
|
||||
printlog(LOG_ERROR, "rail %s already exists at y=%d , new rail %s is from %d to %d", rail->second->name.c_str(),
|
||||
logger.error("rail %s already exists at y=%d , new rail %s is from %d to %d", rail->second->name.c_str(),
|
||||
y, snet->name.c_str(),
|
||||
lx,
|
||||
hx);
|
||||
|
||||
@ -48,7 +48,7 @@ void Pin::getPinCenter(int& x, int& y) {
|
||||
x = iopin->x + (lx + hx) / 2;
|
||||
y = iopin->y + (ly + hy) / 2;
|
||||
} else {
|
||||
printlog(LOG_ERROR, "invalid pin %s:%d", __FILE__, __LINE__);
|
||||
logger.error("invalid pin %s:%d", __FILE__, __LINE__);
|
||||
x = INT_MIN;
|
||||
y = INT_MIN;
|
||||
}
|
||||
|
||||
@ -60,14 +60,14 @@ public:
|
||||
~IOPin();
|
||||
|
||||
const string& netName() const { return _netName; }
|
||||
const int width() const { return this->type->getW(); }
|
||||
const int height() const { return this->type->getH(); }
|
||||
const int lx() const { return x; }
|
||||
const int ly() const { return y; }
|
||||
const int hx() const { return x + width(); }
|
||||
const int hy() const { return y + height(); }
|
||||
const int cx() const { return x + width() / 2; }
|
||||
const int cy() const { return y + height() / 2; }
|
||||
int width() const { return this->type->getW(); }
|
||||
int height() const { return this->type->getH(); }
|
||||
int lx() const { return x; }
|
||||
int ly() const { return y; }
|
||||
int hx() const { return x + width(); }
|
||||
int hy() const { return y + height(); }
|
||||
int cx() const { return x + width() / 2; }
|
||||
int cy() const { return y + height() / 2; }
|
||||
void getBounds(int& lx, int& ly, int& hx, int& hy, int& rIndex) const;
|
||||
int orient() { return _orient; }
|
||||
};
|
||||
|
||||
@ -10,8 +10,8 @@ public:
|
||||
int nRows;
|
||||
|
||||
std::string format;
|
||||
unordered_map<string, int> cellMap;
|
||||
unordered_map<string, int> typeMap;
|
||||
robin_hood::unordered_map<string, int> cellMap;
|
||||
robin_hood::unordered_map<string, int> typeMap;
|
||||
vector<string> cellName;
|
||||
vector<int> cellX;
|
||||
vector<int> cellY;
|
||||
@ -22,7 +22,7 @@ public:
|
||||
vector<int> typeHeight;
|
||||
vector<char> typeFixed;
|
||||
vector<vector<vector<int>>> typeShapes;
|
||||
unordered_map<string, int> typePinMap;
|
||||
robin_hood::unordered_map<string, int> typePinMap;
|
||||
vector<int> typeNPins;
|
||||
vector<vector<string>> typePinName;
|
||||
vector<vector<char>> typePinDir;
|
||||
@ -54,7 +54,7 @@ public:
|
||||
int tileH;
|
||||
double blockagePorosity;
|
||||
vector<int> IOPinRouteLayer;
|
||||
vector<pair<int, vector<int>>> routeBlkgs; // cellID, BlockedLayers
|
||||
vector<pair<int, vector<int>>> routeBlkgs; // cellID, BlockedLayers
|
||||
|
||||
BookshelfData() {
|
||||
nCells = 0;
|
||||
@ -104,7 +104,7 @@ public:
|
||||
typePinY[i][j] *= scale;
|
||||
}
|
||||
for (int j = 0; j < typeShapes[i].size(); j++) {
|
||||
for (int k = 0; k < typeShapes[i][j].size(); k++){
|
||||
for (int k = 0; k < typeShapes[i][j].size(); k++) {
|
||||
typeShapes[i][j][k] *= scale;
|
||||
}
|
||||
}
|
||||
@ -176,11 +176,11 @@ public:
|
||||
}
|
||||
siteWidth = gcd(sizes);
|
||||
siteHeight = gcd(heights);
|
||||
printlog(LOG_INFO, "estimate site size = %d x %d", siteWidth, siteHeight);
|
||||
logger.info("estimate site size = %d x %d", siteWidth, siteHeight);
|
||||
ii = heightSet.begin();
|
||||
ie = heightSet.end();
|
||||
for (; ii != ie; ++ii) {
|
||||
printlog(LOG_INFO, "standard cell heights: %d rows", (*ii) / siteHeight);
|
||||
logger.info("standard cell heights: %d rows", (*ii) / siteHeight);
|
||||
}
|
||||
}
|
||||
int gcd(vector<int>& nums) {
|
||||
@ -199,7 +199,7 @@ public:
|
||||
bool factorValid = true;
|
||||
for (int i = 0; i < (int)nums.size(); i++) {
|
||||
int num = nums[i];
|
||||
// printlog(LOG_INFO, "%d : %d", i, num);
|
||||
// logger.info("%d : %d", i, num);
|
||||
if (num % factor != 0) {
|
||||
factorValid = false;
|
||||
break;
|
||||
@ -308,11 +308,11 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
directory = auxFile.substr(0, found);
|
||||
directory += "/";
|
||||
}
|
||||
printlog(LOG_INFO, "dir = %s", directory.c_str());
|
||||
logger.info("dir = %s", directory.c_str());
|
||||
|
||||
ifstream fs(auxFile.c_str());
|
||||
if (!fs.good()) {
|
||||
printlog(LOG_ERROR, "cannot open file: %s", auxFile.c_str());
|
||||
logger.error("cannot open file: %s", auxFile.c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
@ -351,7 +351,7 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
} else if (ext == ".pl") {
|
||||
filePl = directory + file;
|
||||
} else {
|
||||
printlog(LOG_ERROR, "unrecognized file extension: %s", ext.c_str());
|
||||
logger.error("unrecognized file extension: %s", ext.c_str());
|
||||
}
|
||||
}
|
||||
// step 1: read floorplan, rows from:
|
||||
@ -404,7 +404,8 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
// this->scale = 1;
|
||||
}
|
||||
|
||||
printlog(LOG_INFO, "parsing rows");
|
||||
logger.info("parsing rows");
|
||||
this->LefConvertFactor = 1; // suppose 1 in bookshelf
|
||||
this->dieLX = INT_MAX;
|
||||
this->dieLY = INT_MAX;
|
||||
this->dieHX = INT_MIN;
|
||||
@ -421,12 +422,16 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
this->dieHX = std::max(this->dieHX, row->x() + (int)row->width());
|
||||
this->dieHY = std::max(this->dieHY, row->y() + bsData.siteHeight);
|
||||
// make sure the parsed results are the same as the estimated results
|
||||
assert_msg(bsData.rowXStep[i] == bsData.siteWidth,
|
||||
"Row %s rowXStep (%d) is not equal to siteWidth (%d)",
|
||||
row->name().c_str(), bsData.rowXStep[i], bsData.siteWidth);
|
||||
assert_msg(bsData.rowHeight[i] == bsData.siteHeight,
|
||||
"Row %s rowYStep (%d) is not equal to siteHeight (%d)",
|
||||
row->name().c_str(), bsData.rowHeight[i], bsData.siteHeight);
|
||||
assert_msg(bsData.rowXStep[i] == bsData.siteWidth,
|
||||
"Row %s rowXStep (%d) is not equal to siteWidth (%d)",
|
||||
row->name().c_str(),
|
||||
bsData.rowXStep[i],
|
||||
bsData.siteWidth);
|
||||
assert_msg(bsData.rowHeight[i] == bsData.siteHeight,
|
||||
"Row %s rowYStep (%d) is not equal to siteHeight (%d)",
|
||||
row->name().c_str(),
|
||||
bsData.rowHeight[i],
|
||||
bsData.siteHeight);
|
||||
}
|
||||
|
||||
// parsing gcellgrid
|
||||
@ -463,10 +468,10 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
// }
|
||||
}
|
||||
|
||||
// NOTE: ICCAD/DAC 2012 does not define trackPitch and the capacity of each GCell
|
||||
// NOTE: ICCAD/DAC 2012 does not define trackPitch and the capacity of each GCell
|
||||
// cannot be directly computed by tracks. We restore the bookshelf capacity value
|
||||
// in Database::Others::capV(H) and will handle them in GRDatabase.
|
||||
printlog(LOG_INFO, "parsing layers");
|
||||
logger.info("parsing layers");
|
||||
int defaultPitch = bsData.siteWidth;
|
||||
int defaultWidth = bsData.siteWidth / 2;
|
||||
int defaultSpace = bsData.siteWidth - defaultWidth;
|
||||
@ -516,7 +521,7 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
}
|
||||
}
|
||||
|
||||
printlog(LOG_INFO, "parsing celltype");
|
||||
logger.info("parsing celltype");
|
||||
const Layer& layer = this->layers[0];
|
||||
for (int i = 0; i < bsData.nTypes; i++) {
|
||||
CellType* celltype = this->addCellType(bsData.typeName[i], this->celltypes.size());
|
||||
@ -533,7 +538,7 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
direction = 'o';
|
||||
break;
|
||||
default:
|
||||
printlog(LOG_ERROR, "unknown pin direction: %c", bsData.typePinDir[i][j]);
|
||||
logger.error("unknown pin direction: %c", bsData.typePinDir[i][j]);
|
||||
break;
|
||||
}
|
||||
PinType* pintype = celltype->addPin(bsData.typePinName[i][j], direction, 's');
|
||||
@ -549,7 +554,7 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
}
|
||||
}
|
||||
|
||||
printlog(LOG_INFO, "parsing cells");
|
||||
logger.info("parsing cells");
|
||||
for (int i = 0; i < bsData.nCells; i++) {
|
||||
int typeID = bsData.cellType[i];
|
||||
if (typeID < 0) {
|
||||
@ -576,7 +581,7 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
}
|
||||
}
|
||||
|
||||
printlog(LOG_INFO, "parsing nets");
|
||||
logger.info("parsing nets");
|
||||
for (unsigned i = 0; i != bsData.nNets; ++i) {
|
||||
Net* net = this->addNet(bsData.netName[i]);
|
||||
for (unsigned j = 0; j != bsData.netCells[i].size(); ++j) {
|
||||
@ -587,8 +592,7 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
pin = iopin->pin;
|
||||
if (pin->is_connected) {
|
||||
string netName(net->name);
|
||||
printlog(
|
||||
LOG_WARN, "IO Pin is re-connected: %s %s", netName.c_str(), bsData.cellName[cellID].c_str());
|
||||
logger.warning("IO Pin is re-connected: %s %s", netName.c_str(), bsData.cellName[cellID].c_str());
|
||||
}
|
||||
iopin->is_connected = true;
|
||||
} else {
|
||||
@ -596,11 +600,10 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
pin = cell->pin(bsData.netPins[i][j]);
|
||||
if (pin->is_connected) {
|
||||
string netName(net->name);
|
||||
printlog(LOG_WARN,
|
||||
"Pin is re-connected: %s %s %d",
|
||||
netName.c_str(),
|
||||
bsData.cellName[cellID].c_str(),
|
||||
bsData.netPins[i][j]);
|
||||
logger.warning("Pin is re-connected: %s %s %d",
|
||||
netName.c_str(),
|
||||
bsData.cellName[cellID].c_str(),
|
||||
bsData.netPins[i][j]);
|
||||
}
|
||||
cell->is_connected = true;
|
||||
}
|
||||
@ -635,22 +638,22 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
}
|
||||
|
||||
bsData.clearData();
|
||||
printlog(LOG_INFO, "finish reading bookshelf.");
|
||||
logger.info("finish reading bookshelf.");
|
||||
|
||||
return true;
|
||||
}
|
||||
bool Database::readBSNodes(const std::string& file) {
|
||||
printlog(LOG_INFO, "reading nodes");
|
||||
logger.info("reading nodes");
|
||||
ifstream fs(file.c_str());
|
||||
if (!fs.good()) {
|
||||
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
|
||||
logger.error("cannot open file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
// int nNodes = 0;
|
||||
// int nTerminals = 0;
|
||||
vector<string> tokens;
|
||||
while (readBSLine(fs, tokens)) {
|
||||
// printlog(LOG_INFO, "%d : %s", i++, tokens[0].c_str());
|
||||
// logger.info("%d : %s", i++, tokens[0].c_str());
|
||||
if (tokens[0] == "UCLA") {
|
||||
continue;
|
||||
} else if (tokens[0] == "NumNodes") {
|
||||
@ -668,7 +671,7 @@ bool Database::readBSNodes(const std::string& file) {
|
||||
// cout << cName << '\t' << cType << endl;
|
||||
}
|
||||
if (cType == "terminal" && cWidth > 1 && cHeight > 1) {
|
||||
// printlog(LOG_INFO, "read terminal");
|
||||
// logger.info("read terminal");
|
||||
cType = cName;
|
||||
cFixed = true;
|
||||
}
|
||||
@ -678,7 +681,7 @@ bool Database::readBSNodes(const std::string& file) {
|
||||
}
|
||||
int typeID = -1;
|
||||
if (cType == "terminal") {
|
||||
// printlog(LOG_INFO, "read terminal");
|
||||
// logger.info("read terminal");
|
||||
typeID = -1;
|
||||
cFixed = true;
|
||||
} else if (bsData.typeMap.find(cType) == bsData.typeMap.end()) {
|
||||
@ -716,10 +719,10 @@ bool Database::readBSNodes(const std::string& file) {
|
||||
}
|
||||
|
||||
bool Database::readBSNets(const std::string& file) {
|
||||
printlog(LOG_INFO, "reading net");
|
||||
logger.info("reading net");
|
||||
ifstream fs(file.c_str());
|
||||
if (!fs.good()) {
|
||||
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
|
||||
logger.error("cannot open file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
vector<string> tokens;
|
||||
@ -727,13 +730,13 @@ bool Database::readBSNets(const std::string& file) {
|
||||
if (tokens[0] == "UCLA") {
|
||||
continue;
|
||||
} else if (tokens[0] == "NumNets") {
|
||||
// printlog(LOG_INFO, "#nets : %d", atoi(tokens[1].c_str()));
|
||||
// logger.info("#nets : %d", atoi(tokens[1].c_str()));
|
||||
int numNets = atoi(tokens[1].c_str());
|
||||
bsData.netName.resize(numNets);
|
||||
bsData.netCells.resize(numNets);
|
||||
bsData.netPins.resize(numNets);
|
||||
} else if (tokens[0] == "NumPins") {
|
||||
// printlog(LOG_INFO, "#pins : %d", atoi(tokens[1].c_str()));
|
||||
// logger.info("#pins : %d", atoi(tokens[1].c_str()));
|
||||
} else if (tokens[0] == "NetDegree") {
|
||||
int degree = atoi(tokens[1].c_str());
|
||||
string nName = tokens[2];
|
||||
@ -744,7 +747,7 @@ bool Database::readBSNets(const std::string& file) {
|
||||
string cName = tokens[0];
|
||||
if (bsData.cellMap.find(cName) == bsData.cellMap.end()) {
|
||||
assert(false);
|
||||
printlog(LOG_ERROR, "cell not found : %s", cName.c_str());
|
||||
logger.error("cell not found : %s", cName.c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
@ -773,7 +776,7 @@ bool Database::readBSNets(const std::string& file) {
|
||||
stringstream ss;
|
||||
ss << bsData.typeNPins[typeID];
|
||||
pinName = ss.str();
|
||||
// printlog(LOG_INFO, "pinname = %s", pinName.c_str());
|
||||
// logger.info("pinname = %s", pinName.c_str());
|
||||
}
|
||||
if (typeID >= 0) {
|
||||
tpName.append(bsData.typeName[typeID]);
|
||||
@ -813,10 +816,10 @@ bool Database::readBSNets(const std::string& file) {
|
||||
}
|
||||
|
||||
bool Database::readBSScl(const std::string& file) {
|
||||
printlog(LOG_INFO, "reading scl");
|
||||
logger.info("reading scl");
|
||||
ifstream fs(file.c_str());
|
||||
if (!fs.good()) {
|
||||
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
|
||||
logger.error("cannot open file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
vector<string> tokens;
|
||||
@ -857,10 +860,10 @@ bool Database::readBSScl(const std::string& file) {
|
||||
}
|
||||
|
||||
bool Database::readBSRoute(const std::string& file) {
|
||||
printlog(LOG_INFO, "reading route");
|
||||
logger.info("reading route");
|
||||
ifstream fs(file.c_str());
|
||||
if (!fs.good()) {
|
||||
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
|
||||
logger.error("cannot open file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
vector<string> tokens;
|
||||
@ -912,7 +915,7 @@ bool Database::readBSRoute(const std::string& file) {
|
||||
} else if (status == ReadingPinLayer) {
|
||||
std::string cName = tokens[0];
|
||||
if (bsData.cellMap.find(cName) == bsData.cellMap.end()) {
|
||||
printlog(LOG_ERROR, "pin not found : %s", cName.c_str());
|
||||
logger.error("pin not found : %s", cName.c_str());
|
||||
getchar();
|
||||
}
|
||||
int cellID = bsData.cellMap[cName];
|
||||
@ -921,7 +924,7 @@ bool Database::readBSRoute(const std::string& file) {
|
||||
} else if (status == ReadingBlockages) {
|
||||
std::string cName = tokens[0];
|
||||
if (bsData.cellMap.find(cName) == bsData.cellMap.end()) {
|
||||
printlog(LOG_ERROR, "cell not found : %s", cName.c_str());
|
||||
logger.error("cell not found : %s", cName.c_str());
|
||||
getchar();
|
||||
}
|
||||
int cellID = bsData.cellMap[cName];
|
||||
@ -952,10 +955,10 @@ bool Database::readBSRoute(const std::string& file) {
|
||||
}
|
||||
|
||||
bool Database::readBSShapes(const std::string& file) {
|
||||
printlog(LOG_INFO, "reading shapes");
|
||||
logger.info("reading shapes");
|
||||
ifstream fs(file.c_str());
|
||||
if (!fs.good()) {
|
||||
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
|
||||
logger.error("cannot open file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
vector<string> tokens;
|
||||
@ -987,10 +990,10 @@ bool Database::readBSShapes(const std::string& file) {
|
||||
}
|
||||
|
||||
bool Database::readBSWts(const std::string& file) {
|
||||
printlog(LOG_INFO, "reading weights");
|
||||
logger.info("reading weights");
|
||||
ifstream fs(file.c_str());
|
||||
if (!fs.good()) {
|
||||
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
|
||||
logger.error("cannot open file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
vector<string> tokens;
|
||||
@ -999,10 +1002,10 @@ bool Database::readBSWts(const std::string& file) {
|
||||
}
|
||||
|
||||
bool Database::readBSPl(const std::string& file) {
|
||||
printlog(LOG_INFO, "reading placement");
|
||||
logger.info("reading placement");
|
||||
ifstream fs(file.c_str());
|
||||
if (!fs.good()) {
|
||||
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
|
||||
logger.error("cannot open file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
vector<string> tokens;
|
||||
@ -1010,10 +1013,10 @@ bool Database::readBSPl(const std::string& file) {
|
||||
if (tokens[0] == "UCLA") {
|
||||
continue;
|
||||
} else if (tokens.size() >= 4) {
|
||||
unordered_map<string, int>::iterator itr = bsData.cellMap.find(tokens[0]);
|
||||
robin_hood::unordered_map<string, int>::iterator itr = bsData.cellMap.find(tokens[0]);
|
||||
if (itr == bsData.cellMap.end()) {
|
||||
assert(false);
|
||||
printlog(LOG_ERROR, "cell not found: %s", tokens[0].c_str());
|
||||
logger.error("cell not found: %s", tokens[0].c_str());
|
||||
return false;
|
||||
}
|
||||
int cell = itr->second;
|
||||
@ -1036,7 +1039,7 @@ bool Database::readBSPl(const std::string& file) {
|
||||
bool Database::writeBSPl(const std::string& file) {
|
||||
ofstream fs(file.c_str());
|
||||
if (!fs.good()) {
|
||||
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
|
||||
logger.error("cannot open file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
fs << "UCLA pl 1.0\n";
|
||||
|
||||
@ -6,7 +6,7 @@ bool Database::readConstraints(const std::string& file) {
|
||||
string buffer;
|
||||
ifstream ifs(file.c_str());
|
||||
if (!ifs.good()) {
|
||||
printlog(LOG_ERROR, "cannot open constraint file: %s", file.c_str());
|
||||
logger.error("cannot open constraint file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
while (ifs >> buffer) {
|
||||
@ -18,7 +18,7 @@ bool Database::readConstraints(const std::string& file) {
|
||||
string value = buffer.substr(equal + 1, unit - equal - 1);
|
||||
maxDensity = atof(value.c_str()) / 100.0;
|
||||
} else {
|
||||
printlog(LOG_WARN, "use input max util %f", maxDensity);
|
||||
logger.warning("use input max util %f", maxDensity);
|
||||
}
|
||||
} else if (key == "maximum_movement") {
|
||||
if (!maxDisp) {
|
||||
@ -26,7 +26,7 @@ bool Database::readConstraints(const std::string& file) {
|
||||
string value = buffer.substr(equal + 1, unit - equal - 1);
|
||||
maxDisp = atof(value.c_str());
|
||||
} else {
|
||||
printlog(LOG_WARN, "use input max disp %f", maxDisp);
|
||||
logger.warning("use input max disp %f", maxDisp);
|
||||
}
|
||||
}
|
||||
}
|
||||
@ -37,7 +37,7 @@ bool Database::readConstraints(const std::string& file) {
|
||||
bool Database::readSize(const std::string& file) {
|
||||
ifstream fs(file.c_str());
|
||||
if (!fs.good()) {
|
||||
printlog(LOG_ERROR, "cannot open size file: %s", file.c_str());
|
||||
logger.error("cannot open size file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
@ -54,12 +54,12 @@ bool Database::readSize(const std::string& file) {
|
||||
}
|
||||
}
|
||||
if (siteW == 0 || siteH == 0) {
|
||||
printlog(LOG_INFO, "not enough information to retrieve placement site size");
|
||||
logger.info("not enough information to retrieve placement site size");
|
||||
return false;
|
||||
}
|
||||
|
||||
#ifndef NDEBUG
|
||||
printlog(LOG_INFO, "reading %s", file.c_str());
|
||||
logger.info("reading %s", file.c_str());
|
||||
#endif
|
||||
|
||||
std::set<CellType*> sized;
|
||||
@ -70,7 +70,7 @@ bool Database::readSize(const std::string& file) {
|
||||
if (name != "") {
|
||||
Cell* cell = getCell(name);
|
||||
if (cell == NULL) {
|
||||
printlog(LOG_ERROR, "cell not found : %s", name.c_str());
|
||||
logger.error("cell not found : %s", name.c_str());
|
||||
break;
|
||||
} else {
|
||||
CellType* celltype = cell->ctype();
|
||||
|
||||
@ -146,12 +146,12 @@ inline void fastCopy(char* t, const char* s, size_t n);
|
||||
bool Database::readLEF(const std::string& file) {
|
||||
FILE* fp;
|
||||
if (!(fp = fopen(file.c_str(), "r"))) {
|
||||
printlog(LOG_ERROR, "Unable to open LEF file: %s", file.c_str());
|
||||
logger.error("Unable to open LEF file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
#ifndef NDEBUG
|
||||
printlog(LOG_INFO, "reading %s", file.c_str());
|
||||
logger.info("reading %s", file.c_str());
|
||||
#endif
|
||||
|
||||
lefrSetUnitsCbk(readLefUnits);
|
||||
@ -168,7 +168,7 @@ bool Database::readLEF(const std::string& file) {
|
||||
lefrReset();
|
||||
int res = lefrRead(fp, file.c_str(), (void*)this);
|
||||
if (res) {
|
||||
printlog(LOG_ERROR, "Error in reading LEF");
|
||||
logger.error("Error in reading LEF");
|
||||
return false;
|
||||
}
|
||||
lefrReleaseNResetMemory();
|
||||
@ -183,12 +183,12 @@ bool Database::readLEF(const std::string& file) {
|
||||
bool Database::readDEF(const std::string& file) {
|
||||
FILE* fp;
|
||||
if (!(fp = fopen(file.c_str(), "r"))) {
|
||||
printlog(LOG_ERROR, "Unable to open DEF file: %s", file.c_str());
|
||||
logger.error("Unable to open DEF file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
#ifndef NDEBUG
|
||||
printlog(LOG_INFO, "reading %s", file.c_str());
|
||||
logger.info("reading %s", file.c_str());
|
||||
#endif
|
||||
|
||||
defrSetDesignCbk(readDefDesign);
|
||||
@ -222,7 +222,7 @@ bool Database::readDEF(const std::string& file) {
|
||||
defrReset();
|
||||
int res = defrRead(fp, file.c_str(), (void*)this, 1);
|
||||
if (res) {
|
||||
printlog(LOG_ERROR, "Error in reading DEF");
|
||||
logger.error("Error in reading DEF");
|
||||
return false;
|
||||
}
|
||||
defrReleaseNResetMemory();
|
||||
@ -236,12 +236,12 @@ bool Database::readDEFPG(const std::string& file) {
|
||||
string buffer;
|
||||
ifstream ifs(file.c_str());
|
||||
if (!ifs.good()) {
|
||||
printlog(LOG_ERROR, "Unable to open DEF PG file: %s", file.c_str());
|
||||
logger.error("Unable to open DEF PG file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
#ifndef NDEBUG
|
||||
printlog(LOG_INFO, "reading %s", file.c_str());
|
||||
logger.info("reading %s", file.c_str());
|
||||
#endif
|
||||
|
||||
Database* db = this;
|
||||
@ -280,7 +280,7 @@ bool Database::readDEFPG(const std::string& file) {
|
||||
ifs >> buffer >> buffer >> shape >> buffer >> fx >> fy >> buffer >> vianame;
|
||||
ViaType* viatype = db->getViaType(vianame);
|
||||
if (!viatype) {
|
||||
printlog(LOG_ERROR, "Via type is not defined: %s", vianame.c_str());
|
||||
logger.error("Via type is not defined: %s", vianame.c_str());
|
||||
return false;
|
||||
}
|
||||
// snet->addVia(viatype, fx, fy);
|
||||
@ -305,7 +305,7 @@ bool Database::readDEFPG(const std::string& file) {
|
||||
// }
|
||||
Layer* layer = db->getLayer(layername);
|
||||
if (!layer) {
|
||||
printlog(LOG_ERROR, "Layer is not defined: %s", layername.c_str());
|
||||
logger.error("Layer is not defined: %s", layername.c_str());
|
||||
return false;
|
||||
}
|
||||
// int lx, ly, hx, hy;
|
||||
@ -334,7 +334,7 @@ bool Database::readDEFPG(const std::string& file) {
|
||||
} else if (type == "GROUND") {
|
||||
// snet->type = 'g';
|
||||
} else {
|
||||
printlog(LOG_ERROR, "unknown use: %s", type.c_str());
|
||||
logger.error("unknown use: %s", type.c_str());
|
||||
}
|
||||
}
|
||||
if (buffer == ";") {
|
||||
@ -390,20 +390,20 @@ bool Database::writeComponents(ofstream& ofs) {
|
||||
bool Database::writeICCAD2017(const string& inputDef, const string& outputDef) {
|
||||
ifstream ifs(inputDef.c_str());
|
||||
if (!ifs.good()) {
|
||||
printlog(LOG_ERROR, "Unable to create/open DEF: %s", inputDef.c_str());
|
||||
logger.error("Unable to create/open DEF: %s", inputDef.c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
#ifndef NDEBUG
|
||||
printlog(LOG_INFO, "reading %s", inputDef.c_str());
|
||||
logger.info("reading %s", inputDef.c_str());
|
||||
#endif
|
||||
|
||||
ofstream ofs(outputDef.c_str());
|
||||
if (!ofs.good()) {
|
||||
printlog(LOG_ERROR, "Unable to create/open DEF: %s", outputDef.c_str());
|
||||
logger.error("Unable to create/open DEF: %s", outputDef.c_str());
|
||||
return false;
|
||||
}
|
||||
printlog(LOG_INFO, "writing %s", outputDef.c_str());
|
||||
logger.info("writing %s", outputDef.c_str());
|
||||
|
||||
string line;
|
||||
while (getline(ifs, line)) {
|
||||
@ -431,10 +431,10 @@ bool Database::writeICCAD2017(const string& inputDef, const string& outputDef) {
|
||||
bool Database::writeICCAD2017(const string& outputDef) {
|
||||
ofstream ofs(outputDef.c_str(), ios::app);
|
||||
if (!ofs.good()) {
|
||||
printlog(LOG_ERROR, "Unable to create/open DEF: %s", outputDef.c_str());
|
||||
logger.error("Unable to create/open DEF: %s", outputDef.c_str());
|
||||
return false;
|
||||
}
|
||||
printlog(LOG_INFO, "writing %s", outputDef.c_str());
|
||||
logger.info("writing %s", outputDef.c_str());
|
||||
|
||||
writeComponents(ofs); // just replace the information of components, while others keep remain.
|
||||
|
||||
@ -447,10 +447,10 @@ bool Database::writeICCAD2017(const string& outputDef) {
|
||||
bool Database::writeDEF(const string& file) {
|
||||
ofstream ofs(file.c_str());
|
||||
if (!ofs.good()) {
|
||||
printlog(LOG_ERROR, "Unable to create/open DEF: %s", file.c_str());
|
||||
logger.error("Unable to create/open DEF: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
printlog(LOG_INFO, "writing %s", file.c_str());
|
||||
logger.info("writing %s", file.c_str());
|
||||
|
||||
ofs << "VERSION 5.8 ;" << endl;
|
||||
ofs << "DIVIDERCHAR \"/\" ;" << endl;
|
||||
@ -534,7 +534,7 @@ bool Database::writeDEF(const string& file) {
|
||||
ossr << "\t+ TYPE GUIDE ;\n";
|
||||
break;
|
||||
default:
|
||||
printlog(LOG_ERROR, "region type not recognized: %c", region->type());
|
||||
logger.error("region type not recognized: %c", region->type());
|
||||
ossr << " ;\n";
|
||||
break;
|
||||
}
|
||||
@ -562,7 +562,7 @@ bool Database::writeDEF(const string& file) {
|
||||
ofs << "\n\t\t+ DIRECTION INOUT";
|
||||
break;
|
||||
default:
|
||||
printlog(LOG_ERROR, "iopin direction not recognized: %c", iopin->type->direction());
|
||||
logger.error("iopin direction not recognized: %c", iopin->type->direction());
|
||||
break;
|
||||
}
|
||||
if (iopin->x != INT_MIN || iopin->y != INT_MIN) {
|
||||
@ -662,7 +662,7 @@ int readLefProp(lefrCallbackType_e c, lefiProp* prop, lefiUserData ud) {
|
||||
double microndist;
|
||||
sstable >> type1 >> type2 >> type3;
|
||||
if (type3 == "EXCEPTABUTTED") {
|
||||
printlog(LOG_WARN, "ignore EXCEPTABUTTED between %s and %s", type1.c_str(), type2.c_str());
|
||||
logger.warning("ignore EXCEPTABUTTED between %s and %s", type1.c_str(), type2.c_str());
|
||||
sstable >> microndist;
|
||||
} else {
|
||||
microndist = stod(type3);
|
||||
@ -717,7 +717,7 @@ int readLefLayer(lefrCallbackType_e c, lefiLayer* leflayer, lefiUserData ud) {
|
||||
type = 'r';
|
||||
} else if (!strcmp(leflayer->type(), "CUT")) {
|
||||
if (db->layers.empty()) {
|
||||
printlog(LOG_WARN, "remove cut layer %s below the first metal layer", name.c_str());
|
||||
logger.warning("remove cut layer %s below the first metal layer", name.c_str());
|
||||
return 0;
|
||||
}
|
||||
type = 'c';
|
||||
@ -849,14 +849,14 @@ int readLefLayer(lefrCallbackType_e c, lefiLayer* leflayer, lefiUserData ud) {
|
||||
if (leflayer->hasSpacingNumber()) {
|
||||
switch (leflayer->numSpacing()) {
|
||||
case 0:
|
||||
printlog(LOG_WARN, "layer has no spacing: %s", name.c_str());
|
||||
logger.warning("layer has no spacing: %s", name.c_str());
|
||||
return 0;
|
||||
case 1:
|
||||
layer.spacing = leflayer->spacing(0);
|
||||
return 0;
|
||||
default:
|
||||
layer.spacing = leflayer->spacing(0);
|
||||
printlog(LOG_WARN, "layer has multiple spacing: %s", name.c_str());
|
||||
logger.warning("layer has multiple spacing: %s", name.c_str());
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
@ -877,7 +877,7 @@ int readLefVia(lefrCallbackType_e c, lefiVia* lvia, lefiUserData ud) {
|
||||
string layername(lvia->lefiVia::layerName(i));
|
||||
Layer* layer = db->getLayer(layername);
|
||||
if (!layer) {
|
||||
printlog(LOG_ERROR, "layer not found: %s", layername.c_str());
|
||||
logger.error("layer not found: %s", layername.c_str());
|
||||
}
|
||||
for (int j = 0; j < lvia->lefiVia::numRects(i); ++j) {
|
||||
via->addRect(*layer,
|
||||
@ -954,7 +954,7 @@ int readLefPin(lefrCallbackType_e c, lefiPin* pin, lefiUserData ud) {
|
||||
} else if (use == "SIGNAL") {
|
||||
// Pin is used for regular net connectivity.
|
||||
} else {
|
||||
printlog(LOG_ERROR, "unknown use: %s.%s", celltype->name.c_str(), use.c_str());
|
||||
logger.error("unknown use: %s.%s", celltype->name.c_str(), use.c_str());
|
||||
}
|
||||
}
|
||||
|
||||
@ -964,17 +964,16 @@ int readLefPin(lefrCallbackType_e c, lefiPin* pin, lefiUserData ud) {
|
||||
direction = 'o';
|
||||
} else if (sDir == "OUTPUT TRISTATE") {
|
||||
direction = 'o';
|
||||
printlog(
|
||||
LOG_WARN, "treat pin %s.%s direction %s as OUTPUT", celltype->name.c_str(), name.c_str(), sDir.c_str());
|
||||
logger.warning(
|
||||
"treat pin %s.%s direction %s as OUTPUT", celltype->name.c_str(), name.c_str(), sDir.c_str());
|
||||
} else if (sDir == "INPUT") {
|
||||
direction = 'i';
|
||||
} else if (sDir == "INOUT") {
|
||||
if (name != "VDD" && name != "vdd" && name != "VSS" && name != "vss") {
|
||||
printlog(
|
||||
LOG_WARN, "unknown pin %s.%s direction: %s", celltype->name.c_str(), name.c_str(), sDir.c_str());
|
||||
logger.warning("unknown pin %s.%s direction: %s", celltype->name.c_str(), name.c_str(), sDir.c_str());
|
||||
}
|
||||
} else {
|
||||
printlog(LOG_ERROR, "unknown pin %s.%s direction: %s", celltype->name.c_str(), name.c_str(), sDir.c_str());
|
||||
logger.error("unknown pin %s.%s direction: %s", celltype->name.c_str(), name.c_str(), sDir.c_str());
|
||||
}
|
||||
}
|
||||
|
||||
@ -990,7 +989,7 @@ int readLefPin(lefrCallbackType_e c, lefiPin* pin, lefiUserData ud) {
|
||||
// }
|
||||
|
||||
if (pin->hasTaperRule()) {
|
||||
printlog(LOG_WARN, "pin %s has taper rule %s", name.c_str(), pin->taperRule());
|
||||
logger.warning("pin %s has taper rule %s", name.c_str(), pin->taperRule());
|
||||
}
|
||||
|
||||
PinType* pintype = celltype->addPin(name, direction, type);
|
||||
@ -1006,7 +1005,7 @@ int readLefPin(lefrCallbackType_e c, lefiPin* pin, lefiUserData ud) {
|
||||
for (unsigned i = 0; i != (unsigned)geom->numItems(); ++i) {
|
||||
switch (geom->itemType(i)) {
|
||||
case lefiGeomUnknown:
|
||||
printlog(LOG_WARN, "lefiGeomUnknown: %s.%s", celltype->name.c_str(), name.c_str());
|
||||
logger.warning("lefiGeomUnknown: %s.%s", celltype->name.c_str(), name.c_str());
|
||||
break;
|
||||
case lefiGeomLayerE:
|
||||
layer = db->getLayer(string(geom->getLayer(i)));
|
||||
@ -1132,11 +1131,8 @@ int readLefPin(lefrCallbackType_e c, lefiPin* pin, lefiUserData ud) {
|
||||
}
|
||||
break;
|
||||
default:
|
||||
printlog(LOG_WARN,
|
||||
"unknown lefiGeomEnum %u: %s.%s",
|
||||
geom->itemType(i),
|
||||
celltype->name.c_str(),
|
||||
name.c_str());
|
||||
logger.warning(
|
||||
"unknown lefiGeomEnum %u: %s.%s", geom->itemType(i), celltype->name.c_str(), name.c_str());
|
||||
break;
|
||||
}
|
||||
}
|
||||
@ -1148,7 +1144,7 @@ int readLefSite(lefrCallbackType_e c, lefiSite* site, lefiUserData ud) {
|
||||
int convertFactor = db->LefConvertFactor;
|
||||
int width = lround(site->sizeX() * convertFactor);
|
||||
int height = lround(site->sizeY() * convertFactor);
|
||||
printlog(LOG_INFO, "site name: %s class: %s siteW: %d siteH: %d", site->name(), site->siteClass(), width, height);
|
||||
logger.info("site name: %s class: %s siteW: %d siteH: %d", site->name(), site->siteClass(), width, height);
|
||||
db->addSite(site->name(), site->siteClass(), width, height);
|
||||
return 0;
|
||||
}
|
||||
@ -1171,7 +1167,7 @@ int readLefMacro(lefrCallbackType_e c, lefiMacro* macro, lefiUserData ud) {
|
||||
celltype->cls = 'b';
|
||||
} else {
|
||||
celltype->cls = clsname[0];
|
||||
printlog(LOG_WARN, "Class type is not defined: %s", clsname);
|
||||
logger.warning("Class type is not defined: %s", clsname);
|
||||
}
|
||||
} else {
|
||||
celltype->cls = 'c';
|
||||
@ -1212,17 +1208,17 @@ int readLefMacro(lefrCallbackType_e c, lefiMacro* macro, lefiUserData ud) {
|
||||
} else if (edgeside == "BOTTOM") {
|
||||
static bool missBot = true;
|
||||
if (missBot) {
|
||||
printlog(LOG_WARN, "unknown edge side: %s", edgeside.c_str());
|
||||
logger.warning("unknown edge side: %s", edgeside.c_str());
|
||||
missBot = false;
|
||||
}
|
||||
} else if (edgeside == "TOP") {
|
||||
static bool missTop = true;
|
||||
if (missTop) {
|
||||
printlog(LOG_WARN, "unknown edge side: %s", edgeside.c_str());
|
||||
logger.warning("unknown edge side: %s", edgeside.c_str());
|
||||
missTop = false;
|
||||
}
|
||||
} else {
|
||||
printlog(LOG_WARN, "unknown edge side: %s", edgeside.c_str());
|
||||
logger.warning("unknown edge side: %s", edgeside.c_str());
|
||||
}
|
||||
}
|
||||
}
|
||||
@ -1308,10 +1304,10 @@ int readDefTrack(defrCallbackType_e c, defiTrack* dtrack, defiUserData ud) {
|
||||
} else if (layer->direction != track->direction) {
|
||||
layer->nonPreferDirTrack = *track;
|
||||
} else {
|
||||
printlog(LOG_ERROR, "wrong definition of tracks for layer %s", layername.c_str());
|
||||
logger.error("wrong definition of tracks for layer %s", layername.c_str());
|
||||
}
|
||||
} else {
|
||||
printlog(LOG_ERROR, "layer name not found: %s", layername.c_str());
|
||||
logger.error("layer name not found: %s", layername.c_str());
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
@ -1356,7 +1352,7 @@ int readDefVia(defrCallbackType_e c, defiVia* dvia, defiUserData ud) {
|
||||
if (layer) {
|
||||
via->addRect(*layer, lx, ly, hx, hy);
|
||||
} else {
|
||||
printlog(LOG_INFO, "layer name not found: %s", dvialayer);
|
||||
logger.info("layer name not found: %s", dvialayer);
|
||||
}
|
||||
}
|
||||
|
||||
@ -1431,7 +1427,7 @@ int readDefNdr(defrCallbackType_e c, defiNonDefault* nd, defiUserData ud) {
|
||||
string vianame(nd->viaName(i));
|
||||
ViaType* viatype = db->getViaType(vianame);
|
||||
if (!viatype) {
|
||||
printlog(LOG_WARN, "NDR via type not found: %s", vianame.c_str());
|
||||
logger.warning("NDR via type not found: %s", vianame.c_str());
|
||||
}
|
||||
ndr->vias.push_back(viatype);
|
||||
}
|
||||
@ -1457,20 +1453,16 @@ int readDefComponent(defrCallbackType_e c, defiComponent* co, defiUserData ud) {
|
||||
cell->fixed(false);
|
||||
if (co->placementOrient() % 2 == 1) {
|
||||
// 0:N, 1:W, 2:S, 3:E, 4:FN, 5:FW, 6:FS, 7:FE
|
||||
printlog(LOG_WARN,
|
||||
"Cell [%s]'s placementOrient [%d] is not supported.",
|
||||
cell->name().c_str(),
|
||||
co->placementOrient());
|
||||
logger.warning(
|
||||
"Cell [%s]'s placementOrient [%d] is not supported.", cell->name().c_str(), co->placementOrient());
|
||||
}
|
||||
} else if (co->isFixed()) {
|
||||
cell->place(co->placementX(), co->placementY(), isFlipX(co->placementOrient()), isFlipY(co->placementOrient()));
|
||||
cell->fixed(true);
|
||||
if (co->placementOrient() % 2 == 1) {
|
||||
// 0:N, 1:W, 2:S, 3:E, 4:FN, 5:FW, 6:FS, 7:FE
|
||||
printlog(LOG_WARN,
|
||||
"Cell [%s]'s placementOrient [%d] is not supported.",
|
||||
cell->name().c_str(),
|
||||
co->placementOrient());
|
||||
logger.warning(
|
||||
"Cell [%s]'s placementOrient [%d] is not supported.", cell->name().c_str(), co->placementOrient());
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
@ -1489,11 +1481,11 @@ int readDefPin(defrCallbackType_e c, defiPin* dpin, defiUserData ud) {
|
||||
// OUTPUT to the chip, input to external
|
||||
direction = 'i';
|
||||
} else {
|
||||
printlog(LOG_WARN, "unknown pin signal direction: %s", dpin->direction());
|
||||
logger.warning("unknown pin signal direction: %s", dpin->direction());
|
||||
}
|
||||
} else {
|
||||
string pinName(dpin->pinName());
|
||||
printlog(LOG_WARN, "Pin %s has no pin signal direction", pinName.c_str());
|
||||
logger.warning("Pin %s has no pin signal direction", pinName.c_str());
|
||||
}
|
||||
|
||||
IOPin* iopin = db->addIOPin(string(dpin->pinName()), string(dpin->netName()), direction);
|
||||
@ -1524,7 +1516,7 @@ int readDefBlockage(defrCallbackType_e c, defiBlockage* dblk, defiUserData ud) {
|
||||
string layername(dblk->layerName());
|
||||
Layer* layer = db->getLayer(layername);
|
||||
if (!layer) {
|
||||
printlog(LOG_ERROR, "layer not found: %s", layername.c_str());
|
||||
logger.error("layer not found: %s", layername.c_str());
|
||||
return 1;
|
||||
}
|
||||
for (int i = 0; i < dblk->numRectangles(); ++i) {
|
||||
@ -1556,7 +1548,7 @@ int readDefSNet(defrCallbackType_e c, defiNet* dnet, defiUserData ud) {
|
||||
} else if (use == "GROUND") {
|
||||
snet->type = 'g';
|
||||
} else {
|
||||
printlog(LOG_ERROR, "unknown use: %s", use.c_str());
|
||||
logger.error("unknown use: %s", use.c_str());
|
||||
}
|
||||
}
|
||||
|
||||
@ -1590,7 +1582,7 @@ int readDefSNet(defrCallbackType_e c, defiNet* dnet, defiUserData ud) {
|
||||
layername = dpath->getLayer();
|
||||
layer = db->getLayer(layername);
|
||||
if (!layer) {
|
||||
printlog(LOG_ERROR, "Layer is not defined: %s", layername.c_str());
|
||||
logger.error("Layer is not defined: %s", layername.c_str());
|
||||
return false;
|
||||
}
|
||||
break;
|
||||
@ -1598,7 +1590,7 @@ int readDefSNet(defrCallbackType_e c, defiNet* dnet, defiUserData ud) {
|
||||
vianame = dpath->getVia();
|
||||
viatype = db->getViaType(vianame);
|
||||
if (!viatype) {
|
||||
printlog(LOG_ERROR, "Via type is not defined: %s", vianame.c_str());
|
||||
logger.error("Via type is not defined: %s", vianame.c_str());
|
||||
return false;
|
||||
}
|
||||
snet->addVia(viatype, fx, fy);
|
||||
@ -1689,12 +1681,12 @@ int readDefNet(defrCallbackType_e c, defiNet* dnet, defiUserData ud) {
|
||||
string designrulename(dnet->nonDefaultRule());
|
||||
ndr = db->getNDR(designrulename);
|
||||
if (!ndr) {
|
||||
printlog(LOG_WARN, "NDR rule is not defined: %s", designrulename.c_str());
|
||||
logger.warning("NDR rule is not defined: %s", designrulename.c_str());
|
||||
}
|
||||
}
|
||||
if ((unsigned)dnet->numConnections() == 0) {
|
||||
string netName(dnet->name());
|
||||
printlog(LOG_WARN, "Net %s is 0-Pin net. Ignore.", netName.c_str());
|
||||
logger.warning("Net %s is 0-Pin net. Ignore.", netName.c_str());
|
||||
return 0;
|
||||
}
|
||||
|
||||
@ -1706,12 +1698,12 @@ int readDefNet(defrCallbackType_e c, defiNet* dnet, defiUserData ud) {
|
||||
string iopinname(dnet->pin(i));
|
||||
IOPin* iopin = db->getIOPin(iopinname);
|
||||
if (!iopin) {
|
||||
printlog(LOG_WARN, "IO pin is not defined: %s", iopinname.c_str());
|
||||
logger.warning("IO pin is not defined: %s", iopinname.c_str());
|
||||
}
|
||||
pin = iopin->pin;
|
||||
if (pin->is_connected) {
|
||||
string netName(dnet->name());
|
||||
printlog(LOG_WARN, "IO Pin is re-connected: %s %s", netName.c_str(), iopinname.c_str());
|
||||
logger.warning("IO Pin is re-connected: %s %s", netName.c_str(), iopinname.c_str());
|
||||
}
|
||||
iopin->is_connected = true;
|
||||
} else {
|
||||
@ -1719,16 +1711,16 @@ int readDefNet(defrCallbackType_e c, defiNet* dnet, defiUserData ud) {
|
||||
string pinname(dnet->pin(i));
|
||||
Cell* cell = db->getCell(cellname);
|
||||
if (!cell) {
|
||||
printlog(LOG_WARN, "Cell is not defined: %s", cellname.c_str());
|
||||
logger.warning("Cell is not defined: %s", cellname.c_str());
|
||||
}
|
||||
pin = cell->pin(pinname);
|
||||
if (!pin) {
|
||||
string netName(dnet->name());
|
||||
printlog(LOG_WARN, "Pin is not defined: %s %s %s", netName.c_str(), cellname.c_str(), pinname.c_str());
|
||||
logger.warning("Pin is not defined: %s %s %s", netName.c_str(), cellname.c_str(), pinname.c_str());
|
||||
}
|
||||
if (pin->is_connected) {
|
||||
string netName(dnet->name());
|
||||
printlog(LOG_WARN, "Pin is re-connected: %s %s %s", netName.c_str(), cellname.c_str(), pinname.c_str());
|
||||
logger.warning("Pin is re-connected: %s %s %s", netName.c_str(), cellname.c_str(), pinname.c_str());
|
||||
}
|
||||
cell->is_connected = true;
|
||||
}
|
||||
@ -1818,10 +1810,10 @@ int readDefRegion(defrCallbackType_e c, defiRegion* dreg, defiUserData ud) {
|
||||
} else if (!strcmp(dreg->type(), "GUIDE")) {
|
||||
type = 'g';
|
||||
} else {
|
||||
printlog(LOG_WARN, "Unknown region type: %s", dreg->type());
|
||||
logger.warning("Unknown region type: %s", dreg->type());
|
||||
}
|
||||
} else {
|
||||
printlog(LOG_WARN, "Region is defined without type, use default region type = FENCE");
|
||||
logger.warning("Region is defined without type, use default region type = FENCE");
|
||||
}
|
||||
|
||||
Region* region = db->addRegion(string(dreg->name()), type);
|
||||
@ -1833,7 +1825,7 @@ int readDefRegion(defrCallbackType_e c, defiRegion* dreg, defiUserData ud) {
|
||||
}
|
||||
|
||||
//-----Group-----
|
||||
//#define GROUP_MARKER 9999999 //first mark all group member cell with this marker, then replace the value in one scan
|
||||
// #define GROUP_MARKER 9999999 //first mark all group member cell with this marker, then replace the value in one scan
|
||||
int readDefGroupName(defrCallbackType_e c, const char* cl, defiUserData ud) { return 0; }
|
||||
|
||||
int readDefGroupMember(defrCallbackType_e c, const char* cl, defiUserData ud) {
|
||||
@ -1847,7 +1839,7 @@ int readDefGroup(defrCallbackType_e c, defiGroup* dgp, defiUserData ud) {
|
||||
string regionname(dgp->regionName());
|
||||
Region* region = db->getRegion(regionname);
|
||||
if (!region) {
|
||||
printlog(LOG_WARN, "Region is not defined: %s", regionname.c_str());
|
||||
logger.warning("Region is not defined: %s", regionname.c_str());
|
||||
return 1;
|
||||
}
|
||||
region->members = db->regions[0]->members;
|
||||
|
||||
@ -74,7 +74,7 @@ bool readVerilogLine(istream &is, vector<string> &tokens) {
|
||||
bool Database::readVerilog(const std::string &file) {
|
||||
ifstream fs(file.c_str());
|
||||
if (!fs.good()) {
|
||||
printlog(LOG_ERROR, "cannot open verilog file: %s", file.c_str());
|
||||
logger.error("cannot open verilog file: %s", file.c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
@ -132,7 +132,7 @@ bool Database::readVerilog(const std::string &file) {
|
||||
string pinName(tokens[i]);
|
||||
IOPin *iopin = getIOPin(pinName);
|
||||
if (!iopin) {
|
||||
printlog(LOG_ERROR, "io pin not found: %s", pinName.c_str());
|
||||
logger.error("io pin not found: %s", pinName.c_str());
|
||||
}
|
||||
Net *net = addNet(pinName);
|
||||
net->addPin(iopin->pin);
|
||||
|
||||
@ -65,24 +65,6 @@ double mem_use::get_peak() {
|
||||
#endif
|
||||
}
|
||||
|
||||
void printlog(int log_level, const char* format, ...) {
|
||||
if (!verbose_parser_log) {
|
||||
return;
|
||||
}
|
||||
if (log_level >= GLOBAL_LOG_LEVEL) {
|
||||
std::string curr_log = tstamp.get_time_stamp();
|
||||
if (log_level > LOG_INFO) {
|
||||
curr_log += log_level_ANSI_color(log_level);
|
||||
}
|
||||
std::cout << curr_log;
|
||||
va_list ap;
|
||||
va_start(ap, format);
|
||||
vfprintf(stdout, format, ap);
|
||||
printf("\n");
|
||||
fflush(stdout);
|
||||
}
|
||||
}
|
||||
|
||||
void assert_msg(bool condition, const char* format, ...) {
|
||||
if (!condition) {
|
||||
std::cerr << "Assertion failure: ";
|
||||
@ -120,6 +102,7 @@ std::string log_level_ANSI_color(int& log_level) {
|
||||
}
|
||||
|
||||
|
||||
// C++ 20 only
|
||||
// void PrintfLogger::setup_logger(argparse::ArgumentParser parser) {
|
||||
// std::filesystem::path current_dir(std::filesystem::current_path());
|
||||
// // std::filesystem::path result_dir(parser.get<std::string>("result_dir"));
|
||||
@ -139,7 +122,6 @@ std::string log_level_ANSI_color(int& log_level) {
|
||||
|
||||
// if (write_log) {
|
||||
// f = fopen(log_file_path.c_str(), "w");
|
||||
// this->warning("Logging into file would affect the elapsed time. Disable it before submission.");
|
||||
// if (f == NULL) {
|
||||
// this->error("Cannot open logfile {}", log_file_path);
|
||||
// exit(1);
|
||||
|
||||
@ -28,7 +28,6 @@ extern bool verbose_parser_log;
|
||||
|
||||
// 1. Timer
|
||||
|
||||
|
||||
class timer {
|
||||
using clock = std::chrono::high_resolution_clock;
|
||||
|
||||
@ -54,18 +53,6 @@ public:
|
||||
|
||||
// 3. Easy print
|
||||
|
||||
// print(a, b, c)
|
||||
inline void print() { std::cout << std::endl; }
|
||||
template <typename T, typename... TAIL>
|
||||
void print(const T& t, TAIL... tail) {
|
||||
std::cout << t << ' ';
|
||||
print(tail...);
|
||||
}
|
||||
|
||||
// "printlog(LOG_LEVEL, a, b, c...)" puts a time stamp in beginning
|
||||
// try to make code compatible with old printlog(int level, const char *format, ...)
|
||||
void printlog(int level, const char* format, ...);
|
||||
|
||||
void assert_msg(bool condition, const char* format, ...);
|
||||
|
||||
std::string log_level_ANSI_color(int& log_level);
|
||||
@ -75,6 +62,7 @@ std::string log_level_ANSI_color(int& log_level);
|
||||
class PrintfLogger {
|
||||
static constexpr bool write_log = false;
|
||||
FILE* f;
|
||||
bool tmp_verbose_parser_log = false;
|
||||
|
||||
public:
|
||||
// void setup_logger(argparse::ArgumentParser parser);
|
||||
@ -82,6 +70,20 @@ public:
|
||||
if (f != NULL) fclose(f);
|
||||
}
|
||||
|
||||
void enable_logger() {
|
||||
tmp_verbose_parser_log = verbose_parser_log;
|
||||
verbose_parser_log = true;
|
||||
}
|
||||
|
||||
void disable_logger() {
|
||||
tmp_verbose_parser_log = verbose_parser_log;
|
||||
verbose_parser_log = false;
|
||||
}
|
||||
|
||||
void reset_logger() {
|
||||
verbose_parser_log = tmp_verbose_parser_log;
|
||||
}
|
||||
|
||||
template <typename... Args>
|
||||
void log(int log_level, const char* format, Args&&... args) {
|
||||
if (!verbose_parser_log) {
|
||||
@ -105,37 +107,89 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
void log(int log_level, const char* format) {
|
||||
if (!verbose_parser_log) {
|
||||
return;
|
||||
}
|
||||
if (log_level >= GLOBAL_LOG_LEVEL) {
|
||||
std::string curr_log = tstamp.get_time_stamp();
|
||||
if (log_level > LOG_INFO) {
|
||||
curr_log += log_level_ANSI_color(log_level);
|
||||
}
|
||||
std::cout << curr_log;
|
||||
puts(format);
|
||||
fflush(stdout);
|
||||
if (write_log) {
|
||||
fprintf(f, "%s", curr_log.c_str());
|
||||
fputs(format, f);
|
||||
fflush(f);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename... Args>
|
||||
void debug(const char* format, Args&&... args) {
|
||||
log(LOG_DEBUG, format, args...);
|
||||
if (sizeof...(args) != 0) {
|
||||
log(LOG_DEBUG, format, args...);
|
||||
} else {
|
||||
log(LOG_DEBUG, format);
|
||||
}
|
||||
};
|
||||
template <typename... Args>
|
||||
void verbose(const char* format, Args&&... args) {
|
||||
log(LOG_VERBOSE, format, args...);
|
||||
if (sizeof...(args) != 0) {
|
||||
log(LOG_VERBOSE, format, args...);
|
||||
} else {
|
||||
log(LOG_VERBOSE, format);
|
||||
}
|
||||
};
|
||||
template <typename... Args>
|
||||
void info(const char* format, Args&&... args) {
|
||||
log(LOG_INFO, format, args...);
|
||||
if (sizeof...(args) != 0) {
|
||||
log(LOG_INFO, format, args...);
|
||||
} else {
|
||||
log(LOG_INFO, format);
|
||||
}
|
||||
};
|
||||
template <typename... Args>
|
||||
void notice(const char* format, Args&&... args) {
|
||||
log(LOG_NOTICE, format, args...);
|
||||
if (sizeof...(args) != 0) {
|
||||
log(LOG_NOTICE, format, args...);
|
||||
} else {
|
||||
log(LOG_NOTICE, format);
|
||||
}
|
||||
};
|
||||
template <typename... Args>
|
||||
void warning(const char* format, Args&&... args) {
|
||||
log(LOG_WARN, format, args...);
|
||||
if (sizeof...(args) != 0) {
|
||||
log(LOG_WARN, format, args...);
|
||||
} else {
|
||||
log(LOG_WARN, format);
|
||||
}
|
||||
};
|
||||
template <typename... Args>
|
||||
void error(const char* format, Args&&... args) {
|
||||
log(LOG_ERROR, format, args...);
|
||||
if (sizeof...(args) != 0) {
|
||||
log(LOG_ERROR, format, args...);
|
||||
} else {
|
||||
log(LOG_ERROR, format);
|
||||
}
|
||||
};
|
||||
template <typename... Args>
|
||||
void fatal(const char* format, Args&&... args) {
|
||||
log(LOG_FATAL, format, args...);
|
||||
if (sizeof...(args) != 0) {
|
||||
log(LOG_FATAL, format, args...);
|
||||
} else {
|
||||
log(LOG_FATAL, format);
|
||||
}
|
||||
};
|
||||
template <typename... Args>
|
||||
void ok(const char* format, Args&&... args) {
|
||||
log(LOG_OK, format, args...);
|
||||
if (sizeof...(args) != 0) {
|
||||
log(LOG_OK, format, args...);
|
||||
} else {
|
||||
log(LOG_OK, format);
|
||||
}
|
||||
};
|
||||
};
|
||||
|
||||
|
||||
@ -1 +0,0 @@
|
||||
from . import *
|
||||
@ -9,10 +9,10 @@
|
||||
* except tiny modifications on preprocessing and postprocessing
|
||||
*/
|
||||
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
#include <float.h>
|
||||
#include <math.h>
|
||||
#include <torch/extension.h>
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
|
||||
#include "cuda_runtime.h"
|
||||
|
||||
@ -210,15 +210,15 @@ void dct2dPostprocessCudaLauncher(
|
||||
dim3 blockSize(TPB, TPB, 1);
|
||||
auto stream = at::cuda::getCurrentCUDAStream();
|
||||
dct2dPostprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>((ComplexType<T> *)x,
|
||||
y,
|
||||
M,
|
||||
N,
|
||||
M / 2,
|
||||
N / 2,
|
||||
(T)(2. / (M * N)),
|
||||
(T)(4. / (M * N)),
|
||||
(ComplexType<T> *)expkM,
|
||||
(ComplexType<T> *)expkN);
|
||||
y,
|
||||
M,
|
||||
N,
|
||||
M / 2,
|
||||
N / 2,
|
||||
(T)(2. / (M * N)),
|
||||
(T)(4. / (M * N)),
|
||||
(ComplexType<T> *)expkM,
|
||||
(ComplexType<T> *)expkN);
|
||||
}
|
||||
|
||||
// idct2_fft2
|
||||
@ -712,18 +712,12 @@ void dct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at
|
||||
auto N = x.size(-1);
|
||||
auto M = x.numel() / N;
|
||||
|
||||
AT_DISPATCH_ALL_TYPES(x.scalar_type(), "dct2_fft2_forward_cuda", [&] {
|
||||
dct2dPreprocessCudaLauncher<scalar_t>(x.data_ptr<scalar_t>(), out.data_ptr<scalar_t>(), M, N);
|
||||
dct2dPreprocessCudaLauncher<float>(x.data_ptr<float>(), out.data_ptr<float>(), M, N);
|
||||
|
||||
buf = at::view_as_real(at::fft_rfft2(out, c10::nullopt, {-2, -1}, "backward")).contiguous();
|
||||
buf = at::view_as_real(at::fft_rfft2(out, c10::nullopt, {-2, -1}, "backward")).contiguous();
|
||||
|
||||
dct2dPostprocessCudaLauncher<scalar_t>(buf.data_ptr<scalar_t>(),
|
||||
out.data_ptr<scalar_t>(),
|
||||
M,
|
||||
N,
|
||||
expkM.data_ptr<scalar_t>(),
|
||||
expkN.data_ptr<scalar_t>());
|
||||
});
|
||||
dct2dPostprocessCudaLauncher<float>(
|
||||
buf.data_ptr<float>(), out.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
|
||||
}
|
||||
|
||||
void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
|
||||
@ -737,18 +731,12 @@ void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a
|
||||
auto N = x.size(-1);
|
||||
auto M = x.numel() / N;
|
||||
|
||||
AT_DISPATCH_ALL_TYPES(x.scalar_type(), "idct2_fft2_forward_cuda", [&] {
|
||||
idct2_fft2PreprocessCudaLauncher<scalar_t>(x.data_ptr<scalar_t>(),
|
||||
buf.data_ptr<scalar_t>(),
|
||||
M,
|
||||
N,
|
||||
expkM.data_ptr<scalar_t>(),
|
||||
expkN.data_ptr<scalar_t>());
|
||||
idct2_fft2PreprocessCudaLauncher<float>(
|
||||
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
|
||||
|
||||
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
|
||||
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
|
||||
|
||||
idct2_fft2PostprocessCudaLauncher<scalar_t>(y.data_ptr<scalar_t>(), out.data_ptr<scalar_t>(), M, N);
|
||||
});
|
||||
idct2_fft2PostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
|
||||
}
|
||||
|
||||
void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
|
||||
@ -762,18 +750,12 @@ void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a
|
||||
auto N = x.size(-1);
|
||||
auto M = x.numel() / N;
|
||||
|
||||
AT_DISPATCH_ALL_TYPES(x.scalar_type(), "idct_idxst_forward_cuda", [&] {
|
||||
idct_idxstPreprocessCudaLauncher<scalar_t>(x.data_ptr<scalar_t>(),
|
||||
buf.data_ptr<scalar_t>(),
|
||||
M,
|
||||
N,
|
||||
expkM.data_ptr<scalar_t>(),
|
||||
expkN.data_ptr<scalar_t>());
|
||||
idct_idxstPreprocessCudaLauncher<float>(
|
||||
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
|
||||
|
||||
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
|
||||
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
|
||||
|
||||
idct_idxstPostprocessCudaLauncher<scalar_t>(y.data_ptr<scalar_t>(), out.data_ptr<scalar_t>(), M, N);
|
||||
});
|
||||
idct_idxstPostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
|
||||
}
|
||||
|
||||
void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
|
||||
@ -787,16 +769,10 @@ void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a
|
||||
auto N = x.size(-1);
|
||||
auto M = x.numel() / N;
|
||||
|
||||
AT_DISPATCH_ALL_TYPES(x.scalar_type(), "idxst_idct_forward_cuda", [&] {
|
||||
idxst_idctPreprocessCudaLauncher<scalar_t>(x.data_ptr<scalar_t>(),
|
||||
buf.data_ptr<scalar_t>(),
|
||||
M,
|
||||
N,
|
||||
expkM.data_ptr<scalar_t>(),
|
||||
expkN.data_ptr<scalar_t>());
|
||||
idxst_idctPreprocessCudaLauncher<float>(
|
||||
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
|
||||
|
||||
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
|
||||
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
|
||||
|
||||
idxst_idctPostprocessCudaLauncher<scalar_t>(y.data_ptr<scalar_t>(), out.data_ptr<scalar_t>(), M, N);
|
||||
});
|
||||
idxst_idctPostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
|
||||
}
|
||||
@ -15,7 +15,8 @@ torch::Tensor density_map_cuda_forward(torch::Tensor normalize_node_info,
|
||||
torch::Tensor aux_mat,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_nodes);
|
||||
int num_nodes,
|
||||
bool deterministic);
|
||||
|
||||
torch::Tensor density_map_cuda_backward(torch::Tensor normalize_node_info,
|
||||
torch::Tensor grad_mat,
|
||||
@ -24,7 +25,8 @@ torch::Tensor density_map_cuda_backward(torch::Tensor normalize_node_info,
|
||||
float grad_weight,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_nodes);
|
||||
int num_nodes,
|
||||
bool deterministic);
|
||||
|
||||
torch::Tensor density_map_cuda_forward_naive(torch::Tensor node_pos,
|
||||
torch::Tensor node_size,
|
||||
@ -37,7 +39,8 @@ torch::Tensor density_map_cuda_forward_naive(torch::Tensor node_pos,
|
||||
float min_node_w,
|
||||
float min_node_h,
|
||||
float margin,
|
||||
bool clamp_node);
|
||||
bool clamp_node,
|
||||
bool deterministic);
|
||||
|
||||
#define CHECK_CUDA(x) TORCH_CHECK(x.device().is_cuda(), #x " must be a CUDA tensor")
|
||||
#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
|
||||
@ -61,27 +64,22 @@ torch::Tensor density_map_normalize_node(torch::Tensor node_pos,
|
||||
CHECK_INPUT(unit_len);
|
||||
CHECK_INPUT(normalize_node_info);
|
||||
|
||||
return density_map_cuda_normalize_node(node_pos,
|
||||
node_size,
|
||||
node_weight,
|
||||
expand_ratio,
|
||||
unit_len,
|
||||
normalize_node_info,
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
num_nodes);
|
||||
return density_map_cuda_normalize_node(
|
||||
node_pos, node_size, node_weight, expand_ratio, unit_len, normalize_node_info, num_bin_x, num_bin_y, num_nodes);
|
||||
}
|
||||
torch::Tensor density_map_forward(torch::Tensor normalize_node_info,
|
||||
torch::Tensor sorted_node_map,
|
||||
torch::Tensor aux_mat,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_nodes) {
|
||||
int num_nodes,
|
||||
bool deterministic) {
|
||||
CHECK_INPUT(normalize_node_info);
|
||||
CHECK_INPUT(sorted_node_map);
|
||||
CHECK_INPUT(aux_mat);
|
||||
|
||||
return density_map_cuda_forward(normalize_node_info, sorted_node_map, aux_mat, num_bin_x, num_bin_y, num_nodes);
|
||||
return density_map_cuda_forward(
|
||||
normalize_node_info, sorted_node_map, aux_mat, num_bin_x, num_bin_y, num_nodes, deterministic);
|
||||
}
|
||||
|
||||
torch::Tensor density_map_backward(torch::Tensor normalize_node_info,
|
||||
@ -91,14 +89,22 @@ torch::Tensor density_map_backward(torch::Tensor normalize_node_info,
|
||||
float grad_weight,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_nodes) {
|
||||
int num_nodes,
|
||||
bool deterministic) {
|
||||
CHECK_INPUT(normalize_node_info);
|
||||
CHECK_INPUT(grad_mat);
|
||||
CHECK_INPUT(sorted_node_map);
|
||||
CHECK_INPUT(node_grad);
|
||||
|
||||
return density_map_cuda_backward(
|
||||
normalize_node_info, grad_mat, sorted_node_map, node_grad, grad_weight, num_bin_x, num_bin_y, num_nodes);
|
||||
return density_map_cuda_backward(normalize_node_info,
|
||||
grad_mat,
|
||||
sorted_node_map,
|
||||
node_grad,
|
||||
grad_weight,
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
num_nodes,
|
||||
deterministic);
|
||||
}
|
||||
|
||||
torch::Tensor density_map_forward_naive(torch::Tensor node_pos,
|
||||
@ -112,7 +118,8 @@ torch::Tensor density_map_forward_naive(torch::Tensor node_pos,
|
||||
float min_node_w,
|
||||
float min_node_h,
|
||||
float margin,
|
||||
bool clamp_node) {
|
||||
bool clamp_node,
|
||||
bool deterministic) {
|
||||
CHECK_INPUT(node_pos);
|
||||
CHECK_INPUT(node_size);
|
||||
CHECK_INPUT(node_weight);
|
||||
@ -130,12 +137,13 @@ torch::Tensor density_map_forward_naive(torch::Tensor node_pos,
|
||||
min_node_w,
|
||||
min_node_h,
|
||||
margin,
|
||||
clamp_node);
|
||||
clamp_node,
|
||||
deterministic);
|
||||
}
|
||||
|
||||
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
|
||||
m.def("pre_normalize", &density_map_normalize_node, "normalize bin size to 1");
|
||||
m.def("forward", &density_map_forward, "get density map from node information");
|
||||
m.def("forward_naive", &density_map_cuda_forward_naive, "calculate density map");
|
||||
m.def("forward_naive", &density_map_forward_naive, "calculate density map");
|
||||
m.def("backward", &density_map_backward, "calculate density gradient of each node");
|
||||
}
|
||||
@ -1,59 +1,58 @@
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
#include <cuda.h>
|
||||
#include <cuda_runtime.h>
|
||||
#include <torch/extension.h>
|
||||
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
#include <THC/THCAtomics.cuh>
|
||||
#include <vector>
|
||||
|
||||
template <typename scalar_t>
|
||||
__device__ scalar_t overlap(scalar_t x_l, scalar_t x_h, scalar_t bin_x_l) {
|
||||
template <typename T>
|
||||
__device__ T overlap(T x_l, T x_h, T bin_x_l) {
|
||||
// bin_x_h == bin_x_l + 1
|
||||
return min(x_h, bin_x_l + 1) - max(x_l, bin_x_l);
|
||||
}
|
||||
|
||||
template <typename scalar_t>
|
||||
__global__ void density_map_cuda_normalize_node_kernel(
|
||||
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_pos,
|
||||
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_size,
|
||||
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> node_weight,
|
||||
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> expand_ratio,
|
||||
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> unit_len,
|
||||
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> normalize_node_info,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_size,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> node_weight,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> expand_ratio,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> unit_len,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> normalize_node_info,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_nodes) {
|
||||
const int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < num_nodes) {
|
||||
// normalize bin_size_x and bin_size_y to 1
|
||||
normalize_node_info[i][0] = (node_pos[i][0] - node_size[i][0] / 2) / unit_len[0]; // x_l
|
||||
normalize_node_info[i][1] = (node_pos[i][0] + node_size[i][0] / 2) / unit_len[0]; // x_h
|
||||
normalize_node_info[i][2] = (node_pos[i][1] - node_size[i][1] / 2) / unit_len[1]; // y_l
|
||||
normalize_node_info[i][3] = (node_pos[i][1] + node_size[i][1] / 2) / unit_len[1]; // y_h
|
||||
normalize_node_info[i][4] = node_weight[i] * expand_ratio[i]; // weight
|
||||
if (normalize_node_info[i][1] - normalize_node_info[i][0] < 0 ||
|
||||
normalize_node_info[i][3] - normalize_node_info[i][2] < 0) {
|
||||
normalize_node_info[i][3] - normalize_node_info[i][2] < 0 ||
|
||||
(node_size[i][0] < 1e-6 && node_size[i][1] < 1e-6)) {
|
||||
normalize_node_info[i][4] = -normalize_node_info[i][4]; // we should ignore node whose weight <= 0
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename scalar_t>
|
||||
__global__ void __launch_bounds__(256, 4) density_map_cuda_forward_kernel(
|
||||
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> normalize_node_info,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> normalize_node_info,
|
||||
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> sorted_node_map,
|
||||
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> aux_mat,
|
||||
float *aux_mat,
|
||||
int num_nodes,
|
||||
int num_bin_x,
|
||||
int num_bin_y) {
|
||||
const int index = blockIdx.x * blockDim.z + threadIdx.z;
|
||||
if (index < num_nodes) {
|
||||
const int i = (sorted_node_map[index] >= 0) ? sorted_node_map[index] : index;
|
||||
const scalar_t weight = normalize_node_info[i][4];
|
||||
const float weight = normalize_node_info[i][4];
|
||||
if (weight > 0) {
|
||||
const scalar_t x_l = normalize_node_info[i][0];
|
||||
const scalar_t x_h = normalize_node_info[i][1];
|
||||
const scalar_t y_l = normalize_node_info[i][2];
|
||||
const scalar_t y_h = normalize_node_info[i][3];
|
||||
const float x_l = normalize_node_info[i][0];
|
||||
const float x_h = normalize_node_info[i][1];
|
||||
const float y_l = normalize_node_info[i][2];
|
||||
const float y_h = normalize_node_info[i][3];
|
||||
int x_lf = lround(floor(x_l));
|
||||
int x_hf = lround(floor(x_h));
|
||||
int y_lf = lround(floor(y_l));
|
||||
@ -64,25 +63,65 @@ __global__ void __launch_bounds__(256, 4) density_map_cuda_forward_kernel(
|
||||
y_hf = min(y_hf, num_bin_y - 1);
|
||||
|
||||
for (int j = x_lf + threadIdx.y; j < x_hf + 1; j += blockDim.y) {
|
||||
scalar_t bin_x_l = static_cast<scalar_t>(j);
|
||||
scalar_t overlap_x = overlap(x_l, x_h, bin_x_l);
|
||||
float bin_x_l = static_cast<float>(j);
|
||||
float overlap_x = overlap(x_l, x_h, bin_x_l);
|
||||
for (int k = y_lf + threadIdx.x; k < y_hf + 1; k += blockDim.x) {
|
||||
scalar_t bin_y_l = static_cast<scalar_t>(k);
|
||||
scalar_t overlap_y = overlap(y_l, y_h, bin_y_l);
|
||||
scalar_t overlap_area = overlap_x * overlap_y;
|
||||
gpuAtomicAdd(&aux_mat[j][k], weight * overlap_area);
|
||||
float bin_y_l = static_cast<float>(k);
|
||||
float overlap_y = overlap(y_l, y_h, bin_y_l);
|
||||
float overlap_area = overlap_x * overlap_y;
|
||||
atomicAdd(&aux_mat[j * num_bin_y + k], weight * overlap_area);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename scalar_t>
|
||||
__global__ void __launch_bounds__(256, 4) density_map_cuda_backward_kernel(
|
||||
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> normalize_node_info,
|
||||
const scalar_t *grad_mat,
|
||||
__global__ void __launch_bounds__(256, 4) density_map_cuda_deterministic_forward_kernel(
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> normalize_node_info,
|
||||
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> sorted_node_map,
|
||||
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_grad,
|
||||
unsigned long long *aux_mat,
|
||||
int num_nodes,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
unsigned long long scalar) {
|
||||
const int index = blockIdx.x * blockDim.z + threadIdx.z;
|
||||
if (index < num_nodes) {
|
||||
const int i = (sorted_node_map[index] >= 0) ? sorted_node_map[index] : index;
|
||||
const float weight = normalize_node_info[i][4];
|
||||
if (weight > 0) {
|
||||
const float x_l = normalize_node_info[i][0];
|
||||
const float x_h = normalize_node_info[i][1];
|
||||
const float y_l = normalize_node_info[i][2];
|
||||
const float y_h = normalize_node_info[i][3];
|
||||
int x_lf = lround(floor(x_l));
|
||||
int x_hf = lround(floor(x_h));
|
||||
int y_lf = lround(floor(y_l));
|
||||
int y_hf = lround(floor(y_h));
|
||||
x_lf = max(x_lf, 0);
|
||||
x_hf = min(x_hf, num_bin_x - 1);
|
||||
y_lf = max(y_lf, 0);
|
||||
y_hf = min(y_hf, num_bin_y - 1);
|
||||
|
||||
for (int j = x_lf + threadIdx.y; j < x_hf + 1; j += blockDim.y) {
|
||||
float bin_x_l = static_cast<float>(j);
|
||||
float overlap_x = overlap(x_l, x_h, bin_x_l);
|
||||
for (int k = y_lf + threadIdx.x; k < y_hf + 1; k += blockDim.x) {
|
||||
float bin_y_l = static_cast<float>(k);
|
||||
float overlap_y = overlap(y_l, y_h, bin_y_l);
|
||||
float overlap_area = overlap_x * overlap_y;
|
||||
atomicAdd(&aux_mat[j * num_bin_y + k],
|
||||
static_cast<unsigned long long>(weight * overlap_area * scalar));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void __launch_bounds__(256, 4) density_map_cuda_backward_kernel(
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> normalize_node_info,
|
||||
const float *grad_mat,
|
||||
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> sorted_node_map,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_grad,
|
||||
float grad_weight,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
@ -90,12 +129,12 @@ __global__ void __launch_bounds__(256, 4) density_map_cuda_backward_kernel(
|
||||
const int index = blockIdx.x * blockDim.z + threadIdx.z;
|
||||
if (index < num_nodes) {
|
||||
const int i = (sorted_node_map[index] >= 0) ? sorted_node_map[index] : index;
|
||||
const scalar_t weight = normalize_node_info[i][4];
|
||||
const float weight = normalize_node_info[i][4];
|
||||
if (weight > 0) {
|
||||
const scalar_t x_l = normalize_node_info[i][0];
|
||||
const scalar_t x_h = normalize_node_info[i][1];
|
||||
const scalar_t y_l = normalize_node_info[i][2];
|
||||
const scalar_t y_h = normalize_node_info[i][3];
|
||||
const float x_l = normalize_node_info[i][0];
|
||||
const float x_h = normalize_node_info[i][1];
|
||||
const float y_l = normalize_node_info[i][2];
|
||||
const float y_h = normalize_node_info[i][3];
|
||||
int x_lf = lround(floor(x_l));
|
||||
int x_hf = lround(floor(x_h));
|
||||
int y_lf = lround(floor(y_l));
|
||||
@ -106,33 +145,31 @@ __global__ void __launch_bounds__(256, 4) density_map_cuda_backward_kernel(
|
||||
y_hf = min(y_hf, num_bin_y - 1);
|
||||
|
||||
extern __shared__ unsigned char grad_xy[];
|
||||
scalar_t *grad_x = (scalar_t *)grad_xy;
|
||||
scalar_t *grad_y = grad_x + blockDim.z;
|
||||
float *grad_x = (float *)grad_xy;
|
||||
float *grad_y = grad_x + blockDim.z;
|
||||
if (threadIdx.x == 0 && threadIdx.y == 0) {
|
||||
grad_x[threadIdx.z] = grad_y[threadIdx.z] = 0;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
scalar_t part_grad_x = 0;
|
||||
scalar_t part_grad_y = 0;
|
||||
float part_grad_x = 0;
|
||||
float part_grad_y = 0;
|
||||
|
||||
for (int j = x_lf + threadIdx.y; j < x_hf + 1; j += blockDim.y) {
|
||||
scalar_t bin_x_l = static_cast<scalar_t>(j);
|
||||
scalar_t overlap_x = overlap(x_l, x_h, bin_x_l);
|
||||
float bin_x_l = static_cast<float>(j);
|
||||
float overlap_x = overlap(x_l, x_h, bin_x_l);
|
||||
for (int k = y_lf + threadIdx.x; k < y_hf + 1; k += blockDim.x) {
|
||||
scalar_t bin_y_l = static_cast<scalar_t>(k);
|
||||
scalar_t overlap_y = overlap(y_l, y_h, bin_y_l);
|
||||
scalar_t overlap_area = overlap_x * overlap_y;
|
||||
scalar_t tmp_x = grad_mat[0 * num_bin_x * num_bin_y + j * num_bin_y + k];
|
||||
scalar_t tmp_y = grad_mat[1 * num_bin_x * num_bin_y + j * num_bin_y + k];
|
||||
// part_grad_x += overlap_area * grad_mat[0][j][k];
|
||||
// part_grad_y += overlap_area * grad_mat[1][j][k];
|
||||
float bin_y_l = static_cast<float>(k);
|
||||
float overlap_y = overlap(y_l, y_h, bin_y_l);
|
||||
float overlap_area = overlap_x * overlap_y;
|
||||
float tmp_x = grad_mat[0 * num_bin_x * num_bin_y + j * num_bin_y + k];
|
||||
float tmp_y = grad_mat[1 * num_bin_x * num_bin_y + j * num_bin_y + k];
|
||||
part_grad_x += overlap_area * tmp_x;
|
||||
part_grad_y += overlap_area * tmp_y;
|
||||
}
|
||||
}
|
||||
gpuAtomicAdd(&grad_x[threadIdx.z], part_grad_x);
|
||||
gpuAtomicAdd(&grad_y[threadIdx.z], part_grad_y);
|
||||
atomicAdd(&grad_x[threadIdx.z], part_grad_x);
|
||||
atomicAdd(&grad_y[threadIdx.z], part_grad_y);
|
||||
__syncthreads();
|
||||
|
||||
if (threadIdx.x == 0 && threadIdx.y == 0) {
|
||||
@ -143,6 +180,68 @@ __global__ void __launch_bounds__(256, 4) density_map_cuda_backward_kernel(
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void density_map_cuda_deterministic_backward_kernel(
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> normalize_node_info,
|
||||
const float *grad_mat,
|
||||
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> sorted_node_map,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_grad,
|
||||
float grad_weight,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_nodes) {
|
||||
const int index = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (index < num_nodes) {
|
||||
const int i = (sorted_node_map[index] >= 0) ? sorted_node_map[index] : index;
|
||||
const float weight = normalize_node_info[i][4];
|
||||
if (weight > 0) {
|
||||
const float x_l = normalize_node_info[i][0];
|
||||
const float x_h = normalize_node_info[i][1];
|
||||
const float y_l = normalize_node_info[i][2];
|
||||
const float y_h = normalize_node_info[i][3];
|
||||
int x_lf = lround(floor(x_l));
|
||||
int x_hf = lround(floor(x_h));
|
||||
int y_lf = lround(floor(y_l));
|
||||
int y_hf = lround(floor(y_h));
|
||||
x_lf = max(x_lf, 0);
|
||||
x_hf = min(x_hf, num_bin_x - 1);
|
||||
y_lf = max(y_lf, 0);
|
||||
y_hf = min(y_hf, num_bin_y - 1);
|
||||
|
||||
float gradX = 0;
|
||||
float gradY = 0;
|
||||
for (int j = x_lf; j < x_hf + 1; j++) {
|
||||
float bin_x_l = static_cast<float>(j);
|
||||
float overlap_x = overlap(x_l, x_h, bin_x_l);
|
||||
for (int k = y_lf; k < y_hf + 1; k++) {
|
||||
float bin_y_l = static_cast<float>(k);
|
||||
float overlap_y = overlap(y_l, y_h, bin_y_l);
|
||||
float overlap_area = overlap_x * overlap_y;
|
||||
gradX += grad_mat[0 * num_bin_x * num_bin_y + j * num_bin_y + k] * overlap_area;
|
||||
gradY += grad_mat[1 * num_bin_x * num_bin_y + j * num_bin_y + k] * overlap_area;
|
||||
}
|
||||
}
|
||||
node_grad[i][0] = grad_weight * weight * gradX;
|
||||
node_grad[i][1] = grad_weight * weight * gradY;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void copyFromFloatAuxMat(
|
||||
unsigned long long *aux_mat_uint64, float *aux_mat, unsigned long long scalar, float inv_scalar, int num_bin) {
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < num_bin) {
|
||||
aux_mat_uint64[i] = static_cast<unsigned long long>(aux_mat[i] * scalar);
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void copyToFloatAuxMat(
|
||||
unsigned long long *aux_mat_uint64, float *aux_mat, unsigned long long scalar, float inv_scalar, int num_bin) {
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < num_bin) {
|
||||
aux_mat[i] = static_cast<float>(inv_scalar * aux_mat_uint64[i]);
|
||||
}
|
||||
}
|
||||
|
||||
torch::Tensor density_map_cuda_normalize_node(torch::Tensor node_pos,
|
||||
torch::Tensor node_size,
|
||||
torch::Tensor node_weight,
|
||||
@ -158,18 +257,16 @@ torch::Tensor density_map_cuda_normalize_node(torch::Tensor node_pos,
|
||||
const int threads = 128;
|
||||
const int blocks = (num_nodes + threads - 1) / threads;
|
||||
|
||||
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "density_map_cuda_normalize_node", ([&] {
|
||||
density_map_cuda_normalize_node_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
|
||||
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
|
||||
node_size.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
|
||||
node_weight.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
|
||||
expand_ratio.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
|
||||
unit_len.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
|
||||
normalize_node_info.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
num_nodes);
|
||||
}));
|
||||
density_map_cuda_normalize_node_kernel<<<blocks, threads, 0, stream>>>(
|
||||
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
node_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
node_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
expand_ratio.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
unit_len.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
normalize_node_info.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
num_nodes);
|
||||
|
||||
return normalize_node_info;
|
||||
}
|
||||
@ -179,7 +276,8 @@ torch::Tensor density_map_cuda_forward(torch::Tensor normalize_node_info,
|
||||
torch::Tensor aux_mat,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_nodes) {
|
||||
int num_nodes,
|
||||
bool deterministic) {
|
||||
cudaSetDevice(normalize_node_info.get_device());
|
||||
auto stream = at::cuda::getCurrentCUDAStream();
|
||||
|
||||
@ -187,15 +285,69 @@ torch::Tensor density_map_cuda_forward(torch::Tensor normalize_node_info,
|
||||
dim3 blockSize(2, 2, thread_count);
|
||||
int block_count = (num_nodes - 1 + thread_count) / thread_count;
|
||||
|
||||
AT_DISPATCH_ALL_TYPES(normalize_node_info.scalar_type(), "density_map_cuda_forward", ([&] {
|
||||
density_map_cuda_forward_kernel<scalar_t><<<block_count, blockSize, 0, stream>>>(
|
||||
normalize_node_info.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
|
||||
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
aux_mat.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
|
||||
num_nodes,
|
||||
num_bin_x,
|
||||
num_bin_y);
|
||||
}));
|
||||
if (deterministic) {
|
||||
// each bin size is pre-normalized to 1x1
|
||||
// max_value_bits -> #bits of the maximum density == (num_bin_x * num_bin_y)
|
||||
int max_value_bits = max(static_cast<int>(ceil(log2((num_bin_x + 0.1) * (num_bin_y + 0.1)))) + 1, 32);
|
||||
int scalar_bits = max(64 - max_value_bits, 0);
|
||||
unsigned long long scalar = (1UL << scalar_bits);
|
||||
float inv_scalar = 1.0 / static_cast<float>(scalar);
|
||||
int num_bin = num_bin_x * num_bin_y;
|
||||
|
||||
// use cache to save runtime
|
||||
int cp_threads = 512;
|
||||
int cp_blocks = (num_bin + cp_threads - 1) / cp_threads;
|
||||
static unsigned long long *aux_mat_uint64_ptr = nullptr;
|
||||
static int aux_mat_uint64_size = -1;
|
||||
if (aux_mat_uint64_ptr == nullptr) {
|
||||
aux_mat_uint64_size = num_bin;
|
||||
cudaMalloc(&aux_mat_uint64_ptr, aux_mat_uint64_size * sizeof(unsigned long long));
|
||||
} else if (num_bin != aux_mat_uint64_size) {
|
||||
cudaFree(aux_mat_uint64_ptr);
|
||||
aux_mat_uint64_ptr = nullptr;
|
||||
aux_mat_uint64_size = num_bin;
|
||||
cudaMalloc(&aux_mat_uint64_ptr, aux_mat_uint64_size * sizeof(unsigned long long));
|
||||
}
|
||||
copyFromFloatAuxMat<<<cp_blocks, cp_threads, 0, stream>>>(
|
||||
aux_mat_uint64_ptr, aux_mat.data_ptr<float>(), scalar, inv_scalar, num_bin);
|
||||
density_map_cuda_deterministic_forward_kernel<<<block_count, blockSize, 0, stream>>>(
|
||||
normalize_node_info.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
aux_mat_uint64_ptr,
|
||||
num_nodes,
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
scalar);
|
||||
copyToFloatAuxMat<<<cp_blocks, cp_threads, 0, stream>>>(
|
||||
aux_mat_uint64_ptr, aux_mat.data_ptr<float>(), scalar, inv_scalar, num_bin);
|
||||
|
||||
// without cache, need a lot of cudaMallocAsync...
|
||||
// int cp_threads = 512;
|
||||
// int cp_blocks = (num_bin + cp_threads - 1) / cp_threads;
|
||||
// unsigned long long *aux_mat_uint64 = nullptr;
|
||||
// cudaMallocAsync(&aux_mat_uint64, num_bin * sizeof(unsigned long long), stream);
|
||||
// copyFromFloatAuxMat<<<cp_blocks, cp_threads, 0, stream>>>(
|
||||
// aux_mat_uint64, aux_mat.data_ptr<float>(), scalar, inv_scalar, num_bin);
|
||||
// density_map_cuda_deterministic_forward_kernel<<<block_count, blockSize, 0, stream>>>(
|
||||
// normalize_node_info.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
// sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
// aux_mat_uint64,
|
||||
// num_nodes,
|
||||
// num_bin_x,
|
||||
// num_bin_y,
|
||||
// scalar);
|
||||
// copyToFloatAuxMat<<<cp_blocks, cp_threads, 0, stream>>>(
|
||||
// aux_mat_uint64, aux_mat.data_ptr<float>(), scalar, inv_scalar, num_bin);
|
||||
// cudaFreeAsync(aux_mat_uint64, stream);
|
||||
} else {
|
||||
density_map_cuda_forward_kernel<<<block_count, blockSize, 0, stream>>>(
|
||||
normalize_node_info.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
aux_mat.data_ptr<float>(),
|
||||
num_nodes,
|
||||
num_bin_x,
|
||||
num_bin_y);
|
||||
}
|
||||
|
||||
return aux_mat;
|
||||
}
|
||||
@ -207,27 +359,38 @@ torch::Tensor density_map_cuda_backward(torch::Tensor normalize_node_info,
|
||||
float grad_weight,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_nodes) {
|
||||
int num_nodes,
|
||||
bool deterministic) {
|
||||
cudaSetDevice(normalize_node_info.get_device());
|
||||
auto stream = at::cuda::getCurrentCUDAStream();
|
||||
|
||||
int thread_count = 64;
|
||||
dim3 blockSize(2, 2, thread_count);
|
||||
int block_count = (num_nodes - 1 + thread_count) / thread_count;
|
||||
|
||||
AT_DISPATCH_ALL_TYPES(normalize_node_info.scalar_type(), "density_map_cuda_backward", ([&] {
|
||||
size_t shared_mem_size = sizeof(scalar_t) * thread_count * 2;
|
||||
density_map_cuda_backward_kernel<scalar_t>
|
||||
<<<block_count, blockSize, shared_mem_size, stream>>>(
|
||||
normalize_node_info.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
|
||||
grad_mat.data_ptr<scalar_t>(),
|
||||
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
node_grad.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
|
||||
grad_weight,
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
num_nodes);
|
||||
}));
|
||||
if (deterministic) {
|
||||
int threads = 64;
|
||||
int blocks = (num_nodes + threads - 1) / threads;
|
||||
density_map_cuda_deterministic_backward_kernel<<<blocks, threads, 0, stream>>>(
|
||||
normalize_node_info.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
grad_mat.data_ptr<float>(),
|
||||
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
node_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
grad_weight,
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
num_nodes);
|
||||
} else {
|
||||
int thread_count = 64;
|
||||
dim3 blockSize(2, 2, thread_count);
|
||||
int block_count = (num_nodes - 1 + thread_count) / thread_count;
|
||||
size_t shared_mem_size = sizeof(float) * thread_count * 2;
|
||||
density_map_cuda_backward_kernel<<<block_count, blockSize, shared_mem_size, stream>>>(
|
||||
normalize_node_info.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
grad_mat.data_ptr<float>(),
|
||||
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
node_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
grad_weight,
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
num_nodes);
|
||||
}
|
||||
|
||||
return node_grad;
|
||||
}
|
||||
@ -1,18 +1,16 @@
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
#include <cuda.h>
|
||||
#include <cuda_runtime.h>
|
||||
#include <torch/extension.h>
|
||||
|
||||
#include <ATen/cuda/CUDAContext.h>
|
||||
#include <THC/THCAtomics.cuh>
|
||||
#include <vector>
|
||||
|
||||
template <typename scalar_t>
|
||||
__global__ void density_map_cuda_forward_naive_kernel(
|
||||
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_pos,
|
||||
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_size,
|
||||
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> node_weight,
|
||||
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> unit_len,
|
||||
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> aux_mat,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_size,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> node_weight,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> unit_len,
|
||||
float *aux_mat,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_nodes,
|
||||
@ -22,23 +20,23 @@ __global__ void density_map_cuda_forward_naive_kernel(
|
||||
bool clamp_node) {
|
||||
const int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < num_nodes) {
|
||||
scalar_t node_w = node_size[i][0];
|
||||
scalar_t node_h = node_size[i][1];
|
||||
scalar_t ratio = 1.0;
|
||||
float node_w = node_size[i][0];
|
||||
float node_h = node_size[i][1];
|
||||
float ratio = 1.0;
|
||||
if (clamp_node) {
|
||||
const scalar_t node_area = node_w * node_h;
|
||||
node_w = max(node_w, static_cast<scalar_t>(min_node_w));
|
||||
node_h = max(node_h, static_cast<scalar_t>(min_node_h));
|
||||
const float node_area = node_w * node_h;
|
||||
node_w = max(node_w, static_cast<float>(min_node_w));
|
||||
node_h = max(node_h, static_cast<float>(min_node_h));
|
||||
ratio = node_area / (node_w * node_h);
|
||||
}
|
||||
const scalar_t mgn = static_cast<scalar_t>(margin);
|
||||
const scalar_t num_bin_x_minus_mgn = static_cast<scalar_t>(num_bin_x) - mgn;
|
||||
const scalar_t num_bin_y_minus_mgn = static_cast<scalar_t>(num_bin_y) - mgn;
|
||||
const scalar_t small_mgn = static_cast<scalar_t>(margin * 0.1);
|
||||
scalar_t x_l = max((node_pos[i][0] - node_w / 2) / unit_len[0], mgn);
|
||||
scalar_t x_h = min((node_pos[i][0] + node_w / 2) / unit_len[0], num_bin_x_minus_mgn);
|
||||
scalar_t y_l = max((node_pos[i][1] - node_h / 2) / unit_len[1], mgn);
|
||||
scalar_t y_h = min((node_pos[i][1] + node_h / 2) / unit_len[1], num_bin_y_minus_mgn);
|
||||
const float mgn = static_cast<float>(margin);
|
||||
const float num_bin_x_minus_mgn = static_cast<float>(num_bin_x) - mgn;
|
||||
const float num_bin_y_minus_mgn = static_cast<float>(num_bin_y) - mgn;
|
||||
const float small_mgn = static_cast<float>(margin * 0.1);
|
||||
float x_l = max((node_pos[i][0] - node_w / 2) / unit_len[0], mgn);
|
||||
float x_h = min((node_pos[i][0] + node_w / 2) / unit_len[0], num_bin_x_minus_mgn);
|
||||
float y_l = max((node_pos[i][1] - node_h / 2) / unit_len[1], mgn);
|
||||
float y_h = min((node_pos[i][1] + node_h / 2) / unit_len[1], num_bin_y_minus_mgn);
|
||||
x_l = min(x_l, num_bin_x_minus_mgn);
|
||||
x_h = max(x_h, mgn);
|
||||
y_l = min(y_l, num_bin_y_minus_mgn);
|
||||
@ -46,7 +44,7 @@ __global__ void density_map_cuda_forward_naive_kernel(
|
||||
if (x_h - x_l < small_mgn || y_h - y_l < small_mgn) {
|
||||
return;
|
||||
}
|
||||
const scalar_t p_node_wght = node_weight[i] * ratio;
|
||||
const float p_node_wght = node_weight[i] * ratio;
|
||||
|
||||
const int x_lf = lround(floor(x_l));
|
||||
const int x_hf = lround(floor(x_h));
|
||||
@ -54,28 +52,90 @@ __global__ void density_map_cuda_forward_naive_kernel(
|
||||
const int y_hf = lround(floor(y_h));
|
||||
|
||||
for (int j = x_lf; j < x_hf + 1; j++) {
|
||||
const scalar_t bin_x_l = j;
|
||||
const scalar_t bin_x_h = j + 1;
|
||||
scalar_t overlap_x = min(x_h, bin_x_h) - max(x_l, bin_x_l);
|
||||
const float bin_x_l = j;
|
||||
const float bin_x_h = j + 1;
|
||||
float overlap_x = min(x_h, bin_x_h) - max(x_l, bin_x_l);
|
||||
for (int k = y_lf; k < y_hf + 1; k++) {
|
||||
const scalar_t bin_y_l = k;
|
||||
const scalar_t bin_y_h = k + 1;
|
||||
scalar_t overlap_y = min(y_h, bin_y_h) - max(y_l, bin_y_l);
|
||||
scalar_t overlap_area = overlap_x * overlap_y;
|
||||
gpuAtomicAdd(&aux_mat[j][k], p_node_wght * overlap_area);
|
||||
const float bin_y_l = k;
|
||||
const float bin_y_h = k + 1;
|
||||
float overlap_y = min(y_h, bin_y_h) - max(y_l, bin_y_l);
|
||||
float overlap_area = overlap_x * overlap_y;
|
||||
atomicAdd(&aux_mat[j * num_bin_y + k], p_node_wght * overlap_area);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void density_map_cuda_deterministic_forward_naive_kernel(
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_size,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> node_weight,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> unit_len,
|
||||
unsigned long long *aux_mat,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_nodes,
|
||||
float min_node_w,
|
||||
float min_node_h,
|
||||
float margin,
|
||||
bool clamp_node,
|
||||
unsigned long long scalar) {
|
||||
const int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < num_nodes) {
|
||||
float node_w = node_size[i][0];
|
||||
float node_h = node_size[i][1];
|
||||
float ratio = 1.0;
|
||||
if (clamp_node) {
|
||||
const float node_area = node_w * node_h;
|
||||
node_w = max(node_w, static_cast<float>(min_node_w));
|
||||
node_h = max(node_h, static_cast<float>(min_node_h));
|
||||
ratio = node_area / (node_w * node_h);
|
||||
}
|
||||
const float mgn = static_cast<float>(margin);
|
||||
const float num_bin_x_minus_mgn = static_cast<float>(num_bin_x) - mgn;
|
||||
const float num_bin_y_minus_mgn = static_cast<float>(num_bin_y) - mgn;
|
||||
const float small_mgn = static_cast<float>(margin * 0.1);
|
||||
float x_l = max((node_pos[i][0] - node_w / 2) / unit_len[0], mgn);
|
||||
float x_h = min((node_pos[i][0] + node_w / 2) / unit_len[0], num_bin_x_minus_mgn);
|
||||
float y_l = max((node_pos[i][1] - node_h / 2) / unit_len[1], mgn);
|
||||
float y_h = min((node_pos[i][1] + node_h / 2) / unit_len[1], num_bin_y_minus_mgn);
|
||||
x_l = min(x_l, num_bin_x_minus_mgn);
|
||||
x_h = max(x_h, mgn);
|
||||
y_l = min(y_l, num_bin_y_minus_mgn);
|
||||
y_h = max(y_h, mgn);
|
||||
if (x_h - x_l < small_mgn || y_h - y_l < small_mgn) {
|
||||
return;
|
||||
}
|
||||
const float p_node_wght = node_weight[i] * ratio;
|
||||
|
||||
const int x_lf = lround(floor(x_l));
|
||||
const int x_hf = lround(floor(x_h));
|
||||
const int y_lf = lround(floor(y_l));
|
||||
const int y_hf = lround(floor(y_h));
|
||||
|
||||
for (int j = x_lf; j < x_hf + 1; j++) {
|
||||
const float bin_x_l = j;
|
||||
const float bin_x_h = j + 1;
|
||||
float overlap_x = min(x_h, bin_x_h) - max(x_l, bin_x_l);
|
||||
for (int k = y_lf; k < y_hf + 1; k++) {
|
||||
const float bin_y_l = k;
|
||||
const float bin_y_h = k + 1;
|
||||
float overlap_y = min(y_h, bin_y_h) - max(y_l, bin_y_l);
|
||||
float overlap_area = overlap_x * overlap_y;
|
||||
atomicAdd(&aux_mat[j * num_bin_y + k],
|
||||
static_cast<unsigned long long>(p_node_wght * overlap_area * scalar));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename scalar_t>
|
||||
__global__ void density_map_cuda_backward_naive_kernel(
|
||||
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_pos,
|
||||
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_size,
|
||||
const torch::PackedTensorAccessor32<scalar_t, 3, torch::RestrictPtrTraits> grad_mat,
|
||||
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> unit_len,
|
||||
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> node_weight,
|
||||
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_grad,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_size,
|
||||
const torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> grad_mat,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> unit_len,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> node_weight,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_grad,
|
||||
float grad_weight,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
@ -86,23 +146,23 @@ __global__ void density_map_cuda_backward_naive_kernel(
|
||||
bool clamp_node) {
|
||||
const int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < num_nodes) {
|
||||
scalar_t node_w = node_size[i][0];
|
||||
scalar_t node_h = node_size[i][1];
|
||||
scalar_t ratio = 1.0;
|
||||
float node_w = node_size[i][0];
|
||||
float node_h = node_size[i][1];
|
||||
float ratio = 1.0;
|
||||
if (clamp_node) {
|
||||
const scalar_t node_area = node_w * node_h;
|
||||
node_w = max(node_w, static_cast<scalar_t>(min_node_w));
|
||||
node_h = max(node_h, static_cast<scalar_t>(min_node_h));
|
||||
const float node_area = node_w * node_h;
|
||||
node_w = max(node_w, static_cast<float>(min_node_w));
|
||||
node_h = max(node_h, static_cast<float>(min_node_h));
|
||||
ratio = node_area / (node_w * node_h);
|
||||
}
|
||||
const scalar_t mgn = static_cast<scalar_t>(margin);
|
||||
const scalar_t num_bin_x_minus_mgn = static_cast<scalar_t>(num_bin_x) - mgn;
|
||||
const scalar_t num_bin_y_minus_mgn = static_cast<scalar_t>(num_bin_y) - mgn;
|
||||
const scalar_t small_mgn = static_cast<scalar_t>(margin * 0.1);
|
||||
scalar_t x_l = max((node_pos[i][0] - node_w / 2) / unit_len[0], mgn);
|
||||
scalar_t x_h = min((node_pos[i][0] + node_w / 2) / unit_len[0], num_bin_x_minus_mgn);
|
||||
scalar_t y_l = max((node_pos[i][1] - node_h / 2) / unit_len[1], mgn);
|
||||
scalar_t y_h = min((node_pos[i][1] + node_h / 2) / unit_len[1], num_bin_y_minus_mgn);
|
||||
const float mgn = static_cast<float>(margin);
|
||||
const float num_bin_x_minus_mgn = static_cast<float>(num_bin_x) - mgn;
|
||||
const float num_bin_y_minus_mgn = static_cast<float>(num_bin_y) - mgn;
|
||||
const float small_mgn = static_cast<float>(margin * 0.1);
|
||||
float x_l = max((node_pos[i][0] - node_w / 2) / unit_len[0], mgn);
|
||||
float x_h = min((node_pos[i][0] + node_w / 2) / unit_len[0], num_bin_x_minus_mgn);
|
||||
float y_l = max((node_pos[i][1] - node_h / 2) / unit_len[1], mgn);
|
||||
float y_h = min((node_pos[i][1] + node_h / 2) / unit_len[1], num_bin_y_minus_mgn);
|
||||
x_l = min(x_l, num_bin_x_minus_mgn);
|
||||
x_h = max(x_h, mgn);
|
||||
y_l = min(y_l, num_bin_y_minus_mgn);
|
||||
@ -116,17 +176,17 @@ __global__ void density_map_cuda_backward_naive_kernel(
|
||||
const int y_lf = lround(floor(y_l));
|
||||
const int y_hf = lround(floor(y_h));
|
||||
|
||||
scalar_t gradX = 0;
|
||||
scalar_t gradY = 0;
|
||||
float gradX = 0;
|
||||
float gradY = 0;
|
||||
for (int j = x_lf; j < x_hf + 1; j++) {
|
||||
const scalar_t bin_x_l = j;
|
||||
const scalar_t bin_x_h = j + 1;
|
||||
scalar_t overlap_x = min(x_h, bin_x_h) - max(x_l, bin_x_l);
|
||||
const float bin_x_l = j;
|
||||
const float bin_x_h = j + 1;
|
||||
float overlap_x = min(x_h, bin_x_h) - max(x_l, bin_x_l);
|
||||
for (int k = y_lf; k < y_hf + 1; k++) {
|
||||
const scalar_t bin_y_l = k;
|
||||
const scalar_t bin_y_h = k + 1;
|
||||
scalar_t overlap_y = min(y_h, bin_y_h) - max(y_l, bin_y_l);
|
||||
scalar_t overlap_area = overlap_x * overlap_y;
|
||||
const float bin_y_l = k;
|
||||
const float bin_y_h = k + 1;
|
||||
float overlap_y = min(y_h, bin_y_h) - max(y_l, bin_y_l);
|
||||
float overlap_area = overlap_x * overlap_y;
|
||||
gradX += grad_mat[0][j][k] * overlap_area;
|
||||
gradY += grad_mat[1][j][k] * overlap_area;
|
||||
}
|
||||
@ -136,6 +196,22 @@ __global__ void density_map_cuda_backward_naive_kernel(
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void copyFromFloatAuxMat2(
|
||||
unsigned long long *aux_mat_uint64, float *aux_mat, unsigned long long scalar, float inv_scalar, int num_bin) {
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < num_bin) {
|
||||
aux_mat_uint64[i] = static_cast<unsigned long long>(aux_mat[i] * scalar);
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void copyToFloatAuxMat2(
|
||||
unsigned long long *aux_mat_uint64, float *aux_mat, unsigned long long scalar, float inv_scalar, int num_bin) {
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < num_bin) {
|
||||
aux_mat[i] = static_cast<float>(inv_scalar * aux_mat_uint64[i]);
|
||||
}
|
||||
}
|
||||
|
||||
torch::Tensor density_map_cuda_forward_naive(torch::Tensor node_pos,
|
||||
torch::Tensor node_size,
|
||||
torch::Tensor node_weight,
|
||||
@ -147,28 +223,61 @@ torch::Tensor density_map_cuda_forward_naive(torch::Tensor node_pos,
|
||||
float min_node_w,
|
||||
float min_node_h,
|
||||
float margin,
|
||||
bool clamp_node) {
|
||||
bool clamp_node,
|
||||
bool deterministic) {
|
||||
cudaSetDevice(node_pos.get_device());
|
||||
auto stream = at::cuda::getCurrentCUDAStream();
|
||||
|
||||
const int threads = 64;
|
||||
const int blocks = (num_nodes + threads - 1) / threads;
|
||||
|
||||
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "density_map_cuda_forward_naive", ([&] {
|
||||
density_map_cuda_forward_naive_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
|
||||
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
|
||||
node_size.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
|
||||
node_weight.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
|
||||
unit_len.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
|
||||
aux_mat.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
num_nodes,
|
||||
min_node_w,
|
||||
min_node_h,
|
||||
margin,
|
||||
clamp_node);
|
||||
}));
|
||||
if (deterministic) {
|
||||
// each bin size is internally normalized to 1x1
|
||||
// max_value_bits -> #bits of the maximum density == (num_bin_x * num_bin_y)
|
||||
int max_value_bits = max(static_cast<int>(ceil(log2((num_bin_x + 0.1) * (num_bin_y + 0.1)))) + 1, 32);
|
||||
int scalar_bits = max(64 - max_value_bits, 0);
|
||||
unsigned long long scalar = (1UL << scalar_bits);
|
||||
float inv_scalar = 1.0 / static_cast<float>(scalar);
|
||||
int num_bin = num_bin_x * num_bin_y;
|
||||
|
||||
int cp_threads = 512;
|
||||
int cp_blocks = (num_bin + cp_threads - 1) / cp_threads;
|
||||
unsigned long long *aux_mat_uint64 = nullptr;
|
||||
cudaMallocAsync(&aux_mat_uint64, num_bin * sizeof(unsigned long long), stream);
|
||||
copyFromFloatAuxMat2<<<cp_blocks, cp_threads, 0, stream>>>(
|
||||
aux_mat_uint64, aux_mat.data_ptr<float>(), scalar, inv_scalar, num_bin);
|
||||
density_map_cuda_deterministic_forward_naive_kernel<<<blocks, threads, 0, stream>>>(
|
||||
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
node_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
node_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
unit_len.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
aux_mat_uint64,
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
num_nodes,
|
||||
min_node_w,
|
||||
min_node_h,
|
||||
margin,
|
||||
clamp_node,
|
||||
scalar);
|
||||
copyToFloatAuxMat2<<<cp_blocks, cp_threads, 0, stream>>>(
|
||||
aux_mat_uint64, aux_mat.data_ptr<float>(), scalar, inv_scalar, num_bin);
|
||||
cudaFreeAsync(aux_mat_uint64, stream);
|
||||
} else {
|
||||
density_map_cuda_forward_naive_kernel<<<blocks, threads, 0, stream>>>(
|
||||
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
node_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
node_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
unit_len.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
aux_mat.data_ptr<float>(),
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
num_nodes,
|
||||
min_node_w,
|
||||
min_node_h,
|
||||
margin,
|
||||
clamp_node);
|
||||
}
|
||||
|
||||
return aux_mat;
|
||||
}
|
||||
@ -186,30 +295,29 @@ torch::Tensor density_map_cuda_backward(torch::Tensor node_pos,
|
||||
float min_node_w,
|
||||
float min_node_h,
|
||||
float margin,
|
||||
bool clamp_node) {
|
||||
bool clamp_node,
|
||||
bool deterministic) {
|
||||
cudaSetDevice(node_pos.get_device());
|
||||
auto stream = at::cuda::getCurrentCUDAStream();
|
||||
|
||||
const int threads = 64;
|
||||
const int blocks = (num_nodes + threads - 1) / threads;
|
||||
|
||||
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "density_map_cuda_backward_naive", ([&] {
|
||||
density_map_cuda_backward_naive_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
|
||||
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
|
||||
node_size.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
|
||||
grad_mat.packed_accessor32<scalar_t, 3, torch::RestrictPtrTraits>(),
|
||||
unit_len.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
|
||||
node_weight.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
|
||||
node_grad.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
|
||||
grad_weight,
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
num_nodes,
|
||||
min_node_w,
|
||||
min_node_h,
|
||||
margin,
|
||||
clamp_node);
|
||||
}));
|
||||
density_map_cuda_backward_naive_kernel<<<blocks, threads, 0, stream>>>(
|
||||
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
node_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
grad_mat.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
|
||||
unit_len.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
node_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
node_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
grad_weight,
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
num_nodes,
|
||||
min_node_w,
|
||||
min_node_h,
|
||||
margin,
|
||||
clamp_node);
|
||||
|
||||
return node_grad;
|
||||
}
|
||||
|
||||
37
cpp_to_py/gpudp/CMakeLists.txt
Normal file
37
cpp_to_py/gpudp/CMakeLists.txt
Normal file
@ -0,0 +1,37 @@
|
||||
# Files
|
||||
file(GLOB_RECURSE SRC_FILES_DP ${CMAKE_CURRENT_SOURCE_DIR}/check/*.cpp
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/dp/*.cpp
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/db/*.cpp
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/lg/*.cpp)
|
||||
file(GLOB_RECURSE SRC_FILES_DP_CUDA ${CMAKE_CURRENT_SOURCE_DIR}/*.cu)
|
||||
|
||||
# OpenMP
|
||||
find_package(OpenMP REQUIRED)
|
||||
|
||||
# CUDA DP Kernel
|
||||
cuda_add_library(dp_cuda_tmp STATIC ${SRC_FILES_DP_CUDA})
|
||||
|
||||
set_target_properties(dp_cuda_tmp PROPERTIES CUDA_RESOLVE_DEVICE_SYMBOLS ON)
|
||||
set_target_properties(dp_cuda_tmp PROPERTIES CUDA_SEPARABLE_COMPILATION ON)
|
||||
set_target_properties(dp_cuda_tmp PROPERTIES POSITION_INDEPENDENT_CODE ON)
|
||||
|
||||
target_include_directories(dp_cuda_tmp PRIVATE ${PROJECT_SOURCE_DIR}/cpp_to_py ${TORCH_INCLUDE_DIRS} ${FLUTE_INCLUDE_DIR})
|
||||
target_link_libraries(dp_cuda_tmp torch ${TORCH_PYTHON_LIBRARY} xplace_common flute OpenMP::OpenMP_CXX)
|
||||
|
||||
# CPU DP object
|
||||
add_library(dp SHARED ${CMAKE_CURRENT_SOURCE_DIR}/../io_parser/gp/GPDatabase.cpp
|
||||
${SRC_FILES_DP}
|
||||
${SRC_FILES_DP_CUDA})
|
||||
|
||||
target_include_directories(dp PRIVATE ${PROJECT_SOURCE_DIR}/cpp_to_py ${TORCH_INCLUDE_DIRS} ${FLUTE_INCLUDE_DIR})
|
||||
target_link_libraries(dp PRIVATE torch ${TORCH_PYTHON_LIBRARY} xplace_common flute dp_cuda_tmp pthread)
|
||||
target_compile_options(dp PRIVATE -fPIC)
|
||||
|
||||
install(TARGETS dp DESTINATION ${XPLACE_LIB_DIR})
|
||||
|
||||
# For Pybind
|
||||
add_pytorch_extension(gpudp PyBindCppMain.cpp
|
||||
EXTRA_INCLUDE_DIRS ${PROJECT_SOURCE_DIR}/cpp_to_py ${FLUTE_INCLUDE_DIR}
|
||||
EXTRA_LINK_LIBRARIES xplace_common flute io_parser dp)
|
||||
|
||||
install(TARGETS gpudp DESTINATION ${XPLACE_LIB_DIR})
|
||||
119
cpp_to_py/gpudp/PyBindCppMain.cpp
Normal file
119
cpp_to_py/gpudp/PyBindCppMain.cpp
Normal file
@ -0,0 +1,119 @@
|
||||
#include "common/common.h"
|
||||
#include "common/db/Database.h"
|
||||
#include "gpudp/db/dp_torch.h"
|
||||
|
||||
namespace Xplace {
|
||||
|
||||
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
|
||||
pybind11::class_<dp::DPTorchRawDB, std::shared_ptr<dp::DPTorchRawDB>>(m, "DPTorchRawDB")
|
||||
.def(pybind11::init<torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
float,
|
||||
float,
|
||||
float,
|
||||
float,
|
||||
int,
|
||||
int,
|
||||
float,
|
||||
float>())
|
||||
.def("check", &dp::DPTorchRawDB::check)
|
||||
.def("scale", &dp::DPTorchRawDB::scale)
|
||||
.def("commit", &dp::DPTorchRawDB::commit)
|
||||
.def("rollback", &dp::DPTorchRawDB::rollback)
|
||||
.def("commit_from", &dp::DPTorchRawDB::commit_from)
|
||||
.def("get_curr_cposx", &dp::DPTorchRawDB::get_curr_cposx, py::return_value_policy::move)
|
||||
.def("get_curr_cposy", &dp::DPTorchRawDB::get_curr_cposy, py::return_value_policy::move)
|
||||
.def("get_curr_lposx", &dp::DPTorchRawDB::get_curr_lposx, py::return_value_policy::move)
|
||||
.def("get_curr_lposy", &dp::DPTorchRawDB::get_curr_lposy, py::return_value_policy::move);
|
||||
|
||||
m.def("create_dp_rawdb",
|
||||
[](torch::Tensor node_lpos_init_,
|
||||
torch::Tensor node_size_,
|
||||
torch::Tensor node_weight_,
|
||||
torch::Tensor pin_rel_lpos_,
|
||||
torch::Tensor pin_id2node_id_,
|
||||
torch::Tensor pin_id2net_id_,
|
||||
torch::Tensor node2pin_list_,
|
||||
torch::Tensor node2pin_list_end_,
|
||||
torch::Tensor hyperedge_list_,
|
||||
torch::Tensor hyperedge_list_end_,
|
||||
torch::Tensor net_mask_,
|
||||
torch::Tensor node_id2region_id_,
|
||||
torch::Tensor region_boxes_,
|
||||
torch::Tensor region_boxes_end_,
|
||||
float xl_,
|
||||
float xh_,
|
||||
float yl_,
|
||||
float yh_,
|
||||
int num_movable_nodes_,
|
||||
int num_nodes_,
|
||||
float site_width_,
|
||||
float row_height_) {
|
||||
return std::make_shared<dp::DPTorchRawDB>(node_lpos_init_,
|
||||
node_size_,
|
||||
node_weight_,
|
||||
pin_rel_lpos_,
|
||||
pin_id2node_id_,
|
||||
pin_id2net_id_,
|
||||
node2pin_list_,
|
||||
node2pin_list_end_,
|
||||
hyperedge_list_,
|
||||
hyperedge_list_end_,
|
||||
net_mask_,
|
||||
node_id2region_id_,
|
||||
region_boxes_,
|
||||
region_boxes_end_,
|
||||
xl_,
|
||||
xh_,
|
||||
yl_,
|
||||
yh_,
|
||||
num_movable_nodes_,
|
||||
num_nodes_,
|
||||
site_width_,
|
||||
row_height_);
|
||||
});
|
||||
m.def("macroLegalization", [](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr, int num_bins_x, int num_bins_y) {
|
||||
return dp::macroLegalization(*at_db_ptr, num_bins_x, num_bins_y);
|
||||
});
|
||||
m.def("abacusLegalization", [](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr, int num_bins_x, int num_bins_y) {
|
||||
return dp::abacusLegalization(*at_db_ptr, num_bins_x, num_bins_y);
|
||||
});
|
||||
m.def("greedyLegalization", [](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr, int num_bins_x, int num_bins_y) {
|
||||
return dp::greedyLegalization(*at_db_ptr, num_bins_x, num_bins_y);
|
||||
});
|
||||
m.def("kReorder",
|
||||
[](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr, int num_bins_x, int num_bins_y, int K, int max_iters) {
|
||||
return dp::kReorder(*at_db_ptr, num_bins_x, num_bins_y, K, max_iters);
|
||||
});
|
||||
m.def(
|
||||
"globalSwap",
|
||||
[](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr, int num_bins_x, int num_bins_y, int batch_size, int max_iters) {
|
||||
return dp::globalSwap(*at_db_ptr, num_bins_x, num_bins_y, batch_size, max_iters);
|
||||
});
|
||||
m.def("independentSetMatching",
|
||||
[](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr,
|
||||
int num_bins_x,
|
||||
int num_bins_y,
|
||||
int batch_size,
|
||||
int set_size,
|
||||
int max_iters) {
|
||||
return dp::independentSetMatching(*at_db_ptr, num_bins_x, num_bins_y, batch_size, set_size, max_iters);
|
||||
});
|
||||
m.def("legalityCheck", [](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr, float scale_factor) {
|
||||
return dp::legalityCheck(*at_db_ptr, scale_factor);
|
||||
});
|
||||
}
|
||||
|
||||
} // namespace Xplace
|
||||
1
cpp_to_py/gpudp/README.md
Normal file
1
cpp_to_py/gpudp/README.md
Normal file
@ -0,0 +1 @@
|
||||
This GPU-accelerated detailed placer is adapted from [ABCDPlace](https://ieeexplore.ieee.org/document/8982049).
|
||||
383
cpp_to_py/gpudp/check/legality_check.cpp
Normal file
383
cpp_to_py/gpudp/check/legality_check.cpp
Normal file
@ -0,0 +1,383 @@
|
||||
#include "common/common.h"
|
||||
#include "gpudp/db/dp_torch.h"
|
||||
#include "gpudp/lg/legalization_db.h"
|
||||
|
||||
namespace dp {
|
||||
|
||||
bool boundaryCheck(const float* x,
|
||||
const float* y,
|
||||
const float* node_size_x,
|
||||
const float* node_size_y,
|
||||
const float scale_factor,
|
||||
float xl,
|
||||
float yl,
|
||||
float xh,
|
||||
float yh,
|
||||
int num_movable_nodes) {
|
||||
// use scale factor to control the precision
|
||||
float precision = (scale_factor == 1.0) ? 1e-6 : scale_factor * 0.1;
|
||||
bool legal_flag = true;
|
||||
// check node within boundary
|
||||
for (int i = 0; i < num_movable_nodes; ++i) {
|
||||
float node_xl = x[i];
|
||||
float node_yl = y[i];
|
||||
float node_xh = node_xl + node_size_x[i];
|
||||
float node_yh = node_yl + node_size_y[i];
|
||||
if (node_xl + precision < xl || node_xh > xh + precision || node_yl + precision < yl ||
|
||||
node_yh > yh + precision) {
|
||||
logger.error("node %d (%g, %g, %g, %g) out of boundary\n", i, node_xl, node_yl, node_xh, node_yh);
|
||||
legal_flag = false;
|
||||
}
|
||||
}
|
||||
return legal_flag;
|
||||
}
|
||||
|
||||
bool siteAlignmentCheck(const float* x,
|
||||
const float* y,
|
||||
const float site_width,
|
||||
const float row_height,
|
||||
const float scale_factor,
|
||||
const float xl,
|
||||
const float yl,
|
||||
int num_movable_nodes) {
|
||||
// use scale factor to control the precision
|
||||
float precision = (scale_factor == 1.0) ? 1e-6 : scale_factor * 0.1;
|
||||
bool legal_flag = true;
|
||||
// check row and site alignment
|
||||
for (int i = 0; i < num_movable_nodes; ++i) {
|
||||
float node_xl = x[i];
|
||||
float node_yl = y[i];
|
||||
|
||||
float row_id_f = (node_yl - yl) / row_height;
|
||||
int row_id = floorDiv(node_yl - yl, row_height);
|
||||
float row_yl = yl + row_height * row_id;
|
||||
float row_yh = row_yl + row_height;
|
||||
|
||||
if (std::abs(row_id_f - row_id) > precision) {
|
||||
logger.error("node %d (%g, %g) failed to align to row %d (%g, %g), gap %g, yl %g, row_height %g",
|
||||
i,
|
||||
node_xl,
|
||||
node_yl,
|
||||
row_id,
|
||||
row_yl,
|
||||
row_yh,
|
||||
std::abs(node_yl - row_yl),
|
||||
yl,
|
||||
row_height);
|
||||
legal_flag = false;
|
||||
}
|
||||
|
||||
float site_id_f = (node_xl - xl) / site_width;
|
||||
int site_id = floorDiv(node_xl - xl, site_width);
|
||||
if (std::abs(site_id_f - site_id) > precision) {
|
||||
logger.error("node %d (%g, %g) failed to align to row %d (%g, %g) and site; xl %g, site_width %g",
|
||||
i,
|
||||
node_xl,
|
||||
node_yl,
|
||||
row_id,
|
||||
row_yl,
|
||||
row_yh,
|
||||
xl,
|
||||
site_width);
|
||||
legal_flag = false;
|
||||
}
|
||||
}
|
||||
|
||||
return legal_flag;
|
||||
}
|
||||
|
||||
bool fenceRegionCheck(const float* x,
|
||||
const float* y,
|
||||
const float* node_size_x,
|
||||
const float* node_size_y,
|
||||
const float* flat_region_boxes,
|
||||
const int* flat_region_boxes_start,
|
||||
const int* node2fence_region_map,
|
||||
int num_movable_nodes,
|
||||
int num_regions) {
|
||||
bool legal_flag = true;
|
||||
// check fence regions
|
||||
for (int i = 0; i < num_movable_nodes; ++i) {
|
||||
float node_xl = x[i];
|
||||
float node_yl = y[i];
|
||||
float node_xh = node_xl + node_size_x[i];
|
||||
float node_yh = node_yl + node_size_y[i];
|
||||
|
||||
int region_id = node2fence_region_map[i];
|
||||
if (region_id < num_regions) {
|
||||
int box_bgn = flat_region_boxes_start[region_id];
|
||||
int box_end = flat_region_boxes_start[region_id + 1];
|
||||
float node_area = (node_xh - node_xl) * (node_yh - node_yl);
|
||||
// I assume there is no overlap between boxes of a region
|
||||
// otherwise, preprocessing is required
|
||||
for (int box_id = box_bgn; box_id < box_end; ++box_id) {
|
||||
int box_offset = box_id * 4;
|
||||
float box_xl = flat_region_boxes[box_offset];
|
||||
float box_xh = flat_region_boxes[box_offset + 1];
|
||||
float box_yl = flat_region_boxes[box_offset + 2];
|
||||
float box_yh = flat_region_boxes[box_offset + 3];
|
||||
|
||||
float dx = std::max(std::min(node_xh, box_xh) - std::max(node_xl, box_xl), (float)0);
|
||||
float dy = std::max(std::min(node_yh, box_yh) - std::max(node_yl, box_yl), (float)0);
|
||||
float overlap = dx * dy;
|
||||
if (overlap > 0) {
|
||||
node_area -= overlap;
|
||||
}
|
||||
}
|
||||
if (node_area > 0) { // not consumed by boxes within a region
|
||||
std::string fence_str = "";
|
||||
for (int box_id = box_bgn; box_id < box_end; ++box_id) {
|
||||
int box_offset = box_id * 4;
|
||||
float box_xl = flat_region_boxes[box_offset];
|
||||
float box_xh = flat_region_boxes[box_offset + 1];
|
||||
float box_yl = flat_region_boxes[box_offset + 2];
|
||||
float box_yh = flat_region_boxes[box_offset + 3];
|
||||
fence_str += (" (" + std::to_string(box_xl) + ", " + std::to_string(box_yl) + ", " +
|
||||
std::to_string(box_xh) + ", " + std::to_string(box_yh) + ")");
|
||||
}
|
||||
logger.error("node %d (%g, %g, %g, %g), out of fence region %d: %s",
|
||||
i,
|
||||
node_xl,
|
||||
node_yl,
|
||||
node_xh,
|
||||
node_yh,
|
||||
region_id,
|
||||
fence_str.c_str());
|
||||
legal_flag = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
return legal_flag;
|
||||
}
|
||||
|
||||
bool overlapCheck(const float* x,
|
||||
const float* y,
|
||||
const float* node_size_x,
|
||||
const float* node_size_y,
|
||||
float site_width,
|
||||
float row_height,
|
||||
float scale_factor,
|
||||
float xl,
|
||||
float yl,
|
||||
float xh,
|
||||
float yh,
|
||||
int num_nodes,
|
||||
int num_movable_nodes) {
|
||||
bool legal_flag = true;
|
||||
int num_rows = ceilDiv(yh - yl, row_height);
|
||||
assert(num_rows > 0);
|
||||
std::vector<std::vector<int> > row_nodes(num_rows);
|
||||
|
||||
// general to node and fixed boxes
|
||||
auto getXL = [&](int id) { return x[id]; };
|
||||
auto getYL = [&](int id) { return y[id]; };
|
||||
auto getXH = [&](int id) { return x[id] + node_size_x[id]; };
|
||||
auto getYH = [&](int id) { return y[id] + node_size_y[id]; };
|
||||
|
||||
auto getSiteXL = [&](float xx) { return int(floorDiv(xx - xl, site_width)); };
|
||||
auto getSiteYL = [&](float yy) { return int(floorDiv(yy - yl, row_height)); };
|
||||
auto getSiteXH = [&](float xx) { return int(ceilDiv(xx - xl, site_width)); };
|
||||
auto getSiteYH = [&](float yy) { return int(ceilDiv(yy - yl, row_height)); };
|
||||
|
||||
// add a box to row
|
||||
auto addBox2Row = [&](int id, float bxl, float byl, float bxh, float byh) {
|
||||
int row_idxl = floorDiv(byl - yl, row_height);
|
||||
int row_idxh = ceilDiv(byh - yl, row_height);
|
||||
row_idxl = std::max(row_idxl, 0);
|
||||
row_idxh = std::min(row_idxh, num_rows);
|
||||
|
||||
for (int row_id = row_idxl; row_id < row_idxh; ++row_id) {
|
||||
float row_yl = yl + row_id * row_height;
|
||||
float row_yh = row_yl + row_height;
|
||||
|
||||
if (byl < row_yh && byh > row_yl) // overlap with row
|
||||
{
|
||||
row_nodes[row_id].push_back(id);
|
||||
}
|
||||
}
|
||||
};
|
||||
// distribute movable cells to rows
|
||||
for (int i = 0; i < num_nodes; ++i) {
|
||||
float node_xl = x[i];
|
||||
float node_yl = y[i];
|
||||
float node_xh = node_xl + node_size_x[i];
|
||||
float node_yh = node_yl + node_size_y[i];
|
||||
|
||||
addBox2Row(i, node_xl, node_yl, node_xh, node_yh);
|
||||
}
|
||||
|
||||
// sort cells within rows
|
||||
for (int i = 0; i < num_rows; ++i) {
|
||||
auto& nodes_in_row = row_nodes.at(i);
|
||||
// using left edge
|
||||
std::sort(nodes_in_row.begin(), nodes_in_row.end(), [&](int node_id1, int node_id2) {
|
||||
float x1 = getXL(node_id1);
|
||||
float x2 = getXL(node_id2);
|
||||
return x1 < x2 || (x1 == x2 && (node_id1 < node_id2));
|
||||
});
|
||||
// After sorting by left edge,
|
||||
// there is a special case for fixed cells where
|
||||
// one fixed cell is completely within another in a row.
|
||||
// This will cause failure to detect some overlaps.
|
||||
// We need to remove the "small" fixed cell that is inside another.
|
||||
if (!nodes_in_row.empty()) {
|
||||
std::vector<int> tmp_nodes;
|
||||
tmp_nodes.reserve(nodes_in_row.size());
|
||||
tmp_nodes.push_back(nodes_in_row.front());
|
||||
for (int j = 1, je = nodes_in_row.size(); j < je; ++j) {
|
||||
int node_id1 = nodes_in_row.at(j - 1);
|
||||
int node_id2 = nodes_in_row.at(j);
|
||||
// two fixed cells
|
||||
if (node_id1 >= num_movable_nodes && node_id2 >= num_movable_nodes) {
|
||||
float xh1 = getXH(node_id1);
|
||||
float xh2 = getXH(node_id2);
|
||||
if (xh1 < xh2) {
|
||||
tmp_nodes.push_back(node_id2);
|
||||
}
|
||||
} else {
|
||||
tmp_nodes.push_back(node_id2);
|
||||
}
|
||||
}
|
||||
nodes_in_row.swap(tmp_nodes);
|
||||
}
|
||||
}
|
||||
|
||||
// check overlap
|
||||
// use scale factor to control the precision
|
||||
// auto scaleBack2Integer = [&](float value) {
|
||||
// return (scale_factor == 1.0) ? value : std::round(value / scale_factor);
|
||||
// };
|
||||
for (int i = 0; i < num_rows; ++i) {
|
||||
for (unsigned int j = 0; j < row_nodes.at(i).size(); ++j) {
|
||||
if (j > 0) {
|
||||
int node_id = row_nodes[i][j];
|
||||
int prev_node_id = row_nodes[i][j - 1];
|
||||
|
||||
if (node_id < num_movable_nodes || prev_node_id < num_movable_nodes) // ignore two fixed nodes
|
||||
{
|
||||
float prev_xl = getXL(prev_node_id);
|
||||
float prev_yl = getYL(prev_node_id);
|
||||
float prev_xh = getXH(prev_node_id);
|
||||
float prev_yh = getYH(prev_node_id);
|
||||
float cur_xl = getXL(node_id);
|
||||
float cur_yl = getYL(node_id);
|
||||
float cur_xh = getXH(node_id);
|
||||
float cur_yh = getYH(node_id);
|
||||
int prev_site_xl = getSiteXL(prev_xl);
|
||||
int prev_site_xh = getSiteXH(prev_xh);
|
||||
int cur_site_xl = getSiteXL(cur_xl);
|
||||
int cur_site_xh = getSiteXH(cur_xh);
|
||||
// detect overlap
|
||||
if (prev_site_xh > cur_site_xl) {
|
||||
logger.error(
|
||||
"row %d (%g, %g), overlap node %d (%g, %g, %g, %g) with "
|
||||
"node %d (%g, %g, %g, %g) site (%d, %d), gap %g",
|
||||
i,
|
||||
yl + i * row_height,
|
||||
yl + (i + 1) * row_height,
|
||||
prev_node_id,
|
||||
prev_xl,
|
||||
prev_yl,
|
||||
prev_xh,
|
||||
prev_yh,
|
||||
node_id,
|
||||
cur_xl,
|
||||
cur_yl,
|
||||
cur_xh,
|
||||
cur_yh,
|
||||
cur_site_xl,
|
||||
cur_site_xh,
|
||||
prev_xh - cur_xl);
|
||||
legal_flag = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return legal_flag;
|
||||
}
|
||||
|
||||
bool legalityCheckKernelCPU(const float* x,
|
||||
const float* y,
|
||||
const float* node_size_x,
|
||||
const float* node_size_y,
|
||||
const float* flat_region_boxes,
|
||||
const int* flat_region_boxes_start,
|
||||
const int* node2fence_region_map,
|
||||
float xl,
|
||||
float yl,
|
||||
float xh,
|
||||
float yh,
|
||||
float site_width,
|
||||
float row_height,
|
||||
int num_nodes,
|
||||
int num_movable_nodes,
|
||||
int num_regions,
|
||||
float scale_factor) {
|
||||
bool legal_flag = true;
|
||||
int num_rows = ceil((yh - yl) / row_height);
|
||||
assert(num_rows > 0);
|
||||
fflush(stdout);
|
||||
std::vector<std::vector<int> > row_nodes(num_rows);
|
||||
|
||||
// check node within boundary
|
||||
if (!boundaryCheck(x, y, node_size_x, node_size_y, scale_factor, xl, yl, xh, yh, num_movable_nodes)) {
|
||||
legal_flag = false;
|
||||
std::cerr << "boundary check error!" << std::endl;
|
||||
}
|
||||
|
||||
// check row and site alignment
|
||||
if (!siteAlignmentCheck(x, y, site_width, row_height, scale_factor, xl, yl, num_movable_nodes)) {
|
||||
legal_flag = false;
|
||||
std::cerr << "site alignment check error!" << std::endl;
|
||||
}
|
||||
|
||||
if (!overlapCheck(
|
||||
x, y, node_size_x, node_size_y, site_width, row_height, scale_factor, xl, yl, xh, yh, num_nodes, num_movable_nodes)) {
|
||||
legal_flag = false;
|
||||
std::cerr << "overlap check error!" << std::endl;
|
||||
}
|
||||
|
||||
// check fence regions
|
||||
if (!fenceRegionCheck(x,
|
||||
y,
|
||||
node_size_x,
|
||||
node_size_y,
|
||||
flat_region_boxes,
|
||||
flat_region_boxes_start,
|
||||
node2fence_region_map,
|
||||
num_movable_nodes,
|
||||
num_regions)) {
|
||||
legal_flag = false;
|
||||
std::cerr << "fence region check error!" << std::endl;
|
||||
}
|
||||
|
||||
if (!legal_flag) {
|
||||
logger.error("placement legality check error!");
|
||||
}
|
||||
|
||||
return legal_flag;
|
||||
}
|
||||
|
||||
bool legalityCheck(DPTorchRawDB& at_db, float scale_factor) {
|
||||
return legalityCheckKernelCPU(at_db.x.cpu().data_ptr<float>(),
|
||||
at_db.y.cpu().data_ptr<float>(),
|
||||
at_db.node_size_x.cpu().data_ptr<float>(),
|
||||
at_db.node_size_y.cpu().data_ptr<float>(),
|
||||
at_db.flat_region_boxes.cpu().data_ptr<float>(),
|
||||
at_db.flat_region_boxes_start.cpu().data_ptr<int>(),
|
||||
at_db.node2fence_region_map.cpu().data_ptr<int>(),
|
||||
at_db.xl,
|
||||
at_db.yl,
|
||||
at_db.xh,
|
||||
at_db.yh,
|
||||
at_db.site_width,
|
||||
at_db.row_height,
|
||||
at_db.num_nodes,
|
||||
at_db.num_movable_nodes,
|
||||
at_db.num_regions,
|
||||
scale_factor);
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
164
cpp_to_py/gpudp/db/dp_torch.cpp
Normal file
164
cpp_to_py/gpudp/db/dp_torch.cpp
Normal file
@ -0,0 +1,164 @@
|
||||
#include "gpudp/db/dp_torch.h"
|
||||
|
||||
#include "common/common.h"
|
||||
#include "common/db/Database.h"
|
||||
|
||||
namespace dp {
|
||||
|
||||
DPTorchRawDB::DPTorchRawDB(torch::Tensor node_lpos_init_,
|
||||
torch::Tensor node_size_,
|
||||
torch::Tensor node_weight_,
|
||||
torch::Tensor pin_rel_lpos_,
|
||||
torch::Tensor pin_id2node_id_,
|
||||
torch::Tensor pin_id2net_id_,
|
||||
torch::Tensor node2pin_list_,
|
||||
torch::Tensor node2pin_list_end_,
|
||||
torch::Tensor hyperedge_list_,
|
||||
torch::Tensor hyperedge_list_end_,
|
||||
torch::Tensor net_mask_,
|
||||
torch::Tensor node_id2region_id_,
|
||||
torch::Tensor region_boxes_,
|
||||
torch::Tensor region_boxes_end_,
|
||||
float xl_,
|
||||
float xh_,
|
||||
float yl_,
|
||||
float yh_,
|
||||
int num_movable_nodes_,
|
||||
int num_nodes_,
|
||||
float site_width_,
|
||||
float row_height_) {
|
||||
node_lpos_init = node_lpos_init_;
|
||||
node_size = node_size_;
|
||||
pin_rel_lpos = pin_rel_lpos_;
|
||||
|
||||
node_size_x = node_size.index({"...", 0}).clone().contiguous();
|
||||
node_size_y = node_size.index({"...", 1}).clone().contiguous();
|
||||
init_x = node_lpos_init.index({"...", 0}).clone().contiguous();
|
||||
init_y = node_lpos_init.index({"...", 1}).clone().contiguous();
|
||||
pin_offset_x = pin_rel_lpos.index({"...", 0}).clone().contiguous();
|
||||
pin_offset_y = pin_rel_lpos.index({"...", 1}).clone().contiguous();
|
||||
x = init_x.clone().contiguous();
|
||||
y = init_y.clone().contiguous();
|
||||
|
||||
num_nodes = num_nodes_;
|
||||
num_pins = pin_id2node_id_.size(0);
|
||||
num_nets = hyperedge_list_end_.size(0);
|
||||
num_regions = region_boxes_end_.size(0);
|
||||
num_movable_nodes = num_movable_nodes_;
|
||||
|
||||
flat_node2pin_start_map =
|
||||
torch::cat({torch::zeros({1}, torch::dtype(torch::kInt32).device(torch::Device(node_size.device()))),
|
||||
node2pin_list_end_},
|
||||
0)
|
||||
.contiguous();
|
||||
flat_node2pin_map = node2pin_list_;
|
||||
pin2node_map = pin_id2node_id_;
|
||||
|
||||
flat_net2pin_start_map =
|
||||
torch::cat({torch::zeros({1}, torch::dtype(torch::kInt32).device(torch::Device(node_size.device()))),
|
||||
hyperedge_list_end_},
|
||||
0)
|
||||
.contiguous();
|
||||
flat_net2pin_map = hyperedge_list_;
|
||||
pin2net_map = pin_id2net_id_;
|
||||
|
||||
flat_region_boxes_start =
|
||||
torch::cat({torch::zeros({1}, torch::dtype(torch::kInt32).device(torch::Device(node_size.device()))),
|
||||
region_boxes_end_},
|
||||
0)
|
||||
.contiguous();
|
||||
flat_region_boxes = region_boxes_.flatten().contiguous().clone();
|
||||
node2fence_region_map = node_id2region_id_;
|
||||
|
||||
net_mask = net_mask_;
|
||||
node_weight = node_weight_;
|
||||
|
||||
site_width = site_width_;
|
||||
row_height = row_height_;
|
||||
xl = xl_;
|
||||
xh = xh_;
|
||||
yl = yl_;
|
||||
yh = yh_;
|
||||
|
||||
num_sites_x = std::round((xh - xl) / site_width);
|
||||
num_sites_y = std::round((yh - yl) / row_height);
|
||||
|
||||
num_threads = std::max(db::setting.numThreads, 1);
|
||||
}
|
||||
|
||||
bool DPTorchRawDB::check(float scale_factor) {
|
||||
// NOTE: if tensors are on GPU, legalityCheck would copy large data from GPU to CPU
|
||||
return legalityCheck(*this, scale_factor);
|
||||
}
|
||||
|
||||
void DPTorchRawDB::scale(float scale_factor, bool use_round) {
|
||||
pin_rel_lpos.mul_(scale_factor);
|
||||
if (use_round) {
|
||||
node_size.mul_(scale_factor).round_();
|
||||
node_lpos_init.mul_(scale_factor).round_();
|
||||
x.mul_(scale_factor).round_();
|
||||
y.mul_(scale_factor).round_();
|
||||
flat_region_boxes.mul_(scale_factor).round_();
|
||||
site_width = round(site_width * scale_factor);
|
||||
row_height = round(row_height * scale_factor);
|
||||
xl = round(xl * scale_factor);
|
||||
xh = round(xh * scale_factor);
|
||||
yl = round(yl * scale_factor);
|
||||
yh = round(yh * scale_factor);
|
||||
} else {
|
||||
node_size.mul_(scale_factor);
|
||||
node_lpos_init.mul_(scale_factor);
|
||||
x.mul_(scale_factor);
|
||||
y.mul_(scale_factor);
|
||||
flat_region_boxes.mul_(scale_factor);
|
||||
site_width = site_width * scale_factor;
|
||||
row_height = row_height * scale_factor;
|
||||
xl = xl * scale_factor;
|
||||
xh = xh * scale_factor;
|
||||
yl = yl * scale_factor;
|
||||
yh = yh * scale_factor;
|
||||
}
|
||||
}
|
||||
|
||||
void DPTorchRawDB::commit() {
|
||||
// commit cached pos to original pos
|
||||
init_x.index({torch::indexing::Slice(0, num_movable_nodes)})
|
||||
.data()
|
||||
.copy_(x.index({torch::indexing::Slice(0, num_movable_nodes)}));
|
||||
init_y.index({torch::indexing::Slice(0, num_movable_nodes)})
|
||||
.data()
|
||||
.copy_(y.index({torch::indexing::Slice(0, num_movable_nodes)}));
|
||||
}
|
||||
|
||||
void DPTorchRawDB::rollback() {
|
||||
// rollback cached pos to original pos
|
||||
x.index({torch::indexing::Slice(0, num_movable_nodes)})
|
||||
.data()
|
||||
.copy_(init_x.index({torch::indexing::Slice(0, num_movable_nodes)}));
|
||||
y.index({torch::indexing::Slice(0, num_movable_nodes)})
|
||||
.data()
|
||||
.copy_(init_y.index({torch::indexing::Slice(0, num_movable_nodes)}));
|
||||
}
|
||||
|
||||
void DPTorchRawDB::commit_from(torch::Tensor x_, torch::Tensor y_) {
|
||||
// commit external pos to original pos
|
||||
init_x.index({torch::indexing::Slice(0, num_movable_nodes)})
|
||||
.data()
|
||||
.copy_(x_.index({torch::indexing::Slice(0, num_movable_nodes)}));
|
||||
init_y.index({torch::indexing::Slice(0, num_movable_nodes)})
|
||||
.data()
|
||||
.copy_(y_.index({torch::indexing::Slice(0, num_movable_nodes)}));
|
||||
x.index({torch::indexing::Slice(0, num_movable_nodes)})
|
||||
.data()
|
||||
.copy_(x_.index({torch::indexing::Slice(0, num_movable_nodes)}));
|
||||
y.index({torch::indexing::Slice(0, num_movable_nodes)})
|
||||
.data()
|
||||
.copy_(y_.index({torch::indexing::Slice(0, num_movable_nodes)}));
|
||||
}
|
||||
|
||||
torch::Tensor DPTorchRawDB::get_curr_cposx() { return x + node_size_x / 2; }
|
||||
torch::Tensor DPTorchRawDB::get_curr_cposy() { return y + node_size_y / 2; }
|
||||
torch::Tensor DPTorchRawDB::get_curr_lposx() { return x; }
|
||||
torch::Tensor DPTorchRawDB::get_curr_lposy() { return y; }
|
||||
|
||||
} // namespace dp
|
||||
114
cpp_to_py/gpudp/db/dp_torch.h
Normal file
114
cpp_to_py/gpudp/db/dp_torch.h
Normal file
@ -0,0 +1,114 @@
|
||||
#pragma once
|
||||
|
||||
#include "common/common.h"
|
||||
#include "common/db/Database.h"
|
||||
|
||||
namespace dp {
|
||||
|
||||
class DPTorchRawDB {
|
||||
public:
|
||||
DPTorchRawDB(torch::Tensor node_lpos_init_,
|
||||
torch::Tensor node_size_,
|
||||
torch::Tensor node_weight_,
|
||||
torch::Tensor pin_rel_lpos_,
|
||||
torch::Tensor pin_id2node_id_,
|
||||
torch::Tensor pin_id2net_id_,
|
||||
torch::Tensor node2pin_list_,
|
||||
torch::Tensor node2pin_list_end_,
|
||||
torch::Tensor hyperedge_list_,
|
||||
torch::Tensor hyperedge_list_end_,
|
||||
torch::Tensor net_mask_,
|
||||
torch::Tensor node_id2region_id_,
|
||||
torch::Tensor region_boxes_,
|
||||
torch::Tensor region_boxes_end_,
|
||||
float xl_,
|
||||
float xh_,
|
||||
float yl_,
|
||||
float yh_,
|
||||
int num_movable_nodes_,
|
||||
int num_nodes_,
|
||||
float site_width_,
|
||||
float row_height_);
|
||||
bool check(float scale_factor);
|
||||
void scale(float scale_factor, bool use_round);
|
||||
void commit();
|
||||
void rollback();
|
||||
void commit_from(torch::Tensor x_, torch::Tensor y_);
|
||||
torch::Tensor get_curr_cposx();
|
||||
torch::Tensor get_curr_cposy();
|
||||
torch::Tensor get_curr_lposx();
|
||||
torch::Tensor get_curr_lposy();
|
||||
|
||||
public:
|
||||
/* node info */
|
||||
// for backup
|
||||
torch::Tensor node_lpos_init;
|
||||
torch::Tensor node_size;
|
||||
torch::Tensor pin_rel_lpos;
|
||||
|
||||
torch::Tensor node_weight;
|
||||
|
||||
torch::Tensor init_x; // original pos (keep it const except committing)
|
||||
torch::Tensor init_y; // original pos (keep it const except committing)
|
||||
torch::Tensor x; // mutable/cached pos (current)
|
||||
torch::Tensor y; // mutable/cached pos (current)
|
||||
torch::Tensor node_size_x;
|
||||
torch::Tensor node_size_y;
|
||||
|
||||
/* pin info */
|
||||
torch::Tensor pin_offset_x;
|
||||
torch::Tensor pin_offset_y;
|
||||
|
||||
torch::Tensor flat_node2pin_start_map;
|
||||
torch::Tensor flat_node2pin_map;
|
||||
torch::Tensor pin2node_map;
|
||||
|
||||
/* net info */
|
||||
torch::Tensor flat_net2pin_start_map;
|
||||
torch::Tensor flat_net2pin_map;
|
||||
torch::Tensor pin2net_map;
|
||||
torch::Tensor net_mask;
|
||||
|
||||
/* fence info */
|
||||
torch::Tensor flat_region_boxes_start;
|
||||
torch::Tensor flat_region_boxes;
|
||||
torch::Tensor node2fence_region_map;
|
||||
|
||||
/* chip info */
|
||||
float xl;
|
||||
float yl;
|
||||
float xh;
|
||||
float yh;
|
||||
|
||||
/* row info */
|
||||
int num_sites_x;
|
||||
int num_sites_y;
|
||||
|
||||
int num_pins;
|
||||
int num_nets;
|
||||
int num_nodes;
|
||||
int num_movable_nodes;
|
||||
int num_regions;
|
||||
|
||||
float site_width;
|
||||
float row_height;
|
||||
|
||||
int num_threads;
|
||||
};
|
||||
|
||||
/* API for python */
|
||||
// Legalization
|
||||
bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y);
|
||||
void abacusLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y);
|
||||
void greedyLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y);
|
||||
|
||||
// Detailed Placement
|
||||
void kReorder(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int K, int max_iters);
|
||||
void globalSwap(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int max_iters);
|
||||
void independentSetMatching(
|
||||
DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int set_size, int max_iters);
|
||||
|
||||
// Legality Check
|
||||
bool legalityCheck(DPTorchRawDB& at_db, float scale_factor);
|
||||
|
||||
} // namespace dp
|
||||
39
cpp_to_py/gpudp/dp/compute_hpwl.cu
Normal file
39
cpp_to_py/gpudp/dp/compute_hpwl.cu
Normal file
@ -0,0 +1,39 @@
|
||||
#include <cuda.h>
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
#include "common/common.h"
|
||||
#include "gpudp/db/dp_torch.h"
|
||||
#include "detailed_place_db.cuh"
|
||||
|
||||
namespace dp {
|
||||
|
||||
__global__ void compute_total_hpwl_kernel(DetailedPlaceData db, const float* xx, const float* yy, double* net_hpwls) {
|
||||
for (int i = blockIdx.x * blockDim.x + threadIdx.x; i < db.num_nets; i += blockDim.x * gridDim.x) {
|
||||
net_hpwls[i] = double(db.compute_net_hpwl(i, xx, yy));
|
||||
}
|
||||
}
|
||||
|
||||
float compute_total_hpwl(const DetailedPlaceData& db, const float* xx, const float* yy, double* net_hpwls) {
|
||||
compute_total_hpwl_kernel<<<ceilDiv(db.num_nets, 512), 512>>>(db, xx, yy, net_hpwls);
|
||||
// auto hpwl = thrust::reduce(thrust::device, net_hpwls, net_hpwls+db.num_nets);
|
||||
|
||||
double* d_out = NULL;
|
||||
// Determine temporary device storage requirements
|
||||
void* d_temp_storage = NULL;
|
||||
size_t temp_storage_bytes = 0;
|
||||
cub::DeviceReduce::Sum(d_temp_storage, temp_storage_bytes, net_hpwls, d_out, db.num_nets);
|
||||
// Allocate temporary storage
|
||||
checkCuda(cudaMalloc(&d_temp_storage, temp_storage_bytes));
|
||||
checkCuda(cudaMalloc(&d_out, sizeof(double)));
|
||||
// Run sum-reduction
|
||||
cub::DeviceReduce::Sum(d_temp_storage, temp_storage_bytes, net_hpwls, d_out, db.num_nets);
|
||||
// copy d_out to hpwl
|
||||
double hpwl = 0;
|
||||
checkCuda(cudaMemcpy(&hpwl, d_out, sizeof(double), cudaMemcpyDeviceToHost));
|
||||
cudaFree(d_temp_storage);
|
||||
cudaFree(d_out);
|
||||
|
||||
return float(hpwl);
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
24
cpp_to_py/gpudp/dp/detail_placement.cpp
Normal file
24
cpp_to_py/gpudp/dp/detail_placement.cpp
Normal file
@ -0,0 +1,24 @@
|
||||
#include "common/common.h"
|
||||
#include "gpudp/db/dp_torch.h"
|
||||
|
||||
namespace dp {
|
||||
|
||||
void kReorderCUDA(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int K, int max_iters);
|
||||
void globalSwapCUDA(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int max_iters);
|
||||
void independentSetMatchingCUDA(
|
||||
DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int set_size, int max_iters);
|
||||
|
||||
void kReorder(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int K, int max_iters) {
|
||||
kReorderCUDA(at_db, num_bins_x, num_bins_y, K, max_iters);
|
||||
}
|
||||
|
||||
void globalSwap(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int max_iters) {
|
||||
globalSwapCUDA(at_db, num_bins_x, num_bins_y, batch_size, max_iters);
|
||||
}
|
||||
|
||||
void independentSetMatching(
|
||||
DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int set_size, int max_iters) {
|
||||
independentSetMatchingCUDA(at_db, num_bins_x, num_bins_y, batch_size, set_size, max_iters);
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
417
cpp_to_py/gpudp/dp/detailed_place_db.cuh
Normal file
417
cpp_to_py/gpudp/dp/detailed_place_db.cuh
Normal file
@ -0,0 +1,417 @@
|
||||
#pragma once
|
||||
|
||||
#include <cuda.h>
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
#include "common/common.h"
|
||||
#include "common/db/Database.h"
|
||||
#include "gpudp/db/dp_torch.h"
|
||||
#include "pitch_nested_vector.cuh"
|
||||
#include "utils.cuh"
|
||||
|
||||
namespace dp {
|
||||
|
||||
inline __host__ __device__ int floorDiv(float a, float b, float rtol = 1e-4) { return floor((a + rtol * b) / b); }
|
||||
|
||||
inline __host__ __device__ int ceilDiv(float a, float b, float rtol = 1e-4) { return ceil((a - rtol * b) / b); }
|
||||
|
||||
inline __host__ __device__ int roundDiv(float a, float b) { return round(a / b); }
|
||||
|
||||
template <typename T>
|
||||
struct Space {
|
||||
T xl;
|
||||
T xh;
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
struct Box {
|
||||
T xl;
|
||||
T yl;
|
||||
T xh;
|
||||
T yh;
|
||||
__host__ __device__ Box() {
|
||||
xl = cuda::numeric_limits<T>::max();
|
||||
yl = cuda::numeric_limits<T>::max();
|
||||
xh = cuda::numeric_limits<T>::lowest();
|
||||
yh = cuda::numeric_limits<T>::lowest();
|
||||
}
|
||||
__host__ __device__ Box(T xxl, T yyl, T xxh, T yyh) : xl(xxl), yl(yyl), xh(xxh), yh(yyh) {}
|
||||
|
||||
__host__ __device__ T center_x() const { return (xl + xh) / 2; }
|
||||
__host__ __device__ T center_y() const { return (yl + yh) / 2; }
|
||||
};
|
||||
|
||||
struct RowMapIndex {
|
||||
int row_id;
|
||||
int sub_id;
|
||||
};
|
||||
|
||||
struct BinMapIndex {
|
||||
int bin_id;
|
||||
int sub_id;
|
||||
};
|
||||
|
||||
class DetailedPlaceData {
|
||||
public:
|
||||
DetailedPlaceData() {}
|
||||
DetailedPlaceData(DPTorchRawDB& at_db)
|
||||
: x(at_db.x.data_ptr<float>()),
|
||||
y(at_db.y.data_ptr<float>()),
|
||||
init_x(at_db.init_x.data_ptr<float>()),
|
||||
init_y(at_db.init_y.data_ptr<float>()),
|
||||
node_size_x(at_db.node_size_x.data_ptr<float>()),
|
||||
node_size_y(at_db.node_size_y.data_ptr<float>()),
|
||||
pin_offset_x(at_db.pin_offset_x.data_ptr<float>()),
|
||||
pin_offset_y(at_db.pin_offset_y.data_ptr<float>()),
|
||||
flat_node2pin_start_map(at_db.flat_node2pin_start_map.data_ptr<int>()),
|
||||
flat_node2pin_map(at_db.flat_node2pin_map.data_ptr<int>()),
|
||||
pin2node_map(at_db.pin2node_map.data_ptr<int>()),
|
||||
flat_net2pin_start_map(at_db.flat_net2pin_start_map.data_ptr<int>()),
|
||||
flat_net2pin_map(at_db.flat_net2pin_map.data_ptr<int>()),
|
||||
pin2net_map(at_db.pin2net_map.data_ptr<int>()),
|
||||
flat_region_boxes_start(at_db.flat_region_boxes_start.data_ptr<int>()),
|
||||
flat_region_boxes(at_db.flat_region_boxes.data_ptr<float>()),
|
||||
node2fence_region_map(at_db.node2fence_region_map.data_ptr<int>()),
|
||||
net_mask(at_db.net_mask.data_ptr<bool>()),
|
||||
node_weight(at_db.node_weight.data_ptr<float>()),
|
||||
xl(at_db.xl),
|
||||
xh(at_db.xh),
|
||||
yl(at_db.yl),
|
||||
yh(at_db.yh),
|
||||
row_height(at_db.row_height),
|
||||
site_width(at_db.site_width),
|
||||
num_sites_x(at_db.num_sites_x),
|
||||
num_sites_y(at_db.num_sites_y),
|
||||
num_threads(at_db.num_threads),
|
||||
num_nodes(at_db.num_nodes),
|
||||
num_movable_nodes(at_db.num_movable_nodes),
|
||||
num_nets(at_db.num_nets),
|
||||
num_pins(at_db.num_pins),
|
||||
num_regions(at_db.num_regions) {}
|
||||
|
||||
public:
|
||||
typedef float type;
|
||||
|
||||
float* x;
|
||||
float* y;
|
||||
const float* init_x;
|
||||
const float* init_y;
|
||||
const float* node_size_x;
|
||||
const float* node_size_y;
|
||||
|
||||
const float* pin_offset_x;
|
||||
const float* pin_offset_y;
|
||||
|
||||
const int* flat_node2pin_start_map;
|
||||
const int* flat_node2pin_map;
|
||||
const int* pin2node_map;
|
||||
|
||||
const int* flat_net2pin_start_map;
|
||||
const int* flat_net2pin_map;
|
||||
const int* pin2net_map;
|
||||
|
||||
const int* flat_region_boxes_start;
|
||||
const float* flat_region_boxes;
|
||||
const int* node2fence_region_map;
|
||||
|
||||
const bool* net_mask;
|
||||
const float* node_weight;
|
||||
|
||||
/* chip info */
|
||||
float xl;
|
||||
float yl;
|
||||
float xh;
|
||||
float yh;
|
||||
|
||||
/* row info */
|
||||
int num_sites_x;
|
||||
int num_sites_y;
|
||||
float row_height;
|
||||
float site_width;
|
||||
|
||||
int num_nets;
|
||||
int num_movable_nodes;
|
||||
int num_nodes;
|
||||
int num_pins;
|
||||
int num_regions;
|
||||
|
||||
int num_threads;
|
||||
|
||||
int num_bins_x;
|
||||
int num_bins_y;
|
||||
float bin_size_x;
|
||||
float bin_size_y;
|
||||
|
||||
public:
|
||||
void set_num_bins(int num_bins_x_, int num_bins_y_) {
|
||||
num_bins_x = num_bins_x_;
|
||||
num_bins_y = num_bins_y_;
|
||||
bin_size_x = (xh - xl) / num_bins_x_;
|
||||
bin_size_y = (yh - yl) / num_bins_y_;
|
||||
}
|
||||
inline __device__ int pos2site_x(float xx) const {
|
||||
return min(max((int)floorDiv((xx - xl), site_width), 0), num_sites_x - 1);
|
||||
}
|
||||
inline __device__ int pos2site_y(float yy) const {
|
||||
return min(max((int)floorDiv((yy - yl), row_height), 0), num_sites_y - 1);
|
||||
}
|
||||
inline __device__ int pos2site_ub_x(float xx) const {
|
||||
return min(max(ceilDiv((xx - xl), site_width), 1), num_sites_x);
|
||||
}
|
||||
inline __device__ int pos2site_ub_y(float yy) const {
|
||||
return min(max(ceilDiv((yy - yl), row_height), 1), num_sites_y);
|
||||
}
|
||||
inline __device__ int pos2bin_x(float xx) const {
|
||||
int bx = floorDiv((xx - xl), bin_size_x);
|
||||
bx = max(bx, 0);
|
||||
bx = min(bx, num_bins_x - 1);
|
||||
return bx;
|
||||
}
|
||||
inline __device__ int pos2bin_y(float yy) const {
|
||||
int by = floorDiv((yy - yl), bin_size_y);
|
||||
by = max(by, 0);
|
||||
by = min(by, num_bins_y - 1);
|
||||
return by;
|
||||
}
|
||||
inline __device__ void shift_box_to_layout(Box<float>& box) const {
|
||||
box.xl = max(box.xl, xl);
|
||||
box.xl = min(box.xl, xh);
|
||||
box.xh = max(box.xh, xl);
|
||||
box.xh = min(box.xh, xh);
|
||||
box.yl = max(box.yl, yl);
|
||||
box.yl = min(box.yl, yh);
|
||||
box.yh = max(box.yh, yl);
|
||||
box.yh = min(box.yh, yh);
|
||||
}
|
||||
inline __device__ float align2site(float xx) const {
|
||||
return (int)floorDiv((xx - xl), site_width) * site_width + xl;
|
||||
}
|
||||
inline __device__ Space<float> align2site(Space<float> space) const {
|
||||
space.xl = ceilDiv((space.xl - xl), site_width) * site_width + xl;
|
||||
space.xh = floorDiv((space.xh - xl), site_width) * site_width + xl;
|
||||
return space;
|
||||
}
|
||||
__device__ Box<float> compute_optimal_region(int node_id, const float* xx, const float* yy) const {
|
||||
Box<float> box(xh, yh, xl, yl);
|
||||
for (int node2pin_id = flat_node2pin_start_map[node_id]; node2pin_id < flat_node2pin_start_map[node_id + 1];
|
||||
++node2pin_id) {
|
||||
int node_pin_id = flat_node2pin_map[node2pin_id];
|
||||
int net_id = pin2net_map[node_pin_id];
|
||||
if (net_mask[net_id]) {
|
||||
for (int net2pin_id = flat_net2pin_start_map[net_id]; net2pin_id < flat_net2pin_start_map[net_id + 1];
|
||||
++net2pin_id) {
|
||||
int net_pin_id = flat_net2pin_map[net2pin_id];
|
||||
int other_node_id = pin2node_map[net_pin_id];
|
||||
if (node_id != other_node_id) {
|
||||
box.xl = min(box.xl, xx[other_node_id] + pin_offset_x[net_pin_id]);
|
||||
box.xh = max(box.xh, xx[other_node_id] + pin_offset_x[net_pin_id]);
|
||||
box.yl = min(box.yl, yy[other_node_id] + pin_offset_y[net_pin_id]);
|
||||
box.yh = max(box.yh, yy[other_node_id] + pin_offset_y[net_pin_id]);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
shift_box_to_layout(box);
|
||||
|
||||
return box;
|
||||
}
|
||||
__device__ float compute_net_hpwl(int net_id, const float* xx, const float* yy) const {
|
||||
Box<float> box(xh, yh, xl, yl);
|
||||
for (int net2pin_id = flat_net2pin_start_map[net_id]; net2pin_id < flat_net2pin_start_map[net_id + 1];
|
||||
++net2pin_id) {
|
||||
int net_pin_id = flat_net2pin_map[net2pin_id];
|
||||
int other_node_id = pin2node_map[net_pin_id];
|
||||
box.xl = min(box.xl, xx[other_node_id] + pin_offset_x[net_pin_id]);
|
||||
box.xh = max(box.xh, xx[other_node_id] + pin_offset_x[net_pin_id]);
|
||||
box.yl = min(box.yl, yy[other_node_id] + pin_offset_y[net_pin_id]);
|
||||
box.yh = max(box.yh, yy[other_node_id] + pin_offset_y[net_pin_id]);
|
||||
}
|
||||
if (box.xl == xh || box.yl == yh) {
|
||||
return (float)0;
|
||||
}
|
||||
return (box.xh - box.xl) + (box.yh - box.yl);
|
||||
}
|
||||
// __device__ float compute_total_hpwl() const {
|
||||
// float total_hpwl = 0;
|
||||
// for (int net_id = 0; net_id < num_nets; ++net_id) {
|
||||
// total_hpwl += compute_net_hpwl(net_id, x, y);
|
||||
// }
|
||||
// return total_hpwl;
|
||||
// }
|
||||
__device__ bool inside_fence(int node_id, float xx, float yy) const {
|
||||
float node_xl = xx;
|
||||
float node_yl = yy;
|
||||
float node_xh = node_xl + node_size_x[node_id];
|
||||
float node_yh = node_yl + node_size_y[node_id];
|
||||
|
||||
bool legal_flag = true;
|
||||
int region_id = node2fence_region_map[node_id];
|
||||
if (region_id < num_regions) {
|
||||
int box_bgn = flat_region_boxes_start[region_id];
|
||||
int box_end = flat_region_boxes_start[region_id + 1];
|
||||
float node_area = (node_xh - node_xl) * (node_yh - node_yl);
|
||||
// assume there is no overlap between boxes of a region
|
||||
// otherwise, preprocessing is required
|
||||
for (int box_id = box_bgn; box_id < box_end; ++box_id) {
|
||||
int box_offset = box_id * 4;
|
||||
float box_xl = flat_region_boxes[box_offset];
|
||||
float box_xh = flat_region_boxes[box_offset + 1];
|
||||
float box_yl = flat_region_boxes[box_offset + 2];
|
||||
float box_yh = flat_region_boxes[box_offset + 3];
|
||||
|
||||
float dx = max(min(node_xh, box_xh) - max(node_xl, box_xl), (float)0);
|
||||
float dy = max(min(node_yh, box_yh) - max(node_yl, box_yl), (float)0);
|
||||
float overlap = dx * dy;
|
||||
if (overlap > 0) {
|
||||
node_area -= overlap;
|
||||
}
|
||||
}
|
||||
if (node_area > 0) {
|
||||
// not consumed by boxes within a region
|
||||
legal_flag = false;
|
||||
}
|
||||
}
|
||||
return legal_flag;
|
||||
}
|
||||
|
||||
void make_row2node_map(const float* host_x,
|
||||
const float* host_y,
|
||||
const float* host_node_size_x,
|
||||
const float* host_node_size_y,
|
||||
int host_num_nodes,
|
||||
std::vector<std::vector<int>>& row2node_map) {
|
||||
// distribute cells to rows
|
||||
for (int i = 0; i < host_num_nodes; ++i) {
|
||||
float node_yl = host_y[i];
|
||||
float node_yh = node_yl + host_node_size_y[i];
|
||||
|
||||
int row_idxl = floorDiv(node_yl - yl, row_height);
|
||||
int row_idxh = ceilDiv(node_yh - yl, row_height);
|
||||
row_idxl = max(row_idxl, 0);
|
||||
row_idxh = min(row_idxh, num_sites_y);
|
||||
|
||||
for (int row_id = row_idxl; row_id < row_idxh; ++row_id) {
|
||||
float row_yl = yl + row_id * row_height;
|
||||
float row_yh = row_yl + row_height;
|
||||
|
||||
if (node_yl < row_yh && node_yh > row_yl) // overlap with row
|
||||
{
|
||||
row2node_map[row_id].push_back(i);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// sort cells within rows
|
||||
#pragma omp parallel for num_threads(num_threads) schedule(dynamic, 1)
|
||||
for (int i = 0; i < num_sites_y; ++i) {
|
||||
auto& row2nodes = row2node_map[i];
|
||||
// sort cells within rows according to left edges
|
||||
std::sort(row2nodes.begin(), row2nodes.end(), [&](int node_id1, int node_id2) {
|
||||
float x1 = host_x[node_id1];
|
||||
float x2 = host_x[node_id2];
|
||||
return x1 < x2 || (x1 == x2 && node_id1 < node_id2);
|
||||
});
|
||||
if (!row2nodes.empty()) {
|
||||
std::vector<int> tmp_nodes;
|
||||
tmp_nodes.reserve(row2nodes.size());
|
||||
tmp_nodes.push_back(row2nodes.front());
|
||||
for (int j = 1, je = row2nodes.size(); j < je; ++j) {
|
||||
int node_id1 = row2nodes.at(j - 1);
|
||||
int node_id2 = row2nodes.at(j);
|
||||
// two fixed cells
|
||||
if (node_id1 >= num_movable_nodes && node_id2 >= num_movable_nodes) {
|
||||
float xl1 = host_x[node_id1];
|
||||
float xl2 = host_x[node_id2];
|
||||
float width1 = host_node_size_x[node_id1];
|
||||
float width2 = host_node_size_x[node_id2];
|
||||
float xh1 = xl1 + width1;
|
||||
float xh2 = xl2 + width2;
|
||||
// only collect node_id2 if its right edge is righter than node_id1
|
||||
if (xh1 < xh2) {
|
||||
tmp_nodes.push_back(node_id2);
|
||||
}
|
||||
} else {
|
||||
tmp_nodes.push_back(node_id2);
|
||||
}
|
||||
}
|
||||
row2nodes.swap(tmp_nodes);
|
||||
|
||||
// sort according to center
|
||||
std::sort(row2nodes.begin(), row2nodes.end(), [&](int node_id1, int node_id2) {
|
||||
float x1 = host_x[node_id1] + host_node_size_x[node_id1] / 2;
|
||||
float x2 = host_x[node_id2] + host_node_size_x[node_id2] / 2;
|
||||
return x1 < x2 || (x1 == x2 && node_id1 < node_id2);
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void make_row2node_map_with_spaces(const float* host_x,
|
||||
const float* host_y,
|
||||
const float* host_node_size_x,
|
||||
const float* host_node_size_y,
|
||||
std::vector<std::vector<int>>& row2node_map,
|
||||
std::vector<RowMapIndex>& node2row_map,
|
||||
std::vector<Space<float>>& spaces) {
|
||||
make_row2node_map(host_x, host_y, host_node_size_x, host_node_size_y, num_nodes + 2, row2node_map);
|
||||
|
||||
// construct node2row_map
|
||||
for (int i = 0; i < num_sites_y; ++i) {
|
||||
for (unsigned int j = 0; j < row2node_map[i].size(); ++j) {
|
||||
int node_id = row2node_map[i][j];
|
||||
if (node_id < num_movable_nodes) {
|
||||
RowMapIndex& row_id = node2row_map[node_id];
|
||||
row_id.row_id = i;
|
||||
row_id.sub_id = j;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// construct spaces
|
||||
for (int i = 0; i < num_sites_y; ++i) {
|
||||
for (unsigned int j = 0; j < row2node_map[i].size(); ++j) {
|
||||
int node_id = row2node_map[i][j];
|
||||
if (node_id < num_movable_nodes) {
|
||||
assert(j);
|
||||
int left_node_id = row2node_map[i][j - 1];
|
||||
spaces[node_id].xl = host_x[left_node_id] + host_node_size_x[left_node_id];
|
||||
assert(j + 1 < row2node_map[i].size());
|
||||
int right_node_id = row2node_map[i][j + 1];
|
||||
spaces[node_id].xh = host_x[right_node_id];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void make_bin2node_map(const float* host_x,
|
||||
const float* host_y,
|
||||
const float* host_node_size_x,
|
||||
const float* host_node_size_y,
|
||||
std::vector<std::vector<int>>& bin2node_map,
|
||||
std::vector<BinMapIndex>& node2bin_map) {
|
||||
// construct bin2node_map
|
||||
for (int i = 0; i < num_movable_nodes; ++i) {
|
||||
int node_id = i;
|
||||
float node_x = host_x[node_id] + host_node_size_x[node_id] / 2;
|
||||
float node_y = host_y[node_id] + host_node_size_y[node_id] / 2;
|
||||
|
||||
int bx = min(max((int)floorDiv(node_x - xl, bin_size_x), 0), num_bins_x - 1);
|
||||
int by = min(max((int)floorDiv(node_y - yl, bin_size_y), 0), num_bins_y - 1);
|
||||
int bin_id = bx * num_bins_y + by;
|
||||
int sub_id = bin2node_map.at(bin_id).size();
|
||||
bin2node_map.at(bin_id).push_back(node_id);
|
||||
}
|
||||
for (int bin_id = 0; bin_id < bin2node_map.size(); ++bin_id) {
|
||||
for (int sub_id = 0; sub_id < bin2node_map[bin_id].size(); ++sub_id) {
|
||||
int node_id = bin2node_map[bin_id][sub_id];
|
||||
BinMapIndex& bm_idx = node2bin_map.at(node_id);
|
||||
bm_idx.bin_id = bin_id;
|
||||
bm_idx.sub_id = sub_id;
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
float compute_total_hpwl(const DetailedPlaceData& db, const float* xx, const float* yy, double* net_hpwls);
|
||||
|
||||
} // namespace dp
|
||||
1001
cpp_to_py/gpudp/dp/global_swap_cuda.cu
Normal file
1001
cpp_to_py/gpudp/dp/global_swap_cuda.cu
Normal file
File diff suppressed because it is too large
Load Diff
323
cpp_to_py/gpudp/dp/independent_set_matching_cuda_kernel.cu
Normal file
323
cpp_to_py/gpudp/dp/independent_set_matching_cuda_kernel.cu
Normal file
@ -0,0 +1,323 @@
|
||||
#include <curand.h>
|
||||
#include <curand_kernel.h>
|
||||
|
||||
#include "detailed_place_db.cuh"
|
||||
|
||||
#include "gpudp/dp/ism/apply_solution.cuh"
|
||||
#include "gpudp/dp/ism/auction.cuh"
|
||||
#include "gpudp/dp/ism/collect_independent_sets.cuh"
|
||||
#include "gpudp/dp/ism/cost_matrix_construction.cuh"
|
||||
#include "gpudp/dp/ism/cpu_state.cuh"
|
||||
#include "gpudp/dp/ism/maximal_independent_set.cuh"
|
||||
#include "gpudp/dp/ism/shuffle.cuh"
|
||||
|
||||
namespace dp {
|
||||
|
||||
#define DETERMINISTIC
|
||||
|
||||
#define NUM_NODE_SIZES 64 ///< number of different cell sizes
|
||||
|
||||
struct SizedBinIndex {
|
||||
int size_id;
|
||||
int bin_id;
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
struct IndependentSetMatchingState {
|
||||
typedef T type;
|
||||
typedef int cost_type;
|
||||
|
||||
int* ordered_nodes = nullptr;
|
||||
Space<T>* spaces = nullptr; ///< array of cell spaces, each cell only consider the space on its left side except
|
||||
///< for the left and right boundary
|
||||
int num_node_sizes; ///< number of cell sizes considered
|
||||
int* independent_sets = nullptr; ///< independent sets, length of batch_size*set_size
|
||||
int* independent_set_sizes = nullptr; ///< size of each independent set
|
||||
int* selected_maximal_independent_set = nullptr; ///< storing the selected maximum independent set
|
||||
int* select_scratch = nullptr; ///< temporary storage for selection kernel
|
||||
int num_selected; ///< maximum independent set size
|
||||
int* device_num_selected; ///< maximum independent set size
|
||||
|
||||
double* net_hpwls; ///< HPWL for each net, use integer to get consistent values
|
||||
|
||||
int* selected_markers = nullptr; ///< must be int for cub to compute prefix sum
|
||||
unsigned char* dependent_markers = nullptr;
|
||||
int* independent_set_empty_flag = nullptr; ///< a stopping flag for maximum independent set
|
||||
int num_independent_sets; ///< host copy
|
||||
|
||||
cost_type* cost_matrices = nullptr; ///< cost matrices batch_size*set_size*set_size
|
||||
cost_type* cost_matrices_copy = nullptr; ///< temporary copy of cost matrices
|
||||
int* solutions = nullptr; ///< batch_size*set_size
|
||||
char* auction_scratch = nullptr; ///< temporary memory for auction solver
|
||||
char* stop_flags = nullptr; ///< record stopping status from auction solver
|
||||
T* orig_x = nullptr; ///< original locations of cells for applying solutions
|
||||
T* orig_y = nullptr;
|
||||
cost_type* orig_costs = nullptr; ///< original costs
|
||||
cost_type* solution_costs = nullptr; ///< solution costs
|
||||
Space<T>* orig_spaces = nullptr; ///< original spaces of cells for apply solutions
|
||||
|
||||
int batch_size; ///< pre-allocated number of independent sets
|
||||
int set_size;
|
||||
int cost_matrix_size; ///< set_size*set_size
|
||||
int num_bins; ///< num_bins_x*num_bins_y
|
||||
int* device_num_moved; ///< device copy
|
||||
int num_moved; ///< host copy, number of moved cells
|
||||
int large_number; ///< a large number
|
||||
|
||||
float auction_max_eps; ///< maximum epsilon for auction solver
|
||||
float auction_min_eps; ///< minimum epsilon for auction solver
|
||||
float auction_factor; ///< decay factor for auction epsilon
|
||||
int auction_max_iterations; ///< maximum iteration
|
||||
T skip_threshold; ///< ignore connections if cells are far apart
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
__global__ void iota(T* a, int n) {
|
||||
for (int i = blockIdx.x * blockDim.x + threadIdx.x; i < n; i += blockDim.x * gridDim.x) {
|
||||
a[i] = i;
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void cost_matrix_init(int* cost_matrix, int set_size) {
|
||||
for (int i = blockIdx.x; i < set_size; i += gridDim.x) {
|
||||
for (int j = threadIdx.x; j < set_size; j += blockDim.x) {
|
||||
cost_matrix[i * set_size + j] = (i == j) ? 0 : cuda::numeric_limits<int>::max();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
__global__ void print_global(T* a, int n) {
|
||||
unsigned int tid = threadIdx.x;
|
||||
unsigned int bid = blockIdx.x;
|
||||
if (tid == 0 && bid == 0) {
|
||||
printf("[%d]\n", n);
|
||||
for (int i = 0; i < n; ++i) {
|
||||
printf("%g ", (double)a[i]);
|
||||
}
|
||||
printf("\n");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
__global__ void print_cost_matrix(const T* cost_matrix, int set_size, bool major) {
|
||||
unsigned int tid = threadIdx.x;
|
||||
unsigned int bid = blockIdx.x;
|
||||
if (tid == 0 && bid == 0) {
|
||||
printf("[%dx%d]\n", set_size, set_size);
|
||||
for (int r = 0; r < set_size; ++r) {
|
||||
for (int c = 0; c < set_size; ++c) {
|
||||
if (major) // column major
|
||||
{
|
||||
printf("%g ", (double)cost_matrix[c * set_size + r]);
|
||||
} else {
|
||||
printf("%g ", (double)cost_matrix[r * set_size + c]);
|
||||
}
|
||||
}
|
||||
printf("\n");
|
||||
}
|
||||
printf("\n");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
__global__ void print_solution(const T* solution, int n) {
|
||||
unsigned int tid = threadIdx.x;
|
||||
unsigned int bid = blockIdx.x;
|
||||
if (tid == 0 && bid == 0) {
|
||||
printf("[%d]\n", n);
|
||||
for (int i = 0; i < n; ++i) {
|
||||
printf("%g ", (double)solution[i]);
|
||||
}
|
||||
printf("\n");
|
||||
}
|
||||
}
|
||||
|
||||
void construct_spaces(DetailedPlaceData& db,
|
||||
const float* host_x,
|
||||
const float* host_y,
|
||||
const float* host_node_size_x,
|
||||
const float* host_node_size_y,
|
||||
std::vector<Space<float>>& host_spaces,
|
||||
int num_threads) {
|
||||
std::vector<std::vector<int> > row2node_map(db.num_sites_y);
|
||||
db.make_row2node_map(host_x, host_y, host_node_size_x, host_node_size_y, db.num_nodes, row2node_map);
|
||||
|
||||
// construct spaces
|
||||
host_spaces.resize(db.num_movable_nodes);
|
||||
for (int i = 0; i < db.num_sites_y; ++i) {
|
||||
for (unsigned int j = 0; j < row2node_map[i].size(); ++j) {
|
||||
auto const& row2nodes = row2node_map[i];
|
||||
int node_id = row2nodes[j];
|
||||
auto& space = host_spaces[node_id];
|
||||
if (node_id < db.num_movable_nodes) {
|
||||
auto left_bound = db.xl;
|
||||
if (j) {
|
||||
left_bound = host_x[node_id];
|
||||
}
|
||||
space.xl = ceilDiv(left_bound - db.xl, db.site_width) * db.site_width + db.xl;
|
||||
|
||||
auto right_bound = db.xh;
|
||||
if (j + 1 < row2nodes.size()) {
|
||||
int right_node_id = row2nodes[j + 1];
|
||||
right_bound = min(right_bound, host_x[right_node_id]);
|
||||
}
|
||||
space.xh = std::floor(right_bound);
|
||||
space.xh = floorDiv(space.xh - db.xl, db.site_width) * db.site_width + db.xl;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void independentSetMatchingCUDA(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int set_size, int max_iters) {
|
||||
cudaSetDevice(at_db.node_size_x.get_device());
|
||||
DetailedPlaceData db(at_db);
|
||||
db.set_num_bins(num_bins_x, num_bins_y);
|
||||
// fix random seed
|
||||
std::srand(1000);
|
||||
|
||||
IndependentSetMatchingState<float> state;
|
||||
|
||||
// initialize host database
|
||||
DetailedPlaceCPUDB<float> host_db;
|
||||
init_cpu_db(db, host_db);
|
||||
|
||||
state.batch_size = batch_size;
|
||||
state.set_size = set_size;
|
||||
state.cost_matrix_size = state.set_size * state.set_size;
|
||||
state.num_bins = db.num_bins_x * db.num_bins_y;
|
||||
state.num_moved = 0;
|
||||
state.large_number = ((db.xh - db.xl) + (db.yh - db.yl)) * set_size;
|
||||
state.skip_threshold = ((db.xh - db.xl) + (db.yh - db.yl)) * 0.01;
|
||||
state.auction_max_eps = 10.0;
|
||||
state.auction_min_eps = 1.0;
|
||||
state.auction_factor = 0.1;
|
||||
state.auction_max_iterations = 9999;
|
||||
|
||||
checkCuda(cudaMemcpy(host_db.x.data(), db.x, sizeof(float) * db.num_nodes, cudaMemcpyDeviceToHost));
|
||||
checkCuda(cudaMemcpy(host_db.y.data(), db.y, sizeof(float) * db.num_nodes, cudaMemcpyDeviceToHost));
|
||||
std::vector<Space<float>> host_spaces(db.num_movable_nodes);
|
||||
construct_spaces(db,
|
||||
host_db.x.data(),
|
||||
host_db.y.data(),
|
||||
host_db.node_size_x.data(),
|
||||
host_db.node_size_y.data(),
|
||||
host_spaces,
|
||||
db.num_threads);
|
||||
|
||||
// initialize cuda state
|
||||
|
||||
allocateCopyCuda(state.spaces, host_spaces.data(), db.num_movable_nodes);
|
||||
allocateCuda(state.ordered_nodes, db.num_movable_nodes, int);
|
||||
iota<<<ceilDiv(db.num_movable_nodes, 512), 512>>>(state.ordered_nodes, db.num_movable_nodes);
|
||||
allocateCuda(state.independent_sets, state.batch_size * state.set_size, int);
|
||||
allocateCuda(state.independent_set_sizes, state.batch_size, int);
|
||||
allocateCuda(state.selected_maximal_independent_set, db.num_movable_nodes, int);
|
||||
allocateCuda(state.select_scratch, db.num_movable_nodes, int);
|
||||
allocateCuda(state.device_num_selected, 1, int);
|
||||
allocateCuda(state.orig_x, state.batch_size * state.set_size, float);
|
||||
allocateCuda(state.orig_y, state.batch_size * state.set_size, float);
|
||||
allocateCuda(state.orig_spaces, state.batch_size * state.set_size, Space<float>);
|
||||
allocateCuda(state.selected_markers, db.num_nodes, int);
|
||||
allocateCuda(state.dependent_markers, db.num_nodes, unsigned char);
|
||||
allocateCuda(state.independent_set_empty_flag, 1, int);
|
||||
allocateCuda(state.cost_matrices,
|
||||
state.batch_size * state.set_size * state.set_size,
|
||||
typename IndependentSetMatchingState<float>::cost_type);
|
||||
allocateCuda(state.cost_matrices_copy,
|
||||
state.batch_size * state.set_size * state.set_size,
|
||||
typename IndependentSetMatchingState<float>::cost_type);
|
||||
allocateCuda(state.solutions, state.batch_size * state.set_size, int);
|
||||
allocateCuda(
|
||||
state.orig_costs, state.batch_size * state.set_size, typename IndependentSetMatchingState<float>::cost_type);
|
||||
allocateCuda(state.solution_costs,
|
||||
state.batch_size * state.set_size,
|
||||
typename IndependentSetMatchingState<float>::cost_type);
|
||||
allocateCuda(state.net_hpwls, db.num_nets, typename std::remove_pointer<decltype(state.net_hpwls)>::type);
|
||||
allocateCopyCuda(state.device_num_moved, &state.num_moved, 1);
|
||||
|
||||
init_auction<float>(state.batch_size, state.set_size, state.auction_scratch, state.stop_flags);
|
||||
|
||||
Shuffler<int, unsigned int> shuffler(2023ULL, state.ordered_nodes, db.num_movable_nodes);
|
||||
|
||||
// initialize host state
|
||||
IndependentSetMatchingCPUState<float> host_state;
|
||||
init_cpu_state(db, state, host_state);
|
||||
|
||||
// initialize kmeans state
|
||||
KMeansState<float> kmeans_state;
|
||||
init_kmeans(db, state, kmeans_state);
|
||||
|
||||
std::vector<float> hpwls(max_iters + 1);
|
||||
hpwls[0] = compute_total_hpwl(db, db.x, db.y, state.net_hpwls);
|
||||
logger.info("initial hpwl %g", hpwls[0]);
|
||||
for (int iter = 0; iter < max_iters; ++iter) {
|
||||
shuffler();
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
|
||||
maximal_independent_set(db, state);
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
|
||||
collect_independent_sets(db, state, kmeans_state, host_db, host_state);
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
|
||||
cost_matrix_construction(db, state);
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
|
||||
// solve independent sets
|
||||
// print_cost_matrix<<<1, 1>>>(state.cost_matrices + state.cost_matrix_size*3, state.set_size, 0);
|
||||
linear_assignment_auction(state.cost_matrices,
|
||||
state.solutions,
|
||||
state.num_independent_sets,
|
||||
state.set_size,
|
||||
state.auction_scratch,
|
||||
state.stop_flags,
|
||||
state.auction_max_eps,
|
||||
state.auction_min_eps,
|
||||
state.auction_factor,
|
||||
state.auction_max_iterations);
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
// print_solution<<<1, 1>>>(state.solutions + state.set_size*3, state.set_size);
|
||||
|
||||
// apply solutions
|
||||
apply_solution(db, state);
|
||||
checkCuda(cudaDeviceSynchronize());
|
||||
|
||||
hpwls[iter + 1] = compute_total_hpwl(db, db.x, db.y, state.net_hpwls);
|
||||
if ((iter % (max(max_iters / 10, 1))) == 0 || iter + 1 == max_iters) {
|
||||
logger.info("iteration %d, target hpwl %g, delta %g(%g%%), %d independent sets, moved %g%% cells",
|
||||
iter,
|
||||
hpwls[iter + 1],
|
||||
hpwls[iter + 1] - hpwls[0],
|
||||
(hpwls[iter + 1] - hpwls[0]) / hpwls[0] * 100,
|
||||
state.num_independent_sets,
|
||||
state.num_moved / (double)db.num_movable_nodes * 100);
|
||||
}
|
||||
}
|
||||
|
||||
// destroy state
|
||||
cudaFree(state.spaces);
|
||||
cudaFree(state.ordered_nodes);
|
||||
cudaFree(state.independent_sets);
|
||||
cudaFree(state.independent_set_sizes);
|
||||
cudaFree(state.selected_maximal_independent_set);
|
||||
cudaFree(state.select_scratch);
|
||||
cudaFree(state.device_num_selected);
|
||||
cudaFree(state.net_hpwls);
|
||||
cudaFree(state.cost_matrices);
|
||||
cudaFree(state.cost_matrices_copy);
|
||||
cudaFree(state.solutions);
|
||||
cudaFree(state.orig_costs);
|
||||
cudaFree(state.solution_costs);
|
||||
cudaFree(state.orig_x);
|
||||
cudaFree(state.orig_y);
|
||||
cudaFree(state.orig_spaces);
|
||||
cudaFree(state.selected_markers);
|
||||
cudaFree(state.dependent_markers);
|
||||
cudaFree(state.independent_set_empty_flag);
|
||||
cudaFree(state.device_num_moved);
|
||||
destroy_auction(state.auction_scratch, state.stop_flags);
|
||||
destroy_kmeans(kmeans_state);
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
15
cpp_to_py/gpudp/dp/ism/adjust_pos.cuh
Normal file
15
cpp_to_py/gpudp/dp/ism/adjust_pos.cuh
Normal file
@ -0,0 +1,15 @@
|
||||
#pragma once
|
||||
|
||||
#include "gpudp/dp/detailed_place_db.cuh"
|
||||
|
||||
namespace dp {
|
||||
|
||||
template <typename T>
|
||||
__host__ __device__ bool adjust_pos(T& x, T width, const Space<T>& space) {
|
||||
// the order is very tricky for numerical stability
|
||||
x = min(x, space.xh - width);
|
||||
x = max(x, space.xl);
|
||||
return width + space.xl <= space.xh;
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
281
cpp_to_py/gpudp/dp/ism/apply_solution.cuh
Normal file
281
cpp_to_py/gpudp/dp/ism/apply_solution.cuh
Normal file
@ -0,0 +1,281 @@
|
||||
#pragma once
|
||||
|
||||
#include "adjust_pos.cuh"
|
||||
#include "gpudp/dp/detailed_place_db.cuh"
|
||||
|
||||
namespace dp {
|
||||
|
||||
template <typename T>
|
||||
__global__ void copy_orig_cost_kernel(const T* cost_matrices, const char* stop_flags, const int set_size, T* costs) {
|
||||
int i = blockIdx.x; // set
|
||||
|
||||
if (stop_flags[i]) {
|
||||
auto cost_matrix = cost_matrices + i * set_size * set_size;
|
||||
auto cost = costs + i * set_size;
|
||||
for (int j = threadIdx.x; j < set_size; j += blockDim.x) {
|
||||
cost[j] = cost_matrix[j * set_size + j];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
__global__ void copy_solution_cost_kernel(
|
||||
const T* cost_matrices, const char* stop_flags, const int* solutions, const int set_size, T* costs) {
|
||||
int i = blockIdx.x; // set
|
||||
|
||||
if (stop_flags[i]) {
|
||||
auto cost_matrix = cost_matrices + i * set_size * set_size;
|
||||
auto cost = costs + i * set_size;
|
||||
auto solution = solutions + i * set_size;
|
||||
for (int j = threadIdx.x; j < set_size; j += blockDim.x) {
|
||||
int sol_k = solution[j];
|
||||
cost[j] = cost_matrix[j * set_size + sol_k];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, int BlockDim>
|
||||
__global__ void block_reduce_sum(T* costs, const char* stop_flags, int batch_size, int set_size) {
|
||||
int bid = blockIdx.x; // set
|
||||
int tid = threadIdx.x;
|
||||
|
||||
if (stop_flags[bid]) {
|
||||
// Specialize BlockReduce for a 1D block of BlockDim threads on type int
|
||||
typedef cub::BlockReduce<T, BlockDim> BlockReduce;
|
||||
// Allocate shared memory for BlockReduce
|
||||
__shared__ typename BlockReduce::TempStorage temp_storage;
|
||||
// Obtain a segment of consecutive items that are blocked across threads
|
||||
int thread_data[1];
|
||||
|
||||
thread_data[0] = costs[bid * set_size + tid];
|
||||
|
||||
__syncthreads();
|
||||
|
||||
// Compute the block-wide sum for thread0
|
||||
T aggregate = BlockReduce(temp_storage).Sum(thread_data);
|
||||
|
||||
__syncthreads();
|
||||
|
||||
if (tid == 0) {
|
||||
costs[bid * set_size] = aggregate;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
void compute_costs(const char* stop_flags, const int batch_size, const int set_size, T* costs) {
|
||||
switch (set_size) {
|
||||
case 2:
|
||||
block_reduce_sum<T, 2><<<batch_size, 2>>>(costs, stop_flags, batch_size, set_size);
|
||||
break;
|
||||
case 4:
|
||||
block_reduce_sum<T, 4><<<batch_size, 4>>>(costs, stop_flags, batch_size, set_size);
|
||||
break;
|
||||
case 8:
|
||||
block_reduce_sum<T, 8><<<batch_size, 8>>>(costs, stop_flags, batch_size, set_size);
|
||||
break;
|
||||
case 16:
|
||||
block_reduce_sum<T, 16><<<batch_size, 16>>>(costs, stop_flags, batch_size, set_size);
|
||||
break;
|
||||
case 32:
|
||||
block_reduce_sum<T, 32><<<batch_size, 32>>>(costs, stop_flags, batch_size, set_size);
|
||||
break;
|
||||
case 64:
|
||||
block_reduce_sum<T, 64><<<batch_size, 64>>>(costs, stop_flags, batch_size, set_size);
|
||||
break;
|
||||
case 128:
|
||||
block_reduce_sum<T, 128><<<batch_size, 128>>>(costs, stop_flags, batch_size, set_size);
|
||||
break;
|
||||
case 256:
|
||||
block_reduce_sum<T, 256><<<batch_size, 256>>>(costs, stop_flags, batch_size, set_size);
|
||||
break;
|
||||
case 512:
|
||||
block_reduce_sum<T, 512><<<batch_size, 512>>>(costs, stop_flags, batch_size, set_size);
|
||||
break;
|
||||
case 1024:
|
||||
block_reduce_sum<T, 1024><<<batch_size, 1024>>>(costs, stop_flags, batch_size, set_size);
|
||||
break;
|
||||
default:
|
||||
assert_msg(0, "unsupported set size %d", set_size);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
__global__ void print_copy_costs_kernel(const T* costs, int batch_size, int set_size) {
|
||||
if (blockIdx.x == 0 && threadIdx.x == 0) {
|
||||
for (int i = 0; i < batch_size; ++i) {
|
||||
printf("[%d] orig/solution_costs ", i);
|
||||
for (int j = 0; j < set_size; ++j) {
|
||||
printf("%g ", (float)costs[i * set_size + j]);
|
||||
}
|
||||
printf("\n");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
void compute_orig_cost(
|
||||
const T* cost_matrices, const char* stop_flags, const int batch_size, const int set_size, T* costs) {
|
||||
copy_orig_cost_kernel<<<batch_size, set_size>>>(cost_matrices, stop_flags, set_size, costs);
|
||||
compute_costs(stop_flags, batch_size, set_size, costs);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
void compute_solution_cost(const T* cost_matrices,
|
||||
const int* solutions,
|
||||
const char* stop_flags,
|
||||
const int batch_size,
|
||||
const int set_size,
|
||||
T* costs) {
|
||||
copy_solution_cost_kernel<<<batch_size, set_size>>>(cost_matrices, stop_flags, solutions, set_size, costs);
|
||||
// print_copy_costs_kernel<<<1, 1>>>(costs, batch_size, set_size);
|
||||
compute_costs(stop_flags, batch_size, set_size, costs);
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void store_orig_pos_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
|
||||
int i = blockIdx.x; // set
|
||||
const int* __restrict__ independent_set = state.independent_sets + i * state.set_size;
|
||||
auto orig_x = state.orig_x + i * state.set_size;
|
||||
auto orig_y = state.orig_y + i * state.set_size;
|
||||
auto orig_spaces = state.orig_spaces + i * state.set_size;
|
||||
for (int j = threadIdx.x; j < state.set_size; j += blockDim.x) {
|
||||
int node_id = independent_set[j];
|
||||
if (node_id < db.num_movable_nodes) {
|
||||
assert(node_id >= 0);
|
||||
orig_x[j] = db.x[node_id];
|
||||
orig_y[j] = db.y[node_id];
|
||||
orig_spaces[j] = state.spaces[node_id];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void move_nodes_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
|
||||
int i = blockIdx.x; // set
|
||||
int idx = i * state.set_size;
|
||||
|
||||
if (state.stop_flags[i]) {
|
||||
// encourage movement
|
||||
if (state.orig_costs[idx] <= state.solution_costs[idx]) {
|
||||
const int* __restrict__ independent_set = state.independent_sets + i * state.set_size;
|
||||
const int* __restrict__ solution = state.solutions + i * state.set_size;
|
||||
const typename IndependentSetMatchingStateType::type* __restrict__ orig_x =
|
||||
state.orig_x + i * state.set_size;
|
||||
const typename IndependentSetMatchingStateType::type* __restrict__ orig_y =
|
||||
state.orig_y + i * state.set_size;
|
||||
const Space<typename IndependentSetMatchingStateType::type>* __restrict__ orig_spaces =
|
||||
state.orig_spaces + i * state.set_size;
|
||||
for (int j = threadIdx.x; j < state.set_size; j += blockDim.x) {
|
||||
int node_id = independent_set[j];
|
||||
int sol_k = solution[j];
|
||||
if (node_id < db.num_movable_nodes) {
|
||||
auto node_width = db.node_size_x[node_id];
|
||||
|
||||
auto& x = db.x[node_id];
|
||||
auto& y = db.y[node_id];
|
||||
auto& space = state.spaces[node_id];
|
||||
if (j != sol_k) {
|
||||
atomicAdd(state.device_num_moved, 1);
|
||||
auto const& orig_space = orig_spaces[sol_k];
|
||||
x = orig_x[sol_k];
|
||||
bool ret = adjust_pos(x, node_width, orig_space);
|
||||
assert(ret);
|
||||
y = orig_y[sol_k];
|
||||
space = orig_space;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename IndependentSetMatchingStateType>
|
||||
__global__ void print_orig_and_solution_costs_kernel(IndependentSetMatchingStateType state) {
|
||||
if (blockIdx.x == 0 && threadIdx.x == 0) {
|
||||
for (int i = 0; i < state.num_independent_sets; ++i) {
|
||||
int stop = state.stop_flags[i];
|
||||
printf("[%d] orig_costs %g, solution_costs %g, delta %g, stop_flag %d\n",
|
||||
i,
|
||||
(float)state.orig_costs[i * state.set_size],
|
||||
(float)state.solution_costs[i * state.set_size],
|
||||
(float)(state.solution_costs[i * state.set_size] - state.orig_costs[i * state.set_size]),
|
||||
stop);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void check_hpwl_kernel(DetailedPlaceDBType db,
|
||||
IndependentSetMatchingStateType state,
|
||||
const int* independent_set) {
|
||||
if (blockIdx.x == 0 && threadIdx.x == 0) {
|
||||
for (int i = 0; i < state.set_size; ++i) {
|
||||
int node_id = independent_set[i];
|
||||
if (node_id < db.num_movable_nodes) {
|
||||
printf("node %d (%g, %g)", node_id, db.x[node_id], db.y[node_id]);
|
||||
typename DetailedPlaceDBType::type target_hpwl = 0;
|
||||
for (int node2pin_id = db.flat_node2pin_start_map[node_id];
|
||||
node2pin_id < db.flat_node2pin_start_map[node_id + 1];
|
||||
++node2pin_id) {
|
||||
int node_pin_id = db.flat_node2pin_map[node2pin_id];
|
||||
int net_id = db.pin2net_map[node_pin_id];
|
||||
if (db.net_mask[net_id]) {
|
||||
{
|
||||
Box<typename DetailedPlaceDBType::type> box;
|
||||
box.xl = db.xh;
|
||||
box.yl = db.yh;
|
||||
box.xh = db.xl;
|
||||
box.yh = db.yl;
|
||||
for (int net2pin_id = db.flat_net2pin_start_map[net_id];
|
||||
net2pin_id < db.flat_net2pin_start_map[net_id + 1];
|
||||
++net2pin_id) {
|
||||
int net_pin_id = db.flat_net2pin_map[net2pin_id];
|
||||
int other_node_id = db.pin2node_map[net_pin_id];
|
||||
auto xxl = db.x[other_node_id] + db.pin_offset_x[net_pin_id];
|
||||
auto yyl = db.y[other_node_id] + db.pin_offset_y[net_pin_id];
|
||||
box.xl = min(box.xl, xxl);
|
||||
box.xh = max(box.xh, xxl);
|
||||
box.yl = min(box.yl, yyl);
|
||||
box.yh = max(box.yh, yyl);
|
||||
}
|
||||
typename DetailedPlaceDBType::type hpwl = box.xh - box.xl + box.yh - box.yl;
|
||||
target_hpwl += hpwl;
|
||||
printf(", net %d hpwl %g", net_id, (double)hpwl);
|
||||
}
|
||||
{
|
||||
auto const& box = state.net_boxes[net_id];
|
||||
typename DetailedPlaceDBType::type xxl = db.x[node_id] + db.pin_offset_x[node_pin_id];
|
||||
typename DetailedPlaceDBType::type yyl = db.y[node_id] + db.pin_offset_y[node_pin_id];
|
||||
typename DetailedPlaceDBType::type bxl = min(box.xl, xxl);
|
||||
typename DetailedPlaceDBType::type bxh = max(box.xh, xxl);
|
||||
typename DetailedPlaceDBType::type byl = min(box.yl, yyl);
|
||||
typename DetailedPlaceDBType::type byh = max(box.yh, yyl);
|
||||
typename DetailedPlaceDBType::type hpwl = (bxh - bxl) + (byh - byl);
|
||||
printf(" (%g)", hpwl);
|
||||
}
|
||||
}
|
||||
}
|
||||
printf(", total hpwl %g\n", (double)target_hpwl);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
void apply_solution(DetailedPlaceDBType& db, IndependentSetMatchingStateType& state) {
|
||||
compute_orig_cost(
|
||||
state.cost_matrices, state.stop_flags, state.num_independent_sets, state.set_size, state.orig_costs);
|
||||
compute_solution_cost(state.cost_matrices,
|
||||
state.solutions,
|
||||
state.stop_flags,
|
||||
state.num_independent_sets,
|
||||
state.set_size,
|
||||
state.solution_costs);
|
||||
|
||||
store_orig_pos_kernel<<<state.num_independent_sets, state.set_size>>>(db, state);
|
||||
move_nodes_kernel<<<state.num_independent_sets, state.set_size>>>(db, state);
|
||||
checkCuda(cudaMemcpy(&state.num_moved, state.device_num_moved, sizeof(int), cudaMemcpyDeviceToHost));
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
202
cpp_to_py/gpudp/dp/ism/auction.cuh
Normal file
202
cpp_to_py/gpudp/dp/ism/auction.cuh
Normal file
@ -0,0 +1,202 @@
|
||||
#pragma once
|
||||
|
||||
#include "gpudp/dp/detailed_place_db.cuh"
|
||||
|
||||
namespace dp {
|
||||
|
||||
#define BIG_NEGATIVE -9999999
|
||||
#define MAX_MINIBATCH 64
|
||||
|
||||
template <typename T>
|
||||
inline void init_auction(const int num_graphs, const int num_nodes, char*& scratch, char*& stop_flags) {
|
||||
checkCuda(cudaMalloc(
|
||||
&scratch,
|
||||
num_graphs * (3 * num_nodes + 1) * sizeof(int) + num_graphs * (num_nodes * num_nodes + num_nodes) * sizeof(T)));
|
||||
allocateCuda(stop_flags, num_graphs, char);
|
||||
}
|
||||
|
||||
inline void destroy_auction(char* scratch, char* stop_flags) {
|
||||
cudaFree(scratch);
|
||||
cudaFree(stop_flags);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
__global__ void __launch_bounds__(256, 4) linear_assignment_auction_kernel(const int num_nodes,
|
||||
const T* __restrict__ data_ptr,
|
||||
int* person2item_ptr,
|
||||
int* item2person_ptr,
|
||||
T* bids_ptr,
|
||||
T* prices_ptr,
|
||||
int* sbids_ptr,
|
||||
char* stop_flag_ptr,
|
||||
const float auction_max_eps,
|
||||
const float auction_min_eps,
|
||||
const float auction_factor,
|
||||
const int max_iterations) {
|
||||
const int batch_id = blockIdx.x;
|
||||
const int node_id = threadIdx.x;
|
||||
__shared__ float auction_eps;
|
||||
__shared__ int num_iteration;
|
||||
__shared__ int num_assigned;
|
||||
extern __shared__ T s_prices[];
|
||||
|
||||
if (node_id == 0) {
|
||||
auction_eps = auction_max_eps;
|
||||
num_iteration = 0;
|
||||
}
|
||||
|
||||
const T* __restrict__ data = data_ptr + batch_id * num_nodes * num_nodes;
|
||||
int* person2item = person2item_ptr + batch_id * num_nodes;
|
||||
int* item2person = item2person_ptr + batch_id * num_nodes;
|
||||
T* bids = bids_ptr + batch_id * num_nodes * num_nodes;
|
||||
int* sbids = sbids_ptr + batch_id * num_nodes;
|
||||
T* prices = prices_ptr + batch_id * num_nodes;
|
||||
char* stop_flag = stop_flag_ptr + batch_id;
|
||||
|
||||
__syncthreads();
|
||||
|
||||
while (auction_eps >= auction_min_eps && num_iteration < max_iterations) {
|
||||
// clear num_assigned
|
||||
if (node_id == 0) {
|
||||
num_assigned = 0;
|
||||
}
|
||||
|
||||
// pre-init
|
||||
for (int i = node_id; i < num_nodes; i += blockDim.x) {
|
||||
person2item[i] = -1;
|
||||
item2person[i] = -1;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// start iterative solving
|
||||
while (num_assigned < num_nodes && num_iteration < max_iterations) {
|
||||
// phase 1: init bid and bids
|
||||
for (int i = node_id; i < num_nodes; i += blockDim.x) {
|
||||
sbids[i] = 0;
|
||||
}
|
||||
for (int i = node_id; i < num_nodes * num_nodes; i += blockDim.x) {
|
||||
bids[i] = 0;
|
||||
}
|
||||
|
||||
// preload price
|
||||
s_prices[node_id] = prices[node_id];
|
||||
__syncthreads();
|
||||
|
||||
// phase 2: bidding
|
||||
if (person2item[node_id] == -1) {
|
||||
T top1_val = BIG_NEGATIVE;
|
||||
T top2_val = BIG_NEGATIVE;
|
||||
int top1_col;
|
||||
T tmp_val;
|
||||
|
||||
for (int col = 0; col < num_nodes; col++) {
|
||||
tmp_val = data[node_id * num_nodes + col];
|
||||
if (tmp_val < 0) {
|
||||
continue;
|
||||
}
|
||||
tmp_val = tmp_val - s_prices[col];
|
||||
if (tmp_val >= top1_val) {
|
||||
top2_val = top1_val;
|
||||
top1_col = col;
|
||||
top1_val = tmp_val;
|
||||
} else if (tmp_val > top2_val) {
|
||||
top2_val = tmp_val;
|
||||
}
|
||||
}
|
||||
if (top2_val == BIG_NEGATIVE) {
|
||||
top2_val = top1_val;
|
||||
}
|
||||
T bid = top1_val - top2_val + auction_eps;
|
||||
bids[num_nodes * top1_col + node_id] = bid;
|
||||
atomicMax(sbids + top1_col, 1);
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
// phase 3 : assignment
|
||||
if (sbids[node_id] != 0) {
|
||||
T high_bid = 0;
|
||||
int high_bidder = -1;
|
||||
|
||||
T tmp_bid = -1;
|
||||
for (int i = 0; i < num_nodes; i++) {
|
||||
tmp_bid = bids[node_id * num_nodes + i];
|
||||
if (tmp_bid > high_bid) {
|
||||
high_bid = tmp_bid;
|
||||
high_bidder = i;
|
||||
}
|
||||
}
|
||||
|
||||
int current_person = item2person[node_id];
|
||||
if (current_person >= 0) {
|
||||
person2item[current_person] = -1;
|
||||
} else {
|
||||
atomicAdd(&num_assigned, 1);
|
||||
}
|
||||
|
||||
prices[node_id] += high_bid;
|
||||
person2item[high_bidder] = node_id;
|
||||
item2person[node_id] = high_bidder;
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// update iteration
|
||||
if (node_id == 0) {
|
||||
num_iteration++;
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
// scale auction_eps
|
||||
if (node_id == 0) {
|
||||
auction_eps *= auction_factor;
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
__syncthreads();
|
||||
// report whether finish solving
|
||||
if (node_id == 0) {
|
||||
*stop_flag = (num_assigned == num_nodes);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
void linear_assignment_auction(const T* cost_matrics,
|
||||
int* solutions,
|
||||
const int num_graphs,
|
||||
const int num_nodes,
|
||||
char* scratch,
|
||||
char* stop_flags,
|
||||
const float auction_max_eps,
|
||||
const float auction_min_eps,
|
||||
const float auction_factor,
|
||||
const int max_iterations) {
|
||||
// get pointers from scratch, size of scratch: num_graphs * (4*num_nodes + num_nodes*num_nodes) * 4 bytes
|
||||
int* person2item = (int*)scratch;
|
||||
int* item2person = person2item + num_graphs * num_nodes;
|
||||
int* sbids = item2person + num_graphs * num_nodes;
|
||||
T* prices = (T*)(sbids + num_graphs * num_nodes);
|
||||
T* bids = prices + num_graphs * num_nodes;
|
||||
|
||||
// init
|
||||
cudaMemsetAsync(prices, 0, num_graphs * num_nodes * sizeof(T));
|
||||
|
||||
// launch solver
|
||||
linear_assignment_auction_kernel<T><<<num_graphs, num_nodes, num_nodes * sizeof(T)>>>(num_nodes,
|
||||
cost_matrics,
|
||||
person2item,
|
||||
item2person,
|
||||
bids,
|
||||
prices,
|
||||
sbids,
|
||||
stop_flags,
|
||||
auction_max_eps,
|
||||
auction_min_eps,
|
||||
auction_factor,
|
||||
max_iterations);
|
||||
cudaDeviceSynchronize();
|
||||
|
||||
// copy solutions
|
||||
cudaMemcpy(solutions, person2item, num_graphs * num_nodes * sizeof(int), cudaMemcpyDeviceToDevice);
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
372
cpp_to_py/gpudp/dp/ism/collect_independent_sets.cuh
Normal file
372
cpp_to_py/gpudp/dp/ism/collect_independent_sets.cuh
Normal file
@ -0,0 +1,372 @@
|
||||
#pragma once
|
||||
|
||||
#include "gpudp/dp/detailed_place_db.cuh"
|
||||
#include "cpu_state.cuh"
|
||||
|
||||
namespace dp {
|
||||
|
||||
#define DETERMINISTIC
|
||||
|
||||
template <typename T>
|
||||
struct KMeansState {
|
||||
#ifdef DETERMINISTIC
|
||||
typedef long long int coordinate_type;
|
||||
#else
|
||||
typedef T coordinate_type;
|
||||
#endif
|
||||
|
||||
coordinate_type* centers_x; // To ensure determinism, use fixed point numbers
|
||||
coordinate_type* centers_y;
|
||||
T* weights;
|
||||
int* partition_sizes;
|
||||
int* node2centers_map;
|
||||
int num_seeds;
|
||||
|
||||
#ifdef DETERMINISTIC
|
||||
static constexpr T scale = 16384;
|
||||
#else
|
||||
static constexpr T scale = 1;
|
||||
#endif
|
||||
};
|
||||
|
||||
/// @brief A wrapper for atomicAdd
|
||||
/// As CUDA atomicAdd does not support for long long int, using unsigned long long int is equivalent.
|
||||
template <typename T>
|
||||
inline __device__ T atomicAddWrapper(T* address, T value) {
|
||||
return atomicAdd(address, value);
|
||||
}
|
||||
|
||||
/// @brief Template specialization for long long int
|
||||
template <>
|
||||
inline __device__ long long int atomicAddWrapper<long long int>(long long int* address, long long int value) {
|
||||
return atomicAdd((unsigned long long int*)address, (unsigned long long int)value);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
__global__ void fill_array_kernel(T* array, int n, T v) {
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < n) {
|
||||
array[i] = v;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
inline void fill_array(T* array, int n, T v) {
|
||||
fill_array_kernel<<<ceilDiv(n, 512), 512>>>(array, n, v);
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
void init_kmeans(const DetailedPlaceDBType& db,
|
||||
const IndependentSetMatchingStateType& state,
|
||||
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
|
||||
typedef typename DetailedPlaceDBType::type T;
|
||||
|
||||
allocateCuda(kmeans_state.centers_x,
|
||||
state.batch_size,
|
||||
typename KMeansState<typename DetailedPlaceDBType::type>::coordinate_type);
|
||||
allocateCuda(kmeans_state.centers_y,
|
||||
state.batch_size,
|
||||
typename KMeansState<typename DetailedPlaceDBType::type>::coordinate_type);
|
||||
allocateCuda(kmeans_state.weights, state.batch_size, T);
|
||||
allocateCuda(kmeans_state.partition_sizes, state.batch_size, int);
|
||||
allocateCuda(kmeans_state.node2centers_map, db.num_movable_nodes, int);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
void destroy_kmeans(KMeansState<T>& kmeans_state) {
|
||||
cudaFree(kmeans_state.centers_x);
|
||||
cudaFree(kmeans_state.centers_y);
|
||||
cudaFree(kmeans_state.weights);
|
||||
cudaFree(kmeans_state.partition_sizes);
|
||||
cudaFree(kmeans_state.node2centers_map);
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
void prepare_kmeans(const DetailedPlaceDBType& db,
|
||||
const IndependentSetMatchingStateType& state,
|
||||
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
|
||||
// need at least 1 seed; otherwise, it will cause problem in later kernels
|
||||
kmeans_state.num_seeds = max(min(state.num_selected / state.set_size, state.batch_size), 1);
|
||||
// set weights to 1.0
|
||||
fill_array(kmeans_state.weights, kmeans_state.num_seeds, (typename DetailedPlaceDBType::type)1.0);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
__inline__ __device__ T kmeans_distance(T node_x, T node_y, T center_x, T center_y) {
|
||||
T distance = fabs(node_x - center_x) + fabs(node_y - center_y);
|
||||
return distance;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
struct ItemWithIndex {
|
||||
T value;
|
||||
int index;
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
struct ReduceMinOP {
|
||||
__host__ __device__ ItemWithIndex<T> operator()(const ItemWithIndex<T>& a, const ItemWithIndex<T>& b) const {
|
||||
return (a.value < b.value) ? a : b;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType, int ThreadsPerBlock = 128>
|
||||
__global__ void kmeans_find_centers_kernel(DetailedPlaceDBType db,
|
||||
IndependentSetMatchingStateType state,
|
||||
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
|
||||
assert(blockIdx.x < state.num_selected);
|
||||
int node_id = state.selected_maximal_independent_set[blockIdx.x];
|
||||
assert(node_id < db.num_movable_nodes);
|
||||
auto node_x = db.x[node_id];
|
||||
auto node_y = db.y[node_id];
|
||||
|
||||
typedef cub::BlockReduce<ItemWithIndex<typename DetailedPlaceDBType::type>, ThreadsPerBlock> BlockReduce;
|
||||
|
||||
__shared__ typename BlockReduce::TempStorage temp_storage;
|
||||
|
||||
ItemWithIndex<typename DetailedPlaceDBType::type> thread_data;
|
||||
|
||||
thread_data.value = cuda::numeric_limits<typename DetailedPlaceDBType::type>::max();
|
||||
thread_data.index = cuda::numeric_limits<int>::max();
|
||||
for (int center_id = threadIdx.x; center_id < kmeans_state.num_seeds; center_id += ThreadsPerBlock) {
|
||||
assert(center_id < kmeans_state.num_seeds);
|
||||
// scale back to floating point numbers
|
||||
typename DetailedPlaceDBType::type center_x =
|
||||
kmeans_state.centers_x[center_id] / KMeansState<typename DetailedPlaceDBType::type>::scale;
|
||||
typename DetailedPlaceDBType::type center_y =
|
||||
kmeans_state.centers_y[center_id] / KMeansState<typename DetailedPlaceDBType::type>::scale;
|
||||
typename DetailedPlaceDBType::type weight = kmeans_state.weights[center_id];
|
||||
|
||||
typename DetailedPlaceDBType::type distance = kmeans_distance(node_x, node_y, center_x, center_y) * weight;
|
||||
if (distance < thread_data.value) {
|
||||
thread_data.value = distance;
|
||||
thread_data.index = center_id;
|
||||
}
|
||||
}
|
||||
if (threadIdx.x < kmeans_state.num_seeds) {
|
||||
assert(thread_data.index < kmeans_state.num_seeds);
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
// Compute the block-wide max for thread0
|
||||
ItemWithIndex<typename DetailedPlaceDBType::type> aggregate =
|
||||
BlockReduce(temp_storage)
|
||||
.Reduce(thread_data, ReduceMinOP<typename DetailedPlaceDBType::type>(), kmeans_state.num_seeds);
|
||||
|
||||
__syncthreads();
|
||||
|
||||
if (threadIdx.x == 0) {
|
||||
assert(blockIdx.x < state.num_selected);
|
||||
kmeans_state.node2centers_map[blockIdx.x] = aggregate.index;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void init_kmeans_seeds_kernel(DetailedPlaceDBType db,
|
||||
IndependentSetMatchingStateType state,
|
||||
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < kmeans_state.num_seeds) {
|
||||
assert(db.num_movable_nodes - i - 1 < db.num_movable_nodes && db.num_movable_nodes - i - 1 >= 0);
|
||||
int random_number = state.ordered_nodes[db.num_movable_nodes - i - 1];
|
||||
random_number = random_number % state.num_selected;
|
||||
int node_id = state.selected_maximal_independent_set[random_number];
|
||||
assert(node_id < db.num_movable_nodes);
|
||||
// scale up for fixed point numbers
|
||||
kmeans_state.centers_x[i] = db.x[node_id] * KMeansState<typename DetailedPlaceDBType::type>::scale;
|
||||
kmeans_state.centers_y[i] = db.y[node_id] * KMeansState<typename DetailedPlaceDBType::type>::scale;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
void init_kmeans_seeds(const DetailedPlaceDBType& db,
|
||||
IndependentSetMatchingStateType& state,
|
||||
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
|
||||
init_kmeans_seeds_kernel<<<ceilDiv(kmeans_state.num_seeds, 256), 256>>>(db, state, kmeans_state);
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void reset_kmeans_partition_sizes_kernel(DetailedPlaceDBType db,
|
||||
IndependentSetMatchingStateType state,
|
||||
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < kmeans_state.num_seeds) {
|
||||
kmeans_state.partition_sizes[i] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void compute_kmeans_partition_sizes_kernel(DetailedPlaceDBType db,
|
||||
IndependentSetMatchingStateType state,
|
||||
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < state.num_selected) {
|
||||
int center_id = kmeans_state.node2centers_map[i];
|
||||
assert(center_id < kmeans_state.num_seeds);
|
||||
atomicAdd(kmeans_state.partition_sizes + center_id, 1);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void reset_kmeans_centers_kernel(DetailedPlaceDBType db,
|
||||
IndependentSetMatchingStateType state,
|
||||
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < kmeans_state.num_seeds) {
|
||||
if (kmeans_state.partition_sizes[i]) {
|
||||
kmeans_state.centers_x[i] = 0;
|
||||
kmeans_state.centers_y[i] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void compute_kmeans_centers_sum_kernel(DetailedPlaceDBType db,
|
||||
IndependentSetMatchingStateType state,
|
||||
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < state.num_selected) {
|
||||
int node_id = state.selected_maximal_independent_set[i];
|
||||
int center_id = kmeans_state.node2centers_map[i];
|
||||
assert(center_id < kmeans_state.num_seeds);
|
||||
assert(node_id < db.num_movable_nodes);
|
||||
// scale up for fixed point numbers
|
||||
atomicAddWrapper<typename KMeansState<typename DetailedPlaceDBType::type>::coordinate_type>(
|
||||
kmeans_state.centers_x + center_id, db.x[node_id] * KMeansState<typename DetailedPlaceDBType::type>::scale);
|
||||
atomicAddWrapper<typename KMeansState<typename DetailedPlaceDBType::type>::coordinate_type>(
|
||||
kmeans_state.centers_y + center_id, db.y[node_id] * KMeansState<typename DetailedPlaceDBType::type>::scale);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void compute_kmeans_centers_div_kernel(DetailedPlaceDBType db,
|
||||
IndependentSetMatchingStateType state,
|
||||
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < kmeans_state.num_seeds) {
|
||||
int s = kmeans_state.partition_sizes[i];
|
||||
if (s) {
|
||||
kmeans_state.centers_x[i] /= s;
|
||||
kmeans_state.centers_y[i] /= s;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
void kmeans_update_centers(const DetailedPlaceDBType& db,
|
||||
IndependentSetMatchingStateType& state,
|
||||
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
|
||||
// reset partition_sizes to 0
|
||||
reset_kmeans_partition_sizes_kernel<<<ceilDiv(kmeans_state.num_seeds, 256), 256>>>(db, state, kmeans_state);
|
||||
// compute partition sizes
|
||||
compute_kmeans_partition_sizes_kernel<<<ceilDiv(state.num_selected, 256), 256>>>(db, state, kmeans_state);
|
||||
// reset kmeans centers to 0
|
||||
reset_kmeans_centers_kernel<<<ceilDiv(kmeans_state.num_seeds, 256), 256>>>(db, state, kmeans_state);
|
||||
// compute kmeans centers sum
|
||||
compute_kmeans_centers_sum_kernel<<<ceilDiv(state.num_selected, 256), 256>>>(db, state, kmeans_state);
|
||||
// compute kmeans centers div
|
||||
compute_kmeans_centers_div_kernel<<<ceilDiv(kmeans_state.num_seeds, 256), 256>>>(db, state, kmeans_state);
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void compute_kmeans_weights_kernel(DetailedPlaceDBType db,
|
||||
IndependentSetMatchingStateType state,
|
||||
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < kmeans_state.num_seeds) {
|
||||
int s = kmeans_state.partition_sizes[i];
|
||||
auto& w = kmeans_state.weights[i];
|
||||
if (s > state.set_size) {
|
||||
auto ratio = s / (typename DetailedPlaceDBType::type)state.set_size;
|
||||
ratio = 1.0 + 0.5 * log(ratio);
|
||||
w *= ratio;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
void kmeans_update_weights(const DetailedPlaceDBType& db,
|
||||
IndependentSetMatchingStateType& state,
|
||||
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
|
||||
compute_kmeans_weights_kernel<<<ceilDiv(kmeans_state.num_seeds, 256), 256>>>(db, state, kmeans_state);
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
void kmeans_collect_sets_cuda2cpu(const DetailedPlaceDBType& db,
|
||||
IndependentSetMatchingStateType& state,
|
||||
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
|
||||
std::vector<int> selected_nodes(state.num_selected);
|
||||
checkCuda(cudaMemcpy(selected_nodes.data(),
|
||||
state.selected_maximal_independent_set,
|
||||
sizeof(int) * state.num_selected,
|
||||
cudaMemcpyDeviceToHost));
|
||||
std::vector<int> node2centers_map(state.num_selected);
|
||||
checkCuda(cudaMemcpy(node2centers_map.data(),
|
||||
kmeans_state.node2centers_map,
|
||||
sizeof(int) * state.num_selected,
|
||||
cudaMemcpyDeviceToHost));
|
||||
|
||||
std::vector<int> flat_independent_sets(state.batch_size * state.set_size, std::numeric_limits<int>::max());
|
||||
std::vector<int> independent_set_sizes(state.batch_size, 0);
|
||||
// directly use flat array
|
||||
for (int i = 0; i < state.num_selected; ++i) {
|
||||
int node_id = selected_nodes.at(i);
|
||||
int center_id = node2centers_map.at(i);
|
||||
int& size = independent_set_sizes.at(center_id);
|
||||
if (size < state.set_size) {
|
||||
flat_independent_sets.at(center_id * state.set_size + size) = node_id;
|
||||
++size;
|
||||
}
|
||||
}
|
||||
checkCuda(cudaMemcpy(state.independent_sets,
|
||||
flat_independent_sets.data(),
|
||||
sizeof(int) * state.batch_size * state.set_size,
|
||||
cudaMemcpyHostToDevice));
|
||||
checkCuda(cudaMemcpy(state.independent_set_sizes,
|
||||
independent_set_sizes.data(),
|
||||
sizeof(int) * state.batch_size,
|
||||
cudaMemcpyHostToDevice));
|
||||
|
||||
// statistics
|
||||
logger.debug(
|
||||
"from %d nodes, collect %d sets, avg %d nodes, min/max %d/%d nodes",
|
||||
state.num_selected,
|
||||
state.num_independent_sets,
|
||||
std::accumulate(independent_set_sizes.begin(), independent_set_sizes.begin() + state.num_independent_sets, 0) /
|
||||
state.num_independent_sets,
|
||||
*std::min_element(independent_set_sizes.begin(), independent_set_sizes.begin() + state.num_independent_sets),
|
||||
*std::max_element(independent_set_sizes.begin(), independent_set_sizes.begin() + state.num_independent_sets));
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
void partition_kmeans(const DetailedPlaceDBType& db,
|
||||
IndependentSetMatchingStateType& state,
|
||||
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
|
||||
prepare_kmeans(db, state, kmeans_state);
|
||||
init_kmeans_seeds(db, state, kmeans_state);
|
||||
|
||||
for (int iter = 0; iter < 2; ++iter) {
|
||||
// for each node, find centers
|
||||
kmeans_find_centers_kernel<DetailedPlaceDBType, IndependentSetMatchingStateType, 256>
|
||||
<<<state.num_selected, 256>>>(db, state, kmeans_state);
|
||||
// for each center, adjust itself
|
||||
kmeans_update_centers(db, state, kmeans_state);
|
||||
// for each partition, update weight
|
||||
kmeans_update_weights(db, state, kmeans_state);
|
||||
}
|
||||
|
||||
state.num_independent_sets = kmeans_state.num_seeds;
|
||||
kmeans_collect_sets_cuda2cpu(db, state, kmeans_state);
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
void collect_independent_sets(const DetailedPlaceDBType& db,
|
||||
IndependentSetMatchingStateType& state,
|
||||
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state,
|
||||
DetailedPlaceCPUDB<typename DetailedPlaceDBType::type>& host_db,
|
||||
IndependentSetMatchingCPUState<typename DetailedPlaceDBType::type>& host_state) {
|
||||
partition_kmeans(db, state, kmeans_state);
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
231
cpp_to_py/gpudp/dp/ism/cost_matrix_construction.cuh
Normal file
231
cpp_to_py/gpudp/dp/ism/cost_matrix_construction.cuh
Normal file
@ -0,0 +1,231 @@
|
||||
#pragma once
|
||||
|
||||
#include "adjust_pos.cuh"
|
||||
#include "gpudp/dp/detailed_place_db.cuh"
|
||||
#include "reduce_min.cuh"
|
||||
|
||||
namespace dp {
|
||||
|
||||
#define MAX_NODE_DEGREE 32
|
||||
|
||||
template <typename T>
|
||||
struct SharedBox {
|
||||
T xl;
|
||||
T yl;
|
||||
T xh;
|
||||
T yh;
|
||||
};
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void print_net_boxes_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
|
||||
if (blockIdx.x == 0 && threadIdx.x == 0) {
|
||||
for (int node_id = 0; node_id < db.num_movable_nodes; ++node_id) {
|
||||
if (state.selected_markers[node_id]) {
|
||||
int node2pin_id = db.flat_node2pin_start_map[node_id];
|
||||
const int node2pin_id_end = db.flat_node2pin_start_map[node_id + 1];
|
||||
for (; node2pin_id < node2pin_id_end; ++node2pin_id) {
|
||||
int node_pin_id = db.flat_node2pin_map[node2pin_id];
|
||||
int net_id = db.pin2net_map[node_pin_id];
|
||||
auto const& box = state.net_boxes[net_id];
|
||||
printf("node %d: net %d (%g, %g, %g, %g)\n", node_id, net_id, box.xl, box.yl, box.xh, box.yh);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void compute_cost_matrix_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
|
||||
int i = blockIdx.y; // set
|
||||
int j = blockIdx.x; // node in set
|
||||
const int* __restrict__ independent_set = state.independent_sets + i * state.set_size;
|
||||
auto cost_matrix = state.cost_matrices + i * state.cost_matrix_size + j * state.set_size;
|
||||
__shared__ int node_id;
|
||||
__shared__ typename DetailedPlaceDBType::type node_width;
|
||||
__shared__ SharedBox<typename DetailedPlaceDBType::type> net_boxes[MAX_NODE_DEGREE];
|
||||
__shared__ int node2pin_id_bgn;
|
||||
__shared__ int node2pin_id_end;
|
||||
if (threadIdx.x == 0) {
|
||||
node_id = independent_set[j];
|
||||
node_width = cuda::numeric_limits<typename DetailedPlaceDBType::type>::max();
|
||||
if (node_id < db.num_movable_nodes) {
|
||||
node_width = db.node_size_x[node_id];
|
||||
|
||||
node2pin_id_bgn = db.flat_node2pin_start_map[node_id];
|
||||
node2pin_id_end = db.flat_node2pin_start_map[node_id + 1];
|
||||
node2pin_id_end = min(node2pin_id_bgn + MAX_NODE_DEGREE, node2pin_id_end);
|
||||
|
||||
int idx = 0;
|
||||
for (int node2pin_id = node2pin_id_bgn; node2pin_id < node2pin_id_end; ++node2pin_id, ++idx) {
|
||||
int node_pin_id = db.flat_node2pin_map[node2pin_id];
|
||||
int net_id = db.pin2net_map[node_pin_id];
|
||||
auto& box = net_boxes[idx];
|
||||
box.xl = db.xh;
|
||||
box.yl = db.yh;
|
||||
box.xh = db.xl;
|
||||
box.yh = db.yl;
|
||||
if (db.net_mask[net_id]) {
|
||||
int net2pin_id_bgn = db.flat_net2pin_start_map[net_id];
|
||||
int net2pin_id_end = db.flat_net2pin_start_map[net_id + 1];
|
||||
for (int net2pin_id = net2pin_id_bgn; net2pin_id < net2pin_id_end; ++net2pin_id) {
|
||||
int net_pin_id = db.flat_net2pin_map[net2pin_id];
|
||||
int other_node_id = db.pin2node_map[net_pin_id];
|
||||
if (other_node_id != node_id) {
|
||||
typename DetailedPlaceDBType::type xxl = db.x[other_node_id] + db.pin_offset_x[net_pin_id];
|
||||
typename DetailedPlaceDBType::type yyl = db.y[other_node_id] + db.pin_offset_y[net_pin_id];
|
||||
box.xl = min(box.xl, xxl);
|
||||
box.xh = max(box.xh, xxl);
|
||||
box.yl = min(box.yl, yyl);
|
||||
box.yh = max(box.yh, yyl);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
for (int k = threadIdx.x; k < state.set_size; k += blockDim.x) // pos in set
|
||||
{
|
||||
int pos_id = independent_set[k];
|
||||
auto& cost = cost_matrix[k]; // row major
|
||||
if (node_id < db.num_movable_nodes && pos_id < db.num_movable_nodes) {
|
||||
typename DetailedPlaceDBType::type target_x = db.x[pos_id];
|
||||
typename DetailedPlaceDBType::type target_y = db.y[pos_id];
|
||||
auto const& target_space = state.spaces[pos_id];
|
||||
int target_hpwl = 0;
|
||||
if (adjust_pos(target_x, node_width, target_space)) {
|
||||
// consider FENCE region
|
||||
if (db.num_regions && !db.inside_fence(node_id, target_x, target_y)) {
|
||||
cost = BIG_NEGATIVE; // as a marker for post processing
|
||||
} else {
|
||||
int idx = 0;
|
||||
for (int node2pin_id = node2pin_id_bgn; node2pin_id < node2pin_id_end; ++node2pin_id, ++idx) {
|
||||
int node_pin_id = db.flat_node2pin_map[node2pin_id];
|
||||
int net_id = db.pin2net_map[node_pin_id];
|
||||
auto const& box = net_boxes[idx];
|
||||
if (db.net_mask[net_id]) {
|
||||
typename DetailedPlaceDBType::type xxl = target_x + db.pin_offset_x[node_pin_id];
|
||||
typename DetailedPlaceDBType::type yyl = target_y + db.pin_offset_y[node_pin_id];
|
||||
typename DetailedPlaceDBType::type bxl = min(box.xl, xxl);
|
||||
typename DetailedPlaceDBType::type bxh = max(box.xh, xxl);
|
||||
typename DetailedPlaceDBType::type byl = min(box.yl, yyl);
|
||||
typename DetailedPlaceDBType::type byh = max(box.yh, yyl);
|
||||
target_hpwl += (bxh - bxl) + (byh - byl);
|
||||
}
|
||||
}
|
||||
// target_hpwl = target_hpwl*db.row_height + (abs(target_x-node_x) + abs(target_y-node_y));
|
||||
// row major
|
||||
cost = target_hpwl;
|
||||
}
|
||||
} else {
|
||||
cost = BIG_NEGATIVE; // as a marker for post processing
|
||||
}
|
||||
} else {
|
||||
// cost = state.large_number*(j != k);
|
||||
cost = BIG_NEGATIVE; // as a marker for post processing
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// @brief change from minimization problem for maximization problem with non-negative edge weights
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void postprocess_cost_matrix_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
|
||||
int i = blockIdx.y; // set
|
||||
int j = blockIdx.x; // node in set
|
||||
const int* __restrict__ independent_set = state.independent_sets + i * state.set_size;
|
||||
auto cost_matrix = state.cost_matrices + i * state.cost_matrix_size + j * state.set_size;
|
||||
auto max_cost = state.cost_matrices_copy[i * state.cost_matrix_size];
|
||||
for (int k = threadIdx.x; k < state.set_size; k += blockDim.x) // pos in set
|
||||
{
|
||||
int node_id = independent_set[j];
|
||||
int pos_id = independent_set[k];
|
||||
auto& cost = cost_matrix[k]; // row major
|
||||
if (node_id < db.num_movable_nodes && pos_id < db.num_movable_nodes) {
|
||||
if (cost >= 0) {
|
||||
cost = max_cost - cost;
|
||||
}
|
||||
// cost < 0 is already assigned to negative
|
||||
} else if (j == k) {
|
||||
cost = max_cost; // dummy cells or positions
|
||||
}
|
||||
// j != k is already assigned to negative
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
__global__ void print_cost_matrix_kernel(const T* cost_matrix, int set_size) {
|
||||
unsigned int tid = threadIdx.x;
|
||||
unsigned int bid = blockIdx.x;
|
||||
if (tid == 0 && bid == 0) {
|
||||
printf("[%dx%d]\n", set_size, set_size);
|
||||
for (int r = 0; r < set_size; ++r) {
|
||||
for (int c = 0; c < set_size; ++c) {
|
||||
auto cost = cost_matrix[r * set_size + c];
|
||||
if (cost == BIG_NEGATIVE) {
|
||||
printf("X ");
|
||||
} else {
|
||||
printf("%g ", (double)cost);
|
||||
}
|
||||
}
|
||||
printf("\n");
|
||||
}
|
||||
printf("\n");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename IndependentSetMatchingStateType>
|
||||
__global__ void print_max_cost_kernel(IndependentSetMatchingStateType state) {
|
||||
unsigned int tid = threadIdx.x;
|
||||
unsigned int bid = blockIdx.x;
|
||||
if (tid == 0 && bid == 0) {
|
||||
printf("[%d]\n", state.num_independent_sets);
|
||||
for (int i = 0; i < state.num_independent_sets; ++i) {
|
||||
printf("%g ", (double)state.cost_matrices_copy[i * state.cost_matrix_size]);
|
||||
}
|
||||
printf("\n");
|
||||
}
|
||||
}
|
||||
|
||||
template <typename IndependentSetMatchingStateType>
|
||||
__global__ void check_cost_matrices_kernel(IndependentSetMatchingStateType state) {
|
||||
unsigned int tid = threadIdx.x;
|
||||
unsigned int bid = blockIdx.x;
|
||||
if (tid == 0 && bid == 0) {
|
||||
for (int i = 0; i < state.num_independent_sets; ++i) {
|
||||
for (int j = 0; j < state.cost_matrix_size; ++j) {
|
||||
auto cost = state.cost_matrices[i * state.cost_matrix_size + j];
|
||||
assert(cost == cuda::numeric_limits<typename IndependentSetMatchingStateType::cost_type>::lowest() ||
|
||||
cost >= 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
struct CompareCost {
|
||||
__host__ __device__ bool operator()(T cost1, T cost2) const { return cost1 > cost2; }
|
||||
};
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
void cost_matrix_construction(const DetailedPlaceDBType& db, IndependentSetMatchingStateType& state) {
|
||||
dim3 grid(state.set_size, state.num_independent_sets, 1);
|
||||
compute_cost_matrix_kernel<<<grid, state.set_size>>>(db, state);
|
||||
|
||||
checkCuda(cudaMemcpy(state.cost_matrices_copy,
|
||||
state.cost_matrices,
|
||||
sizeof(typename IndependentSetMatchingStateType::cost_type) * state.num_independent_sets *
|
||||
state.cost_matrix_size,
|
||||
cudaMemcpyDeviceToDevice));
|
||||
typename IndependentSetMatchingStateType::cost_type ref = 0;
|
||||
reduce_2d(state.cost_matrices_copy,
|
||||
state.num_independent_sets,
|
||||
state.cost_matrix_size,
|
||||
ref,
|
||||
CompareCost<typename IndependentSetMatchingStateType::cost_type>());
|
||||
|
||||
postprocess_cost_matrix_kernel<<<grid, state.set_size>>>(db, state);
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
92
cpp_to_py/gpudp/dp/ism/cpu_state.cuh
Normal file
92
cpp_to_py/gpudp/dp/ism/cpu_state.cuh
Normal file
@ -0,0 +1,92 @@
|
||||
#pragma once
|
||||
|
||||
#include "gpudp/dp/detailed_place_db.cuh"
|
||||
#include "diamond_search.h"
|
||||
|
||||
namespace dp {
|
||||
|
||||
template <typename T>
|
||||
struct DetailedPlaceCPUDB {
|
||||
typedef T type;
|
||||
int num_movable_nodes;
|
||||
int num_bins_x;
|
||||
int num_bins_y;
|
||||
T bin_size_x;
|
||||
T bin_size_y;
|
||||
T xl, yl, xh, yh;
|
||||
std::vector<T> node_size_x;
|
||||
std::vector<T> node_size_y;
|
||||
std::vector<T> x;
|
||||
std::vector<T> y;
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
struct IndependentSetMatchingCPUState {
|
||||
typedef T type;
|
||||
int batch_size;
|
||||
int set_size;
|
||||
int grid_size;
|
||||
int max_diamond_search_sequence;
|
||||
int num_independent_sets;
|
||||
std::vector<std::vector<int>> independent_sets;
|
||||
std::vector<int> flat_independent_sets; ///< flat version of storage
|
||||
std::vector<int> independent_set_sizes; ///< size of each set
|
||||
std::vector<int> selected_nodes;
|
||||
std::vector<unsigned char> selected_markers;
|
||||
std::vector<int> ordered_nodes;
|
||||
std::vector<BinMapIndex> node2bin_map;
|
||||
std::vector<std::vector<int>>
|
||||
bin2node_map; ///< the first dimension is size, all the cells are categorized by width
|
||||
std::vector<GridIndex<int>> search_grids;
|
||||
};
|
||||
|
||||
inline int ceil_power2(int v) { return (1 << (int)ceil(log2((float)v))); }
|
||||
|
||||
template <typename DetailedPlaceDBType>
|
||||
void init_cpu_db(const DetailedPlaceDBType& db, DetailedPlaceCPUDB<typename DetailedPlaceDBType::type>& host_db) {
|
||||
host_db.num_movable_nodes = db.num_movable_nodes;
|
||||
host_db.num_bins_x = db.num_bins_x;
|
||||
host_db.num_bins_y = db.num_bins_y;
|
||||
host_db.bin_size_x = db.bin_size_x;
|
||||
host_db.bin_size_y = db.bin_size_y;
|
||||
host_db.xl = db.xl;
|
||||
host_db.yl = db.yl;
|
||||
host_db.xh = db.xh;
|
||||
host_db.yh = db.yh;
|
||||
host_db.node_size_x.resize(db.num_nodes);
|
||||
checkCuda(cudaMemcpy(host_db.node_size_x.data(),
|
||||
db.node_size_x,
|
||||
sizeof(typename DetailedPlaceDBType::type) * db.num_nodes,
|
||||
cudaMemcpyDeviceToHost));
|
||||
host_db.node_size_y.resize(db.num_nodes);
|
||||
checkCuda(cudaMemcpy(host_db.node_size_y.data(),
|
||||
db.node_size_y,
|
||||
sizeof(typename DetailedPlaceDBType::type) * db.num_nodes,
|
||||
cudaMemcpyDeviceToHost));
|
||||
host_db.x.resize(db.num_nodes);
|
||||
host_db.y.resize(db.num_nodes);
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
void init_cpu_state(const DetailedPlaceDBType& db,
|
||||
const IndependentSetMatchingStateType& state,
|
||||
IndependentSetMatchingCPUState<typename DetailedPlaceDBType::type>& host_state) {
|
||||
host_state.batch_size = state.batch_size;
|
||||
host_state.set_size = state.set_size;
|
||||
host_state.grid_size = ceil_power2(std::max(db.num_bins_x, db.num_bins_y) / 8);
|
||||
host_state.max_diamond_search_sequence = host_state.grid_size * host_state.grid_size / 2;
|
||||
logger.info("diamond search grid size %d, sequence length %d",
|
||||
host_state.grid_size,
|
||||
host_state.max_diamond_search_sequence);
|
||||
host_state.selected_nodes.reserve(db.num_movable_nodes);
|
||||
host_state.selected_markers.assign(db.num_movable_nodes, 1);
|
||||
host_state.ordered_nodes.resize(db.num_movable_nodes);
|
||||
host_state.search_grids = diamond_search_sequence(host_state.grid_size, host_state.grid_size);
|
||||
host_state.independent_sets.resize(state.batch_size, std::vector<int>(state.set_size));
|
||||
host_state.flat_independent_sets.resize(state.batch_size * state.set_size);
|
||||
host_state.independent_set_sizes.resize(state.batch_size);
|
||||
host_state.node2bin_map.resize(db.num_movable_nodes);
|
||||
host_state.bin2node_map.resize(db.num_bins_x * db.num_bins_y);
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
121
cpp_to_py/gpudp/dp/ism/diamond_search.h
Normal file
121
cpp_to_py/gpudp/dp/ism/diamond_search.h
Normal file
@ -0,0 +1,121 @@
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <limits>
|
||||
#include <numeric>
|
||||
#include <type_traits>
|
||||
#include <vector>
|
||||
|
||||
namespace dp {
|
||||
|
||||
/// @brief grid index
|
||||
template <typename T>
|
||||
struct GridIndex {
|
||||
T ir; ///< row index
|
||||
T ic; ///< column index
|
||||
|
||||
GridIndex() : ir(std::numeric_limits<T>::max()), ic(std::numeric_limits<T>::max()) {}
|
||||
|
||||
GridIndex(T r, T c) : ir(r), ic(c) {}
|
||||
|
||||
T manhattan_distance(const GridIndex& rhs) const { return fabs(ir - rhs.ir) + fabs(ic - rhs.ic); }
|
||||
double angle(const GridIndex& rhs) const { return atan2(ic - rhs.ic, ir - rhs.ir); }
|
||||
};
|
||||
|
||||
/// @brief compare grid (row index, column index) by its manhattan distance to a
|
||||
/// target grid
|
||||
template <typename T>
|
||||
struct CompareGridByDistance2Target {
|
||||
GridIndex<T> target;
|
||||
CompareGridByDistance2Target(const GridIndex<T>& g) : target(g) {}
|
||||
bool operator()(const GridIndex<T>& g1, const GridIndex<T>& g2) const {
|
||||
T d1 = g1.manhattan_distance(target);
|
||||
double angle1 = g1.angle(target);
|
||||
T d2 = g2.manhattan_distance(target);
|
||||
double angle2 = g2.angle(target);
|
||||
return d1 < d2 || (d1 == d2 && (angle1 < angle2));
|
||||
}
|
||||
};
|
||||
|
||||
/// @brief kernel to generate the sequence for diamond search
|
||||
/// @tparam the template must be a signed integer
|
||||
/// @param num_rows number of rows
|
||||
/// @param num_cols number of columns
|
||||
/// @return the sequence in order from small distance to the center grid to
|
||||
/// large
|
||||
template <typename T>
|
||||
std::vector<GridIndex<T> > diamond_search_sequence_kernel(T num_rows, T num_cols) {
|
||||
//// 2D grid map in row major
|
||||
//// each element is the (row index, column index)
|
||||
// std::vector<GridIndex<T> > grid_map (num_rows*num_cols, GridIndex<T>(0,
|
||||
// 0)); for (T ir = 0; ir < num_rows; ++ir)
|
||||
//{
|
||||
// for (T ic = 0; ic < num_cols; ++ic)
|
||||
// {
|
||||
// grid_map[ir*num_cols+ic] = GridIndex<T>(-(T)num_rows/2+ir,
|
||||
// -(T)num_cols/2+ic);
|
||||
// }
|
||||
//}
|
||||
|
||||
//// sort from small distance to large
|
||||
// std::sort(grid_map.begin(), grid_map.end(),
|
||||
// CompareGridByDistance2Target<T>(GridIndex<T>(0, 0)));
|
||||
|
||||
// directly generate diamond shape grids
|
||||
// in clock-wise direction
|
||||
// the sequence covers the following shape
|
||||
// 1
|
||||
// 111
|
||||
// 11111
|
||||
// 111
|
||||
// 1
|
||||
std::vector<GridIndex<T> > grid_map;
|
||||
grid_map.reserve(num_rows * num_cols / 2);
|
||||
T max_sum = std::min(num_rows, num_cols) / 2;
|
||||
grid_map.push_back(GridIndex<T>(0, 0));
|
||||
for (T sum = 1; sum <= max_sum; ++sum) {
|
||||
// y > 0, x [-sum, sum]
|
||||
for (T ir = -sum; ir < sum; ++ir) {
|
||||
grid_map.push_back(GridIndex<T>(ir, sum - std::abs(ir)));
|
||||
}
|
||||
// y < 0, x [sum, -sum]
|
||||
for (T ir = sum; ir > -sum; --ir) {
|
||||
grid_map.push_back(GridIndex<T>(ir, -(sum - std::abs(ir))));
|
||||
}
|
||||
}
|
||||
|
||||
return grid_map;
|
||||
}
|
||||
|
||||
/// @brief top API to generate the sequence for diamond search
|
||||
/// @param num_rows number of rows
|
||||
/// @param num_cols number of columns
|
||||
/// @return the sequence in order from small distance to the center grid to
|
||||
/// large
|
||||
template <typename T>
|
||||
std::vector<GridIndex<typename std::make_signed<T>::type> > diamond_search_sequence(T num_rows, T num_cols) {
|
||||
return diamond_search_sequence_kernel<typename std::make_signed<T>::type>(num_rows, num_cols);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
void diamond_search_print(const std::vector<GridIndex<T> >& grid_sequence) {
|
||||
unsigned int sum = 0;
|
||||
unsigned int count = 0;
|
||||
GridIndex<T> target(0, 0);
|
||||
printf("[0] ");
|
||||
for (typename std::vector<GridIndex<T> >::const_iterator it = grid_sequence.begin(); it != grid_sequence.end();
|
||||
++it, ++count) {
|
||||
T distance = it->manhattan_distance(target);
|
||||
if (sum != distance) {
|
||||
sum = distance;
|
||||
printf("\n[%u] ", count);
|
||||
}
|
||||
printf("(%d,%d) ", it->ir, it->ic);
|
||||
}
|
||||
printf("\n");
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
179
cpp_to_py/gpudp/dp/ism/maximal_independent_set.cuh
Normal file
179
cpp_to_py/gpudp/dp/ism/maximal_independent_set.cuh
Normal file
@ -0,0 +1,179 @@
|
||||
#pragma once
|
||||
|
||||
#include "gpudp/dp/detailed_place_db.cuh"
|
||||
|
||||
namespace dp {
|
||||
|
||||
__global__ void collect_kernel(const int* d_flags, int* d_sums, int* d_results, const int length) {
|
||||
const int tid = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (d_flags[tid] == 1 && tid < length) {
|
||||
d_results[d_sums[tid]] = tid;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename V>
|
||||
__global__ void select_kernel_add(const T* a, const V* b, int* c) {
|
||||
if (blockIdx.x == 0 && threadIdx.x == 0) {
|
||||
*c = (int)(*a) + (int)(*b);
|
||||
}
|
||||
}
|
||||
|
||||
void select(const int* d_flags, int* d_results, const int length, int* scratch, int* num_collected) {
|
||||
size_t temp_storage_bytes = 0;
|
||||
void* d_temp_storage = NULL; // need this NULL pointer to get temp_storage_bytes
|
||||
int* prefix_sum = scratch;
|
||||
|
||||
checkCuda(cub::DeviceScan::ExclusiveSum(d_temp_storage, temp_storage_bytes, d_flags, prefix_sum, length));
|
||||
|
||||
// Run exclusive prefix sum
|
||||
checkCuda(cub::DeviceScan::ExclusiveSum((void*)d_results, temp_storage_bytes, d_flags, prefix_sum, length));
|
||||
// cudaDeviceSynchronize();
|
||||
|
||||
select_kernel_add<<<1, 1>>>(prefix_sum + (length - 1), d_flags + (length - 1), num_collected);
|
||||
|
||||
collect_kernel<<<(length + 256 - 1) / 256, 256>>>(d_flags, prefix_sum, d_results, length);
|
||||
cudaDeviceSynchronize();
|
||||
}
|
||||
|
||||
/// @brief for each node, check its first level neighbors, if they are selected, mark itself as dependent
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__device__ void mark_dependent_nodes_self(const DetailedPlaceDBType& db,
|
||||
IndependentSetMatchingStateType& state,
|
||||
int node_id) {
|
||||
if (state.selected_markers[node_id]) {
|
||||
state.dependent_markers[node_id] = 1;
|
||||
return;
|
||||
}
|
||||
typename DetailedPlaceDBType::type node_xl = db.x[node_id];
|
||||
typename DetailedPlaceDBType::type node_yl = db.y[node_id];
|
||||
// in case all nets are masked
|
||||
int node2pin_start = db.flat_node2pin_start_map[node_id];
|
||||
int node2pin_end = db.flat_node2pin_start_map[node_id + 1];
|
||||
for (int node2pin_id = node2pin_start; node2pin_id < node2pin_end; ++node2pin_id) {
|
||||
int node_pin_id = db.flat_node2pin_map[node2pin_id];
|
||||
int net_id = db.pin2net_map[node_pin_id];
|
||||
if (db.net_mask[net_id]) {
|
||||
int net2pin_start = db.flat_net2pin_start_map[net_id];
|
||||
int net2pin_end = db.flat_net2pin_start_map[net_id + 1];
|
||||
for (int net2pin_id = net2pin_start; net2pin_id < net2pin_end; ++net2pin_id) {
|
||||
int net_pin_id = db.flat_net2pin_map[net2pin_id];
|
||||
int other_node_id = db.pin2node_map[net_pin_id];
|
||||
typename DetailedPlaceDBType::type other_node_xl = db.x[other_node_id];
|
||||
typename DetailedPlaceDBType::type other_node_yl = db.y[other_node_id];
|
||||
if (std::abs(node_xl - other_node_xl) + std::abs(node_yl - other_node_yl) < state.skip_threshold) {
|
||||
if (other_node_id < db.num_movable_nodes && state.selected_markers[other_node_id]) {
|
||||
state.dependent_markers[node_id] = 1;
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void maximal_independent_set_kernel(DetailedPlaceDBType db,
|
||||
IndependentSetMatchingStateType state,
|
||||
int* empty) {
|
||||
const int from = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
const int incr = gridDim.x * blockDim.x;
|
||||
|
||||
// do
|
||||
//{
|
||||
// empty = true;
|
||||
for (int node_id = from; node_id < db.num_movable_nodes; node_id += incr) {
|
||||
if (!state.dependent_markers[node_id]) {
|
||||
if (*empty) {
|
||||
atomicExch(empty, false);
|
||||
}
|
||||
// empty = false;
|
||||
bool min_node_flag = true;
|
||||
{
|
||||
typename DetailedPlaceDBType::type node_xl = db.x[node_id];
|
||||
typename DetailedPlaceDBType::type node_yl = db.y[node_id];
|
||||
int node_rank = state.ordered_nodes[node_id];
|
||||
// in case all nets are masked
|
||||
int node2pin_start = db.flat_node2pin_start_map[node_id];
|
||||
int node2pin_end = db.flat_node2pin_start_map[node_id + 1];
|
||||
for (int node2pin_id = node2pin_start; node2pin_id < node2pin_end; ++node2pin_id) {
|
||||
int node_pin_id = db.flat_node2pin_map[node2pin_id];
|
||||
int net_id = db.pin2net_map[node_pin_id];
|
||||
if (db.net_mask[net_id]) {
|
||||
int net2pin_start = db.flat_net2pin_start_map[net_id];
|
||||
int net2pin_end = db.flat_net2pin_start_map[net_id + 1];
|
||||
for (int net2pin_id = net2pin_start; net2pin_id < net2pin_end; ++net2pin_id) {
|
||||
int net_pin_id = db.flat_net2pin_map[net2pin_id];
|
||||
int other_node_id = db.pin2node_map[net_pin_id];
|
||||
typename DetailedPlaceDBType::type other_node_xl = db.x[other_node_id];
|
||||
typename DetailedPlaceDBType::type other_node_yl = db.y[other_node_id];
|
||||
typename DetailedPlaceDBType::type distance =
|
||||
abs(node_xl - other_node_xl) + abs(node_yl - other_node_yl);
|
||||
if (other_node_id < db.num_movable_nodes && (distance < state.skip_threshold) &&
|
||||
(state.selected_markers[other_node_id] ||
|
||||
(state.dependent_markers[other_node_id] == 0 &&
|
||||
state.ordered_nodes[other_node_id] < node_rank))) {
|
||||
min_node_flag = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!min_node_flag) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (min_node_flag) {
|
||||
state.selected_markers[node_id] = 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
//} while (!empty);
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void mark_dependent_nodes_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
|
||||
for (int node_id = blockIdx.x * blockDim.x + threadIdx.x; node_id < db.num_movable_nodes;
|
||||
node_id += blockDim.x * gridDim.x) {
|
||||
if (!state.dependent_markers[node_id]) {
|
||||
mark_dependent_nodes_self(db, state, node_id);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
__global__ void init_markers_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
|
||||
for (int i = blockIdx.x * blockDim.x + threadIdx.x; i < db.num_nodes; i += blockDim.x * gridDim.x) {
|
||||
state.selected_markers[i] = 0;
|
||||
// make sure multi-row height cells are not selected
|
||||
state.dependent_markers[i] = (db.node_size_y[i] > db.row_height);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
|
||||
void maximal_independent_set(DetailedPlaceDBType const& db, IndependentSetMatchingStateType& state) {
|
||||
// if dependent_markers is 1, it means "cannot be selected"
|
||||
// if selected_markers is 1, it means "already selected"
|
||||
init_markers_kernel<<<ceilDiv(db.num_nodes, 256), 256>>>(db, state);
|
||||
|
||||
int host_empty;
|
||||
|
||||
int iteration = 0;
|
||||
do {
|
||||
host_empty = true;
|
||||
checkCuda(cudaMemcpy(state.independent_set_empty_flag, &host_empty, sizeof(int), cudaMemcpyHostToDevice));
|
||||
maximal_independent_set_kernel<<<ceilDiv(db.num_movable_nodes, 256), 256>>>(
|
||||
db, state, state.independent_set_empty_flag);
|
||||
mark_dependent_nodes_kernel<<<ceilDiv(db.num_movable_nodes, 256), 256>>>(db, state);
|
||||
checkCuda(cudaMemcpy(&host_empty, state.independent_set_empty_flag, sizeof(int), cudaMemcpyDeviceToHost));
|
||||
++iteration;
|
||||
} while (!host_empty && iteration < 10);
|
||||
|
||||
select(state.selected_markers,
|
||||
state.selected_maximal_independent_set,
|
||||
db.num_movable_nodes,
|
||||
state.select_scratch,
|
||||
state.device_num_selected);
|
||||
checkCuda(cudaMemcpy(&state.num_selected, state.device_num_selected, sizeof(int), cudaMemcpyDeviceToHost));
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
155
cpp_to_py/gpudp/dp/ism/reduce_min.cuh
Normal file
155
cpp_to_py/gpudp/dp/ism/reduce_min.cuh
Normal file
@ -0,0 +1,155 @@
|
||||
#pragma once
|
||||
|
||||
#include "gpudp/dp/utils.cuh"
|
||||
|
||||
namespace dp {
|
||||
|
||||
template <typename T, typename V>
|
||||
__device__ void warpReduce(/*volatile*/ T* sdata, int tid, V comp) {
|
||||
sdata[tid] = sdata[tid + 32 * !comp(sdata[tid], sdata[tid + 32])];
|
||||
sdata[tid] = sdata[tid + 16 * !comp(sdata[tid], sdata[tid + 16])];
|
||||
sdata[tid] = sdata[tid + 8 * !comp(sdata[tid], sdata[tid + 8])];
|
||||
sdata[tid] = sdata[tid + 4 * !comp(sdata[tid], sdata[tid + 4])];
|
||||
sdata[tid] = sdata[tid + 2 * !comp(sdata[tid], sdata[tid + 2])];
|
||||
sdata[tid] = sdata[tid + 1 * !comp(sdata[tid], sdata[tid + 1])];
|
||||
}
|
||||
|
||||
/**
|
||||
* 优化:解决了 reduce3 中存在的多余同步操作(每个warp默认自动同步)。
|
||||
* globalInputData 输入数据,位于全局内存
|
||||
* globalOutputData 输出数据,位于全局内存
|
||||
* n length of array
|
||||
* ref reference value
|
||||
* comp compare function which returns the target element
|
||||
*/
|
||||
template <typename T, typename V, unsigned int BlockSize = 256>
|
||||
__global__ void reduce4(T* globalInputData, T* globalOutputData, int n, T ref, V comp) {
|
||||
__shared__ T sdata[BlockSize];
|
||||
|
||||
// 坐标索引
|
||||
int tid = threadIdx.x;
|
||||
int index = blockIdx.x * (blockDim.x * 2) + threadIdx.x;
|
||||
int indexWithOffset = index + blockDim.x;
|
||||
|
||||
if (index >= n)
|
||||
sdata[tid] = ref;
|
||||
else if (indexWithOffset >= n)
|
||||
sdata[tid] = globalInputData[index];
|
||||
else {
|
||||
// printf("tid = %d, index = %d, indexWithOffset = %d, index+blockDim.x*!comp(globalInputData[index],
|
||||
// globalInputData[indexWithOffset]) = %d\n",
|
||||
// tid, index, indexWithOffset, index+blockDim.x*!comp(globalInputData[index],
|
||||
// globalInputData[indexWithOffset])
|
||||
// );
|
||||
sdata[tid] = (comp(globalInputData[index], globalInputData[indexWithOffset]))
|
||||
? globalInputData[index]
|
||||
: globalInputData[indexWithOffset];
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
// 在共享内存中对每一个块进行规约计算
|
||||
for (int s = blockDim.x / 2; s > 0; s >>= 1) {
|
||||
if (tid < s) {
|
||||
sdata[tid] = sdata[tid + s * !comp(sdata[tid], sdata[tid + s])];
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
}
|
||||
// if (tid < 32)
|
||||
//{
|
||||
// warpReduce(sdata, tid, comp);
|
||||
//}
|
||||
|
||||
// 把计算结果从共享内存写回全局内存
|
||||
if (tid == 0) {
|
||||
globalOutputData[blockIdx.x] = sdata[0];
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename V, unsigned int BlockSize = 256>
|
||||
void reduce(T* fMatrix_Device, int iMatrixSize, const T& ref, const V& comp) {
|
||||
for (int i = 1, iNum = iMatrixSize; i < iMatrixSize; i = 2 * i * BlockSize) {
|
||||
int iBlockNum = (iNum + (2 * BlockSize) - 1) / (2 * BlockSize);
|
||||
reduce4<T, V, BlockSize><<<iBlockNum, BlockSize>>>(fMatrix_Device, fMatrix_Device, iNum, ref, comp);
|
||||
iNum = iBlockNum;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename V, unsigned int BlockSize = 256>
|
||||
void reduce(T* fMatrix_Device, int iMatrixSize, const T& ref, const V& comp, cudaStream_t& stream) {
|
||||
for (int i = 1, iNum = iMatrixSize; i < iMatrixSize; i = 2 * i * BlockSize) {
|
||||
int iBlockNum = (iNum + (2 * BlockSize) - 1) / (2 * BlockSize);
|
||||
reduce4<T, V, BlockSize><<<iBlockNum, BlockSize, 0, stream>>>(fMatrix_Device, fMatrix_Device, iNum, ref, comp);
|
||||
iNum = iBlockNum;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* improvement: resolved redundant synchronization in reduce3, i.e., synchronize each warp
|
||||
* globalInputData input data, located in global memory
|
||||
* globalOutputData output data, located in global memory
|
||||
* nc number of initial columns before reduction
|
||||
* n number of columns
|
||||
* ref reference value
|
||||
* comp compare function which returns the target element
|
||||
*/
|
||||
template <typename T, typename V, unsigned int BlockSize = 256>
|
||||
__global__ void reduce4_2d(T* globalInputData, T* globalOutputData, int nc, int n, T ref, V comp) {
|
||||
__shared__ T sdata[BlockSize];
|
||||
|
||||
// compute indices
|
||||
int tid = threadIdx.x;
|
||||
int yOffset = blockIdx.y * nc;
|
||||
int index = yOffset + blockIdx.x * (blockDim.x * 2) + threadIdx.x;
|
||||
|
||||
int indexWithOffset = index + blockDim.x;
|
||||
|
||||
if (index >= yOffset + n)
|
||||
sdata[tid] = ref;
|
||||
else if (indexWithOffset >= yOffset + n)
|
||||
sdata[tid] = globalInputData[index];
|
||||
else {
|
||||
// printf("tid = %d, index = %d, indexWithOffset = %d, index+blockDim.x*!comp(globalInputData[index],
|
||||
// globalInputData[indexWithOffset]) = %d\n",
|
||||
// tid, index, indexWithOffset, index+blockDim.x*!comp(globalInputData[index],
|
||||
// globalInputData[indexWithOffset])
|
||||
// );
|
||||
sdata[tid] = (comp(globalInputData[index], globalInputData[indexWithOffset]))
|
||||
? globalInputData[index]
|
||||
: globalInputData[indexWithOffset];
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
// reduction for data in shared memory
|
||||
for (int s = blockDim.x / 2; s > 0; s >>= 1) {
|
||||
if (tid < s) {
|
||||
sdata[tid] = sdata[tid + s * !comp(sdata[tid], sdata[tid + s])];
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
}
|
||||
// if (tid < 32)
|
||||
//{
|
||||
// warpReduce(sdata, tid, comp);
|
||||
//}
|
||||
|
||||
// write data back from shared memory to global memory
|
||||
if (tid == 0) {
|
||||
globalOutputData[yOffset + blockIdx.x] = sdata[0];
|
||||
// printf("globalOutputData[%d] = %g\n", blockIdx.y*nc + blockIdx.x, sdata[0].cost);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, typename V, unsigned int BlockSize = 256>
|
||||
void reduce_2d(T* fMatrix_Device, int m, int n, const T& ref, const V& comp) {
|
||||
for (int i = 1, iNum = n; i < n; i = 2 * i * BlockSize) {
|
||||
int iBlockNum = (iNum + (2 * BlockSize) - 1) / (2 * BlockSize);
|
||||
dim3 grid(iBlockNum, m, 1);
|
||||
reduce4_2d<T, V, BlockSize><<<grid, BlockSize>>>(fMatrix_Device, fMatrix_Device, n, iNum, ref, comp);
|
||||
iNum = iBlockNum;
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
108
cpp_to_py/gpudp/dp/ism/shuffle.cuh
Normal file
108
cpp_to_py/gpudp/dp/ism/shuffle.cuh
Normal file
@ -0,0 +1,108 @@
|
||||
#pragma once
|
||||
|
||||
#include <curand.h>
|
||||
#include <curand_kernel.h>
|
||||
|
||||
#include "gpudp/dp/utils.cuh"
|
||||
|
||||
namespace dp {
|
||||
|
||||
template <typename T, typename V>
|
||||
__global__ void print_shuffle(const T* values, const V* keys, int n) {
|
||||
if (blockIdx.x == 0 && threadIdx.x == 0) {
|
||||
printf("values[%d]\n", n);
|
||||
for (int i = 0; i < n; ++i) {
|
||||
printf("%d ", int(values[i]));
|
||||
}
|
||||
printf("\n");
|
||||
printf("keys[%d]\n", n);
|
||||
for (int i = 0; i < n; ++i) {
|
||||
printf("%d ", int(keys[i]));
|
||||
}
|
||||
printf("\n");
|
||||
}
|
||||
}
|
||||
|
||||
/// @brief A shuffler that can be repeatedly called.
|
||||
/// @tparam T value type
|
||||
/// @tparam V key type
|
||||
template <typename T, typename V>
|
||||
class Shuffler {
|
||||
public:
|
||||
/// @brief constructor
|
||||
/// @param seed random seed
|
||||
/// @param values data array that will be manipulated
|
||||
/// @param n length of array
|
||||
Shuffler(size_t seed, T* values, int n) {
|
||||
/* Create pseudo-random number generator */
|
||||
checkCurand(curandCreateGenerator(&m_gen, CURAND_RNG_PSEUDO_DEFAULT));
|
||||
|
||||
/* Set seed */
|
||||
checkCurand(curandSetPseudoRandomGeneratorSeed(m_gen, seed));
|
||||
|
||||
m_values_in = values;
|
||||
allocateCuda(m_keys_in, n, V);
|
||||
allocateCuda(m_keys_out, n, V);
|
||||
allocateCuda(m_values_out, n, T);
|
||||
m_temp_storage = NULL;
|
||||
m_temp_storage_bytes = 0;
|
||||
m_num_items = n;
|
||||
}
|
||||
/// @brief destructor
|
||||
~Shuffler() {
|
||||
if (m_temp_storage) {
|
||||
cudaFree(m_temp_storage);
|
||||
}
|
||||
cudaFree(m_keys_in);
|
||||
cudaFree(m_keys_out);
|
||||
cudaFree(m_values_out);
|
||||
checkCurand(curandDestroyGenerator(m_gen));
|
||||
}
|
||||
/// @brief top API to shuffle data. It can be called repeatedly.
|
||||
void operator()() {
|
||||
/* Generate n floats on device */
|
||||
checkCurand(curandGenerate(m_gen, m_keys_in, m_num_items));
|
||||
|
||||
// Determine temporary device storage requirements
|
||||
void* d_temp_storage = NULL;
|
||||
size_t temp_storage_bytes = 0;
|
||||
cub::DeviceRadixSort::SortPairs(
|
||||
d_temp_storage, temp_storage_bytes, m_keys_in, m_keys_out, m_values_in, m_values_out, m_num_items);
|
||||
|
||||
// Allocate temporary storage
|
||||
// re-allocate if different size
|
||||
if (m_temp_storage_bytes != temp_storage_bytes) {
|
||||
if (m_temp_storage_bytes) {
|
||||
cudaFree(m_temp_storage);
|
||||
m_temp_storage = NULL;
|
||||
}
|
||||
m_temp_storage_bytes = temp_storage_bytes;
|
||||
logger.debug("allocate %lu bytes in shuffler for length %d*(%d+%d)",
|
||||
m_temp_storage_bytes,
|
||||
m_num_items,
|
||||
sizeof(T),
|
||||
sizeof(V));
|
||||
checkCuda(cudaMalloc(&m_temp_storage, m_temp_storage_bytes));
|
||||
}
|
||||
|
||||
// Run sorting operation
|
||||
cub::DeviceRadixSort::SortPairs(
|
||||
m_temp_storage, m_temp_storage_bytes, m_keys_in, m_keys_out, m_values_in, m_values_out, m_num_items);
|
||||
|
||||
// copy back to m_values_in, not necessary
|
||||
// As m_values_in corresponds to external data, copying back the output can allow in-place manipulation
|
||||
checkCuda(cudaMemcpy(m_values_in, m_values_out, sizeof(T) * m_num_items, cudaMemcpyDeviceToDevice));
|
||||
}
|
||||
|
||||
protected:
|
||||
curandGenerator_t m_gen; ///< random number generator
|
||||
V* m_keys_in; ///< on device, to store real key
|
||||
T* m_values_in; ///< on device, to store real data
|
||||
V* m_keys_out; ///< on device, a buffer
|
||||
T* m_values_out; ///< on device, a buffer
|
||||
void* m_temp_storage; ///< on device, temporary storage for sorting
|
||||
size_t m_temp_storage_bytes; ///< number of bytes for m_temp_storage
|
||||
int m_num_items; ///< length of array
|
||||
};
|
||||
|
||||
} // namespace dp
|
||||
1093
cpp_to_py/gpudp/dp/k_reorder_cuda.cu
Normal file
1093
cpp_to_py/gpudp/dp/k_reorder_cuda.cu
Normal file
File diff suppressed because it is too large
Load Diff
108
cpp_to_py/gpudp/dp/pitch_nested_vector.cuh
Normal file
108
cpp_to_py/gpudp/dp/pitch_nested_vector.cuh
Normal file
@ -0,0 +1,108 @@
|
||||
#pragma once
|
||||
|
||||
#include <cuda.h>
|
||||
#include <cuda_runtime.h>
|
||||
#include "utils.cuh"
|
||||
|
||||
template <typename T>
|
||||
struct PitchNestedVector {
|
||||
T* flat_element_map; ///< allocate on device, length of size1*size2
|
||||
unsigned int* dim2_sizes; ///< sizes of dimension 2
|
||||
unsigned int size1; ///< length in dimension 1
|
||||
unsigned int size2; ///< maximum length in dimension 2
|
||||
unsigned int num_elements; ///< total number of elements
|
||||
|
||||
/// @brief constructor
|
||||
__host__ PitchNestedVector() : flat_element_map(nullptr), dim2_sizes(nullptr), size1(0), size2(0) {}
|
||||
|
||||
/// @brief initialization
|
||||
__host__ void initialize(const std::vector<std::vector<T> >& nested_map) {
|
||||
// construct flat map on host
|
||||
unsigned int max_num_elements = 0;
|
||||
num_elements = 0;
|
||||
for (typename std::vector<std::vector<T> >::const_iterator it = nested_map.begin(); it != nested_map.end();
|
||||
++it) {
|
||||
max_num_elements = max(max_num_elements, (unsigned int)it->size());
|
||||
num_elements += it->size();
|
||||
}
|
||||
std::vector<T> host_flat_element_map(nested_map.size() * max_num_elements, std::numeric_limits<T>::max());
|
||||
std::vector<unsigned int> host_dim2_sizes(nested_map.size());
|
||||
|
||||
for (unsigned int i = 0; i < nested_map.size(); ++i) {
|
||||
const std::vector<T>& vec = nested_map[i];
|
||||
std::copy(vec.begin(), vec.end(), host_flat_element_map.begin() + max_num_elements * i);
|
||||
host_dim2_sizes[i] = vec.size();
|
||||
}
|
||||
|
||||
// copy to device
|
||||
size1 = nested_map.size();
|
||||
size2 = max_num_elements;
|
||||
allocateCopyCuda(flat_element_map, host_flat_element_map.data(), host_flat_element_map.size());
|
||||
allocateCopyCuda(dim2_sizes, host_dim2_sizes.data(), host_dim2_sizes.size());
|
||||
}
|
||||
|
||||
__host__ void destroy() {
|
||||
if (flat_element_map) {
|
||||
cudaFree(flat_element_map);
|
||||
cudaFree(dim2_sizes);
|
||||
}
|
||||
}
|
||||
|
||||
/// @brief access element
|
||||
inline __device__ const T& operator()(unsigned int i, unsigned int j) const {
|
||||
#ifdef DEBUG
|
||||
if (!(i < size1 && j < size(i))) {
|
||||
printf("%u < %u && %u < %u\n", i, size1, j, size(i));
|
||||
}
|
||||
#endif
|
||||
assert(i < size1 && j < size(i));
|
||||
return flat_element_map[i * size2 + j];
|
||||
}
|
||||
|
||||
/// @brief access element
|
||||
inline __device__ T& operator()(unsigned int i, unsigned int j) {
|
||||
#ifdef DEBUG
|
||||
if (!(i < size1 && j < size(i))) {
|
||||
printf("%u < %u && %u < %u\n", i, size1, j, size(i));
|
||||
}
|
||||
#endif
|
||||
assert(i < size1 && j < size(i));
|
||||
return flat_element_map[i * size2 + j];
|
||||
}
|
||||
|
||||
/// @brief access each row
|
||||
inline __device__ const T* operator()(unsigned int i) const {
|
||||
#ifdef DEBUG
|
||||
if (!(i < size1)) {
|
||||
printf("%u < %u\n", i, size1);
|
||||
}
|
||||
#endif
|
||||
assert(i < size1);
|
||||
return flat_element_map + i * size2;
|
||||
}
|
||||
|
||||
/// @brief access each row
|
||||
inline __device__ T* operator()(unsigned int i) {
|
||||
#ifdef DEBUG
|
||||
if (!(i < size1)) {
|
||||
printf("%u < %u\n", i, size1);
|
||||
}
|
||||
#endif
|
||||
assert(i < size1);
|
||||
return flat_element_map + i * size2;
|
||||
}
|
||||
|
||||
/// @brief length of each row
|
||||
inline __device__ unsigned int size(unsigned int i) const {
|
||||
#ifdef DEBUG
|
||||
if (!(i < size1)) {
|
||||
printf("%u < %u\n", i, size1);
|
||||
}
|
||||
#endif
|
||||
assert(i < size1);
|
||||
return dim2_sizes[i];
|
||||
}
|
||||
|
||||
/// @brief total number of elements
|
||||
inline __device__ unsigned int size() const { return num_elements; }
|
||||
};
|
||||
279
cpp_to_py/gpudp/dp/utils.cuh
Normal file
279
cpp_to_py/gpudp/dp/utils.cuh
Normal file
@ -0,0 +1,279 @@
|
||||
#pragma once
|
||||
|
||||
#include <cuda.h>
|
||||
#include <cuda_runtime.h>
|
||||
#include <float.h>
|
||||
#include <limits.h>
|
||||
|
||||
#include "cub/cub.cuh"
|
||||
|
||||
#define checkCuda(expression) \
|
||||
{ \
|
||||
cudaError_t status = (expression); \
|
||||
if (status != cudaSuccess) { \
|
||||
printf("CUDA Runtime Error: %s at %s:%d\n", cudaGetErrorString(expression), __FILE__, __LINE__); \
|
||||
std::exit(EXIT_FAILURE); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define checkCurand(expression) \
|
||||
{ \
|
||||
curandStatus_t status = (expression); \
|
||||
if (status != CURAND_STATUS_SUCCESS) { \
|
||||
printf("Curand Error at %s:%d\n", __FILE__, __LINE__); \
|
||||
std::exit(EXIT_FAILURE); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define allocateCuda(var, size, type) \
|
||||
{ \
|
||||
cudaError_t status = cudaMalloc(&(var), (size) * sizeof(type)); \
|
||||
if (status != cudaSuccess) { \
|
||||
printf("cudaMalloc failed for " #var " at %s:%d\n", __FILE__, __LINE__); \
|
||||
} \
|
||||
}
|
||||
|
||||
#define allocateCopyCuda(var, rhs, size) \
|
||||
{ \
|
||||
allocateCuda(var, size, decltype(*rhs)); \
|
||||
checkCuda(cudaMemcpy(var, rhs, sizeof(decltype(*rhs)) * (size), cudaMemcpyHostToDevice)); \
|
||||
}
|
||||
|
||||
#define allocateCopyCpu(var, rhs, size, T) \
|
||||
{ \
|
||||
var = (T*)malloc(sizeof(T) * (size)); \
|
||||
checkCuda(cudaMemcpy((void*)var, (void*)rhs, sizeof(T) * (size), cudaMemcpyDeviceToHost)); \
|
||||
}
|
||||
|
||||
|
||||
// For cuda::numeric_limits
|
||||
namespace cuda { // namespace cuda
|
||||
|
||||
template <typename T>
|
||||
struct numeric_limits_base {
|
||||
typedef T type;
|
||||
};
|
||||
template <typename T>
|
||||
struct numeric_limits : public numeric_limits_base<T> {};
|
||||
|
||||
template <>
|
||||
struct numeric_limits<char> : public numeric_limits_base<char> {
|
||||
/** The minimum finite value, or for floating types with
|
||||
denormalization, the minimum positive normalized value. */
|
||||
__host__ __device__ static constexpr type min() noexcept { return CHAR_MIN; }
|
||||
|
||||
/** The maximum finite value. */
|
||||
__host__ __device__ static constexpr type max() noexcept { return CHAR_MAX; }
|
||||
|
||||
/** A finite value x such that there is no other finite value y
|
||||
* where y < x. */
|
||||
__host__ __device__ static constexpr type lowest() noexcept { return CHAR_MIN; }
|
||||
|
||||
/** A the machine epsilon. */
|
||||
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct numeric_limits<unsigned char> : public numeric_limits_base<unsigned char> {
|
||||
/** The minimum finite value, or for floating types with
|
||||
denormalization, the minimum positive normalized value. */
|
||||
__host__ __device__ static constexpr type min() noexcept { return 0; }
|
||||
|
||||
/** The maximum finite value. */
|
||||
__host__ __device__ static constexpr type max() noexcept { return UCHAR_MAX; }
|
||||
|
||||
/** A finite value x such that there is no other finite value y
|
||||
* where y < x. */
|
||||
__host__ __device__ static constexpr type lowest() noexcept { return 0; }
|
||||
|
||||
/** A the machine epsilon. */
|
||||
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct numeric_limits<short> : public numeric_limits_base<short> {
|
||||
/** The minimum finite value, or for floating types with
|
||||
denormalization, the minimum positive normalized value. */
|
||||
__host__ __device__ static constexpr type min() noexcept { return SHRT_MIN; }
|
||||
|
||||
/** The maximum finite value. */
|
||||
__host__ __device__ static constexpr type max() noexcept { return SHRT_MAX; }
|
||||
|
||||
/** A finite value x such that there is no other finite value y
|
||||
* where y < x. */
|
||||
__host__ __device__ static constexpr type lowest() noexcept { return SHRT_MIN; }
|
||||
|
||||
/** A the machine epsilon. */
|
||||
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct numeric_limits<unsigned short> : public numeric_limits_base<unsigned short> {
|
||||
/** The minimum finite value, or for floating types with
|
||||
denormalization, the minimum positive normalized value. */
|
||||
__host__ __device__ static constexpr type min() noexcept { return 0; }
|
||||
|
||||
/** The maximum finite value. */
|
||||
__host__ __device__ static constexpr type max() noexcept { return USHRT_MAX; }
|
||||
|
||||
/** A finite value x such that there is no other finite value y
|
||||
* where y < x. */
|
||||
__host__ __device__ static constexpr type lowest() noexcept { return 0; }
|
||||
|
||||
/** A the machine epsilon. */
|
||||
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct numeric_limits<int> : public numeric_limits_base<int> {
|
||||
/** The minimum finite value, or for floating types with
|
||||
denormalization, the minimum positive normalized value. */
|
||||
__host__ __device__ static constexpr type min() noexcept { return INT_MIN; }
|
||||
|
||||
/** The maximum finite value. */
|
||||
__host__ __device__ static constexpr type max() noexcept { return INT_MAX; }
|
||||
|
||||
/** A finite value x such that there is no other finite value y
|
||||
* where y < x. */
|
||||
__host__ __device__ static constexpr type lowest() noexcept { return INT_MIN; }
|
||||
|
||||
/** A the machine epsilon. */
|
||||
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct numeric_limits<unsigned int> : public numeric_limits_base<unsigned int> {
|
||||
/** The minimum finite value, or for floating types with
|
||||
denormalization, the minimum positive normalized value. */
|
||||
__host__ __device__ static constexpr type min() noexcept { return 0; }
|
||||
|
||||
/** The maximum finite value. */
|
||||
__host__ __device__ static constexpr type max() noexcept { return UINT_MAX; }
|
||||
|
||||
/** A finite value x such that there is no other finite value y
|
||||
* where y < x. */
|
||||
__host__ __device__ static constexpr type lowest() noexcept { return 0; }
|
||||
|
||||
/** A the machine epsilon. */
|
||||
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct numeric_limits<long> : public numeric_limits_base<long> {
|
||||
/** The minimum finite value, or for floating types with
|
||||
denormalization, the minimum positive normalized value. */
|
||||
__host__ __device__ static constexpr type min() noexcept { return LONG_MIN; }
|
||||
|
||||
/** The maximum finite value. */
|
||||
__host__ __device__ static constexpr type max() noexcept { return LONG_MAX; }
|
||||
|
||||
/** A finite value x such that there is no other finite value y
|
||||
* where y < x. */
|
||||
__host__ __device__ static constexpr type lowest() noexcept { return LONG_MIN; }
|
||||
|
||||
/** A the machine epsilon. */
|
||||
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct numeric_limits<unsigned long> : public numeric_limits_base<unsigned long> {
|
||||
/** The minimum finite value, or for floating types with
|
||||
denormalization, the minimum positive normalized value. */
|
||||
__host__ __device__ static constexpr type min() noexcept { return 0; }
|
||||
|
||||
/** The maximum finite value. */
|
||||
__host__ __device__ static constexpr type max() noexcept { return ULONG_MAX; }
|
||||
|
||||
/** A finite value x such that there is no other finite value y
|
||||
* where y < x. */
|
||||
__host__ __device__ static constexpr type lowest() noexcept { return 0; }
|
||||
|
||||
/** A the machine epsilon. */
|
||||
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct numeric_limits<long long> : public numeric_limits_base<long long> {
|
||||
/** The minimum finite value, or for floating types with
|
||||
denormalization, the minimum positive normalized value. */
|
||||
__host__ __device__ static constexpr type min() noexcept { return LLONG_MIN; }
|
||||
|
||||
/** The maximum finite value. */
|
||||
__host__ __device__ static constexpr type max() noexcept { return LLONG_MAX; }
|
||||
|
||||
/** A finite value x such that there is no other finite value y
|
||||
* where y < x. */
|
||||
__host__ __device__ static constexpr type lowest() noexcept { return LLONG_MIN; }
|
||||
|
||||
/** A the machine epsilon. */
|
||||
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct numeric_limits<unsigned long long> : public numeric_limits_base<unsigned long long> {
|
||||
/** The minimum finite value, or for floating types with
|
||||
denormalization, the minimum positive normalized value. */
|
||||
__host__ __device__ static constexpr type min() noexcept { return 0; }
|
||||
|
||||
/** The maximum finite value. */
|
||||
__host__ __device__ static constexpr type max() noexcept { return ULLONG_MAX; }
|
||||
|
||||
/** A finite value x such that there is no other finite value y
|
||||
* where y < x. */
|
||||
__host__ __device__ static constexpr type lowest() noexcept { return 0; }
|
||||
|
||||
/** A the machine epsilon. */
|
||||
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct numeric_limits<float> : public numeric_limits_base<float> {
|
||||
/** The minimum finite value, or for floating types with
|
||||
denormalization, the minimum positive normalized value. */
|
||||
__host__ __device__ static constexpr type min() noexcept { return FLT_MIN; }
|
||||
|
||||
/** The maximum finite value. */
|
||||
__host__ __device__ static constexpr type max() noexcept { return FLT_MAX; }
|
||||
|
||||
/** A finite value x such that there is no other finite value y
|
||||
* where y < x. */
|
||||
__host__ __device__ static constexpr type lowest() noexcept { return -FLT_MAX; }
|
||||
|
||||
/** A the machine epsilon. */
|
||||
__host__ __device__ static constexpr type epsilon() noexcept { return FLT_EPSILON; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct numeric_limits<double> : public numeric_limits_base<double> {
|
||||
/** The minimum finite value, or for floating types with
|
||||
denormalization, the minimum positive normalized value. */
|
||||
__host__ __device__ static constexpr type min() noexcept { return DBL_MIN; }
|
||||
|
||||
/** The maximum finite value. */
|
||||
__host__ __device__ static constexpr type max() noexcept { return DBL_MAX; }
|
||||
|
||||
/** A finite value x such that there is no other finite value y
|
||||
* where y < x. */
|
||||
__host__ __device__ static constexpr type lowest() noexcept { return -DBL_MAX; }
|
||||
|
||||
/** A the machine epsilon. */
|
||||
__host__ __device__ static constexpr type epsilon() noexcept { return DBL_EPSILON; }
|
||||
};
|
||||
|
||||
template <>
|
||||
struct numeric_limits<long double> : public numeric_limits_base<long double> {
|
||||
/** The minimum finite value, or for floating types with
|
||||
denormalization, the minimum positive normalized value. */
|
||||
__host__ __device__ static constexpr type min() noexcept { return LDBL_MIN; }
|
||||
|
||||
/** The maximum finite value. */
|
||||
__host__ __device__ static constexpr type max() noexcept { return LDBL_MAX; }
|
||||
|
||||
/** A finite value x such that there is no other finite value y
|
||||
* where y < x. */
|
||||
__host__ __device__ static constexpr type lowest() noexcept { return -LDBL_MAX; }
|
||||
|
||||
/** A the machine epsilon. */
|
||||
__host__ __device__ static constexpr type epsilon() noexcept { return LDBL_EPSILON; }
|
||||
};
|
||||
} // namespace cuda
|
||||
429
cpp_to_py/gpudp/lg/abacus_legalize.cpp
Normal file
429
cpp_to_py/gpudp/lg/abacus_legalize.cpp
Normal file
@ -0,0 +1,429 @@
|
||||
#include "gpudp/lg/legalization_db.h"
|
||||
|
||||
namespace dp {
|
||||
|
||||
struct AbacusCluster {
|
||||
int prev_cluster_id; ///< previous cluster, set to INT_MIN if the cluster is
|
||||
///< invalid
|
||||
int next_cluster_id; ///< next cluster, set to INT_MIN if the cluster is
|
||||
///< invalid
|
||||
int bgn_row_node_id; ///< id of first node in the row
|
||||
int end_row_node_id; ///< id of last node in the row
|
||||
float e; ///< weight of displacement in the objective
|
||||
float q; ///< x = q/e
|
||||
float w; ///< width
|
||||
float x; ///< optimal location
|
||||
|
||||
/// @return whether this is a valid cluster
|
||||
bool valid() const { return prev_cluster_id != INT_MIN && next_cluster_id != INT_MIN; }
|
||||
};
|
||||
|
||||
/// @brief helper function for distributing cells to rows
|
||||
/// sort cells within a row and clean overlapping fixed cells
|
||||
void sortNodesInRow(const float* host_x,
|
||||
const float* host_y,
|
||||
const float* host_node_size_x,
|
||||
const float* host_node_size_y,
|
||||
int num_movable_nodes,
|
||||
std::vector<int>& nodes_in_row) {
|
||||
// sort cells within rows according to left edges
|
||||
std::sort(nodes_in_row.begin(), nodes_in_row.end(), [&](int node_id1, int node_id2) {
|
||||
float x1 = host_x[node_id1];
|
||||
float x2 = host_x[node_id2];
|
||||
// put larger width front will help remove
|
||||
// overlapping fixed cells, especially when
|
||||
// x1 == x2, then we need the wider one comes first
|
||||
float w1 = host_node_size_x[node_id1];
|
||||
float w2 = host_node_size_x[node_id2];
|
||||
return x1 < x2 || (x1 == x2 && (w1 > w2 || (w1 == w2 && node_id1 < node_id2)));
|
||||
});
|
||||
// After sorting by left edge,
|
||||
// there is a special case for fixed cells where
|
||||
// one fixed cell is completely within another in a row.
|
||||
// This will cause failure to detect some overlaps.
|
||||
// We need to remove the "small" fixed cell that is inside another.
|
||||
if (!nodes_in_row.empty()) {
|
||||
std::vector<int> tmp_nodes;
|
||||
tmp_nodes.reserve(nodes_in_row.size());
|
||||
tmp_nodes.push_back(nodes_in_row.front());
|
||||
int j_1 = 0;
|
||||
for (int j = 1, je = nodes_in_row.size(); j < je; ++j) {
|
||||
int node_id1 = nodes_in_row.at(j_1);
|
||||
int node_id2 = nodes_in_row.at(j);
|
||||
// two fixed cells
|
||||
if (node_id1 >= num_movable_nodes && node_id2 >= num_movable_nodes) {
|
||||
float xl1 = host_x[node_id1];
|
||||
float xl2 = host_x[node_id2];
|
||||
float width1 = host_node_size_x[node_id1];
|
||||
float width2 = host_node_size_x[node_id2];
|
||||
float xh1 = xl1 + width1;
|
||||
float xh2 = xl2 + width2;
|
||||
// only collect node_id2 if its right edge is righter than node_id1
|
||||
if (xh1 < xh2) {
|
||||
tmp_nodes.push_back(node_id2);
|
||||
j_1 = j;
|
||||
}
|
||||
} else {
|
||||
tmp_nodes.push_back(node_id2);
|
||||
j_1 = j;
|
||||
}
|
||||
}
|
||||
nodes_in_row.swap(tmp_nodes);
|
||||
|
||||
// sort according to center
|
||||
std::sort(nodes_in_row.begin(), nodes_in_row.end(), [&](int node_id1, int node_id2) {
|
||||
float x1 = host_x[node_id1] + host_node_size_x[node_id1] / 2;
|
||||
float x2 = host_x[node_id2] + host_node_size_x[node_id2] / 2;
|
||||
return x1 < x2 || (x1 == x2 && node_id1 < node_id2);
|
||||
});
|
||||
|
||||
for (int j = 1, je = nodes_in_row.size(); j < je; ++j) {
|
||||
int node_id1 = nodes_in_row.at(j - 1);
|
||||
int node_id2 = nodes_in_row.at(j);
|
||||
float xl1 = host_x[node_id1];
|
||||
float xl2 = host_x[node_id2];
|
||||
float width1 = host_node_size_x[node_id1];
|
||||
float width2 = host_node_size_x[node_id2];
|
||||
float xh1 = xl1 + width1;
|
||||
float xh2 = xl2 + width2;
|
||||
float yl1 = host_y[node_id1];
|
||||
float yl2 = host_y[node_id2];
|
||||
float yh1 = yl1 + host_node_size_y[node_id1];
|
||||
float yh2 = yl2 + host_node_size_y[node_id2];
|
||||
assert_msg(xl1 < xl2 && xh1 < xh2,
|
||||
"node %d (%g, %g, %g, %g) overlaps with node %d (%g, %g, %g, %g)",
|
||||
node_id1,
|
||||
xl1,
|
||||
yl1,
|
||||
xh1,
|
||||
yh1,
|
||||
node_id2,
|
||||
xl2,
|
||||
yl2,
|
||||
xh2,
|
||||
yh2);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void distributeMovableAndFixedCells2Bins(const float* x,
|
||||
const float* y,
|
||||
const float* node_size_x,
|
||||
const float* node_size_y,
|
||||
float bin_size_x,
|
||||
float bin_size_y,
|
||||
float xl,
|
||||
float yl,
|
||||
float xh,
|
||||
float yh,
|
||||
float site_width,
|
||||
int num_bins_x,
|
||||
int num_bins_y,
|
||||
int num_nodes,
|
||||
int num_movable_nodes,
|
||||
std::vector<std::vector<int>>& bin_cells) {
|
||||
for (int i = 0; i < num_nodes; i += 1) {
|
||||
if (i < num_movable_nodes && roundDiv(node_size_y[i], bin_size_y) <= 1) {
|
||||
// single-row movable nodes only distribute to one bin
|
||||
int bin_id_x = (x[i] + node_size_x[i] / 2 - xl) / bin_size_x;
|
||||
int bin_id_y = (y[i] + node_size_y[i] / 2 - yl) / bin_size_y;
|
||||
|
||||
bin_id_x = std::min(std::max(bin_id_x, 0), num_bins_x - 1);
|
||||
bin_id_y = std::min(std::max(bin_id_y, 0), num_bins_y - 1);
|
||||
|
||||
int bin_id = bin_id_x * num_bins_y + bin_id_y;
|
||||
|
||||
bin_cells[bin_id].push_back(i);
|
||||
} else {
|
||||
// fixed nodes may distribute to multiple bins
|
||||
int node_id = i;
|
||||
int bin_id_xl = std::max((x[node_id] - xl) / bin_size_x, (float)0);
|
||||
int bin_id_xh = std::min((int)ceil((x[node_id] + node_size_x[node_id] - xl) / bin_size_x), num_bins_x);
|
||||
int bin_id_yl = std::max((y[node_id] - yl) / bin_size_y, (float)0);
|
||||
int bin_id_yh = std::min((int)ceil((y[node_id] + node_size_y[node_id] - yl) / bin_size_y), num_bins_y);
|
||||
|
||||
for (int bin_id_x = bin_id_xl; bin_id_x < bin_id_xh; ++bin_id_x) {
|
||||
for (int bin_id_y = bin_id_yl; bin_id_y < bin_id_yh; ++bin_id_y) {
|
||||
int bin_id = bin_id_x * num_bins_y + bin_id_y;
|
||||
|
||||
bin_cells[bin_id].push_back(node_id);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// @param row_nodes node indices in this row
|
||||
/// @param clusters pre-allocated clusters in this row with the same length as
|
||||
/// that of row_nodes
|
||||
/// @param num_row_nodes length of row_nodes
|
||||
/// @return true if succeed, otherwise false
|
||||
bool abacusPlaceRowCPU(const float* init_x,
|
||||
const float* node_size_x,
|
||||
const float* node_size_y,
|
||||
float* x,
|
||||
float row_height,
|
||||
float xl,
|
||||
float xh,
|
||||
int num_nodes,
|
||||
int num_movable_nodes,
|
||||
int* row_nodes,
|
||||
AbacusCluster* clusters,
|
||||
int num_row_nodes) {
|
||||
// a very large number
|
||||
float M = std::pow(10, ceilDiv(std::log((xh - xl) * num_row_nodes), log(10)));
|
||||
bool ret_flag = true;
|
||||
|
||||
// merge two clusters
|
||||
// the second cluster will be invalid
|
||||
auto merge_cluster = [&](int dst_cluster_id, int src_cluster_id) {
|
||||
assert(dst_cluster_id < num_row_nodes);
|
||||
AbacusCluster& dst_cluster = clusters[dst_cluster_id];
|
||||
assert(src_cluster_id < num_row_nodes);
|
||||
AbacusCluster& src_cluster = clusters[src_cluster_id];
|
||||
|
||||
assert(dst_cluster.valid() && src_cluster.valid());
|
||||
for (int i = dst_cluster_id + 1; i < src_cluster_id; ++i) {
|
||||
assert(!clusters[i].valid());
|
||||
}
|
||||
dst_cluster.end_row_node_id = src_cluster.end_row_node_id;
|
||||
assert(dst_cluster.e < M && src_cluster.e < M);
|
||||
dst_cluster.e += src_cluster.e;
|
||||
dst_cluster.q += src_cluster.q - src_cluster.e * dst_cluster.w;
|
||||
dst_cluster.w += src_cluster.w;
|
||||
// update linked list
|
||||
if (src_cluster.next_cluster_id < num_row_nodes) {
|
||||
clusters[src_cluster.next_cluster_id].prev_cluster_id = dst_cluster_id;
|
||||
}
|
||||
dst_cluster.next_cluster_id = src_cluster.next_cluster_id;
|
||||
src_cluster.prev_cluster_id = std::numeric_limits<int>::min();
|
||||
src_cluster.next_cluster_id = std::numeric_limits<int>::min();
|
||||
};
|
||||
|
||||
// collapse clusters between [0, cluster_id]
|
||||
// compute the locations and merge clusters
|
||||
auto collapse = [&](int cluster_id, float range_xl, float range_xh) {
|
||||
int cur_cluster_id = cluster_id;
|
||||
assert(cur_cluster_id < num_row_nodes);
|
||||
int prev_cluster_id = clusters[cur_cluster_id].prev_cluster_id;
|
||||
AbacusCluster* cluster = nullptr;
|
||||
AbacusCluster* prev_cluster = nullptr;
|
||||
|
||||
while (true) {
|
||||
assert(cur_cluster_id < num_row_nodes);
|
||||
cluster = &clusters[cur_cluster_id];
|
||||
cluster->x = cluster->q / cluster->e;
|
||||
// make sure cluster >= range_xl, so fixed nodes will not be moved
|
||||
// in illegal case, cluster+w > range_xh may occur, but it is OK.
|
||||
// We can collect failed clusters later
|
||||
cluster->x = std::max(std::min(cluster->x, range_xh - cluster->w), range_xl);
|
||||
assert(cluster->x >= range_xl && cluster->x + cluster->w <= range_xh);
|
||||
|
||||
prev_cluster_id = cluster->prev_cluster_id;
|
||||
if (prev_cluster_id >= 0) {
|
||||
prev_cluster = &clusters[prev_cluster_id];
|
||||
if (prev_cluster->x + prev_cluster->w > cluster->x) {
|
||||
merge_cluster(prev_cluster_id, cur_cluster_id);
|
||||
cur_cluster_id = prev_cluster_id;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// initial cluster has only one cell
|
||||
for (int i = 0; i < num_row_nodes; ++i) {
|
||||
int node_id = row_nodes[i];
|
||||
AbacusCluster& cluster = clusters[i];
|
||||
cluster.prev_cluster_id = i - 1;
|
||||
cluster.next_cluster_id = i + 1;
|
||||
cluster.bgn_row_node_id = i;
|
||||
cluster.end_row_node_id = i;
|
||||
cluster.e = (node_id < num_movable_nodes && node_size_y[node_id] <= row_height) ? 1.0 : M;
|
||||
cluster.q = cluster.e * init_x[node_id];
|
||||
cluster.w = node_size_x[node_id];
|
||||
// this is required since we also include fixed nodes
|
||||
cluster.x = (node_id < num_movable_nodes && node_size_y[node_id] > row_height) ? x[node_id] : init_x[node_id];
|
||||
}
|
||||
|
||||
// kernel algorithm for placeRow
|
||||
float range_xl = xl;
|
||||
float range_xh = xh;
|
||||
for (int j = 0; j < num_row_nodes; ++j) {
|
||||
const AbacusCluster& next_cluster = clusters[j];
|
||||
if (next_cluster.e >= M) // fixed node
|
||||
{
|
||||
range_xh = std::min(next_cluster.x, range_xh);
|
||||
break;
|
||||
} else {
|
||||
assert(std::abs(node_size_y[row_nodes[j]] - row_height) < 1e-6);
|
||||
}
|
||||
}
|
||||
for (int i = 0; i < num_row_nodes; ++i) {
|
||||
const AbacusCluster& cluster = clusters[i];
|
||||
if (cluster.e < M) {
|
||||
assert(std::abs(node_size_y[row_nodes[i]] - row_height) < 1e-6);
|
||||
collapse(i, range_xl, range_xh);
|
||||
} else // set range xl/xh according to fixed nodes
|
||||
{
|
||||
range_xl = cluster.x + cluster.w;
|
||||
range_xh = xh;
|
||||
for (int j = i + 1; j < num_row_nodes; ++j) {
|
||||
const AbacusCluster& next_cluster = clusters[j];
|
||||
if (next_cluster.e >= M) // fixed node
|
||||
{
|
||||
range_xh = std::min(next_cluster.x, range_xh);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// apply solution
|
||||
for (int i = 0; i < num_row_nodes; ++i) {
|
||||
if (clusters[i].valid()) {
|
||||
const AbacusCluster& cluster = clusters[i];
|
||||
float xc = cluster.x;
|
||||
for (int j = cluster.bgn_row_node_id; j <= cluster.end_row_node_id; ++j) {
|
||||
int node_id = row_nodes[j];
|
||||
if (node_id < num_movable_nodes && std::abs(node_size_y[node_id] - row_height) < 1e-6) {
|
||||
x[node_id] = xc;
|
||||
} else if (xc != x[node_id]) {
|
||||
if (node_id < num_movable_nodes)
|
||||
logger.warning(
|
||||
"multi-row node %d tends to move from %.12f to "
|
||||
"%.12f, ignored",
|
||||
node_id,
|
||||
x[node_id],
|
||||
xc);
|
||||
else
|
||||
logger.warning(
|
||||
"fixed node %d tends to move from %.12f to %.12f, ignored", node_id, x[node_id], xc);
|
||||
ret_flag = false;
|
||||
}
|
||||
xc += node_size_x[node_id];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return ret_flag;
|
||||
}
|
||||
|
||||
void abacusLegalizeRow(const float* init_x,
|
||||
const float* node_size_x,
|
||||
const float* node_size_y,
|
||||
float* x,
|
||||
float* y,
|
||||
float xl,
|
||||
float xh,
|
||||
float bin_size_x,
|
||||
float bin_size_y,
|
||||
int num_bins_x,
|
||||
int num_bins_y,
|
||||
int num_nodes,
|
||||
int num_movable_nodes,
|
||||
std::vector<std::vector<int>>& bin_cells,
|
||||
std::vector<std::vector<AbacusCluster>>& bin_clusters) {
|
||||
for (unsigned int i = 0; i < bin_cells.size(); i += 1) {
|
||||
auto& row2nodes = bin_cells.at(i);
|
||||
|
||||
// sort bin cells from left to right
|
||||
sortNodesInRow(x, y, node_size_x, node_size_y, num_movable_nodes, row2nodes);
|
||||
|
||||
auto& clusters = bin_clusters.at(i);
|
||||
int num_row_nodes = row2nodes.size();
|
||||
|
||||
int bin_id_x = i / num_bins_y;
|
||||
// int bin_id_y = i-bin_id_x*num_bins_y;
|
||||
|
||||
float bin_xl = xl + bin_size_x * bin_id_x;
|
||||
float bin_xh = std::min(bin_xl + bin_size_x, xh);
|
||||
|
||||
abacusPlaceRowCPU(init_x,
|
||||
node_size_x,
|
||||
node_size_y,
|
||||
x,
|
||||
bin_size_y, // must be equal to row_height
|
||||
bin_xl,
|
||||
bin_xh,
|
||||
num_nodes,
|
||||
num_movable_nodes,
|
||||
row2nodes.data(),
|
||||
clusters.data(),
|
||||
num_row_nodes);
|
||||
}
|
||||
float displace = 0;
|
||||
for (int i = 0; i < num_movable_nodes; ++i) {
|
||||
displace += fabs(x[i] - init_x[i]);
|
||||
}
|
||||
logger.debug("average displace = %g", displace / num_movable_nodes);
|
||||
}
|
||||
|
||||
void abacusLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
|
||||
LegalizationData db(at_db);
|
||||
db.set_num_bins(num_bins_x, num_bins_y);
|
||||
// adjust bin sizes
|
||||
float bin_size_x = (db.xh - db.xl) / num_bins_x;
|
||||
float bin_size_y = db.row_height;
|
||||
num_bins_y = ceilDiv(db.yh - db.yl, bin_size_y);
|
||||
|
||||
// include both movable and fixed nodes
|
||||
std::vector<std::vector<int>> bin_cells(num_bins_x * num_bins_y);
|
||||
// distribute cells to bins
|
||||
distributeMovableAndFixedCells2Bins(db.x,
|
||||
db.y,
|
||||
db.node_size_x,
|
||||
db.node_size_y,
|
||||
bin_size_x,
|
||||
bin_size_y,
|
||||
db.xl,
|
||||
db.yl,
|
||||
db.xh,
|
||||
db.yh,
|
||||
db.site_width,
|
||||
num_bins_x,
|
||||
num_bins_y,
|
||||
db.num_nodes,
|
||||
db.num_movable_nodes,
|
||||
bin_cells);
|
||||
|
||||
std::vector<std::vector<AbacusCluster>> bin_clusters(num_bins_x * num_bins_y);
|
||||
for (unsigned int i = 0; i < bin_cells.size(); ++i) {
|
||||
bin_clusters[i].resize(bin_cells[i].size());
|
||||
}
|
||||
|
||||
abacusLegalizeRow(db.init_x,
|
||||
db.node_size_x,
|
||||
db.node_size_y,
|
||||
db.x,
|
||||
db.y,
|
||||
db.xl,
|
||||
db.xh,
|
||||
bin_size_x,
|
||||
bin_size_y,
|
||||
num_bins_x,
|
||||
num_bins_y,
|
||||
db.num_nodes,
|
||||
db.num_movable_nodes,
|
||||
bin_cells,
|
||||
bin_clusters);
|
||||
// need to align nodes to sites
|
||||
// this also considers cell width which is not integral times of site_width
|
||||
for (auto const& cells : bin_cells) {
|
||||
float xxl = db.xl;
|
||||
for (auto node_id : cells) {
|
||||
if (node_id < db.num_movable_nodes) {
|
||||
db.x[node_id] = std::max(std::min(db.x[node_id], db.xh - db.node_size_x[node_id]), xxl);
|
||||
db.x[node_id] = floorDiv(db.x[node_id] - db.xl, db.site_width) * db.site_width + db.xl;
|
||||
xxl = db.x[node_id] + db.node_size_x[node_id];
|
||||
} else if (node_id < db.num_nodes) {
|
||||
xxl = ceilDiv(db.x[node_id] + db.node_size_x[node_id] - db.xl, db.site_width) * db.site_width + db.xl;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
709
cpp_to_py/gpudp/lg/greedy_legalize.cpp
Normal file
709
cpp_to_py/gpudp/lg/greedy_legalize.cpp
Normal file
@ -0,0 +1,709 @@
|
||||
#include "gpudp/lg/legalization_db.h"
|
||||
|
||||
namespace dp {
|
||||
|
||||
template <typename T>
|
||||
struct Interval {
|
||||
T xl;
|
||||
T xh;
|
||||
|
||||
Interval(T l, T h) : xl(l), xh(h) {}
|
||||
|
||||
void intersect(T rhs_xl, T rhs_xh) {
|
||||
xl = std::max(xl, rhs_xl);
|
||||
xh = std::min(xh, rhs_xh);
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
struct Blank {
|
||||
T xl;
|
||||
T yl;
|
||||
T xh;
|
||||
T yh;
|
||||
|
||||
void intersect(const Blank& rhs) {
|
||||
xl = std::max(xl, rhs.xl);
|
||||
xh = std::min(xh, rhs.xh);
|
||||
yl = std::max(yl, rhs.yl);
|
||||
yh = std::min(yh, rhs.yh);
|
||||
}
|
||||
};
|
||||
|
||||
void distributeCells2Bins(const LegalizationData& db,
|
||||
const float* x,
|
||||
const float* y,
|
||||
const float* node_size_x,
|
||||
const float* node_size_y,
|
||||
float bin_size_x,
|
||||
float bin_size_y,
|
||||
float xl,
|
||||
float yl,
|
||||
float xh,
|
||||
float yh,
|
||||
int num_bins_x,
|
||||
int num_bins_y,
|
||||
int num_nodes,
|
||||
int num_movable_nodes,
|
||||
std::vector<std::vector<int>>& bin_cells) {
|
||||
// do not handle large macros
|
||||
// one cell cannot be distributed to one bin
|
||||
for (int i = 0; i < num_movable_nodes; i += 1) {
|
||||
if (!db.is_dummy_fixed(i)) {
|
||||
int bin_id_x = (x[i] + node_size_x[i] / 2 - xl) / bin_size_x;
|
||||
int bin_id_y = (y[i] + node_size_y[i] / 2 - yl) / bin_size_y;
|
||||
|
||||
bin_id_x = std::min(std::max(bin_id_x, 0), num_bins_x - 1);
|
||||
bin_id_y = std::min(std::max(bin_id_y, 0), num_bins_y - 1);
|
||||
|
||||
int bin_id = bin_id_x * num_bins_y + bin_id_y;
|
||||
|
||||
bin_cells[bin_id].push_back(i);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void distributeFixedCells2Bins(const LegalizationData& db,
|
||||
const float* x,
|
||||
const float* y,
|
||||
const float* node_size_x,
|
||||
const float* node_size_y,
|
||||
float bin_size_x,
|
||||
float bin_size_y,
|
||||
float xl,
|
||||
float yl,
|
||||
float xh,
|
||||
float yh,
|
||||
int num_bins_x,
|
||||
int num_bins_y,
|
||||
int num_nodes,
|
||||
int num_movable_nodes,
|
||||
std::vector<std::vector<int>>& bin_cells) {
|
||||
// one cell can be assigned to multiple bins
|
||||
for (int i = 0; i < num_nodes; i += 1) {
|
||||
if (db.is_dummy_fixed(i) || i >= num_movable_nodes) {
|
||||
int node_id = i;
|
||||
int bin_id_xl = std::max((int)floorDiv(x[node_id] - xl, bin_size_x), 0);
|
||||
int bin_id_xh = std::min((int)ceilDiv((x[node_id] + node_size_x[node_id] - xl), bin_size_x), num_bins_x);
|
||||
int bin_id_yl = std::max((int)floorDiv(y[node_id] - yl, bin_size_y), 0);
|
||||
int bin_id_yh = std::min((int)ceilDiv((y[node_id] + node_size_y[node_id] - yl), bin_size_y), num_bins_y);
|
||||
|
||||
for (int bin_id_x = bin_id_xl; bin_id_x < bin_id_xh; ++bin_id_x) {
|
||||
for (int bin_id_y = bin_id_yl; bin_id_y < bin_id_yh; ++bin_id_y) {
|
||||
int bin_id = bin_id_x * num_bins_y + bin_id_y;
|
||||
|
||||
bin_cells[bin_id].push_back(node_id);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void distributeBlanks2Bins(const float* x,
|
||||
const float* y,
|
||||
const float* node_size_x,
|
||||
const float* node_size_y,
|
||||
const std::vector<std::vector<int>>& bin_fixed_cells,
|
||||
float bin_size_x,
|
||||
float bin_size_y,
|
||||
float blank_bin_size_y,
|
||||
float xl,
|
||||
float yl,
|
||||
float xh,
|
||||
float yh,
|
||||
float site_width,
|
||||
float row_height,
|
||||
int num_bins_x,
|
||||
int num_bins_y,
|
||||
int blank_num_bins_y,
|
||||
std::vector<std::vector<Blank<float>>>& bin_blanks) {
|
||||
for (int i = 0; i < num_bins_x * num_bins_y; i += 1) {
|
||||
int bin_id_x = i / num_bins_y;
|
||||
int bin_id_y = i - bin_id_x * num_bins_y;
|
||||
int blank_num_bins_per_bin = roundDiv(bin_size_y, blank_bin_size_y);
|
||||
int blank_bin_id_yl = bin_id_y * blank_num_bins_per_bin;
|
||||
int blank_bin_id_yh = std::min(blank_bin_id_yl + blank_num_bins_per_bin, blank_num_bins_y);
|
||||
for (int blank_bin_id_y = blank_bin_id_yl; blank_bin_id_y < blank_bin_id_yh; ++blank_bin_id_y) {
|
||||
float bin_xl = xl + bin_id_x * bin_size_x;
|
||||
float bin_xh = std::min(bin_xl + bin_size_x, xh);
|
||||
float bin_yl = yl + blank_bin_id_y * blank_bin_size_y;
|
||||
float bin_yh = std::min(bin_yl + blank_bin_size_y, yh);
|
||||
int blank_bin_id = bin_id_x * blank_num_bins_y + blank_bin_id_y;
|
||||
|
||||
for (float by = bin_yl; by < bin_yh; by += row_height) {
|
||||
Blank<float> blank;
|
||||
blank.xl = floorDiv((bin_xl - xl), site_width) * site_width + xl; // align blanks to sites
|
||||
blank.xh = floorDiv((bin_xh - xl), site_width) * site_width + xl; // align blanks to sites
|
||||
blank.yl = by;
|
||||
blank.yh = by + row_height;
|
||||
|
||||
bin_blanks.at(blank_bin_id).push_back(blank);
|
||||
}
|
||||
|
||||
const std::vector<int>& cells = bin_fixed_cells.at(i);
|
||||
std::vector<Blank<float>>& blanks = bin_blanks.at(blank_bin_id);
|
||||
|
||||
for (unsigned int bi = 0; bi < blanks.size(); ++bi) {
|
||||
Blank<float>& blank = blanks.at(bi);
|
||||
for (unsigned int ci = 0; ci < cells.size(); ++ci) {
|
||||
int node_id = cells.at(ci);
|
||||
float node_xl = x[node_id];
|
||||
float node_yl = y[node_id];
|
||||
float node_xh = node_xl + node_size_x[node_id];
|
||||
float node_yh = node_yl + node_size_y[node_id];
|
||||
|
||||
if (node_yh > blank.yl && node_yl < blank.yh && node_xh > blank.xl &&
|
||||
node_xl < blank.xh) // overlap
|
||||
{
|
||||
if (node_xl <= blank.xl && node_xh >= blank.xh) // erase
|
||||
{
|
||||
bin_blanks.at(blank_bin_id).erase(bin_blanks.at(blank_bin_id).begin() + bi);
|
||||
--bi;
|
||||
break;
|
||||
} else if (node_xl <= blank.xl) { // one blank
|
||||
blank.xl = ceilDiv((node_xh - xl), site_width) * site_width + xl; // align blanks to sites
|
||||
} else if (node_xh >= blank.xh) { // one blank
|
||||
blank.xh = floorDiv((node_xl - xl), site_width) * site_width + xl; // align blanks to sites
|
||||
} else { // two blanks
|
||||
Blank<float> new_blank = blank;
|
||||
blank.xh = floorDiv((node_xl - xl), site_width) * site_width + xl; // align blanks to sites
|
||||
new_blank.xl =
|
||||
floorDiv((node_xh - xl), site_width) * site_width + xl; // align blanks to sites
|
||||
bin_blanks.at(blank_bin_id).insert(bin_blanks.at(blank_bin_id).begin() + bi + 1, new_blank);
|
||||
--bi;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void legalizeBin(
|
||||
const float* init_x,
|
||||
const float* init_y,
|
||||
const float* node_size_x,
|
||||
const float* node_size_y,
|
||||
std::vector<std::vector<Blank<float>>>& bin_blanks, // blanks in each bin, sorted from low to high, left to right
|
||||
std::vector<std::vector<int>>& bin_cells, // unplaced cells in each bin
|
||||
float* x,
|
||||
float* y,
|
||||
int num_bins_x,
|
||||
int num_bins_y,
|
||||
int blank_num_bins_y,
|
||||
float bin_size_x,
|
||||
float bin_size_y,
|
||||
float blank_bin_size_y,
|
||||
float site_width,
|
||||
float row_height,
|
||||
float xl,
|
||||
float yl,
|
||||
float xh,
|
||||
float yh,
|
||||
float alpha, // a parameter to tune anchor initial locations and current locations
|
||||
float beta, // a parameter to tune space reserving
|
||||
bool lr_flag, // from left to right
|
||||
int* num_unplaced_cells) {
|
||||
for (int i = 0; i < num_bins_x * num_bins_y; i += 1) {
|
||||
int bin_id_x = i / num_bins_y;
|
||||
int bin_id_y = i - bin_id_x * num_bins_y;
|
||||
int blank_num_bins_per_bin = roundDiv(bin_size_y, blank_bin_size_y);
|
||||
int blank_bin_id_yl = bin_id_y * blank_num_bins_per_bin;
|
||||
int blank_bin_id_yh = std::min(blank_bin_id_yl + blank_num_bins_per_bin, blank_num_bins_y);
|
||||
|
||||
// cells in this bin
|
||||
std::vector<int>& cells = bin_cells.at(i);
|
||||
|
||||
// sort cells according to width
|
||||
if (lr_flag) {
|
||||
std::sort(cells.begin(), cells.end(), [&](int i, int j) -> bool {
|
||||
float wi = -1000 * (init_x[i] + node_size_x[i] / 2) + node_size_x[i] + node_size_y[i];
|
||||
float wj = -1000 * (init_x[j] + node_size_x[j] / 2) + node_size_x[j] + node_size_y[j];
|
||||
return wi < wj || (wi == wj && (init_y[i] > init_y[j] || (init_y[i] == init_y[j] && i < j)));
|
||||
});
|
||||
} else {
|
||||
std::sort(cells.begin(), cells.end(), [&](int i, int j) -> bool {
|
||||
float wi = 1000 * (init_x[i] + node_size_x[i] / 2) + node_size_x[i] + node_size_y[i];
|
||||
float wj = 1000 * (init_x[j] + node_size_x[j] / 2) + node_size_x[j] + node_size_y[j];
|
||||
return wi < wj || (wi == wj && (init_y[i] < init_y[j] || (init_y[i] == init_y[j] && i < j)));
|
||||
});
|
||||
}
|
||||
|
||||
for (int ci = bin_cells.at(i).size() - 1; ci >= 0; --ci) {
|
||||
int node_id = cells.at(ci);
|
||||
// align to site
|
||||
float init_xl =
|
||||
floorDiv(((alpha * init_x[node_id] + (1 - alpha) * x[node_id]) - xl), site_width) * site_width + xl;
|
||||
float init_yl = (alpha * init_y[node_id] + (1 - alpha) * y[node_id]);
|
||||
float width = ceilDiv(node_size_x[node_id], site_width) * site_width;
|
||||
float height = node_size_y[node_id];
|
||||
|
||||
int num_node_rows = ceilDiv(height, row_height); // may take multiple rows
|
||||
int blank_index_offset[num_node_rows];
|
||||
std::fill(blank_index_offset, blank_index_offset + num_node_rows, 0);
|
||||
|
||||
int blank_initial_bin_id_y = floorDiv((init_yl - yl), blank_bin_size_y);
|
||||
blank_initial_bin_id_y = std::min(blank_bin_id_yh - 1, std::max(blank_bin_id_yl, blank_initial_bin_id_y));
|
||||
int blank_bin_id_dist_y = std::max(blank_initial_bin_id_y + 1, blank_bin_id_yh - blank_initial_bin_id_y);
|
||||
|
||||
int best_blank_bin_id_y = -1;
|
||||
int best_blank_bi[num_node_rows];
|
||||
std::fill(best_blank_bi, best_blank_bi + num_node_rows, -1);
|
||||
float best_cost = xh - xl + yh - yl;
|
||||
float best_xl = -1;
|
||||
float best_yl = -1;
|
||||
for (int bin_id_offset_y = 0; abs(bin_id_offset_y) < blank_bin_id_dist_y;
|
||||
bin_id_offset_y = (bin_id_offset_y > 0) ? -bin_id_offset_y : -(bin_id_offset_y - 1)) {
|
||||
int blank_bin_id_y = blank_initial_bin_id_y + bin_id_offset_y;
|
||||
if (blank_bin_id_y < blank_bin_id_yl || blank_bin_id_y + num_node_rows > blank_bin_id_yh) {
|
||||
continue;
|
||||
}
|
||||
int blank_bin_id = bin_id_x * blank_num_bins_y + blank_bin_id_y;
|
||||
// blanks in this bin
|
||||
const std::vector<Blank<float>>& blanks = bin_blanks.at(blank_bin_id);
|
||||
|
||||
int row_best_blank_bi[num_node_rows];
|
||||
std::fill(row_best_blank_bi, row_best_blank_bi + num_node_rows, -1);
|
||||
float row_best_cost = xh - xl + yh - yl;
|
||||
float row_best_xl = -1;
|
||||
float row_best_yl = -1;
|
||||
bool search_flag = true;
|
||||
for (unsigned int bi = 0; search_flag && bi < bin_blanks.at(blank_bin_id).size(); ++bi) {
|
||||
const Blank<float>& blank = blanks[bi];
|
||||
// for multi-row height cells, check blanks in upper rows
|
||||
// find blanks with maximum intersection
|
||||
blank_index_offset[0] = bi;
|
||||
std::fill(blank_index_offset + 1, blank_index_offset + num_node_rows, -1);
|
||||
|
||||
while (true) {
|
||||
Interval<float> intersect_blank(blank.xl, blank.xh);
|
||||
for (int row_offset = 1; row_offset < num_node_rows; ++row_offset) {
|
||||
int next_blank_bin_id_y = blank_bin_id_y + row_offset;
|
||||
int next_blank_bin_id = bin_id_x * blank_num_bins_y + next_blank_bin_id_y;
|
||||
unsigned int next_bi = blank_index_offset[row_offset] + 1;
|
||||
for (; next_bi < bin_blanks.at(next_blank_bin_id).size(); ++next_bi) {
|
||||
const Blank<float>& next_blank = bin_blanks.at(next_blank_bin_id)[next_bi];
|
||||
Interval<float> intersect_blank_tmp = intersect_blank;
|
||||
intersect_blank_tmp.intersect(next_blank.xl, next_blank.xh);
|
||||
if (intersect_blank_tmp.xh - intersect_blank_tmp.xl >= width) {
|
||||
intersect_blank = intersect_blank_tmp;
|
||||
blank_index_offset[row_offset] = next_bi;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (next_bi == bin_blanks.at(next_blank_bin_id).size()) // not found
|
||||
{
|
||||
intersect_blank.xl = intersect_blank.xh = 0;
|
||||
break;
|
||||
}
|
||||
}
|
||||
float intersect_blank_width = intersect_blank.xh - intersect_blank.xl;
|
||||
if (intersect_blank_width >= width) {
|
||||
// compute displacement
|
||||
float target_xl = init_xl;
|
||||
float target_yl = blank.yl;
|
||||
// alow tolerance to avoid more dead space
|
||||
float beta = 4;
|
||||
float tolerance = std::min(beta * width, intersect_blank_width / beta);
|
||||
if (target_xl <= intersect_blank.xl + tolerance) {
|
||||
target_xl = intersect_blank.xl;
|
||||
} else if (target_xl + width >= intersect_blank.xh - tolerance) {
|
||||
target_xl = (intersect_blank.xh - width);
|
||||
}
|
||||
float cost = fabs(target_xl - init_xl) + fabs(target_yl - init_yl);
|
||||
// update best cost
|
||||
if (cost < row_best_cost) {
|
||||
std::copy(blank_index_offset, blank_index_offset + num_node_rows, row_best_blank_bi);
|
||||
row_best_cost = cost;
|
||||
row_best_xl = target_xl;
|
||||
row_best_yl = target_yl;
|
||||
} else { // early exit since we iterate within rows from left to right
|
||||
search_flag = false;
|
||||
}
|
||||
} else { // not found
|
||||
break;
|
||||
}
|
||||
if (num_node_rows < 2) { // for single-row height cells
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (row_best_cost < best_cost) {
|
||||
best_blank_bin_id_y = blank_bin_id_y;
|
||||
std::copy(row_best_blank_bi, row_best_blank_bi + num_node_rows, best_blank_bi);
|
||||
best_cost = row_best_cost;
|
||||
best_xl = row_best_xl;
|
||||
best_yl = row_best_yl;
|
||||
} else if (best_cost + row_height < bin_id_offset_y * row_height) {
|
||||
break; // early exit since we iterate from close row to far-away row
|
||||
}
|
||||
}
|
||||
|
||||
// found blank
|
||||
if (best_blank_bin_id_y >= 0) {
|
||||
x[node_id] = best_xl;
|
||||
y[node_id] = best_yl;
|
||||
// update cell position and blank
|
||||
for (int row_offset = 0; row_offset < num_node_rows; ++row_offset) {
|
||||
assert(best_blank_bi[row_offset] >= 0);
|
||||
// blanks in this bin
|
||||
int best_blank_bin_id = bin_id_x * blank_num_bins_y + best_blank_bin_id_y + row_offset;
|
||||
std::vector<Blank<float>>& blanks = bin_blanks.at(best_blank_bin_id);
|
||||
Blank<float>& blank = blanks.at(best_blank_bi[row_offset]);
|
||||
assert(best_xl >= blank.xl && best_xl + width <= blank.xh);
|
||||
assert(best_yl + row_height * row_offset == blank.yl);
|
||||
if (best_xl == blank.xl) {
|
||||
// update blank
|
||||
blank.xl += width;
|
||||
if (floorDiv((blank.xl - xl), site_width) * site_width != blank.xl - xl) {
|
||||
logger.debug("1. move node %d from %g to %g, blank (%g, %g)",
|
||||
node_id,
|
||||
x[node_id],
|
||||
blank.xl,
|
||||
blank.xl,
|
||||
blank.xh);
|
||||
}
|
||||
if (blank.xl >= blank.xh) {
|
||||
bin_blanks.at(best_blank_bin_id)
|
||||
.erase(bin_blanks.at(best_blank_bin_id).begin() + best_blank_bi[row_offset]);
|
||||
}
|
||||
} else if (best_xl + width == blank.xh) {
|
||||
// update blank
|
||||
blank.xh -= width;
|
||||
if (floorDiv((blank.xh - xl), site_width) * site_width != blank.xh - xl) {
|
||||
logger.debug("2. move node %d from %g to %g, blank (%g, %g)",
|
||||
node_id,
|
||||
x[node_id],
|
||||
blank.xh - width,
|
||||
blank.xl,
|
||||
blank.xh);
|
||||
}
|
||||
if (blank.xl >= blank.xh) {
|
||||
bin_blanks.at(best_blank_bin_id)
|
||||
.erase(bin_blanks.at(best_blank_bin_id).begin() + best_blank_bi[row_offset]);
|
||||
}
|
||||
} else {
|
||||
// need to update current blank and insert one more blank
|
||||
Blank<float> new_blank;
|
||||
new_blank.xl = best_xl + width;
|
||||
new_blank.xh = blank.xh;
|
||||
new_blank.yl = blank.yl;
|
||||
new_blank.yh = blank.yh;
|
||||
blank.xh = best_xl;
|
||||
if (floorDiv((blank.xl - xl), site_width) * site_width != blank.xl - xl ||
|
||||
floorDiv((blank.xh - xl), site_width) * site_width != blank.xh - xl ||
|
||||
floorDiv((new_blank.xl - xl), site_width) * site_width != new_blank.xl - xl ||
|
||||
floorDiv((new_blank.xh - xl), site_width) * site_width != new_blank.xh - xl) {
|
||||
logger.debug("3. move node %d from %g to %g, blank (%g, %g), new_blank (%g, %g)",
|
||||
node_id,
|
||||
x[node_id],
|
||||
init_xl,
|
||||
blank.xl,
|
||||
blank.xh,
|
||||
new_blank.xl,
|
||||
new_blank.xh);
|
||||
}
|
||||
bin_blanks.at(best_blank_bin_id)
|
||||
.insert(bin_blanks.at(best_blank_bin_id).begin() + best_blank_bi[row_offset] + 1,
|
||||
new_blank);
|
||||
}
|
||||
}
|
||||
|
||||
// remove from cells
|
||||
bin_cells.at(i).erase(bin_cells.at(i).begin() + ci);
|
||||
}
|
||||
}
|
||||
*num_unplaced_cells += bin_cells.at(i).size();
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
void resizeBinObjects(std::vector<std::vector<T>>& bin_objs, int num_bins_x, int num_bins_y) {
|
||||
bin_objs.resize(num_bins_x * num_bins_y);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
void countBinObjects(const std::vector<std::vector<T>>& bin_objs) {
|
||||
int count = 0;
|
||||
for (unsigned int i = 0; i < bin_objs.size(); ++i) {
|
||||
count += bin_objs.at(i).size();
|
||||
}
|
||||
}
|
||||
|
||||
void mergeBinBlanks(const std::vector<std::vector<Blank<float>>>& src_bin_blanks,
|
||||
int src_num_bins_x,
|
||||
int src_num_bins_y, // dimensions for the src
|
||||
std::vector<std::vector<Blank<float>>>& dst_bin_blanks,
|
||||
int dst_num_bins_x,
|
||||
int dst_num_bins_y, // dimensions for the dst
|
||||
int scale_ratio_x, // roughly src_num_bins_x/dst_num_bins_x
|
||||
float min_blank_width // minimum blank width to consider
|
||||
) {
|
||||
for (int i = 0; i < dst_num_bins_x * dst_num_bins_y; i += 1) {
|
||||
// assume src_num_bins_y == dst_num_bins_y
|
||||
int dst_bin_id_x = i / dst_num_bins_y;
|
||||
int dst_bin_id_y = i - dst_bin_id_x * dst_num_bins_y;
|
||||
|
||||
int src_bin_id_x_bgn = dst_bin_id_x * scale_ratio_x;
|
||||
int src_bin_id_x_end = std::min(src_bin_id_x_bgn + scale_ratio_x, src_num_bins_x);
|
||||
|
||||
std::vector<Blank<float>>& dst_bin_blank = dst_bin_blanks.at(i);
|
||||
|
||||
for (int ix = src_bin_id_x_bgn; ix < src_bin_id_x_end; ++ix) {
|
||||
int iy = dst_bin_id_y; // same as src_bin_id_y
|
||||
int src_bin_id = ix * src_num_bins_y + iy;
|
||||
const std::vector<Blank<float>>& src_bin_blank = src_bin_blanks.at(src_bin_id);
|
||||
|
||||
int offset = 0;
|
||||
if (!dst_bin_blank.empty() && !src_bin_blank.empty()) {
|
||||
const Blank<float>& first_blank = src_bin_blank.at(0);
|
||||
Blank<float>& last_blank = dst_bin_blank.at(dst_bin_blank.size() - 1);
|
||||
if (last_blank.yl == first_blank.yl && last_blank.xh == first_blank.xl) {
|
||||
last_blank.xh = first_blank.xh;
|
||||
offset = 1;
|
||||
}
|
||||
}
|
||||
for (unsigned int k = offset; k < src_bin_blank.size(); ++k) {
|
||||
const Blank<float>& blank = src_bin_blank.at(k);
|
||||
// prune small blanks
|
||||
if (blank.xh - blank.xl >= min_blank_width) {
|
||||
dst_bin_blanks.at(i).push_back(blank);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void mergeBinCells(
|
||||
const std::vector<std::vector<int>>& src_bin_cells,
|
||||
int src_num_bins_x,
|
||||
int src_num_bins_y, // dimensions for the src
|
||||
std::vector<std::vector<int>>& dst_bin_cells,
|
||||
int dst_num_bins_x,
|
||||
int dst_num_bins_y, // dimensions for the dst
|
||||
int scale_ratio_x,
|
||||
int scale_ratio_y // roughly src_num_bins_x/dst_num_bins_x, but may not be exactly the same due to even/odd numbers
|
||||
) {
|
||||
for (int i = 0; i < dst_num_bins_x * dst_num_bins_y; i += 1) {
|
||||
int dst_bin_id_x = i / dst_num_bins_y;
|
||||
int dst_bin_id_y = i - dst_bin_id_x * dst_num_bins_y;
|
||||
|
||||
int src_bin_id_x_bgn = dst_bin_id_x * scale_ratio_x;
|
||||
int src_bin_id_y_bgn = dst_bin_id_y * scale_ratio_y;
|
||||
int src_bin_id_x_end = std::min(src_bin_id_x_bgn + scale_ratio_x, src_num_bins_x);
|
||||
int src_bin_id_y_end = std::min(src_bin_id_y_bgn + scale_ratio_y, src_num_bins_y);
|
||||
|
||||
for (int ix = src_bin_id_x_bgn; ix < src_bin_id_x_end; ++ix) {
|
||||
for (int iy = src_bin_id_y_bgn; iy < src_bin_id_y_end; ++iy) {
|
||||
int src_bin_id = ix * src_num_bins_y + iy;
|
||||
|
||||
const std::vector<int>& src_bin_cell = src_bin_cells.at(src_bin_id);
|
||||
|
||||
dst_bin_cells.at(i).insert(dst_bin_cells.at(i).end(), src_bin_cell.begin(), src_bin_cell.end());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void minNodeSize(const std::vector<std::vector<int>>& bin_cells,
|
||||
const float* node_size_x,
|
||||
const float* node_size_y,
|
||||
float site_width,
|
||||
float row_height,
|
||||
int num_bins_x,
|
||||
int num_bins_y,
|
||||
int* min_node_size_x) {
|
||||
for (int i = 0; i < num_bins_x * num_bins_y; i += 1) {
|
||||
const std::vector<int>& cells = bin_cells.at(i);
|
||||
float min_size_x = std::numeric_limits<int>::max();
|
||||
for (unsigned int k = 0; k < cells.size(); ++k) {
|
||||
int node_id = cells.at(k);
|
||||
min_size_x = std::min(min_size_x, node_size_x[node_id]);
|
||||
}
|
||||
if (min_size_x != std::numeric_limits<int>::max()) {
|
||||
*min_node_size_x = std::min(*min_node_size_x, (int)ceilDiv(min_size_x, site_width));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void greedyLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
|
||||
LegalizationData db(at_db);
|
||||
db.set_num_bins(num_bins_x, num_bins_y);
|
||||
// first from right to left
|
||||
// then from left to right
|
||||
for (int i = 0; i < 2; ++i) {
|
||||
num_bins_x = 1;
|
||||
num_bins_y = 1;
|
||||
// adjust bin sizes
|
||||
float bin_size_x = (db.xh - db.xl) / static_cast<float>(num_bins_x);
|
||||
float bin_size_y = (db.yh - db.yl) / static_cast<float>(num_bins_y);
|
||||
bin_size_y = std::max((float)(ceilDiv(bin_size_y, db.row_height) * db.row_height), db.row_height);
|
||||
num_bins_y = ceilDiv((db.yh - db.yl), bin_size_y);
|
||||
|
||||
// bin dimension in y direction for blanks is different from that for cells
|
||||
float blank_bin_size_y = db.row_height;
|
||||
int blank_num_bins_y = floorDiv((db.yh - db.yl), blank_bin_size_y);
|
||||
logger.debug("%s blank_num_bins_y = %d", "Standard cell legalization", blank_num_bins_y);
|
||||
|
||||
// allocate bin cells
|
||||
std::vector<std::vector<int>> bin_cells(num_bins_x * num_bins_y);
|
||||
std::vector<std::vector<int>> bin_cells_copy(num_bins_x * num_bins_y);
|
||||
|
||||
// distribute cells to bins
|
||||
distributeCells2Bins(db,
|
||||
db.x,
|
||||
db.y,
|
||||
db.node_size_x,
|
||||
db.node_size_y,
|
||||
bin_size_x,
|
||||
bin_size_y,
|
||||
db.xl,
|
||||
db.yl,
|
||||
db.xh,
|
||||
db.yh,
|
||||
num_bins_x,
|
||||
num_bins_y,
|
||||
db.num_nodes,
|
||||
db.num_movable_nodes,
|
||||
bin_cells);
|
||||
|
||||
// allocate bin fixed cells
|
||||
std::vector<std::vector<int>> bin_fixed_cells(num_bins_x * num_bins_y);
|
||||
|
||||
// distribute fixed cells to bins
|
||||
distributeFixedCells2Bins(db,
|
||||
db.init_x,
|
||||
db.init_y,
|
||||
db.node_size_x,
|
||||
db.node_size_y,
|
||||
bin_size_x,
|
||||
bin_size_y,
|
||||
db.xl,
|
||||
db.yl,
|
||||
db.xh,
|
||||
db.yh,
|
||||
num_bins_x,
|
||||
num_bins_y,
|
||||
db.num_nodes,
|
||||
db.num_movable_nodes,
|
||||
bin_fixed_cells);
|
||||
|
||||
// allocate bin blanks
|
||||
std::vector<std::vector<Blank<float>>> bin_blanks(num_bins_x * blank_num_bins_y);
|
||||
std::vector<std::vector<Blank<float>>> bin_blanks_copy(num_bins_x * blank_num_bins_y);
|
||||
|
||||
// distribute blanks to bins
|
||||
distributeBlanks2Bins(db.init_x,
|
||||
db.init_y,
|
||||
db.node_size_x,
|
||||
db.node_size_y,
|
||||
bin_fixed_cells,
|
||||
bin_size_x,
|
||||
bin_size_y,
|
||||
blank_bin_size_y,
|
||||
db.xl,
|
||||
db.yl,
|
||||
db.xh,
|
||||
db.yh,
|
||||
db.site_width,
|
||||
db.row_height,
|
||||
num_bins_x,
|
||||
num_bins_y,
|
||||
blank_num_bins_y,
|
||||
bin_blanks);
|
||||
|
||||
int num_unplaced_cells_host;
|
||||
// minimum width in sites
|
||||
int min_unplaced_node_size_x_host;
|
||||
int num_iters = floor(log((float)std::min(num_bins_x, num_bins_y)) / log(2.0)) + 1;
|
||||
for (int iter = 0; iter < num_iters; ++iter) {
|
||||
logger.debug(
|
||||
"%s iteration %d with %dx%d bins", "Standard cell legalization", iter, num_bins_x, num_bins_y);
|
||||
num_unplaced_cells_host = 0;
|
||||
logger.debug("%s #bin_blanks", "Standard cell legalization");
|
||||
countBinObjects(bin_blanks);
|
||||
|
||||
legalizeBin(db.init_x,
|
||||
db.init_y,
|
||||
db.node_size_x,
|
||||
db.node_size_y,
|
||||
bin_blanks, // blanks in each bin, sorted from low to high, left to right
|
||||
bin_cells, // unplaced cells in each bin
|
||||
db.x,
|
||||
db.y,
|
||||
num_bins_x,
|
||||
num_bins_y,
|
||||
blank_num_bins_y,
|
||||
bin_size_x,
|
||||
bin_size_y,
|
||||
blank_bin_size_y,
|
||||
db.site_width,
|
||||
db.row_height,
|
||||
db.xl,
|
||||
db.yl,
|
||||
db.xh,
|
||||
db.yh,
|
||||
0.5,
|
||||
4.0,
|
||||
i % 2,
|
||||
&num_unplaced_cells_host);
|
||||
logger.debug("%s num_unplaced_cells = %d", "Standard cell legalization", num_unplaced_cells_host);
|
||||
|
||||
if (num_unplaced_cells_host == 0 || iter + 1 == num_iters) {
|
||||
break;
|
||||
}
|
||||
|
||||
// compute minimum size of unplaced cells
|
||||
min_unplaced_node_size_x_host = floorDiv((db.xh - db.xl), db.site_width);
|
||||
minNodeSize(bin_cells,
|
||||
db.node_size_x,
|
||||
db.node_size_y,
|
||||
db.site_width,
|
||||
db.row_height,
|
||||
num_bins_x,
|
||||
num_bins_y,
|
||||
&min_unplaced_node_size_x_host);
|
||||
logger.debug("%s minimum unplaced node_size_x %d sites",
|
||||
"Standard cell legalization",
|
||||
min_unplaced_node_size_x_host);
|
||||
|
||||
// ceil(num_bins_x/2), ceil(num_bins_y/2)
|
||||
int dst_num_bins_x = (num_bins_x >> 1) + (num_bins_x & 1);
|
||||
int dst_num_bins_y = (num_bins_y >> 1) + (num_bins_y & 1);
|
||||
int scale_ratio_x = (num_bins_x == dst_num_bins_x) ? 1 : num_bins_x / dst_num_bins_x;
|
||||
int scale_ratio_y = (num_bins_y == dst_num_bins_y) ? 1 : num_bins_y / dst_num_bins_y;
|
||||
|
||||
resizeBinObjects(bin_cells_copy, dst_num_bins_x, dst_num_bins_y);
|
||||
mergeBinCells(bin_cells,
|
||||
num_bins_x,
|
||||
num_bins_y, // dimensions for the src
|
||||
bin_cells_copy, // ceil(src_num_bins_x/2) * ceil(src_num_bins_y/2)
|
||||
dst_num_bins_x,
|
||||
dst_num_bins_y,
|
||||
scale_ratio_x,
|
||||
scale_ratio_y);
|
||||
resizeBinObjects(bin_blanks_copy, dst_num_bins_x, blank_num_bins_y);
|
||||
mergeBinBlanks(bin_blanks,
|
||||
num_bins_x,
|
||||
blank_num_bins_y, // dimensions for the src
|
||||
bin_blanks_copy, // ceil(src_num_bins_x/2) * ceil(src_num_bins_y/2)
|
||||
dst_num_bins_x,
|
||||
blank_num_bins_y,
|
||||
scale_ratio_x,
|
||||
min_unplaced_node_size_x_host * db.site_width);
|
||||
|
||||
// update bin dimensions
|
||||
num_bins_x = dst_num_bins_x;
|
||||
num_bins_y = dst_num_bins_y;
|
||||
|
||||
bin_size_x = bin_size_x * 2;
|
||||
bin_size_y = bin_size_y * 2;
|
||||
|
||||
std::swap(bin_cells, bin_cells_copy);
|
||||
std::swap(bin_blanks, bin_blanks_copy);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
381
cpp_to_py/gpudp/lg/hannan_legalize.h
Normal file
381
cpp_to_py/gpudp/lg/hannan_legalize.h
Normal file
@ -0,0 +1,381 @@
|
||||
#pragma once
|
||||
#include <vector>
|
||||
|
||||
#include "gpudp/dp/ism/diamond_search.h"
|
||||
#include "gpudp/lg/legalization_db.h"
|
||||
|
||||
namespace dp {
|
||||
|
||||
/// @brief A class models Hannan grids.
|
||||
class HannanGrids {
|
||||
public:
|
||||
HannanGrids(const float* x,
|
||||
const float* y,
|
||||
const float* width,
|
||||
const float* height,
|
||||
std::size_t n,
|
||||
const float xl,
|
||||
const float yl,
|
||||
const float xh,
|
||||
const float yh,
|
||||
const float spacing_x,
|
||||
const float spacing_y) {
|
||||
build(x, y, width, height, n, xl, yl, xh, yh, spacing_x, spacing_y);
|
||||
}
|
||||
|
||||
std::size_t dim_x() const { return m_coordx.size(); }
|
||||
std::size_t dim_y() const { return m_coordy.size(); }
|
||||
/// @brief query x index in log(n) time complexity
|
||||
std::size_t grid_x(float x) const {
|
||||
auto it = std::lower_bound(m_coordx.begin(), m_coordx.end(), x);
|
||||
std::size_t ix = std::min((std::size_t)std::distance(m_coordx.begin(), it), dim_x() - 1);
|
||||
float gxl = m_coordx[ix];
|
||||
if (gxl > x && ix) {
|
||||
ix -= 1;
|
||||
}
|
||||
return ix;
|
||||
}
|
||||
/// @brief query y index in log(n) time complexity
|
||||
std::size_t grid_y(float y) const {
|
||||
auto it = std::lower_bound(m_coordy.begin(), m_coordy.end(), y);
|
||||
std::size_t iy = std::min((std::size_t)std::distance(m_coordy.begin(), it), dim_y() - 1);
|
||||
float gyl = m_coordy[iy];
|
||||
if (gyl > y && iy) {
|
||||
iy -= 1;
|
||||
}
|
||||
return iy;
|
||||
}
|
||||
/// @brief get x coordinate of a grid
|
||||
float coord_x(std::size_t ix) const { return m_coordx[ix]; }
|
||||
/// @brief get y coordinate of a grid
|
||||
float coord_y(std::size_t iy) const { return m_coordy[iy]; }
|
||||
/// @brief check whether a grid overlaps with a rectangle.
|
||||
/// Touching is not considered as overlap.
|
||||
bool overlap(std::size_t ix, std::size_t iy, float xl, float yl, float xh, float yh) const {
|
||||
float gxl = m_coordx[ix];
|
||||
float gxh = (ix + 1 == dim_x()) ? std::numeric_limits<float>::max() : m_coordx[ix + 1];
|
||||
float gyl = m_coordy[iy];
|
||||
float gyh = (iy + 1 == dim_y()) ? std::numeric_limits<float>::max() : m_coordy[iy + 1];
|
||||
|
||||
return std::max(gxl, xl) < std::min(gxh, xh) && std::max(gyl, yl) < std::min(gyh, yh);
|
||||
}
|
||||
|
||||
protected:
|
||||
/// @brief build grids from rectangles and boundaries
|
||||
void build(const float* x,
|
||||
const float* y,
|
||||
const float* width,
|
||||
const float* height,
|
||||
std::size_t n,
|
||||
const float xl,
|
||||
const float yl,
|
||||
const float xh,
|
||||
const float yh,
|
||||
const float spacing_x,
|
||||
const float spacing_y) {
|
||||
// collect all scan lines
|
||||
m_coordx.reserve((n << 1) + 2);
|
||||
m_coordy.reserve((n << 1) + 2);
|
||||
m_coordx.push_back(xl);
|
||||
m_coordx.push_back(xh);
|
||||
m_coordy.push_back(yl);
|
||||
m_coordy.push_back(yh);
|
||||
for (std::size_t i = 0; i < n; ++i) {
|
||||
m_coordx.push_back(x[i]);
|
||||
m_coordx.push_back(x[i] + width[i]);
|
||||
m_coordy.push_back(y[i]);
|
||||
m_coordy.push_back(y[i] + height[i]);
|
||||
}
|
||||
|
||||
// sort and make them unique
|
||||
std::sort(m_coordx.begin(), m_coordx.end());
|
||||
std::sort(m_coordy.begin(), m_coordy.end());
|
||||
m_coordx.resize(std::distance(m_coordx.begin(), std::unique(m_coordx.begin(), m_coordx.end())));
|
||||
m_coordy.resize(std::distance(m_coordy.begin(), std::unique(m_coordy.begin(), m_coordy.end())));
|
||||
|
||||
// in case some grids are too large
|
||||
// add more scan lines with step size spacing_x and spacing_y
|
||||
for (std::size_t i = 1, ie = m_coordx.size(); i < ie; ++i) {
|
||||
float gxl = m_coordx[i - 1];
|
||||
float gxh = m_coordx[i];
|
||||
|
||||
for (float xl = gxl + spacing_x; xl < gxh; xl += spacing_x) {
|
||||
m_coordx.push_back(xl);
|
||||
}
|
||||
}
|
||||
for (std::size_t i = 1, ie = m_coordy.size(); i < ie; ++i) {
|
||||
float gyl = m_coordy[i - 1];
|
||||
float gyh = m_coordy[i];
|
||||
|
||||
for (float yl = gyl + spacing_y; yl < gyh; yl += spacing_y) {
|
||||
m_coordy.push_back(yl);
|
||||
}
|
||||
}
|
||||
|
||||
// they should already be unique
|
||||
std::sort(m_coordx.begin(), m_coordx.end());
|
||||
std::sort(m_coordy.begin(), m_coordy.end());
|
||||
}
|
||||
|
||||
std::vector<float> m_coordx; ///< coordinates of grid lines in x direction
|
||||
std::vector<float> m_coordy; ///< coordinates of grid lines in y direction
|
||||
};
|
||||
|
||||
/// @brief A class models binary maps on Hannan grids.
|
||||
class HannanGridMap : public HannanGrids {
|
||||
public:
|
||||
HannanGridMap(const float* x,
|
||||
const float* y,
|
||||
const float* width,
|
||||
const float* height,
|
||||
std::size_t n,
|
||||
const float xl,
|
||||
const float yl,
|
||||
const float xh,
|
||||
const float yh,
|
||||
const float spacing_x,
|
||||
const float spacing_y)
|
||||
: HannanGrids(x, y, width, height, n, xl, yl, xh, yh, spacing_x, spacing_y) {
|
||||
// construct 2D binary map
|
||||
m_map.assign(this->dim_x() * this->dim_y(), 0);
|
||||
}
|
||||
|
||||
/// @brief set an entry in grid map
|
||||
void set(std::size_t ix, std::size_t iy, bool value) { m_map[ix * this->dim_y() + iy] = value; }
|
||||
|
||||
/// @brief get an entry in grid map
|
||||
bool at(std::size_t ix, std::size_t iy) const { return m_map[ix * this->dim_y() + iy]; }
|
||||
|
||||
/// @brief check whether a rectangle overlaps with any grid in the map
|
||||
bool overlap(float xl, float yl, float xh, float yh) const {
|
||||
std::size_t ixl = this->grid_x(xl);
|
||||
std::size_t ixh = this->grid_x(xh) + 1;
|
||||
std::size_t iyl = this->grid_y(yl);
|
||||
std::size_t iyh = this->grid_y(yh) + 1;
|
||||
|
||||
for (std::size_t ix = ixl; ix < ixh; ++ix) {
|
||||
for (std::size_t iy = iyl; iy < iyh; ++iy) {
|
||||
if (this->HannanGrids::overlap(ix, iy, xl, yl, xh, yh) && this->at(ix, iy)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// @brief add a rectangle to the grid map
|
||||
void add(float xl, float yl, float xh, float yh) {
|
||||
std::size_t ixl = this->grid_x(xl);
|
||||
std::size_t ixh = this->grid_x(xh) + 1;
|
||||
std::size_t iyl = this->grid_y(yl);
|
||||
std::size_t iyh = this->grid_y(yh) + 1;
|
||||
|
||||
for (std::size_t ix = ixl; ix < ixh; ++ix) {
|
||||
for (std::size_t iy = iyl; iy < iyh; ++iy) {
|
||||
if (this->HannanGrids::overlap(ix, iy, xl, yl, xh, yh)) {
|
||||
this->set(ix, iy, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
protected:
|
||||
std::vector<unsigned char> m_map; ///< 2D map indicating whether a grid is taken or not
|
||||
};
|
||||
|
||||
/// @brief A greedy macro legalization algorithm manipulating on Hannan grids.
|
||||
/// The procedure of the algorithm is as follows.
|
||||
/// For each macro:
|
||||
/// Perfrom spiral/diamond search to the locations;
|
||||
/// Find the first one with minimum displacement;
|
||||
/// Update the grid map;
|
||||
/// If the layout is very tight, it may not be able to find a solution.
|
||||
/// @return true if all macros legalized
|
||||
bool hannanLegalize(LegalizationData& db,
|
||||
std::vector<int>& macros,
|
||||
const std::vector<int>& fixed_macros,
|
||||
int max_iters) {
|
||||
logger.info("Legalize movable macros on Hannan grids");
|
||||
|
||||
// count number of failures to control the order
|
||||
std::vector<int> failure_counts(db.num_movable_nodes, 0);
|
||||
std::vector<float> x(db.num_movable_nodes, 0);
|
||||
std::vector<float> y(db.num_movable_nodes, 0);
|
||||
bool legal = true;
|
||||
|
||||
for (int iter = 0; iter < max_iters; ++iter) {
|
||||
logger.info("round %d", iter);
|
||||
// copy location to working array
|
||||
for (auto node_id : macros) {
|
||||
x[node_id] = db.x[node_id];
|
||||
y[node_id] = db.y[node_id];
|
||||
}
|
||||
// sort from left to right, large to small
|
||||
std::sort(macros.begin(), macros.end(), [&](int node_id1, int node_id2) {
|
||||
int factor1 = (1 + failure_counts[node_id1]);
|
||||
int factor2 = (1 + failure_counts[node_id2]);
|
||||
float a1 = db.node_size_x[node_id1] * db.node_size_y[node_id1]; // * factor1;
|
||||
float a2 = db.node_size_x[node_id2] * db.node_size_y[node_id2]; // * factor2;
|
||||
float x1 = x[node_id1] / factor1;
|
||||
float x2 = x[node_id2] / factor2;
|
||||
float y1 = y[node_id1] / factor1;
|
||||
float y2 = y[node_id2] / factor2;
|
||||
// return a1 > a2 || (a1 == a2 && (x1 < x2 || (x1 == x2 && (y1 < y2 || (y1 == y2 && node_id1 <
|
||||
// node_id2))))); return x1 < x2 || (x1 == x2 && (a1 > a2 || (a1 == a2 && (y1 < y2 || (y1 == y2 && node_id1
|
||||
// < node_id2)))));
|
||||
return x1 < x2 || (x1 == x2 && (y1 < y2 || (y1 == y2 && (a1 > a2 || (a1 == a2 && node_id1 < node_id2)))));
|
||||
});
|
||||
|
||||
float spacing_x = std::numeric_limits<float>::max();
|
||||
float spacing_y = std::numeric_limits<float>::max();
|
||||
for (auto node_id : macros) {
|
||||
spacing_x = std::min(spacing_x, db.node_size_x[node_id]);
|
||||
spacing_y = std::min(spacing_y, db.node_size_y[node_id]);
|
||||
}
|
||||
// make sure the grid is not too small
|
||||
spacing_x = std::max(spacing_x, (db.xh - db.xl) / db.num_bins_x);
|
||||
spacing_y = std::max(spacing_y, (db.yh - db.yl) / db.num_bins_y);
|
||||
logger.debug("maximum grid spacing %gx%g, equivalent to %dx%d bins",
|
||||
(double)spacing_x,
|
||||
(double)spacing_y,
|
||||
(int)((db.xh - db.xl) / spacing_x),
|
||||
(int)((db.yh - db.yl) / spacing_y));
|
||||
|
||||
// construct hannan grid map for fixed macros
|
||||
// collect fixed and dummy fixed nodes
|
||||
std::vector<float> vx;
|
||||
std::vector<float> vy;
|
||||
std::vector<float> node_size_x;
|
||||
std::vector<float> node_size_y;
|
||||
vx.reserve(db.num_nodes);
|
||||
vy.reserve(db.num_nodes);
|
||||
node_size_x.reserve(db.num_nodes);
|
||||
node_size_y.reserve(db.num_nodes);
|
||||
for (auto node_id : fixed_macros) {
|
||||
vx.push_back(db.x[node_id]);
|
||||
vy.push_back(db.y[node_id]);
|
||||
node_size_x.push_back(db.node_size_x[node_id]);
|
||||
node_size_y.push_back(db.node_size_y[node_id]);
|
||||
}
|
||||
for (auto node_id : macros) {
|
||||
vx.push_back(x[node_id]);
|
||||
vy.push_back(y[node_id]);
|
||||
node_size_x.push_back(db.node_size_x[node_id]);
|
||||
node_size_y.push_back(db.node_size_y[node_id]);
|
||||
}
|
||||
|
||||
HannanGridMap grid_map(vx.data(),
|
||||
vy.data(),
|
||||
node_size_x.data(),
|
||||
node_size_y.data(),
|
||||
vx.size(),
|
||||
db.xl,
|
||||
db.yl,
|
||||
db.xh,
|
||||
db.yh,
|
||||
spacing_x,
|
||||
spacing_y);
|
||||
|
||||
// the right and top boundary should always be occupied
|
||||
for (std::size_t ix = 0; ix < grid_map.dim_x(); ++ix) {
|
||||
grid_map.set(ix, grid_map.dim_y() - 1, 1);
|
||||
}
|
||||
for (std::size_t iy = 0; iy < grid_map.dim_y(); ++iy) {
|
||||
grid_map.set(grid_map.dim_x() - 1, iy, 1);
|
||||
}
|
||||
// set fixed nodes to occupy the grid map
|
||||
for (auto node_id : fixed_macros) {
|
||||
float xl = db.init_x[node_id];
|
||||
float xh = xl + db.node_size_x[node_id];
|
||||
float yl = db.init_y[node_id];
|
||||
float yh = yl + db.node_size_y[node_id];
|
||||
std::size_t ixl = grid_map.grid_x(xl);
|
||||
std::size_t ixh = grid_map.grid_x(xh);
|
||||
std::size_t iyl = grid_map.grid_y(yl);
|
||||
std::size_t iyh = grid_map.grid_y(yh);
|
||||
|
||||
for (std::size_t ix = ixl; ix <= ixh; ++ix) {
|
||||
for (std::size_t iy = iyl; iy <= iyh; ++iy) {
|
||||
if (grid_map.HannanGrids::overlap(ix, iy, xl, yl, xh, yh)) {
|
||||
grid_map.set(ix, iy, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
auto search_grids = diamond_search_sequence(grid_map.dim_y(), grid_map.dim_x());
|
||||
logger.debug("Construct %lux%lu Hannan grids, diamond search sequence %lu",
|
||||
grid_map.dim_x(),
|
||||
grid_map.dim_y(),
|
||||
search_grids.size());
|
||||
|
||||
legal = true;
|
||||
for (auto node_id : macros) {
|
||||
float node_x = x[node_id];
|
||||
float node_y = y[node_id];
|
||||
float width = db.node_size_x[node_id];
|
||||
float height = db.node_size_y[node_id];
|
||||
std::size_t init_ix = grid_map.grid_x(node_x);
|
||||
std::size_t init_iy = grid_map.grid_y(node_y);
|
||||
|
||||
bool found = false;
|
||||
for (auto grid_offset : search_grids) {
|
||||
std::size_t ix = init_ix + grid_offset.ic;
|
||||
std::size_t iy = init_iy + grid_offset.ir;
|
||||
|
||||
// valid grid
|
||||
if (ix < grid_map.dim_x() && iy < grid_map.dim_y()) {
|
||||
float xl = grid_map.coord_x(ix);
|
||||
float yl = grid_map.coord_y(iy);
|
||||
if (grid_offset.ic == 0 && grid_offset.ir == 0) {
|
||||
assert_msg(xl == node_x, "%g != %g", xl, node_x);
|
||||
assert_msg(yl == node_y, "%g != %g", yl, node_y);
|
||||
}
|
||||
|
||||
// make sure the coordinates are aligned to row and site
|
||||
float aligned_xl = db.align2site(xl, width);
|
||||
float aligned_yl = db.align2row(yl, height);
|
||||
if (aligned_xl < xl) {
|
||||
xl = aligned_xl + db.site_width;
|
||||
}
|
||||
if (aligned_yl < yl) {
|
||||
yl = aligned_yl + db.row_height;
|
||||
}
|
||||
float xh = xl + width;
|
||||
float yh = yl + height;
|
||||
|
||||
if (!grid_map.overlap(xl, yl, xh, yh)) {
|
||||
x[node_id] = xl;
|
||||
y[node_id] = yl;
|
||||
grid_map.add(xl, yl, xh, yh);
|
||||
found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!found) {
|
||||
logger.error("failed to find legal position for macro %d (%g, %g, %g, %g)",
|
||||
node_id,
|
||||
node_x,
|
||||
node_y,
|
||||
node_x + width,
|
||||
node_y + height);
|
||||
failure_counts[node_id] += 1;
|
||||
legal = false;
|
||||
}
|
||||
}
|
||||
if (legal) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// copy solutions back
|
||||
for (auto node_id : macros) {
|
||||
db.x[node_id] = x[node_id];
|
||||
db.y[node_id] = y[node_id];
|
||||
}
|
||||
|
||||
return legal;
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
165
cpp_to_py/gpudp/lg/legalization_db.h
Normal file
165
cpp_to_py/gpudp/lg/legalization_db.h
Normal file
@ -0,0 +1,165 @@
|
||||
#pragma once
|
||||
|
||||
#include "common/common.h"
|
||||
#include "common/db/Database.h"
|
||||
#include "gpudp/db/dp_torch.h"
|
||||
|
||||
namespace dp {
|
||||
|
||||
inline int floorDiv(float a, float b, float rtol = 1e-4) { return std::floor((a + rtol * b) / b); }
|
||||
|
||||
inline int ceilDiv(float a, float b, float rtol = 1e-4) { return std::ceil((a - rtol * b) / b); }
|
||||
|
||||
inline int roundDiv(float a, float b) { return std::round(a / b); }
|
||||
|
||||
template <typename T>
|
||||
struct Space {
|
||||
T xl;
|
||||
T xh;
|
||||
};
|
||||
|
||||
struct RowMapIndex {
|
||||
int row_id;
|
||||
int sub_id;
|
||||
};
|
||||
|
||||
struct BinMapIndex {
|
||||
int bin_id;
|
||||
int sub_id;
|
||||
};
|
||||
|
||||
struct Box {
|
||||
float xl;
|
||||
float yl;
|
||||
float xh;
|
||||
float yh;
|
||||
Box() {
|
||||
xl = std::numeric_limits<float>::max();
|
||||
yl = std::numeric_limits<float>::max();
|
||||
xh = std::numeric_limits<float>::lowest();
|
||||
yh = std::numeric_limits<float>::lowest();
|
||||
}
|
||||
Box(float xxl, float yyl, float xxh, float yyh) : xl(xxl), yl(yyl), xh(xxh), yh(yyh) {}
|
||||
|
||||
float center_x() const { return (xl + xh) / 2; }
|
||||
float center_y() const { return (yl + yh) / 2; }
|
||||
float width() const { return (xh - xl); }
|
||||
float height() const { return (yh - yl); }
|
||||
float area() const { return (xh - xl) * (yh - yl); }
|
||||
};
|
||||
|
||||
class LegalizationData {
|
||||
public:
|
||||
LegalizationData() {}
|
||||
LegalizationData(DPTorchRawDB& at_db)
|
||||
: x(at_db.x.data_ptr<float>()),
|
||||
y(at_db.y.data_ptr<float>()),
|
||||
init_x(at_db.init_x.data_ptr<float>()),
|
||||
init_y(at_db.init_y.data_ptr<float>()),
|
||||
node_size_x(at_db.node_size_x.data_ptr<float>()),
|
||||
node_size_y(at_db.node_size_y.data_ptr<float>()),
|
||||
pin_offset_x(at_db.pin_offset_x.data_ptr<float>()),
|
||||
pin_offset_y(at_db.pin_offset_y.data_ptr<float>()),
|
||||
flat_node2pin_start_map(at_db.flat_node2pin_start_map.data_ptr<int>()),
|
||||
flat_node2pin_map(at_db.flat_node2pin_map.data_ptr<int>()),
|
||||
pin2node_map(at_db.pin2node_map.data_ptr<int>()),
|
||||
flat_net2pin_start_map(at_db.flat_net2pin_start_map.data_ptr<int>()),
|
||||
flat_net2pin_map(at_db.flat_net2pin_map.data_ptr<int>()),
|
||||
pin2net_map(at_db.pin2net_map.data_ptr<int>()),
|
||||
flat_region_boxes_start(at_db.flat_region_boxes_start.data_ptr<int>()),
|
||||
flat_region_boxes(at_db.flat_region_boxes.data_ptr<float>()),
|
||||
node2fence_region_map(at_db.node2fence_region_map.data_ptr<int>()),
|
||||
net_mask(at_db.net_mask.data_ptr<bool>()),
|
||||
node_weight(at_db.node_weight.data_ptr<float>()),
|
||||
xl(at_db.xl),
|
||||
xh(at_db.xh),
|
||||
yl(at_db.yl),
|
||||
yh(at_db.yh),
|
||||
row_height(at_db.row_height),
|
||||
site_width(at_db.site_width),
|
||||
num_sites_x(at_db.num_sites_x),
|
||||
num_sites_y(at_db.num_sites_y),
|
||||
num_threads(at_db.num_threads),
|
||||
num_nodes(at_db.num_nodes),
|
||||
num_movable_nodes(at_db.num_movable_nodes),
|
||||
num_nets(at_db.num_nets),
|
||||
num_pins(at_db.num_pins),
|
||||
num_regions(at_db.num_regions) {}
|
||||
|
||||
public:
|
||||
float* x; // new pos x, need to be checked their legality
|
||||
float* y; // new pos y, need to be checked their legality
|
||||
const float* init_x; // original pos x
|
||||
const float* init_y; // original pos y
|
||||
const float* node_size_x;
|
||||
const float* node_size_y;
|
||||
|
||||
const float* pin_offset_x;
|
||||
const float* pin_offset_y;
|
||||
|
||||
const int* flat_node2pin_start_map;
|
||||
const int* flat_node2pin_map;
|
||||
const int* pin2node_map;
|
||||
|
||||
const int* flat_net2pin_start_map;
|
||||
const int* flat_net2pin_map;
|
||||
const int* pin2net_map;
|
||||
|
||||
const int* flat_region_boxes_start;
|
||||
const float* flat_region_boxes;
|
||||
const int* node2fence_region_map;
|
||||
|
||||
const bool* net_mask;
|
||||
const float* node_weight;
|
||||
|
||||
/* chip info */
|
||||
float xl;
|
||||
float yl;
|
||||
float xh;
|
||||
float yh;
|
||||
|
||||
/* row info */
|
||||
int num_sites_x;
|
||||
int num_sites_y;
|
||||
float row_height;
|
||||
float site_width;
|
||||
|
||||
int num_nets;
|
||||
int num_movable_nodes;
|
||||
int num_nodes;
|
||||
int num_pins;
|
||||
int num_regions;
|
||||
|
||||
int num_threads;
|
||||
|
||||
int num_bins_x;
|
||||
int num_bins_y;
|
||||
float bin_size_x;
|
||||
float bin_size_y;
|
||||
|
||||
public:
|
||||
void set_num_bins(int num_bins_x_, int num_bins_y_) {
|
||||
num_bins_x = num_bins_x_;
|
||||
num_bins_y = num_bins_y_;
|
||||
bin_size_x = (xh - xl) / num_bins_x_;
|
||||
bin_size_y = (yh - yl) / num_bins_y_;
|
||||
}
|
||||
inline bool is_dummy_fixed(int node_id) const {
|
||||
// DUMMY_FIXED_NUM_ROWS == 2
|
||||
return (node_id < num_movable_nodes && node_size_y[node_id] > (row_height * 2));
|
||||
}
|
||||
|
||||
inline float align2row(float y, float height) const {
|
||||
float yy = std::max(std::min(y, yh - height), yl);
|
||||
yy = floorDiv(yy - yl, row_height) * row_height + yl;
|
||||
return yy;
|
||||
}
|
||||
|
||||
inline float align2site(float x, float width) const {
|
||||
float xx = std::max(std::min(x, xh - width), xl);
|
||||
xx = floorDiv(xx - xl, site_width) * site_width + xl;
|
||||
return xx;
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace dp
|
||||
377
cpp_to_py/gpudp/lg/macro_legalize.cpp
Normal file
377
cpp_to_py/gpudp/lg/macro_legalize.cpp
Normal file
@ -0,0 +1,377 @@
|
||||
#include "gpudp/lg/hannan_legalize.h"
|
||||
#include "gpudp/lg/legalization_db.h"
|
||||
|
||||
namespace dp {
|
||||
/// @brief The macro legalization follows the way of floorplanning,
|
||||
/// because macros have quite different sizes.
|
||||
|
||||
bool check_macro_legality(LegalizationData& db, const std::vector<int>& macros, bool fast_check) {
|
||||
// check legality between movable and fixed macros
|
||||
// for debug only, so it is slow
|
||||
auto checkOverlap2Nodes = [&](int i,
|
||||
int node_id1,
|
||||
float xl1,
|
||||
float yl1,
|
||||
float width1,
|
||||
float height1,
|
||||
int j,
|
||||
int node_id2,
|
||||
float xl2,
|
||||
float yl2,
|
||||
float width2,
|
||||
float height2) {
|
||||
float xh1 = xl1 + width1;
|
||||
float yh1 = yl1 + height1;
|
||||
float xh2 = xl2 + width2;
|
||||
float yh2 = yl2 + height2;
|
||||
if (std::min(xh1, xh2) > std::max(xl1, xl2) && std::min(yh1, yh2) > std::max(yl1, yl2)) {
|
||||
logger.error(
|
||||
"macro %d (%g, %g, %g, %g) var %d overlaps with macro %d "
|
||||
"(%g, %g, %g, %g) var %d, fixed: %d",
|
||||
node_id1,
|
||||
xl1,
|
||||
yl1,
|
||||
xh1,
|
||||
yh1,
|
||||
i,
|
||||
node_id2,
|
||||
xl2,
|
||||
yl2,
|
||||
xh2,
|
||||
yh2,
|
||||
j,
|
||||
(int)(node_id2 >= db.num_movable_nodes));
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
};
|
||||
|
||||
bool legal = true;
|
||||
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
|
||||
int node_id1 = macros[i];
|
||||
float xl1 = db.x[node_id1];
|
||||
float yl1 = db.y[node_id1];
|
||||
float width1 = db.node_size_x[node_id1];
|
||||
float height1 = db.node_size_y[node_id1];
|
||||
// constraints with other macros
|
||||
for (unsigned int j = i + 1; j < ie; ++j) {
|
||||
int node_id2 = macros[j];
|
||||
float xl2 = db.x[node_id2];
|
||||
float yl2 = db.y[node_id2];
|
||||
float width2 = db.node_size_x[node_id2];
|
||||
float height2 = db.node_size_y[node_id2];
|
||||
|
||||
bool overlap =
|
||||
checkOverlap2Nodes(i, node_id1, xl1, yl1, width1, height1, j, node_id2, xl2, yl2, width2, height2);
|
||||
if (overlap) {
|
||||
legal = false;
|
||||
if (fast_check) {
|
||||
return legal;
|
||||
}
|
||||
}
|
||||
}
|
||||
// constraints with fixed macros
|
||||
// when considering fixed macros, there is no guarantee to find legal
|
||||
// solution with current ad-hoc constraint graphs
|
||||
for (int j = db.num_movable_nodes; j < db.num_nodes; ++j) {
|
||||
int node_id2 = j;
|
||||
float xl2 = db.init_x[node_id2];
|
||||
float yl2 = db.init_y[node_id2];
|
||||
float width2 = db.node_size_x[node_id2];
|
||||
float height2 = db.node_size_y[node_id2];
|
||||
|
||||
bool overlap =
|
||||
checkOverlap2Nodes(i, node_id1, xl1, yl1, width1, height1, j, node_id2, xl2, yl2, width2, height2);
|
||||
if (overlap) {
|
||||
legal = false;
|
||||
if (fast_check) {
|
||||
return legal;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (legal) {
|
||||
logger.debug("Macro legality check PASSED");
|
||||
} else {
|
||||
logger.error("Macro legality check FAILED");
|
||||
}
|
||||
|
||||
return legal;
|
||||
}
|
||||
|
||||
struct MacroLegalizeStats {
|
||||
float total_displace;
|
||||
float max_displace;
|
||||
float total_weighted_displace; ///< displacement weighted by macro area ratio to
|
||||
///< average macro area
|
||||
float max_weighted_displace;
|
||||
// float average_macro_area;
|
||||
};
|
||||
|
||||
MacroLegalizeStats compute_displace(const LegalizationData& db, const std::vector<int>& macros) {
|
||||
MacroLegalizeStats stats;
|
||||
stats.total_displace = 0;
|
||||
stats.max_displace = 0;
|
||||
stats.total_weighted_displace = 0;
|
||||
stats.max_weighted_displace = 0;
|
||||
// stats.average_macro_area = 0;
|
||||
|
||||
// for (auto node_id : macros)
|
||||
//{
|
||||
// stats.average_macro_area += db.node_size_x[node_id] *
|
||||
// db.node_size_y[node_id];
|
||||
//}
|
||||
// stats.average_macro_area /= macros.size();
|
||||
|
||||
for (auto node_id : macros) {
|
||||
float displace = std::abs(db.init_x[node_id] - db.x[node_id]) + std::abs(db.init_y[node_id] - db.y[node_id]);
|
||||
stats.total_displace += displace;
|
||||
stats.max_displace = std::max(stats.max_displace, displace);
|
||||
|
||||
displace *= db.node_weight[node_id];
|
||||
stats.total_weighted_displace += displace;
|
||||
stats.max_weighted_displace = std::max(stats.max_weighted_displace, displace);
|
||||
}
|
||||
return stats;
|
||||
}
|
||||
|
||||
/// @brief Rough legalize some special macros
|
||||
/// 1. macros that form small clusters overlapping with each other
|
||||
/// 2. macros blocked by big ones
|
||||
/// All the other macros are regarded as fixed.
|
||||
/// @param small_clusters_flag controls whether to perform the legalization for
|
||||
/// 1
|
||||
/// @param blocked_macros_flag controls whether to perform the legalization for
|
||||
/// 2
|
||||
bool roughLegalize(LegalizationData& db,
|
||||
const std::vector<int>& macros,
|
||||
const std::vector<int>& fixed_macros,
|
||||
bool small_clusters_flag,
|
||||
bool blocked_macros_flag) {
|
||||
std::vector<unsigned char> markers(db.num_nodes, false);
|
||||
std::vector<int> macros_for_rough_legalize;
|
||||
std::vector<int> fixed_macros_for_rough_legalize;
|
||||
|
||||
// collect small clusters
|
||||
if (small_clusters_flag) {
|
||||
std::vector<std::vector<int> > clusters(macros.size());
|
||||
float cluster_area_ratio = 2;
|
||||
float cluster_overlap_ratio = 0.5;
|
||||
unsigned int cluster_macro_numbers_threshold = 2;
|
||||
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
|
||||
int node_id1 = macros[i];
|
||||
Box box1(db.x[node_id1],
|
||||
db.y[node_id1],
|
||||
db.x[node_id1] + db.node_size_x[node_id1],
|
||||
db.y[node_id1] + db.node_size_y[node_id1]);
|
||||
float a1 = box1.area();
|
||||
clusters.at(i).push_back(node_id1);
|
||||
for (unsigned int j = i + 1; j < ie; ++j) {
|
||||
int node_id2 = macros[j];
|
||||
Box box2(db.x[node_id2],
|
||||
db.y[node_id2],
|
||||
db.x[node_id2] + db.node_size_x[node_id2],
|
||||
db.y[node_id2] + db.node_size_y[node_id2]);
|
||||
float a2 = box2.area();
|
||||
|
||||
if (a1 >= a2 / cluster_area_ratio && a1 <= a2 * cluster_area_ratio) {
|
||||
float overlap = std::max((float)0, std::min(box1.xh, box2.xh) - std::max(box1.xl, box2.xl)) *
|
||||
std::max((float)0, std::min(box1.yh, box2.yh) - std::max(box1.yl, box2.yl));
|
||||
if (overlap >= std::min(a1, a2) * cluster_overlap_ratio) {
|
||||
clusters.at(i).push_back(node_id2);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
|
||||
if (clusters.at(i).size() >= cluster_macro_numbers_threshold) {
|
||||
markers.at(macros.at(i)) = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
// collect small macros blocked by large ones
|
||||
// If a small macro is blocked by two big macros, it is easier to move the
|
||||
// small one around. We detect such blocks by checking whether the macro is
|
||||
// blocked from left, right, bottom, top 4 directions. Any macro with (left,
|
||||
// right) or (bottom, top) blocked will be collected.
|
||||
if (blocked_macros_flag) {
|
||||
float blocked_macros_area_ratio = 10; // the area ratio of macros to be regarded as large
|
||||
float blocked_macros_direct_threshold = 0.9; // determine the direction blocked
|
||||
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
|
||||
int node_id1 = macros[i];
|
||||
if (!markers[node_id1]) {
|
||||
Box box1(db.x[node_id1],
|
||||
db.y[node_id1],
|
||||
db.x[node_id1] + db.node_size_x[node_id1],
|
||||
db.y[node_id1] + db.node_size_y[node_id1]);
|
||||
float a1 = box1.area();
|
||||
std::array<unsigned char, 4> intersect_directs; // from L, R, B, float
|
||||
// direction, the box
|
||||
// is overlapped
|
||||
intersect_directs.fill(0);
|
||||
for (unsigned int j = 0; j < ie; ++j) {
|
||||
int node_id2 = macros[j];
|
||||
if (i != j && !markers[node_id2]) {
|
||||
Box box2(db.x[node_id2],
|
||||
db.y[node_id2],
|
||||
db.x[node_id2] + db.node_size_x[node_id2],
|
||||
db.y[node_id2] + db.node_size_y[node_id2]);
|
||||
float a2 = box2.area();
|
||||
|
||||
if (a1 * blocked_macros_area_ratio < a2) {
|
||||
Box intersect_box(std::max(box1.xl, box2.xl),
|
||||
std::max(box1.yl, box2.yl),
|
||||
std::min(box1.xh, box2.xh),
|
||||
std::min(box1.yh, box2.yh));
|
||||
if (intersect_box.xl < intersect_box.xh && intersect_box.yl < intersect_box.yh) {
|
||||
if (intersect_box.height() > box1.height() * blocked_macros_direct_threshold) {
|
||||
if (box2.xl <= box1.xl) {
|
||||
intersect_directs[0] = 1; // xl
|
||||
}
|
||||
if (box2.xh >= box1.xh) {
|
||||
intersect_directs[1] = 1; // xh
|
||||
}
|
||||
}
|
||||
if (intersect_box.width() > box1.width() * blocked_macros_direct_threshold) {
|
||||
if (box2.yl <= box1.yl) {
|
||||
intersect_directs[2] = 1; // yl
|
||||
}
|
||||
if (box2.yh >= box1.yh) {
|
||||
intersect_directs[3] = 1; // yh
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if ((intersect_directs[0] && intersect_directs[1]) ||
|
||||
(intersect_directs[2] && intersect_directs[3])) {
|
||||
markers[node_id1] = true;
|
||||
logger.debug("collect %d", node_id1);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fixed_macros_for_rough_legalize = fixed_macros;
|
||||
for (auto node_id : macros) {
|
||||
if (markers[node_id]) {
|
||||
macros_for_rough_legalize.push_back(node_id);
|
||||
} else {
|
||||
fixed_macros_for_rough_legalize.push_back(node_id);
|
||||
}
|
||||
}
|
||||
|
||||
logger.info("Rough legalize small clusters with %lu macros", macros_for_rough_legalize.size());
|
||||
return hannanLegalize(db, macros_for_rough_legalize, fixed_macros_for_rough_legalize, 1);
|
||||
}
|
||||
|
||||
bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
|
||||
LegalizationData db(at_db);
|
||||
db.set_num_bins(num_bins_x, num_bins_y);
|
||||
// collect macros
|
||||
std::vector<int> macros;
|
||||
for (int i = 0; i < db.num_movable_nodes; ++i) {
|
||||
if (db.is_dummy_fixed(i)) {
|
||||
// in some extreme case, some macros with 0 area should be ignored
|
||||
float area = db.node_size_x[i] * db.node_size_y[i];
|
||||
if (area > 0) {
|
||||
macros.push_back(i);
|
||||
}
|
||||
}
|
||||
}
|
||||
logger.info("Macro legalization: regard %lu cells as dummy fixed (movable macros)", macros.size());
|
||||
|
||||
// in case there is no movable macros
|
||||
if (macros.empty()) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// fixed macros
|
||||
std::vector<int> fixed_macros;
|
||||
fixed_macros.reserve(db.num_nodes - db.num_movable_nodes);
|
||||
for (int i = db.num_movable_nodes; i < db.num_nodes; ++i) {
|
||||
// in some extreme case, some fixed macros with 0 area should be ignored
|
||||
float area = db.node_size_x[i] * db.node_size_y[i];
|
||||
if (area > 0) {
|
||||
fixed_macros.push_back(i);
|
||||
}
|
||||
}
|
||||
|
||||
// store the best legalization solution found
|
||||
std::vector<float> best_x(macros.size());
|
||||
std::vector<float> best_y(macros.size());
|
||||
MacroLegalizeStats best_displace;
|
||||
best_displace.total_displace = std::numeric_limits<float>::max();
|
||||
best_displace.max_displace = std::numeric_limits<float>::max();
|
||||
best_displace.total_weighted_displace = std::numeric_limits<float>::max();
|
||||
best_displace.max_weighted_displace = std::numeric_limits<float>::max();
|
||||
|
||||
// update current best solution
|
||||
auto update_best = [&](bool legal, const MacroLegalizeStats& displace) {
|
||||
if (legal && displace.total_displace < best_displace.total_displace) {
|
||||
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
|
||||
int macro_id = macros[i];
|
||||
best_x[i] = db.x[macro_id];
|
||||
best_y[i] = db.y[macro_id];
|
||||
}
|
||||
best_displace = displace;
|
||||
}
|
||||
};
|
||||
|
||||
// first round rough legalization with Hannan grid for clusters
|
||||
bool small_clusters_flag = true;
|
||||
bool blocked_macros_flag = false;
|
||||
roughLegalize(db, macros, fixed_macros, small_clusters_flag, blocked_macros_flag);
|
||||
auto displace = compute_displace(db, macros);
|
||||
logger.info("Macro displacement total %g, max %g, weighted total %g, max %g",
|
||||
displace.total_displace,
|
||||
displace.max_displace,
|
||||
displace.total_weighted_displace,
|
||||
displace.max_weighted_displace);
|
||||
bool legal = check_macro_legality(db, macros, true);
|
||||
|
||||
// try Hannan grid legalization if still not legal
|
||||
if (!legal) {
|
||||
legal = hannanLegalize(db, macros, fixed_macros, 10);
|
||||
auto displace = compute_displace(db, macros);
|
||||
logger.info("Macro displacement total %g, max %g, weighted total %g, max %g",
|
||||
displace.total_displace,
|
||||
displace.max_displace,
|
||||
displace.total_weighted_displace,
|
||||
displace.max_weighted_displace);
|
||||
legal = check_macro_legality(db, macros, true);
|
||||
update_best(legal, displace);
|
||||
|
||||
// apply best solution
|
||||
if (best_displace.total_displace < std::numeric_limits<float>::max()) {
|
||||
logger.info(
|
||||
"use best macro displacement total %g, max %g, weighted "
|
||||
"total %g, max %g",
|
||||
best_displace.total_displace,
|
||||
best_displace.max_displace,
|
||||
best_displace.total_weighted_displace,
|
||||
best_displace.max_weighted_displace);
|
||||
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
|
||||
int macro_id = macros[i];
|
||||
db.x[macro_id] = best_x[i];
|
||||
db.y[macro_id] = best_y[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
logger.info("Align macros to site and rows");
|
||||
// align the lower left corner to row and site
|
||||
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
|
||||
int node_id = macros[i];
|
||||
db.x[node_id] = db.align2site(db.x[node_id], db.node_size_x[node_id]);
|
||||
db.y[node_id] = db.align2row(db.y[node_id], db.node_size_y[node_id]);
|
||||
}
|
||||
|
||||
legal = check_macro_legality(db, macros, false);
|
||||
|
||||
return legal;
|
||||
}
|
||||
|
||||
} // namespace dp
|
||||
69
cpp_to_py/gpugr/CMakeLists.txt
Normal file
69
cpp_to_py/gpugr/CMakeLists.txt
Normal file
@ -0,0 +1,69 @@
|
||||
# Files
|
||||
file(GLOB_RECURSE SRC_FILES_GGR ${CMAKE_CURRENT_SOURCE_DIR}/db/*.cpp
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/gr/*.cpp)
|
||||
file(GLOB_RECURSE SRC_FILES_GGR_CUDA ${CMAKE_CURRENT_SOURCE_DIR}/*.cu)
|
||||
|
||||
# CUDA GGR Kernel
|
||||
cuda_add_library(ggr_cuda_tmp STATIC ${SRC_FILES_GGR_CUDA})
|
||||
|
||||
set_target_properties(ggr_cuda_tmp PROPERTIES CUDA_RESOLVE_DEVICE_SYMBOLS ON)
|
||||
set_target_properties(ggr_cuda_tmp PROPERTIES CUDA_SEPARABLE_COMPILATION ON)
|
||||
set_target_properties(ggr_cuda_tmp PROPERTIES POSITION_INDEPENDENT_CODE ON)
|
||||
|
||||
target_include_directories(ggr_cuda_tmp PRIVATE ${PROJECT_SOURCE_DIR}/cpp_to_py ${TORCH_INCLUDE_DIRS} ${FLUTE_INCLUDE_DIR})
|
||||
target_link_libraries(ggr_cuda_tmp torch ${TORCH_PYTHON_LIBRARY} xplace_common flute)
|
||||
# target_compile_options(ggr_cuda_tmp PRIVATE "$<$<COMPILE_LANGUAGE:CUDA>:SHELL:-use_fast_math>") # not work...
|
||||
|
||||
|
||||
# CPU GGR object
|
||||
add_library(ggr SHARED ${CMAKE_CURRENT_SOURCE_DIR}/../io_parser/gp/GPDatabase.cpp
|
||||
${SRC_FILES_GGR}
|
||||
${SRC_FILES_GGR_CUDA})
|
||||
|
||||
target_include_directories(ggr PRIVATE ${PROJECT_SOURCE_DIR}/cpp_to_py ${TORCH_INCLUDE_DIRS} ${FLUTE_INCLUDE_DIR})
|
||||
target_link_libraries(ggr PRIVATE torch ${TORCH_PYTHON_LIBRARY} xplace_common flute ggr_cuda_tmp pthread)
|
||||
target_compile_options(ggr PRIVATE -fPIC)
|
||||
|
||||
install(TARGETS ggr DESTINATION ${XPLACE_LIB_DIR})
|
||||
|
||||
# For Pybind
|
||||
add_pytorch_extension(gpugr PyBindCppMain.cpp
|
||||
EXTRA_INCLUDE_DIRS ${PROJECT_SOURCE_DIR}/cpp_to_py ${FLUTE_INCLUDE_DIR}
|
||||
EXTRA_LINK_LIBRARIES xplace_common flute io_parser ggr)
|
||||
|
||||
install(TARGETS gpugr DESTINATION ${XPLACE_LIB_DIR})
|
||||
|
||||
################ For debug only ##################
|
||||
|
||||
# set(CMAKE_BUILD_TYPE Release)
|
||||
# set(CMAKE_CXX_STANDARD 17)
|
||||
# set(CMAKE_CUDA17_EXTENSION_COMPILE_OPTION "-std=c++17")
|
||||
# set(CMAKE_CUDA_FLAGS "${CMAKE_CUDA_FLAGS} -arch sm_86 --extended-lambda --use_fast_math ")
|
||||
# project(ggr LANGUAGES C CXX CUDA)
|
||||
|
||||
# file(GLOB_RECURSE SRC_FILES_GR2 ${CMAKE_CURRENT_SOURCE_DIR}/*.cpp)
|
||||
# file(GLOB_RECURSE SRC_FILES_GR2_CUDA ${CMAKE_CURRENT_SOURCE_DIR}/*.cu)
|
||||
|
||||
# add_executable(gpugr_cpp PyBindCppMain.cpp
|
||||
# ${CMAKE_CURRENT_SOURCE_DIR}/../io_parser/gp/GPDatabase.cpp
|
||||
# ${SRC_FILES_GR2}
|
||||
# ${SRC_FILES_GR2_CUDA})
|
||||
|
||||
# set_target_properties(gpugr_cpp PROPERTIES CUDA_RESOLVE_DEVICE_SYMBOLS ON)
|
||||
# set_target_properties(gpugr_cpp PROPERTIES CUDA_SEPARABLE_COMPILATION ON)
|
||||
# set_target_properties(gpugr_cpp PROPERTIES LINK_FLAGS "-Wl,--whole-archive -rdynamic -lpthread -Wl,--no-whole-archive")
|
||||
|
||||
# target_include_directories(
|
||||
# gpugr_cpp PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/.. ${PROJECT_SOURCE_DIR}/cpp_to_py ${FLUTE_INCLUDE_DIR} ${TORCH_INCLUDE_DIRS})
|
||||
# target_link_libraries(
|
||||
# gpugr_cpp PRIVATE torch ${TORCH_PYTHON_LIBRARY} flute xplace_common)
|
||||
|
||||
# target_compile_options(gpugr_cpp PRIVATE $<$<COMPILE_LANGUAGE:CUDA>:
|
||||
# -arch=sm_86
|
||||
# --use_fast_math
|
||||
# -std=c++17
|
||||
# >)
|
||||
|
||||
# install(TARGETS
|
||||
# gpugr_cpp
|
||||
# DESTINATION ${XPLACE_LIB_DIR})
|
||||
80
cpp_to_py/gpugr/PyBindCppMain.cpp
Normal file
80
cpp_to_py/gpugr/PyBindCppMain.cpp
Normal file
@ -0,0 +1,80 @@
|
||||
#include "common/common.h"
|
||||
#include "common/db/Database.h"
|
||||
#include "gpugr/db/GRDatabase.h"
|
||||
#include "gpugr/gr/RouteForce.h"
|
||||
#include "flute.h"
|
||||
|
||||
namespace Xplace {
|
||||
|
||||
bool loadGRParams(const pybind11::dict& kwargs) {
|
||||
gr::grSetting.reset();
|
||||
// ----- design related options -----
|
||||
|
||||
if (kwargs.contains("device_id")) {
|
||||
gr::grSetting.deviceId = kwargs["device_id"].cast<int>();
|
||||
}
|
||||
|
||||
if (kwargs.contains("route_xSize")) {
|
||||
gr::grSetting.routeXSize = kwargs["route_xSize"].cast<int>();
|
||||
}
|
||||
|
||||
if (kwargs.contains("route_ySize")) {
|
||||
gr::grSetting.routeYSize = kwargs["route_ySize"].cast<int>();
|
||||
}
|
||||
|
||||
if (kwargs.contains("csrn_scale")) {
|
||||
gr::grSetting.csrnScale = kwargs["csrn_scale"].cast<int>();
|
||||
}
|
||||
|
||||
if (kwargs.contains("rrrIters")) {
|
||||
gr::grSetting.rrrIters = kwargs["rrrIters"].cast<int>();
|
||||
}
|
||||
|
||||
if (kwargs.contains("route_guide")) {
|
||||
gr::grSetting.routeGuideFile = kwargs["route_guide"].cast<std::string>();
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
|
||||
pybind11::class_<gr::GRDatabase, std::shared_ptr<gr::GRDatabase>>(m, "GRDatabase")
|
||||
.def(pybind11::init<std::shared_ptr<db::Database>, std::shared_ptr<gp::GPDatabase>>())
|
||||
.def("report_gr_stat", &gr::GRDatabase::reportGRStat)
|
||||
.def("setup_capacity", &gr::GRDatabase::setupCapacity)
|
||||
.def("setup_wiredist", &gr::GRDatabase::setupWireDist)
|
||||
.def("setup_obs", &gr::GRDatabase::setupObs)
|
||||
.def("setup_grnets", &gr::GRDatabase::setupGrNets);
|
||||
pybind11::class_<gr::RouteForce, std::shared_ptr<gr::RouteForce>>(m, "RouteForce")
|
||||
.def(pybind11::init<std::shared_ptr<gr::GRDatabase>>())
|
||||
.def("run_ggr", &gr::RouteForce::run_ggr)
|
||||
.def("num_ovfl_nets", &gr::RouteForce::getNumOvflNets)
|
||||
.def("gcell_steps", &gr::RouteForce::getGcellStep)
|
||||
.def("microns", &gr::RouteForce::getMicrons)
|
||||
.def("layer_pitch", &gr::RouteForce::getLayerPitch)
|
||||
.def("layer_width", &gr::RouteForce::getLayerWidth)
|
||||
.def("dmd_map", &gr::RouteForce::getDemandMap, py::return_value_policy::move)
|
||||
.def("cap_map", &gr::RouteForce::getCapacityMap, py::return_value_policy::move)
|
||||
.def("route_grad", &gr::RouteForce::calcRouteGrad, py::return_value_policy::move)
|
||||
.def("filler_route_grad", &gr::RouteForce::calcFillerRouteGrad, py::return_value_policy::move)
|
||||
.def("pseudo_grad", &gr::RouteForce::calcPseudoPinGrad, py::return_value_policy::move)
|
||||
.def("inflate_ratio", &gr::RouteForce::calcNodeInflateRatio, py::return_value_policy::move)
|
||||
.def("inflate_pin_rel_cpos", &gr::RouteForce::calcInflatedPinRelCpos, py::return_value_policy::move);
|
||||
|
||||
m.def("create_grdatabase", [](std::shared_ptr<db::Database> rawdb, std::shared_ptr<gp::GPDatabase> gpdb) {
|
||||
logger.enable_logger();
|
||||
std::shared_ptr<gr::GRDatabase> grdb = std::make_shared<gr::GRDatabase>(rawdb, gpdb);
|
||||
logger.reset_logger();
|
||||
return grdb;
|
||||
});
|
||||
m.def("create_routeforce", [](std::shared_ptr<gr::GRDatabase> grdb) {
|
||||
logger.enable_logger();
|
||||
std::shared_ptr<gr::RouteForce> routeforce = std::make_shared<gr::RouteForce>(grdb);
|
||||
logger.reset_logger();
|
||||
return routeforce;
|
||||
});
|
||||
m.def("load_gr_params", &loadGRParams, "Parse input args to DB and return graph information");
|
||||
m.def("read_flute", &Flute::readLUT, "Read Flute LUT");
|
||||
}
|
||||
|
||||
} // namespace Xplace
|
||||
132
cpp_to_py/gpugr/README.md
Normal file
132
cpp_to_py/gpugr/README.md
Normal file
@ -0,0 +1,132 @@
|
||||
# GGR: Superfast Full-Scale GPU-Accelerated Global Routing
|
||||
GGR is a superfast full-scale GPU-accelerated global router developed by the research team supervised by Prof. Evangeline F. Y. Young and Prof. Martin D.F. Wong at The Chinese University of Hong Kong (CUHK). It includes an an efficient and high quality Z-shape pattern routing and a GPU-accelerated maze router GAMER.
|
||||
|
||||
More details are in the following paper:
|
||||
|
||||
Shiju Lin, Jinwei Liu, Evangeline F.Y. Young and Martin D.F. Wong. "[GAMER: GPU-Accelerated Maze Routing](https://ieeexplore.ieee.org/document/9799536)". In IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems, vol. 42, no. 2, pp. 583-593, Feb. 2023.
|
||||
|
||||
Shiju Lin and Martin D. F. Wong. "[Superfast Full-Scale CPU-Accelerated Global Routing](https://doi.org/10.1145/3508352.3549474)". In Proceedings of the 41st IEEE/ACM International Conference on Computer-Aided Design (ICCAD '22). Association for Computing Machinery, New York, NY, USA, Article 51, 1–8.
|
||||
|
||||
**GGR is integrated in Xplace now!**
|
||||
|
||||
## Notes and Limitations
|
||||
- We defaultly use `N=0` (without CUDA graph optimization). When `N >= 1`, unknown error would occur during terminating the Python program. To enable CUDA graph optimization, please set `use_tf = true` in `cpp_to_py/gpugr/gr/MazeRoute.cu` and re-compile the project.
|
||||
- GGR is **deterministic** when `N=0` but not test in `N >= 1`.
|
||||
- We currently only support LEF/DEF format.
|
||||
- The runtime of this version is a little bit slower than the version described in the GGR paper because we use a stronger but slower parser and make the algorithm deterministic yet robust while sacrificing the runtime.
|
||||
|
||||
## Parameters
|
||||
Please refer to `cpp_to_py/gpugr/PyBindCppMain.cpp`.
|
||||
|
||||
- `device_id`: GPU id.
|
||||
- `route_xSize` / `route_ySize`: given GR GridGraph size. If the GridGraph size is `0`, use the GridGraph definition in DEF file. If the GridGraph size is `0` and there is no definition in DEF, we defaultly set it as `512`.
|
||||
- `rrrIters`: the number of rip-up and re-route iterations (maze route). If `rrrIters = 0`, perform pattern route only.
|
||||
- `csrn_scale`: the size of coarsen grid in maze routing.
|
||||
- `route_guide`: the file name of output route guide.
|
||||
|
||||
## Citation
|
||||
If you find **GGR** useful in your research, please consider to cite:
|
||||
```bibtex
|
||||
@inproceedings{lin2022ggr,
|
||||
author = {Lin, Shiju and Wong, Martin D. F.},
|
||||
booktitle = {Proceedings of the 41st IEEE/ACM International Conference on Computer-Aided Design},
|
||||
title = {Superfast Full-Scale CPU-Accelerated Global Routing},
|
||||
year = {2022},
|
||||
}
|
||||
|
||||
@article{lin2023gamer,
|
||||
author={Lin, Shiju and Liu, Jinwei and Young, Evangeline F. Y. and Wong, Martin D. F.},
|
||||
journal={IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems},
|
||||
title={GAMER: GPU-Accelerated Maze Routing},
|
||||
year={2023},
|
||||
```
|
||||
|
||||
|
||||
## Example
|
||||
```python
|
||||
# main_test_gr.py
|
||||
import torch
|
||||
import time
|
||||
|
||||
from utils import IOParser
|
||||
from cpp_to_py import gpugr
|
||||
from src import Flute
|
||||
from src.core.route_force import calc_gr_wl_via, estimate_num_shorts
|
||||
|
||||
num_threads = 20
|
||||
gpu_id = 0
|
||||
Flute.register(num_threads)
|
||||
torch.cuda.synchronize("cuda:{}".format(gpu_id))
|
||||
|
||||
# 1) Benchmark Setting
|
||||
root = "your_path"
|
||||
design_name = "ispd19_test9"
|
||||
params = {
|
||||
"benchmark": "iccad2019",
|
||||
"lef": "%s/%s/%s.input.lef" % (root, design_name, design_name),
|
||||
"def": "%s/%s/%s.input.def" % (root, design_name, design_name),
|
||||
"design_name": design_name,
|
||||
}
|
||||
route_guide_file = "test.guide"
|
||||
|
||||
# 2) LEF/DEF Parser
|
||||
print("--- Start GR ---")
|
||||
start_gr_time = time.time()
|
||||
parser = IOParser()
|
||||
rawdb, gpdb = parser.read(
|
||||
params, verbose_log=True, lite_mode=True, random_place=False, num_threads=num_threads
|
||||
)
|
||||
|
||||
# 3) Construct Global Routing Database and Run GGR
|
||||
gpugr.load_gr_params(
|
||||
{
|
||||
"device_id": gpu_id,
|
||||
"route_xSize": 0,
|
||||
"route_ySize": 0,
|
||||
"rrrIters": 1,
|
||||
"route_guide": route_guide_file,
|
||||
}
|
||||
)
|
||||
grdb = gpugr.create_grdatabase(rawdb, gpdb)
|
||||
routeforce = gpugr.create_routeforce(grdb)
|
||||
routeforce.run_ggr()
|
||||
end_gr_time = time.time()
|
||||
print("--- End GR ---")
|
||||
|
||||
# 4) Report Global Routing Statistics
|
||||
skip_m1_route = True
|
||||
m1direction = gpdb.m1direction() # 0 for H, 1 for V, metal1's layer idx is 0
|
||||
hId = 1 if m1direction else 0
|
||||
vId = 0 if m1direction else 1
|
||||
if skip_m1_route:
|
||||
hId = hId + 2 if hId == 0 else hId
|
||||
vId = vId + 2 if vId == 0 else vId
|
||||
|
||||
dmd_map, wire_dmd_map, via_dmd_map = routeforce.dmd_map()
|
||||
cap_map: torch.Tensor = routeforce.cap_map()
|
||||
|
||||
cg_mapH = dmd_map[hId::2].sum(dim=0) / cap_map[hId::2].sum(dim=0)
|
||||
cg_mapV = dmd_map[vId::2].sum(dim=0) / cap_map[vId::2].sum(dim=0)
|
||||
cg_mapHV = torch.stack((cg_mapH, cg_mapV))
|
||||
cg_mapHV = torch.where(cg_mapHV > 1, cg_mapHV - 1, 0)
|
||||
|
||||
numOvflNets = routeforce.num_ovfl_nets()
|
||||
gr_wirelength, gr_numVias = calc_gr_wl_via(grdb, routeforce)
|
||||
gr_numShorts = estimate_num_shorts(routeforce, gpdb, cap_map, wire_dmd_map, via_dmd_map)
|
||||
|
||||
gr_time = end_gr_time - start_gr_time
|
||||
|
||||
print(
|
||||
"#OvflNets: %d, GR WL: %d, GR #Vias: %d, #EstShorts: %d | GR Time: %.4f"
|
||||
% (numOvflNets, gr_wirelength, gr_numVias, gr_numShorts, gr_time)
|
||||
)
|
||||
```
|
||||
|
||||
## Contact
|
||||
|
||||
[Shiju Lin](https://appsrv.cse.cuhk.edu.hk/~sjlin/) (sjlin@cse.cuhk.edu.hk) and [Lixin Liu](https://liulixinkerry.github.io/) (lxliu@cse.cuhk.edu.hk)
|
||||
|
||||
|
||||
## License
|
||||
|
||||
GGR is an open source project licensed under a BSD 3-Clause License that can be found in the [LICENSE](../../LICENSE) file.
|
||||
1021
cpp_to_py/gpugr/db/GRDatabase.cpp
Normal file
1021
cpp_to_py/gpugr/db/GRDatabase.cpp
Normal file
File diff suppressed because it is too large
Load Diff
106
cpp_to_py/gpugr/db/GRDatabase.h
Normal file
106
cpp_to_py/gpugr/db/GRDatabase.h
Normal file
@ -0,0 +1,106 @@
|
||||
#pragma once
|
||||
#include "GRSetting.h"
|
||||
#include "GrNet.h"
|
||||
#include "common/common.h"
|
||||
#include "common/db/Database.h"
|
||||
#include "io_parser/gp/GPDatabase.h"
|
||||
|
||||
namespace gr {
|
||||
|
||||
enum class AggrParaRunSpace { DEFAULT, LARGER_WIDTH, LARGER_LENGTH };
|
||||
|
||||
class RectOnLayer {
|
||||
public:
|
||||
int layer = -1;
|
||||
int lx, ly, hx, hy;
|
||||
|
||||
RectOnLayer() {}
|
||||
RectOnLayer(const int layer_, const int lx_, const int ly_, const int hx_, const int hy_)
|
||||
: layer(layer_), lx(lx_), ly(ly_), hx(hx_), hy(hy_) {}
|
||||
|
||||
int getDirRange(unsigned i) { return (i == 0) ? hx - lx : hy - ly; }
|
||||
};
|
||||
|
||||
// class GRNet {
|
||||
// std::vector<std::vector<std::tuple<int, int, int>>> global_pins;
|
||||
// std::vector<std::vector<RectOnLayer>> pins; // GCell locations
|
||||
// };
|
||||
|
||||
class GRDatabase {
|
||||
public:
|
||||
db::Database& rawdb;
|
||||
gp::GPDatabase& gpdb;
|
||||
|
||||
// GR Obstacle
|
||||
std::vector<RectOnLayer> fixObs;
|
||||
std::vector<RectOnLayer> movObs; // for movable nodes
|
||||
std::vector<std::vector<int>> tracks;
|
||||
|
||||
// Metal Layer
|
||||
std::vector<int> layerWidth;
|
||||
std::vector<int> layerPitch;
|
||||
std::vector<int> defaultSpacing;
|
||||
std::vector<int> maxEOLSpacingVec; // from all spacing types
|
||||
std::vector<int> maxEOLWidthVec; // from all spacing types
|
||||
|
||||
// Design
|
||||
int nLayers;
|
||||
int xSize;
|
||||
int ySize;
|
||||
int nMaxGrid;
|
||||
int gridGraphSize;
|
||||
int m1direction; // layer 0 direction, 'v': 1, 'h': 0
|
||||
double m2pitch;
|
||||
int microns;
|
||||
|
||||
int mainGcellStepX, mainGcellStepY;
|
||||
std::vector<std::vector<int>> gridlines;
|
||||
std::vector<std::vector<int>> gridCenters;
|
||||
|
||||
int ISPD19 = 0;
|
||||
int ISPD18 = 0;
|
||||
int METAL5 = 0;
|
||||
int csrnScale = 0;
|
||||
int cgxsize, cgysize;
|
||||
|
||||
// GR variables
|
||||
std::vector<float> capacity, wireDist; // init once
|
||||
std::vector<float> fixTmpUsage, fixTmpLength; // init once
|
||||
std::vector<float> movTmpUsage, movTmpLength; // update dynamically
|
||||
std::vector<float> fixedUsage, fixedLength; // variables for GR
|
||||
|
||||
std::vector<GrNet> grNets;
|
||||
std::vector<int> gpdbPinId2gbPinId;
|
||||
|
||||
GRDatabase(std::shared_ptr<db::Database> rawdb_, std::shared_ptr<gp::GPDatabase> gpdb_);
|
||||
~GRDatabase();
|
||||
|
||||
void setupCapacity();
|
||||
void setupCapacityBookshelf();
|
||||
void setupWireDist();
|
||||
|
||||
void addFixObs();
|
||||
void addMovObs();
|
||||
void updateUsageLength();
|
||||
void setupObs();
|
||||
|
||||
void setupGrNets();
|
||||
void resetGrNetsRoute();
|
||||
void resetGrNets() { grNets.clear(); };
|
||||
|
||||
std::pair<int, int> reportGRStat();
|
||||
void writeGuides(std::string outputFile);
|
||||
|
||||
int encodeId(int l, int x, int y);
|
||||
tuple<int, int, int, int> getOrientOffset(int orient, int lx, int ly, int hx, int hy);
|
||||
int getEOLSpace(int width, int l);
|
||||
int getParallelRunSpace(int l, int width, int length);
|
||||
utils::PointT<int> getObsMargin(RectOnLayer box, AggrParaRunSpace aggr);
|
||||
utils::IntervalT<int> rangeSearchTracks(const utils::IntervalT<int>& locRange, int layerIdx);
|
||||
void markObs(std::vector<RectOnLayer>& allObs, std::vector<float>& wireUsage, std::vector<float>& wireTotalLength);
|
||||
void markObsBookShelf(std::vector<RectOnLayer>& allObs,
|
||||
std::vector<float>& wireUsage,
|
||||
std::vector<float>& wireTotalLength);
|
||||
void addCellObs(std::vector<RectOnLayer>& allObs, db::Cell* cell);
|
||||
};
|
||||
} // namespace gr
|
||||
19
cpp_to_py/gpugr/db/GRSetting.cpp
Normal file
19
cpp_to_py/gpugr/db/GRSetting.cpp
Normal file
@ -0,0 +1,19 @@
|
||||
#include "GRSetting.h"
|
||||
|
||||
namespace gr {
|
||||
|
||||
void GRSetting::reset() {
|
||||
deviceId = 0;
|
||||
|
||||
routeXSize = 0;
|
||||
routeYSize = 0;
|
||||
csrnScale = 0;
|
||||
|
||||
rrrIters = 0;
|
||||
|
||||
routeGuideFile = "";
|
||||
}
|
||||
|
||||
GRSetting grSetting;
|
||||
|
||||
} // namespace gr
|
||||
25
cpp_to_py/gpugr/db/GRSetting.h
Normal file
25
cpp_to_py/gpugr/db/GRSetting.h
Normal file
@ -0,0 +1,25 @@
|
||||
#pragma once
|
||||
#include <string>
|
||||
|
||||
namespace gr {
|
||||
|
||||
class GRSetting {
|
||||
public:
|
||||
// 1. SystemSetting
|
||||
int deviceId = 0;
|
||||
|
||||
// 2. Gridgraph setting
|
||||
int routeXSize = 0;
|
||||
int routeYSize = 0;
|
||||
int csrnScale = 0;
|
||||
|
||||
// 3. The number of Rip-up and Reroute iterations (if 0, only PR is invoked)
|
||||
int rrrIters = 0;
|
||||
|
||||
std::string routeGuideFile = "";
|
||||
|
||||
void reset();
|
||||
};
|
||||
|
||||
extern GRSetting grSetting;
|
||||
} // namespace gr
|
||||
20
cpp_to_py/gpugr/db/GrNet.cpp
Normal file
20
cpp_to_py/gpugr/db/GrNet.cpp
Normal file
@ -0,0 +1,20 @@
|
||||
#include "GrNet.h"
|
||||
|
||||
namespace gr {
|
||||
|
||||
bool GrNet::needToRoute() {
|
||||
std::unordered_map<int, int> cnt;
|
||||
for (auto e : pins) {
|
||||
for (auto f : e) {
|
||||
cnt[f]++;
|
||||
}
|
||||
}
|
||||
for (auto e : pins) {
|
||||
for (auto f : e) {
|
||||
if (cnt[f] == pins.size()) return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
}
|
||||
36
cpp_to_py/gpugr/db/GrNet.h
Normal file
36
cpp_to_py/gpugr/db/GrNet.h
Normal file
@ -0,0 +1,36 @@
|
||||
#pragma once
|
||||
#include <vector>
|
||||
#include <unordered_map>
|
||||
namespace gr {
|
||||
|
||||
class GrNet {
|
||||
public:
|
||||
void setPins(const std::vector<std::vector<int>>& p) { pins = p; }
|
||||
void setBoundingBox(int lx, int ly, int ux, int uy) {
|
||||
lowerx = lx;
|
||||
lowery = ly;
|
||||
upperx = ux;
|
||||
uppery = uy;
|
||||
}
|
||||
void setWires(const std::vector<int>& w) { wires = w; }
|
||||
void setVias(const std::vector<int>& v) { vias = v; }
|
||||
void resetRoute() { wires.clear(), vias.clear(); }
|
||||
void addVias();
|
||||
bool needToRoute();
|
||||
void setNoRoute() { noroute = 1; }
|
||||
int area() { return (upperx - lowerx + 1) * (uppery - lowery + 1); }
|
||||
int hpwl() { return upperx + uppery - lowerx - lowery; }
|
||||
const std::vector<int>& getWires() { return wires; }
|
||||
const std::vector<int>& getVias() { return vias; }
|
||||
const std::vector<std::vector<int>>& getPins() { return pins; }
|
||||
int lowerx, lowery, upperx, uppery, noroute = 0;
|
||||
std::vector<int> points;
|
||||
std::vector<int> pin2gbpinId;
|
||||
std::vector<std::vector<int>> pin2gpdbPinIds;
|
||||
|
||||
private:
|
||||
std::vector<std::vector<int>> pins;
|
||||
std::vector<int> wires, vias;
|
||||
};
|
||||
|
||||
} // namespace gr
|
||||
962
cpp_to_py/gpugr/gr/GPURouter.cu
Normal file
962
cpp_to_py/gpugr/gr/GPURouter.cu
Normal file
@ -0,0 +1,962 @@
|
||||
#include "GPURouter.h"
|
||||
#include "InCellUsage.cuh"
|
||||
#include <cstdio>
|
||||
|
||||
namespace gr {
|
||||
|
||||
constexpr int MAX_ROUTE_LEN_PER_PIN = 130; // too large may exceed the maximum GPU memory
|
||||
|
||||
constexpr int INF = 10000000;
|
||||
constexpr int MAX_COST = 10000000;
|
||||
|
||||
#define BLOCK_SIZE 512
|
||||
#define BLOCK_NUMBER(n) (((n) + (BLOCK_SIZE) - 1) / BLOCK_SIZE)
|
||||
|
||||
__managed__ int STAMP = 0, wireLen, viaLen;
|
||||
|
||||
void GPURouter::initialize(int device_id, int layer, int x, int y, int N_, int cgxsize_, int cgysize_, int direction, int csrn_scale) {
|
||||
gpuMR.startGPU(device_id, layer, cgxsize_, cgysize_);
|
||||
DEVICE_ID = device_id;
|
||||
DIRECTION = direction;
|
||||
COARSENING_SCALE = csrn_scale;
|
||||
cgxsize = cgxsize_;
|
||||
cgysize = cgysize_;
|
||||
|
||||
LAYER = layer;
|
||||
N = N_;
|
||||
X = x;
|
||||
Y = y;
|
||||
int gridGraphSize = LAYER * N * N;
|
||||
cudaMalloc(&dist, (MAX_BATCH_SIZE + 6) * gridGraphSize * sizeof(int));
|
||||
cudaMalloc(&prev, (MAX_BATCH_SIZE + 6) * gridGraphSize * sizeof(int));
|
||||
cudaMalloc(&capacity, gridGraphSize * sizeof(float));
|
||||
cudaMalloc(&wireDist, gridGraphSize * sizeof(float));
|
||||
cudaMalloc(&fixedLength, gridGraphSize * sizeof(float));
|
||||
cudaMalloc(&fixed, gridGraphSize * sizeof(float));
|
||||
cudaMalloc(&wires, gridGraphSize * sizeof(int));
|
||||
cudaMalloc(&vias, gridGraphSize * sizeof(int));
|
||||
cudaMemset(wires, 0, sizeof(int) * gridGraphSize);
|
||||
cudaMemset(vias, 0, sizeof(int) * gridGraphSize);
|
||||
cudaMalloc(&modifiedWire, gridGraphSize * sizeof(int));
|
||||
cudaMalloc(&modifiedVia, gridGraphSize * sizeof(int));
|
||||
cudaMalloc(&viaCost, gridGraphSize * sizeof(dtype));
|
||||
cudaMalloc(&cost, gridGraphSize * sizeof(dtype));
|
||||
cudaMalloc(&costSum, gridGraphSize * sizeof(int64_t));
|
||||
cudaMalloc(&cell_resource, gridGraphSize * sizeof(float));
|
||||
cudaMalloc(&isOverflowWire, gridGraphSize * sizeof(int));
|
||||
cudaMalloc(&isOverflowVia, gridGraphSize * sizeof(int));
|
||||
cudaMalloc(&unitShortCostDiscounted, LAYER * sizeof(float));
|
||||
cudaMallocManaged(&allpins, MAX_BATCH_SIZE * MAX_PIN_SIZE_PER_NET * sizeof(int));
|
||||
}
|
||||
|
||||
GPURouter::~GPURouter() {
|
||||
gpuMR.endGPU();
|
||||
cudaFree(dist);
|
||||
cudaFree(prev);
|
||||
cudaFree(capacity);
|
||||
cudaFree(wireDist);
|
||||
cudaFree(fixedLength);
|
||||
cudaFree(fixed);
|
||||
cudaFree(wires);
|
||||
cudaFree(vias);
|
||||
cudaFree(modifiedWire);
|
||||
cudaFree(modifiedVia);
|
||||
cudaFree(viaCost);
|
||||
cudaFree(cost);
|
||||
cudaFree(costSum);
|
||||
cudaFree(cell_resource);
|
||||
cudaFree(isOverflowWire);
|
||||
cudaFree(isOverflowVia);
|
||||
cudaFree(unitShortCostDiscounted);
|
||||
cudaFree(allpins);
|
||||
|
||||
if(pins != nullptr) cudaFree(pins);
|
||||
if(pinNum != nullptr) cudaFree(pinNum);
|
||||
if(pinNumOffset != nullptr) cudaFree(pinNumOffset);
|
||||
if(routes != nullptr) cudaFree(routes);
|
||||
if(routesOffset != nullptr) cudaFree(routesOffset);
|
||||
if(isOverflowNet != nullptr) cudaFree(isOverflowNet);
|
||||
if(points != nullptr) cudaFree(points);
|
||||
if(gbpoints != nullptr) cudaFree(gbpoints);
|
||||
if(gbpinRoutes != nullptr) cudaFree(gbpinRoutes);
|
||||
if(gbpin2netId != nullptr) cudaFree(gbpin2netId);
|
||||
if(plPinId2gbPinId != nullptr) cudaFree(plPinId2gbPinId);
|
||||
|
||||
if(routesOffsetCPU != nullptr) { delete[] routesOffsetCPU; }
|
||||
if(pinNumCPU != nullptr) { delete[] pinNumCPU; }
|
||||
}
|
||||
|
||||
void GPURouter::setUnitViaMultiplier(float value) {
|
||||
unitViaMultiplier = value;
|
||||
}
|
||||
|
||||
void GPURouter::setUnitVioCost(vector<float>& values, float discount) {
|
||||
//printf("??? setUnitVioCost %.2f\n", discount);
|
||||
float temp[100];
|
||||
for(int i = 0; i < LAYER; i++)
|
||||
temp[i] = values[i] * discount;
|
||||
cudaMemcpy(unitShortCostDiscounted, temp, LAYER * sizeof(float), cudaMemcpyHostToDevice);
|
||||
}
|
||||
|
||||
void GPURouter::setLogisticSlope(float value) {
|
||||
logisticSlope = value;
|
||||
}
|
||||
|
||||
void GPURouter::setUnitViaCost(float value) {
|
||||
unitViaCost = value;
|
||||
}
|
||||
|
||||
void GPURouter::setMap(const vector<float> &cap, const vector<float> &wir, const vector<float> &fixedL, const vector<float> &fix) {
|
||||
int gridGraphSize = LAYER * N * N;
|
||||
auto copy = [&] (const vector<float> &vec, float *target) {
|
||||
cudaMemcpy(target, vec.data(), gridGraphSize * sizeof(float), cudaMemcpyHostToDevice);
|
||||
};
|
||||
copy(cap, capacity);
|
||||
copy(wir, wireDist);
|
||||
copy(fixedL, fixedLength);
|
||||
copy(fix, fixed);
|
||||
}
|
||||
|
||||
__global__ void calculateCellResource(float *cell_resource, int *wires, float *fixed, int *vias, const float *capacity, int N, int LAYER, int tot) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if(idx >= tot) return;
|
||||
cell_resource[idx] = cellResource(idx, wires, fixed, vias, capacity, N, LAYER);
|
||||
}
|
||||
|
||||
__global__ void calculateCoarseCost(float *cell_resource, int *cost, int *wires, float *fixed, int *vias, float *capacity, int N, int xsize, int ysize, int X, int Y, int LAYER, int DIRECTION, int COARSENING_SCALE) {
|
||||
int layer = blockIdx.x / xsize, x = blockIdx.x % xsize, y = threadIdx.x;
|
||||
if(layer == 0) {
|
||||
cost[blockIdx.x * blockDim.x + threadIdx.x] = 10000000;
|
||||
return;
|
||||
}
|
||||
int minx = x * COARSENING_SCALE, maxx = min(X - 1, x * COARSENING_SCALE + COARSENING_SCALE - 1);
|
||||
int miny = y * COARSENING_SCALE, maxy = min(Y - 1, y * COARSENING_SCALE + COARSENING_SCALE - 1);
|
||||
float ans = 0;
|
||||
if(DIRECTION ^ (layer & 1)) {
|
||||
if(y + 1 < ysize) {
|
||||
float sum = 0;
|
||||
for(int i = minx; i <= maxx; i++)
|
||||
for(int j = miny; j <= maxy; j++) {
|
||||
//if(layer == 1 && x == 3 && y == 18)
|
||||
// printf("GPU %d %d: %.2lf\n", i, j, cellResource(layer * N * N + i * N + j, wires, fixed, vias, capacity, N, LAYER));
|
||||
sum += cell_resource[layer * N * N + i * N + j];
|
||||
}
|
||||
ans += 1.0 * COARSENING_SCALE / max(0.1, sum / (maxx - minx + 1) / (maxy - miny + 1));
|
||||
//if(layer == 1 && x == 3 && y == 18)
|
||||
// printf("sum: %.2lf\n", sum);
|
||||
sum = 0;
|
||||
for(int i = minx; i <= maxx; i++)
|
||||
for(int j = miny + COARSENING_SCALE; j <= min(Y - 1, maxy + COARSENING_SCALE); j++) {
|
||||
//if(layer == 1 && x == 3 && y == 18)
|
||||
// printf("GPU %d %d: %.2lf\n", i, j, cellResource(layer * N * N + i * N + j, wires, fixed, vias, capacity, N, LAYER));
|
||||
sum += cell_resource[layer * N * N + i * N + j];
|
||||
}
|
||||
ans += 1.0 * COARSENING_SCALE / max(0.1, sum / (maxx - minx + 1) / (min(Y - 1, maxy + COARSENING_SCALE) - miny - COARSENING_SCALE + 1));
|
||||
//if(layer == 1 && x == 3 && y == 18)
|
||||
// printf("sum: %.2lf\n", sum);
|
||||
cost[layer * xsize * ysize + x * ysize + y] = 100 * ans;
|
||||
}
|
||||
} else {
|
||||
if(x + 1 < xsize) {
|
||||
float sum = 0;
|
||||
for(int i = minx; i <= maxx; i++)
|
||||
for(int j = miny; j <= maxy; j++)
|
||||
sum += cell_resource[layer * N * N + j * N + i];
|
||||
ans += 1.0 * COARSENING_SCALE / max(0.1, sum / (maxx - minx + 1) / (maxy - miny + 1));
|
||||
sum = 0;
|
||||
for(int i = minx + COARSENING_SCALE; i <= min(X - 1, maxx + COARSENING_SCALE); i++)
|
||||
for(int j = miny; j <= maxy; j++)
|
||||
sum += cell_resource[layer * N * N + j * N + i];
|
||||
ans += 1.0 * COARSENING_SCALE / max(0.1, sum / (min(X - 1, maxx + COARSENING_SCALE) - minx - COARSENING_SCALE + 1) / (maxy - miny + 1));
|
||||
cost[layer * xsize * ysize + x * ysize + y] = 100 * ans;
|
||||
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void calculateCoarseVia(float *cell_resource, int *coarseVia, int *wires, float *fixed, int *vias, float *capacity, int N, int LAYER, int xsize, int ysize, int X, int Y, int DIRECTION, int COARSENING_SCALE) {
|
||||
int layer = blockIdx.x / xsize, x = blockIdx.x % xsize, y = threadIdx.x;
|
||||
int minx = x * COARSENING_SCALE, maxx = min(X - 1, x * COARSENING_SCALE + COARSENING_SCALE - 1);
|
||||
int miny = y * COARSENING_SCALE, maxy = min(Y - 1, y * COARSENING_SCALE + COARSENING_SCALE - 1);
|
||||
if(layer + 1 < LAYER) {
|
||||
float sum = 0, ans = 0;
|
||||
for(int i = minx; i <= maxx; i++)
|
||||
for(int j = miny; j <= maxy; j++)
|
||||
if((layer & 1) ^ DIRECTION)
|
||||
sum += cell_resource[layer * N * N + i * N + j];
|
||||
else
|
||||
sum += cell_resource[layer * N * N + j * N + i];
|
||||
|
||||
ans += 1.0 / max(0.1, sum / (maxx - minx + 1) / (maxy - miny + 1));
|
||||
sum = 0;
|
||||
for(int i = minx; i <= maxx; i++)
|
||||
for(int j = miny; j <= maxy; j++)
|
||||
if(!(layer & 1) ^ DIRECTION)
|
||||
sum += cell_resource[(layer + 1) * N * N + i * N + j];
|
||||
else
|
||||
sum += cell_resource[(layer + 1) * N * N + j * N + i];
|
||||
ans += 1.0 / max(0.1, sum / (maxx - minx + 1) / (maxy - miny + 1));
|
||||
coarseVia[layer * xsize * ysize + x * ysize + y] = 100 * ans;
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void initMap(dtype *dist, int *prev, int total, int N) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if(idx < N) {
|
||||
dist[idx] = INF;
|
||||
prev[idx] = idx % total;
|
||||
//for(int i = 0; i < N; i++)
|
||||
// dist[idx + i * total] = INF, prev[idx + i * total] = idx;
|
||||
}
|
||||
}
|
||||
|
||||
/*
|
||||
__managed__ float minval, maxval = 0;
|
||||
__managed__ unsigned long long allroutecost = 0;
|
||||
#define debug -1
|
||||
__global__ void traceBack(int *modifiedWire, int *modifiedVia, dtype *dist, int *prev, int *wires, int *vias, int *pins, int *routes, int *routesOffset, int *cudaPos, int netId, int N, int flag) {
|
||||
routes += routesOffset[netId];
|
||||
dtype minDist = INF;
|
||||
int p = -1, lef = -1, rig = -1, finished = 1;
|
||||
for(int i = 0, cur = 1; i < pins[0]; i++) {
|
||||
//if(netId == debug || debug == -2)
|
||||
// printf("pin %d\n", i);
|
||||
for(int j = 1; j <= pins[cur]; j++) {
|
||||
int pos = pins[cur + j];
|
||||
//if(netId == debug || debug == -2)
|
||||
// printf("access point %d=(%d,%d,%d) dist: %d\n", cudaPos[pos], cudaPos[pos] / N / N, cudaPos[pos] % (N * N) / N, cudaPos[pos] % N, dist[pos]);
|
||||
if(dist[pos] == INF) finished = 0;
|
||||
if(dist[pos] > 0 && dist[pos] < minDist) {
|
||||
minDist = dist[pos];
|
||||
p = pos;
|
||||
lef = cur + 1;
|
||||
rig = cur + pins[cur];
|
||||
}
|
||||
}
|
||||
cur += pins[cur] + 1;
|
||||
}
|
||||
if(flag && finished == 0)
|
||||
printf("REAL ERROR: DISCONNECTED NET%d\n", netId);
|
||||
if(p == -1) {
|
||||
//printf("WARNING: No pin is connected in this round. NET ID: %d\n", netId);
|
||||
return;
|
||||
}
|
||||
//if(netId == debug)
|
||||
// printf("Start tracing result: %d %d %d\n", p / N / N, p % (N * N) / N, p % N);
|
||||
maxval = max(maxval, 1.0 * dist[p]);
|
||||
int expected = p;
|
||||
while(dist[expected] > 0)
|
||||
expected = prev[expected];
|
||||
while(dist[p] > 0) {
|
||||
//if(netId == debug || debug == -2)
|
||||
// printf("%d %d (%d, %d, %d): %d; prev: %d %d (%d, %d, %d): %d\n", p, cudaPos[p], cudaPos[p] / N / N, cudaPos[p] % (N * N) / N, cudaPos[p] % N, dist[p], prev[p], cudaPos[prev[p]], cudaPos[prev[p]] / N / N, cudaPos[prev[p]] % (N * N) / N, cudaPos[prev[p]] % N, dist[prev[p]]);
|
||||
int minp = cudaPos[p], maxp = cudaPos[prev[p]], pre = prev[p];
|
||||
if(minp > maxp) {
|
||||
int temp = minp;
|
||||
minp = maxp;
|
||||
maxp = temp;
|
||||
}
|
||||
int lmin = minp / N / N, lmax = maxp / N / N, x = minp % (N * N) / N, y = minp % N;
|
||||
if(lmin != lmax) {
|
||||
for(int i = lmin; i <= lmax; i++) {
|
||||
int idx = i * N * N + ((i - lmin) % 2 ? y * N + x : x * N + y);
|
||||
if(i < lmax) {
|
||||
atomicAdd(vias + idx, 1);
|
||||
routes[routes[0]++] = idx;
|
||||
routes[routes[0]++] = -1;
|
||||
modifiedVia[idx] = STAMP;
|
||||
}
|
||||
//if(idx != cudaPos[pre])
|
||||
// dist[idx] = 0;
|
||||
}
|
||||
} else {
|
||||
routes[routes[0]++] = minp;
|
||||
routes[routes[0]++] = maxp - minp;
|
||||
for(int i = minp; i <= maxp; i++) {
|
||||
if(i < maxp) {
|
||||
atomicAdd(wires + i, 1);
|
||||
modifiedWire[i] = STAMP;
|
||||
}
|
||||
//if(i != cudaPos[pre])
|
||||
// dist[i] = 0;
|
||||
}
|
||||
}
|
||||
//if(dist[pre] + sum != distp)
|
||||
// printf("ERROR: INCONSISTENT COST; net %d sum %d\n", netId, sum);
|
||||
dist[p] = 0;
|
||||
p = pre;
|
||||
}
|
||||
if(expected != p)
|
||||
printf("netid %d: TRACEBACK ERROR\n", netId);
|
||||
//printf("Expected tracing result: %d %d %d\n", expected / N / N, expected % (N * N) / N, expected % N);
|
||||
//if(netId == debug)
|
||||
// printf("Final tracing result: %d %d %d\n", p / N / N, p % (N * N) / N, p % N);
|
||||
for(int i = lef; i <= rig; i++) {
|
||||
dist[pins[i]] = 0;
|
||||
//if(debug == netId)
|
||||
// printf("setting 0 distance: %d=%d,%d,%d\n", pins[i], pins[i] / N / N, pins[i] / N % N, pins[i] % N);
|
||||
}
|
||||
if(routes[0] > routesOffset[netId + 1] - routesOffset[netId])
|
||||
printf("%d **ERROR: ROUTE_LEN_PER_PIN INSUFFICENT; \n", routes[0]);
|
||||
}
|
||||
*/
|
||||
|
||||
__global__ void calculateWireCost(dtype *cost, float *wireDist, float *fixed, float *fixedLength, int *wires, int *vias, float *capacity, float *unitShortCost, float logisticSlope, int N, int LAYER) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if(!(idx < LAYER * N * N && idx % N + 1 < N)) return;
|
||||
if(capacity[idx] < 0.01) {
|
||||
cost[idx] = INF;
|
||||
return;
|
||||
}
|
||||
int expectedLen = (fixed[idx] * fixedLength[idx] + wires[idx] * wireDist[idx]) / capacity[idx];
|
||||
float remain = capacity[idx] - (fixed[idx] + wires[idx] + 1 + twoCellsViaUsage(idx, vias, N, LAYER));
|
||||
//if(idx == 2 * N * N + 63)
|
||||
// for(int i = 0; i < 9; i++)
|
||||
// printf("%d unit %.2lf\n", i, unitShortCost[i]);
|
||||
float result = wireDist[idx] + expectedLen / (1.0 + exp(logisticSlope * remain)) * unitShortCost[idx / N / N];
|
||||
int result_r = static_cast<int>(result);
|
||||
cost[idx] = (result_r > MAX_COST ? MAX_COST : result_r);
|
||||
}
|
||||
|
||||
__global__ void calculateViaCost(int *wires, float *fixed, float *capacity, dtype *viaCost, float unitViaMultiplier, float unitViaCost, float logisticSlope, int N, int LAYER) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
//if(idx >= viaLen) return;
|
||||
//idx = ids[idx];
|
||||
//if(modified[idx] != STAMP) return;
|
||||
if(idx >= (LAYER - 1) * N * N) return;
|
||||
int layer = idx / N / N + 1, y = idx % (N * N) / N, x = idx % N;
|
||||
float result = unitViaCost * (unitViaMultiplier + inCellViaCost(idx, wires, fixed, capacity, logisticSlope, N) + inCellViaCost(layer * N * N + x * N + y, wires, fixed, capacity, logisticSlope, N));
|
||||
//result /= 100;
|
||||
int result_r = static_cast<int>(result);
|
||||
viaCost[idx] = (result_r > MAX_COST ? MAX_COST : result_r);
|
||||
}
|
||||
|
||||
__global__ void setStartCells(dtype *dist, int *pins, int N, int T) {
|
||||
pins += blockIdx.x * N + 2;
|
||||
dist += blockIdx.x * T;
|
||||
for(int i = 1; i <= pins[0]; i++) {
|
||||
dist[pins[i]] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void markUnrouteUsage(int *pins, int *vias, int cnt) {
|
||||
int cur = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if(cur < cnt) atomicAdd(vias + pins[cur], 1);
|
||||
}
|
||||
|
||||
__global__ void markOverflowWires(const float *capacity, int *wires, int *vias, float *fixed, int *isOverflow, int N, int LAYER) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if(idx < LAYER * N * N && idx % N + 1 < N)
|
||||
isOverflow[idx] = (wires[idx] + fixed[idx] + twoCellsViaUsage(idx, vias, N, LAYER) > capacity[idx]);
|
||||
}
|
||||
|
||||
__global__ void markOverflowVias(const float *capacity, int *wires, int *vias, float *fixed, int *isOverflow, int N, int LAYER) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if(idx >= (LAYER - 1) * N * N) return;
|
||||
int layer = idx / N / N + 1, y = idx % (N * N) / N, x = idx % N;
|
||||
int upper_idx = layer * N * N + x * N + y;
|
||||
isOverflow[idx] = (inCellUsedArea(idx, wires, fixed, N) > capacity[idx] || inCellUsedArea(upper_idx, wires, fixed, N) > capacity[upper_idx]);
|
||||
}
|
||||
|
||||
__global__ void markOverflowNets(int *isOverflowVia, int *isOverflowWire, int *isOverflowNet, int *routes, int *routesOffset, int NET_NUM) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if(idx >= NET_NUM) return;
|
||||
routes += routesOffset[idx];
|
||||
isOverflowNet[idx] = 0;
|
||||
if(routes[0] == -1) {
|
||||
// printf("resolving failed net from pattern routing. netId: %d\n", idx);
|
||||
isOverflowNet[idx] = 1;
|
||||
return;
|
||||
}
|
||||
//if(routes[0] == 0) return;
|
||||
for(int i = 1; i < routes[0]; i += 2) if(routes[i + 1] == -1) {
|
||||
if(isOverflowVia[routes[i]]) {
|
||||
isOverflowNet[idx] = 1;
|
||||
return;
|
||||
}
|
||||
} else {
|
||||
for(int j = 0; j < routes[i + 1]; j++)
|
||||
if(isOverflowWire[routes[i] + j]) {
|
||||
isOverflowNet[idx] = 1;
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void commit(int *routes, int *routesOffset, int *wires, int *vias, int NET_NUM) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if(idx >= NET_NUM) return;
|
||||
routes += routesOffset[idx];
|
||||
for(int i = 1; i < routes[0]; i += 2) if(routes[i + 1] > 0) {
|
||||
for(int j = 0; j < routes[i + 1]; j++)
|
||||
atomicAdd(wires + routes[i] + j, 1);
|
||||
} else
|
||||
atomicAdd(vias + routes[i], 1);
|
||||
}
|
||||
|
||||
__global__ void ripupOverflowNets(int *isOverflowNet, int *routes, int *routesOffset, int *wires, int *vias, int NET_NUM) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if(idx >= NET_NUM || isOverflowNet[idx] == 0) return;
|
||||
routes += routesOffset[idx];
|
||||
for(int i = 1; i < routes[0]; i += 2) if(routes[i + 1] > 0) {
|
||||
for(int j = 0; j < routes[i + 1]; j++)
|
||||
atomicAdd(wires + routes[i] + j, -1);
|
||||
} else
|
||||
atomicAdd(vias + routes[i], -1);
|
||||
routes[0] = 1;
|
||||
}
|
||||
|
||||
__global__ void getWires(int N, int LAYER, int *wires, int *ids) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int layer = idx / (N * N), x = idx / N % N, y = idx % N;
|
||||
// bool needsUpdate = false;
|
||||
for(int dl = -1; dl <= 1; dl++) if(0 <= layer + dl && layer + dl < LAYER)
|
||||
for(int dx = -1; dx <= 1; dx++) if(0 <= x + dx && x + dx < N)
|
||||
for(int dy = -1; dy <= 1; dy++) if(0 <= y + dy && y + dy < N)
|
||||
if(wires[(layer + dl) * N * N + (x + dx) * N + (y + dy)] == STAMP) {
|
||||
ids[atomicAdd(&wireLen, 1)] = idx;
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void getVias(int N, int LAYER, int *vias, int *ids) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int layer = idx / (N * N), x = idx / N % N, y = idx % N;
|
||||
// bool needsUpdate = false;
|
||||
for(int dl = -1; dl <= 1; dl++) if(0 <= layer + dl && layer + dl < LAYER)
|
||||
for(int dx = -1; dx <= 1; dx++) if(0 <= x + dx && x + dx < N)
|
||||
for(int dy = -1; dy <= 1; dy++) if(0 <= y + dy && y + dy < N)
|
||||
if(vias[(layer + dl) * N * N + (x + dx) * N + (y + dy)] == STAMP) {
|
||||
ids[atomicAdd(&viaLen, 1)] = idx;
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void calculateCostSum(int LAYER, int N, int *cost, int64_t *costSum) {
|
||||
extern __shared__ int64_t sum[];
|
||||
int a = threadIdx.x << 1, b = threadIdx.x << 1 | 1;
|
||||
for(int i = 1; i < LAYER; i++) {
|
||||
sum[a] = cost[i * N * N + blockIdx.x * N + N - 1 - a];
|
||||
sum[b] = cost[i * N * N + blockIdx.x * N + N - 1 - b];
|
||||
__syncthreads();
|
||||
for(int d = 0; (1 << d) < N; d++) {
|
||||
if(a >> d & 1)
|
||||
sum[a] += sum[(a >> d << d) - 1];
|
||||
if(b >> d & 1)
|
||||
sum[b] += sum[(b >> d << d) - 1];
|
||||
__syncthreads();
|
||||
}
|
||||
costSum[i * N * N + blockIdx.x * N + N - 1 - a] = sum[a];
|
||||
costSum[i * N * N + blockIdx.x * N + N - 1 - b] = sum[b];
|
||||
__syncthreads();
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void printWires(int *wires, int LAYER, int N) {
|
||||
for(int i = 0; i < LAYER; i++)
|
||||
for(int j = 0; j < N; j++)
|
||||
for(int k = 0; k < N; k++)
|
||||
if(wires[i * N * N + j * N + k]) printf("%d %d ", i * N * N + j * N + k, wires[i * N * N + j * N + k]);
|
||||
}
|
||||
|
||||
__global__ void output(int *wires, int *vias, int LAYER, int X, int Y, int N, int DIRECTION) {
|
||||
for(int i = 2; i < 3; i++)
|
||||
for(int j = 0; j < X; j++)
|
||||
for(int k = 0; k < Y; k++)
|
||||
if((i & 1) ^ DIRECTION) {
|
||||
if(k + 1 < Y) printf("%d %d %d: %d\n", i, j, k, wires[i * N * N + j * N + k]);
|
||||
} else {
|
||||
if(j + 1 < X) printf("%d %d %d: %d\n", i, j, k, wires[i * N * N + j * N + k]);
|
||||
}
|
||||
/*if((i & 1) ^ DIRECTION)
|
||||
for(int j = 0; j < X; j++)
|
||||
for(int k = 0; k < Y - 1; k++)
|
||||
printf("%d %d %d: %d\n", i, j, k, wires[i * N * N + j * N + k]);
|
||||
else
|
||||
for(int j = 0; j < Y; j++)
|
||||
for(int k = 0; k < X - 1; k++)
|
||||
printf("%d %d %d: %d\n", i, j, k, wires[i * N * N + j * N + k]);*/
|
||||
}
|
||||
/*
|
||||
__global__ void generateBatch(int n, int *minx, int *maxx, int *miny, int *maxy, int *res, int *siz, int *used, int *flag) {
|
||||
//printf("thid %d %d\n", threadIdx.x, n);
|
||||
extern __shared__ int selected[];
|
||||
for(int i = threadIdx.x; i < n; i += blockDim.x) used[i] = 0;
|
||||
if(threadIdx.x == 0) selected[1] = 0;
|
||||
__syncthreads();
|
||||
while(selected[1] < n) {
|
||||
int cur = selected[1];
|
||||
if(threadIdx.x == 0) selected[0] = n;
|
||||
__syncthreads();
|
||||
for(int i = threadIdx.x; i < n; i += blockDim.x) {
|
||||
if(used[i] == 0) atomicMin(selected, i);
|
||||
flag[i] = 0;
|
||||
}
|
||||
__syncthreads();
|
||||
#define allowed_overlap 2
|
||||
while(selected[0] < n) {
|
||||
int t = selected[0];
|
||||
__syncthreads();
|
||||
if(threadIdx.x == 0) res[cur++] = t, used[t] = 1, selected[0] = n;//, printf("%d ", t);
|
||||
__syncthreads();
|
||||
for(int offset = 0; offset < n; offset += blockDim.x) {
|
||||
int i = threadIdx.x + offset;
|
||||
if(i < n) {
|
||||
if(used[i] == 0 && flag[i] == 0) {
|
||||
if(maxx[i] + allowed_overlap < minx[t] ||
|
||||
maxx[t] + allowed_overlap < minx[i] ||
|
||||
maxy[i] + allowed_overlap < miny[t] ||
|
||||
maxy[t] + allowed_overlap < miny[i]) atomicMin(selected, i);
|
||||
else
|
||||
flag[i] = 1;
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
if(selected[0] < n) break;
|
||||
}
|
||||
}
|
||||
//if(threadIdx.x == 0) printf("\n");
|
||||
if(threadIdx.x == 0)
|
||||
siz[++siz[0]] = cur - selected[1], selected[1] = cur;
|
||||
__syncthreads();
|
||||
}
|
||||
}
|
||||
*/
|
||||
__global__ void generateBatch(int n, int *minx, int *maxx, int *miny, int *maxy, int *res, int *siz, int *used) {
|
||||
extern __shared__ int shared[];
|
||||
if(threadIdx.x == 0) shared[0] = 0;
|
||||
__syncthreads();
|
||||
while(shared[0] < n) {
|
||||
if(threadIdx.x == 0) shared[2] = 0;
|
||||
__syncthreads();
|
||||
for(int i = 0; i < n; i++) if(used[i] == 0) {
|
||||
if(threadIdx.x == 0) shared[1] = 0;
|
||||
__syncthreads();
|
||||
#define allowed_overlap 2
|
||||
if(threadIdx.x < shared[2]) {
|
||||
if (maxx[i] + allowed_overlap < minx[res[threadIdx.x + shared[0]]] ||
|
||||
maxx[res[threadIdx.x + shared[0]]] + allowed_overlap < minx[i] ||
|
||||
maxy[i] + allowed_overlap < miny[res[threadIdx.x + shared[0]]] ||
|
||||
maxy[res[threadIdx.x + shared[0]]] + allowed_overlap < miny[i]) {}
|
||||
else
|
||||
shared[1] = 1;
|
||||
}
|
||||
__syncthreads();
|
||||
if(threadIdx.x == 0 && shared[1] == 0)
|
||||
res[shared[0] + shared[2]++] = i, used[i] = 1;
|
||||
__syncthreads();
|
||||
}
|
||||
if(threadIdx.x == 0) {
|
||||
siz[++siz[0]] = shared[2], shared[0] += shared[2];
|
||||
if(shared[2] > blockDim.x) printf("ERROR in kernel\n");
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
}
|
||||
|
||||
void GPURouter::route(vector<GrNet> &nets, int iter) {
|
||||
logger.info("GPU Routing start... DIRECTION: %d", DIRECTION);
|
||||
|
||||
double prtime = 0, prpreparetime = 0, batchgentime = 0, timer2 = 0;
|
||||
|
||||
vector<int> netsToRoute;
|
||||
if(iter > 0) {
|
||||
logger.info("Maze Routing...");
|
||||
ripupOverflowNets<<<BLOCK_NUMBER(NET_NUM), BLOCK_SIZE>>> (isOverflowNet, routes, routesOffset, wires, vias, NET_NUM);
|
||||
cudaDeviceSynchronize();
|
||||
for(size_t netId = 0; netId < nets.size(); netId++)
|
||||
if(isOverflowNet[netId] && !nets[netId].noroute) {
|
||||
netsToRoute.emplace_back(netId);
|
||||
// if(nets[netId].noroute) printf("ERROR: net %d is noroute but overflow\n", netId);
|
||||
}
|
||||
} else {
|
||||
logger.info("Pattern Routing...");
|
||||
// int cnt = 0;
|
||||
for(int i = 0; i < nets.size(); i++)
|
||||
if(!nets[i].noroute) netsToRoute.emplace_back(i);
|
||||
/*else {
|
||||
auto &temp = nets[i].getPins();
|
||||
for(auto e : temp)
|
||||
for(auto f : e)
|
||||
allpins[cnt++] = f;
|
||||
if(cnt > MAX_BATCH_SIZE * MAX_PIN_SIZE_PER_NET) std::cerr << "ERROR!\n";
|
||||
}
|
||||
markUnrouteUsage<<<BLOCK_NUMBER(cnt), BLOCK_SIZE>>> (allpins, vias, cnt);
|
||||
cudaDeviceSynchronize();*/
|
||||
}
|
||||
vector<int> batchSizes;
|
||||
{
|
||||
double t = clock();
|
||||
constexpr bool check_vis_correctness = false;
|
||||
std::vector<int> s = netsToRoute;
|
||||
int margin = 0; // NOTE: margin == 0 is okay for PR, but do not check for MR
|
||||
//int margin = iter ? 0 : 2;
|
||||
std::sort(s.begin(), s.end(), [&] (int l, int r) {
|
||||
int area_l = nets[l].area();
|
||||
int area_r = nets[r].area();
|
||||
if (area_l == area_r) {
|
||||
return l > r;
|
||||
}
|
||||
return area_l > area_r;
|
||||
//return nets[l].area() * 1.0 / nets[l].getPins().size() > nets[r].area() * 1.0 / nets[r].getPins().size();
|
||||
});
|
||||
// std::sort(s.begin(), s.end(), [&] (int l, int r) {
|
||||
// int hpwl_l = nets[l].hpwl();
|
||||
// int hpwl_r = nets[r].hpwl();
|
||||
// if (hpwl_l == hpwl_r) return l > r;
|
||||
// return hpwl_l < hpwl_r;
|
||||
// });
|
||||
//for(int i = 0; i < s.size(); i++)
|
||||
// printf("%d, [%d, %d] [%d, %d]\n", s[i], nets[s[i]].lowerx, nets[s[i]].upperx, nets[s[i]].lowery, nets[s[i]].uppery);
|
||||
/*int n = netsToRoute.size();
|
||||
nt *minx, *maxx, *miny, *maxy, *res, *siz, *used, *flag;
|
||||
cudaMallocManaged(&minx, n * sizeof(int));
|
||||
cudaMallocManaged(&maxx, n * sizeof(int));
|
||||
cudaMallocManaged(&miny, n * sizeof(int));
|
||||
cudaMallocManaged(&maxy, n * sizeof(int));
|
||||
cudaMallocManaged(&res, n * sizeof(int));
|
||||
cudaMallocManaged(&siz, (n + 1) * sizeof(int));
|
||||
cudaMalloc(&used, n * sizeof(int));
|
||||
cudaMalloc(&flag, n * sizeof(int));
|
||||
|
||||
siz[0] = 0;
|
||||
for(int i = 0; i < n; i++) {
|
||||
auto &net = nets[s[i]];
|
||||
minx[i] = net.lowerx;
|
||||
maxx[i] = net.upperx;
|
||||
miny[i] = net.lowery;
|
||||
maxy[i] = net.uppery;
|
||||
}
|
||||
|
||||
generateBatch<<<1, 1024, 3 * sizeof(int)>>> (n, minx, maxx, miny, maxy, res, siz, used);
|
||||
cudaDeviceSynchronize();
|
||||
|
||||
for(int i = 0; i < n; i++)
|
||||
netsToRoute[i] = s[res[i]];
|
||||
for(int i = 1; i <= siz[0]; i++)
|
||||
batchSizes.emplace_back(siz[i]);*/
|
||||
/*int startpos = 0;
|
||||
for(auto e : batchSizes) {
|
||||
for(int i = startpos; i < startpos + e; i++)
|
||||
for(int j= startpos; j < i; j++) {
|
||||
int n1 = netsToRoute[i], n2 = netsToRoute[j];
|
||||
if(maxx[n1] + 2 < minx[n2] ||
|
||||
maxx[n2] + 2 < minx[n1] ||
|
||||
maxy[n1] + 2 < miny[n2] ||
|
||||
maxy[n2] + 2 < miny[n2]) {}
|
||||
else
|
||||
printf("ERROR: overlap %d %d!\n", i, j);
|
||||
}
|
||||
|
||||
startpos += e;
|
||||
}*/
|
||||
//static std::vector<std::vector<int>> vis(2000, std::vector<int> (2000, 0));
|
||||
/*RangedBitset<1600> test;
|
||||
test.set(10, 128, 1);
|
||||
printf("%d\n", test.check_all(9, 128, 1));
|
||||
printf("%d\n", test.check_all(13, 128, 1));
|
||||
printf("%d\n", test.check_all(9, 128, 0));
|
||||
printf("%d\n", test.check_all(13, 128, 0));
|
||||
exit(0);*/
|
||||
// const int LEN = 10;
|
||||
if (vis.size() == 0) {
|
||||
vis.resize(2000, std::vector<short>(2000, 0));
|
||||
visLL.resize(2000, std::vector<short>(2000, 0));
|
||||
visRR.resize(2000, std::vector<short>(2000, 0));
|
||||
}
|
||||
auto noConflict = [&] (int netId) {
|
||||
const auto &net = nets[netId];
|
||||
//int blockL = net.lowery / LEN, blockR = net.uppery / LEN;
|
||||
//if(blockL == blockR) {
|
||||
// for(int i = net.lowerx; i <= net.upperx; i++)
|
||||
// for(int j = net.lowery; j <= net.uppery; j++)
|
||||
// if(vis[i][j]) return false;
|
||||
for(int i = max(0, net.lowerx - margin); i <= net.upperx + margin; i++)
|
||||
for(int j = max(0, net.lowery - margin); j <= net.uppery + margin; j++)
|
||||
if(vis[i][j]) return false;
|
||||
/*} else {
|
||||
for(int i = net.lowerx; i <= net.upperx; i++) {
|
||||
if(visLL[i][net.lowery] || visRR[i][net.uppery]) return false;
|
||||
for(int j = blockL + 1; j < blockR; j++)
|
||||
if(visLL[i][j * LEN]) return false;
|
||||
}
|
||||
}*/
|
||||
return true;
|
||||
};
|
||||
auto insert = [&] (int netId) {
|
||||
//double t = clock();
|
||||
const auto &net = nets[netId];
|
||||
int xl = max(0, net.lowerx - margin), xr = net.upperx + margin;
|
||||
int yl = max(0, net.lowery - margin), yr = net.uppery + margin;
|
||||
//int blockL = yl / LEN - (yl % LEN == 0), blockR = yr / LEN + (yr % LEN == 0);
|
||||
for(int i = xl; i <= xr; i++) {
|
||||
for(int j = yl; j <= yr; j++) {
|
||||
if (check_vis_correctness) {
|
||||
vis[i][j] += 1;
|
||||
} else {
|
||||
vis[i][j] = 1;
|
||||
}
|
||||
}
|
||||
/*for(int j = yl; j <= (blockR - 1) * LEN; j++)
|
||||
visLL[i][j] = 1;
|
||||
for(int j = (blockL + 1) * LEN; j <= yr; j++)
|
||||
visRR[i][j] = 1;*/
|
||||
}
|
||||
//modify_cnt += clock() - t;
|
||||
};
|
||||
auto remove = [&] (int netId) {
|
||||
const auto &net = nets[netId];
|
||||
int xl = max(0, net.lowerx - margin), xr = net.upperx + margin;
|
||||
int yl = max(0, net.lowery - margin), yr = net.uppery + margin;
|
||||
//int blockL = yl / LEN - (yl % LEN == 0), blockR = yr / LEN + (yr % LEN == 0);
|
||||
for(int i = xl; i <= xr; i++) {
|
||||
for(int j = yl; j <= yr; j++) {
|
||||
if (check_vis_correctness) {
|
||||
vis[i][j] -= 1;
|
||||
} else {
|
||||
vis[i][j] = 0;
|
||||
}
|
||||
}
|
||||
/*for(int j = yl; j <= (blockR - 1) * LEN; j++)
|
||||
visLL[i][j] = 0;
|
||||
for(int j = (blockL + 1) * LEN; j <= yr; j++)
|
||||
visRR[i][j] = 0;*/
|
||||
}
|
||||
};
|
||||
auto checkVis = [&] () {
|
||||
for (int i = 0; i < 2000; i++) {
|
||||
for (int j = 0; j < 2000; j++) {
|
||||
if (vis[i][j] > 1) {
|
||||
std::cout << "ERROR in batch generation" << std::endl;
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
netsToRoute.clear();
|
||||
int lastUnroute = 0;
|
||||
// FIXME: if batch generation allows bbox overlap, data race would happen in PR kernel
|
||||
while(netsToRoute.size() < s.size()) {
|
||||
int sz = netsToRoute.size();
|
||||
// int last = lastUnroute;
|
||||
int cnt = 0;
|
||||
for(size_t i = lastUnroute; i < s.size(); i++) if(s[i] != -1) {
|
||||
bool no_conflict = 1;
|
||||
no_conflict = noConflict(s[i]);
|
||||
// older version, allow bbox overlap but much faster
|
||||
// if(nets[s[i]].area() * 2 < netsToRoute.size() - sz)
|
||||
// no_conflict = noConflict(s[i]);
|
||||
// else for(size_t j = sz; j < netsToRoute.size(); j++) {
|
||||
// const auto &a = nets[s[i]];
|
||||
// const auto &b = nets[netsToRoute[j]];
|
||||
// if(!(a.upperx + margin < b.lowerx || b.upperx + margin < a.lowerx ||
|
||||
// a.uppery + margin < b.lowery || b.uppery + margin < a.lowery)) {
|
||||
// no_conflict = 0;
|
||||
// break;
|
||||
// }
|
||||
// }
|
||||
if(no_conflict)
|
||||
netsToRoute.emplace_back(s[i]), insert(s[i]), s[i] = -1, cnt = 0;
|
||||
else
|
||||
cnt++;
|
||||
//if(cnt >= 100) break;
|
||||
if(iter && netsToRoute.size() - sz == MAX_BATCH_SIZE) break;
|
||||
if(!iter && netsToRoute.size() - sz == 250) break;
|
||||
}
|
||||
while(lastUnroute < s.size() && s[lastUnroute] == -1) lastUnroute++;
|
||||
if (check_vis_correctness) checkVis();
|
||||
for(int i = sz; i < netsToRoute.size(); i++)
|
||||
remove(netsToRoute[i]);
|
||||
batchSizes.emplace_back(netsToRoute.size() - sz);
|
||||
}
|
||||
batchgentime += clock() - t;
|
||||
logger.info("INFO: Batch Generation Time %.4f", batchgentime / CLOCKS_PER_SEC);
|
||||
}
|
||||
reverse(batchSizes.begin(), batchSizes.end());
|
||||
reverse(netsToRoute.begin(), netsToRoute.end());
|
||||
logger.info("number of batches: %d; number of nets: %d", batchSizes.size(), netsToRoute.size());
|
||||
int startpos = 0;
|
||||
int sumofpins = 0;
|
||||
// int lowestpins = 0;
|
||||
// int batch_cnt = 0;
|
||||
double mrtime = 0, costtime = 0;
|
||||
for(auto batchSize : batchSizes) {
|
||||
calculateWireCost<<<BLOCK_NUMBER(LAYER * N * N), BLOCK_SIZE>>> (cost, wireDist, fixed, fixedLength, wires, vias, capacity, unitShortCostDiscounted, logisticSlope, N, LAYER);
|
||||
calculateViaCost<<<BLOCK_NUMBER((LAYER - 1) * N * N), BLOCK_SIZE>>> (wires, fixed, capacity, viaCost, unitViaMultiplier, unitViaCost, logisticSlope, N, LAYER);
|
||||
calculateCostSum<<<N, N / 2, N * sizeof(int64_t)>>> (LAYER, N, cost, costSum);
|
||||
if(iter == 0) {
|
||||
int offset = batchSize;
|
||||
int gbPinOffset = batchSize;
|
||||
if(points == nullptr) {
|
||||
cudaMallocManaged(&points, 20000000 * sizeof(int));
|
||||
cudaMallocManaged(&gbpoints, 10000000 * sizeof(int));
|
||||
}
|
||||
// double prepare_part_time = 0;
|
||||
double prepare_detailed_time = 0;
|
||||
double t = clock();
|
||||
|
||||
//if(startpos == 11394)
|
||||
//logger.info("net id %d", netsToRoute[startpos]);
|
||||
//netsToRoute[startpos] = 8026;
|
||||
for(int i = 0; i < batchSize; i++) {
|
||||
int netId = netsToRoute[startpos + i];
|
||||
points[i] = offset;
|
||||
gbpoints[i] = gbPinOffset;
|
||||
offset += prepare(prepare_detailed_time, nets[netId], points + offset, routesOffsetCPU[netId], gbpoints + gbPinOffset, gbPinOffset, X, Y, N, LAYER, DIRECTION);
|
||||
}
|
||||
if(offset > 20000000 || gbPinOffset > 10000000)
|
||||
logger.error("ERROR offset %d %d", offset, gbPinOffset);
|
||||
timer2 += prepare_detailed_time;
|
||||
prpreparetime += clock() - t;
|
||||
t = clock();
|
||||
//cudaMemcpy(cuda_points, points, offset * sizeof(int), cudaMemcpyHostToDevice);
|
||||
patternRoute(points, batchSize, costSum, viaCost, dist, prev, wires, vias, routes, gbpoints, gbpinRoutes, X, Y, N, LAYER, DIRECTION);
|
||||
|
||||
prtime += clock() - t;
|
||||
} else {
|
||||
calculateCellResource<<<BLOCK_NUMBER(LAYER * N * N), BLOCK_SIZE>>> (cell_resource, wires, fixed, vias, capacity, N, LAYER, LAYER * N * N);
|
||||
calculateCoarseCost<<<LAYER * cgxsize, cgysize>>> (cell_resource, gpuMR.cost, wires, fixed, vias, capacity, N, cgxsize, cgysize, X, Y, LAYER, DIRECTION, COARSENING_SCALE);
|
||||
calculateCoarseVia<<<(LAYER - 1) * cgxsize, cgysize>>> (cell_resource, gpuMR.via, wires, fixed, vias, capacity, N, LAYER, cgxsize, cgysize, X, Y, DIRECTION, COARSENING_SCALE);
|
||||
gpuMR.run(DIRECTION, iter);
|
||||
int pin10 = 0;
|
||||
for(int i = 0; i < batchSize; i++) {
|
||||
int netId = netsToRoute[startpos + i], cur = 0, *pins = allpins + i * MAX_PIN_SIZE_PER_NET;
|
||||
auto& vec = nets[netId].getPins();
|
||||
pins[cur++] = routesOffsetCPU[netId];
|
||||
pins[cur++] = vec.size();
|
||||
sumofpins += vec.size();
|
||||
if(vec.size() <= 10) pin10++;
|
||||
for(auto a : vec) {
|
||||
pins[cur++] = a.size();
|
||||
for(auto b : a) {
|
||||
pins[cur++] = b;
|
||||
//if(b / N / N >= 5)
|
||||
//std::cerr << "upper lower pins exist. " << b / N / N << std::endl;
|
||||
}
|
||||
}
|
||||
if(cur >= MAX_PIN_SIZE_PER_NET) {
|
||||
std::cerr << "ERROR: NOT ENOUGH FOR PINS cur: " << cur << " MAX_PIN_SIZE_PER_NET: " << MAX_PIN_SIZE_PER_NET << std::endl;
|
||||
exit(-1);
|
||||
}
|
||||
}
|
||||
//printf("%d / %d = %.2lf\n", pin10, batchSize, pin10 * 1.0 / batchSize);
|
||||
initMap<<<BLOCK_NUMBER(batchSize * LAYER * N * N), BLOCK_SIZE>>> (dist, prev, N * N * LAYER, batchSize * N * N * LAYER);
|
||||
setStartCells<<<batchSize, 1>>> (dist, allpins, MAX_PIN_SIZE_PER_NET, N * N * LAYER);
|
||||
double t = clock();
|
||||
gpuMR.getResults(costtime, costSum, allpins, dist, prev, cost, viaCost, wires, vias, routes, N, COARSENING_SCALE, DIRECTION, batchSize, netsToRoute[startpos]);
|
||||
cudaDeviceSynchronize();
|
||||
mrtime += (clock() - t) * 1.0 / CLOCKS_PER_SEC;
|
||||
}
|
||||
//std::cerr << "pins: " << sumofpins << ' ' << 1.0 * sumofpins / netsToRoute.size() << std::endl;
|
||||
startpos += batchSize;
|
||||
}
|
||||
if(!iter) {
|
||||
logger.info("INFO: PR Prepare Time %.4f", prpreparetime / CLOCKS_PER_SEC);
|
||||
logger.info("INFO: PR Kernel Time %.4f", prtime / CLOCKS_PER_SEC);
|
||||
} else {
|
||||
logger.info("INFO: MR Func time %.4f", mrtime);
|
||||
logger.info("INFO: MR Cost calc time %.4f", costtime / CLOCKS_PER_SEC);
|
||||
}
|
||||
//output<<<1, 1>>> (wires, vias, LAYER, X, Y, N, DIRECTION);
|
||||
|
||||
//gpuMR.query();
|
||||
//if(iter == db::setting.rrrIterLimit - 1) {
|
||||
if(1) {
|
||||
markOverflowWires<<<BLOCK_NUMBER(LAYER * N * N), BLOCK_SIZE>>> (capacity, wires, vias, fixed, isOverflowWire, N, LAYER);
|
||||
markOverflowVias<<<BLOCK_NUMBER((LAYER - 1) * N * N), BLOCK_SIZE>>> (capacity, wires, vias, fixed, isOverflowVia, N, LAYER);
|
||||
markOverflowNets<<<BLOCK_NUMBER(NET_NUM), BLOCK_SIZE>>> (isOverflowVia, isOverflowWire, isOverflowNet, routes, routesOffset, NET_NUM);
|
||||
cudaDeviceSynchronize();
|
||||
int cnt = 0;
|
||||
for(size_t netId = 0; netId < nets.size(); netId++)
|
||||
if(isOverflowNet[netId]) {
|
||||
cnt++;
|
||||
if(nets[netId].noroute) printf("ERROR: net %zu is noroute but overflow\n", netId);
|
||||
}
|
||||
logger.info("Final Overflow Net Number: %d", cnt);
|
||||
numOvflNets = cnt;
|
||||
}
|
||||
}
|
||||
|
||||
void GPURouter::setFromNets(vector<GrNet> &nets, int numPlPin_) {
|
||||
NET_NUM = nets.size();
|
||||
pinNumCPU = new int[NET_NUM];
|
||||
routesOffsetCPU = new int[NET_NUM + 1];
|
||||
routesOffsetCPU[0] = 0;
|
||||
for(int i = 0; i < NET_NUM; i++) {
|
||||
pinNumCPU[i] = nets[i].getPins().size();
|
||||
routesOffsetCPU[i + 1] = pinNumCPU[i] * MAX_ROUTE_LEN_PER_PIN + routesOffsetCPU[i];
|
||||
}
|
||||
cudaMallocManaged(&isOverflowNet, NET_NUM * sizeof(int));
|
||||
cudaMalloc(&routes, routesOffsetCPU[NET_NUM] * sizeof(int));
|
||||
cudaMemset(routes, 0, routesOffsetCPU[NET_NUM] * sizeof(int));
|
||||
logger.info("total routes: %d", routesOffsetCPU[NET_NUM]);
|
||||
cudaMalloc(&routesOffset, (NET_NUM + 1) * sizeof(int));
|
||||
cudaMemcpy(routesOffset, routesOffsetCPU, sizeof(int) * (NET_NUM + 1), cudaMemcpyHostToDevice);
|
||||
|
||||
for (int netId = nets.size() - 1; netId >= 0; netId--) {
|
||||
auto& lastGrNet = nets[netId];
|
||||
if (lastGrNet.pin2gbpinId.size() > 0) {
|
||||
numGbPin = lastGrNet.pin2gbpinId[lastGrNet.pin2gbpinId.size() - 1] + 1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
// numRoutes, routeId, routeId, routeId, routeId, numVias
|
||||
cudaMalloc(&gbpinRoutes, 6 * numGbPin * sizeof(int));
|
||||
cudaMemset(gbpinRoutes, 0, 6 * numGbPin * sizeof(int));
|
||||
|
||||
// number of pins in placement database
|
||||
numPlPin = numPlPin_;
|
||||
std::vector<int> gbpin2netIdCPU(numGbPin);
|
||||
std::vector<int> plPinId2gbPinIdCPU(numPlPin, -1);
|
||||
for (int netId = 0; netId < nets.size(); netId ++) {
|
||||
for (int pinId = 0; pinId < nets[netId].getPins().size(); pinId++) {
|
||||
int gbpinId = nets[netId].pin2gbpinId[pinId];
|
||||
gbpin2netIdCPU[gbpinId] = netId;
|
||||
std::vector<int>& gpdbPinIds = nets[netId].pin2gpdbPinIds[pinId];
|
||||
for (int gpdbPinId : gpdbPinIds) {
|
||||
plPinId2gbPinIdCPU[gpdbPinId] = gbpinId;
|
||||
}
|
||||
}
|
||||
}
|
||||
cudaMalloc(&gbpin2netId, numGbPin * sizeof(int));
|
||||
cudaMemcpy(gbpin2netId, gbpin2netIdCPU.data(), numGbPin * sizeof(int), cudaMemcpyHostToDevice);
|
||||
cudaMalloc(&plPinId2gbPinId, numPlPin * sizeof(int));
|
||||
cudaMemcpy(plPinId2gbPinId, plPinId2gbPinIdCPU.data(), numPlPin * sizeof(int), cudaMemcpyHostToDevice);
|
||||
|
||||
cudaDeviceSynchronize();
|
||||
}
|
||||
|
||||
void GPURouter::setToNets(vector<GrNet> &nets) {
|
||||
int *routesCPU = new int[routesOffsetCPU[NET_NUM]];
|
||||
int mx = 0;
|
||||
cudaMemcpy(routesCPU, routes, sizeof(int) * routesOffsetCPU[NET_NUM], cudaMemcpyDeviceToHost);
|
||||
int num_net_use_too_many_route = 0;
|
||||
for(size_t netId = 0; netId < nets.size(); netId++) {
|
||||
vector<int> wires, vias;
|
||||
int *routesSub = routesCPU + routesOffsetCPU[netId];
|
||||
mx = max(mx, routesSub[0] / pinNumCPU[netId]);
|
||||
if(routesSub[0] > routesOffsetCPU[netId + 1] - routesOffsetCPU[netId]) {
|
||||
num_net_use_too_many_route++;
|
||||
// std::cerr << "ERROR: too many routesSub! Please set MAX_ROUTE_LEN_PER_PIN larger than " << routesSub[0] / pinNumCPU[netId] << std::endl;
|
||||
}
|
||||
for(int i = 1; i < routesSub[0]; i += 2) {
|
||||
if (routesSub[i + 1] > 0) {
|
||||
wires.emplace_back(routesSub[i]);
|
||||
wires.emplace_back(routesSub[i + 1]);
|
||||
} else if (routesSub[i + 1] == -1) {
|
||||
vias.emplace_back(routesSub[i]);
|
||||
}
|
||||
}
|
||||
nets[netId].setWires(wires);
|
||||
nets[netId].setVias(vias);
|
||||
//nets[netId].useExtraVias();
|
||||
};
|
||||
logger.info("max routes[0] = %d", mx);
|
||||
if (num_net_use_too_many_route) {
|
||||
std::cerr << "ERROR: there are " << num_net_use_too_many_route << " nets use too many route segments!";
|
||||
std::cerr << " Please set MAX_ROUTE_LEN_PER_PIN (" << MAX_ROUTE_LEN_PER_PIN << ") larger than " << mx << std::endl;
|
||||
}
|
||||
delete[] routesCPU;
|
||||
}
|
||||
|
||||
} // namespace gr
|
||||
116
cpp_to_py/gpugr/gr/GPURouter.h
Normal file
116
cpp_to_py/gpugr/gr/GPURouter.h
Normal file
@ -0,0 +1,116 @@
|
||||
#pragma once
|
||||
#include "MazeRoute.h"
|
||||
#include "PatternRoute.h"
|
||||
#include "common/common.h"
|
||||
#include "gpugr/db/GrNet.h"
|
||||
|
||||
namespace gr {
|
||||
|
||||
typedef int dtype;
|
||||
|
||||
class GPURouter {
|
||||
public:
|
||||
GPURouter(){};
|
||||
GPURouter(
|
||||
int device_id, int layer, int x, int y, int N_, int cgxsize_, int cgysize_, int direction, int csrn_scale) {
|
||||
initialize(device_id, layer, x, y, N_, cgxsize_, cgysize_, direction, csrn_scale);
|
||||
}
|
||||
~GPURouter();
|
||||
|
||||
void initialize(
|
||||
int device_id, int layer, int x, int y, int N_, int cgxsize_, int cgysize_, int direction, int csrn_scale);
|
||||
|
||||
void setMap(const vector<float> &cap,
|
||||
const vector<float> &wir,
|
||||
const vector<float> &fixedL,
|
||||
const vector<float> &fix);
|
||||
void setFromNets(vector<GrNet> &nets, int numPlPin_);
|
||||
void setToNets(vector<GrNet> &nets);
|
||||
void route(vector<GrNet> &nets, int iterleft);
|
||||
void setUnitViaMultiplier(float w);
|
||||
void setUnitVioCost(vector<float>& cost, float discount);
|
||||
void setLogisticSlope(float value);
|
||||
void setUnitViaCost(float value);
|
||||
void query();
|
||||
|
||||
public:
|
||||
std::tuple<torch::Tensor, torch::Tensor, torch::Tensor> getDemandMap();
|
||||
torch::Tensor getCapacityMap();
|
||||
torch::Tensor calcRouteGrad(torch::Tensor mask_map,
|
||||
torch::Tensor wire_dmd_map_2d,
|
||||
torch::Tensor via_dmd_map_2d,
|
||||
torch::Tensor cap_map_2d,
|
||||
torch::Tensor dist_weights,
|
||||
torch::Tensor wirelength_weights,
|
||||
torch::Tensor route_gradmat,
|
||||
torch::Tensor node2pin_list,
|
||||
torch::Tensor node2pin_list_end,
|
||||
float grad_weight,
|
||||
float unit_wire_cost,
|
||||
float unit_via_cost,
|
||||
int num_nodes);
|
||||
torch::Tensor calcFillerRouteGrad(torch::Tensor filler_pos,
|
||||
torch::Tensor filler_size,
|
||||
torch::Tensor filler_weight,
|
||||
torch::Tensor expand_ratio,
|
||||
torch::Tensor grad_mat,
|
||||
float grad_weight,
|
||||
float unit_len_x,
|
||||
float unit_len_y,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_fillers);
|
||||
torch::Tensor calcPseudoPinGrad(torch::Tensor node_pos, torch::Tensor pseudo_pin_pos, float gamma);
|
||||
torch::Tensor calcNodeInflateRatio(torch::Tensor node_pos,
|
||||
torch::Tensor node_size,
|
||||
torch::Tensor node_weight,
|
||||
torch::Tensor expand_ratio,
|
||||
torch::Tensor inflate_mat,
|
||||
float grad_weight,
|
||||
float unit_len_x,
|
||||
float unit_len_y,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
bool use_weighted_inflation);
|
||||
torch::Tensor calcInflatedPinRelCpos(torch::Tensor node_inflate_ratio,
|
||||
torch::Tensor old_pin_rel_cpos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
int num_movable_nodes);
|
||||
int getNumOvflNets() { return numOvflNets; }
|
||||
|
||||
private:
|
||||
GPUMazeRouter gpuMR;
|
||||
// routes:
|
||||
// (x, y): starting point x, length |y|; negative y implies vias
|
||||
|
||||
int DEVICE_ID;
|
||||
int LAYER, N, X, Y, NET_NUM, DIRECTION;
|
||||
int COARSENING_SCALE;
|
||||
int cgxsize, cgysize;
|
||||
|
||||
int *pinNum = nullptr, *pinNumOffset = nullptr, *pins = nullptr;
|
||||
int *routes = nullptr, *routesOffset = nullptr, *routesOffsetCPU = nullptr, *pinNumCPU = nullptr;
|
||||
int *allpins;
|
||||
int *points = nullptr, *gbpoints = nullptr;
|
||||
int *gbpinRoutes = nullptr, *gbpin2netId = nullptr, *plPinId2gbPinId = nullptr;
|
||||
float *capacity, *wireDist, *fixedLength, *fixed;
|
||||
int *wires, *vias, *prev;
|
||||
int *isOverflowWire, *isOverflowVia, *isOverflowNet = nullptr;
|
||||
int *boundaries, *isLocked;
|
||||
int *wiresCPU, *viasCPU;
|
||||
int *cudaIndex, *cudaCostIndex;
|
||||
int *modifiedVia, *modifiedWire, *viasToBeUpdated, *wiresToBeUpdated;
|
||||
dtype *dist, *cost, *viaCost;
|
||||
int64_t *costSum;
|
||||
float *unitShortCostDiscounted, unitViaCost, unitViaMultiplier = 1, logisticSlope = 1, *cell_resource;
|
||||
|
||||
int numGbPin, numPlPin;
|
||||
|
||||
int numOvflNets = 0;
|
||||
|
||||
const int MAX_BATCH_SIZE = 100, MAX_PIN_SIZE_PER_NET = 500000;
|
||||
|
||||
std::vector<std::vector<short>> vis, visLL, visRR;
|
||||
};
|
||||
|
||||
} // namespace gr
|
||||
577
cpp_to_py/gpugr/gr/GPURouterTorch.cu
Normal file
577
cpp_to_py/gpugr/gr/GPURouterTorch.cu
Normal file
@ -0,0 +1,577 @@
|
||||
#include "GPURouter.h"
|
||||
#include "InCellUsage.cuh"
|
||||
|
||||
namespace gr {
|
||||
|
||||
#define BLOCK_SIZE 512
|
||||
#define BLOCK_NUMBER(n) (((n) + (BLOCK_SIZE) - 1) / BLOCK_SIZE)
|
||||
|
||||
__device__ void inline cudaSwapInt(int &a, int &b) {
|
||||
int c(a);
|
||||
a = b;
|
||||
b = c;
|
||||
}
|
||||
|
||||
__device__ float overlap(float x_l, float x_h, float bin_x_l) {
|
||||
// bin_x_h == bin_x_l + 1
|
||||
return min(x_h, bin_x_l + 1) - max(x_l, bin_x_l);
|
||||
}
|
||||
|
||||
__global__ void calc_node_grad_deterministic_cuda_kernel(
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_grad,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_grad,
|
||||
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> node2pin_list,
|
||||
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> node2pin_list_end,
|
||||
int num_nodes) {
|
||||
const int index = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
const int i = index >> 1; // node index
|
||||
if (i < num_nodes) {
|
||||
const int c = index & 1; // channel index
|
||||
int64_t start_idx = 0;
|
||||
if (i != 0) {
|
||||
start_idx = node2pin_list_end[i - 1];
|
||||
}
|
||||
int64_t end_idx = node2pin_list_end[i];
|
||||
if (end_idx != start_idx) {
|
||||
node_grad[i][c] += pin_grad[node2pin_list[start_idx]][c];
|
||||
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
|
||||
node_grad[i][c] += pin_grad[node2pin_list[idx]][c];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void getDmdTensor(
|
||||
torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> dmdMap,
|
||||
torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> wireDmdMap,
|
||||
torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> viaDmdMap,
|
||||
const float *capacity, int *wires, int *vias, float *fixed, int N, int LAYER, int xSize, int ySize, int DIRECTION) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if(idx < LAYER * N * N && idx % N + 1 < N) {
|
||||
int layer = idx / N / N, x = idx / N % N, y = idx % N;
|
||||
if (!(layer & 1) ^ DIRECTION) cudaSwapInt(x, y);
|
||||
if (layer < LAYER && x < xSize && y < ySize) {
|
||||
float wireDmd = wires[idx] + fixed[idx];
|
||||
float viaDmd = twoCellsViaUsage(idx, vias, N, LAYER);
|
||||
dmdMap[layer][x][y] = wireDmd + viaDmd;
|
||||
wireDmdMap[layer][x][y] = wireDmd;
|
||||
viaDmdMap[layer][x][y] = viaDmd;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void getCapTensor(
|
||||
torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> capMap,
|
||||
const float *capacity, int *wires, int *vias, float *fixed, int N, int LAYER, int xSize, int ySize, int DIRECTION) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if(idx < LAYER * N * N && idx % N + 1 < N) {
|
||||
int layer = idx / N / N, x = idx / N % N, y = idx % N;
|
||||
if (!(layer & 1) ^ DIRECTION) cudaSwapInt(x, y);
|
||||
if (layer < LAYER && x < xSize && y < ySize) {
|
||||
capMap[layer][x][y] = capacity[idx];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void compGcellRouteForce(
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> gbpin_grad,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> mask_map,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> wire_dmd_map_2d,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> via_dmd_map_2d,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> cap_map_2d,
|
||||
torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> dist_weights,
|
||||
torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> wirelength_weights,
|
||||
torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> route_gradmat,
|
||||
float grad_weight, float unit_wire_cost, float unit_via_cost,
|
||||
int *gbpinRoutes, int *gbpin2netId, int *routes, int *routesOffset,
|
||||
int numGbPin, int N, int LAYER, int xSize, int ySize, int DIRECTION
|
||||
) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (idx < numGbPin) {
|
||||
int netId = gbpin2netId[idx];
|
||||
routes += routesOffset[netId];
|
||||
if (routes[0] == -1) {
|
||||
// FIXME: failed PR, use floating cost instead of int cost
|
||||
return;
|
||||
}
|
||||
gbpinRoutes += idx * 6;
|
||||
int numGbpinRoutes = gbpinRoutes[0];
|
||||
// int numGbpinVias = gbpinRoutes[5]; // TODO: consider #Vias in route grad
|
||||
for (int i = 1; i < 1 + numGbpinRoutes; i++) {
|
||||
int routeId = gbpinRoutes[i];
|
||||
bool reverseRoute = false;
|
||||
if (routeId < 0) {
|
||||
// for a segment (lx, hx), if reverseRoute is true, the gbpin is located at hx, else at lx
|
||||
routeId = -routeId;
|
||||
reverseRoute = true;
|
||||
}
|
||||
int p = routes[routeId];
|
||||
int l = p / N / N, x = p % (N * N) / N, y = p % N;
|
||||
if (!(l & 1) ^ DIRECTION) cudaSwapInt(x, y);
|
||||
int lx = x, hx = x, ly = y, hy = y;
|
||||
if ((l & 1) ^ DIRECTION) {
|
||||
hy += routes[routeId + 1];
|
||||
} else {
|
||||
hx += routes[routeId + 1];
|
||||
}
|
||||
if (lx != hx) {
|
||||
// x direction route segement
|
||||
float grad = 0, total_weight = 0;
|
||||
for (int j = lx; j <= hx; j++) {
|
||||
float cur_dist_weight;
|
||||
if (reverseRoute) {
|
||||
cur_dist_weight = dist_weights[hx - j];
|
||||
} else {
|
||||
cur_dist_weight = dist_weights[j - lx];
|
||||
}
|
||||
// TODO: 1) should we also consider X direction?
|
||||
// 2) consider via cost?
|
||||
float cost = unit_wire_cost / min(cap_map_2d[j][ly], 0.2);
|
||||
grad += cost * route_gradmat[1][j][ly] * mask_map[j][ly] * cur_dist_weight;
|
||||
total_weight += cur_dist_weight;
|
||||
}
|
||||
gbpin_grad[idx][1] = grad_weight * grad / total_weight * wirelength_weights[hx - lx + 1];
|
||||
} else if (ly != hy) {
|
||||
// y direction route segement
|
||||
float grad = 0, total_weight = 0;
|
||||
for (int j = ly; j<= hy; j++) {
|
||||
float cur_dist_weight;
|
||||
if (reverseRoute) {
|
||||
cur_dist_weight = dist_weights[hy - j];
|
||||
} else {
|
||||
cur_dist_weight = dist_weights[j - ly];
|
||||
}
|
||||
float cost = unit_wire_cost / min(cap_map_2d[lx][j], 0.2);
|
||||
grad += cost * route_gradmat[0][lx][j] * mask_map[lx][j] * cur_dist_weight;
|
||||
total_weight += cur_dist_weight;
|
||||
}
|
||||
gbpin_grad[idx][0] = grad_weight * grad / total_weight * wirelength_weights[hy - ly + 1];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void assignRouteForceToPlPin(
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> plpin_grad,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> gbpin_grad,
|
||||
int *plPinId2gbPinId, int numPlPin
|
||||
) {
|
||||
int plPinId = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (plPinId < numPlPin) {
|
||||
int gbPinId = plPinId2gbPinId[plPinId];
|
||||
plpin_grad[plPinId][0] = gbpin_grad[gbPinId][0];
|
||||
plpin_grad[plPinId][1] = gbpin_grad[gbPinId][1];
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void fillerRouteForce(
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> filler_pos,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> filler_size,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> filler_weight,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> expand_ratio,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> filler_grad,
|
||||
const float *grad_mat,
|
||||
float grad_weight,
|
||||
float unit_len_x,
|
||||
float unit_len_y,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_fillers
|
||||
) {
|
||||
const int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < num_fillers) {
|
||||
float weight = filler_weight[i];
|
||||
if (weight > 0) {
|
||||
const float x_l = (filler_pos[i][0] - filler_size[i][0] / 2) / unit_len_x;
|
||||
const float x_h = (filler_pos[i][0] + filler_size[i][0] / 2) / unit_len_x;
|
||||
const float y_l = (filler_pos[i][1] - filler_size[i][1] / 2) / unit_len_y;
|
||||
const float y_h = (filler_pos[i][1] + filler_size[i][1] / 2) / unit_len_y;
|
||||
if (x_h - x_l < 0 || y_h - y_l < 0) return;
|
||||
weight *= expand_ratio[i];
|
||||
int x_lf = lround(floor(x_l));
|
||||
int x_hf = lround(floor(x_h));
|
||||
int y_lf = lround(floor(y_l));
|
||||
int y_hf = lround(floor(y_h));
|
||||
x_lf = max(x_lf, 0);
|
||||
x_hf = min(x_hf, num_bin_x - 1);
|
||||
y_lf = max(y_lf, 0);
|
||||
y_hf = min(y_hf, num_bin_y - 1);
|
||||
|
||||
float gradX = 0;
|
||||
float gradY = 0;
|
||||
for (int j = x_lf; j < x_hf + 1; j++) {
|
||||
float bin_x_l = static_cast<float>(j);
|
||||
float overlap_x = overlap(x_l, x_h, bin_x_l);
|
||||
for (int k = y_lf; k < y_hf + 1; k++) {
|
||||
float bin_y_l = static_cast<float>(k);
|
||||
float overlap_y = overlap(y_l, y_h, bin_y_l);
|
||||
float overlap_area = overlap_x * overlap_y;
|
||||
gradX += grad_mat[0 * num_bin_x * num_bin_y + j * num_bin_y + k] * overlap_area;
|
||||
gradY += grad_mat[1 * num_bin_x * num_bin_y + j * num_bin_y + k] * overlap_area;
|
||||
}
|
||||
}
|
||||
filler_grad[i][0] = grad_weight * weight * gradX;
|
||||
filler_grad[i][1] = grad_weight * weight * gradY;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void inflateNodeRatioWeighted(
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_size,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> node_weight,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> expand_ratio,
|
||||
const torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> inflate_mat,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_inflate_ratio,
|
||||
float grad_weight,
|
||||
float unit_len_x,
|
||||
float unit_len_y,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_nodes
|
||||
) {
|
||||
const int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < num_nodes) {
|
||||
float weight = node_weight[i] * expand_ratio[i];
|
||||
if (weight > 0) {
|
||||
const float x_l = (node_pos[i][0] - node_size[i][0] / 2) / unit_len_x;
|
||||
const float x_h = (node_pos[i][0] + node_size[i][0] / 2) / unit_len_x;
|
||||
const float y_l = (node_pos[i][1] - node_size[i][1] / 2) / unit_len_y;
|
||||
const float y_h = (node_pos[i][1] + node_size[i][1] / 2) / unit_len_y;
|
||||
if (x_h - x_l < 0 || y_h - y_l < 0) return;
|
||||
int x_lf = lround(floor(x_l));
|
||||
int x_hf = lround(floor(x_h));
|
||||
int y_lf = lround(floor(y_l));
|
||||
int y_hf = lround(floor(y_h));
|
||||
x_lf = max(x_lf, 0);
|
||||
x_hf = min(x_hf, num_bin_x - 1);
|
||||
y_lf = max(y_lf, 0);
|
||||
y_hf = min(y_hf, num_bin_y - 1);
|
||||
const float node_area = (x_h - x_l) * (y_h - y_l);
|
||||
|
||||
float inflate_x = 0, inflate_y = 0;
|
||||
for (int j = x_lf; j < x_hf + 1; j++) {
|
||||
float bin_x_l = static_cast<float>(j);
|
||||
float overlap_x = overlap(x_l, x_h, bin_x_l);
|
||||
for (int k = y_lf; k < y_hf + 1; k++) {
|
||||
float bin_y_l = static_cast<float>(k);
|
||||
float overlap_y = overlap(y_l, y_h, bin_y_l);
|
||||
float overlap_area_ratio = overlap_x * overlap_y / node_area;
|
||||
inflate_x += overlap_area_ratio * inflate_mat[0][j][k];
|
||||
inflate_y += overlap_area_ratio * inflate_mat[1][j][k];
|
||||
}
|
||||
}
|
||||
node_inflate_ratio[i][0] = grad_weight * weight * inflate_x;
|
||||
node_inflate_ratio[i][1] = grad_weight * weight * inflate_y;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void inflateNodeRatioMax(
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_size,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> node_weight,
|
||||
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> expand_ratio,
|
||||
const torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> inflate_mat,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_inflate_ratio,
|
||||
float grad_weight,
|
||||
float unit_len_x,
|
||||
float unit_len_y,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_nodes
|
||||
) {
|
||||
const int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < num_nodes) {
|
||||
float weight = node_weight[i] * expand_ratio[i];
|
||||
if (weight > 0) {
|
||||
const float x_l = (node_pos[i][0] - node_size[i][0] / 2) / unit_len_x;
|
||||
const float x_h = (node_pos[i][0] + node_size[i][0] / 2) / unit_len_x;
|
||||
const float y_l = (node_pos[i][1] - node_size[i][1] / 2) / unit_len_y;
|
||||
const float y_h = (node_pos[i][1] + node_size[i][1] / 2) / unit_len_y;
|
||||
if (x_h - x_l < 0 || y_h - y_l < 0) return;
|
||||
int x_lf = lround(floor(x_l));
|
||||
int x_hf = lround(floor(x_h));
|
||||
int y_lf = lround(floor(y_l));
|
||||
int y_hf = lround(floor(y_h));
|
||||
x_lf = max(x_lf, 0);
|
||||
x_hf = min(x_hf, num_bin_x - 1);
|
||||
y_lf = max(y_lf, 0);
|
||||
y_hf = min(y_hf, num_bin_y - 1);
|
||||
|
||||
float inflate_x = 0, inflate_y = 0;
|
||||
for (int j = x_lf; j < x_hf + 1; j++) {
|
||||
for (int k = y_lf; k < y_hf + 1; k++) {
|
||||
inflate_x = max(inflate_x, inflate_mat[0][j][k]);
|
||||
inflate_y = max(inflate_y, inflate_mat[1][j][k]);
|
||||
}
|
||||
}
|
||||
node_inflate_ratio[i][0] = grad_weight * weight * inflate_x;
|
||||
node_inflate_ratio[i][1] = grad_weight * weight * inflate_y;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void inflatePinRelCpos(
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_inflate_ratio,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> old_pin_rel_cpos,
|
||||
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> pin_id2node_id,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> new_pin_rel_cpos,
|
||||
int num_movable_nodes,
|
||||
int num_pins
|
||||
) {
|
||||
const int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if (i < num_pins) {
|
||||
int64_t node_id = pin_id2node_id[i];
|
||||
if (node_id < num_movable_nodes) {
|
||||
new_pin_rel_cpos[i][0] = old_pin_rel_cpos[i][0] * node_inflate_ratio[node_id][0];
|
||||
new_pin_rel_cpos[i][1] = old_pin_rel_cpos[i][1] * node_inflate_ratio[node_id][1];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void pseudoPinForce(
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pseudo_pin_pos,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_grad,
|
||||
int num_nodes,
|
||||
float inv_gamma) {
|
||||
const int index = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
const int pin_id = index >> 1; // net index
|
||||
if (pin_id < num_nodes) {
|
||||
const int c = index & 1; // channel index
|
||||
|
||||
float x1 = node_pos[pin_id][c];
|
||||
float x2 = pseudo_pin_pos[pin_id][c];
|
||||
float x_max = max(x1, x2);
|
||||
float x_min = min(x1, x2);
|
||||
|
||||
float sum_x_exp_x = 0;
|
||||
float sum_x_exp_nx = 0;
|
||||
float sum_exp_x = 0;
|
||||
float sum_exp_nx = 0;
|
||||
|
||||
for (int i = 0; i < 2; i++) {
|
||||
float cur_x = i == 0 ? x1 : x2;
|
||||
float recenter_exp_x = exp((cur_x - x_max) * inv_gamma);
|
||||
float recenter_exp_nx = exp((x_min - cur_x) * inv_gamma);
|
||||
|
||||
sum_x_exp_x += cur_x * recenter_exp_x;
|
||||
sum_x_exp_nx += cur_x * recenter_exp_nx;
|
||||
sum_exp_x += recenter_exp_x;
|
||||
sum_exp_nx += recenter_exp_nx;
|
||||
}
|
||||
float inv_sum_exp_x = 1 / sum_exp_x;
|
||||
float inv_sum_exp_nx = 1 / sum_exp_nx;
|
||||
|
||||
float s_x = sum_x_exp_x * inv_sum_exp_x;
|
||||
float ns_nx = sum_x_exp_nx * inv_sum_exp_nx;
|
||||
float x_coeff = inv_gamma * inv_sum_exp_x;
|
||||
float nx_coeff = -inv_gamma * inv_sum_exp_nx;
|
||||
float grad_const = (1 - inv_gamma * s_x) * inv_sum_exp_x;
|
||||
float grad_nconst = (1 + inv_gamma * ns_nx) * inv_sum_exp_nx;
|
||||
|
||||
// calc x1 (original pin)'s gradient
|
||||
float recenter_exp_x = exp((x1 - x_max) * inv_gamma);
|
||||
float recenter_exp_nx = exp((x_min - x1) * inv_gamma);
|
||||
node_grad[pin_id][c] = (grad_const + x_coeff * x1) * recenter_exp_x -
|
||||
(grad_nconst + nx_coeff * x1) * recenter_exp_nx;
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
std::tuple<torch::Tensor, torch::Tensor, torch::Tensor> GPURouter::getDemandMap() {
|
||||
int xSize = X;
|
||||
int ySize = Y;
|
||||
torch::Tensor dmdMap = torch::zeros({LAYER, xSize, ySize},
|
||||
torch::dtype(torch::kFloat32).device(torch::Device(torch::kCUDA, DEVICE_ID)));
|
||||
torch::Tensor wireDmdMap = torch::zeros({LAYER, xSize, ySize},
|
||||
torch::dtype(torch::kFloat32).device(torch::Device(torch::kCUDA, DEVICE_ID)));
|
||||
torch::Tensor viaDmdMap = torch::zeros({LAYER, xSize, ySize},
|
||||
torch::dtype(torch::kFloat32).device(torch::Device(torch::kCUDA, DEVICE_ID)));
|
||||
getDmdTensor<<<BLOCK_NUMBER(LAYER * N * N), BLOCK_SIZE>>>(
|
||||
dmdMap.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
|
||||
wireDmdMap.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
|
||||
viaDmdMap.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
|
||||
capacity, wires, vias, fixed, N, LAYER, xSize, ySize, DIRECTION);
|
||||
return {dmdMap, wireDmdMap, viaDmdMap};
|
||||
}
|
||||
|
||||
torch::Tensor GPURouter::getCapacityMap() {
|
||||
int xSize = X;
|
||||
int ySize = Y;
|
||||
torch::Tensor capMap = torch::zeros({LAYER, xSize, ySize},
|
||||
torch::dtype(torch::kFloat32).device(torch::Device(torch::kCUDA, DEVICE_ID)));
|
||||
getCapTensor<<<BLOCK_NUMBER(LAYER * N * N), BLOCK_SIZE>>>(
|
||||
capMap.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
|
||||
capacity, wires, vias, fixed, N, LAYER, xSize, ySize, DIRECTION);
|
||||
return capMap;
|
||||
}
|
||||
|
||||
torch::Tensor GPURouter::calcRouteGrad(torch::Tensor mask_map,
|
||||
torch::Tensor wire_dmd_map_2d,
|
||||
torch::Tensor via_dmd_map_2d,
|
||||
torch::Tensor cap_map_2d,
|
||||
torch::Tensor dist_weights,
|
||||
torch::Tensor wirelength_weights,
|
||||
torch::Tensor route_gradmat,
|
||||
torch::Tensor node2pin_list,
|
||||
torch::Tensor node2pin_list_end,
|
||||
float grad_weight,
|
||||
float unit_wire_cost,
|
||||
float unit_via_cost,
|
||||
int num_nodes) {
|
||||
// 1. compute demand map and capacity map
|
||||
// 2. compute route_gradmat for each gcell (use torchDCT)
|
||||
|
||||
// 3. compute route force for each global pin
|
||||
torch::Tensor gbpin_grad = torch::zeros({numGbPin, 2},
|
||||
torch::dtype(torch::kFloat32).device(torch::Device(torch::kCUDA, DEVICE_ID)));
|
||||
compGcellRouteForce<<<BLOCK_NUMBER(numGbPin), BLOCK_SIZE>>>(
|
||||
gbpin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
mask_map.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
wire_dmd_map_2d.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
via_dmd_map_2d.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
cap_map_2d.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
dist_weights.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
wirelength_weights.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
route_gradmat.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
|
||||
grad_weight, unit_wire_cost, unit_via_cost,
|
||||
gbpinRoutes, gbpin2netId, routes, routesOffset,
|
||||
numGbPin, N, LAYER, X, Y, DIRECTION
|
||||
);
|
||||
|
||||
// 4. assign route force to placement pins (node's pins)
|
||||
torch::Tensor plpin_grad = torch::zeros({numPlPin, 2},
|
||||
torch::dtype(torch::kFloat32).device(torch::Device(torch::kCUDA, DEVICE_ID)));
|
||||
assignRouteForceToPlPin<<<BLOCK_NUMBER(numPlPin), BLOCK_SIZE>>>(
|
||||
plpin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
gbpin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
plPinId2gbPinId, numPlPin
|
||||
);
|
||||
|
||||
// 5. calc node grad (use deterministic summation)
|
||||
auto node_grad = torch::zeros({num_nodes, 2}, torch::dtype(plpin_grad.dtype()).device(plpin_grad.device()));
|
||||
const int threads = 128;
|
||||
const int blocks = (num_nodes * 2 + threads - 1) / threads;
|
||||
calc_node_grad_deterministic_cuda_kernel<<<blocks, threads, 0>>>(
|
||||
node_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
plpin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
node2pin_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
node2pin_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
num_nodes);
|
||||
|
||||
return node_grad;
|
||||
}
|
||||
|
||||
torch::Tensor GPURouter::calcFillerRouteGrad(torch::Tensor filler_pos,
|
||||
torch::Tensor filler_size,
|
||||
torch::Tensor filler_weight,
|
||||
torch::Tensor expand_ratio,
|
||||
torch::Tensor grad_mat,
|
||||
float grad_weight,
|
||||
float unit_len_x,
|
||||
float unit_len_y,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_fillers) {
|
||||
int threads = 64;
|
||||
int blocks = (num_fillers + threads - 1) / threads;
|
||||
|
||||
auto filler_grad = torch::zeros({num_fillers, 2}, torch::dtype(filler_pos.dtype()).device(filler_pos.device()));
|
||||
|
||||
fillerRouteForce<<<blocks, threads, 0>>>(
|
||||
filler_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
filler_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
filler_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
expand_ratio.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
filler_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
grad_mat.data_ptr<float>(),
|
||||
grad_weight, unit_len_x, unit_len_y, num_bin_x, num_bin_y, num_fillers
|
||||
);
|
||||
|
||||
return filler_grad;
|
||||
}
|
||||
|
||||
torch::Tensor GPURouter::calcPseudoPinGrad(torch::Tensor node_pos, torch::Tensor pseudo_pin_pos, float gamma) {
|
||||
const auto num_nodes = node_pos.size(0);
|
||||
|
||||
const int threads = 128;
|
||||
const int blocks = (num_nodes * 2 + threads - 1) / threads;
|
||||
float inv_gamma = 1 / gamma;
|
||||
|
||||
auto node_grad = torch::zeros({num_nodes, 2}, torch::dtype(node_pos.dtype()).device(node_pos.device()));
|
||||
pseudoPinForce<<<blocks, threads, 0>>>(
|
||||
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
pseudo_pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
node_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
num_nodes, inv_gamma
|
||||
);
|
||||
|
||||
return node_grad;
|
||||
}
|
||||
|
||||
torch::Tensor GPURouter::calcNodeInflateRatio(torch::Tensor node_pos,
|
||||
torch::Tensor node_size,
|
||||
torch::Tensor node_weight,
|
||||
torch::Tensor expand_ratio,
|
||||
torch::Tensor inflate_mat,
|
||||
float grad_weight,
|
||||
float unit_len_x,
|
||||
float unit_len_y,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
bool use_weighted_inflation) {
|
||||
const auto num_nodes = node_pos.size(0);
|
||||
|
||||
const int threads = 128;
|
||||
const int blocks = (num_nodes + threads - 1) / threads;
|
||||
|
||||
auto node_inflate_ratio = torch::ones({num_nodes, 2}, torch::dtype(node_pos.dtype()).device(node_pos.device()));
|
||||
if (use_weighted_inflation) {
|
||||
inflateNodeRatioWeighted<<<blocks, threads, 0>>>(
|
||||
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
node_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
node_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
expand_ratio.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
inflate_mat.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
|
||||
node_inflate_ratio.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
grad_weight, unit_len_x, unit_len_y, num_bin_x, num_bin_y, num_nodes
|
||||
);
|
||||
} else {
|
||||
inflateNodeRatioMax<<<blocks, threads, 0>>>(
|
||||
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
node_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
node_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
expand_ratio.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||
inflate_mat.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
|
||||
node_inflate_ratio.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
grad_weight, unit_len_x, unit_len_y, num_bin_x, num_bin_y, num_nodes
|
||||
);
|
||||
}
|
||||
|
||||
return node_inflate_ratio;
|
||||
}
|
||||
|
||||
torch::Tensor GPURouter::calcInflatedPinRelCpos(torch::Tensor node_inflate_ratio,
|
||||
torch::Tensor old_pin_rel_cpos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
int num_movable_nodes) {
|
||||
const auto num_pins = old_pin_rel_cpos.size(0);
|
||||
|
||||
const int threads = 128;
|
||||
const int blocks = (num_pins + threads - 1) / threads;
|
||||
|
||||
auto new_pin_rel_cpos = old_pin_rel_cpos.clone();
|
||||
inflatePinRelCpos<<<blocks, threads, 0>>>(
|
||||
node_inflate_ratio.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
old_pin_rel_cpos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
pin_id2node_id.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
new_pin_rel_cpos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
num_movable_nodes, num_pins
|
||||
);
|
||||
|
||||
return new_pin_rel_cpos;
|
||||
}
|
||||
|
||||
} // namespace gr
|
||||
41
cpp_to_py/gpugr/gr/InCellUsage.cuh
Normal file
41
cpp_to_py/gpugr/gr/InCellUsage.cuh
Normal file
@ -0,0 +1,41 @@
|
||||
#pragma once
|
||||
#include "common/common.h"
|
||||
|
||||
namespace gr {
|
||||
|
||||
__device__ __forceinline__ float myExp(float x) { return (1 << min(30, static_cast<int>(x))); }
|
||||
|
||||
__device__ __forceinline__ int inCellViaUsage(int idx, int *vias, int N, int LAYER) {
|
||||
int layer = idx / N / N - 1, y = idx / N % N, x = idx % N;
|
||||
int ans = 0;
|
||||
if (layer + 2 < LAYER) ans += vias[idx]; // a via from botLayer to currlayer
|
||||
if (layer >= 0) ans += vias[layer * N * N + x * N + y]; // a via from currlayer to topLayer
|
||||
return ans;
|
||||
}
|
||||
|
||||
__device__ __forceinline__ float twoCellsViaUsage(int idx, int *vias, int N, int LAYER) {
|
||||
return sqrt(0.5 * (inCellViaUsage(idx, vias, N, LAYER) + inCellViaUsage(idx + 1, vias, N, LAYER))) * 1.5;
|
||||
}
|
||||
|
||||
__device__ __forceinline__ float inCellUsedArea(int idx, int *wires, float *fixed, int N) {
|
||||
float ans = 0;
|
||||
if (idx % N > 0) ans += fixed[idx - 1] + wires[idx - 1];
|
||||
if (idx % N + 1 < N) ans += fixed[idx] + wires[idx];
|
||||
return ans / 2;
|
||||
}
|
||||
|
||||
__device__ __forceinline__ float inCellViaCost(
|
||||
int idx, int *wires, float *fixed, const float *capacity, float logisticSlope, int N) {
|
||||
return 1.0 / (1.0 + myExp(logisticSlope * (capacity[idx] - inCellUsedArea(idx, wires, fixed, N))));
|
||||
}
|
||||
|
||||
__device__ __forceinline__ float cellResource(
|
||||
int idx, int *wires, float *fixed, int *vias, const float *capacity, int N, int LAYER) {
|
||||
float ans = wires[idx] + fixed[idx];
|
||||
if (idx % N) ans += wires[idx - 1] + fixed[idx - 1];
|
||||
ans /= 2;
|
||||
ans += sqrt(1.0 * inCellViaUsage(idx, vias, N, LAYER)) * 1.5;
|
||||
return capacity[idx] - ans;
|
||||
}
|
||||
|
||||
} // namespace gr
|
||||
1061
cpp_to_py/gpugr/gr/MazeRoute.cu
Normal file
1061
cpp_to_py/gpugr/gr/MazeRoute.cu
Normal file
File diff suppressed because it is too large
Load Diff
50
cpp_to_py/gpugr/gr/MazeRoute.h
Normal file
50
cpp_to_py/gpugr/gr/MazeRoute.h
Normal file
@ -0,0 +1,50 @@
|
||||
#pragma once
|
||||
#include <algorithm>
|
||||
#include <cassert>
|
||||
#include <cstdio>
|
||||
#include <iostream>
|
||||
#include <vector>
|
||||
|
||||
namespace gr {
|
||||
|
||||
extern const int MAX_LAYER_NUM, MAX_PIN_NUM;
|
||||
|
||||
class GPUMazeRouter {
|
||||
public:
|
||||
void startGPU(int device_id, int layer, int x, int y);
|
||||
void endGPU();
|
||||
void query();
|
||||
void run(int DIRECTION, int iterleft);
|
||||
|
||||
void getResults(double &t,
|
||||
int64_t *costSum,
|
||||
int *pins,
|
||||
int *dist,
|
||||
int *fgprev,
|
||||
int *wireCost,
|
||||
int *viaCost,
|
||||
int *wires,
|
||||
int *vias,
|
||||
int *routes,
|
||||
int N,
|
||||
int SCALE,
|
||||
int DIRECTION,
|
||||
int routesOffset,
|
||||
int netId);
|
||||
|
||||
const static int MAX_TOT_PIN_NUM = 1000000, MAX_NUM_NET = 100;
|
||||
|
||||
int MAX_TURN_NUM = 10;
|
||||
int LAYER, X, Y, NX, NY; // 0: x-1,x,x+1... 1: y-1,y,y+1...
|
||||
int *costMap, *viaMap, *markMap, *cudaRoutedPin;
|
||||
|
||||
int *cost, *via, *costL, *costR, *cudaMap, *reset;
|
||||
int *cudaPrev;
|
||||
const int MAX_PIN_SIZE_PER_NET = 500000, MAX_PIN_NUM = 10000;
|
||||
|
||||
int firstTime = 0; // varible to determine whether the first run of TF
|
||||
};
|
||||
extern int counter1, counter2, counter3, counter4;
|
||||
extern double time1;
|
||||
|
||||
} // namespace gr
|
||||
552
cpp_to_py/gpugr/gr/PatternRoute.cpp
Normal file
552
cpp_to_py/gpugr/gr/PatternRoute.cpp
Normal file
@ -0,0 +1,552 @@
|
||||
#include "PatternRoute.h"
|
||||
|
||||
#include <iostream>
|
||||
#include <set>
|
||||
|
||||
#include "common/db/Database.h"
|
||||
#include "common/utils/robin_hood.h"
|
||||
#include "flute.h"
|
||||
|
||||
namespace gr {
|
||||
|
||||
using namespace Flute;
|
||||
|
||||
void prepareSingeNet(gr::GrNet &grNet, int routesOffset, int X, int Y, int N, int LAYER, int DIRECTION) {
|
||||
const std::vector<std::vector<int>> &pins = grNet.getPins();
|
||||
std::vector<int> &points = grNet.points;
|
||||
points.clear();
|
||||
|
||||
robin_hood::unordered_map<int, std::vector<int>> loc2Pins;
|
||||
// double startTimer = clock();
|
||||
std::vector<int> xpos(pins.size()), ypos(pins.size());
|
||||
for (int i = 0; i < pins.size(); i++) {
|
||||
int layer = pins[i][0] / N / N, _x = pins[i][0] / N % N, _y = pins[i][0] % N;
|
||||
if (!(layer & 1) ^ DIRECTION) std::swap(_x, _y);
|
||||
xpos[i] = _x;
|
||||
ypos[i] = _y;
|
||||
loc2Pins[_x * N + _y].emplace_back(layer);
|
||||
}
|
||||
|
||||
std::sort(xpos.begin(), xpos.end());
|
||||
std::sort(ypos.begin(), ypos.end());
|
||||
xpos.erase(std::unique(xpos.begin(), xpos.end()), xpos.end());
|
||||
ypos.erase(std::unique(ypos.begin(), ypos.end()), ypos.end());
|
||||
int degree = loc2Pins.size(), cur = 0;
|
||||
if (degree == 0) std::cerr << "ERROR: degree 0" << std::endl;
|
||||
// const int MAX_DEGREE = 100000;
|
||||
// if (degree > MAX_DEGREE) std::cerr << "Not Enough X and Y in Pattern Routing" << std::endl;
|
||||
|
||||
int x[degree * 4], y[degree * 4];
|
||||
for (auto e : loc2Pins) x[cur] = e.first / N, y[cur] = e.first % N, cur++;
|
||||
|
||||
Tree flutetree = flute(degree, x, y, 3);
|
||||
|
||||
robin_hood::unordered_map<int, int> loc2node, node2loc;
|
||||
std::set<int> locations;
|
||||
int node_cnt = 0;
|
||||
for (int i = 0; i < degree * 2 - 2; i++) locations.insert(flutetree.branch[i].x * N + flutetree.branch[i].y);
|
||||
for (auto e : locations) node2loc[loc2node[e] = node_cnt++] = e;
|
||||
std::vector<robin_hood::unordered_set<int>> graph(node_cnt);
|
||||
|
||||
std::vector<std::vector<int>> cntx(xpos.size(), std::vector<int>(ypos.size(), 0));
|
||||
std::vector<std::vector<int>> cnty(xpos.size(), std::vector<int>(ypos.size(), 0));
|
||||
std::vector<std::vector<int>> idx(xpos.size(), std::vector<int>(ypos.size(), -1));
|
||||
for (auto e : loc2node) {
|
||||
int x = std::lower_bound(xpos.begin(), xpos.end(), e.first / N) - xpos.begin();
|
||||
int y = std::lower_bound(ypos.begin(), ypos.end(), e.first % N) - ypos.begin();
|
||||
// printf("%d %d -> %d\n", x, y, e.second);
|
||||
idx[x][y] = e.second;
|
||||
}
|
||||
|
||||
for (int i = 0; i < degree * 2 - 2; i++) {
|
||||
Branch &branch1 = flutetree.branch[i], &branch2 = flutetree.branch[branch1.n];
|
||||
int id1 = loc2node[branch1.x * N + branch1.y], id2 = loc2node[branch2.x * N + branch2.y];
|
||||
if (id1 == id2) continue;
|
||||
int x1 = node2loc[id1] / N, y1 = node2loc[id1] % N;
|
||||
int x2 = node2loc[id2] / N, y2 = node2loc[id2] % N;
|
||||
// printf("%d %d %d %d\n", x1, y1, x2, y2);
|
||||
x1 = std::lower_bound(xpos.begin(), xpos.end(), x1) - xpos.begin();
|
||||
x2 = std::lower_bound(xpos.begin(), xpos.end(), x2) - xpos.begin();
|
||||
y1 = std::lower_bound(ypos.begin(), ypos.end(), y1) - ypos.begin();
|
||||
y2 = std::lower_bound(ypos.begin(), ypos.end(), y2) - ypos.begin();
|
||||
// printf("%d %d %d %d\n", x1, y1, x2, y2);
|
||||
if (x1 != x2 && y1 != y2) {
|
||||
graph[id1].insert(id2), graph[id2].insert(id1);
|
||||
/* if(locations.count(x1 * N + y2) || locations.count(x2 * N + y1))
|
||||
std::cerr << "ERROR & ERROR: BAD FLUTE RESULTS\n";
|
||||
for(int i = std::min(x1, x2); i <= std::max(x1, x2); i++)
|
||||
if(locations.count(i * N + y1) || locations.count(i * N + y2))
|
||||
std::cerr << "ERROR: BAD FLUTE RESULTS\n";
|
||||
for(int i = std::min(y1, y2); i <= std::max(y1, y2); i++)
|
||||
if(locations.count(x1 * N + i) || locations.count(x2 * N + i))
|
||||
std::cerr << "ERROR: BAD FLUTE RESULTS\n";
|
||||
*/
|
||||
} else {
|
||||
if (x1 == x2)
|
||||
for (int t = std::min(y1, y2); t < std::max(y1, y2); t++) cnty[x1][t]++;
|
||||
else
|
||||
for (int t = std::min(x1, x2); t < std::max(x1, x2); t++) cntx[t][y2]++;
|
||||
}
|
||||
}
|
||||
free(flutetree.branch);
|
||||
// printf("cnt = %d, %d %d\n", cntx[0][0], idx[0][0], idx[1][0]);
|
||||
for (int i = 0; i < xpos.size(); i++) {
|
||||
int last = -1;
|
||||
for (int j = 0; j < (int)ypos.size(); j++) {
|
||||
if (j && cnty[i][j - 1] == 0) last = -1;
|
||||
int cur = -1;
|
||||
if (idx[i][j] >= 0) cur = idx[i][j];
|
||||
if (cur >= 0 && last >= 0 && cur != last) graph[cur].insert(last), graph[last].insert(cur);
|
||||
if (cur >= 0) last = cur;
|
||||
}
|
||||
}
|
||||
for (int i = 0; i < ypos.size(); i++) {
|
||||
int last = -1;
|
||||
for (int j = 0; j < (int)xpos.size(); j++) {
|
||||
if (j && cntx[j - 1][i] == 0) last = -1;
|
||||
int cur = -1;
|
||||
if (idx[j][i] >= 0) cur = idx[j][i];
|
||||
if (cur >= 0 && last >= 0 && cur != last) graph[cur].insert(last), graph[last].insert(cur);
|
||||
// printf("i=%d, j=%d, cur=%d,last=%d\n", i, j, cur, last);
|
||||
if (cur >= 0) last = cur;
|
||||
}
|
||||
}
|
||||
std::vector<int> vis(node_cnt, 0);
|
||||
for (int i = 0; i < node_cnt; i++) {
|
||||
if (graph[i].size() > 4) {
|
||||
std::cerr << "ERROR in FLUTE Results\n";
|
||||
exit(-1);
|
||||
}
|
||||
}
|
||||
points.emplace_back(routesOffset);
|
||||
points.emplace_back(node_cnt);
|
||||
int len = 0;
|
||||
// for each point in points
|
||||
// points[0]: location
|
||||
// points[1, 2]: min and max layer
|
||||
// points[3, 4, 5, 6] children locations
|
||||
|
||||
std::function<void(int)> dfs = [&](int x) {
|
||||
vis[x] = 1;
|
||||
int startlen = len;
|
||||
points.emplace_back(node2loc[x]);
|
||||
len++;
|
||||
if (loc2Pins.count(node2loc[x])) {
|
||||
auto temp = loc2Pins[node2loc[x]];
|
||||
points.emplace_back(*std::min_element(temp.begin(), temp.end()));
|
||||
points.emplace_back(*std::max_element(temp.begin(), temp.end()));
|
||||
len += 2;
|
||||
} else {
|
||||
points.emplace_back(-1);
|
||||
points.emplace_back(-1);
|
||||
len += 2;
|
||||
}
|
||||
for (auto e : graph[x]) {
|
||||
if (!vis[e]) {
|
||||
points.emplace_back(node2loc[e]);
|
||||
len++;
|
||||
}
|
||||
}
|
||||
if (len - startlen > 6) {
|
||||
printf(" %d ERROR in len\n", (int)graph[x].size());
|
||||
}
|
||||
while (len % 6 != 0) {
|
||||
points.emplace_back(-1);
|
||||
len++;
|
||||
}
|
||||
for (auto e : graph[x]) {
|
||||
if (!vis[e]) dfs(e);
|
||||
}
|
||||
};
|
||||
|
||||
dfs(0);
|
||||
if (len != 6 * node_cnt) {
|
||||
for (int i = 0; i < pins.size(); i++) {
|
||||
int layer = pins[i][0] / N / N, _x = pins[i][0] / N % N, _y = pins[i][0] % N;
|
||||
if (!(layer & 1) ^ DIRECTION) std::swap(_x, _y);
|
||||
printf("(%d, %d)\n", _x, _y);
|
||||
}
|
||||
for (auto e : loc2node) printf("(%d, %d)_%d ", e.first / N, e.first % N, e.second);
|
||||
puts("");
|
||||
for (int i = 0; i < node_cnt; i++)
|
||||
for (auto e : graph[i]) printf("E(%d, %d) ", i, e);
|
||||
puts("");
|
||||
std::cerr << "ERROR in pattern routing preparation" << std::endl;
|
||||
}
|
||||
}
|
||||
|
||||
void prepareGrNets(std::vector<gr::GrNet> &grNets,
|
||||
std::vector<int> &netsToRoute,
|
||||
std::vector<int> &batchSizes,
|
||||
std::vector<std::vector<int>> &points_cpu_vec,
|
||||
std::vector<std::tuple<int, int, int>> &batchId2vec_info,
|
||||
int *routesOffsetCPU,
|
||||
int X,
|
||||
int Y,
|
||||
int N,
|
||||
int LAYER,
|
||||
int DIRECTION) {
|
||||
// FIXME: the vanilla version of FLUTE cannot support multi-threads. If need MT, please
|
||||
// change the FLUTE to the version in https://github.com/The-OpenROAD-Project-Attic/flute3
|
||||
int totalBs = 0;
|
||||
int numThreads = 1; // multi-thread is not supported by FLUTE
|
||||
for (int batchSize : batchSizes) {
|
||||
totalBs += batchSize;
|
||||
}
|
||||
auto thread_func = [&](int threadIdx) {
|
||||
for (int i = threadIdx; i < totalBs; i += numThreads) {
|
||||
int netId = netsToRoute[i];
|
||||
prepareSingeNet(grNets[netId], routesOffsetCPU[netId], X, Y, N, LAYER, DIRECTION);
|
||||
}
|
||||
};
|
||||
std::thread threads[numThreads];
|
||||
for (int j = 0; j < numThreads; j++) {
|
||||
threads[j] = std::thread(thread_func, j);
|
||||
}
|
||||
for (auto &t : threads) {
|
||||
t.join();
|
||||
}
|
||||
logger.info("Finish Flute %d");
|
||||
|
||||
points_cpu_vec.clear();
|
||||
batchId2vec_info.clear();
|
||||
|
||||
batchId2vec_info.resize(batchSizes.size());
|
||||
int startpos = 0;
|
||||
constexpr int MAX_POINTS_SIZE = 20000000;
|
||||
points_cpu_vec.push_back(std::vector<int>());
|
||||
points_cpu_vec.back().reserve(MAX_POINTS_SIZE);
|
||||
for (int batchId = 0; batchId < batchSizes.size(); batchId++) {
|
||||
int batchSize = batchSizes[batchId];
|
||||
if (batchSize == 0) continue;
|
||||
int offset = batchSize;
|
||||
std::vector<int> curBatch_points(batchSize, -1);
|
||||
for (int i = 0; i < batchSize; i++) {
|
||||
// the first batchSize elements indicate the offset
|
||||
curBatch_points[i] = offset;
|
||||
int netId = netsToRoute[startpos + i];
|
||||
offset += grNets[netId].points.size();
|
||||
// std::cout << batchSize << " " << i << " BigVecId " << points_cpu_vec.size() << " " <<
|
||||
// curBatch_points.size() << " " << points_cpu_vec.back().size() << " " << grNets[netId].getPins().size() <<
|
||||
// " " << grNets[netId].points.size() << std::endl;
|
||||
curBatch_points.insert(curBatch_points.end(),
|
||||
std::make_move_iterator(grNets[netId].points.begin()),
|
||||
std::make_move_iterator(grNets[netId].points.end()));
|
||||
}
|
||||
int startIdx, endIdx, inBigVecId;
|
||||
if (points_cpu_vec.back().size() + curBatch_points.size() < MAX_POINTS_SIZE) {
|
||||
startIdx = points_cpu_vec.back().size();
|
||||
endIdx = startIdx + curBatch_points.size();
|
||||
auto &tmp = points_cpu_vec.back();
|
||||
tmp.insert(tmp.end(),
|
||||
std::make_move_iterator(curBatch_points.begin()),
|
||||
std::make_move_iterator(curBatch_points.end()));
|
||||
inBigVecId = points_cpu_vec.size() - 1;
|
||||
} else {
|
||||
startIdx = 0;
|
||||
endIdx = curBatch_points.size();
|
||||
points_cpu_vec.emplace_back(std::move(curBatch_points));
|
||||
points_cpu_vec.back().reserve(MAX_POINTS_SIZE);
|
||||
inBigVecId = points_cpu_vec.size() - 1;
|
||||
}
|
||||
batchId2vec_info[batchId] = {inBigVecId, startIdx, endIdx};
|
||||
startpos += batchSize;
|
||||
}
|
||||
logger.info("#BigVec %d", points_cpu_vec.size());
|
||||
}
|
||||
|
||||
int prepare(double &count,
|
||||
gr::GrNet &grNet,
|
||||
int *points,
|
||||
int routesOffset,
|
||||
int *gbpoints,
|
||||
int &gbPinOffset,
|
||||
int X,
|
||||
int Y,
|
||||
int N,
|
||||
int LAYER,
|
||||
int DIRECTION) {
|
||||
auto &pins = grNet.getPins();
|
||||
// std::map<int, std::vector<int>> loc2Pins;
|
||||
robin_hood::unordered_map<int, std::vector<int>> loc2Pins;
|
||||
robin_hood::unordered_map<int, int> loc2pinIds;
|
||||
// double startTimer = clock();
|
||||
std::vector<int> xpos(pins.size()), ypos(pins.size());
|
||||
for (int i = 0; i < pins.size(); i++) {
|
||||
int layer = pins[i][0] / N / N, _x = pins[i][0] / N % N, _y = pins[i][0] % N;
|
||||
if (!(layer & 1) ^ DIRECTION) std::swap(_x, _y);
|
||||
xpos[i] = _x;
|
||||
ypos[i] = _y;
|
||||
auto& loc2PinsLayerVec = loc2Pins[_x * N + _y];
|
||||
loc2PinsLayerVec.emplace_back(layer);
|
||||
loc2pinIds[_x * N + _y] = i;
|
||||
// if(pins[i].size() > 1) {
|
||||
// int layer = pins[i][pins[i].size() - 1] / N / N;
|
||||
// loc2PinsLayerVec.emplace_back(layer);
|
||||
// }
|
||||
}
|
||||
std::sort(xpos.begin(), xpos.end());
|
||||
std::sort(ypos.begin(), ypos.end());
|
||||
xpos.erase(unique(xpos.begin(), xpos.end()), xpos.end());
|
||||
ypos.erase(unique(ypos.begin(), ypos.end()), ypos.end());
|
||||
int degree = loc2Pins.size(), cur = 0;
|
||||
if (degree == 0) std::cerr << "ERROR: degree 0" << std::endl;
|
||||
constexpr int MAX_DEGREE = 100000;
|
||||
if (degree > MAX_DEGREE) std::cerr << "Not Enough X and Y in Pattern Routing" << std::endl;
|
||||
int x[degree * 4], y[degree * 4];
|
||||
for (auto e : loc2Pins) x[cur] = e.first / N, y[cur] = e.first % N, cur++;
|
||||
|
||||
Tree flutetree = flute(degree, x, y, 3);
|
||||
// count += clock() - startTimer;
|
||||
robin_hood::unordered_map<int, int> loc2node, node2loc;
|
||||
std::set<int> locations;
|
||||
int node_cnt = 0;
|
||||
for (int i = 0; i < degree * 2 - 2; i++) locations.insert(flutetree.branch[i].x * N + flutetree.branch[i].y);
|
||||
for (auto e : locations) {
|
||||
node2loc[loc2node[e] = node_cnt++] = e;
|
||||
if (!loc2pinIds.contains(e)) {
|
||||
// e is not a real pin position but is a pseudo pin generated by RSMT
|
||||
loc2pinIds[e] = -1;
|
||||
}
|
||||
}
|
||||
// std::vector<std::set<int>> graph(node_cnt);
|
||||
std::vector<robin_hood::unordered_set<int>> graph(node_cnt);
|
||||
|
||||
std::vector<std::vector<int>> cntx(xpos.size(), std::vector<int>(ypos.size(), 0));
|
||||
std::vector<std::vector<int>> cnty(xpos.size(), std::vector<int>(ypos.size(), 0));
|
||||
std::vector<std::vector<int>> idx(xpos.size(), std::vector<int>(ypos.size(), -1));
|
||||
for (auto e : loc2node) {
|
||||
int x = lower_bound(xpos.begin(), xpos.end(), e.first / N) - xpos.begin();
|
||||
int y = lower_bound(ypos.begin(), ypos.end(), e.first % N) - ypos.begin();
|
||||
// printf("%d %d -> %d\n", x, y, e.second);
|
||||
idx[x][y] = e.second;
|
||||
}
|
||||
|
||||
for (int i = 0; i < degree * 2 - 2; i++) {
|
||||
Branch &branch1 = flutetree.branch[i], &branch2 = flutetree.branch[branch1.n];
|
||||
int id1 = loc2node[branch1.x * N + branch1.y], id2 = loc2node[branch2.x * N + branch2.y];
|
||||
if (id1 == id2) continue;
|
||||
int x1 = node2loc[id1] / N, y1 = node2loc[id1] % N;
|
||||
int x2 = node2loc[id2] / N, y2 = node2loc[id2] % N;
|
||||
// printf("%d %d %d %d\n", x1, y1, x2, y2);
|
||||
x1 = lower_bound(xpos.begin(), xpos.end(), x1) - xpos.begin();
|
||||
x2 = lower_bound(xpos.begin(), xpos.end(), x2) - xpos.begin();
|
||||
y1 = lower_bound(ypos.begin(), ypos.end(), y1) - ypos.begin();
|
||||
y2 = lower_bound(ypos.begin(), ypos.end(), y2) - ypos.begin();
|
||||
// printf("%d %d %d %d\n", x1, y1, x2, y2);
|
||||
if (x1 != x2 && y1 != y2) {
|
||||
graph[id1].insert(id2), graph[id2].insert(id1);
|
||||
/* if(locations.count(x1 * N + y2) || locations.count(x2 * N + y1))
|
||||
std::cerr << "ERROR & ERROR: BAD FLUTE RESULTS\n";
|
||||
for(int i = min(x1, x2); i <= max(x1, x2); i++)
|
||||
if(locations.count(i * N + y1) || locations.count(i * N + y2))
|
||||
std::cerr << "ERROR: BAD FLUTE RESULTS\n";
|
||||
for(int i = min(y1, y2); i <= max(y1, y2); i++)
|
||||
if(locations.count(x1 * N + i) || locations.count(x2 * N + i))
|
||||
std::cerr << "ERROR: BAD FLUTE RESULTS\n";
|
||||
*/
|
||||
} else {
|
||||
if (x1 == x2)
|
||||
for (int t = min(y1, y2); t < max(y1, y2); t++) cnty[x1][t]++;
|
||||
else
|
||||
for (int t = min(x1, x2); t < max(x1, x2); t++) cntx[t][y2]++;
|
||||
}
|
||||
}
|
||||
free(flutetree.branch);
|
||||
// NOTE: When a_x < b_x < c_x and a_y == b_y == c_y, FLUTE may report two edges A-B, A-C,
|
||||
// here we fix it to A-B, B-C
|
||||
// printf("cnt = %d, %d %d\n", cntx[0][0], idx[0][0], idx[1][0]);
|
||||
for (int i = 0; i < xpos.size(); i++) {
|
||||
int last = -1;
|
||||
for (int j = 0; j < (int)ypos.size(); j++) {
|
||||
if (j && cnty[i][j - 1] == 0) last = -1;
|
||||
int cur = -1;
|
||||
if (idx[i][j] >= 0) cur = idx[i][j];
|
||||
if (cur >= 0 && last >= 0 && cur != last) graph[cur].insert(last), graph[last].insert(cur);
|
||||
if (cur >= 0) last = cur;
|
||||
}
|
||||
}
|
||||
for (int i = 0; i < ypos.size(); i++) {
|
||||
int last = -1;
|
||||
for (int j = 0; j < (int)xpos.size(); j++) {
|
||||
if (j && cntx[j - 1][i] == 0) last = -1;
|
||||
int cur = -1;
|
||||
if (idx[j][i] >= 0) cur = idx[j][i];
|
||||
if (cur >= 0 && last >= 0 && cur != last) graph[cur].insert(last), graph[last].insert(cur);
|
||||
// printf("i=%d, j=%d, cur=%d,last=%d\n", i, j, cur, last);
|
||||
if (cur >= 0) last = cur;
|
||||
}
|
||||
}
|
||||
// NOTE: Fix corner cases, graph[i] includes 4 straight edges and >= 1 bevel edges
|
||||
for (int i = 0; i < node_cnt; i++) {
|
||||
if (graph[i].size() > 4) {
|
||||
std::vector<int> movedIds;
|
||||
int thisX = node2loc[i] / N, thisY = node2loc[i] % N;
|
||||
for (auto childId : graph[i]) {
|
||||
int childX = node2loc[childId] / N, childY = node2loc[childId] % N;
|
||||
if (childX != thisX && childY != thisY) {
|
||||
movedIds.emplace_back(childId);
|
||||
}
|
||||
}
|
||||
for (auto childId : movedIds) {
|
||||
std::queue<int> q;
|
||||
std::vector<bool> possibleSet(node_cnt, false);
|
||||
q.push(i);
|
||||
while (q.size() > 0) {
|
||||
int cur = q.front();
|
||||
q.pop();
|
||||
if (possibleSet[cur]) continue;
|
||||
possibleSet[cur] = true;
|
||||
for (auto c : graph[cur]) {
|
||||
if (!possibleSet[c] && c != childId) {
|
||||
q.push(c);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (graph[i].size() == 4) break;
|
||||
int childX = node2loc[childId] / N, childY = node2loc[childId] % N;
|
||||
int minDist = std::numeric_limits<int>::max();
|
||||
int new_i = -1;
|
||||
for (int j = 0; j < node_cnt; j++) {
|
||||
if (!possibleSet[j]) continue;
|
||||
if (j == i || j == childId) continue;
|
||||
if (graph[j].size() < 4) {
|
||||
int tarX = node2loc[j] / N, tarY = node2loc[j] % N;
|
||||
int dist = std::abs(tarX - childX) + std::abs(tarY - childY);
|
||||
if (dist == 0) continue;
|
||||
if (dist < minDist) {
|
||||
minDist = dist;
|
||||
new_i = j;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (new_i == -1) {
|
||||
continue;
|
||||
}
|
||||
graph[i].erase(childId);
|
||||
graph[childId].erase(i);
|
||||
graph[childId].insert(new_i);
|
||||
graph[new_i].insert(childId);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int i = 0; i < node_cnt; i++) {
|
||||
if (graph[i].size() > 4) {
|
||||
std::cerr << "ERROR in FLUTE Results\n";
|
||||
printf("(%d %d) Childs: ", node2loc[i] / N, node2loc[i] % N);
|
||||
for (auto e : graph[i]) {
|
||||
printf("(%d %d) ", node2loc[e] / N, node2loc[e] % N);
|
||||
}
|
||||
printf("\nAll pts: ");
|
||||
for (int j = 0; j < node_cnt; j++) {
|
||||
printf("(%d %d) ", node2loc[j] / N, node2loc[j] % N);
|
||||
}
|
||||
std::cout << std::endl;
|
||||
exit(-1);
|
||||
}
|
||||
}
|
||||
std::vector<int> vis(node_cnt, 0);
|
||||
points[0] = routesOffset;
|
||||
points[1] = node_cnt;
|
||||
int len = 0;
|
||||
points += 2;
|
||||
// points[0]: location
|
||||
// points[1, 2]: min and max layer
|
||||
// points[3, 4, 5] children locations
|
||||
std::function<void(int)> dfs = [&](int x) {
|
||||
vis[x] = 1;
|
||||
int startlen = len;
|
||||
// points[0]: location
|
||||
int loc = node2loc[x];
|
||||
points[len++] = loc;
|
||||
if (loc2Pins.count(loc)) {
|
||||
// points[1, 2]: min and max layer
|
||||
auto temp = loc2Pins[loc];
|
||||
points[len++] = *std::min_element(temp.begin(), temp.end());
|
||||
points[len++] = *std::max_element(temp.begin(), temp.end());
|
||||
} else
|
||||
points[len++] = -1, points[len++] = -1;
|
||||
|
||||
// points[3, 4, 5] children locations
|
||||
for (auto e : graph[x])
|
||||
if (!vis[e]) points[len++] = node2loc[e];
|
||||
if (len - startlen > 6) printf(" %d ERROR in len\n", (int)graph[x].size());
|
||||
while (len % 6 != 0) points[len++] = -1;
|
||||
for (auto e : graph[x])
|
||||
if (!vis[e]) dfs(e);
|
||||
};
|
||||
|
||||
dfs(0);
|
||||
if (len != 6 * node_cnt) {
|
||||
printf("len: %d node_cnt: %d\n", len, node_cnt);
|
||||
for (int i = 0; i < pins.size(); i++) {
|
||||
int layer = pins[i][0] / N / N, _x = pins[i][0] / N % N, _y = pins[i][0] % N;
|
||||
if (!(layer & 1) ^ DIRECTION) std::swap(_x, _y);
|
||||
printf("(%d, %d)\n", _x, _y);
|
||||
}
|
||||
for (auto e : loc2node) printf("(%d, %d)_%d ", e.first / N, e.first % N, e.second);
|
||||
puts("");
|
||||
for (int i = 0; i < node_cnt; i++)
|
||||
for (auto e : graph[i]) printf("E(%d, %d) ", i, e);
|
||||
puts("");
|
||||
std::cerr << "ERROR in pattern routing preparation" << std::endl;
|
||||
}
|
||||
|
||||
// rewrite points child
|
||||
robin_hood::unordered_map<int, int> loc2point_id;
|
||||
for (int i = 0; i < node_cnt; i++) {
|
||||
loc2point_id[points[i * 6]] = i * 6;
|
||||
}
|
||||
for (int i = 0; i < node_cnt; i++) {
|
||||
for (int j = 3; j < 6; j++) {
|
||||
if (points[i * 6 + j] == -1) continue;
|
||||
points[i * 6 + j] = loc2point_id[points[i * 6 + j]];
|
||||
}
|
||||
}
|
||||
|
||||
// for (int i = 0; i < node_cnt; i++) {
|
||||
// int x = points[i * 6] / N, y = points[i * 6] % N;
|
||||
// if (x > grNet.upperx || x < grNet.lowerx || y > grNet.uppery || y < grNet.lowery) {
|
||||
// printf("pin: (%d, %d) out of boundary of net_bbox: (%d, %d, %d, %d)\n",
|
||||
// x, y, grNet.lowerx, grNet.lowery, grNet.upperx, grNet.uppery);
|
||||
// }
|
||||
// }
|
||||
|
||||
// gbpoints
|
||||
for (int i = 0; i < node_cnt; i++) {
|
||||
int pinId = loc2pinIds[points[i * 6]];
|
||||
if (pinId == -1) {
|
||||
gbpoints[i] = -1;
|
||||
} else {
|
||||
gbpoints[i] = grNet.pin2gbpinId[pinId];
|
||||
}
|
||||
}
|
||||
gbPinOffset += node_cnt;
|
||||
// if (node_cnt > pins.size() && node_cnt > 10) {
|
||||
// std::cout << "node_cnt " << node_cnt << ", #gbpins " << pins.size() << std::endl;
|
||||
// for (int i = 0; i < node_cnt; i++) {
|
||||
// std::cout << points[i * 6] << " ";
|
||||
// }
|
||||
// std::cout << std::endl;
|
||||
// for (int i = 0; i < node_cnt; i++) {
|
||||
// std::cout << gbpoints[i] << " ";
|
||||
// }
|
||||
// std::cout << std::endl;
|
||||
// for (int i = 0; i < node_cnt; i++) {
|
||||
// int loc = points[i * 6];
|
||||
// if (loc2pinIds[loc] != -1) {
|
||||
// int pinid = loc2pinIds[loc];
|
||||
// int layer = pins[pinid][0] / N / N, _x = pins[pinid][0] / N % N, _y = pins[pinid][0] % N;
|
||||
// std::cout << _x * N + _y << " ";
|
||||
// } else {
|
||||
// std::cout << "xxxxxx" << " ";
|
||||
// }
|
||||
// }
|
||||
// std::cout << std::endl;
|
||||
// exit(0);
|
||||
// }
|
||||
|
||||
return len + 2;
|
||||
}
|
||||
|
||||
} // namespace gr
|
||||
1215
cpp_to_py/gpugr/gr/PatternRoute.cu
Normal file
1215
cpp_to_py/gpugr/gr/PatternRoute.cu
Normal file
File diff suppressed because it is too large
Load Diff
49
cpp_to_py/gpugr/gr/PatternRoute.h
Normal file
49
cpp_to_py/gpugr/gr/PatternRoute.h
Normal file
@ -0,0 +1,49 @@
|
||||
#pragma once
|
||||
#include "common/common.h"
|
||||
#include "gpugr/db/GrNet.h"
|
||||
|
||||
namespace gr {
|
||||
|
||||
int prepare(double &count,
|
||||
gr::GrNet &grNet,
|
||||
int *points,
|
||||
int routesOffset,
|
||||
int *gbpoints,
|
||||
int &gbPinOffset,
|
||||
int X,
|
||||
int Y,
|
||||
int N,
|
||||
int LAYER,
|
||||
int DIRECTION);
|
||||
|
||||
void prepareSingeNet(gr::GrNet &grNet, int routesOffset, int X, int Y, int N, int LAYER, int DIRECTION);
|
||||
|
||||
void prepareGrNets(std::vector<gr::GrNet> &grNets,
|
||||
std::vector<int> &netsToRoute,
|
||||
std::vector<int> &batchSizes,
|
||||
std::vector<std::vector<int>> &points_cpu_vec,
|
||||
std::vector<std::tuple<int, int, int>> &batchId2vec_info,
|
||||
int *routesOffsetCPU,
|
||||
int X,
|
||||
int Y,
|
||||
int N,
|
||||
int LAYER,
|
||||
int DIRECTION);
|
||||
|
||||
void patternRoute(int *points,
|
||||
int batchSize,
|
||||
int64_t *wireCostSum,
|
||||
int *viaCost,
|
||||
int *map,
|
||||
int *prev,
|
||||
int *wires,
|
||||
int *vias,
|
||||
int *routes,
|
||||
int *gbpoints,
|
||||
int *gbpinRoutes,
|
||||
int X,
|
||||
int Y,
|
||||
int N,
|
||||
int LAYER,
|
||||
int DIRECTION);
|
||||
} // namespace gr
|
||||
170
cpp_to_py/gpugr/gr/RouteForce.cpp
Normal file
170
cpp_to_py/gpugr/gr/RouteForce.cpp
Normal file
@ -0,0 +1,170 @@
|
||||
#include "RouteForce.h"
|
||||
|
||||
namespace gr {
|
||||
|
||||
RouteForce::RouteForce(std::shared_ptr<gr::GRDatabase> grdb_) : grdb(*grdb_) {}
|
||||
|
||||
void RouteForce::run_ggr() {
|
||||
logger.enable_logger();
|
||||
utils::timer T_total;
|
||||
T_total.start();
|
||||
// we only need the PR segment, our current data structure unsupport MR route force
|
||||
// if rrrIters > 0, this router can only be used for congestion map computation or solution evaluation
|
||||
int runMazeRouteTimes = grSetting.rrrIters;
|
||||
// Parameters
|
||||
int rrrIterLimit = 1 + runMazeRouteTimes;
|
||||
double _unitWireCostRaw = 0.5 * grdb.microns / grdb.m2pitch;
|
||||
double _unitViaCostRaw = 4;
|
||||
double _unitViaCost = _unitViaCostRaw / _unitWireCostRaw * grdb.microns;
|
||||
double _unitShortVioCostRaw = 500;
|
||||
double rrrInitVioCostDiscount = 0.1;
|
||||
|
||||
router.initialize(grSetting.deviceId,
|
||||
grdb.nLayers,
|
||||
grdb.xSize,
|
||||
grdb.ySize,
|
||||
grdb.nMaxGrid,
|
||||
grdb.cgxsize,
|
||||
grdb.cgysize,
|
||||
grdb.m1direction,
|
||||
grdb.csrnScale);
|
||||
router.setMap(grdb.capacity, grdb.wireDist, grdb.fixedLength, grdb.fixedUsage);
|
||||
|
||||
std::vector<float> _unitShortVioCost(grdb.nLayers), _unitShortVioCostDiscounted(grdb.nLayers);
|
||||
router.setFromNets(grdb.grNets, grdb.gpdb.getPins().size());
|
||||
router.setUnitViaCost(_unitViaCost);
|
||||
for (int i = 0; i < grdb.nLayers; ++i) {
|
||||
_unitShortVioCost[i] =
|
||||
_unitShortVioCostRaw * grdb.layerWidth[i] * grdb.microns / grdb.m2pitch / grdb.m2pitch / _unitWireCostRaw;
|
||||
}
|
||||
double tot_time = 0;
|
||||
for (int iter = 0; iter < rrrIterLimit; iter++) {
|
||||
router.setLogisticSlope(1 << iter);
|
||||
router.setUnitVioCost(_unitShortVioCost, 0.1);
|
||||
if (iter == 0) {
|
||||
router.setUnitViaMultiplier(1);
|
||||
} else {
|
||||
router.setUnitViaMultiplier(max(100 / pow(5, iter - 1), 4.0));
|
||||
router.setUnitVioCost(_unitShortVioCost,
|
||||
rrrInitVioCostDiscount + (1.0 - rrrInitVioCostDiscount) / (rrrIterLimit - 1) * iter);
|
||||
}
|
||||
utils::timer T;
|
||||
T.start();
|
||||
router.route(grdb.grNets, iter);
|
||||
tot_time += T.elapsed();
|
||||
logger.info("##### GPU Routing Iter: %d Time: %.4f #####", iter, T.elapsed());
|
||||
// break;
|
||||
}
|
||||
router.setToNets(grdb.grNets);
|
||||
logger.info("Total GPU Routing time: %.4f", tot_time);
|
||||
|
||||
if (grSetting.routeGuideFile != "") {
|
||||
grdb.writeGuides(grSetting.routeGuideFile);
|
||||
}
|
||||
|
||||
logger.info("Total GPU GR Time: %.4f", T_total.elapsed());
|
||||
logger.reset_logger();
|
||||
}
|
||||
|
||||
std::tuple<torch::Tensor, torch::Tensor, torch::Tensor> RouteForce::getDemandMap() { return router.getDemandMap(); }
|
||||
|
||||
torch::Tensor RouteForce::getCapacityMap() { return router.getCapacityMap(); }
|
||||
|
||||
torch::Tensor RouteForce::calcRouteGrad(torch::Tensor mask_map,
|
||||
torch::Tensor wire_dmd_map_2d,
|
||||
torch::Tensor via_dmd_map_2d,
|
||||
torch::Tensor cap_map_2d,
|
||||
torch::Tensor dist_weights,
|
||||
torch::Tensor wirelength_weights,
|
||||
torch::Tensor route_gradmat,
|
||||
torch::Tensor node2pin_list,
|
||||
torch::Tensor node2pin_list_end,
|
||||
float grad_weight,
|
||||
float unit_wire_cost,
|
||||
float unit_via_cost,
|
||||
int num_nodes) {
|
||||
return router.calcRouteGrad(mask_map,
|
||||
wire_dmd_map_2d,
|
||||
via_dmd_map_2d,
|
||||
cap_map_2d,
|
||||
dist_weights,
|
||||
wirelength_weights,
|
||||
route_gradmat,
|
||||
node2pin_list,
|
||||
node2pin_list_end,
|
||||
grad_weight,
|
||||
unit_wire_cost,
|
||||
unit_via_cost,
|
||||
num_nodes);
|
||||
};
|
||||
|
||||
torch::Tensor RouteForce::calcFillerRouteGrad(torch::Tensor filler_pos,
|
||||
torch::Tensor filler_size,
|
||||
torch::Tensor filler_weight,
|
||||
torch::Tensor expand_ratio,
|
||||
torch::Tensor grad_mat,
|
||||
float grad_weight,
|
||||
float unit_len_x,
|
||||
float unit_len_y,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_fillers) {
|
||||
return router.calcFillerRouteGrad(filler_pos,
|
||||
filler_size,
|
||||
filler_weight,
|
||||
expand_ratio,
|
||||
grad_mat,
|
||||
grad_weight,
|
||||
unit_len_x,
|
||||
unit_len_y,
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
num_fillers);
|
||||
}
|
||||
|
||||
torch::Tensor RouteForce::calcPseudoPinGrad(torch::Tensor node_pos, torch::Tensor pseudo_pin_pos, float gamma) {
|
||||
return router.calcPseudoPinGrad(node_pos, pseudo_pin_pos, gamma);
|
||||
}
|
||||
|
||||
torch::Tensor RouteForce::calcNodeInflateRatio(torch::Tensor node_pos,
|
||||
torch::Tensor node_size,
|
||||
torch::Tensor node_weight,
|
||||
torch::Tensor expand_ratio,
|
||||
torch::Tensor inflate_mat,
|
||||
float grad_weight,
|
||||
float unit_len_x,
|
||||
float unit_len_y,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
bool use_weighted_inflation) {
|
||||
return router.calcNodeInflateRatio(node_pos,
|
||||
node_size,
|
||||
node_weight,
|
||||
expand_ratio,
|
||||
inflate_mat,
|
||||
grad_weight,
|
||||
unit_len_x,
|
||||
unit_len_y,
|
||||
num_bin_x,
|
||||
num_bin_y,
|
||||
use_weighted_inflation);
|
||||
}
|
||||
|
||||
torch::Tensor RouteForce::calcInflatedPinRelCpos(torch::Tensor node_inflate_ratio,
|
||||
torch::Tensor old_pin_rel_cpos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
int num_movable_conn_nodes) {
|
||||
return router.calcInflatedPinRelCpos(node_inflate_ratio, old_pin_rel_cpos, pin_id2node_id, num_movable_conn_nodes);
|
||||
}
|
||||
|
||||
int RouteForce::getNumOvflNets() { return router.getNumOvflNets(); }
|
||||
|
||||
int RouteForce::getMicrons() { return grdb.microns; }
|
||||
|
||||
std::tuple<int, int> RouteForce::getGcellStep() { return {grdb.mainGcellStepX, grdb.mainGcellStepY}; }
|
||||
|
||||
std::vector<int> RouteForce::getLayerPitch() { return grdb.layerPitch; }
|
||||
|
||||
std::vector<int> RouteForce::getLayerWidth() { return grdb.layerWidth; }
|
||||
|
||||
} // namespace gr
|
||||
68
cpp_to_py/gpugr/gr/RouteForce.h
Normal file
68
cpp_to_py/gpugr/gr/RouteForce.h
Normal file
@ -0,0 +1,68 @@
|
||||
#pragma once
|
||||
#include "common/common.h"
|
||||
#include "common/db/Database.h"
|
||||
#include "gpugr/db/GRDatabase.h"
|
||||
#include "gpugr/gr/GPURouter.h"
|
||||
|
||||
namespace gr {
|
||||
|
||||
class RouteForce {
|
||||
public:
|
||||
RouteForce(std::shared_ptr<gr::GRDatabase> grdb_);
|
||||
void run_ggr();
|
||||
|
||||
public:
|
||||
std::tuple<torch::Tensor, torch::Tensor, torch::Tensor> getDemandMap();
|
||||
torch::Tensor getCapacityMap();
|
||||
torch::Tensor calcRouteGrad(torch::Tensor mask_map,
|
||||
torch::Tensor wire_dmd_map_2d,
|
||||
torch::Tensor via_dmd_map_2d,
|
||||
torch::Tensor cap_map_2d,
|
||||
torch::Tensor dist_weights,
|
||||
torch::Tensor wirelength_weights,
|
||||
torch::Tensor route_gradmat,
|
||||
torch::Tensor node2pin_list,
|
||||
torch::Tensor node2pin_list_end,
|
||||
float grad_weight,
|
||||
float unit_wire_cost,
|
||||
float unit_via_cost,
|
||||
int num_nodes);
|
||||
torch::Tensor calcFillerRouteGrad(torch::Tensor filler_pos,
|
||||
torch::Tensor filler_size,
|
||||
torch::Tensor filler_weight,
|
||||
torch::Tensor expand_ratio,
|
||||
torch::Tensor grad_mat,
|
||||
float grad_weight,
|
||||
float unit_len_x,
|
||||
float unit_len_y,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
int num_fillers);
|
||||
torch::Tensor calcPseudoPinGrad(torch::Tensor node_pos, torch::Tensor pseudo_pin_pos, float gamma);
|
||||
torch::Tensor calcNodeInflateRatio(torch::Tensor node_pos,
|
||||
torch::Tensor node_size,
|
||||
torch::Tensor node_weight,
|
||||
torch::Tensor expand_ratio,
|
||||
torch::Tensor inflate_mat,
|
||||
float grad_weight,
|
||||
float unit_len_x,
|
||||
float unit_len_y,
|
||||
int num_bin_x,
|
||||
int num_bin_y,
|
||||
bool use_weighted_inflation);
|
||||
torch::Tensor calcInflatedPinRelCpos(torch::Tensor node_inflate_ratio,
|
||||
torch::Tensor old_pin_rel_cpos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
int num_movable_nodes);
|
||||
int getNumOvflNets();
|
||||
int getMicrons();
|
||||
std::tuple<int, int> getGcellStep();
|
||||
std::vector<int> getLayerPitch();
|
||||
std::vector<int> getLayerWidth();
|
||||
|
||||
private:
|
||||
gr::GRDatabase& grdb;
|
||||
gr::GPURouter router;
|
||||
};
|
||||
|
||||
} // namespace gr
|
||||
78
cpp_to_py/gpugr/taskflow/core/algorithm/critical.hpp
Normal file
78
cpp_to_py/gpugr/taskflow/core/algorithm/critical.hpp
Normal file
@ -0,0 +1,78 @@
|
||||
#pragma once
|
||||
|
||||
#include "../task.hpp"
|
||||
|
||||
/**
|
||||
@file critical.hpp
|
||||
@brief critical include file
|
||||
*/
|
||||
|
||||
namespace tf {
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// CriticalSection
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@class CriticalSection
|
||||
|
||||
@brief class to create a critical region of limited workers to run tasks
|
||||
|
||||
tf::CriticalSection is a warpper over tf::Semaphore and is specialized for
|
||||
limiting the maximum concurrency over a set of tasks.
|
||||
A critical section starts with an initial count representing that limit.
|
||||
When a task is added to the critical section,
|
||||
the task acquires and releases the semaphore internal to the critical section.
|
||||
This design avoids explicit call of tf::Task::acquire and tf::Task::release.
|
||||
The following example creates a critical section of one worker and adds
|
||||
the five tasks to the critical section.
|
||||
|
||||
@code{.cpp}
|
||||
tf::Executor executor(8); // create an executor of 8 workers
|
||||
tf::Taskflow taskflow;
|
||||
|
||||
// create a critical section of 1 worker
|
||||
tf::CriticalSection critical_section(1);
|
||||
|
||||
tf::Task A = taskflow.emplace([](){ std::cout << "A" << std::endl; });
|
||||
tf::Task B = taskflow.emplace([](){ std::cout << "B" << std::endl; });
|
||||
tf::Task C = taskflow.emplace([](){ std::cout << "C" << std::endl; });
|
||||
tf::Task D = taskflow.emplace([](){ std::cout << "D" << std::endl; });
|
||||
tf::Task E = taskflow.emplace([](){ std::cout << "E" << std::endl; });
|
||||
|
||||
critical_section.add(A, B, C, D, E);
|
||||
|
||||
executor.run(taskflow).wait();
|
||||
@endcode
|
||||
|
||||
*/
|
||||
class CriticalSection : public Semaphore {
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
@brief constructs a critical region of a limited number of workers
|
||||
*/
|
||||
explicit CriticalSection(int max_workers = 1);
|
||||
|
||||
/**
|
||||
@brief adds a task into the critical region
|
||||
*/
|
||||
template <typename... Tasks>
|
||||
void add(Tasks...tasks);
|
||||
};
|
||||
|
||||
inline CriticalSection::CriticalSection(int max_workers) :
|
||||
Semaphore {max_workers} {
|
||||
}
|
||||
|
||||
template <typename... Tasks>
|
||||
void CriticalSection::add(Tasks... tasks) {
|
||||
(tasks.acquire(*this), ...);
|
||||
(tasks.release(*this), ...);
|
||||
}
|
||||
|
||||
|
||||
} // end of namespace tf. ---------------------------------------------------
|
||||
|
||||
|
||||
203
cpp_to_py/gpugr/taskflow/core/algorithm/for_each.hpp
Normal file
203
cpp_to_py/gpugr/taskflow/core/algorithm/for_each.hpp
Normal file
@ -0,0 +1,203 @@
|
||||
// reference:
|
||||
// - gomp: https://github.com/gcc-mirror/gcc/blob/master/libgomp/iter.c
|
||||
// - komp: https://github.com/llvm-mirror/openmp/blob/master/runtime/src/kmp_dispatch.cpp
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "../executor.hpp"
|
||||
|
||||
namespace tf {
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// default parallel for
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Function: for_each
|
||||
template <typename B, typename E, typename C>
|
||||
Task FlowBuilder::for_each(B&& beg, E&& end, C c) {
|
||||
|
||||
using I = stateful_iterator_t<B, E>;
|
||||
using namespace std::string_literals;
|
||||
|
||||
Task task = emplace(
|
||||
[b=std::forward<B>(beg), e=std::forward<E>(end), c] (Subflow& sf) mutable {
|
||||
|
||||
// fetch the stateful values
|
||||
I beg = b;
|
||||
I end = e;
|
||||
|
||||
if(beg == end) {
|
||||
return;
|
||||
}
|
||||
|
||||
size_t chunk_size = 1;
|
||||
size_t W = sf._executor.num_workers();
|
||||
size_t N = std::distance(beg, end);
|
||||
|
||||
// only myself - no need to spawn another graph
|
||||
if(W <= 1 || N <= chunk_size) {
|
||||
std::for_each(beg, end, c);
|
||||
return;
|
||||
}
|
||||
|
||||
if(N < W) {
|
||||
W = N;
|
||||
}
|
||||
|
||||
std::atomic<size_t> next(0);
|
||||
|
||||
for(size_t w=0; w<W; w++) {
|
||||
|
||||
//sf.emplace([&next, beg, N, chunk_size, W, c] () mutable {
|
||||
sf.silent_async([&next, beg, N, chunk_size, W, c] () mutable {
|
||||
|
||||
size_t z = 0;
|
||||
size_t p1 = 2 * W * (chunk_size + 1);
|
||||
double p2 = 0.5 / static_cast<double>(W);
|
||||
size_t s0 = next.load(std::memory_order_relaxed);
|
||||
|
||||
while(s0 < N) {
|
||||
|
||||
size_t r = N - s0;
|
||||
|
||||
// fine-grained
|
||||
if(r < p1) {
|
||||
while(1) {
|
||||
s0 = next.fetch_add(chunk_size, std::memory_order_relaxed);
|
||||
if(s0 >= N) {
|
||||
return;
|
||||
}
|
||||
size_t e0 = (chunk_size <= (N - s0)) ? s0 + chunk_size : N;
|
||||
std::advance(beg, s0-z);
|
||||
for(size_t x=s0; x<e0; x++) {
|
||||
c(*beg++);
|
||||
}
|
||||
z = e0;
|
||||
}
|
||||
break;
|
||||
}
|
||||
// coarse-grained
|
||||
else {
|
||||
size_t q = static_cast<size_t>(p2 * r);
|
||||
if(q < chunk_size) {
|
||||
q = chunk_size;
|
||||
}
|
||||
size_t e0 = (q <= r) ? s0 + q : N;
|
||||
if(next.compare_exchange_strong(s0, e0, std::memory_order_relaxed,
|
||||
std::memory_order_relaxed)) {
|
||||
std::advance(beg, s0-z);
|
||||
for(size_t x = s0; x< e0; x++) {
|
||||
c(*beg++);
|
||||
}
|
||||
z = e0;
|
||||
s0 = next.load(std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
}
|
||||
//}).name("pfg_"s + std::to_string(w));
|
||||
});
|
||||
}
|
||||
|
||||
sf.join();
|
||||
});
|
||||
|
||||
return task;
|
||||
}
|
||||
|
||||
// Function: for_each_index
|
||||
template <typename B, typename E, typename S, typename C>
|
||||
Task FlowBuilder::for_each_index(B&& beg, E&& end, S&& inc, C c){
|
||||
|
||||
using I = stateful_index_t<B, E, S>;
|
||||
using namespace std::string_literals;
|
||||
|
||||
Task task = emplace(
|
||||
[b=std::forward<B>(beg), e=std::forward<E>(end), a=std::forward<S>(inc), c]
|
||||
(Subflow& sf) mutable {
|
||||
|
||||
// fetch the iterator values
|
||||
I beg = b;
|
||||
I end = e;
|
||||
I inc = a;
|
||||
|
||||
if(is_range_invalid(beg, end, inc)) {
|
||||
TF_THROW("invalid range [", beg, ", ", end, ") with step size ", inc);
|
||||
}
|
||||
|
||||
size_t chunk_size = 1;
|
||||
size_t W = sf._executor.num_workers();
|
||||
size_t N = distance(beg, end, inc);
|
||||
|
||||
// only myself - no need to spawn another graph
|
||||
if(W <= 1 || N <= chunk_size) {
|
||||
for(size_t x=0; x<N; x++, beg+=inc) {
|
||||
c(beg);
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
if(N < W) {
|
||||
W = N;
|
||||
}
|
||||
|
||||
std::atomic<size_t> next(0);
|
||||
|
||||
for(size_t w=0; w<W; w++) {
|
||||
|
||||
//sf.emplace([&next, beg, inc, N, chunk_size, W, c] () mutable {
|
||||
sf.silent_async([&next, beg, inc, N, chunk_size, W, c] () mutable {
|
||||
|
||||
size_t p1 = 2 * W * (chunk_size + 1);
|
||||
double p2 = 0.5 / static_cast<double>(W);
|
||||
size_t s0 = next.load(std::memory_order_relaxed);
|
||||
|
||||
while(s0 < N) {
|
||||
|
||||
size_t r = N - s0;
|
||||
|
||||
// find-grained
|
||||
if(r < p1) {
|
||||
while(1) {
|
||||
s0 = next.fetch_add(chunk_size, std::memory_order_relaxed);
|
||||
if(s0 >= N) {
|
||||
return;
|
||||
}
|
||||
size_t e0 = (chunk_size <= (N - s0)) ? s0 + chunk_size : N;
|
||||
auto s = static_cast<I>(s0) * inc + beg;
|
||||
for(size_t x=s0; x<e0; x++, s+=inc) {
|
||||
c(s);
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
// coarse-grained
|
||||
else {
|
||||
size_t q = static_cast<size_t>(p2 * r);
|
||||
if(q < chunk_size) {
|
||||
q = chunk_size;
|
||||
}
|
||||
size_t e0 = (q <= r) ? s0 + q : N;
|
||||
if(next.compare_exchange_strong(s0, e0, std::memory_order_relaxed,
|
||||
std::memory_order_relaxed)) {
|
||||
auto s = static_cast<I>(s0) * inc + beg;
|
||||
for(size_t x=s0; x<e0; x++, s+= inc) {
|
||||
c(s);
|
||||
}
|
||||
s0 = next.load(std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
}
|
||||
//}).name("pfg_"s + std::to_string(w));
|
||||
});
|
||||
}
|
||||
|
||||
sf.join();
|
||||
});
|
||||
|
||||
return task;
|
||||
}
|
||||
|
||||
} // end of namespace tf -----------------------------------------------------
|
||||
|
||||
|
||||
|
||||
262
cpp_to_py/gpugr/taskflow/core/algorithm/reduce.hpp
Normal file
262
cpp_to_py/gpugr/taskflow/core/algorithm/reduce.hpp
Normal file
@ -0,0 +1,262 @@
|
||||
#pragma once
|
||||
|
||||
#include "../executor.hpp"
|
||||
|
||||
namespace tf {
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// default reduction
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
template <typename B, typename E, typename T, typename O>
|
||||
Task FlowBuilder::reduce(B&& beg, E&& end, T& init, O bop) {
|
||||
|
||||
using I = stateful_iterator_t<B, E>;
|
||||
using namespace std::string_literals;
|
||||
|
||||
Task task = emplace(
|
||||
[b=std::forward<B>(beg), e=std::forward<E>(end), &r=init, bop]
|
||||
(Subflow& sf) mutable {
|
||||
|
||||
// fetch the iterator values
|
||||
I beg = b;
|
||||
I end = e;
|
||||
|
||||
if(beg == end) {
|
||||
return;
|
||||
}
|
||||
|
||||
//size_t C = (c == 0) ? 1 : c;
|
||||
size_t C = 1;
|
||||
size_t W = sf._executor.num_workers();
|
||||
size_t N = std::distance(beg, end);
|
||||
|
||||
// only myself - no need to spawn another graph
|
||||
if(W <= 1 || N <= C) {
|
||||
for(; beg!=end; r = bop(r, *beg++));
|
||||
return;
|
||||
}
|
||||
|
||||
if(N < W) {
|
||||
W = N;
|
||||
}
|
||||
|
||||
std::mutex mutex;
|
||||
std::atomic<size_t> next(0);
|
||||
|
||||
for(size_t w=0; w<W; w++) {
|
||||
|
||||
if(w*2 >= N) {
|
||||
break;
|
||||
}
|
||||
|
||||
//sf.emplace([&mutex, &next, &r, beg, N, W, o, C] () mutable {
|
||||
sf.silent_async([&mutex, &next, &r, beg, N, W, bop, C] () mutable {
|
||||
|
||||
size_t s0 = next.fetch_add(2, std::memory_order_relaxed);
|
||||
|
||||
if(s0 >= N) {
|
||||
return;
|
||||
}
|
||||
|
||||
std::advance(beg, s0);
|
||||
|
||||
if(N - s0 == 1) {
|
||||
std::lock_guard<std::mutex> lock(mutex);
|
||||
r = bop(r, *beg);
|
||||
return;
|
||||
}
|
||||
|
||||
auto beg1 = beg++;
|
||||
auto beg2 = beg++;
|
||||
|
||||
T sum = bop(*beg1, *beg2);
|
||||
|
||||
size_t z = s0 + 2;
|
||||
size_t p1 = 2 * W * (C + 1);
|
||||
double p2 = 0.5 / static_cast<double>(W);
|
||||
s0 = next.load(std::memory_order_relaxed);
|
||||
|
||||
while(s0 < N) {
|
||||
|
||||
size_t r = N - s0;
|
||||
|
||||
// fine-grained
|
||||
if(r < p1) {
|
||||
while(1) {
|
||||
s0 = next.fetch_add(C, std::memory_order_relaxed);
|
||||
if(s0 >= N) {
|
||||
break;
|
||||
}
|
||||
size_t e0 = (C <= (N - s0)) ? s0 + C : N;
|
||||
std::advance(beg, s0-z);
|
||||
for(size_t x=s0; x<e0; x++, beg++) {
|
||||
sum = bop(sum, *beg);
|
||||
}
|
||||
z = e0;
|
||||
}
|
||||
break;
|
||||
}
|
||||
// coarse-grained
|
||||
else {
|
||||
size_t q = static_cast<size_t>(p2 * r);
|
||||
if(q < C) {
|
||||
q = C;
|
||||
}
|
||||
size_t e0 = (q <= r) ? s0 + q : N;
|
||||
if(next.compare_exchange_strong(s0, e0, std::memory_order_relaxed,
|
||||
std::memory_order_relaxed)) {
|
||||
std::advance(beg, s0-z);
|
||||
for(size_t x = s0; x<e0; x++, beg++) {
|
||||
sum = bop(sum, *beg);
|
||||
}
|
||||
z = e0;
|
||||
s0 = next.load(std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::lock_guard<std::mutex> lock(mutex);
|
||||
r = bop(r, sum);
|
||||
//}).name("prg_"s + std::to_string(w));
|
||||
});
|
||||
}
|
||||
|
||||
sf.join();
|
||||
});
|
||||
|
||||
return task;
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// default transform and reduction
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
template <typename B, typename E, typename T, typename BOP, typename UOP>
|
||||
Task FlowBuilder::transform_reduce(
|
||||
B&& beg, E&& end, T& init, BOP bop, UOP uop
|
||||
) {
|
||||
|
||||
using I = stateful_iterator_t<B, E>;
|
||||
using namespace std::string_literals;
|
||||
|
||||
Task task = emplace(
|
||||
[b=std::forward<B>(beg), e=std::forward<E>(end), &r=init, bop, uop]
|
||||
(Subflow& sf) mutable {
|
||||
|
||||
// fetch the iterator values
|
||||
I beg = b;
|
||||
I end = e;
|
||||
|
||||
if(beg == end) {
|
||||
return;
|
||||
}
|
||||
|
||||
//size_t C = (c == 0) ? 1 : c;
|
||||
size_t C = 1;
|
||||
size_t W = sf._executor.num_workers();
|
||||
size_t N = std::distance(beg, end);
|
||||
|
||||
// only myself - no need to spawn another graph
|
||||
if(W <= 1 || N <= C) {
|
||||
for(; beg!=end; r = bop(r, uop(*beg++)));
|
||||
return;
|
||||
}
|
||||
|
||||
if(N < W) {
|
||||
W = N;
|
||||
}
|
||||
|
||||
std::mutex mutex;
|
||||
std::atomic<size_t> next(0);
|
||||
|
||||
for(size_t w=0; w<W; w++) {
|
||||
|
||||
if(w*2 >= N) {
|
||||
break;
|
||||
}
|
||||
|
||||
//sf.emplace([&mutex, &next, &r, beg, N, W, bop, uop, C] () mutable {
|
||||
sf.silent_async([&mutex, &next, &r, beg, N, W, bop, uop, C] () mutable {
|
||||
|
||||
size_t s0 = next.fetch_add(2, std::memory_order_relaxed);
|
||||
|
||||
if(s0 >= N) {
|
||||
return;
|
||||
}
|
||||
|
||||
std::advance(beg, s0);
|
||||
|
||||
if(N - s0 == 1) {
|
||||
std::lock_guard<std::mutex> lock(mutex);
|
||||
r = bop(r, uop(*beg));
|
||||
return;
|
||||
}
|
||||
|
||||
auto beg1 = beg++;
|
||||
auto beg2 = beg++;
|
||||
|
||||
T sum = bop(uop(*beg1), uop(*beg2));
|
||||
|
||||
size_t z = s0 + 2;
|
||||
size_t p1 = 2 * W * (C + 1);
|
||||
double p2 = 0.5 / static_cast<double>(W);
|
||||
s0 = next.load(std::memory_order_relaxed);
|
||||
|
||||
while(s0 < N) {
|
||||
|
||||
size_t r = N - s0;
|
||||
|
||||
// fine-grained
|
||||
if(r < p1) {
|
||||
while(1) {
|
||||
s0 = next.fetch_add(C, std::memory_order_relaxed);
|
||||
if(s0 >= N) {
|
||||
break;
|
||||
}
|
||||
size_t e0 = (C <= (N - s0)) ? s0 + C : N;
|
||||
std::advance(beg, s0-z);
|
||||
for(size_t x=s0; x<e0; x++, beg++) {
|
||||
sum = bop(sum, uop(*beg));
|
||||
}
|
||||
z = e0;
|
||||
}
|
||||
break;
|
||||
}
|
||||
// coarse-grained
|
||||
else {
|
||||
size_t q = static_cast<size_t>(p2 * r);
|
||||
if(q < C) {
|
||||
q = C;
|
||||
}
|
||||
size_t e0 = (q <= r) ? s0 + q : N;
|
||||
if(next.compare_exchange_strong(s0, e0, std::memory_order_relaxed,
|
||||
std::memory_order_relaxed)) {
|
||||
std::advance(beg, s0-z);
|
||||
for(size_t x = s0; x<e0; x++, beg++) {
|
||||
sum = bop(sum, uop(*beg));
|
||||
}
|
||||
z = e0;
|
||||
s0 = next.load(std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::lock_guard<std::mutex> lock(mutex);
|
||||
r = bop(r, sum);
|
||||
|
||||
//}).name("prg_"s + std::to_string(w));
|
||||
});
|
||||
}
|
||||
|
||||
sf.join();
|
||||
});
|
||||
|
||||
return task;
|
||||
}
|
||||
|
||||
} // end of namespace tf -----------------------------------------------------
|
||||
|
||||
|
||||
|
||||
|
||||
479
cpp_to_py/gpugr/taskflow/core/algorithm/sort.hpp
Normal file
479
cpp_to_py/gpugr/taskflow/core/algorithm/sort.hpp
Normal file
@ -0,0 +1,479 @@
|
||||
#pragma once
|
||||
|
||||
#include "../executor.hpp"
|
||||
|
||||
namespace tf {
|
||||
|
||||
// threshold whether or not to perform parallel sort
|
||||
template <typename I>
|
||||
constexpr size_t parallel_sort_cutoff() {
|
||||
|
||||
//using value_type = std::decay_t<decltype(*std::declval<I>())>;
|
||||
using value_type = typename std::iterator_traits<I>::value_type;
|
||||
|
||||
constexpr size_t object_size = sizeof(value_type);
|
||||
|
||||
if constexpr(std::is_same_v<value_type, std::string>) {
|
||||
return 128;
|
||||
}
|
||||
else {
|
||||
if constexpr(object_size < 16) return 4096;
|
||||
else if constexpr(object_size < 32) return 2048;
|
||||
else if constexpr(object_size < 64) return 1024;
|
||||
else if constexpr(object_size < 128) return 768;
|
||||
else if constexpr(object_size < 256) return 512;
|
||||
else if constexpr(object_size < 512) return 256;
|
||||
else return 128;
|
||||
}
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// pattern-defeating quick sort (pdqsort)
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Sorts [begin, end) using insertion sort with the given comparison function.
|
||||
template<typename RandItr, typename Compare>
|
||||
void insertion_sort(RandItr begin, RandItr end, Compare comp) {
|
||||
|
||||
using T = typename std::iterator_traits<RandItr>::value_type;
|
||||
|
||||
if (begin == end) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (RandItr cur = begin + 1; cur != end; ++cur) {
|
||||
|
||||
RandItr shift = cur;
|
||||
RandItr shift_1 = cur - 1;
|
||||
|
||||
// Compare first to avoid 2 moves for an element
|
||||
// already positioned correctly.
|
||||
if (comp(*shift, *shift_1)) {
|
||||
T tmp = std::move(*shift);
|
||||
do {
|
||||
*shift-- = std::move(*shift_1);
|
||||
}while (shift != begin && comp(tmp, *--shift_1));
|
||||
*shift = std::move(tmp);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Sorts [begin, end) using insertion sort with the given comparison function.
|
||||
// Assumes *(begin - 1) is an element smaller than or equal to any element
|
||||
// in [begin, end).
|
||||
template<typename RandItr, typename Compare>
|
||||
void unguarded_insertion_sort(RandItr begin, RandItr end, Compare comp) {
|
||||
|
||||
using T = typename std::iterator_traits<RandItr>::value_type;
|
||||
|
||||
if (begin == end) {
|
||||
return;
|
||||
}
|
||||
|
||||
for (RandItr cur = begin + 1; cur != end; ++cur) {
|
||||
RandItr shift = cur;
|
||||
RandItr shift_1 = cur - 1;
|
||||
|
||||
// Compare first so we can avoid 2 moves
|
||||
// for an element already positioned correctly.
|
||||
if (comp(*shift, *shift_1)) {
|
||||
T tmp = std::move(*shift);
|
||||
|
||||
do {
|
||||
*shift-- = std::move(*shift_1);
|
||||
}while (comp(tmp, *--shift_1));
|
||||
|
||||
*shift = std::move(tmp);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Attempts to use insertion sort on [begin, end).
|
||||
// Will return false if more than
|
||||
// partial_insertion_sort_limit elements were moved,
|
||||
// and abort sorting. Otherwise it will successfully sort and return true.
|
||||
template<typename RandItr, typename Compare>
|
||||
bool partial_insertion_sort(RandItr begin, RandItr end, Compare comp) {
|
||||
|
||||
using T = typename std::iterator_traits<RandItr>::value_type;
|
||||
using D = typename std::iterator_traits<RandItr>::difference_type;
|
||||
|
||||
// When we detect an already sorted partition, attempt an insertion sort
|
||||
// that allows this amount of element moves before giving up.
|
||||
constexpr auto partial_insertion_sort_limit = D{8};
|
||||
|
||||
if (begin == end) return true;
|
||||
|
||||
auto limit = D{0};
|
||||
|
||||
for (RandItr cur = begin + 1; cur != end; ++cur) {
|
||||
|
||||
if (limit > partial_insertion_sort_limit) {
|
||||
return false;
|
||||
}
|
||||
|
||||
RandItr shift = cur;
|
||||
RandItr shift_1 = cur - 1;
|
||||
|
||||
// Compare first so we can avoid 2 moves
|
||||
// for an element already positioned correctly.
|
||||
if (comp(*shift, *shift_1)) {
|
||||
T tmp = std::move(*shift);
|
||||
|
||||
do {
|
||||
*shift-- = std::move(*shift_1);
|
||||
}while (shift != begin && comp(tmp, *--shift_1));
|
||||
|
||||
*shift = std::move(tmp);
|
||||
limit += cur - shift;
|
||||
}
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
// Partitions [begin, end) around pivot *begin using comparison function comp.
|
||||
// Elements equal to the pivot are put in the right-hand partition.
|
||||
// Returns the position of the pivot after partitioning and whether the passed
|
||||
// sequence already was correctly partitioned.
|
||||
// Assumes the pivot is a median of at least 3 elements and that [begin, end)
|
||||
// is at least insertion_sort_threshold long.
|
||||
template<typename Iter, typename Compare>
|
||||
std::pair<Iter, bool> partition_right(Iter begin, Iter end, Compare comp) {
|
||||
|
||||
using T = typename std::iterator_traits<Iter>::value_type;
|
||||
|
||||
// Move pivot into local for speed.
|
||||
T pivot(std::move(*begin));
|
||||
|
||||
Iter first = begin;
|
||||
Iter last = end;
|
||||
|
||||
// Find the first element greater than or equal than the pivot
|
||||
// (the median of 3 guarantees/ this exists).
|
||||
while (comp(*++first, pivot));
|
||||
|
||||
// Find the first element strictly smaller than the pivot.
|
||||
// We have to guard this search if there was no element before *first.
|
||||
if (first - 1 == begin) while (first < last && !comp(*--last, pivot));
|
||||
else while (!comp(*--last, pivot));
|
||||
|
||||
// If the first pair of elements that should be swapped to partition
|
||||
// are the same element, the passed in sequence already was correctly
|
||||
// partitioned.
|
||||
bool already_partitioned = first >= last;
|
||||
|
||||
// Keep swapping pairs of elements that are on the wrong side of the pivot.
|
||||
// Previously swapped pairs guard the searches,
|
||||
// which is why the first iteration is special-cased above.
|
||||
while (first < last) {
|
||||
std::iter_swap(first, last);
|
||||
while (comp(*++first, pivot));
|
||||
while (!comp(*--last, pivot));
|
||||
}
|
||||
|
||||
// Put the pivot in the right place.
|
||||
Iter pivot_pos = first - 1;
|
||||
*begin = std::move(*pivot_pos);
|
||||
*pivot_pos = std::move(pivot);
|
||||
|
||||
return std::make_pair(pivot_pos, already_partitioned);
|
||||
}
|
||||
|
||||
// Similar function to the one above, except elements equal to the pivot
|
||||
// are put to the left of the pivot and it doesn't check or return
|
||||
// if the passed sequence already was partitioned.
|
||||
// Since this is rarely used (the many equal case),
|
||||
// and in that case pdqsort already has O(n) performance,
|
||||
// no block quicksort is applied here for simplicity.
|
||||
template<typename RandItr, typename Compare>
|
||||
RandItr partition_left(RandItr begin, RandItr end, Compare comp) {
|
||||
|
||||
using T = typename std::iterator_traits<RandItr>::value_type;
|
||||
|
||||
T pivot(std::move(*begin));
|
||||
|
||||
RandItr first = begin;
|
||||
RandItr last = end;
|
||||
|
||||
while (comp(pivot, *--last));
|
||||
|
||||
if (last + 1 == end) {
|
||||
while (first < last && !comp(pivot, *++first));
|
||||
}
|
||||
else {
|
||||
while (!comp(pivot, *++first));
|
||||
}
|
||||
|
||||
while (first < last) {
|
||||
std::iter_swap(first, last);
|
||||
while (comp(pivot, *--last));
|
||||
while (!comp(pivot, *++first));
|
||||
}
|
||||
|
||||
RandItr pivot_pos = last;
|
||||
*begin = std::move(*pivot_pos);
|
||||
*pivot_pos = std::move(pivot);
|
||||
|
||||
return pivot_pos;
|
||||
}
|
||||
|
||||
template<typename Iter, typename Compare>
|
||||
void parallel_pdqsort(
|
||||
tf::Subflow& sf,
|
||||
Iter begin, Iter end, Compare comp,
|
||||
int bad_allowed, bool leftmost = true
|
||||
) {
|
||||
|
||||
// Partitions below this size are sorted sequentially
|
||||
constexpr auto cutoff = parallel_sort_cutoff<Iter>();
|
||||
|
||||
// Partitions below this size are sorted using insertion sort
|
||||
constexpr auto insertion_sort_threshold = 24;
|
||||
|
||||
// Partitions above this size use Tukey's ninther to select the pivot.
|
||||
constexpr auto ninther_threshold = 128;
|
||||
|
||||
//using diff_t = typename std::iterator_traits<Iter>::difference_type;
|
||||
|
||||
// Use a while loop for tail recursion elimination.
|
||||
while (true) {
|
||||
|
||||
//diff_t size = end - begin;
|
||||
size_t size = end - begin;
|
||||
|
||||
if(size <= cutoff) {
|
||||
std::sort(begin, end, comp);
|
||||
return;
|
||||
}
|
||||
//// Insertion sort is faster for small arrays.
|
||||
//if (size < insertion_sort_threshold) {
|
||||
// if (leftmost) {
|
||||
// insertion_sort(begin, end, comp);
|
||||
// }
|
||||
// else {
|
||||
// unguarded_insertion_sort(begin, end, comp);
|
||||
// }
|
||||
// return;
|
||||
//}
|
||||
|
||||
// Choose pivot as median of 3 or pseudomedian of 9.
|
||||
//diff_t s2 = size / 2;
|
||||
size_t s2 = size >> 1;
|
||||
if (size > ninther_threshold) {
|
||||
sort3(begin, begin + s2, end - 1, comp);
|
||||
sort3(begin + 1, begin + (s2 - 1), end - 2, comp);
|
||||
sort3(begin + 2, begin + (s2 + 1), end - 3, comp);
|
||||
sort3(begin + (s2 - 1), begin + s2, begin + (s2 + 1), comp);
|
||||
std::iter_swap(begin, begin + s2);
|
||||
}
|
||||
else {
|
||||
sort3(begin + s2, begin, end - 1, comp);
|
||||
}
|
||||
|
||||
// If *(begin - 1) is the end of the right partition
|
||||
// of a previous partition operation, there is no element in [begin, end)
|
||||
// that is smaller than *(begin - 1).
|
||||
// Then if our pivot compares equal to *(begin - 1) we change strategy,
|
||||
// putting equal elements in the left partition,
|
||||
// greater elements in the right partition.
|
||||
// We do not have to recurse on the left partition,
|
||||
// since it's sorted (all equal).
|
||||
if (!leftmost && !comp(*(begin - 1), *begin)) {
|
||||
begin = partition_left(begin, end, comp) + 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Partition and get results.
|
||||
auto pair = partition_right(begin, end, comp);
|
||||
auto pivot_pos = pair.first;
|
||||
auto already_partitioned = pair.second;
|
||||
|
||||
// Check for a highly unbalanced partition.
|
||||
//diff_t l_size = pivot_pos - begin;
|
||||
//diff_t r_size = end - (pivot_pos + 1);
|
||||
size_t l_size = pivot_pos - begin;
|
||||
size_t r_size = end - (pivot_pos + 1);
|
||||
bool highly_unbalanced = l_size < size / 8 || r_size < size / 8;
|
||||
|
||||
// If we got a highly unbalanced partition we shuffle elements
|
||||
// to break many patterns.
|
||||
if (highly_unbalanced) {
|
||||
// If we had too many bad partitions, switch to heapsort
|
||||
// to guarantee O(n log n).
|
||||
if (--bad_allowed == 0) {
|
||||
std::make_heap(begin, end, comp);
|
||||
std::sort_heap(begin, end, comp);
|
||||
return;
|
||||
}
|
||||
|
||||
if (l_size >= insertion_sort_threshold) {
|
||||
std::iter_swap(begin, begin + l_size / 4);
|
||||
std::iter_swap(pivot_pos - 1, pivot_pos - l_size / 4);
|
||||
if (l_size > ninther_threshold) {
|
||||
std::iter_swap(begin + 1, begin + (l_size / 4 + 1));
|
||||
std::iter_swap(begin + 2, begin + (l_size / 4 + 2));
|
||||
std::iter_swap(pivot_pos - 2, pivot_pos - (l_size / 4 + 1));
|
||||
std::iter_swap(pivot_pos - 3, pivot_pos - (l_size / 4 + 2));
|
||||
}
|
||||
}
|
||||
|
||||
if (r_size >= insertion_sort_threshold) {
|
||||
std::iter_swap(pivot_pos + 1, pivot_pos + (1 + r_size / 4));
|
||||
std::iter_swap(end - 1, end - r_size / 4);
|
||||
if (r_size > ninther_threshold) {
|
||||
std::iter_swap(pivot_pos + 2, pivot_pos + (2 + r_size / 4));
|
||||
std::iter_swap(pivot_pos + 3, pivot_pos + (3 + r_size / 4));
|
||||
std::iter_swap(end - 2, end - (1 + r_size / 4));
|
||||
std::iter_swap(end - 3, end - (2 + r_size / 4));
|
||||
}
|
||||
}
|
||||
}
|
||||
// decently balanced
|
||||
else {
|
||||
// sequence try to use insertion sort.
|
||||
if (already_partitioned &&
|
||||
partial_insertion_sort(begin, pivot_pos, comp) &&
|
||||
partial_insertion_sort(pivot_pos + 1, end, comp)
|
||||
) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
// Sort the left partition first using recursion and
|
||||
// do tail recursion elimination for the right-hand partition.
|
||||
sf.silent_async(
|
||||
[&sf, begin, pivot_pos, comp, bad_allowed, leftmost] () mutable {
|
||||
parallel_pdqsort(sf, begin, pivot_pos, comp, bad_allowed, leftmost);
|
||||
}
|
||||
);
|
||||
begin = pivot_pos + 1;
|
||||
leftmost = false;
|
||||
}
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// 3-way quick sort
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// 3-way quick sort
|
||||
template <typename RandItr, typename C>
|
||||
void parallel_3wqsort(tf::Subflow& sf, RandItr first, RandItr last, C compare) {
|
||||
|
||||
using namespace std::string_literals;
|
||||
|
||||
constexpr auto cutoff = parallel_sort_cutoff<RandItr>();
|
||||
|
||||
sort_partition:
|
||||
|
||||
if(static_cast<size_t>(last - first) < cutoff) {
|
||||
std::sort(first, last+1, compare);
|
||||
return;
|
||||
}
|
||||
|
||||
auto m = pseudo_median_of_nine(first, last, compare);
|
||||
|
||||
if(m != first) {
|
||||
std::iter_swap(first, m);
|
||||
}
|
||||
|
||||
auto l = first;
|
||||
auto r = last;
|
||||
auto f = std::next(first, 1);
|
||||
bool is_swapped_l = false;
|
||||
bool is_swapped_r = false;
|
||||
|
||||
while(f <= r) {
|
||||
if(compare(*f, *l)) {
|
||||
is_swapped_l = true;
|
||||
std::iter_swap(l, f);
|
||||
l++;
|
||||
f++;
|
||||
}
|
||||
else if(compare(*l, *f)) {
|
||||
is_swapped_r = true;
|
||||
std::iter_swap(r, f);
|
||||
r--;
|
||||
}
|
||||
else {
|
||||
f++;
|
||||
}
|
||||
}
|
||||
|
||||
if(l - first > 1 && is_swapped_l) {
|
||||
//sf.emplace([&](tf::Subflow& sfl) mutable {
|
||||
// parallel_3wqsort(sfl, first, l-1, compare);
|
||||
//});
|
||||
sf.silent_async([&sf, first, l, &compare] () mutable {
|
||||
parallel_3wqsort(sf, first, l-1, compare);
|
||||
});
|
||||
}
|
||||
|
||||
if(last - r > 1 && is_swapped_r) {
|
||||
//sf.emplace([&](tf::Subflow& sfr) mutable {
|
||||
// parallel_3wqsort(sfr, r+1, last, compare);
|
||||
//});
|
||||
//sf.silent_async([&sf, r, last, &compare] () mutable {
|
||||
// parallel_3wqsort(sf, r+1, last, compare);
|
||||
//});
|
||||
first = r+1;
|
||||
goto sort_partition;
|
||||
}
|
||||
|
||||
//sf.join();
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// tf::Taskflow::sort
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Function: sort
|
||||
template <typename B, typename E, typename C>
|
||||
Task FlowBuilder::sort(B&& beg, E&& end, C cmp) {
|
||||
|
||||
using I = stateful_iterator_t<B, E>;
|
||||
|
||||
Task task = emplace(
|
||||
[b=std::forward<B>(beg), e=std::forward<E>(end), cmp] (Subflow& sf) mutable {
|
||||
|
||||
// fetch the iterator values
|
||||
I beg = b;
|
||||
I end = e;
|
||||
|
||||
if(beg == end) {
|
||||
return;
|
||||
}
|
||||
|
||||
size_t W = sf._executor.num_workers();
|
||||
size_t N = std::distance(beg, end);
|
||||
|
||||
// only myself - no need to spawn another graph
|
||||
if(W <= 1 || N <= parallel_sort_cutoff<I>()) {
|
||||
std::sort(beg, end, cmp);
|
||||
return;
|
||||
}
|
||||
|
||||
//parallel_3wqsort(sf, beg, end-1, c);
|
||||
parallel_pdqsort(sf, beg, end, cmp, log2(end - beg));
|
||||
|
||||
sf.join();
|
||||
});
|
||||
|
||||
return task;
|
||||
}
|
||||
|
||||
// Function: sort
|
||||
template <typename B, typename E>
|
||||
Task FlowBuilder::sort(B&& beg, E&& end) {
|
||||
|
||||
using I = stateful_iterator_t<B, E>;
|
||||
//using value_type = std::decay_t<decltype(*std::declval<I>())>;
|
||||
using value_type = typename std::iterator_traits<I>::value_type;
|
||||
|
||||
return sort(
|
||||
std::forward<B>(beg), std::forward<E>(end), std::less<value_type>{}
|
||||
);
|
||||
}
|
||||
|
||||
} // namespace tf ------------------------------------------------------------
|
||||
|
||||
50
cpp_to_py/gpugr/taskflow/core/declarations.hpp
Normal file
50
cpp_to_py/gpugr/taskflow/core/declarations.hpp
Normal file
@ -0,0 +1,50 @@
|
||||
#pragma once
|
||||
|
||||
namespace tf {
|
||||
|
||||
// taskflow
|
||||
class AsyncTopology;
|
||||
class Node;
|
||||
class Graph;
|
||||
class FlowBuilder;
|
||||
class Semaphore;
|
||||
class Subflow;
|
||||
class Task;
|
||||
class TaskView;
|
||||
class Taskflow;
|
||||
class Topology;
|
||||
class TopologyBase;
|
||||
class Executor;
|
||||
class WorkerView;
|
||||
class ObserverInterface;
|
||||
class ChromeTracingObserver;
|
||||
class TFProfObserver;
|
||||
class TFProfManager;
|
||||
|
||||
template <typename T>
|
||||
class Future;
|
||||
|
||||
// cudaFlow
|
||||
class cudaNode;
|
||||
class cudaGraph;
|
||||
class cudaTask;
|
||||
class cudaFlow;
|
||||
class cudaFlowCapturer;
|
||||
class cudaFlowCapturerBase;
|
||||
class cudaCapturingBase;
|
||||
class cudaLinearCapturing;
|
||||
class cudaSequentialCapturing;
|
||||
class cudaRoundRobinCapturing;
|
||||
|
||||
// syclFlow
|
||||
class syclNode;
|
||||
class syclGraph;
|
||||
class syclTask;
|
||||
class syclFlow;
|
||||
|
||||
|
||||
} // end of namespace tf -----------------------------------------------------
|
||||
|
||||
|
||||
|
||||
|
||||
8
cpp_to_py/gpugr/taskflow/core/environment.hpp
Normal file
8
cpp_to_py/gpugr/taskflow/core/environment.hpp
Normal file
@ -0,0 +1,8 @@
|
||||
#pragma once
|
||||
|
||||
#define TF_ENABLE_PROFILER "TF_ENABLE_PROFILER"
|
||||
|
||||
namespace tf {
|
||||
|
||||
} // end of namespace tf -----------------------------------------------------
|
||||
|
||||
26
cpp_to_py/gpugr/taskflow/core/error.hpp
Normal file
26
cpp_to_py/gpugr/taskflow/core/error.hpp
Normal file
@ -0,0 +1,26 @@
|
||||
#pragma once
|
||||
|
||||
#include <iostream>
|
||||
#include <sstream>
|
||||
#include <exception>
|
||||
|
||||
#include "../utility/stream.hpp"
|
||||
|
||||
namespace tf {
|
||||
|
||||
// Procedure: throw_se
|
||||
// Throws the system error under a given error code.
|
||||
template <typename... ArgsT>
|
||||
//void throw_se(const char* fname, const size_t line, Error::Code c, ArgsT&&... args) {
|
||||
void throw_re(const char* fname, const size_t line, ArgsT&&... args) {
|
||||
std::ostringstream oss;
|
||||
oss << "[" << fname << ":" << line << "] ";
|
||||
//ostreamize(oss, std::forward<ArgsT>(args)...);
|
||||
(oss << ... << args);
|
||||
throw std::runtime_error(oss.str());
|
||||
}
|
||||
|
||||
} // ------------------------------------------------------------------------
|
||||
|
||||
#define TF_THROW(...) tf::throw_re(__FILE__, __LINE__, __VA_ARGS__);
|
||||
|
||||
1375
cpp_to_py/gpugr/taskflow/core/executor.hpp
Normal file
1375
cpp_to_py/gpugr/taskflow/core/executor.hpp
Normal file
File diff suppressed because it is too large
Load Diff
754
cpp_to_py/gpugr/taskflow/core/flow_builder.hpp
Normal file
754
cpp_to_py/gpugr/taskflow/core/flow_builder.hpp
Normal file
@ -0,0 +1,754 @@
|
||||
#pragma once
|
||||
|
||||
#include "task.hpp"
|
||||
|
||||
/**
|
||||
@file flow_builder.hpp
|
||||
@brief flow builder include file
|
||||
*/
|
||||
|
||||
namespace tf {
|
||||
|
||||
/**
|
||||
@class FlowBuilder
|
||||
|
||||
@brief building methods of a task dependency graph
|
||||
|
||||
*/
|
||||
class FlowBuilder {
|
||||
|
||||
friend class Executor;
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
@brief creates a static task
|
||||
|
||||
@tparam C callable type constructible from std::function<void()>
|
||||
|
||||
@param callable callable to construct a static task
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
The following example creates a static task.
|
||||
|
||||
@code{.cpp}
|
||||
tf::Task static_task = taskflow.emplace([](){});
|
||||
@endcode
|
||||
|
||||
Please refer to @ref StaticTasking for details.
|
||||
*/
|
||||
template <typename C,
|
||||
std::enable_if_t<is_static_task_v<C>, void>* = nullptr
|
||||
>
|
||||
Task emplace(C&& callable);
|
||||
|
||||
/**
|
||||
@brief creates a dynamic task
|
||||
|
||||
@tparam C callable type constructible from std::function<void(tf::Subflow&)>
|
||||
|
||||
@param callable callable to construct a dynamic task
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
The following example creates a dynamic task (tf::Subflow)
|
||||
that spawns two static tasks.
|
||||
|
||||
@code{.cpp}
|
||||
tf::Task dynamic_task = taskflow.emplace([](tf::Subflow& sf){
|
||||
tf::Task static_task1 = sf.emplace([](){});
|
||||
tf::Task static_task2 = sf.emplace([](){});
|
||||
});
|
||||
@endcode
|
||||
|
||||
Please refer to @ref DynamicTasking for details.
|
||||
*/
|
||||
template <typename C,
|
||||
std::enable_if_t<is_dynamic_task_v<C>, void>* = nullptr
|
||||
>
|
||||
Task emplace(C&& callable);
|
||||
|
||||
/**
|
||||
@brief creates a condition task
|
||||
|
||||
@tparam C callable type constructible from std::function<int()>
|
||||
|
||||
@param callable callable to construct a condition task
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
The following example creates an if-else block using one condition task
|
||||
and three static tasks.
|
||||
|
||||
@code{.cpp}
|
||||
tf::Taskflow taskflow;
|
||||
|
||||
auto [init, cond, yes, no] = taskflow.emplace(
|
||||
[] () { },
|
||||
[] () { return 0; },
|
||||
[] () { std::cout << "yes\n"; },
|
||||
[] () { std::cout << "no\n"; }
|
||||
);
|
||||
|
||||
// executes yes if cond returns 0, or no if cond returns 1
|
||||
cond.precede(yes, no);
|
||||
cond.succeed(init);
|
||||
@endcode
|
||||
|
||||
Please refer to @ref ConditionalTasking for details.
|
||||
*/
|
||||
template <typename C,
|
||||
std::enable_if_t<is_condition_task_v<C>, void>* = nullptr
|
||||
>
|
||||
Task emplace(C&& callable);
|
||||
|
||||
/**
|
||||
@brief creates multiple tasks from a list of callable objects
|
||||
|
||||
@tparam C callable types
|
||||
|
||||
@param callables one or multiple callable objects constructible from each task category
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
The method returns a tuple of tasks each corresponding to the given
|
||||
callable target. You can use structured binding to get the return tasks
|
||||
one by one.
|
||||
The following example creates four static tasks and assign them to
|
||||
@c A, @c B, @c C, and @c D using structured binding.
|
||||
|
||||
@code{.cpp}
|
||||
auto [A, B, C, D] = taskflow.emplace(
|
||||
[] () { std::cout << "A"; },
|
||||
[] () { std::cout << "B"; },
|
||||
[] () { std::cout << "C"; },
|
||||
[] () { std::cout << "D"; }
|
||||
);
|
||||
@endcode
|
||||
*/
|
||||
template <typename... C, std::enable_if_t<(sizeof...(C)>1), void>* = nullptr>
|
||||
auto emplace(C&&... callables);
|
||||
|
||||
/**
|
||||
@brief removes a task from a taskflow
|
||||
|
||||
@param task task to remove
|
||||
|
||||
Removes a task and its input and output dependencies from this graph.
|
||||
If the task does not belong to this graph, nothing will happen.
|
||||
*/
|
||||
void erase(Task task);
|
||||
|
||||
/**
|
||||
@brief creates a module task from a taskflow
|
||||
|
||||
@param taskflow a taskflow object for the module
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
Please refer to @ref ComposableTasking for details.
|
||||
*/
|
||||
Task composed_of(Taskflow& taskflow);
|
||||
|
||||
/**
|
||||
@brief creates a placeholder task
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
A placeholder task maps to a node in the taskflow graph, but
|
||||
it does not have any callable work assigned yet.
|
||||
A placeholder task is different from an empty task handle that
|
||||
does not point to any node in a graph.
|
||||
|
||||
@code{.cpp}
|
||||
// create a placeholder task with no callable target assigned
|
||||
tf::Task placeholder = taskflow.placeholder();
|
||||
assert(placeholder.empty() == false && placeholder.has_work() == false);
|
||||
|
||||
// create an empty task handle
|
||||
tf::Task task;
|
||||
assert(task.empty() == true);
|
||||
|
||||
// assign the task handle to the placeholder task
|
||||
task = placeholder;
|
||||
assert(task.empty() == false && task.has_work() == false);
|
||||
@endcode
|
||||
*/
|
||||
Task placeholder();
|
||||
|
||||
/**
|
||||
@brief creates a %cudaFlow task on the caller's GPU device context
|
||||
|
||||
@tparam C callable type constructible from @c std::function<void(tf::cudaFlow&)>
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
This method is equivalent to calling tf::FlowBuilder::emplace_on(callable, d)
|
||||
where @c d is the caller's device context.
|
||||
The following example creates a %cudaFlow of two kernel tasks, @c task1 and
|
||||
@c task2, where @c task1 runs before @c task2.
|
||||
|
||||
@code{.cpp}
|
||||
taskflow.emplace([&](tf::cudaFlow& cf){
|
||||
// create two kernel tasks
|
||||
tf::cudaTask task1 = cf.kernel(grid1, block1, shm1, kernel1, args1);
|
||||
tf::cudaTask task2 = cf.kernel(grid2, block2, shm2, kernel2, args2);
|
||||
|
||||
// kernel1 runs before kernel2
|
||||
task1.precede(task2);
|
||||
});
|
||||
@endcode
|
||||
|
||||
Please refer to @ref GPUTaskingcudaFlow and @ref GPUTaskingcudaFlowCapturer
|
||||
for details.
|
||||
*/
|
||||
template <typename C,
|
||||
std::enable_if_t<is_cudaflow_task_v<C>, void>* = nullptr
|
||||
>
|
||||
Task emplace(C&& callable);
|
||||
|
||||
/**
|
||||
@brief creates a %cudaFlow task on the given device
|
||||
|
||||
@tparam C callable type constructible from std::function<void(tf::cudaFlow&)>
|
||||
@tparam D device type, either @c int or @c std::ref<int> (stateful)
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
The following example creates a %cudaFlow of two kernel tasks, @c task1 and
|
||||
@c task2 on GPU @c 2, where @c task1 runs before @c task2
|
||||
|
||||
@code{.cpp}
|
||||
taskflow.emplace_on([&](tf::cudaFlow& cf){
|
||||
// create two kernel tasks
|
||||
tf::cudaTask task1 = cf.kernel(grid1, block1, shm1, kernel1, args1);
|
||||
tf::cudaTask task2 = cf.kernel(grid2, block2, shm2, kernel2, args2);
|
||||
|
||||
// kernel1 runs before kernel2
|
||||
task1.precede(task2);
|
||||
}, 2);
|
||||
@endcode
|
||||
*/
|
||||
template <typename C, typename D,
|
||||
std::enable_if_t<is_cudaflow_task_v<C>, void>* = nullptr
|
||||
>
|
||||
Task emplace_on(C&& callable, D&& device);
|
||||
|
||||
/**
|
||||
@brief creates a %syclFlow task on the default queue
|
||||
|
||||
@tparam C callable type constructible from std::function<void(tf::syclFlow&)>
|
||||
|
||||
@param callable a callable that takes a referenced tf::syclFlow object
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
The following example creates a %syclFlow on the default queue to submit
|
||||
two kernel tasks, @c task1 and @c task2, where @c task1 runs before @c task2.
|
||||
|
||||
@code{.cpp}
|
||||
taskflow.emplace([&](tf::syclFlow& cf){
|
||||
// create two single-thread kernel tasks
|
||||
tf::syclTask task1 = cf.single_task([](){});
|
||||
tf::syclTask task2 = cf.single_task([](){});
|
||||
|
||||
// kernel1 runs before kernel2
|
||||
task1.precede(task2);
|
||||
});
|
||||
@endcode
|
||||
*/
|
||||
template <typename C, std::enable_if_t<is_syclflow_task_v<C>, void>* = nullptr>
|
||||
Task emplace(C&& callable);
|
||||
|
||||
/**
|
||||
@brief creates a %syclFlow task on the given queue
|
||||
|
||||
@tparam C callable type constructible from std::function<void(tf::syclFlow&)>
|
||||
@tparam Q queue type
|
||||
|
||||
@param callable a callable that takes a referenced tf::syclFlow object
|
||||
@param queue a queue of type sycl::queue
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
The following example creates a %syclFlow on the given queue to submit
|
||||
two kernel tasks, @c task1 and @c task2, where @c task1 runs before @c task2.
|
||||
|
||||
@code{.cpp}
|
||||
taskflow.emplace_on([&](tf::syclFlow& cf){
|
||||
// create two single-thread kernel tasks
|
||||
tf::syclTask task1 = cf.single_task([](){});
|
||||
tf::syclTask task2 = cf.single_task([](){});
|
||||
|
||||
// kernel1 runs before kernel2
|
||||
task1.precede(task2);
|
||||
}, queue);
|
||||
@endcode
|
||||
*/
|
||||
template <typename C, typename Q,
|
||||
std::enable_if_t<is_syclflow_task_v<C>, void>* = nullptr
|
||||
>
|
||||
Task emplace_on(C&& callable, Q&& queue);
|
||||
|
||||
/**
|
||||
@brief adds adjacent dependency links to a linear list of tasks
|
||||
|
||||
@param tasks a vector of tasks
|
||||
*/
|
||||
void linearize(std::vector<Task>& tasks);
|
||||
|
||||
/**
|
||||
@brief adds adjacent dependency links to a linear list of tasks
|
||||
|
||||
@param tasks an initializer list of tasks
|
||||
*/
|
||||
void linearize(std::initializer_list<Task> tasks);
|
||||
|
||||
// ------------------------------------------------------------------------
|
||||
// parallel iterations
|
||||
// ------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@brief constructs a STL-styled parallel-for task
|
||||
|
||||
@tparam B beginning iterator type
|
||||
@tparam E ending iterator type
|
||||
@tparam C callable type
|
||||
|
||||
@param first iterator to the beginning (inclusive)
|
||||
@param last iterator to the end (exclusive)
|
||||
@param callable a callable object to apply to the dereferenced iterator
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
The task spawns a subflow that applies the callable object to each object obtained by dereferencing every iterator in the range <tt>[first, last)</tt>.
|
||||
This method is equivalent to the parallel execution of the following loop:
|
||||
|
||||
@code{.cpp}
|
||||
for(auto itr=first; itr!=last; itr++) {
|
||||
callable(*itr);
|
||||
}
|
||||
@endcode
|
||||
|
||||
Arguments templated to enable stateful passing using std::reference_wrapper.
|
||||
The callable needs to take a single argument of
|
||||
the dereferenced iterator type.
|
||||
|
||||
Please refer to @ref ParallelIterations for details.
|
||||
*/
|
||||
template <typename B, typename E, typename C>
|
||||
Task for_each(B&& first, E&& last, C callable);
|
||||
|
||||
/**
|
||||
@brief constructs an index-based parallel-for task
|
||||
|
||||
@tparam B beginning index type (must be integral)
|
||||
@tparam E ending index type (must be integral)
|
||||
@tparam S step type (must be integral)
|
||||
@tparam C callable type
|
||||
|
||||
@param first index of the beginning (inclusive)
|
||||
@param last index of the end (exclusive)
|
||||
@param step step size
|
||||
@param callable a callable object to apply to each valid index
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
The task spawns a subflow that applies the callable object to each index in the range <tt>[first, last)</tt> with the step size.
|
||||
|
||||
This method is equivalent to the parallel execution of the following loop:
|
||||
|
||||
@code{.cpp}
|
||||
// case 1: step size is positive
|
||||
for(auto i=first; i<last; i+=step) {
|
||||
callable(i);
|
||||
}
|
||||
|
||||
// case 2: step size is negative
|
||||
for(auto i=first, i>last; i+=step) {
|
||||
callable(i);
|
||||
}
|
||||
@endcode
|
||||
|
||||
Arguments are templated to enable stateful passing using std::reference_wrapper.
|
||||
The callable needs to take a single argument of the integral index type.
|
||||
|
||||
Please refer to @ref ParallelIterations for details.
|
||||
*/
|
||||
template <typename B, typename E, typename S, typename C>
|
||||
Task for_each_index(B&& first, E&& last, S&& step, C callable);
|
||||
|
||||
// ------------------------------------------------------------------------
|
||||
// reduction
|
||||
// ------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@brief constructs a STL-styled parallel-reduce task
|
||||
|
||||
@tparam B beginning iterator type
|
||||
@tparam E ending iterator type
|
||||
@tparam T result type
|
||||
@tparam O binary reducer type
|
||||
|
||||
@param first iterator to the beginning (inclusive)
|
||||
@param last iterator to the end (exclusive)
|
||||
@param init initial value of the reduction and the storage for the reduced result
|
||||
@param bop binary operator that will be applied
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
The task spawns a subflow to perform parallel reduction over @c init and the elements in the range <tt>[first, last)</tt>. The reduced result is store in @c init.
|
||||
|
||||
This method is equivalent to the parallel execution of the following loop:
|
||||
|
||||
@code{.cpp}
|
||||
for(auto itr=first; itr!=last; itr++) {
|
||||
init = bop(init, *itr);
|
||||
}
|
||||
@endcode
|
||||
|
||||
Arguments are templated to enable stateful passing using std::reference_wrapper.
|
||||
|
||||
Please refer to @ref ParallelReduction for details.
|
||||
*/
|
||||
template <typename B, typename E, typename T, typename O>
|
||||
Task reduce(B&& first, E&& last, T& init, O bop);
|
||||
|
||||
// ------------------------------------------------------------------------
|
||||
// transfrom and reduction
|
||||
// ------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@brief constructs a STL-styled parallel transform-reduce task
|
||||
|
||||
@tparam B beginning iterator type
|
||||
@tparam E ending iterator type
|
||||
@tparam T result type
|
||||
@tparam BOP binary reducer type
|
||||
@tparam UOP unary transformion type
|
||||
|
||||
@param first iterator to the beginning (inclusive)
|
||||
@param last iterator to the end (exclusive)
|
||||
@param init initial value of the reduction and the storage for the reduced result
|
||||
@param bop binary operator that will be applied in unspecified order to the results of @c uop
|
||||
@param uop unary operator that will be applied to transform each element in the range to the result type
|
||||
|
||||
@return a tf::Task handle
|
||||
|
||||
The task spawns a subflow to perform parallel reduction over @c init and the transformed elements in the range <tt>[first, last)</tt>.
|
||||
The reduced result is store in @c init.
|
||||
|
||||
This method is equivalent to the parallel execution of the following loop:
|
||||
|
||||
@code{.cpp}
|
||||
for(auto itr=first; itr!=last; itr++) {
|
||||
init = bop(init, uop(*itr));
|
||||
}
|
||||
@endcode
|
||||
|
||||
Arguments are templated to enable stateful passing using std::reference_wrapper.
|
||||
|
||||
Please refer to @ref ParallelReduction for details.
|
||||
*/
|
||||
template <typename B, typename E, typename T, typename BOP, typename UOP>
|
||||
Task transform_reduce(B&& first, E&& last, T& init, BOP bop, UOP uop);
|
||||
|
||||
// ------------------------------------------------------------------------
|
||||
// sort
|
||||
// ------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@brief constructs a dynamic task to perform STL-styled parallel sort
|
||||
|
||||
@tparam B beginning iterator type (random-accessible)
|
||||
@tparam E ending iterator type (random-accessible)
|
||||
@tparam C comparator type
|
||||
|
||||
@param first iterator to the beginning (inclusive)
|
||||
@param last iterator to the end (exclusive)
|
||||
@param cmp comparison function object
|
||||
|
||||
The task spawns a subflow to parallelly sort elements in the range
|
||||
<tt>[first, last)</tt>.
|
||||
|
||||
Arguments are templated to enable stateful passing using std::reference_wrapper.
|
||||
|
||||
Please refer to @ref ParallelSort for details.
|
||||
*/
|
||||
template <typename B, typename E, typename C>
|
||||
Task sort(B&& first, E&& last, C cmp);
|
||||
|
||||
/**
|
||||
@brief constructs a dynamic task to perform STL-styled parallel sort using
|
||||
the @c std::less<T> comparator, where @c T is the element type
|
||||
|
||||
@tparam B beginning iterator type (random-accessible)
|
||||
@tparam E ending iterator type (random-accessible)
|
||||
|
||||
@param first iterator to the beginning (inclusive)
|
||||
@param last iterator to the end (exclusive)
|
||||
|
||||
The task spawns a subflow to parallelly sort elements in the range
|
||||
<tt>[first, last)</tt> using the @c std::less<T> comparator,
|
||||
where @c T is the dereferenced iterator type.
|
||||
|
||||
Arguments are templated to enable stateful passing using std::reference_wrapper.
|
||||
|
||||
Please refer to @ref ParallelSort for details.
|
||||
*/
|
||||
template <typename B, typename E>
|
||||
Task sort(B&& first, E&& last);
|
||||
|
||||
protected:
|
||||
|
||||
/**
|
||||
@brief constructs a flow builder with a graph
|
||||
*/
|
||||
FlowBuilder(Graph& graph);
|
||||
|
||||
/**
|
||||
@brief associated graph object
|
||||
*/
|
||||
Graph& _graph;
|
||||
|
||||
private:
|
||||
|
||||
template <typename L>
|
||||
void _linearize(L&);
|
||||
};
|
||||
|
||||
// Constructor
|
||||
inline FlowBuilder::FlowBuilder(Graph& graph) :
|
||||
_graph {graph} {
|
||||
}
|
||||
|
||||
// Function: emplace
|
||||
template <typename C, std::enable_if_t<is_static_task_v<C>, void>*>
|
||||
Task FlowBuilder::emplace(C&& c) {
|
||||
return Task(_graph.emplace_back(
|
||||
std::in_place_type_t<Node::Static>{}, std::forward<C>(c)
|
||||
));
|
||||
}
|
||||
|
||||
// Function: emplace
|
||||
template <typename C, std::enable_if_t<is_dynamic_task_v<C>, void>*>
|
||||
Task FlowBuilder::emplace(C&& c) {
|
||||
return Task(_graph.emplace_back(
|
||||
std::in_place_type_t<Node::Dynamic>{}, std::forward<C>(c)
|
||||
));
|
||||
}
|
||||
|
||||
// Function: emplace
|
||||
template <typename C, std::enable_if_t<is_condition_task_v<C>, void>*>
|
||||
Task FlowBuilder::emplace(C&& c) {
|
||||
return Task(_graph.emplace_back(
|
||||
std::in_place_type_t<Node::Condition>{}, std::forward<C>(c)
|
||||
));
|
||||
}
|
||||
|
||||
// Function: emplace
|
||||
template <typename... C, std::enable_if_t<(sizeof...(C)>1), void>*>
|
||||
auto FlowBuilder::emplace(C&&... cs) {
|
||||
return std::make_tuple(emplace(std::forward<C>(cs))...);
|
||||
}
|
||||
|
||||
// Function: erase
|
||||
inline void FlowBuilder::erase(Task task) {
|
||||
|
||||
if (!task._node) {
|
||||
return;
|
||||
}
|
||||
|
||||
task.for_each_dependent([&] (Task dependent) {
|
||||
auto& S = dependent._node->_successors;
|
||||
if(auto I = std::find(S.begin(), S.end(), task._node); I != S.end()) {
|
||||
S.erase(I);
|
||||
}
|
||||
});
|
||||
|
||||
task.for_each_successor([&] (Task dependent) {
|
||||
auto& D = dependent._node->_dependents;
|
||||
if(auto I = std::find(D.begin(), D.end(), task._node); I != D.end()) {
|
||||
D.erase(I);
|
||||
}
|
||||
});
|
||||
|
||||
_graph.erase(task._node);
|
||||
}
|
||||
|
||||
// Function: composed_of
|
||||
inline Task FlowBuilder::composed_of(Taskflow& taskflow) {
|
||||
auto node = _graph.emplace_back(
|
||||
std::in_place_type_t<Node::Module>{}, &taskflow
|
||||
);
|
||||
return Task(node);
|
||||
}
|
||||
|
||||
// Function: placeholder
|
||||
inline Task FlowBuilder::placeholder() {
|
||||
auto node = _graph.emplace_back();
|
||||
return Task(node);
|
||||
}
|
||||
|
||||
// Procedure: _linearize
|
||||
template <typename L>
|
||||
void FlowBuilder::_linearize(L& keys) {
|
||||
|
||||
auto itr = keys.begin();
|
||||
auto end = keys.end();
|
||||
|
||||
if(itr == end) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto nxt = itr;
|
||||
|
||||
for(++nxt; nxt != end; ++nxt, ++itr) {
|
||||
itr->_node->_precede(nxt->_node);
|
||||
}
|
||||
}
|
||||
|
||||
// Procedure: linearize
|
||||
inline void FlowBuilder::linearize(std::vector<Task>& keys) {
|
||||
_linearize(keys);
|
||||
}
|
||||
|
||||
// Procedure: linearize
|
||||
inline void FlowBuilder::linearize(std::initializer_list<Task> keys) {
|
||||
_linearize(keys);
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@class Subflow
|
||||
|
||||
@brief class to construct a subflow graph from the execution of a dynamic task
|
||||
|
||||
By default, a subflow automatically @em joins its parent node.
|
||||
You may explicitly join or detach a subflow by calling tf::Subflow::join
|
||||
or tf::Subflow::detach, respectively.
|
||||
The following example creates a taskflow graph that spawns a subflow from
|
||||
the execution of task @c B, and the subflow contains three tasks, @c B1,
|
||||
@c B2, and @c B3, where @c B3 runs after @c B1 and @c B2.
|
||||
|
||||
@code{.cpp}
|
||||
// create three regular tasks
|
||||
tf::Task A = taskflow.emplace([](){}).name("A");
|
||||
tf::Task C = taskflow.emplace([](){}).name("C");
|
||||
tf::Task D = taskflow.emplace([](){}).name("D");
|
||||
|
||||
// create a subflow graph (dynamic tasking)
|
||||
tf::Task B = taskflow.emplace([] (tf::Subflow& subflow) {
|
||||
tf::Task B1 = subflow.emplace([](){}).name("B1");
|
||||
tf::Task B2 = subflow.emplace([](){}).name("B2");
|
||||
tf::Task B3 = subflow.emplace([](){}).name("B3");
|
||||
B1.precede(B3);
|
||||
B2.precede(B3);
|
||||
}).name("B");
|
||||
|
||||
A.precede(B); // B runs after A
|
||||
A.precede(C); // C runs after A
|
||||
B.precede(D); // D runs after B
|
||||
C.precede(D); // D runs after C
|
||||
@endcode
|
||||
|
||||
*/
|
||||
class Subflow : public FlowBuilder {
|
||||
|
||||
friend class Executor;
|
||||
friend class FlowBuilder;
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
@brief enables the subflow to join its parent task
|
||||
|
||||
Performs an immediate action to join the subflow. Once the subflow is joined,
|
||||
it is considered finished and you may not modify the subflow anymore.
|
||||
*/
|
||||
void join();
|
||||
|
||||
/**
|
||||
@brief enables the subflow to detach from its parent task
|
||||
|
||||
Performs an immediate action to detach the subflow. Once the subflow is detached,
|
||||
it is considered finished and you may not modify the subflow anymore.
|
||||
*/
|
||||
void detach();
|
||||
|
||||
/**
|
||||
@brief queries if the subflow is joinable
|
||||
|
||||
When a subflow is joined or detached, it becomes not joinable.
|
||||
*/
|
||||
bool joinable() const;
|
||||
|
||||
/**
|
||||
@brief runs a given function asynchronously
|
||||
|
||||
@tparam F callable type
|
||||
@tparam ArgsT parameter types
|
||||
|
||||
@param f callable object to call
|
||||
@param args parameters to pass to the callable
|
||||
|
||||
@return a tf::Future that will holds the result of the execution
|
||||
|
||||
This method is thread-safe and can be called by multiple tasks in the
|
||||
subflow at the same time.
|
||||
The difference to tf::Executor::async is that the created asynchronous task
|
||||
pertains to the subflow.
|
||||
When the subflow joins, all asynchronous tasks created from the subflow
|
||||
are guaranteed to finish before the join.
|
||||
For example:
|
||||
|
||||
@code{.cpp}
|
||||
std::atomic<int> counter(0);
|
||||
taskflow.empalce([&](tf::Subflow& sf){
|
||||
for(int i=0; i<100; i++) {
|
||||
sf.async([&](){ counter++; });
|
||||
}
|
||||
sf.join();
|
||||
assert(counter == 100);
|
||||
});
|
||||
@endcode
|
||||
|
||||
You cannot create asynchronous tasks from a detached subflow.
|
||||
Doing this results in undefined behavior.
|
||||
*/
|
||||
template <typename F, typename... ArgsT>
|
||||
auto async(F&& f, ArgsT&&... args);
|
||||
|
||||
/**
|
||||
@brief similar to tf::Subflow::async but did not return a future object
|
||||
*/
|
||||
template <typename F, typename... ArgsT>
|
||||
void silent_async(F&& f, ArgsT&&... args);
|
||||
|
||||
private:
|
||||
|
||||
Subflow(Executor&, Node*, Graph&);
|
||||
|
||||
Executor& _executor;
|
||||
Node* _parent;
|
||||
|
||||
bool _joinable {true};
|
||||
};
|
||||
|
||||
// Constructor
|
||||
inline Subflow::Subflow(Executor& executor, Node* parent, Graph& graph) :
|
||||
FlowBuilder {graph},
|
||||
_executor {executor},
|
||||
_parent {parent} {
|
||||
}
|
||||
|
||||
// Function: joined
|
||||
inline bool Subflow::joinable() const {
|
||||
return _joinable;
|
||||
}
|
||||
|
||||
} // end of namespace tf. ---------------------------------------------------
|
||||
|
||||
|
||||
572
cpp_to_py/gpugr/taskflow/core/graph.hpp
Normal file
572
cpp_to_py/gpugr/taskflow/core/graph.hpp
Normal file
@ -0,0 +1,572 @@
|
||||
#pragma once
|
||||
|
||||
#include "../utility/iterator.hpp"
|
||||
#include "../utility/object_pool.hpp"
|
||||
#include "../utility/traits.hpp"
|
||||
#include "../utility/singleton.hpp"
|
||||
#include "../utility/os.hpp"
|
||||
#include "../utility/math.hpp"
|
||||
#include "../utility/small_vector.hpp"
|
||||
#include "../utility/serializer.hpp"
|
||||
#include "error.hpp"
|
||||
#include "declarations.hpp"
|
||||
#include "semaphore.hpp"
|
||||
#include "environment.hpp"
|
||||
#include "topology.hpp"
|
||||
|
||||
namespace tf {
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Class: CustomGraphBase
|
||||
// ----------------------------------------------------------------------------
|
||||
class CustomGraphBase {
|
||||
|
||||
public:
|
||||
|
||||
virtual void dump(std::ostream&, const void*, const std::string&) const = 0;
|
||||
virtual ~CustomGraphBase() = default;
|
||||
};
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Class: Graph
|
||||
// ----------------------------------------------------------------------------
|
||||
class Graph {
|
||||
|
||||
friend class Node;
|
||||
friend class Taskflow;
|
||||
friend class Executor;
|
||||
friend class Sanitizer;
|
||||
|
||||
public:
|
||||
|
||||
Graph() = default;
|
||||
Graph(const Graph&) = delete;
|
||||
Graph(Graph&&);
|
||||
|
||||
~Graph();
|
||||
|
||||
Graph& operator = (const Graph&) = delete;
|
||||
Graph& operator = (Graph&&);
|
||||
|
||||
void clear();
|
||||
void clear_detached();
|
||||
void merge(Graph&&);
|
||||
|
||||
bool empty() const;
|
||||
|
||||
size_t size() const;
|
||||
|
||||
template <typename ...Args>
|
||||
Node* emplace_back(Args&& ...);
|
||||
|
||||
Node* emplace_back();
|
||||
|
||||
void erase(Node*);
|
||||
|
||||
private:
|
||||
|
||||
std::vector<Node*> _nodes;
|
||||
};
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Class: Node
|
||||
class Node {
|
||||
|
||||
friend class Graph;
|
||||
friend class Task;
|
||||
friend class TaskView;
|
||||
friend class Taskflow;
|
||||
friend class Executor;
|
||||
friend class FlowBuilder;
|
||||
friend class Subflow;
|
||||
friend class Sanitizer;
|
||||
|
||||
TF_ENABLE_POOLABLE_ON_THIS;
|
||||
|
||||
// state bit flag
|
||||
constexpr static int BRANCHED = 0x1;
|
||||
constexpr static int DETACHED = 0x2;
|
||||
constexpr static int ACQUIRED = 0x4;
|
||||
constexpr static int READY = 0x8;
|
||||
|
||||
// static work handle
|
||||
struct Static {
|
||||
|
||||
template <typename C>
|
||||
Static(C&&);
|
||||
|
||||
std::function<void()> work;
|
||||
};
|
||||
|
||||
// dynamic work handle
|
||||
struct Dynamic {
|
||||
|
||||
template <typename C>
|
||||
Dynamic(C&&);
|
||||
|
||||
std::function<void(Subflow&)> work;
|
||||
Graph subgraph;
|
||||
};
|
||||
|
||||
// condition work handle
|
||||
struct Condition {
|
||||
|
||||
template <typename C>
|
||||
Condition(C&&);
|
||||
|
||||
std::function<int()> work;
|
||||
};
|
||||
|
||||
// module work handle
|
||||
struct Module {
|
||||
|
||||
template <typename T>
|
||||
Module(T&&);
|
||||
|
||||
Taskflow* module {nullptr};
|
||||
};
|
||||
|
||||
// Async work
|
||||
struct Async {
|
||||
|
||||
template <typename T>
|
||||
Async(T&&, std::shared_ptr<AsyncTopology>);
|
||||
|
||||
std::function<void(bool)> work;
|
||||
|
||||
std::shared_ptr<AsyncTopology> topology;
|
||||
};
|
||||
|
||||
// Silent async work
|
||||
struct SilentAsync {
|
||||
|
||||
template <typename C>
|
||||
SilentAsync(C&&);
|
||||
|
||||
std::function<void()> work;
|
||||
};
|
||||
|
||||
// cudaFlow work handle
|
||||
struct cudaFlow {
|
||||
|
||||
template <typename C, typename G>
|
||||
cudaFlow(C&& c, G&& g);
|
||||
|
||||
std::function<void(Executor&, Node*)> work;
|
||||
|
||||
std::unique_ptr<CustomGraphBase> graph;
|
||||
};
|
||||
|
||||
// syclFlow work handle
|
||||
struct syclFlow {
|
||||
|
||||
template <typename C, typename G>
|
||||
syclFlow(C&& c, G&& g);
|
||||
|
||||
std::function<void(Executor&, Node*)> work;
|
||||
|
||||
std::unique_ptr<CustomGraphBase> graph;
|
||||
};
|
||||
|
||||
using handle_t = std::variant<
|
||||
std::monostate, // placeholder
|
||||
Static, // static tasking
|
||||
Dynamic, // dynamic tasking
|
||||
Condition, // conditional tasking
|
||||
Module, // composable tasking
|
||||
Async, // async tasking
|
||||
SilentAsync, // async tasking (no future)
|
||||
cudaFlow, // cudaFlow
|
||||
syclFlow // syclFlow
|
||||
>;
|
||||
|
||||
struct Semaphores {
|
||||
std::vector<Semaphore*> to_acquire;
|
||||
std::vector<Semaphore*> to_release;
|
||||
};
|
||||
|
||||
public:
|
||||
|
||||
// variant index
|
||||
constexpr static auto PLACEHOLDER = get_index_v<std::monostate, handle_t>;
|
||||
constexpr static auto STATIC = get_index_v<Static, handle_t>;
|
||||
constexpr static auto DYNAMIC = get_index_v<Dynamic, handle_t>;
|
||||
constexpr static auto CONDITION = get_index_v<Condition, handle_t>;
|
||||
constexpr static auto MODULE = get_index_v<Module, handle_t>;
|
||||
constexpr static auto ASYNC = get_index_v<Async, handle_t>;
|
||||
constexpr static auto SILENT_ASYNC = get_index_v<SilentAsync, handle_t>;
|
||||
constexpr static auto CUDAFLOW = get_index_v<cudaFlow, handle_t>;
|
||||
constexpr static auto SYCLFLOW = get_index_v<syclFlow, handle_t>;
|
||||
|
||||
template <typename... Args>
|
||||
Node(Args&&... args);
|
||||
|
||||
~Node();
|
||||
|
||||
size_t num_successors() const;
|
||||
size_t num_dependents() const;
|
||||
size_t num_strong_dependents() const;
|
||||
size_t num_weak_dependents() const;
|
||||
|
||||
const std::string& name() const;
|
||||
|
||||
private:
|
||||
|
||||
std::string _name;
|
||||
|
||||
void* _data {nullptr};
|
||||
|
||||
handle_t _handle;
|
||||
|
||||
SmallVector<Node*> _successors;
|
||||
SmallVector<Node*> _dependents;
|
||||
|
||||
std::unique_ptr<Semaphores> _semaphores;
|
||||
|
||||
Topology* _topology {nullptr};
|
||||
|
||||
Node* _parent {nullptr};
|
||||
|
||||
std::atomic<int> _state {0};
|
||||
std::atomic<size_t> _join_counter {0};
|
||||
|
||||
void _precede(Node*);
|
||||
void _set_up_join_counter();
|
||||
|
||||
bool _has_state(int) const;
|
||||
bool _is_cancelled() const;
|
||||
bool _acquire_all(std::vector<Node*>&);
|
||||
|
||||
std::vector<Node*> _release_all();
|
||||
};
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Node Object Pool
|
||||
// ----------------------------------------------------------------------------
|
||||
inline ObjectPool<Node> node_pool;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Definition for Node::Static
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Constructor
|
||||
template <typename C>
|
||||
Node::Static::Static(C&& c) : work {std::forward<C>(c)} {
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Definition for Node::Dynamic
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Constructor
|
||||
template <typename C>
|
||||
Node::Dynamic::Dynamic(C&& c) : work {std::forward<C>(c)} {
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Definition for Node::Condition
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Constructor
|
||||
template <typename C>
|
||||
Node::Condition::Condition(C&& c) : work {std::forward<C>(c)} {
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Definition for Node::cudaFlow
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
template <typename C, typename G>
|
||||
Node::cudaFlow::cudaFlow(C&& c, G&& g) :
|
||||
work {std::forward<C>(c)},
|
||||
graph {std::forward<G>(g)} {
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Definition for Node::syclFlow
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
template <typename C, typename G>
|
||||
Node::syclFlow::syclFlow(C&& c, G&& g) :
|
||||
work {std::forward<C>(c)},
|
||||
graph {std::forward<G>(g)} {
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Definition for Node::Module
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Constructor
|
||||
template <typename T>
|
||||
Node::Module::Module(T&& tf) : module {tf} {
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Definition for Node::Async
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Constructor
|
||||
template <typename C>
|
||||
Node::Async::Async(C&& c, std::shared_ptr<AsyncTopology>tpg) :
|
||||
work {std::forward<C>(c)},
|
||||
topology {std::move(tpg)} {
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Definition for Node::SilentAsync
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Constructor
|
||||
template <typename C>
|
||||
Node::SilentAsync::SilentAsync(C&& c) :
|
||||
work {std::forward<C>(c)} {
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Definition for Node
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Constructor
|
||||
template <typename... Args>
|
||||
Node::Node(Args&&... args): _handle{std::forward<Args>(args)...} {
|
||||
}
|
||||
|
||||
// Destructor
|
||||
inline Node::~Node() {
|
||||
// this is to avoid stack overflow
|
||||
|
||||
if(_handle.index() == DYNAMIC) {
|
||||
|
||||
auto& subgraph = std::get<Dynamic>(_handle).subgraph;
|
||||
|
||||
std::vector<Node*> nodes;
|
||||
nodes.reserve(subgraph.size());
|
||||
|
||||
std::move(
|
||||
subgraph._nodes.begin(), subgraph._nodes.end(), std::back_inserter(nodes)
|
||||
);
|
||||
subgraph._nodes.clear();
|
||||
|
||||
size_t i = 0;
|
||||
|
||||
while(i < nodes.size()) {
|
||||
|
||||
if(nodes[i]->_handle.index() == DYNAMIC) {
|
||||
|
||||
auto& sbg = std::get<Dynamic>(nodes[i]->_handle).subgraph;
|
||||
std::move(
|
||||
sbg._nodes.begin(), sbg._nodes.end(), std::back_inserter(nodes)
|
||||
);
|
||||
sbg._nodes.clear();
|
||||
}
|
||||
|
||||
++i;
|
||||
}
|
||||
|
||||
//auto& np = Graph::_node_pool();
|
||||
for(i=0; i<nodes.size(); ++i) {
|
||||
node_pool.recycle(nodes[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Procedure: _precede
|
||||
inline void Node::_precede(Node* v) {
|
||||
_successors.push_back(v);
|
||||
v->_dependents.push_back(this);
|
||||
}
|
||||
|
||||
// Function: num_successors
|
||||
inline size_t Node::num_successors() const {
|
||||
return _successors.size();
|
||||
}
|
||||
|
||||
// Function: dependents
|
||||
inline size_t Node::num_dependents() const {
|
||||
return _dependents.size();
|
||||
}
|
||||
|
||||
// Function: num_weak_dependents
|
||||
inline size_t Node::num_weak_dependents() const {
|
||||
size_t n = 0;
|
||||
for(size_t i=0; i<_dependents.size(); i++) {
|
||||
if(_dependents[i]->_handle.index() == Node::CONDITION) {
|
||||
n++;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
// Function: num_strong_dependents
|
||||
inline size_t Node::num_strong_dependents() const {
|
||||
size_t n = 0;
|
||||
for(size_t i=0; i<_dependents.size(); i++) {
|
||||
if(_dependents[i]->_handle.index() != Node::CONDITION) {
|
||||
n++;
|
||||
}
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
// Function: name
|
||||
inline const std::string& Node::name() const {
|
||||
return _name;
|
||||
}
|
||||
|
||||
// Function: _is_cancelled
|
||||
inline bool Node::_is_cancelled() const {
|
||||
if(_handle.index() == Node::ASYNC) {
|
||||
auto& h = std::get<Node::Async>(_handle);
|
||||
if(h.topology && h.topology->_is_cancelled) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
// async tasks spawned from subflow does not have topology
|
||||
return _topology && _topology->_is_cancelled;
|
||||
}
|
||||
|
||||
// Procedure: _set_up_join_counter
|
||||
inline void Node::_set_up_join_counter() {
|
||||
|
||||
size_t c = 0;
|
||||
|
||||
for(auto p : _dependents) {
|
||||
if(p->_handle.index() == Node::CONDITION) {
|
||||
//_set_state(Node::BRANCHED);
|
||||
_state.fetch_or(Node::BRANCHED, std::memory_order_relaxed);
|
||||
}
|
||||
else {
|
||||
c++;
|
||||
}
|
||||
}
|
||||
|
||||
_join_counter.store(c, std::memory_order_release);
|
||||
}
|
||||
|
||||
|
||||
// Function: _acquire_all
|
||||
inline bool Node::_acquire_all(std::vector<Node*>& nodes) {
|
||||
|
||||
auto& to_acquire = _semaphores->to_acquire;
|
||||
|
||||
for(size_t i = 0; i < to_acquire.size(); ++i) {
|
||||
if(!to_acquire[i]->_try_acquire_or_wait(this)) {
|
||||
for(size_t j = 1; j <= i; ++j) {
|
||||
auto r = to_acquire[i-j]->_release();
|
||||
nodes.insert(end(nodes), begin(r), end(r));
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// Function: _release_all
|
||||
inline std::vector<Node*> Node::_release_all() {
|
||||
|
||||
auto& to_release = _semaphores->to_release;
|
||||
|
||||
std::vector<Node*> nodes;
|
||||
for(const auto& sem : to_release) {
|
||||
auto r = sem->_release();
|
||||
nodes.insert(end(nodes), begin(r), end(r));
|
||||
}
|
||||
return nodes;
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Graph definition
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
//// Function: _node_pool
|
||||
//inline ObjectPool<Node>& Graph::_node_pool() {
|
||||
// static ObjectPool<Node> pool;
|
||||
// return pool;
|
||||
//}
|
||||
|
||||
// Destructor
|
||||
inline Graph::~Graph() {
|
||||
clear();
|
||||
}
|
||||
|
||||
// Move constructor
|
||||
inline Graph::Graph(Graph&& other) :
|
||||
_nodes {std::move(other._nodes)} {
|
||||
}
|
||||
|
||||
// Move assignment
|
||||
inline Graph& Graph::operator = (Graph&& other) {
|
||||
clear();
|
||||
_nodes = std::move(other._nodes);
|
||||
return *this;
|
||||
}
|
||||
|
||||
// Procedure: clear
|
||||
inline void Graph::clear() {
|
||||
for(auto node : _nodes) {
|
||||
node_pool.recycle(node);
|
||||
}
|
||||
_nodes.clear();
|
||||
}
|
||||
|
||||
// Procedure: clear_detached
|
||||
inline void Graph::clear_detached() {
|
||||
|
||||
auto mid = std::partition(_nodes.begin(), _nodes.end(), [] (Node* node) {
|
||||
return !(node->_state.load(std::memory_order_relaxed) & Node::DETACHED);
|
||||
});
|
||||
|
||||
for(auto itr = mid; itr != _nodes.end(); ++itr) {
|
||||
node_pool.recycle(*itr);
|
||||
}
|
||||
_nodes.resize(std::distance(_nodes.begin(), mid));
|
||||
}
|
||||
|
||||
// Procedure: merge
|
||||
inline void Graph::merge(Graph&& g) {
|
||||
for(auto n : g._nodes) {
|
||||
_nodes.push_back(n);
|
||||
}
|
||||
g._nodes.clear();
|
||||
}
|
||||
|
||||
// Function: size
|
||||
inline size_t Graph::size() const {
|
||||
return _nodes.size();
|
||||
}
|
||||
|
||||
// Function: empty
|
||||
inline bool Graph::empty() const {
|
||||
return _nodes.empty();
|
||||
}
|
||||
|
||||
// Function: emplace_back
|
||||
// create a node from a give argument; constructor is called if necessary
|
||||
template <typename ...ArgsT>
|
||||
Node* Graph::emplace_back(ArgsT&&... args) {
|
||||
_nodes.push_back(node_pool.animate(std::forward<ArgsT>(args)...));
|
||||
return _nodes.back();
|
||||
}
|
||||
|
||||
// Function: emplace_back
|
||||
// create a node from a give argument; constructor is called if necessary
|
||||
inline Node* Graph::emplace_back() {
|
||||
_nodes.push_back(node_pool.animate());
|
||||
return _nodes.back();
|
||||
}
|
||||
|
||||
// Function: erase
|
||||
inline void Graph::erase(Node* node) {
|
||||
if(auto I = std::find(_nodes.begin(), _nodes.end(), node); I != _nodes.end()) {
|
||||
_nodes.erase(I);
|
||||
node_pool.recycle(node);
|
||||
}
|
||||
}
|
||||
|
||||
} // end of namespace tf. ---------------------------------------------------
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
267
cpp_to_py/gpugr/taskflow/core/notifier.hpp
Normal file
267
cpp_to_py/gpugr/taskflow/core/notifier.hpp
Normal file
@ -0,0 +1,267 @@
|
||||
// 2019/02/09 - created by Tsung-Wei Huang
|
||||
// - modified the event count from Eigen
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <iostream>
|
||||
#include <vector>
|
||||
#include <cstdlib>
|
||||
#include <cstdio>
|
||||
#include <atomic>
|
||||
#include <memory>
|
||||
#include <deque>
|
||||
#include <mutex>
|
||||
#include <condition_variable>
|
||||
#include <thread>
|
||||
#include <algorithm>
|
||||
#include <numeric>
|
||||
#include <cassert>
|
||||
|
||||
// This file is part of Eigen, a lightweight C++ template library
|
||||
// for linear algebra.
|
||||
//
|
||||
// Copyright (C) 2016 Dmitry Vyukov <dvyukov@google.com>
|
||||
//
|
||||
// This Source Code Form is subject to the terms of the Mozilla
|
||||
// Public License v. 2.0. If a copy of the MPL was not distributed
|
||||
// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
namespace tf {
|
||||
|
||||
// Notifier allows to wait for arbitrary predicates in non-blocking
|
||||
// algorithms. Think of condition variable, but wait predicate does not need to
|
||||
// be protected by a mutex. Usage:
|
||||
// Waiting thread does:
|
||||
//
|
||||
// if (predicate)
|
||||
// return act();
|
||||
// Notifier::Waiter& w = waiters[my_index];
|
||||
// ec.prepare_wait(&w);
|
||||
// if (predicate) {
|
||||
// ec.cancel_wait(&w);
|
||||
// return act();
|
||||
// }
|
||||
// ec.commit_wait(&w);
|
||||
//
|
||||
// Notifying thread does:
|
||||
//
|
||||
// predicate = true;
|
||||
// ec.notify(true);
|
||||
//
|
||||
// notify is cheap if there are no waiting threads. prepare_wait/commit_wait are not
|
||||
// cheap, but they are executed only if the preceeding predicate check has
|
||||
// failed.
|
||||
//
|
||||
// Algorihtm outline:
|
||||
// There are two main variables: predicate (managed by user) and _state.
|
||||
// Operation closely resembles Dekker mutual algorithm:
|
||||
// https://en.wikipedia.org/wiki/Dekker%27s_algorithm
|
||||
// Waiting thread sets _state then checks predicate, Notifying thread sets
|
||||
// predicate then checks _state. Due to seq_cst fences in between these
|
||||
// operations it is guaranteed than either waiter will see predicate change
|
||||
// and won't block, or notifying thread will see _state change and will unblock
|
||||
// the waiter, or both. But it can't happen that both threads don't see each
|
||||
// other changes, which would lead to deadlock.
|
||||
class Notifier {
|
||||
|
||||
friend class Executor;
|
||||
|
||||
public:
|
||||
|
||||
struct Waiter {
|
||||
std::atomic<Waiter*> next;
|
||||
std::mutex mu;
|
||||
std::condition_variable cv;
|
||||
uint64_t epoch;
|
||||
unsigned state;
|
||||
enum {
|
||||
kNotSignaled,
|
||||
kWaiting,
|
||||
kSignaled,
|
||||
};
|
||||
};
|
||||
|
||||
explicit Notifier(size_t N) : _waiters{N} {
|
||||
assert(_waiters.size() < (1 << kWaiterBits) - 1);
|
||||
// Initialize epoch to something close to overflow to test overflow.
|
||||
_state = kStackMask | (kEpochMask - kEpochInc * _waiters.size() * 2);
|
||||
}
|
||||
|
||||
~Notifier() {
|
||||
// Ensure there are no waiters.
|
||||
assert((_state.load() & (kStackMask | kWaiterMask)) == kStackMask);
|
||||
}
|
||||
|
||||
// prepare_wait prepares for waiting.
|
||||
// After calling this function the thread must re-check the wait predicate
|
||||
// and call either cancel_wait or commit_wait passing the same Waiter object.
|
||||
void prepare_wait(Waiter* w) {
|
||||
w->epoch = _state.fetch_add(kWaiterInc, std::memory_order_relaxed);
|
||||
std::atomic_thread_fence(std::memory_order_seq_cst);
|
||||
}
|
||||
|
||||
// commit_wait commits waiting.
|
||||
void commit_wait(Waiter* w) {
|
||||
w->state = Waiter::kNotSignaled;
|
||||
// Modification epoch of this waiter.
|
||||
uint64_t epoch =
|
||||
(w->epoch & kEpochMask) +
|
||||
(((w->epoch & kWaiterMask) >> kWaiterShift) << kEpochShift);
|
||||
uint64_t state = _state.load(std::memory_order_seq_cst);
|
||||
for (;;) {
|
||||
if (int64_t((state & kEpochMask) - epoch) < 0) {
|
||||
// The preceeding waiter has not decided on its fate. Wait until it
|
||||
// calls either cancel_wait or commit_wait, or is notified.
|
||||
std::this_thread::yield();
|
||||
state = _state.load(std::memory_order_seq_cst);
|
||||
continue;
|
||||
}
|
||||
// We've already been notified.
|
||||
if (int64_t((state & kEpochMask) - epoch) > 0) return;
|
||||
// Remove this thread from prewait counter and add it to the waiter list.
|
||||
assert((state & kWaiterMask) != 0);
|
||||
uint64_t newstate = state - kWaiterInc + kEpochInc;
|
||||
//newstate = (newstate & ~kStackMask) | (w - &_waiters[0]);
|
||||
newstate = static_cast<uint64_t>((newstate & ~kStackMask) | static_cast<uint64_t>(w - &_waiters[0]));
|
||||
if ((state & kStackMask) == kStackMask)
|
||||
w->next.store(nullptr, std::memory_order_relaxed);
|
||||
else
|
||||
w->next.store(&_waiters[state & kStackMask], std::memory_order_relaxed);
|
||||
if (_state.compare_exchange_weak(state, newstate,
|
||||
std::memory_order_release))
|
||||
break;
|
||||
}
|
||||
_park(w);
|
||||
}
|
||||
|
||||
// cancel_wait cancels effects of the previous prepare_wait call.
|
||||
void cancel_wait(Waiter* w) {
|
||||
uint64_t epoch =
|
||||
(w->epoch & kEpochMask) +
|
||||
(((w->epoch & kWaiterMask) >> kWaiterShift) << kEpochShift);
|
||||
uint64_t state = _state.load(std::memory_order_relaxed);
|
||||
for (;;) {
|
||||
if (int64_t((state & kEpochMask) - epoch) < 0) {
|
||||
// The preceeding waiter has not decided on its fate. Wait until it
|
||||
// calls either cancel_wait or commit_wait, or is notified.
|
||||
std::this_thread::yield();
|
||||
state = _state.load(std::memory_order_relaxed);
|
||||
continue;
|
||||
}
|
||||
// We've already been notified.
|
||||
if (int64_t((state & kEpochMask) - epoch) > 0) return;
|
||||
// Remove this thread from prewait counter.
|
||||
assert((state & kWaiterMask) != 0);
|
||||
if (_state.compare_exchange_weak(state, state - kWaiterInc + kEpochInc,
|
||||
std::memory_order_relaxed))
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
// notify wakes one or all waiting threads.
|
||||
// Must be called after changing the associated wait predicate.
|
||||
void notify(bool all) {
|
||||
std::atomic_thread_fence(std::memory_order_seq_cst);
|
||||
uint64_t state = _state.load(std::memory_order_acquire);
|
||||
for (;;) {
|
||||
// Easy case: no waiters.
|
||||
if ((state & kStackMask) == kStackMask && (state & kWaiterMask) == 0)
|
||||
return;
|
||||
uint64_t waiters = (state & kWaiterMask) >> kWaiterShift;
|
||||
uint64_t newstate;
|
||||
if (all) {
|
||||
// Reset prewait counter and empty wait list.
|
||||
newstate = (state & kEpochMask) + (kEpochInc * waiters) + kStackMask;
|
||||
} else if (waiters) {
|
||||
// There is a thread in pre-wait state, unblock it.
|
||||
newstate = state + kEpochInc - kWaiterInc;
|
||||
} else {
|
||||
// Pop a waiter from list and unpark it.
|
||||
Waiter* w = &_waiters[state & kStackMask];
|
||||
Waiter* wnext = w->next.load(std::memory_order_relaxed);
|
||||
uint64_t next = kStackMask;
|
||||
//if (wnext != nullptr) next = wnext - &_waiters[0];
|
||||
if (wnext != nullptr) next = static_cast<uint64_t>(wnext - &_waiters[0]);
|
||||
// Note: we don't add kEpochInc here. ABA problem on the lock-free stack
|
||||
// can't happen because a waiter is re-pushed onto the stack only after
|
||||
// it was in the pre-wait state which inevitably leads to epoch
|
||||
// increment.
|
||||
newstate = (state & kEpochMask) + next;
|
||||
}
|
||||
if (_state.compare_exchange_weak(state, newstate,
|
||||
std::memory_order_acquire)) {
|
||||
if (!all && waiters) return; // unblocked pre-wait thread
|
||||
if ((state & kStackMask) == kStackMask) return;
|
||||
Waiter* w = &_waiters[state & kStackMask];
|
||||
if (!all) w->next.store(nullptr, std::memory_order_relaxed);
|
||||
_unpark(w);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// notify n workers
|
||||
void notify_n(size_t n) {
|
||||
if(n >= _waiters.size()) {
|
||||
notify(true);
|
||||
}
|
||||
else {
|
||||
for(size_t k=0; k<n; ++k) {
|
||||
notify(false);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
size_t size() const {
|
||||
return _waiters.size();
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
// State_ layout:
|
||||
// - low kStackBits is a stack of waiters committed wait.
|
||||
// - next kWaiterBits is count of waiters in prewait state.
|
||||
// - next kEpochBits is modification counter.
|
||||
static const uint64_t kStackBits = 16;
|
||||
static const uint64_t kStackMask = (1ull << kStackBits) - 1;
|
||||
static const uint64_t kWaiterBits = 16;
|
||||
static const uint64_t kWaiterShift = 16;
|
||||
static const uint64_t kWaiterMask = ((1ull << kWaiterBits) - 1)
|
||||
<< kWaiterShift;
|
||||
static const uint64_t kWaiterInc = 1ull << kWaiterBits;
|
||||
static const uint64_t kEpochBits = 32;
|
||||
static const uint64_t kEpochShift = 32;
|
||||
static const uint64_t kEpochMask = ((1ull << kEpochBits) - 1) << kEpochShift;
|
||||
static const uint64_t kEpochInc = 1ull << kEpochShift;
|
||||
std::atomic<uint64_t> _state;
|
||||
std::vector<Waiter> _waiters;
|
||||
|
||||
void _park(Waiter* w) {
|
||||
std::unique_lock<std::mutex> lock(w->mu);
|
||||
while (w->state != Waiter::kSignaled) {
|
||||
w->state = Waiter::kWaiting;
|
||||
w->cv.wait(lock);
|
||||
}
|
||||
}
|
||||
|
||||
void _unpark(Waiter* waiters) {
|
||||
Waiter* next = nullptr;
|
||||
for (Waiter* w = waiters; w; w = next) {
|
||||
next = w->next.load(std::memory_order_relaxed);
|
||||
unsigned state;
|
||||
{
|
||||
std::unique_lock<std::mutex> lock(w->mu);
|
||||
state = w->state;
|
||||
w->state = Waiter::kSignaled;
|
||||
}
|
||||
// Avoid notifying if it wasn't waiting.
|
||||
if (state == Waiter::kWaiting) w->cv.notify_one();
|
||||
}
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
|
||||
|
||||
} // namespace tf ------------------------------------------------------------
|
||||
|
||||
735
cpp_to_py/gpugr/taskflow/core/observer.hpp
Normal file
735
cpp_to_py/gpugr/taskflow/core/observer.hpp
Normal file
@ -0,0 +1,735 @@
|
||||
#pragma once
|
||||
|
||||
#include "task.hpp"
|
||||
#include "worker.hpp"
|
||||
|
||||
/**
|
||||
@file observer.hpp
|
||||
@brief observer include file
|
||||
*/
|
||||
|
||||
namespace tf {
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// timeline data structure
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@brief default time point type of observers
|
||||
*/
|
||||
using observer_stamp_t = std::chrono::time_point<std::chrono::steady_clock>;
|
||||
|
||||
/**
|
||||
@private
|
||||
*/
|
||||
struct Segment {
|
||||
|
||||
std::string name;
|
||||
TaskType type;
|
||||
|
||||
observer_stamp_t beg;
|
||||
observer_stamp_t end;
|
||||
|
||||
template <typename Archiver>
|
||||
auto save(Archiver& ar) const {
|
||||
return ar(name, type, beg, end);
|
||||
}
|
||||
|
||||
template <typename Archiver>
|
||||
auto load(Archiver& ar) {
|
||||
return ar(name, type, beg, end);
|
||||
}
|
||||
|
||||
Segment() = default;
|
||||
|
||||
Segment(
|
||||
const std::string& n, TaskType t, observer_stamp_t b, observer_stamp_t e
|
||||
) : name {n}, type {t}, beg {b}, end {e} {
|
||||
}
|
||||
|
||||
auto span() const {
|
||||
return end-beg;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
@private
|
||||
*/
|
||||
struct Timeline {
|
||||
|
||||
size_t uid;
|
||||
|
||||
observer_stamp_t origin;
|
||||
std::vector<std::vector<std::vector<Segment>>> segments;
|
||||
|
||||
Timeline() = default;
|
||||
|
||||
Timeline(const Timeline& rhs) = delete;
|
||||
Timeline(Timeline&& rhs) = default;
|
||||
|
||||
Timeline& operator = (const Timeline& rhs) = delete;
|
||||
Timeline& operator = (Timeline&& rhs) = default;
|
||||
|
||||
template <typename Archiver>
|
||||
auto save(Archiver& ar) const {
|
||||
return ar(uid, origin, segments);
|
||||
}
|
||||
|
||||
template <typename Archiver>
|
||||
auto load(Archiver& ar) {
|
||||
return ar(uid, origin, segments);
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
@private
|
||||
*/
|
||||
struct ProfileData {
|
||||
|
||||
std::vector<Timeline> timelines;
|
||||
|
||||
ProfileData() = default;
|
||||
|
||||
ProfileData(const ProfileData& rhs) = delete;
|
||||
ProfileData(ProfileData&& rhs) = default;
|
||||
|
||||
ProfileData& operator = (const ProfileData& rhs) = delete;
|
||||
ProfileData& operator = (ProfileData&&) = default;
|
||||
|
||||
template <typename Archiver>
|
||||
auto save(Archiver& ar) const {
|
||||
return ar(timelines);
|
||||
}
|
||||
|
||||
template <typename Archiver>
|
||||
auto load(Archiver& ar) {
|
||||
return ar(timelines);
|
||||
}
|
||||
};
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// observer interface
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@class: ObserverInterface
|
||||
|
||||
@brief The interface class for creating an executor observer.
|
||||
|
||||
The tf::ObserverInterface class let users define custom methods to monitor
|
||||
the behaviors of an executor. This is particularly useful when you want to
|
||||
inspect the performance of an executor and visualize when each thread
|
||||
participates in the execution of a task.
|
||||
To prevent users from direct access to the internal threads and tasks,
|
||||
tf::ObserverInterface provides immutable wrappers,
|
||||
tf::WorkerView and tf::TaskView, over workers and tasks.
|
||||
|
||||
Please refer to tf::WorkerView and tf::TaskView for details.
|
||||
|
||||
Example usage:
|
||||
|
||||
@code{.cpp}
|
||||
|
||||
struct MyObserver : public tf::ObserverInterface {
|
||||
|
||||
MyObserver(const std::string& name) {
|
||||
std::cout << "constructing observer " << name << '\n';
|
||||
}
|
||||
|
||||
void set_up(size_t num_workers) override final {
|
||||
std::cout << "setting up observer with " << num_workers << " workers\n";
|
||||
}
|
||||
|
||||
void on_entry(WorkerView w, tf::TaskView tv) override final {
|
||||
std::ostringstream oss;
|
||||
oss << "worker " << w.id() << " ready to run " << tv.name() << '\n';
|
||||
std::cout << oss.str();
|
||||
}
|
||||
|
||||
void on_exit(WorkerView w, tf::TaskView tv) override final {
|
||||
std::ostringstream oss;
|
||||
oss << "worker " << w.id() << " finished running " << tv.name() << '\n';
|
||||
std::cout << oss.str();
|
||||
}
|
||||
};
|
||||
|
||||
tf::Taskflow taskflow;
|
||||
tf::Executor executor;
|
||||
|
||||
// insert tasks into taskflow
|
||||
// ...
|
||||
|
||||
// create a custom observer
|
||||
std::shared_ptr<MyObserver> observer = executor.make_observer<MyObserver>("MyObserver");
|
||||
|
||||
// run the taskflow
|
||||
executor.run(taskflow).wait();
|
||||
@endcode
|
||||
*/
|
||||
class ObserverInterface {
|
||||
|
||||
friend class Executor;
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
@brief virtual destructor
|
||||
*/
|
||||
virtual ~ObserverInterface() = default;
|
||||
|
||||
/**
|
||||
@brief constructor-like method to call when the executor observer is fully created
|
||||
@param num_workers the number of the worker threads in the executor
|
||||
*/
|
||||
virtual void set_up(size_t num_workers) = 0;
|
||||
|
||||
/**
|
||||
@brief method to call before a worker thread executes a closure
|
||||
@param w an immutable view of this worker thread
|
||||
@param task_view a constant wrapper object to the task
|
||||
*/
|
||||
virtual void on_entry(WorkerView w, TaskView task_view) = 0;
|
||||
|
||||
/**
|
||||
@brief method to call after a worker thread executed a closure
|
||||
@param w an immutable view of this worker thread
|
||||
@param task_view a constant wrapper object to the task
|
||||
*/
|
||||
virtual void on_exit(WorkerView w, TaskView task_view) = 0;
|
||||
};
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// ChromeObserver definition
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@class: ChromeObserver
|
||||
|
||||
@brief observer interface based on Chrome tracing format
|
||||
|
||||
A tf::ChromeObserver inherits tf::ObserverInterface and defines methods to dump
|
||||
the observed thread activities into a format that can be visualized through
|
||||
@ChromeTracing.
|
||||
|
||||
@code{.cpp}
|
||||
tf::Taskflow taskflow;
|
||||
tf::Executor executor;
|
||||
|
||||
// insert tasks into taskflow
|
||||
// ...
|
||||
|
||||
// create a custom observer
|
||||
std::shared_ptr<tf::ChromeObserver> observer = executor.make_observer<tf::ChromeObserver>();
|
||||
|
||||
// run the taskflow
|
||||
executor.run(taskflow).wait();
|
||||
|
||||
// dump the thread activities to a chrome-tracing format.
|
||||
observer->dump(std::cout);
|
||||
@endcode
|
||||
*/
|
||||
class ChromeObserver : public ObserverInterface {
|
||||
|
||||
friend class Executor;
|
||||
|
||||
// data structure to record each task execution
|
||||
struct Segment {
|
||||
|
||||
std::string name;
|
||||
|
||||
observer_stamp_t beg;
|
||||
observer_stamp_t end;
|
||||
|
||||
Segment(
|
||||
const std::string& n,
|
||||
observer_stamp_t b,
|
||||
observer_stamp_t e
|
||||
);
|
||||
};
|
||||
|
||||
// data structure to store the entire execution timeline
|
||||
struct Timeline {
|
||||
observer_stamp_t origin;
|
||||
std::vector<std::vector<Segment>> segments;
|
||||
std::vector<std::stack<observer_stamp_t>> stacks;
|
||||
};
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
@brief dumps the timelines into a @ChromeTracing format through
|
||||
an output stream
|
||||
*/
|
||||
void dump(std::ostream& ostream) const;
|
||||
|
||||
/**
|
||||
@brief dumps the timelines into a @ChromeTracing format
|
||||
*/
|
||||
inline std::string dump() const;
|
||||
|
||||
/**
|
||||
@brief clears the timeline data
|
||||
*/
|
||||
inline void clear();
|
||||
|
||||
/**
|
||||
@brief queries the number of tasks observed
|
||||
*/
|
||||
inline size_t num_tasks() const;
|
||||
|
||||
private:
|
||||
|
||||
inline void set_up(size_t num_workers) override final;
|
||||
inline void on_entry(WorkerView w, TaskView task_view) override final;
|
||||
inline void on_exit(WorkerView w, TaskView task_view) override final;
|
||||
|
||||
Timeline _timeline;
|
||||
};
|
||||
|
||||
// constructor
|
||||
inline ChromeObserver::Segment::Segment(
|
||||
const std::string& n, observer_stamp_t b, observer_stamp_t e
|
||||
) :
|
||||
name {n}, beg {b}, end {e} {
|
||||
}
|
||||
|
||||
// Procedure: set_up
|
||||
inline void ChromeObserver::set_up(size_t num_workers) {
|
||||
_timeline.segments.resize(num_workers);
|
||||
_timeline.stacks.resize(num_workers);
|
||||
|
||||
for(size_t w=0; w<num_workers; ++w) {
|
||||
_timeline.segments[w].reserve(32);
|
||||
}
|
||||
|
||||
_timeline.origin = observer_stamp_t::clock::now();
|
||||
}
|
||||
|
||||
// Procedure: on_entry
|
||||
inline void ChromeObserver::on_entry(WorkerView wv, TaskView) {
|
||||
_timeline.stacks[wv.id()].push(observer_stamp_t::clock::now());
|
||||
}
|
||||
|
||||
// Procedure: on_exit
|
||||
inline void ChromeObserver::on_exit(WorkerView wv, TaskView tv) {
|
||||
|
||||
size_t w = wv.id();
|
||||
|
||||
assert(!_timeline.stacks[w].empty());
|
||||
|
||||
auto beg = _timeline.stacks[w].top();
|
||||
_timeline.stacks[w].pop();
|
||||
|
||||
_timeline.segments[w].emplace_back(
|
||||
tv.name(), beg, observer_stamp_t::clock::now()
|
||||
);
|
||||
}
|
||||
|
||||
// Function: clear
|
||||
inline void ChromeObserver::clear() {
|
||||
for(size_t w=0; w<_timeline.segments.size(); ++w) {
|
||||
_timeline.segments[w].clear();
|
||||
while(!_timeline.stacks[w].empty()) {
|
||||
_timeline.stacks[w].pop();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Procedure: dump
|
||||
inline void ChromeObserver::dump(std::ostream& os) const {
|
||||
|
||||
size_t first;
|
||||
|
||||
for(first = 0; first<_timeline.segments.size(); ++first) {
|
||||
if(_timeline.segments[first].size() > 0) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
os << '[';
|
||||
|
||||
for(size_t w=first; w<_timeline.segments.size(); w++) {
|
||||
|
||||
if(w != first && _timeline.segments[w].size() > 0) {
|
||||
os << ',';
|
||||
}
|
||||
|
||||
for(size_t i=0; i<_timeline.segments[w].size(); i++) {
|
||||
|
||||
os << '{'
|
||||
<< "\"cat\":\"ChromeObserver\",";
|
||||
|
||||
// name field
|
||||
os << "\"name\":\"";
|
||||
if(_timeline.segments[w][i].name.empty()) {
|
||||
os << w << '_' << i;
|
||||
}
|
||||
else {
|
||||
os << _timeline.segments[w][i].name;
|
||||
}
|
||||
os << "\",";
|
||||
|
||||
// segment field
|
||||
os << "\"ph\":\"X\","
|
||||
<< "\"pid\":1,"
|
||||
<< "\"tid\":" << w << ','
|
||||
<< "\"ts\":" << std::chrono::duration_cast<std::chrono::microseconds>(
|
||||
_timeline.segments[w][i].beg - _timeline.origin
|
||||
).count() << ','
|
||||
<< "\"dur\":" << std::chrono::duration_cast<std::chrono::microseconds>(
|
||||
_timeline.segments[w][i].end - _timeline.segments[w][i].beg
|
||||
).count();
|
||||
|
||||
if(i != _timeline.segments[w].size() - 1) {
|
||||
os << "},";
|
||||
}
|
||||
else {
|
||||
os << '}';
|
||||
}
|
||||
}
|
||||
}
|
||||
os << "]\n";
|
||||
}
|
||||
|
||||
// Function: dump
|
||||
inline std::string ChromeObserver::dump() const {
|
||||
std::ostringstream oss;
|
||||
dump(oss);
|
||||
return oss.str();
|
||||
}
|
||||
|
||||
// Function: num_tasks
|
||||
inline size_t ChromeObserver::num_tasks() const {
|
||||
return std::accumulate(
|
||||
_timeline.segments.begin(), _timeline.segments.end(), size_t{0},
|
||||
[](size_t sum, const auto& exe){
|
||||
return sum + exe.size();
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// TFProfObserver definition
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@class TFProfObserver
|
||||
|
||||
@brief observer interface based on the built-in taskflow profiler format
|
||||
|
||||
A tf::TFProfObserver inherits tf::ObserverInterface and defines methods to dump
|
||||
the observed thread activities into a format that can be visualized through
|
||||
@TFProf.
|
||||
|
||||
@code{.cpp}
|
||||
tf::Taskflow taskflow;
|
||||
tf::Executor executor;
|
||||
|
||||
// insert tasks into taskflow
|
||||
// ...
|
||||
|
||||
// create a custom observer
|
||||
std::shared_ptr<tf::TFProfObserver> observer = executor.make_observer<tf::TFProfObserver>();
|
||||
|
||||
// run the taskflow
|
||||
executor.run(taskflow).wait();
|
||||
|
||||
// dump the thread activities to Taskflow Profiler format.
|
||||
observer->dump(std::cout);
|
||||
@endcode
|
||||
|
||||
We recommend using our @TFProf python script to observe thread activities
|
||||
instead of the raw function call.
|
||||
The script will turn on environment variables needed for observing all executors
|
||||
in a taskflow program and dump the result to a valid, clean JSON file
|
||||
compatible with the format of @TFProf.
|
||||
*/
|
||||
class TFProfObserver : public ObserverInterface {
|
||||
|
||||
friend class Executor;
|
||||
friend class TFProfManager;
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
@brief dumps the timelines into a @TFProf format through
|
||||
an output stream
|
||||
*/
|
||||
void dump(std::ostream& ostream) const;
|
||||
|
||||
/**
|
||||
@brief dumps the timelines into a JSON string
|
||||
*/
|
||||
std::string dump() const;
|
||||
|
||||
/**
|
||||
@brief clears the timeline data
|
||||
*/
|
||||
void clear();
|
||||
|
||||
/**
|
||||
@brief queries the number of tasks observed
|
||||
*/
|
||||
size_t num_tasks() const;
|
||||
|
||||
private:
|
||||
|
||||
Timeline _timeline;
|
||||
|
||||
std::vector<std::stack<observer_stamp_t>> _stacks;
|
||||
|
||||
inline void set_up(size_t num_workers) override final;
|
||||
inline void on_entry(WorkerView, TaskView) override final;
|
||||
inline void on_exit(WorkerView, TaskView) override final;
|
||||
};
|
||||
|
||||
// Procedure: set_up
|
||||
inline void TFProfObserver::set_up(size_t num_workers) {
|
||||
_timeline.uid = unique_id<size_t>();
|
||||
_timeline.origin = observer_stamp_t::clock::now();
|
||||
_timeline.segments.resize(num_workers);
|
||||
_stacks.resize(num_workers);
|
||||
}
|
||||
|
||||
// Procedure: on_entry
|
||||
inline void TFProfObserver::on_entry(WorkerView wv, TaskView) {
|
||||
_stacks[wv.id()].push(observer_stamp_t::clock::now());
|
||||
}
|
||||
|
||||
// Procedure: on_exit
|
||||
inline void TFProfObserver::on_exit(WorkerView wv, TaskView tv) {
|
||||
|
||||
size_t w = wv.id();
|
||||
|
||||
assert(!_stacks[w].empty());
|
||||
|
||||
if(_stacks[w].size() > _timeline.segments[w].size()) {
|
||||
_timeline.segments[w].resize(_stacks[w].size());
|
||||
}
|
||||
|
||||
auto beg = _stacks[w].top();
|
||||
_stacks[w].pop();
|
||||
|
||||
_timeline.segments[w][_stacks[w].size()].emplace_back(
|
||||
tv.name(), tv.type(), beg, observer_stamp_t::clock::now()
|
||||
);
|
||||
}
|
||||
|
||||
// Function: clear
|
||||
inline void TFProfObserver::clear() {
|
||||
for(size_t w=0; w<_timeline.segments.size(); ++w) {
|
||||
for(size_t l=0; l<_timeline.segments[w].size(); ++l) {
|
||||
_timeline.segments[w][l].clear();
|
||||
}
|
||||
while(!_stacks[w].empty()) {
|
||||
_stacks[w].pop();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Procedure: dump
|
||||
inline void TFProfObserver::dump(std::ostream& os) const {
|
||||
|
||||
size_t first;
|
||||
|
||||
for(first = 0; first<_timeline.segments.size(); ++first) {
|
||||
if(_timeline.segments[first].size() > 0) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// not timeline data to dump
|
||||
if(first == _timeline.segments.size()) {
|
||||
os << "{}\n";
|
||||
return;
|
||||
}
|
||||
|
||||
os << "{\"executor\":\"" << _timeline.uid << "\",\"data\":[";
|
||||
|
||||
bool comma = false;
|
||||
|
||||
for(size_t w=first; w<_timeline.segments.size(); w++) {
|
||||
for(size_t l=0; l<_timeline.segments[w].size(); l++) {
|
||||
|
||||
if(_timeline.segments[w][l].empty()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if(comma) {
|
||||
os << ',';
|
||||
}
|
||||
else {
|
||||
comma = true;
|
||||
}
|
||||
|
||||
os << "{\"worker\":" << w << ",\"level\":" << l << ",\"data\":[";
|
||||
for(size_t i=0; i<_timeline.segments[w][l].size(); ++i) {
|
||||
|
||||
const auto& s = _timeline.segments[w][l][i];
|
||||
|
||||
if(i) os << ',';
|
||||
|
||||
// span
|
||||
os << "{\"span\":["
|
||||
<< std::chrono::duration_cast<std::chrono::microseconds>(
|
||||
s.beg - _timeline.origin
|
||||
).count() << ","
|
||||
<< std::chrono::duration_cast<std::chrono::microseconds>(
|
||||
s.end - _timeline.origin
|
||||
).count() << "],";
|
||||
|
||||
// name
|
||||
os << "\"name\":\"";
|
||||
if(s.name.empty()) {
|
||||
os << w << '_' << i;
|
||||
}
|
||||
else {
|
||||
os << s.name;
|
||||
}
|
||||
os << "\",";
|
||||
|
||||
// category "type": "Condition Task",
|
||||
os << "\"type\":\"" << to_string(s.type) << "\"";
|
||||
|
||||
os << "}";
|
||||
}
|
||||
os << "]}";
|
||||
}
|
||||
}
|
||||
|
||||
os << "]}\n";
|
||||
}
|
||||
|
||||
// Function: dump
|
||||
inline std::string TFProfObserver::dump() const {
|
||||
std::ostringstream oss;
|
||||
dump(oss);
|
||||
return oss.str();
|
||||
}
|
||||
|
||||
// Function: num_tasks
|
||||
inline size_t TFProfObserver::num_tasks() const {
|
||||
return std::accumulate(
|
||||
_timeline.segments.begin(), _timeline.segments.end(), size_t{0},
|
||||
[](size_t sum, const auto& exe){
|
||||
return sum + exe.size();
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// TFProfManager
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@private
|
||||
*/
|
||||
class TFProfManager {
|
||||
|
||||
friend class Executor;
|
||||
|
||||
public:
|
||||
|
||||
~TFProfManager();
|
||||
|
||||
TFProfManager(const TFProfManager&) = delete;
|
||||
TFProfManager& operator=(const TFProfManager&) = delete;
|
||||
|
||||
static TFProfManager& get();
|
||||
|
||||
void dump(std::ostream& ostream) const;
|
||||
|
||||
private:
|
||||
|
||||
const std::string _fpath;
|
||||
|
||||
std::mutex _mutex;
|
||||
std::vector<std::shared_ptr<TFProfObserver>> _observers;
|
||||
|
||||
TFProfManager();
|
||||
|
||||
void _manage(std::shared_ptr<TFProfObserver> observer);
|
||||
};
|
||||
|
||||
// constructor
|
||||
inline TFProfManager::TFProfManager() :
|
||||
_fpath {get_env(TF_ENABLE_PROFILER)} {
|
||||
|
||||
}
|
||||
|
||||
// Procedure: manage
|
||||
inline void TFProfManager::_manage(std::shared_ptr<TFProfObserver> observer) {
|
||||
std::lock_guard lock(_mutex);
|
||||
_observers.push_back(std::move(observer));
|
||||
}
|
||||
|
||||
// Procedure: dump
|
||||
inline void TFProfManager::dump(std::ostream& os) const {
|
||||
for(size_t i=0; i<_observers.size(); ++i) {
|
||||
if(i) os << ',';
|
||||
_observers[i]->dump(os);
|
||||
}
|
||||
}
|
||||
|
||||
// Destructor
|
||||
inline TFProfManager::~TFProfManager() {
|
||||
std::ofstream ofs(_fpath);
|
||||
if(ofs) {
|
||||
// .tfp
|
||||
if(_fpath.rfind(".tfp") != std::string::npos) {
|
||||
ProfileData data;
|
||||
data.timelines.reserve(_observers.size());
|
||||
for(size_t i=0; i<_observers.size(); ++i) {
|
||||
data.timelines.push_back(std::move(_observers[i]->_timeline));
|
||||
}
|
||||
Serializer<std::ofstream> serializer(ofs);
|
||||
serializer(data);
|
||||
}
|
||||
// .json
|
||||
else {
|
||||
ofs << "[\n";
|
||||
for(size_t i=0; i<_observers.size(); ++i) {
|
||||
if(i) ofs << ',';
|
||||
_observers[i]->dump(ofs);
|
||||
}
|
||||
ofs << "]\n";
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Function: get
|
||||
inline TFProfManager& TFProfManager::get() {
|
||||
static TFProfManager mgr;
|
||||
return mgr;
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Identifier for Each Built-in Observer
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/** @enum ObserverType
|
||||
|
||||
@brief enumeration of all observer types
|
||||
|
||||
*/
|
||||
enum class ObserverType : int {
|
||||
TFPROF = 0,
|
||||
CHROME,
|
||||
UNDEFINED
|
||||
};
|
||||
|
||||
/**
|
||||
@brief convert an observer type to a human-readable string
|
||||
*/
|
||||
inline const char* to_string(ObserverType type) {
|
||||
switch(type) {
|
||||
case ObserverType::TFPROF: return "tfprof";
|
||||
case ObserverType::CHROME: return "chrome";
|
||||
default: return "undefined";
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
} // end of namespace tf -----------------------------------------------------
|
||||
|
||||
|
||||
125
cpp_to_py/gpugr/taskflow/core/semaphore.hpp
Normal file
125
cpp_to_py/gpugr/taskflow/core/semaphore.hpp
Normal file
@ -0,0 +1,125 @@
|
||||
#pragma once
|
||||
|
||||
#include <vector>
|
||||
#include <mutex>
|
||||
|
||||
#include "declarations.hpp"
|
||||
|
||||
/**
|
||||
@file semaphore.hpp
|
||||
@brief semaphore include file
|
||||
*/
|
||||
|
||||
namespace tf {
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Semaphore
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@class Semaphore
|
||||
|
||||
@brief class to create a semophore object for building a concurrency constraint
|
||||
|
||||
A semaphore creates a constraint that limits the maximum concurrency,
|
||||
i.e., the number of workers, in a set of tasks.
|
||||
You can let a task acquire/release one or multiple semaphores before/after
|
||||
executing its work.
|
||||
A task can acquire and release a semaphore,
|
||||
or just acquire or just release it.
|
||||
A tf::Semaphore object starts with an initial count.
|
||||
As long as that count is above 0, tasks can acquire the semaphore and do
|
||||
their work.
|
||||
If the count is 0 or less, a task trying to acquire the semaphore will not run
|
||||
but goes to a waiting list of that semaphore.
|
||||
When the semaphore is released by another task,
|
||||
it reschedules all tasks on that waiting list.
|
||||
|
||||
@code{.cpp}
|
||||
tf::Executor executor(8); // create an executor of 8 workers
|
||||
tf::Taskflow taskflow;
|
||||
|
||||
tf::Semaphore semaphore(1); // create a semaphore with initial count 1
|
||||
|
||||
std::vector<tf::Task> tasks {
|
||||
taskflow.emplace([](){ std::cout << "A" << std::endl; }),
|
||||
taskflow.emplace([](){ std::cout << "B" << std::endl; }),
|
||||
taskflow.emplace([](){ std::cout << "C" << std::endl; }),
|
||||
taskflow.emplace([](){ std::cout << "D" << std::endl; }),
|
||||
taskflow.emplace([](){ std::cout << "E" << std::endl; })
|
||||
};
|
||||
|
||||
for(auto & task : tasks) { // each task acquires and release the semaphore
|
||||
task.acquire(semaphore);
|
||||
task.release(semaphore);
|
||||
}
|
||||
|
||||
executor.run(taskflow).wait();
|
||||
@endcode
|
||||
|
||||
The above example creates five tasks with no dependencies between them.
|
||||
Under normal circumstances, the five tasks would be executed concurrently.
|
||||
However, this example has a semaphore with initial count 1,
|
||||
and all tasks need to acquire that semaphore before running and release that
|
||||
semaphore after they are done.
|
||||
This organization limits the number of concurrently running tasks to only one.
|
||||
|
||||
*/
|
||||
class Semaphore {
|
||||
|
||||
friend class Node;
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
@brief constructs a semaphore with the given counter
|
||||
*/
|
||||
explicit Semaphore(int max_workers);
|
||||
|
||||
/**
|
||||
@brief queries the counter value (not thread-safe during the run)
|
||||
*/
|
||||
int count() const;
|
||||
|
||||
private:
|
||||
|
||||
std::mutex _mtx;
|
||||
|
||||
int _counter;
|
||||
|
||||
std::vector<Node*> _waiters;
|
||||
|
||||
bool _try_acquire_or_wait(Node*);
|
||||
|
||||
std::vector<Node*> _release();
|
||||
};
|
||||
|
||||
inline Semaphore::Semaphore(int max_workers) :
|
||||
_counter(max_workers) {
|
||||
}
|
||||
|
||||
inline bool Semaphore::_try_acquire_or_wait(Node* me) {
|
||||
std::lock_guard<std::mutex> lock(_mtx);
|
||||
if(_counter > 0) {
|
||||
--_counter;
|
||||
return true;
|
||||
}
|
||||
else {
|
||||
_waiters.push_back(me);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
inline std::vector<Node*> Semaphore::_release() {
|
||||
std::lock_guard<std::mutex> lock(_mtx);
|
||||
++_counter;
|
||||
std::vector<Node*> r{std::move(_waiters)};
|
||||
return r;
|
||||
}
|
||||
|
||||
inline int Semaphore::count() const {
|
||||
return _counter;
|
||||
}
|
||||
|
||||
} // end of namespace tf. ---------------------------------------------------
|
||||
|
||||
713
cpp_to_py/gpugr/taskflow/core/task.hpp
Normal file
713
cpp_to_py/gpugr/taskflow/core/task.hpp
Normal file
@ -0,0 +1,713 @@
|
||||
#pragma once
|
||||
|
||||
#include "graph.hpp"
|
||||
|
||||
/**
|
||||
@file task.hpp
|
||||
@brief task include file
|
||||
*/
|
||||
|
||||
namespace tf {
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Task Types
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@enum TaskType
|
||||
|
||||
@brief enumeration of all task types
|
||||
*/
|
||||
enum class TaskType : int {
|
||||
/** @brief placeholder task type */
|
||||
PLACEHOLDER = 0,
|
||||
/** @brief cudaFlow task type */
|
||||
CUDAFLOW,
|
||||
/** @brief syclFlow task type */
|
||||
SYCLFLOW,
|
||||
/** @brief static task type */
|
||||
STATIC,
|
||||
/** @brief dynamic (subflow) task type */
|
||||
DYNAMIC,
|
||||
/** @brief condition task type */
|
||||
CONDITION,
|
||||
/** @brief module task type */
|
||||
MODULE,
|
||||
/** @brief asynchronous task type */
|
||||
ASYNC,
|
||||
/** @brief undefined task type (for internal use only) */
|
||||
UNDEFINED
|
||||
};
|
||||
|
||||
/**
|
||||
@brief array of all task types (used for iterating task types)
|
||||
*/
|
||||
inline constexpr std::array<TaskType, 8> TASK_TYPES = {
|
||||
TaskType::PLACEHOLDER,
|
||||
TaskType::CUDAFLOW,
|
||||
TaskType::SYCLFLOW,
|
||||
TaskType::STATIC,
|
||||
TaskType::DYNAMIC,
|
||||
TaskType::CONDITION,
|
||||
TaskType::MODULE,
|
||||
TaskType::ASYNC
|
||||
};
|
||||
|
||||
/**
|
||||
@brief convert a task type to a human-readable string
|
||||
*/
|
||||
inline const char* to_string(TaskType type) {
|
||||
|
||||
const char* val;
|
||||
|
||||
switch(type) {
|
||||
case TaskType::PLACEHOLDER: val = "placeholder"; break;
|
||||
case TaskType::CUDAFLOW: val = "cudaflow"; break;
|
||||
case TaskType::SYCLFLOW: val = "syclflow"; break;
|
||||
case TaskType::STATIC: val = "static"; break;
|
||||
case TaskType::DYNAMIC: val = "subflow"; break;
|
||||
case TaskType::CONDITION: val = "condition"; break;
|
||||
case TaskType::MODULE: val = "module"; break;
|
||||
case TaskType::ASYNC: val = "async"; break;
|
||||
default: val = "undefined"; break;
|
||||
}
|
||||
|
||||
return val;
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Task Traits
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@brief determines if a callable is a static task
|
||||
|
||||
A static task is a callable object constructible from std::function<void()>.
|
||||
*/
|
||||
template <typename C>
|
||||
constexpr bool is_static_task_v = std::is_invocable_r_v<void, C> &&
|
||||
!std::is_invocable_r_v<int, C>;
|
||||
|
||||
/**
|
||||
@brief determines if a callable is a dynamic task
|
||||
|
||||
A dynamic task is a callable object constructible from std::function<void(Subflow&)>.
|
||||
*/
|
||||
template <typename C>
|
||||
constexpr bool is_dynamic_task_v = std::is_invocable_r_v<void, C, Subflow&>;
|
||||
|
||||
/**
|
||||
@brief determines if a callable is a condition task
|
||||
|
||||
A condition task is a callable object constructible from std::function<int()>.
|
||||
*/
|
||||
template <typename C>
|
||||
constexpr bool is_condition_task_v = std::is_invocable_r_v<int, C>;
|
||||
|
||||
/**
|
||||
@brief determines if a callable is a %cudaFlow task
|
||||
|
||||
A cudaFlow task is a callable object constructible from
|
||||
std::function<void(tf::cudaFlow&)> or std::function<void(tf::cudaFlowCapturer&)>.
|
||||
*/
|
||||
template <typename C>
|
||||
constexpr bool is_cudaflow_task_v = std::is_invocable_r_v<void, C, cudaFlow&> ||
|
||||
std::is_invocable_r_v<void, C, cudaFlowCapturer&>;
|
||||
|
||||
/**
|
||||
@brief determines if a callable is a %syclFlow task
|
||||
|
||||
A syclFlow task is a callable object constructible from
|
||||
std::function<void(tf::syclFlow&)>.
|
||||
*/
|
||||
template <typename C>
|
||||
constexpr bool is_syclflow_task_v = std::is_invocable_r_v<void, C, syclFlow&>;
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Task
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@class Task
|
||||
|
||||
@brief handle to a node in a task dependency graph
|
||||
|
||||
A Task is handle to manipulate a node in a taskflow graph.
|
||||
It provides a set of methods for users to access and modify the attributes of
|
||||
the associated graph node without directly touching internal node data.
|
||||
|
||||
*/
|
||||
class Task {
|
||||
|
||||
friend class FlowBuilder;
|
||||
friend class Taskflow;
|
||||
friend class TaskView;
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
@brief constructs an empty task
|
||||
*/
|
||||
Task() = default;
|
||||
|
||||
/**
|
||||
@brief constructs the task with the copy of the other task
|
||||
*/
|
||||
Task(const Task& other);
|
||||
|
||||
/**
|
||||
@brief replaces the contents with a copy of the other task
|
||||
*/
|
||||
Task& operator = (const Task&);
|
||||
|
||||
/**
|
||||
@brief replaces the contents with a null pointer
|
||||
*/
|
||||
Task& operator = (std::nullptr_t);
|
||||
|
||||
/**
|
||||
@brief compares if two tasks are associated with the same graph node
|
||||
*/
|
||||
bool operator == (const Task& rhs) const;
|
||||
|
||||
/**
|
||||
@brief compares if two tasks are not associated with the same graph node
|
||||
*/
|
||||
bool operator != (const Task& rhs) const;
|
||||
|
||||
/**
|
||||
@brief queries the name of the task
|
||||
*/
|
||||
const std::string& name() const;
|
||||
|
||||
/**
|
||||
@brief queries the number of successors of the task
|
||||
*/
|
||||
size_t num_successors() const;
|
||||
|
||||
/**
|
||||
@brief queries the number of predecessors of the task
|
||||
*/
|
||||
size_t num_dependents() const;
|
||||
|
||||
/**
|
||||
@brief queries the number of strong dependents of the task
|
||||
*/
|
||||
size_t num_strong_dependents() const;
|
||||
|
||||
/**
|
||||
@brief queries the number of weak dependents of the task
|
||||
*/
|
||||
size_t num_weak_dependents() const;
|
||||
|
||||
/**
|
||||
@brief assigns a name to the task
|
||||
|
||||
@param name a @std_string acceptable string
|
||||
|
||||
@return @c *this
|
||||
*/
|
||||
Task& name(const std::string& name);
|
||||
|
||||
/**
|
||||
@brief assigns a callable
|
||||
|
||||
@tparam C callable type
|
||||
|
||||
@param callable callable to construct one of the static, dynamic, condition, and cudaFlow tasks
|
||||
|
||||
@return @c *this
|
||||
*/
|
||||
template <typename C>
|
||||
Task& work(C&& callable);
|
||||
|
||||
/**
|
||||
@brief creates a module task from a taskflow
|
||||
|
||||
@param taskflow a taskflow object for the module
|
||||
|
||||
@return @c *this
|
||||
*/
|
||||
Task& composed_of(Taskflow& taskflow);
|
||||
|
||||
/**
|
||||
@brief adds precedence links from this to other tasks
|
||||
|
||||
@tparam Ts parameter pack
|
||||
|
||||
@param tasks one or multiple tasks
|
||||
|
||||
@return @c *this
|
||||
*/
|
||||
template <typename... Ts>
|
||||
Task& precede(Ts&&... tasks);
|
||||
|
||||
/**
|
||||
@brief adds precedence links from other tasks to this
|
||||
|
||||
@tparam Ts parameter pack
|
||||
|
||||
@param tasks one or multiple tasks
|
||||
|
||||
@return @c *this
|
||||
*/
|
||||
template <typename... Ts>
|
||||
Task& succeed(Ts&&... tasks);
|
||||
|
||||
/**
|
||||
@brief makes the task release this semaphore
|
||||
*/
|
||||
Task& release(Semaphore& semaphore);
|
||||
|
||||
/**
|
||||
@brief makes the task acquire this semaphore
|
||||
*/
|
||||
Task& acquire(Semaphore& semaphore);
|
||||
|
||||
/**
|
||||
@brief assigns pointer to user data
|
||||
|
||||
@param data pointer to user data
|
||||
|
||||
@return @c *this
|
||||
*/
|
||||
Task& data(void* data);
|
||||
|
||||
/**
|
||||
@brief resets the task handle to null
|
||||
*/
|
||||
void reset();
|
||||
|
||||
/**
|
||||
@brief resets the associated work to a placeholder
|
||||
*/
|
||||
void reset_work();
|
||||
|
||||
/**
|
||||
@brief queries if the task handle points to a task node
|
||||
*/
|
||||
bool empty() const;
|
||||
|
||||
/**
|
||||
@brief queries if the task has a work assigned
|
||||
*/
|
||||
bool has_work() const;
|
||||
|
||||
/**
|
||||
@brief applies an visitor callable to each successor of the task
|
||||
*/
|
||||
template <typename V>
|
||||
void for_each_successor(V&& visitor) const;
|
||||
|
||||
/**
|
||||
@brief applies an visitor callable to each dependents of the task
|
||||
*/
|
||||
template <typename V>
|
||||
void for_each_dependent(V&& visitor) const;
|
||||
|
||||
/**
|
||||
@brief obtains a hash value of the underlying node
|
||||
*/
|
||||
size_t hash_value() const;
|
||||
|
||||
/**
|
||||
@brief returns the task type
|
||||
*/
|
||||
TaskType type() const;
|
||||
|
||||
/**
|
||||
@brief dumps the task through an output stream
|
||||
*/
|
||||
void dump(std::ostream& ostream) const;
|
||||
|
||||
/**
|
||||
@brief queries pointer to user data
|
||||
*/
|
||||
void* data() const;
|
||||
|
||||
|
||||
private:
|
||||
|
||||
Task(Node*);
|
||||
|
||||
Node* _node {nullptr};
|
||||
};
|
||||
|
||||
// Constructor
|
||||
inline Task::Task(Node* node) : _node {node} {
|
||||
}
|
||||
|
||||
// Constructor
|
||||
inline Task::Task(const Task& rhs) : _node {rhs._node} {
|
||||
}
|
||||
|
||||
// Function: precede
|
||||
template <typename... Ts>
|
||||
Task& Task::precede(Ts&&... tasks) {
|
||||
(_node->_precede(tasks._node), ...);
|
||||
//_precede(std::forward<Ts>(tasks)...);
|
||||
return *this;
|
||||
}
|
||||
|
||||
// Function: succeed
|
||||
template <typename... Ts>
|
||||
Task& Task::succeed(Ts&&... tasks) {
|
||||
(tasks._node->_precede(_node), ...);
|
||||
//_succeed(std::forward<Ts>(tasks)...);
|
||||
return *this;
|
||||
}
|
||||
|
||||
// Function: composed_of
|
||||
inline Task& Task::composed_of(Taskflow& tf) {
|
||||
_node->_handle.emplace<Node::Module>(&tf);
|
||||
return *this;
|
||||
}
|
||||
|
||||
// Operator =
|
||||
inline Task& Task::operator = (const Task& rhs) {
|
||||
_node = rhs._node;
|
||||
return *this;
|
||||
}
|
||||
|
||||
// Operator =
|
||||
inline Task& Task::operator = (std::nullptr_t ptr) {
|
||||
_node = ptr;
|
||||
return *this;
|
||||
}
|
||||
|
||||
// Operator ==
|
||||
inline bool Task::operator == (const Task& rhs) const {
|
||||
return _node == rhs._node;
|
||||
}
|
||||
|
||||
// Operator !=
|
||||
inline bool Task::operator != (const Task& rhs) const {
|
||||
return _node != rhs._node;
|
||||
}
|
||||
|
||||
// Function: name
|
||||
inline Task& Task::name(const std::string& name) {
|
||||
_node->_name = name;
|
||||
return *this;
|
||||
}
|
||||
|
||||
// Function: acquire
|
||||
inline Task& Task::acquire(Semaphore& s) {
|
||||
if(!_node->_semaphores) {
|
||||
//_node->_semaphores.emplace();
|
||||
_node->_semaphores = std::make_unique<Node::Semaphores>();
|
||||
}
|
||||
_node->_semaphores->to_acquire.push_back(&s);
|
||||
return *this;
|
||||
}
|
||||
|
||||
// Function: release
|
||||
inline Task& Task::release(Semaphore& s) {
|
||||
if(!_node->_semaphores) {
|
||||
//_node->_semaphores.emplace();
|
||||
_node->_semaphores = std::make_unique<Node::Semaphores>();
|
||||
}
|
||||
_node->_semaphores->to_release.push_back(&s);
|
||||
return *this;
|
||||
}
|
||||
|
||||
// Procedure: reset
|
||||
inline void Task::reset() {
|
||||
_node = nullptr;
|
||||
}
|
||||
|
||||
// Procedure: reset_work
|
||||
inline void Task::reset_work() {
|
||||
_node->_handle.emplace<std::monostate>();
|
||||
}
|
||||
|
||||
// Function: name
|
||||
inline const std::string& Task::name() const {
|
||||
return _node->_name;
|
||||
}
|
||||
|
||||
// Function: num_dependents
|
||||
inline size_t Task::num_dependents() const {
|
||||
return _node->num_dependents();
|
||||
}
|
||||
|
||||
// Function: num_strong_dependents
|
||||
inline size_t Task::num_strong_dependents() const {
|
||||
return _node->num_strong_dependents();
|
||||
}
|
||||
|
||||
// Function: num_weak_dependents
|
||||
inline size_t Task::num_weak_dependents() const {
|
||||
return _node->num_weak_dependents();
|
||||
}
|
||||
|
||||
// Function: num_successors
|
||||
inline size_t Task::num_successors() const {
|
||||
return _node->num_successors();
|
||||
}
|
||||
|
||||
// Function: empty
|
||||
inline bool Task::empty() const {
|
||||
return _node == nullptr;
|
||||
}
|
||||
|
||||
// Function: has_work
|
||||
inline bool Task::has_work() const {
|
||||
return _node ? _node->_handle.index() != 0 : false;
|
||||
}
|
||||
|
||||
// Function: task_type
|
||||
inline TaskType Task::type() const {
|
||||
switch(_node->_handle.index()) {
|
||||
case Node::PLACEHOLDER: return TaskType::PLACEHOLDER;
|
||||
case Node::STATIC: return TaskType::STATIC;
|
||||
case Node::DYNAMIC: return TaskType::DYNAMIC;
|
||||
case Node::CONDITION: return TaskType::CONDITION;
|
||||
case Node::MODULE: return TaskType::MODULE;
|
||||
case Node::ASYNC: return TaskType::ASYNC;
|
||||
case Node::SILENT_ASYNC: return TaskType::ASYNC;
|
||||
case Node::CUDAFLOW: return TaskType::CUDAFLOW;
|
||||
case Node::SYCLFLOW: return TaskType::SYCLFLOW;
|
||||
default: return TaskType::UNDEFINED;
|
||||
}
|
||||
}
|
||||
|
||||
// Function: for_each_successor
|
||||
template <typename V>
|
||||
void Task::for_each_successor(V&& visitor) const {
|
||||
for(size_t i=0; i<_node->_successors.size(); ++i) {
|
||||
visitor(Task(_node->_successors[i]));
|
||||
}
|
||||
}
|
||||
|
||||
// Function: for_each_dependent
|
||||
template <typename V>
|
||||
void Task::for_each_dependent(V&& visitor) const {
|
||||
for(size_t i=0; i<_node->_dependents.size(); ++i) {
|
||||
visitor(Task(_node->_dependents[i]));
|
||||
}
|
||||
}
|
||||
|
||||
// Function: hash_value
|
||||
inline size_t Task::hash_value() const {
|
||||
return std::hash<Node*>{}(_node);
|
||||
}
|
||||
|
||||
// Procedure: dump
|
||||
inline void Task::dump(std::ostream& os) const {
|
||||
os << "task ";
|
||||
if(name().empty()) os << _node;
|
||||
else os << name();
|
||||
os << " [type=" << to_string(type()) << ']';
|
||||
}
|
||||
|
||||
// Function: work
|
||||
template <typename C>
|
||||
Task& Task::work(C&& c) {
|
||||
if constexpr(is_static_task_v<C>) {
|
||||
_node->_handle.emplace<Node::Static>(std::forward<C>(c));
|
||||
}
|
||||
else if constexpr(is_dynamic_task_v<C>) {
|
||||
_node->_handle.emplace<Node::Dynamic>(std::forward<C>(c));
|
||||
}
|
||||
else if constexpr(is_condition_task_v<C>) {
|
||||
_node->_handle.emplace<Node::Condition>(std::forward<C>(c));
|
||||
}
|
||||
else if constexpr(is_cudaflow_task_v<C>) {
|
||||
_node->_handle.emplace<Node::cudaFlow>(std::forward<C>(c));
|
||||
}
|
||||
else {
|
||||
static_assert(dependent_false_v<C>, "invalid task callable");
|
||||
}
|
||||
return *this;
|
||||
}
|
||||
|
||||
// Function: name
|
||||
inline void* Task::data() const {
|
||||
return _node->_data;
|
||||
}
|
||||
|
||||
// Function: name
|
||||
inline Task& Task::data(void* data) {
|
||||
_node->_data = data;
|
||||
return *this;
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// global ostream
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@brief overload of ostream inserter operator for cudaTask
|
||||
*/
|
||||
inline std::ostream& operator << (std::ostream& os, const Task& task) {
|
||||
task.dump(os);
|
||||
return os;
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@class TaskView
|
||||
|
||||
@brief class to access task information from the observer interface
|
||||
*/
|
||||
class TaskView {
|
||||
|
||||
friend class Executor;
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
@brief queries the name of the task
|
||||
*/
|
||||
const std::string& name() const;
|
||||
|
||||
/**
|
||||
@brief queries the number of successors of the task
|
||||
*/
|
||||
size_t num_successors() const;
|
||||
|
||||
/**
|
||||
@brief queries the number of predecessors of the task
|
||||
*/
|
||||
size_t num_dependents() const;
|
||||
|
||||
/**
|
||||
@brief queries the number of strong dependents of the task
|
||||
*/
|
||||
size_t num_strong_dependents() const;
|
||||
|
||||
/**
|
||||
@brief queries the number of weak dependents of the task
|
||||
*/
|
||||
size_t num_weak_dependents() const;
|
||||
|
||||
/**
|
||||
@brief applies an visitor callable to each successor of the task
|
||||
*/
|
||||
template <typename V>
|
||||
void for_each_successor(V&& visitor) const;
|
||||
|
||||
/**
|
||||
@brief applies an visitor callable to each dependents of the task
|
||||
*/
|
||||
template <typename V>
|
||||
void for_each_dependent(V&& visitor) const;
|
||||
|
||||
/**
|
||||
@brief queries the task type
|
||||
*/
|
||||
TaskType type() const;
|
||||
|
||||
/**
|
||||
@brief obtains a hash value of the underlying node
|
||||
*/
|
||||
size_t hash_value() const;
|
||||
|
||||
private:
|
||||
|
||||
TaskView(const Node&);
|
||||
TaskView(const TaskView&) = default;
|
||||
|
||||
const Node& _node;
|
||||
};
|
||||
|
||||
// Constructor
|
||||
inline TaskView::TaskView(const Node& node) : _node {node} {
|
||||
}
|
||||
|
||||
// Function: name
|
||||
inline const std::string& TaskView::name() const {
|
||||
return _node._name;
|
||||
}
|
||||
|
||||
// Function: num_dependents
|
||||
inline size_t TaskView::num_dependents() const {
|
||||
return _node.num_dependents();
|
||||
}
|
||||
|
||||
// Function: num_strong_dependents
|
||||
inline size_t TaskView::num_strong_dependents() const {
|
||||
return _node.num_strong_dependents();
|
||||
}
|
||||
|
||||
// Function: num_weak_dependents
|
||||
inline size_t TaskView::num_weak_dependents() const {
|
||||
return _node.num_weak_dependents();
|
||||
}
|
||||
|
||||
// Function: num_successors
|
||||
inline size_t TaskView::num_successors() const {
|
||||
return _node.num_successors();
|
||||
}
|
||||
|
||||
// Function: type
|
||||
inline TaskType TaskView::type() const {
|
||||
switch(_node._handle.index()) {
|
||||
case Node::PLACEHOLDER: return TaskType::PLACEHOLDER;
|
||||
case Node::STATIC: return TaskType::STATIC;
|
||||
case Node::DYNAMIC: return TaskType::DYNAMIC;
|
||||
case Node::CONDITION: return TaskType::CONDITION;
|
||||
case Node::MODULE: return TaskType::MODULE;
|
||||
case Node::ASYNC: return TaskType::ASYNC;
|
||||
case Node::SILENT_ASYNC: return TaskType::ASYNC;
|
||||
case Node::CUDAFLOW: return TaskType::CUDAFLOW;
|
||||
case Node::SYCLFLOW: return TaskType::SYCLFLOW;
|
||||
default: return TaskType::UNDEFINED;
|
||||
}
|
||||
}
|
||||
|
||||
// Function: hash_value
|
||||
inline size_t TaskView::hash_value() const {
|
||||
return std::hash<const Node*>{}(&_node);
|
||||
}
|
||||
|
||||
// Function: for_each_successor
|
||||
template <typename V>
|
||||
void TaskView::for_each_successor(V&& visitor) const {
|
||||
for(size_t i=0; i<_node._successors.size(); ++i) {
|
||||
visitor(TaskView(_node._successors[i]));
|
||||
}
|
||||
}
|
||||
|
||||
// Function: for_each_dependent
|
||||
template <typename V>
|
||||
void TaskView::for_each_dependent(V&& visitor) const {
|
||||
for(size_t i=0; i<_node._dependents.size(); ++i) {
|
||||
visitor(TaskView(_node._dependents[i]));
|
||||
}
|
||||
}
|
||||
|
||||
} // end of namespace tf. ---------------------------------------------------
|
||||
|
||||
namespace std {
|
||||
|
||||
/**
|
||||
@struct hash
|
||||
|
||||
@brief hash specialization for std::hash<tf::Task>
|
||||
*/
|
||||
template <>
|
||||
struct hash<tf::Task> {
|
||||
auto operator() (const tf::Task& task) const noexcept {
|
||||
return task.hash_value();
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
@struct hash
|
||||
|
||||
@brief hash specialization for std::hash<tf::TaskView>
|
||||
*/
|
||||
template <>
|
||||
struct hash<tf::TaskView> {
|
||||
auto operator() (const tf::TaskView& task_view) const noexcept {
|
||||
return task_view.hash_value();
|
||||
}
|
||||
};
|
||||
|
||||
} // end of namespace std ----------------------------------------------------
|
||||
|
||||
|
||||
|
||||
540
cpp_to_py/gpugr/taskflow/core/taskflow.hpp
Normal file
540
cpp_to_py/gpugr/taskflow/core/taskflow.hpp
Normal file
@ -0,0 +1,540 @@
|
||||
#pragma once
|
||||
|
||||
#include "flow_builder.hpp"
|
||||
|
||||
/**
|
||||
@file core/taskflow.hpp
|
||||
@brief taskflow include file
|
||||
*/
|
||||
|
||||
namespace tf {
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@class Taskflow
|
||||
|
||||
@brief main entry to create a task dependency graph
|
||||
|
||||
A %taskflow manages a task dependency graph where each task represents a
|
||||
callable object (e.g., @std_lambda, @std_function) and an edge represents a
|
||||
dependency between two tasks. A task is one of the following types:
|
||||
|
||||
1. static task : the callable constructible from
|
||||
@c std::function<void()>
|
||||
2. dynamic task : the callable constructible from
|
||||
@c std::function<void(tf::Subflow&)>
|
||||
3. condition task: the callable constructible from
|
||||
@c std::function<int()>
|
||||
4. module task : the task constructed from tf::Taskflow::composed_of
|
||||
5. %cudaFlow task: the callable constructible from
|
||||
@c std::function<void(tf::cudaFlow&)> or
|
||||
@c std::function<void(tf::cudaFlowCapturer&)>
|
||||
6. %syclFlow task: the callable constructible from
|
||||
@c std::function<void(tf::syclFlow&)>
|
||||
|
||||
Each task is a basic computation unit and is run by one worker thread
|
||||
from an executor.
|
||||
The following example creates a simple taskflow graph of four static tasks,
|
||||
@c A, @c B, @c C, and @c D, where
|
||||
@c A runs before @c B and @c C and
|
||||
@c D runs after @c B and @c C.
|
||||
|
||||
@code{.cpp}
|
||||
tf::Executor executor;
|
||||
tf::Taskflow taskflow("simple");
|
||||
|
||||
tf::Task A = taskflow.emplace([](){ std::cout << "TaskA\n"; });
|
||||
tf::Task B = taskflow.emplace([](){ std::cout << "TaskB\n"; });
|
||||
tf::Task C = taskflow.emplace([](){ std::cout << "TaskC\n"; });
|
||||
tf::Task D = taskflow.emplace([](){ std::cout << "TaskD\n"; });
|
||||
|
||||
A.precede(B, C); // A runs before B and C
|
||||
D.succeed(B, C); // D runs after B and C
|
||||
|
||||
executor.run(taskflow).wait();
|
||||
@endcode
|
||||
|
||||
Please refer to @ref Cookbook to learn more about each task type
|
||||
and how to submit a taskflow to an executor.
|
||||
*/
|
||||
class Taskflow : public FlowBuilder {
|
||||
|
||||
friend class Topology;
|
||||
friend class Executor;
|
||||
friend class FlowBuilder;
|
||||
|
||||
struct Dumper {
|
||||
std::stack<const Taskflow*> stack;
|
||||
std::unordered_set<const Taskflow*> visited;
|
||||
};
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
@brief constructs a taskflow with the given name
|
||||
*/
|
||||
Taskflow(const std::string& name);
|
||||
|
||||
/**
|
||||
@brief constructs a taskflow
|
||||
*/
|
||||
Taskflow();
|
||||
|
||||
/**
|
||||
@brief constructs a taskflow from a moved taskflow
|
||||
|
||||
Move a running taskflow can result in undefined behavior.
|
||||
You should only move a taskflow to another if it is not being run by
|
||||
an executor.
|
||||
*/
|
||||
Taskflow(Taskflow&& rhs);
|
||||
|
||||
/**
|
||||
@brief move assignment operator
|
||||
|
||||
Move a running taskflow can result in undefined behavior.
|
||||
You should only move a taskflow to another if it is not being run by
|
||||
an executor.
|
||||
*/
|
||||
Taskflow& operator = (Taskflow&& rhs);
|
||||
|
||||
/**
|
||||
@brief default destructor
|
||||
|
||||
When the destructor is called, all tasks and their associated data
|
||||
(e.g., captured data) will be destroyed.
|
||||
It is your responsibility to ensure all submitted execution of this
|
||||
taskflow have completed before destroying it.
|
||||
*/
|
||||
~Taskflow() = default;
|
||||
|
||||
/**
|
||||
@brief dumps the taskflow to a DOT format through a std::ostream target
|
||||
*/
|
||||
void dump(std::ostream& ostream) const;
|
||||
|
||||
/**
|
||||
@brief dumps the taskflow to a std::string of DOT format
|
||||
*/
|
||||
std::string dump() const;
|
||||
|
||||
/**
|
||||
@brief queries the number of tasks
|
||||
*/
|
||||
size_t num_tasks() const;
|
||||
|
||||
/**
|
||||
@brief queries the emptiness of the taskflow
|
||||
*/
|
||||
bool empty() const;
|
||||
|
||||
/**
|
||||
@brief assigns a name to the taskflow
|
||||
*/
|
||||
void name(const std::string&);
|
||||
|
||||
/**
|
||||
@brief queries the name of the taskflow
|
||||
*/
|
||||
const std::string& name() const ;
|
||||
|
||||
/**
|
||||
@brief clears the associated task dependency graph
|
||||
|
||||
When you clear a taskflow, all tasks and their associated data
|
||||
(e.g., captured data) will be destroyed.
|
||||
You should never clean a taskflow while it is being run by an executor.
|
||||
*/
|
||||
void clear();
|
||||
|
||||
/**
|
||||
@brief applies a visitor to each task in the taskflow
|
||||
|
||||
A visitor is a callable that takes an argument of type tf::Task
|
||||
and returns nothing. The following example iterates each task in a
|
||||
taskflow and prints its name:
|
||||
|
||||
@code{.cpp}
|
||||
taskflow.for_each_task([](tf::Task task){
|
||||
std::cout << task.name() << '\n';
|
||||
});
|
||||
@endcode
|
||||
*/
|
||||
template <typename V>
|
||||
void for_each_task(V&& visitor) const;
|
||||
|
||||
private:
|
||||
|
||||
mutable std::mutex _mutex;
|
||||
|
||||
std::string _name;
|
||||
|
||||
Graph _graph;
|
||||
|
||||
std::queue<std::shared_ptr<Topology>> _topologies;
|
||||
|
||||
std::optional<std::list<Taskflow>::iterator> _satellite;
|
||||
|
||||
void _dump(std::ostream&, const Taskflow*) const;
|
||||
void _dump(std::ostream&, const Node*, Dumper&) const;
|
||||
void _dump(std::ostream&, const Graph&, Dumper&) const;
|
||||
};
|
||||
|
||||
// Constructor
|
||||
inline Taskflow::Taskflow(const std::string& name) :
|
||||
FlowBuilder {_graph},
|
||||
_name {name} {
|
||||
}
|
||||
|
||||
// Constructor
|
||||
inline Taskflow::Taskflow() : FlowBuilder{_graph} {
|
||||
}
|
||||
|
||||
// Move constructor
|
||||
inline Taskflow::Taskflow(Taskflow&& rhs) : FlowBuilder{_graph} {
|
||||
|
||||
std::scoped_lock<std::mutex> lock(rhs._mutex);
|
||||
|
||||
_name = std::move(rhs._name);
|
||||
_graph = std::move(rhs._graph);
|
||||
_topologies = std::move(rhs._topologies);
|
||||
_satellite = rhs._satellite;
|
||||
|
||||
rhs._satellite.reset();
|
||||
}
|
||||
|
||||
// Move assignment
|
||||
inline Taskflow& Taskflow::operator = (Taskflow&& rhs) {
|
||||
if(this != &rhs) {
|
||||
std::scoped_lock<std::mutex, std::mutex> lock(_mutex, rhs._mutex);
|
||||
_name = std::move(rhs._name);
|
||||
_graph = std::move(rhs._graph);
|
||||
_topologies = std::move(rhs._topologies);
|
||||
_satellite = rhs._satellite;
|
||||
rhs._satellite.reset();
|
||||
}
|
||||
return *this;
|
||||
}
|
||||
|
||||
// Procedure:
|
||||
inline void Taskflow::clear() {
|
||||
_graph.clear();
|
||||
}
|
||||
|
||||
// Function: num_tasks
|
||||
inline size_t Taskflow::num_tasks() const {
|
||||
return _graph.size();
|
||||
}
|
||||
|
||||
// Function: empty
|
||||
inline bool Taskflow::empty() const {
|
||||
return _graph.empty();
|
||||
}
|
||||
|
||||
// Function: name
|
||||
inline void Taskflow::name(const std::string &name) {
|
||||
_name = name;
|
||||
}
|
||||
|
||||
// Function: name
|
||||
inline const std::string& Taskflow::name() const {
|
||||
return _name;
|
||||
}
|
||||
|
||||
// Function: for_each_task
|
||||
template <typename V>
|
||||
void Taskflow::for_each_task(V&& visitor) const {
|
||||
for(size_t i=0; i<_graph._nodes.size(); ++i) {
|
||||
visitor(Task(_graph._nodes[i]));
|
||||
}
|
||||
}
|
||||
|
||||
// Procedure: dump
|
||||
inline std::string Taskflow::dump() const {
|
||||
std::ostringstream oss;
|
||||
dump(oss);
|
||||
return oss.str();
|
||||
}
|
||||
|
||||
// Function: dump
|
||||
inline void Taskflow::dump(std::ostream& os) const {
|
||||
os << "digraph Taskflow {\n";
|
||||
_dump(os, this);
|
||||
os << "}\n";
|
||||
}
|
||||
|
||||
// Procedure: _dump
|
||||
inline void Taskflow::_dump(std::ostream& os, const Taskflow* top) const {
|
||||
|
||||
Dumper dumper;
|
||||
|
||||
dumper.stack.push(top);
|
||||
dumper.visited.insert(top);
|
||||
|
||||
while(!dumper.stack.empty()) {
|
||||
|
||||
auto f = dumper.stack.top();
|
||||
dumper.stack.pop();
|
||||
|
||||
os << "subgraph cluster_p" << f << " {\nlabel=\"Taskflow: ";
|
||||
if(f->_name.empty()) os << 'p' << f;
|
||||
else os << f->_name;
|
||||
os << "\";\n";
|
||||
_dump(os, f->_graph, dumper);
|
||||
os << "}\n";
|
||||
}
|
||||
}
|
||||
|
||||
// Procedure: _dump
|
||||
inline void Taskflow::_dump(
|
||||
std::ostream& os, const Node* node, Dumper& dumper
|
||||
) const {
|
||||
|
||||
os << 'p' << node << "[label=\"";
|
||||
if(node->_name.empty()) os << 'p' << node;
|
||||
else os << node->_name;
|
||||
os << "\" ";
|
||||
|
||||
// shape for node
|
||||
switch(node->_handle.index()) {
|
||||
|
||||
case Node::CONDITION:
|
||||
os << "shape=diamond color=black fillcolor=aquamarine style=filled";
|
||||
break;
|
||||
|
||||
case Node::CUDAFLOW:
|
||||
os << " style=\"filled\""
|
||||
<< " color=\"black\" fillcolor=\"purple\""
|
||||
<< " fontcolor=\"white\""
|
||||
<< " shape=\"folder\"";
|
||||
break;
|
||||
|
||||
case Node::SYCLFLOW:
|
||||
os << " style=\"filled\""
|
||||
<< " color=\"black\" fillcolor=\"red\""
|
||||
<< " fontcolor=\"white\""
|
||||
<< " shape=\"folder\"";
|
||||
break;
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
os << "];\n";
|
||||
|
||||
for(size_t s=0; s<node->_successors.size(); ++s) {
|
||||
if(node->_handle.index() == Node::CONDITION) {
|
||||
// case edge is dashed
|
||||
os << 'p' << node << " -> p" << node->_successors[s]
|
||||
<< " [style=dashed label=\"" << s << "\"];\n";
|
||||
}
|
||||
else {
|
||||
os << 'p' << node << " -> p" << node->_successors[s] << ";\n";
|
||||
}
|
||||
}
|
||||
|
||||
// subflow join node
|
||||
if(node->_parent && node->_successors.size() == 0) {
|
||||
os << 'p' << node << " -> p" << node->_parent << ";\n";
|
||||
}
|
||||
|
||||
switch(node->_handle.index()) {
|
||||
|
||||
case Node::DYNAMIC: {
|
||||
auto& sbg = std::get<Node::Dynamic>(node->_handle).subgraph;
|
||||
if(!sbg.empty()) {
|
||||
os << "subgraph cluster_p" << node << " {\nlabel=\"Subflow: ";
|
||||
if(node->_name.empty()) os << 'p' << node;
|
||||
else os << node->_name;
|
||||
|
||||
os << "\";\n" << "color=blue\n";
|
||||
_dump(os, sbg, dumper);
|
||||
os << "}\n";
|
||||
}
|
||||
}
|
||||
break;
|
||||
|
||||
case Node::CUDAFLOW: {
|
||||
std::get<Node::cudaFlow>(node->_handle).graph->dump(
|
||||
os, node, node->_name
|
||||
);
|
||||
}
|
||||
break;
|
||||
|
||||
case Node::SYCLFLOW: {
|
||||
std::get<Node::syclFlow>(node->_handle).graph->dump(
|
||||
os, node, node->_name
|
||||
);
|
||||
}
|
||||
break;
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Procedure: _dump
|
||||
inline void Taskflow::_dump(
|
||||
std::ostream& os, const Graph& graph, Dumper& dumper
|
||||
) const {
|
||||
|
||||
for(const auto& n : graph._nodes) {
|
||||
|
||||
// regular task
|
||||
if(n->_handle.index() != Node::MODULE) {
|
||||
_dump(os, n, dumper);
|
||||
}
|
||||
// module task
|
||||
else {
|
||||
|
||||
auto module = std::get<Node::Module>(n->_handle).module;
|
||||
|
||||
os << 'p' << n << "[shape=box3d, color=blue, label=\"";
|
||||
if(n->_name.empty()) os << n;
|
||||
else os << n->_name;
|
||||
os << " [Taskflow: ";
|
||||
if(module->_name.empty()) os << 'p' << module;
|
||||
else os << module->_name;
|
||||
os << "]\"];\n";
|
||||
|
||||
if(dumper.visited.find(module) == dumper.visited.end()) {
|
||||
dumper.visited.insert(module);
|
||||
dumper.stack.push(module);
|
||||
}
|
||||
|
||||
for(const auto s : n->_successors) {
|
||||
os << 'p' << n << "->" << 'p' << s << ";\n";
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// class definition: Future
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@class Future
|
||||
|
||||
@brief class to access the result of task execution
|
||||
|
||||
tf::Future is a derived class from std::future that will eventually hold the
|
||||
execution result of a submitted taskflow (e.g., tf::Executor::run)
|
||||
or an asynchronous task (e.g., tf::Executor::async).
|
||||
In addition to base methods of std::future,
|
||||
you can call tf::Future::cancel to cancel the execution of the running taskflow
|
||||
associated with this future object.
|
||||
The following example cancels a submission of a taskflow that contains
|
||||
1000 tasks each running one second.
|
||||
|
||||
@code{.cpp}
|
||||
tf::Executor executor;
|
||||
tf::Taskflow taskflow;
|
||||
|
||||
for(int i=0; i<1000; i++) {
|
||||
taskflow.emplace([](){
|
||||
std::this_thread::sleep_for(std::chrono::seconds(1));
|
||||
});
|
||||
}
|
||||
|
||||
// submit the taskflow
|
||||
tf::Future fu = executor.run(taskflow);
|
||||
|
||||
// request to cancel the submitted execution above
|
||||
fu.cancel();
|
||||
|
||||
// wait until the cancellation finishes
|
||||
fu.get();
|
||||
@endcode
|
||||
*/
|
||||
template <typename T>
|
||||
class Future : public std::future<T> {
|
||||
|
||||
friend class Executor;
|
||||
friend class Subflow;
|
||||
|
||||
using handle_t = std::variant<
|
||||
std::monostate, std::weak_ptr<Topology>, std::weak_ptr<AsyncTopology>
|
||||
>;
|
||||
|
||||
// variant index
|
||||
constexpr static auto ASYNC = get_index_v<std::weak_ptr<AsyncTopology>, handle_t>;
|
||||
constexpr static auto TASKFLOW = get_index_v<std::weak_ptr<Topology>, handle_t>;
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
@brief default constructor
|
||||
*/
|
||||
Future() = default;
|
||||
|
||||
/**
|
||||
@brief disabled copy constructor
|
||||
*/
|
||||
Future(const Future&) = delete;
|
||||
|
||||
/**
|
||||
@brief default move constructor
|
||||
*/
|
||||
Future(Future&&) = default;
|
||||
|
||||
/**
|
||||
@brief disabled copy assignment
|
||||
*/
|
||||
Future& operator = (const Future&) = delete;
|
||||
|
||||
/**
|
||||
@brief default move assignment
|
||||
*/
|
||||
Future& operator = (Future&&) = default;
|
||||
|
||||
/**
|
||||
@brief cancels the execution of the running taskflow associated with
|
||||
this future object
|
||||
|
||||
@return @c true if the execution can be cancelled or
|
||||
@c false if the execution has already completed
|
||||
*/
|
||||
bool cancel();
|
||||
|
||||
private:
|
||||
|
||||
handle_t _handle;
|
||||
|
||||
template <typename P>
|
||||
Future(std::future<T>&&, P&&);
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
template <typename P>
|
||||
Future<T>::Future(std::future<T>&& fu, P&& p) :
|
||||
std::future<T> {std::move(fu)},
|
||||
_handle {std::forward<P>(p)} {
|
||||
}
|
||||
|
||||
// Function: cancel
|
||||
template <typename T>
|
||||
bool Future<T>::cancel() {
|
||||
return std::visit([](auto&& arg){
|
||||
using P = std::decay_t<decltype(arg)>;
|
||||
if constexpr(std::is_same_v<P, std::monostate>) {
|
||||
return false;
|
||||
}
|
||||
else {
|
||||
auto ptr = arg.lock();
|
||||
if(ptr) {
|
||||
ptr->_is_cancelled = true;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}, _handle);
|
||||
}
|
||||
|
||||
|
||||
} // end of namespace tf. ---------------------------------------------------
|
||||
|
||||
|
||||
|
||||
|
||||
61
cpp_to_py/gpugr/taskflow/core/topology.hpp
Normal file
61
cpp_to_py/gpugr/taskflow/core/topology.hpp
Normal file
@ -0,0 +1,61 @@
|
||||
#pragma once
|
||||
|
||||
namespace tf {
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// class: TopologyBase
|
||||
class TopologyBase {
|
||||
|
||||
friend class Executor;
|
||||
friend class Node;
|
||||
|
||||
template <typename T>
|
||||
friend class Future;
|
||||
|
||||
protected:
|
||||
|
||||
std::atomic<bool> _is_cancelled { false };
|
||||
};
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// class: AsyncTopology
|
||||
class AsyncTopology : public TopologyBase {
|
||||
};
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// class: Topology
|
||||
class Topology : public TopologyBase {
|
||||
|
||||
friend class Executor;
|
||||
|
||||
public:
|
||||
|
||||
template <typename P, typename C>
|
||||
Topology(Taskflow&, P&&, C&&);
|
||||
|
||||
private:
|
||||
|
||||
Taskflow& _taskflow;
|
||||
|
||||
std::promise<void> _promise;
|
||||
|
||||
std::vector<Node*> _sources;
|
||||
|
||||
std::function<bool()> _pred;
|
||||
std::function<void()> _call;
|
||||
|
||||
std::atomic<size_t> _join_counter {0};
|
||||
};
|
||||
|
||||
// Constructor
|
||||
template <typename P, typename C>
|
||||
Topology::Topology(Taskflow& tf, P&& p, C&& c):
|
||||
_taskflow(tf),
|
||||
_pred {std::forward<P>(p)},
|
||||
_call {std::forward<C>(c)} {
|
||||
}
|
||||
|
||||
} // end of namespace tf. ----------------------------------------------------
|
||||
248
cpp_to_py/gpugr/taskflow/core/tsq.hpp
Normal file
248
cpp_to_py/gpugr/taskflow/core/tsq.hpp
Normal file
@ -0,0 +1,248 @@
|
||||
#pragma once
|
||||
|
||||
#include <atomic>
|
||||
#include <vector>
|
||||
#include <cassert>
|
||||
#include <cstdint>
|
||||
#include <cstddef>
|
||||
#include <cstdlib>
|
||||
|
||||
namespace tf {
|
||||
|
||||
/**
|
||||
@class: TaskQueue
|
||||
|
||||
@tparam T data type (must be a pointer)
|
||||
|
||||
@brief Lock-free unbounded single-producer multiple-consumer queue.
|
||||
|
||||
This class implements the work stealing queue described in the paper,
|
||||
"Correct and Efficient Work-Stealing for Weak Memory Models,"
|
||||
available at https://www.di.ens.fr/~zappa/readings/ppopp13.pdf.
|
||||
|
||||
Only the queue owner can perform pop and push operations,
|
||||
while others can steal data from the queue.
|
||||
*/
|
||||
template <typename T>
|
||||
class TaskQueue {
|
||||
|
||||
static_assert(std::is_pointer_v<T>, "T must be a pointer type");
|
||||
|
||||
struct Array {
|
||||
|
||||
int64_t C;
|
||||
int64_t M;
|
||||
std::atomic<T>* S;
|
||||
|
||||
explicit Array(int64_t c) :
|
||||
C {c},
|
||||
M {c-1},
|
||||
S {new std::atomic<T>[static_cast<size_t>(C)]} {
|
||||
}
|
||||
|
||||
~Array() {
|
||||
delete [] S;
|
||||
}
|
||||
|
||||
int64_t capacity() const noexcept {
|
||||
return C;
|
||||
}
|
||||
|
||||
void push(int64_t i, T o) noexcept {
|
||||
S[i & M].store(o, std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
T pop(int64_t i) noexcept {
|
||||
return S[i & M].load(std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
Array* resize(int64_t b, int64_t t) {
|
||||
Array* ptr = new Array {2*C};
|
||||
for(int64_t i=t; i!=b; ++i) {
|
||||
ptr->push(i, pop(i));
|
||||
}
|
||||
return ptr;
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
std::atomic<int64_t> _top;
|
||||
std::atomic<int64_t> _bottom;
|
||||
std::atomic<Array*> _array;
|
||||
std::vector<Array*> _garbage;
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
@brief constructs the queue with a given capacity
|
||||
|
||||
@param capacity the capacity of the queue (must be power of 2)
|
||||
*/
|
||||
explicit TaskQueue(int64_t capacity = 1024);
|
||||
|
||||
/**
|
||||
@brief destructs the queue
|
||||
*/
|
||||
~TaskQueue();
|
||||
|
||||
/**
|
||||
@brief queries if the queue is empty at the time of this call
|
||||
*/
|
||||
bool empty() const noexcept;
|
||||
|
||||
/**
|
||||
@brief queries the number of items at the time of this call
|
||||
*/
|
||||
size_t size() const noexcept;
|
||||
|
||||
/**
|
||||
@brief queries the capacity of the queue
|
||||
*/
|
||||
int64_t capacity() const noexcept;
|
||||
|
||||
/**
|
||||
@brief inserts an item to the queue
|
||||
|
||||
Only the owner thread can insert an item to the queue.
|
||||
The operation can trigger the queue to resize its capacity
|
||||
if more space is required.
|
||||
|
||||
@tparam O data type
|
||||
|
||||
@param item the item to perfect-forward to the queue
|
||||
*/
|
||||
void push(T item);
|
||||
|
||||
/**
|
||||
@brief pops out an item from the queue
|
||||
|
||||
Only the owner thread can pop out an item from the queue.
|
||||
The return can be a nullptr if this operation failed (empty queue).
|
||||
*/
|
||||
T pop();
|
||||
|
||||
/**
|
||||
@brief steals an item from the queue
|
||||
|
||||
Any threads can try to steal an item from the queue.
|
||||
The return can be a nullptr if this operation failed (not necessary empty).
|
||||
*/
|
||||
T steal();
|
||||
};
|
||||
|
||||
// Constructor
|
||||
template <typename T>
|
||||
TaskQueue<T>::TaskQueue(int64_t c) {
|
||||
assert(c && (!(c & (c-1))));
|
||||
_top.store(0, std::memory_order_relaxed);
|
||||
_bottom.store(0, std::memory_order_relaxed);
|
||||
_array.store(new Array{c}, std::memory_order_relaxed);
|
||||
_garbage.reserve(32);
|
||||
}
|
||||
|
||||
// Destructor
|
||||
template <typename T>
|
||||
TaskQueue<T>::~TaskQueue() {
|
||||
for(auto a : _garbage) {
|
||||
delete a;
|
||||
}
|
||||
delete _array.load();
|
||||
}
|
||||
|
||||
// Function: empty
|
||||
template <typename T>
|
||||
bool TaskQueue<T>::empty() const noexcept {
|
||||
int64_t b = _bottom.load(std::memory_order_relaxed);
|
||||
int64_t t = _top.load(std::memory_order_relaxed);
|
||||
return b <= t;
|
||||
}
|
||||
|
||||
// Function: size
|
||||
template <typename T>
|
||||
size_t TaskQueue<T>::size() const noexcept {
|
||||
int64_t b = _bottom.load(std::memory_order_relaxed);
|
||||
int64_t t = _top.load(std::memory_order_relaxed);
|
||||
return static_cast<size_t>(b >= t ? b - t : 0);
|
||||
}
|
||||
|
||||
// Function: push
|
||||
template <typename T>
|
||||
void TaskQueue<T>::push(T o) {
|
||||
int64_t b = _bottom.load(std::memory_order_relaxed);
|
||||
int64_t t = _top.load(std::memory_order_acquire);
|
||||
Array* a = _array.load(std::memory_order_relaxed);
|
||||
|
||||
// queue is full
|
||||
if(a->capacity() - 1 < (b - t)) {
|
||||
Array* tmp = a->resize(b, t);
|
||||
_garbage.push_back(a);
|
||||
std::swap(a, tmp);
|
||||
_array.store(a, std::memory_order_release);
|
||||
// Note: the original paper using relaxed causes t-san to complain
|
||||
//_array.store(a, std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
a->push(b, o);
|
||||
std::atomic_thread_fence(std::memory_order_release);
|
||||
_bottom.store(b + 1, std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
// Function: pop
|
||||
template <typename T>
|
||||
T TaskQueue<T>::pop() {
|
||||
int64_t b = _bottom.load(std::memory_order_relaxed) - 1;
|
||||
Array* a = _array.load(std::memory_order_relaxed);
|
||||
_bottom.store(b, std::memory_order_relaxed);
|
||||
std::atomic_thread_fence(std::memory_order_seq_cst);
|
||||
int64_t t = _top.load(std::memory_order_relaxed);
|
||||
|
||||
T item {nullptr};
|
||||
|
||||
if(t <= b) {
|
||||
item = a->pop(b);
|
||||
if(t == b) {
|
||||
// the last item just got stolen
|
||||
if(!_top.compare_exchange_strong(t, t+1,
|
||||
std::memory_order_seq_cst,
|
||||
std::memory_order_relaxed)) {
|
||||
item = nullptr;
|
||||
}
|
||||
_bottom.store(b + 1, std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
else {
|
||||
_bottom.store(b + 1, std::memory_order_relaxed);
|
||||
}
|
||||
|
||||
return item;
|
||||
}
|
||||
|
||||
// Function: steal
|
||||
template <typename T>
|
||||
T TaskQueue<T>::steal() {
|
||||
int64_t t = _top.load(std::memory_order_acquire);
|
||||
std::atomic_thread_fence(std::memory_order_seq_cst);
|
||||
int64_t b = _bottom.load(std::memory_order_acquire);
|
||||
|
||||
T item {nullptr};
|
||||
|
||||
if(t < b) {
|
||||
Array* a = _array.load(std::memory_order_consume);
|
||||
item = a->pop(t);
|
||||
if(!_top.compare_exchange_strong(t, t+1,
|
||||
std::memory_order_seq_cst,
|
||||
std::memory_order_relaxed)) {
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
return item;
|
||||
}
|
||||
|
||||
// Function: capacity
|
||||
template <typename T>
|
||||
int64_t TaskQueue<T>::capacity() const noexcept {
|
||||
return _array.load(std::memory_order_relaxed)->capacity();
|
||||
}
|
||||
|
||||
} // end of namespace tf -----------------------------------------------------
|
||||
127
cpp_to_py/gpugr/taskflow/core/worker.hpp
Normal file
127
cpp_to_py/gpugr/taskflow/core/worker.hpp
Normal file
@ -0,0 +1,127 @@
|
||||
#pragma once
|
||||
|
||||
#include "declarations.hpp"
|
||||
#include "tsq.hpp"
|
||||
#include "notifier.hpp"
|
||||
|
||||
/**
|
||||
@file worker.hpp
|
||||
@brief worker include file
|
||||
*/
|
||||
|
||||
namespace tf {
|
||||
|
||||
/**
|
||||
@private
|
||||
*/
|
||||
class Worker {
|
||||
|
||||
friend class Executor;
|
||||
friend class WorkerView;
|
||||
|
||||
private:
|
||||
|
||||
size_t _id;
|
||||
size_t _vtm;
|
||||
Executor* _executor;
|
||||
Notifier::Waiter* _waiter;
|
||||
std::default_random_engine _rdgen { std::random_device{}() };
|
||||
TaskQueue<Node*> _wsq;
|
||||
};
|
||||
|
||||
/**
|
||||
@private
|
||||
*/
|
||||
struct PerThreadWorker {
|
||||
|
||||
Worker* worker;
|
||||
|
||||
PerThreadWorker() : worker {nullptr} {}
|
||||
|
||||
PerThreadWorker(const PerThreadWorker&) = delete;
|
||||
PerThreadWorker(PerThreadWorker&&) = delete;
|
||||
|
||||
PerThreadWorker& operator = (const PerThreadWorker&) = delete;
|
||||
PerThreadWorker& operator = (PerThreadWorker&&) = delete;
|
||||
};
|
||||
|
||||
/**
|
||||
@private
|
||||
*/
|
||||
inline PerThreadWorker& this_worker() {
|
||||
thread_local PerThreadWorker worker;
|
||||
return worker;
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Class Definition: WorkerView
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@class WorkerView
|
||||
|
||||
@brief class to create an immutable view of a worker in an executor
|
||||
|
||||
An executor keeps a set of internal worker threads to run tasks.
|
||||
A worker view provides users an immutable interface to observe
|
||||
when a worker runs a task, and the view object is only accessible
|
||||
from an observer derived from tf::ObserverInterface.
|
||||
*/
|
||||
class WorkerView {
|
||||
|
||||
friend class Executor;
|
||||
|
||||
public:
|
||||
|
||||
/**
|
||||
@brief queries the worker id associated with the executor
|
||||
|
||||
A worker id is a unsigned integer in the range <tt>[0, N)</tt>,
|
||||
where @c N is the number of workers spawned at the construction
|
||||
time of the executor.
|
||||
*/
|
||||
size_t id() const;
|
||||
|
||||
/**
|
||||
@brief queries the size of the queue (i.e., number of pending tasks to
|
||||
run) associated with the worker
|
||||
*/
|
||||
size_t queue_size() const;
|
||||
|
||||
/**
|
||||
@brief queries the current capacity of the queue
|
||||
*/
|
||||
size_t queue_capacity() const;
|
||||
|
||||
private:
|
||||
|
||||
WorkerView(const Worker&);
|
||||
WorkerView(const WorkerView&) = default;
|
||||
|
||||
const Worker& _worker;
|
||||
|
||||
};
|
||||
|
||||
// Constructor
|
||||
inline WorkerView::WorkerView(const Worker& w) : _worker{w} {
|
||||
}
|
||||
|
||||
// function: id
|
||||
inline size_t WorkerView::id() const {
|
||||
return _worker._id;
|
||||
}
|
||||
|
||||
// Function: queue_size
|
||||
inline size_t WorkerView::queue_size() const {
|
||||
return _worker._wsq.size();
|
||||
}
|
||||
|
||||
// Function: queue_capacity
|
||||
inline size_t WorkerView::queue_capacity() const {
|
||||
return static_cast<size_t>(_worker._wsq.capacity());
|
||||
}
|
||||
|
||||
|
||||
} // end of namespact tf -----------------------------------------------------
|
||||
|
||||
|
||||
486
cpp_to_py/gpugr/taskflow/cuda/cuda_algorithm/cuda_find.hpp
Normal file
486
cpp_to_py/gpugr/taskflow/cuda/cuda_algorithm/cuda_find.hpp
Normal file
@ -0,0 +1,486 @@
|
||||
#pragma once
|
||||
|
||||
#include "../cuda_flow.hpp"
|
||||
#include "../cuda_capturer.hpp"
|
||||
#include "../cuda_meta.hpp"
|
||||
|
||||
/**
|
||||
@file cuda_find.hpp
|
||||
@brief cuda find algorithms include file
|
||||
*/
|
||||
|
||||
namespace tf::detail {
|
||||
|
||||
/** @private */
|
||||
template <typename T>
|
||||
struct cudaFindPair {
|
||||
|
||||
T key;
|
||||
unsigned index;
|
||||
|
||||
__device__ operator unsigned () const { return index; }
|
||||
};
|
||||
|
||||
/** @private */
|
||||
template <typename P, typename I, typename U>
|
||||
void cuda_find_if_loop(P&& p, I input, unsigned count, unsigned* idx, U pred) {
|
||||
|
||||
if(count == 0) {
|
||||
cuda_single_task(p, [=] __device__ () { *idx = 0; });
|
||||
return;
|
||||
}
|
||||
|
||||
using E = std::decay_t<P>;
|
||||
|
||||
auto B = (count + E::nv - 1) / E::nv;
|
||||
|
||||
// set the index to the maximum
|
||||
cuda_single_task(p, [=] __device__ () { *idx = count; });
|
||||
|
||||
// launch the kernel to atomic-find the minimum
|
||||
cuda_kernel<<<B, E::nt, 0, p.stream()>>>([=] __device__ (auto tid, auto bid) {
|
||||
|
||||
__shared__ unsigned shm_id;
|
||||
|
||||
if(!tid) {
|
||||
shm_id = count;
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
auto tile = cuda_get_tile(bid, E::nv, count);
|
||||
|
||||
auto x = cuda_mem_to_reg_strided<E::nt, E::vt>(
|
||||
input + tile.begin, tid, tile.count()
|
||||
);
|
||||
|
||||
auto id = count;
|
||||
|
||||
for(unsigned i=0; i<E::vt; i++) {
|
||||
auto j = E::nt*i + tid;
|
||||
if(j < tile.count() && pred(x[i])) {
|
||||
id = j + tile.begin;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Note: the reduce version is not faster though
|
||||
// reduce to a scalar per block.
|
||||
//__shared__ typename cudaBlockReduce<E::nt, unsigned>::Storage shm;
|
||||
|
||||
//id = cudaBlockReduce<E::nt, unsigned>()(
|
||||
// tid,
|
||||
// id,
|
||||
// shm,
|
||||
// (tile.count() < E::nt ? tile.count() : E::nt),
|
||||
// cuda_minimum<unsigned>{},
|
||||
// false
|
||||
//);
|
||||
|
||||
// only need the minimum id
|
||||
atomicMin(&shm_id, id);
|
||||
__syncthreads();
|
||||
|
||||
// reduce all to the global memory
|
||||
if(!tid) {
|
||||
atomicMin(idx, shm_id);
|
||||
//atomicMin(idx, id);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/** @private */
|
||||
template <typename P, typename I, typename O>
|
||||
void cuda_min_element_loop(
|
||||
P&& p, I input, unsigned count, unsigned* idx, O op, void* ptr
|
||||
) {
|
||||
|
||||
if(count == 0) {
|
||||
cuda_single_task(p, [=] __device__ () { *idx = 0; });
|
||||
return;
|
||||
}
|
||||
|
||||
using T = cudaFindPair<typename std::iterator_traits<I>::value_type>;
|
||||
|
||||
cuda_uninitialized_reduce_loop(p,
|
||||
cuda_make_load_iterator<T>([=]__device__(auto i){
|
||||
return T{*(input+i), i};
|
||||
}),
|
||||
count,
|
||||
idx,
|
||||
[=] __device__ (const auto& a, const auto& b) {
|
||||
return op(a.key, b.key) ? a : b;
|
||||
},
|
||||
ptr
|
||||
);
|
||||
}
|
||||
|
||||
/** @private */
|
||||
template <typename P, typename I, typename O>
|
||||
void cuda_max_element_loop(
|
||||
P&& p, I input, unsigned count, unsigned* idx, O op, void* ptr
|
||||
) {
|
||||
|
||||
if(count == 0) {
|
||||
cuda_single_task(p, [=] __device__ () { *idx = 0; });
|
||||
return;
|
||||
}
|
||||
|
||||
using T = cudaFindPair<typename std::iterator_traits<I>::value_type>;
|
||||
|
||||
cuda_uninitialized_reduce_loop(p,
|
||||
cuda_make_load_iterator<T>([=]__device__(auto i){
|
||||
return T{*(input+i), i};
|
||||
}),
|
||||
count,
|
||||
idx,
|
||||
[=] __device__ (const auto& a, const auto& b) {
|
||||
return op(a.key, b.key) ? b : a;
|
||||
},
|
||||
ptr
|
||||
);
|
||||
}
|
||||
|
||||
} // end of namespace tf::detail ---------------------------------------------
|
||||
|
||||
namespace tf {
|
||||
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// cuda_find_if
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@brief finds the index of the first element that satisfies the given criteria
|
||||
|
||||
@tparam P execution policy type
|
||||
@tparam I input iterator type
|
||||
@tparam U unary operator type
|
||||
|
||||
@param p execution policy
|
||||
@param first iterator to the beginning of the range
|
||||
@param last iterator to the end of the range
|
||||
@param idx pointer to the index of the found element
|
||||
@param op unary operator which returns @c true for the required element
|
||||
|
||||
The function launches kernels asynchronously to find the index @c idx of the
|
||||
first element in the range <tt>[first, last)</tt>
|
||||
such that <tt>op(*(first+idx))</tt> is true.
|
||||
This is equivalent to the parallel execution of the following loop:
|
||||
|
||||
@code{.cpp}
|
||||
unsigned idx = 0;
|
||||
for(; first != last; ++first, ++idx) {
|
||||
if (p(*first)) {
|
||||
return idx;
|
||||
}
|
||||
}
|
||||
return idx;
|
||||
@endcode
|
||||
*/
|
||||
template <typename P, typename I, typename U>
|
||||
void cuda_find_if(
|
||||
P&& p, I first, I last, unsigned* idx, U op
|
||||
) {
|
||||
detail::cuda_find_if_loop(p, first, std::distance(first, last), idx, op);
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// cudaFlow
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Function: find_if
|
||||
template <typename I, typename U>
|
||||
cudaTask cudaFlow::find_if(I first, I last, unsigned* idx, U op) {
|
||||
return capture([=](cudaFlowCapturer& cap){
|
||||
cap.make_optimizer<cudaLinearCapturing>();
|
||||
cap.find_if(first, last, idx, op);
|
||||
});
|
||||
}
|
||||
|
||||
// Function: find_if
|
||||
template <typename I, typename U>
|
||||
void cudaFlow::find_if(cudaTask task, I first, I last, unsigned* idx, U op) {
|
||||
capture(task, [=](cudaFlowCapturer& cap){
|
||||
cap.make_optimizer<cudaLinearCapturing>();
|
||||
cap.find_if(first, last, idx, op);
|
||||
});
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// cudaFlowCapturer
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Function: find_if
|
||||
template <typename I, typename U>
|
||||
cudaTask cudaFlowCapturer::find_if(I first, I last, unsigned* idx, U op) {
|
||||
return on([=](cudaStream_t stream) mutable {
|
||||
cudaDefaultExecutionPolicy p(stream);
|
||||
cuda_find_if(p, first, last, idx, op);
|
||||
});
|
||||
}
|
||||
|
||||
// Function: find_if
|
||||
template <typename I, typename U>
|
||||
void cudaFlowCapturer::find_if(
|
||||
cudaTask task, I first, I last, unsigned* idx, U op
|
||||
) {
|
||||
on(task, [=](cudaStream_t stream) mutable {
|
||||
cudaDefaultExecutionPolicy p(stream);
|
||||
cuda_find_if(p, first, last, idx, op);
|
||||
});
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// cuda_min_element
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@brief queries the buffer size in bytes needed to call tf::cuda_min_element
|
||||
|
||||
@tparam P execution policy type
|
||||
@tparam T value type
|
||||
|
||||
@param count number of elements to search
|
||||
|
||||
The function is used to decide the buffer size in bytes for calling
|
||||
tf::cuda_min_element.
|
||||
*/
|
||||
template <typename P, typename T>
|
||||
unsigned cuda_min_element_buffer_size(unsigned count) {
|
||||
return cuda_reduce_buffer_size<P, detail::cudaFindPair<T>>(count);
|
||||
}
|
||||
|
||||
/**
|
||||
@brief finds the index of the minimum element in a range
|
||||
|
||||
@tparam P execution policy type
|
||||
@tparam I input iterator type
|
||||
@tparam O comparator type
|
||||
|
||||
@param p execution policy object
|
||||
@param first iterator to the beginning of the range
|
||||
@param last iterator to the end of the range
|
||||
@param idx solution index of the minimum element
|
||||
@param op comparison function object
|
||||
@param buf pointer to the buffer
|
||||
|
||||
The function launches kernels asynchronously to find
|
||||
the smallest element in the range <tt>[first, last)</tt>
|
||||
using the given comparator @c op.
|
||||
You need to provide a buffer that holds at least
|
||||
tf::cuda_min_element_buffer_size bytes for internal use.
|
||||
The function is equivalent to a parallel execution of the following loop:
|
||||
|
||||
@code{.cpp}
|
||||
if(first == last) {
|
||||
return 0;
|
||||
}
|
||||
auto smallest = first;
|
||||
for (++first; first != last; ++first) {
|
||||
if (op(*first, *smallest)) {
|
||||
smallest = first;
|
||||
}
|
||||
}
|
||||
return std::distance(first, smallest);
|
||||
@endcode
|
||||
*/
|
||||
template <typename P, typename I, typename O>
|
||||
void cuda_min_element(P&& p, I first, I last, unsigned* idx, O op, void* buf) {
|
||||
detail::cuda_min_element_loop(
|
||||
p, first, std::distance(first, last), idx, op, buf
|
||||
);
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// cudaFlowCapturer::min_element
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Function: min_element
|
||||
template <typename I, typename O>
|
||||
cudaTask cudaFlowCapturer::min_element(I first, I last, unsigned* idx, O op) {
|
||||
|
||||
using T = typename std::iterator_traits<I>::value_type;
|
||||
|
||||
auto bufsz = cuda_min_element_buffer_size<cudaDefaultExecutionPolicy, T>(
|
||||
std::distance(first, last)
|
||||
);
|
||||
|
||||
return on([=, buf=MoC{cudaScopedDeviceMemory<std::byte>(bufsz)}]
|
||||
(cudaStream_t stream) mutable {
|
||||
cudaDefaultExecutionPolicy p(stream);
|
||||
cuda_min_element(p, first, last, idx, op, buf.get().data());
|
||||
});
|
||||
}
|
||||
|
||||
// Function: min_element
|
||||
template <typename I, typename O>
|
||||
void cudaFlowCapturer::min_element(
|
||||
cudaTask task, I first, I last, unsigned* idx, O op
|
||||
) {
|
||||
|
||||
using T = typename std::iterator_traits<I>::value_type;
|
||||
|
||||
auto bufsz = cuda_min_element_buffer_size<cudaDefaultExecutionPolicy, T>(
|
||||
std::distance(first, last)
|
||||
);
|
||||
|
||||
on(task, [=, buf=MoC{cudaScopedDeviceMemory<std::byte>(bufsz)}]
|
||||
(cudaStream_t stream) mutable {
|
||||
cudaDefaultExecutionPolicy p(stream);
|
||||
cuda_min_element(p, first, last, idx, op, buf.get().data());
|
||||
});
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// cudaFlow::min_element
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Function: min_element
|
||||
template <typename I, typename O>
|
||||
cudaTask cudaFlow::min_element(I first, I last, unsigned* idx, O op) {
|
||||
return capture([=](cudaFlowCapturer& cap){
|
||||
cap.make_optimizer<cudaLinearCapturing>();
|
||||
cap.min_element(first, last, idx, op);
|
||||
});
|
||||
}
|
||||
|
||||
// Function: min_element
|
||||
template <typename I, typename O>
|
||||
void cudaFlow::min_element(
|
||||
cudaTask task, I first, I last, unsigned* idx, O op
|
||||
) {
|
||||
capture(task, [=](cudaFlowCapturer& cap){
|
||||
cap.make_optimizer<cudaLinearCapturing>();
|
||||
cap.min_element(first, last, idx, op);
|
||||
});
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// cuda_max_element
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@brief queries the buffer size in bytes needed to call tf::cuda_max_element
|
||||
|
||||
@tparam P execution policy type
|
||||
@tparam T value type
|
||||
|
||||
@param count number of elements to search
|
||||
|
||||
The function is used to decide the buffer size in bytes for calling
|
||||
tf::cuda_max_element.
|
||||
*/
|
||||
template <typename P, typename T>
|
||||
unsigned cuda_max_element_buffer_size(unsigned count) {
|
||||
return cuda_reduce_buffer_size<P, detail::cudaFindPair<T>>(count);
|
||||
}
|
||||
|
||||
/**
|
||||
@brief finds the index of the maximum element in a range
|
||||
|
||||
@tparam P execution policy type
|
||||
@tparam I input iterator type
|
||||
@tparam O comparator type
|
||||
|
||||
@param p execution policy object
|
||||
@param first iterator to the beginning of the range
|
||||
@param last iterator to the end of the range
|
||||
@param idx solution index of the maximum element
|
||||
@param op comparison function object
|
||||
@param buf pointer to the buffer
|
||||
|
||||
The function launches kernels asynchronously to find
|
||||
the largest element in the range <tt>[first, last)</tt>
|
||||
using the given comparator @c op.
|
||||
You need to provide a buffer that holds at least
|
||||
tf::cuda_max_element_buffer_size bytes for internal use.
|
||||
The function is equivalent to a parallel execution of the following loop:
|
||||
|
||||
@code{.cpp}
|
||||
if(first == last) {
|
||||
return 0;
|
||||
}
|
||||
auto largest = first;
|
||||
for (++first; first != last; ++first) {
|
||||
if (op(*largest, *first)) {
|
||||
largest = first;
|
||||
}
|
||||
}
|
||||
return std::distance(first, largest);
|
||||
@endcode
|
||||
*/
|
||||
template <typename P, typename I, typename O>
|
||||
void cuda_max_element(P&& p, I first, I last, unsigned* idx, O op, void* buf) {
|
||||
detail::cuda_max_element_loop(
|
||||
p, first, std::distance(first, last), idx, op, buf
|
||||
);
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// cudaFlowCapturer::max_element
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Function: max_element
|
||||
template <typename I, typename O>
|
||||
cudaTask cudaFlowCapturer::max_element(I first, I last, unsigned* idx, O op) {
|
||||
|
||||
using T = typename std::iterator_traits<I>::value_type;
|
||||
|
||||
auto bufsz = cuda_max_element_buffer_size<cudaDefaultExecutionPolicy, T>(
|
||||
std::distance(first, last)
|
||||
);
|
||||
|
||||
return on([=, buf=MoC{cudaScopedDeviceMemory<std::byte>(bufsz)}]
|
||||
(cudaStream_t stream) mutable {
|
||||
cudaDefaultExecutionPolicy p(stream);
|
||||
cuda_max_element(p, first, last, idx, op, buf.get().data());
|
||||
});
|
||||
}
|
||||
|
||||
// Function: max_element
|
||||
template <typename I, typename O>
|
||||
void cudaFlowCapturer::max_element(
|
||||
cudaTask task, I first, I last, unsigned* idx, O op
|
||||
) {
|
||||
|
||||
using T = typename std::iterator_traits<I>::value_type;
|
||||
|
||||
auto bufsz = cuda_max_element_buffer_size<cudaDefaultExecutionPolicy, T>(
|
||||
std::distance(first, last)
|
||||
);
|
||||
|
||||
on(task, [=, buf=MoC{cudaScopedDeviceMemory<std::byte>(bufsz)}]
|
||||
(cudaStream_t stream) mutable {
|
||||
cudaDefaultExecutionPolicy p(stream);
|
||||
cuda_max_element(p, first, last, idx, op, buf.get().data());
|
||||
});
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// cudaFlow::max_element
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Function: max_element
|
||||
template <typename I, typename O>
|
||||
cudaTask cudaFlow::max_element(I first, I last, unsigned* idx, O op) {
|
||||
return capture([=](cudaFlowCapturer& cap){
|
||||
cap.make_optimizer<cudaLinearCapturing>();
|
||||
cap.max_element(first, last, idx, op);
|
||||
});
|
||||
}
|
||||
|
||||
// Function: max_element
|
||||
template <typename I, typename O>
|
||||
void cudaFlow::max_element(
|
||||
cudaTask task, I first, I last, unsigned* idx, O op
|
||||
) {
|
||||
capture(task, [=](cudaFlowCapturer& cap){
|
||||
cap.make_optimizer<cudaLinearCapturing>();
|
||||
cap.max_element(first, last, idx, op);
|
||||
});
|
||||
}
|
||||
|
||||
} // end of namespace tf -----------------------------------------------------
|
||||
|
||||
|
||||
283
cpp_to_py/gpugr/taskflow/cuda/cuda_algorithm/cuda_for_each.hpp
Normal file
283
cpp_to_py/gpugr/taskflow/cuda/cuda_algorithm/cuda_for_each.hpp
Normal file
@ -0,0 +1,283 @@
|
||||
#pragma once
|
||||
|
||||
#include "../cuda_flow.hpp"
|
||||
#include "../cuda_capturer.hpp"
|
||||
#include "../cuda_meta.hpp"
|
||||
|
||||
/**
|
||||
@file cuda_for_each.hpp
|
||||
@brief cuda parallel-iteration algorithms include file
|
||||
*/
|
||||
|
||||
namespace tf {
|
||||
|
||||
namespace detail {
|
||||
|
||||
/** @private */
|
||||
template <typename P, typename I, typename C>
|
||||
void cuda_for_each_loop(P&& p, I first, unsigned count, C c) {
|
||||
|
||||
using E = std::decay_t<P>;
|
||||
|
||||
unsigned B = (count + E::nv - 1) / E::nv;
|
||||
|
||||
cuda_kernel<<<B, E::nt, 0, p.stream()>>>(
|
||||
[=] __device__ (auto tid, auto bid) {
|
||||
auto tile = cuda_get_tile(bid, E::nv, count);
|
||||
cuda_strided_iterate<E::nt, E::vt>([=](auto, auto j) {
|
||||
c(*(first + tile.begin + j));
|
||||
}, tid, tile.count());
|
||||
});
|
||||
}
|
||||
|
||||
/** @private */
|
||||
template <typename P, typename I, typename C>
|
||||
void cuda_for_each_index_loop(
|
||||
P&& p, I first, I inc, unsigned count, C c
|
||||
) {
|
||||
|
||||
using E = std::decay_t<P>;
|
||||
|
||||
unsigned B = (count + E::nv - 1) / E::nv;
|
||||
|
||||
cuda_kernel<<<B, E::nt, 0, p.stream()>>>(
|
||||
[=]__device__(auto tid, auto bid) {
|
||||
auto tile = cuda_get_tile(bid, E::nv, count);
|
||||
cuda_strided_iterate<E::nt, E::vt>([=]__device__(auto, auto j) {
|
||||
c(first + inc*(tile.begin+j));
|
||||
}, tid, tile.count());
|
||||
});
|
||||
}
|
||||
|
||||
} // end of namespace detail -------------------------------------------------
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// cuda standard algorithms: single_task/for_each/for_each_index
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@brief runs a callable asynchronously using one kernel thread
|
||||
|
||||
@tparam P execution policy type
|
||||
@tparam C closure type
|
||||
|
||||
@param p execution policy
|
||||
@param c closure to run by one kernel thread
|
||||
|
||||
The function launches a single kernel thread to run the given callable
|
||||
through the stream in the execution policy object.
|
||||
*/
|
||||
template <typename P, typename C>
|
||||
void cuda_single_task(P&& p, C c) {
|
||||
cuda_kernel<<<1, 1, 0, p.stream()>>>(
|
||||
[=]__device__(auto, auto) mutable { c(); }
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
@brief performs asynchronous parallel iterations over a range of items
|
||||
|
||||
@tparam P execution policy type
|
||||
@tparam I input iterator type
|
||||
@tparam C unary operator type
|
||||
|
||||
@param p execution policy object
|
||||
@param first iterator to the beginning of the range
|
||||
@param last iterator to the end of the range
|
||||
@param c unary operator to apply to each dereferenced iterator
|
||||
|
||||
This function is equivalent to a parallel execution of the following loop
|
||||
on a GPU:
|
||||
|
||||
@code{.cpp}
|
||||
for(auto itr = first; itr != last; itr++) {
|
||||
c(*itr);
|
||||
}
|
||||
@endcode
|
||||
*/
|
||||
template <typename P, typename I, typename C>
|
||||
void cuda_for_each(P&& p, I first, I last, C c) {
|
||||
|
||||
unsigned count = std::distance(first, last);
|
||||
|
||||
if(count == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
detail::cuda_for_each_loop(p, first, count, c);
|
||||
}
|
||||
|
||||
/**
|
||||
@brief performs asynchronous parallel iterations over
|
||||
an index-based range of items
|
||||
|
||||
@tparam P execution policy type
|
||||
@tparam I input index type
|
||||
@tparam C unary operator type
|
||||
|
||||
@param p execution policy object
|
||||
@param first index to the beginning of the range
|
||||
@param last index to the end of the range
|
||||
@param inc step size between successive iterations
|
||||
@param c unary operator to apply to each index
|
||||
|
||||
This function is equivalent to a parallel execution of
|
||||
the following loop on a GPU:
|
||||
|
||||
@code{.cpp}
|
||||
// step is positive [first, last)
|
||||
for(auto i=first; i<last; i+=step) {
|
||||
c(i);
|
||||
}
|
||||
|
||||
// step is negative [first, last)
|
||||
for(auto i=first; i>last; i+=step) {
|
||||
c(i);
|
||||
}
|
||||
@endcode
|
||||
*/
|
||||
template <typename P, typename I, typename C>
|
||||
void cuda_for_each_index(P&& p, I first, I last, I inc, C c) {
|
||||
|
||||
if(is_range_invalid(first, last, inc)) {
|
||||
TF_THROW("invalid range [", first, ", ", last, ") with inc size ", inc);
|
||||
}
|
||||
|
||||
unsigned count = distance(first, last, inc);
|
||||
|
||||
if(count == 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
detail::cuda_for_each_index_loop(p, first, inc, count, c);
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// single_task
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
/** @private */
|
||||
template <typename C>
|
||||
__global__ void cuda_single_task(C callable) {
|
||||
callable();
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// cudaFlow
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Function: single_task
|
||||
template <typename C>
|
||||
cudaTask cudaFlow::single_task(C c) {
|
||||
return kernel(1, 1, 0, cuda_single_task<C>, c);
|
||||
}
|
||||
|
||||
// Function: single_task
|
||||
template <typename C>
|
||||
void cudaFlow::single_task(cudaTask task, C c) {
|
||||
return kernel(task, 1, 1, 0, cuda_single_task<C>, c);
|
||||
}
|
||||
|
||||
// Function: for_each
|
||||
template <typename I, typename C>
|
||||
cudaTask cudaFlow::for_each(I first, I last, C c) {
|
||||
return capture([=](cudaFlowCapturer& cap) mutable {
|
||||
cap.make_optimizer<cudaLinearCapturing>();
|
||||
cap.for_each(first, last, c);
|
||||
});
|
||||
}
|
||||
|
||||
// Function: for_each_index
|
||||
template <typename I, typename C>
|
||||
cudaTask cudaFlow::for_each_index(I first, I last, I inc, C c) {
|
||||
return capture([=](cudaFlowCapturer& cap) mutable {
|
||||
cap.make_optimizer<cudaLinearCapturing>();
|
||||
cap.for_each_index(first, last, inc, c);
|
||||
});
|
||||
}
|
||||
|
||||
// Function: for_each
|
||||
template <typename I, typename C>
|
||||
void cudaFlow::for_each(cudaTask task, I first, I last, C c) {
|
||||
capture(task, [=](cudaFlowCapturer& cap) mutable {
|
||||
cap.make_optimizer<cudaLinearCapturing>();
|
||||
cap.for_each(first, last, c);
|
||||
});
|
||||
}
|
||||
|
||||
// Function: for_each_index
|
||||
template <typename I, typename C>
|
||||
void cudaFlow::for_each_index(cudaTask task, I first, I last, I inc, C c) {
|
||||
capture(task, [=](cudaFlowCapturer& cap) mutable {
|
||||
cap.make_optimizer<cudaLinearCapturing>();
|
||||
cap.for_each_index(first, last, inc, c);
|
||||
});
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// cudaFlowCapturer
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// Function: for_each
|
||||
template <typename I, typename C>
|
||||
cudaTask cudaFlowCapturer::for_each(I first, I last, C c) {
|
||||
return on([=](cudaStream_t stream) mutable {
|
||||
cudaDefaultExecutionPolicy p(stream);
|
||||
cuda_for_each(p, first, last, c);
|
||||
});
|
||||
}
|
||||
|
||||
// Function: for_each_index
|
||||
template <typename I, typename C>
|
||||
cudaTask cudaFlowCapturer::for_each_index(I beg, I end, I inc, C c) {
|
||||
return on([=] (cudaStream_t stream) mutable {
|
||||
cudaDefaultExecutionPolicy p(stream);
|
||||
cuda_for_each_index(p, beg, end, inc, c);
|
||||
});
|
||||
}
|
||||
|
||||
// Function: for_each
|
||||
template <typename I, typename C>
|
||||
void cudaFlowCapturer::for_each(cudaTask task, I first, I last, C c) {
|
||||
on(task, [=](cudaStream_t stream) mutable {
|
||||
cudaDefaultExecutionPolicy p(stream);
|
||||
cuda_for_each(p, first, last, c);
|
||||
});
|
||||
}
|
||||
|
||||
// Function: for_each_index
|
||||
template <typename I, typename C>
|
||||
void cudaFlowCapturer::for_each_index(
|
||||
cudaTask task, I beg, I end, I inc, C c
|
||||
) {
|
||||
on(task, [=] (cudaStream_t stream) mutable {
|
||||
cudaDefaultExecutionPolicy p(stream);
|
||||
cuda_for_each_index(p, beg, end, inc, c);
|
||||
});
|
||||
}
|
||||
|
||||
// Function: single_task
|
||||
template <typename C>
|
||||
cudaTask cudaFlowCapturer::single_task(C callable) {
|
||||
return on([=] (cudaStream_t stream) mutable {
|
||||
cudaDefaultExecutionPolicy p(stream);
|
||||
cuda_single_task(p, callable);
|
||||
});
|
||||
}
|
||||
|
||||
// Function: single_task
|
||||
template <typename C>
|
||||
void cudaFlowCapturer::single_task(cudaTask task, C callable) {
|
||||
on(task, [=] (cudaStream_t stream) mutable {
|
||||
cudaDefaultExecutionPolicy p(stream);
|
||||
cuda_single_task(p, callable);
|
||||
});
|
||||
}
|
||||
|
||||
} // end of namespace tf -----------------------------------------------------
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Loading…
Reference in New Issue
Block a user