update to Xplace 2.0

This commit is contained in:
liulixinkerry 2023-04-06 13:34:26 +08:00
parent fa107b8924
commit 81257ecf08
299 changed files with 1058482 additions and 1449 deletions

1
.gitignore vendored
View File

@ -12,5 +12,6 @@ build/
*.json
result*
data/cad
data/raw
misc
.venv/

38
BENCHMARK.md Normal file
View File

@ -0,0 +1,38 @@
# Experimental Results (Last Updated on April 2023)
## Detailed Routing Performance of Xplace-Route on ISPD 2015
Xplace-Route: Routability GP + DP Flow:
```bash
python main.py --dataset ispd2015_fix --run_all True --load_from_raw True --detail_placement True --use_cell_inflate True
```
and use Innovus® to detailedly route the placement solution.
<div align="center">
<img src="img/exp_dr.png">
</div>
## Performance of Xplace on ISPD 2005
1. Xplace GP + DP Deterministic Flow:
```bash
python main.py --dataset ispd2005 --run_all True --load_from_raw True --detail_placement True
```
2. Xplace GP + DP Non-deterministic Flow:
```bash
python main.py --dataset ispd2005 --run_all True --load_from_raw True --detail_placement True --deterministic False
```
<div align="center">
<img src="img/exp_ispd2005.png">
</div>
## Performance of Xplace on ISPD 2015
1. Xplace GP + DP Deterministic Flow:
```bash
python main.py --dataset ispd2015_fix --run_all True --load_from_raw True --detail_placement True
```
2. Xplace GP + DP Non-deterministic Flow:
```bash
python main.py --dataset ispd2015_fix --run_all True --load_from_raw True --detail_placement True --deterministic False
```
<div align="center">
<img src="img/exp_ispd2015.png">
</div>

View File

@ -9,6 +9,7 @@ message(STATUS "CMAKE_BUILD_TYPE: ${CMAKE_BUILD_TYPE}")
set(CMAKE_CXX_STANDARD 17)
set(CMAKE_CXX_STANDARD_REQUIRED ON)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
message(STATUS PROJECT_SOURCE_DIR=${PROJECT_SOURCE_DIR})
set(PATH_THIRDPARTY_ROOT ${PROJECT_SOURCE_DIR}/thirdparty)
@ -34,6 +35,14 @@ message(STATUS "PYTHON_INCLUDE_DIRS: ${PYTHON_INCLUDE_DIRS}")
add_subdirectory(${PATH_THIRDPARTY_ROOT}/flute)
message(STATUS "FLUTE_INCLUDE_DIR: ${FLUTE_INCLUDE_DIR}")
# Lemon
set(LEMON_INCLUDE_DIR "${PATH_THIRDPARTY_ROOT}/lemon/include")
set(LEMON_INCLUDE_DIRS "${LEMON_INCLUDE_DIR}")
find_library(LEMON_LIBRARY emon ${PATH_THIRDPARTY_ROOT}/lemon/lib)
set(LEMON_LIBRARIES "${LEMON_LIBRARY}")
message(STATUS "LEMON_INCLUDE_DIRS: ${LEMON_INCLUDE_DIRS}")
message(STATUS "LEMON_LIBRARIES: ${LEMON_LIBRARIES}")
# Cairo
find_package(Cairo)
message(STATUS "CAIRO_INCLUDE_DIRS: ${CAIRO_INCLUDE_DIRS}")
@ -48,14 +57,14 @@ list(GET TORCH_OUTPUT_LIST 0 TORCH_INSTALL_PREFIX)
list(GET TORCH_OUTPUT_LIST 1 TORCH_ENABLE_CUDA)
list(GET TORCH_OUTPUT_LIST 2 TORCH_VERSION)
string(REPLACE "." ";" TORCH_VERSION_LIST ${TORCH_VERSION})
list(GET TORCH_VERSION_LIST 0 TORCH_MAJOR_VERSION)
list(GET TORCH_VERSION_LIST 1 TORCH_MINOR_VERSION)
list(GET TORCH_VERSION_LIST 0 TORCH_VERSION_MAJOR)
list(GET TORCH_VERSION_LIST 1 TORCH_VERSION_MINOR)
message(STATUS TORCH_INSTALL_PREFIX=${TORCH_INSTALL_PREFIX})
message(STATUS TORCH_VERSION=${TORCH_MAJOR_VERSION}.${TORCH_MINOR_VERSION})
message(STATUS TORCH_VERSION=${TORCH_VERSION_MAJOR}.${TORCH_VERSION_MINOR})
# find CUDA
if (TORCH_ENABLE_CUDA)
find_package(CUDA 11.0)
find_package(CUDA 11.4)
if (NOT CUDA_FOUND)
set(TORCH_ENABLE_CUDA 0 CACHE BOOL "Whether enable CUDA" FORCE)
message(FATAL_ERROR "Xplace only supports CUDA mode, CMake will exit." )
@ -66,16 +75,17 @@ message(STATUS TORCH_ENABLE_CUDA=${TORCH_ENABLE_CUDA})
# set cuda arch and nvcc flags
if (CUDA_FOUND)
if (NOT CUDA_ARCH_LIST)
set(CUDA_ARCH_LIST 7.0 7.5 8.0 8.6)
set(CUDA_ARCH_LIST 8.0 8.6)
endif(NOT CUDA_ARCH_LIST)
# for cuda_add_library
cuda_select_nvcc_arch_flags(CUDA_ARCH_FLAGS ${CUDA_ARCH_LIST})
message(STATUS "CUDA_ARCH_FLAGS: ${CUDA_ARCH_FLAGS}")
# set nvcc flags
set(CUDA_USE_STATIC_CUDA_RUNTIME OFF)
set(CMAKE_CUDA17_EXTENSION_COMPILE_OPTION "-std=c++17")
list(APPEND CUDA_NVCC_FLAGS ${CUDA_ARCH_FLAGS} --compiler-options;-fPIC;-std=c++17)
list(APPEND CUDA_NVCC_FLAGS ${CUDA_ARCH_FLAGS} --extended-lambda)
list(APPEND TORCH_NVCC_FLAGS -D__CUDA_NO_HALF_OPERATORS__;-D__CUDA_NO_HALF_CONVERSIONS__;-D__CUDA_NO_BFLOAT16_CONVERSIONS__;-D__CUDA_NO_HALF2_OPERATORS__;--expt-relaxed-constexpr)
list(APPEND TORCH_NVCC_FLAGS --expt-relaxed-constexpr)
list(APPEND CUDA_NVCC_FLAGS ${TORCH_NVCC_FLAGS})
message(STATUS "CUDA_NVCC_FLAGS: ${CUDA_NVCC_FLAGS}")
endif(CUDA_FOUND)
@ -125,8 +135,8 @@ function(add_pytorch_extension target_name)
target_link_libraries(${target_name}_cuda_tmp ${ARG_EXTRA_LINK_LIBRARIES} ${TORCH_LIBRARY})
target_compile_definitions(${target_name}_cuda_tmp PRIVATE
TORCH_EXTENSION_NAME=${target_name}
TORCH_MAJOR_VERSION=${TORCH_MAJOR_VERSION}
TORCH_MINOR_VERSION=${TORCH_MINOR_VERSION}
TORCH_VERSION_MAJOR=${TORCH_VERSION_MAJOR}
TORCH_VERSION_MINOR=${TORCH_VERSION_MINOR}
ENABLE_CUDA=${TORCH_ENABLE_CUDA}
${ARG_EXTRA_DEFINITIONS})
set_target_properties(${target_name}_cuda_tmp PROPERTIES
@ -146,8 +156,8 @@ function(add_pytorch_extension target_name)
endif()
target_compile_definitions(${target_name} PRIVATE
TORCH_EXTENSION_NAME=${target_name}
TORCH_MAJOR_VERSION=${TORCH_MAJOR_VERSION}
TORCH_MINOR_VERSION=${TORCH_MINOR_VERSION}
TORCH_VERSION_MAJOR=${TORCH_VERSION_MAJOR}
TORCH_VERSION_MINOR=${TORCH_VERSION_MINOR}
ENABLE_CUDA=${TORCH_ENABLE_CUDA}
${ARG_EXTRA_DEFINITIONS})
endfunction()

100
README.md
View File

@ -1,12 +1,24 @@
# Xplace
# Xplace: An Extremely Fast and Extensible Global Placement Framework
Xplace is a fast and extensible GPU accelerated global placement framework developed by the research team supervised by Prof. Evangeline F. Y. Young at The Chinese University of Hong Kong (CUHK). It achieves around 3x speedup per GP iteration compared to the state-of-the-art global placer DREAMPlace and shows high extensiblity.
## News 🚀
We are happy to announce that Xplace 2.0 is now released. Comparing to [Xplace 1.0](https://dl.acm.org/doi/abs/10.1145/3489517.3530485), this version supports the following new features:
- Support deterministic mode with only 5~25% extra GP runtime overhead.
- Implement an extremly fast GPU-accelerated detailed-routability-driven placement algorithm.
- Integrate with a GPU-accelerated detailed placer and a GPU-accelerated global router.
- Provide benchmark download and preprocess scripts, and three routability evalution scripts.
- Code refactoring.
😄 Detailed Experimental results of Xplace 2.0 are given in [BENCHMARK.md](BENCHMARK.md).
## About
Xplace is a fast and extensible GPU accelerated global placement framework developed by the research team supervised by Prof. Evangeline F. Y. Young at The Chinese University of Hong Kong (CUHK). It achieves around 3x speedup per GP iteration compared to DREAMPlace and shows high extensiblity.
As shown in the following figure, Xplace framework is built on top of PyTorch and consists of serveral independent modules. One can easily extend Xplace by applying new scheduling techniques, new gradient functions, new placement metrics and so on.
<div align="center">
<img src="assets/xplace_overview.png" width="300"/>
<img src="img/xplace_overview.png" width="300"/>
</div>
More details are in the following paper:
@ -16,41 +28,61 @@ Lixin Liu, Bangqi Fu, Martin D. F. Wong, and Evangeline F. Y. Young. "[Xplace: a
(For the Xplace-NN, please refer to branch [neural](https://github.com/cuhk-eda/Xplace/tree/neural))
## Requirements
- CMake >= 3.12
- GCC >= 7.5.0
- Boost >= 1.56.0
- CUDA >= 11.0
- Python >= 3.8
- PyTorch >= 1.10.1
- Cairo
- [CMake](https://cmake.org/) >= 3.12
- [GCC](https://gcc.gnu.org/) >= 7.5.0
- [Boost](https://www.boost.org/) >= 1.56.0
- [CUDA](https://developer.nvidia.com/cuda-toolkit) >= 11.3
- [Python](https://www.python.org/) >= 3.8
- [PyTorch](https://pytorch.org/) >= 1.12.0
- [Cairo](https://www.cairographics.org/)
- [Innovus®](https://www.cadence.com/content/cadence-www/global/en_US/home/tools/digital-design-and-signoff/soc-implementation-and-floorplanning/innovus-implementation-system.html) (version 20.14, optional, for detailed routing and design rule checking)
## Setup
1. Clone the Xplace repository. We'll call the directory that you cloned Xplace as `$XPLACE_HOME`.
```console
```bash
git clone --recursive https://github.com/cuhk-eda/Xplace
```
2. Build the shared libraries used in Xplace.
```console
```bash
cd $XPLACE_HOME
mkdir build && cd build
cmake -DPYTHON_EXECUTABLE=$(which python) ..
make -j40 && make install
```
## Prepare Data
The following script will automatically download `ispd2005`, `ispd2015` and `iccad2019` in `./data/raw`. It also preprocesses `ispd2015` benchmark to fix some errors reported by Innovus.
```bash
cd $XPLACE_HOME/data
./download_data.sh
```
## Get started
- To run GP + DP flow for ISPD2005 dataset:
```bash
# only run adaptec1
python main.py --dataset ispd2005 --design_name adaptec1 --load_from_raw True --detail_placement True
- To run GP only flow for all the designs in ISPD2005 dataset:
```console
python main.py --dataset_root your_path --dataset ispd2005 --run_all True --load_from_raw True --write_placement True
# run all the designs in ispd2005
python main.py --dataset ispd2005 --run_all True --load_from_raw True --detail_placement True
```
- To run GP + DP flow for `adaptec1` in ISPD2005 dataset:
```console
python main.py --dataset_root your_path --dataset ispd2005 --design_name adaptec1 --load_from_raw True --write_placement True --detail_placement True
- To run GP + DP flow for ISPD2015 dataset:
```bash
# only run mgc_fft_1
python main.py --dataset ispd2015_fix --design_name mgc_fft_1 --load_from_raw True --detail_placement True
# run all the designs in ispd2015
python main.py --dataset ispd2015_fix --run_all True --load_from_raw True --detail_placement True
```
**Note**: For ISPD2005 dataset, [NTUplace3](http://eda.ee.ntu.edu.tw/research.htm) is used as the detailed placement engine. For ISPD2015 dataset, please run GP only flow and launch [ABCDPlace](https://github.com/limbo018/DREAMPlace) to perform detailed placement.
- To run Routability GP + DP flow for ISPD2015 dataset:
```bash
# run all the designs in ispd2015 with routability optimization
python main.py --dataset ispd2015_fix --run_all True --load_from_raw True --detail_placement True --use_cell_inflate True
```
**NOTE**: we defaultly enable the deterministic mode. If you don't need determinism and want to run placement in an extremely fast mode, please try to set `--deterministic False` in the python arguments.
- Each run will generate serveral output files in `./result/exp_id`. These files can provide valuable information for parameter tuning.
```
@ -67,26 +99,34 @@ Please refer to `main.py`.
## Load design from preprocessed `pt` file (Optional)
The following script will dump the parsed design into a single torch `pt` file so Xplace can load the design from the `pt` file instead of parsing the input file from scratch.
```console
cd $XPLACE_HOME
python utils/convert_design_to_torch_data.py --dataset_root your_path --dataset ispd2005
```bash
cd $XPLACE_HOME/data
python utils/convert_design_to_torch_data.py --dataset ispd2005
python utils/convert_design_to_torch_data.py --dataset ispd2015_fix
python utils/convert_design_to_torch_data.py --dataset iccad2019
```
Preprocessed data is saved in `./data/cad`.
When developing a new global placement technique in Xplace, we highly suggest using the `pt` mode to save the parser time. (set `--load_from_raw False`)
```console
```bash
python main.py --dataset ispd2005 --run_all True --load_from_raw False
```
**Note**: Please remember to use the raw mode (set `--load_from_raw True`) when running detailed placement or measuring the total running time.
**Note**:
1. Please remember to use the raw mode (set `--load_from_raw True`) when measuring the total running time.
2. We currently not support `pt` mode in routability-driven mode.
## Xplace Placement Results
## Evaluate the Routability of Xplace's Solution
We provide three ways to evaluate the routability:
1. Set `--final_route_eval True` in python arguments to invoke the internal global router [GGR](https://dl.acm.org/doi/10.1145/3508352.3549474) to evaluate the placement solution. The evaluation metrics are reported in the log and recorded in `./result/exp_id/log/route.csv`. Besies, the route guide file is written in `./result/exp_id/output/design_name.guide` and
More details about using GGR in Xplace can be found in [cpp_to_py/gpugr](cpp_to_py/gpugr).
2. Use [CU-GR](https://github.com/cuhk-eda/cu-gr) to global route the placement solution. refer to [tool/cugr_ispd2015_fix](tool/cugr_ispd2015_fix) for more instructions.
3. (Optional). If Innovus® has been properly installed in your OS, you may try to use Innovus® to detailedly route the placement solution. Please refer to [tool/innovus_ispd2015_fix](tool/innovus_ispd2015_fix) for more instructions.
Benchmark | Placement Solutions
|:---:|:---:|
ISPD2005 | [Google Drive](https://drive.google.com/drive/folders/1fUzkT9ymV3n0XxfWXA0mR3WQX55hR1PB?usp=sharing)
ISPD2015 (w/o fence) | [Google Drive](https://drive.google.com/drive/folders/1UsKQ1FQ4fFi4pdJ0VoCoCCjLakhoS20Q?usp=sharing)
## Citation
If you find **Xplace** useful in your research, please consider to cite:

Binary file not shown.

Before

Width:  |  Height:  |  Size: 319 KiB

View File

@ -3,7 +3,9 @@ add_subdirectory(dct_cuda)
add_subdirectory(density_map_cuda)
add_subdirectory(draw_placement)
add_subdirectory(flute_cpp)
add_subdirectory(gpudp)
add_subdirectory(gpugr)
add_subdirectory(hpwl_cuda)
add_subdirectory(io_parser)
add_subdirectory(routedp)
add_subdirectory(wa_wirelength_hpwl_cuda)
add_subdirectory(node_pos_to_pin_pos_cuda)

View File

@ -1,3 +1,3 @@
# Add new module
To add new module, please modify `__init__.py` and `CMakeLists.txt`.
To add new module, please modify `cpp_to_py/__init__.py` and `cpp_to_py/CMakeLists.txt`.

View File

@ -6,8 +6,10 @@ __all__ = [
"io_parser",
"density_map_cuda",
"draw_placement",
"node_pos_to_pin_pos_cuda",
"wa_wirelength_hpwl_cuda",
"gpugr",
"gpudp",
"routedp",
]
from .cpybin import (
dct_cuda,
@ -16,7 +18,9 @@ from .cpybin import (
io_parser,
density_map_cuda,
draw_placement,
node_pos_to_pin_pos_cuda,
wa_wirelength_hpwl_cuda,
gpugr,
gpudp,
routedp,
)

View File

@ -45,5 +45,4 @@
using namespace std;
using utils::assert_msg;
using utils::print;
using utils::printlog;
using utils::logger;

View File

@ -24,7 +24,7 @@ void Cell::ctype(CellType* t) {
return;
}
if (_type) {
printlog(LOG_ERROR, "type of cell %s already set", _name.c_str());
logger.error("type of cell %s already set", _name.c_str());
return;
}
_type = t;
@ -58,15 +58,45 @@ bool Cell::placed() const { return (lx() != INT_MIN) && (ly() != INT_MIN); }
void Cell::place(int x, int y) {
if (_fixed) {
printlog(LOG_WARN, "moving fixed cell %s to (%d,%d)", _name.c_str(), x, y);
logger.warning("moving fixed cell %s to (%d,%d)", _name.c_str(), x, y);
}
_lx = x;
_ly = y;
}
void Cell::place(int x, int y, int orient) {
if (_fixed) {
logger.warning("moving fixed cell %s to (%d,%d)", _name.c_str(), x, y);
}
_lx = x;
_ly = y;
switch (orient) {
case 0:
_flipX = false;
_flipY = false;
break;
case 2:
_flipX = true;
_flipY = true;
break;
case 4:
_flipX = true;
_flipY = false;
break;
case 6:
_flipX = false;
_flipY = true;
break;
default:
_flipX = false;
_flipY = false;
break;
}
}
void Cell::place(int x, int y, bool flipX, bool flipY) {
if (_fixed) {
printlog(LOG_WARN, "moving fixed cell %s to (%d,%d)", _name.c_str(), x, y);
logger.warning("moving fixed cell %s to (%d,%d)", _name.c_str(), x, y);
}
_lx = x;
_ly = y;
@ -76,7 +106,7 @@ void Cell::place(int x, int y, bool flipX, bool flipY) {
void Cell::unplace() {
if (_fixed) {
printlog(LOG_WARN, "unplace fixed cell %s", _name.c_str());
logger.warning("unplace fixed cell %s", _name.c_str());
}
_lx = _ly = INT_MIN;
_flipX = _flipY = false;

View File

@ -118,6 +118,7 @@ public:
void fixed(bool fix) { _fixed = fix; }
bool placed() const;
void place(int x, int y);
void place(int x, int y, int orient);
void place(int x, int y, bool flipX, bool flipY);
void unplace();
unsigned numPins() const { return _pins.size(); }

View File

@ -14,7 +14,7 @@ Database::~Database() {
clear();
// for regions.push_back(new Region("default"));
CLEAR_POINTER_LIST(regions);
printlog(LOG_INFO, "destruct rawdb");
logger.info("destruct rawdb");
}
void Database::load() {
@ -52,7 +52,7 @@ void Database::load() {
// if (setting.Verilog != "") {
// readVerilog(setting.Verilog);
// }
printlog(LOG_INFO, "Finish loading rawdb");
logger.info("Finish loading rawdb");
}
void Database::reset() {
@ -244,20 +244,24 @@ void Database::SetupFloorplan() {
for (Site& site : sites) {
if (site.siteClassName() == "CORE") {
if (siteW != (unsigned)site.width()) {
printlog(LOG_WARN,
"siteW %d in DEF is inconsistent with siteW %d in LEF.",
static_cast<int>(siteW),
static_cast<int>(site.width()));
logger.warning("siteW %d in DEF is inconsistent with siteW %d in LEF.",
static_cast<int>(siteW),
static_cast<int>(site.width()));
}
if (siteH != site.height()) {
printlog(LOG_WARN,
"siteH %d in DEF is inconsistent with siteH %d in LEF.",
static_cast<int>(siteH),
static_cast<int>(site.height()));
logger.warning("siteH %d in DEF is inconsistent with siteH %d in LEF.",
static_cast<int>(siteH),
static_cast<int>(site.height()));
}
break;
}
}
nSitesX = (coreHX - coreLX) / siteW;
nSitesY = (coreHY - coreLY) / siteH;
if (!maxDisp) {
maxDisp = nSitesX;
}
}
void Database::SetupRegions() {
@ -294,8 +298,7 @@ void Database::SetupRegions() {
// member group name is a particular cell name
Cell* cell = getCell(member);
if (!cell) {
printlog(
LOG_ERROR, "cell name (%s) not found for group (%s)", member.c_str(), region->name().c_str());
logger.error("cell name (%s) not found for group (%s)", member.c_str(), region->name().c_str());
}
cell->region = region;
}
@ -311,8 +314,6 @@ void Database::SetupRegions() {
void Database::SetupSiteMap() {
// set up site map
nSitesX = (coreHX - coreLX) / siteW;
nSitesY = (coreHY - coreLY) / siteH;
siteMap.siteL = coreLX;
siteMap.siteR = coreHX;
siteMap.siteB = coreLY;
@ -323,17 +324,13 @@ void Database::SetupSiteMap() {
siteMap.siteNY = nSitesY;
siteMap.initSiteMap(nSitesX, nSitesY);
if (!maxDisp) {
maxDisp = nSitesX;
}
// mark site partially overlapped by fence
int nRegions = regions.size();
for (int i = 1; i < nRegions; i++) {
// skipped the default region
Region* region = regions[i];
printlog(LOG_VERBOSE, "region : %s", region->name().c_str());
logger.verbose("region : %s", region->name().c_str());
// partially overlap at left/right
vector<Rectangle> hSlices = region->rects;
Rectangle::sliceH(hSlices);
@ -478,17 +475,14 @@ void Database::SetupSiteMap() {
}
}
printlog(LOG_VERBOSE, "core area: %ld", siteMap.nSites);
printlog(LOG_VERBOSE,
"placeable: %ld (%lf%%)",
siteMap.nPlaceable,
(double)siteMap.nPlaceable / (double)siteMap.nSites * 100.0);
logger.verbose("core area: %ld", siteMap.nSites);
logger.verbose(
"placeable: %ld (%lf%%)", siteMap.nPlaceable, (double)siteMap.nPlaceable / (double)siteMap.nSites * 100.0);
for (int i = 0; i < (int)regions.size(); i++) {
printlog(LOG_VERBOSE,
"region %d : %ld (%lf%%)",
i,
siteMap.nRegionSites[i],
(double)siteMap.nRegionSites[i] / (double)siteMap.nPlaceable);
logger.verbose("region %d : %ld (%lf%%)",
i,
siteMap.nRegionSites[i],
(double)siteMap.nRegionSites[i] / (double)siteMap.nPlaceable);
}
}
@ -502,17 +496,17 @@ void Database::SetupRows() {
if (flip[y] == 0) {
flip[y] = isFlip;
} else if (flip[y] != isFlip) {
printlog(LOG_ERROR, "row flip conflict %d : %d", y, isFlip);
logger.error("row flip conflict %d : %d", y, isFlip);
flipCheckPass = false;
}
}
if (!flipCheckPass) {
printlog(LOG_ERROR, "row flip checking fail");
logger.error("row flip checking fail");
}
if (rows.size() != nSitesY) {
printlog(LOG_ERROR, "resize rows %d->%d", (int)rows.size(), nSitesY);
logger.error("resize rows %d->%d", (int)rows.size(), nSitesY);
for (Row*& row : rows) {
delete row;
row = nullptr;
@ -544,27 +538,24 @@ void Database::SetupRows() {
if (!powerNet.getRowPower(ly, hy, row->_topPower, row->_botPower)) {
if (topNormal && row->topPower() == 'x') {
if (y + 1 == nSitesY) {
printlog(LOG_WARN, "Top power rail of the row at y=%d is not connected to power rail", row->y());
logger.warning("Top power rail of the row at y=%d is not connected to power rail", row->y());
} else {
printlog(LOG_ERROR, "Top power rail of the row at y=%d is not connected to power rail", row->y());
logger.error("Top power rail of the row at y=%d is not connected to power rail", row->y());
topNormal = false;
}
}
if (botNormal && row->botPower() == 'x') {
if (y) {
printlog(
LOG_ERROR, "Bottom power rail of the row at y=%d is not connected to power rail", row->y());
logger.error("Bottom power rail of the row at y=%d is not connected to power rail", row->y());
botNormal = false;
} else {
printlog(LOG_WARN, "Bottom power rail of the row at y=%d is not connected to power rail", row->y());
logger.warning("Bottom power rail of the row at y=%d is not connected to power rail", row->y());
}
}
}
if (shrNormal && row->topPower() == row->botPower()) {
printlog(LOG_ERROR,
"Top and Bottom power rail of the row at y=%d share the same power %c",
row->y(),
row->topPower());
logger.error(
"Top and Bottom power rail of the row at y=%d share the same power %c", row->y(), row->topPower());
shrNormal = false;
}
}
@ -628,7 +619,7 @@ void Database::setup() {
SetupRows();
SetupRowSegments();
}
printlog(LOG_INFO, "Finish setting up rawdb");
logger.info("Finish setting up rawdb");
}
Layer& Database::addLayer(const string& name, const char type) {
@ -654,7 +645,7 @@ Layer& Database::addLayer(const string& name, const char type) {
Site& Database::addSite(const string& name, const string& siteClassName, const int w, const int h) {
for (unsigned i = 0; i < sites.size(); i++) {
if (name == sites[i].name()) {
printlog(LOG_WARN, "site re-defined: %s", name.c_str());
logger.warning("site re-defined: %s", name.c_str());
return sites[i];
}
}
@ -665,7 +656,7 @@ Site& Database::addSite(const string& name, const string& siteClassName, const i
ViaType* Database::addViaType(const string& name, bool isDef) {
ViaType* viatype = getViaType(name);
if (viatype) {
printlog(LOG_WARN, "via type re-defined: %s", name.c_str());
logger.warning("via type re-defined: %s", name.c_str());
return viatype;
}
viatype = new ViaType(name, isDef);
@ -677,7 +668,7 @@ ViaType* Database::addViaType(const string& name, bool isDef) {
CellType* Database::addCellType(const string& name, unsigned libcell) {
CellType* celltype = getCellType(name);
if (celltype) {
printlog(LOG_WARN, "cell type re-defined: %s", name.c_str());
logger.warning("cell type re-defined: %s", name.c_str());
return celltype;
}
celltype = new CellType(name, libcell);
@ -689,7 +680,7 @@ CellType* Database::addCellType(const string& name, unsigned libcell) {
Cell* Database::addCell(const string& name, CellType* type) {
Cell* cell = getCell(name);
if (cell) {
printlog(LOG_WARN, "cell re-defined: %s", name.c_str());
logger.warning("cell re-defined: %s", name.c_str());
if (!cell->ctype()) {
cell->ctype(type);
}
@ -704,7 +695,7 @@ Cell* Database::addCell(const string& name, CellType* type) {
IOPin* Database::addIOPin(const string& name, const string& netName, const char direction) {
IOPin* iopin = getIOPin(name);
if (iopin) {
printlog(LOG_WARN, "IO pin re-defined: %s", name.c_str());
logger.warning("IO pin re-defined: %s", name.c_str());
return iopin;
}
iopin = new IOPin(name, netName, direction);
@ -716,7 +707,7 @@ IOPin* Database::addIOPin(const string& name, const string& netName, const char
Net* Database::addNet(const string& name, const NDR* ndr) {
Net* net = getNet(name);
if (net) {
printlog(LOG_WARN, "Net re-defined: %s", name.c_str());
logger.warning("Net re-defined: %s", name.c_str());
return net;
}
net = new Net(name, ndr);
@ -748,7 +739,7 @@ Track* Database::addTrack(char direction, double start, double num, double step)
Region* Database::addRegion(const string& name, const char type) {
Region* region = getRegion(name);
if (region) {
printlog(LOG_WARN, "Region re-defined: %s", name.c_str());
logger.warning("Region re-defined: %s", name.c_str());
return region;
}
region = new Region(name, type);
@ -759,7 +750,7 @@ Region* Database::addRegion(const string& name, const char type) {
NDR* Database::addNDR(const string& name, const bool hardSpacing) {
NDR* ndr = getNDR(name);
if (ndr) {
printlog(LOG_WARN, "NDR re-defined: %s", name.c_str());
logger.warning("NDR re-defined: %s", name.c_str());
return ndr;
}
ndr = new NDR(name, hardSpacing);
@ -856,7 +847,7 @@ const Layer* Database::getCLayer(const unsigned index) const {
/* get cell type by name */
CellType* Database::getCellType(const string& name) {
unordered_map<string, CellType*>::iterator mi = name_celltypes.find(name);
robin_hood::unordered_map<string, CellType*>::iterator mi = name_celltypes.find(name);
if (mi == name_celltypes.end()) {
return nullptr;
}
@ -864,7 +855,7 @@ CellType* Database::getCellType(const string& name) {
}
Cell* Database::getCell(const string& name) {
unordered_map<string, Cell*>::iterator mi = name_cells.find(name);
robin_hood::unordered_map<string, Cell*>::iterator mi = name_cells.find(name);
if (mi == name_cells.end()) {
return nullptr;
}
@ -872,7 +863,7 @@ Cell* Database::getCell(const string& name) {
}
Net* Database::getNet(const string& name) {
unordered_map<string, Net*>::iterator mi = name_nets.find(name);
robin_hood::unordered_map<string, Net*>::iterator mi = name_nets.find(name);
if (mi == name_nets.end()) {
return nullptr;
}
@ -904,7 +895,7 @@ NDR* Database::getNDR(const string& name) const {
}
IOPin* Database::getIOPin(const string& name) const {
unordered_map<string, IOPin*>::const_iterator mi = name_iopins.find(name);
robin_hood::unordered_map<string, IOPin*>::const_iterator mi = name_iopins.find(name);
if (mi == name_iopins.end()) {
return nullptr;
}
@ -912,7 +903,7 @@ IOPin* Database::getIOPin(const string& name) const {
}
ViaType* Database::getViaType(const string& name) const {
unordered_map<string, ViaType*>::const_iterator mi = name_viatypes.find(name);
robin_hood::unordered_map<string, ViaType*>::const_iterator mi = name_viatypes.find(name);
if (mi == name_viatypes.end()) {
return nullptr;
}
@ -1031,19 +1022,19 @@ void Database::errorCheck(bool autoFix) {
for (int i = 0; i < (int)dbIssues.size(); i++) {
switch (dbIssues[i]) {
case E_ROW_EXCEED_DIE:
printlog(LOG_WARN, "row is placed out of die area");
logger.warning("row is placed out of die area");
break;
case W_NON_UNIFORM_SITE_WIDTH:
printlog(LOG_WARN, "non uniform site width detected");
logger.warning("non uniform site width detected");
break;
case W_NON_HORIZONTAL_ROW:
printlog(LOG_WARN, "non horizontal row detected");
logger.warning("non horizontal row detected");
break;
case E_NO_NET_DRIVING_PIN:
printlog(LOG_WARN, "missing net driving pin");
logger.warning("missing net driving pin");
break;
case E_MULTIPLE_NET_DRIVING_PIN:
printlog(LOG_WARN, "multiple net driving pin");
logger.warning("multiple net driving pin");
break;
default:
break;
@ -1052,7 +1043,7 @@ void Database::errorCheck(bool autoFix) {
}
void Database::checkPlaceError() {
printlog(LOG_INFO, "starting checking...");
logger.info("starting checking...");
int nError = 0;
vector<Cell*> cells = this->cells;
sort(cells.begin(), cells.end(), [](const Cell* a, const Cell* b) { return a->lx() < b->lx(); });
@ -1075,11 +1066,11 @@ void Database::checkPlaceError() {
}
}
printlog(LOG_INFO, "#overlap=%d", nError);
logger.info("#overlap=%d", nError);
}
void Database::checkDRCError() {
printlog(LOG_INFO, "starting checking...");
logger.info("starting checking...");
vector<int> nOverlapErrors(3);
vector<int> nSpacingErrors(3);
@ -1154,7 +1145,7 @@ void Database::checkDRCError() {
}
for (unsigned i = 0; i != 3; ++i) {
printlog(LOG_INFO, "m%d = %u", i + 1, metals[i].size());
logger.info("m%d = %u", i + 1, metals[i].size());
sort(metals[i].begin(), metals[i].end(), [](const Metal& a, const Metal& b) {
return (a.rect.lx == b.rect.lx) ? (a.rect.ly < b.rect.ly) : (a.rect.lx < b.rect.lx);
});
@ -1217,7 +1208,7 @@ void Database::checkDRCError() {
*/
for (unsigned i = 0; i != 3; ++i) {
printlog(LOG_INFO, "#M%u overlaps = %d", i + 1, nOverlapErrors[i]);
printlog(LOG_INFO, "#M%u spacings = %d", i + 1, nSpacingErrors[i]);
logger.info("#M%u overlaps = %d", i + 1, nOverlapErrors[i]);
logger.info("#M%u spacings = %d", i + 1, nSpacingErrors[i]);
}
}

View File

@ -75,11 +75,11 @@ public:
E_NO_NET_DRIVING_PIN
};
unordered_map<string, CellType*> name_celltypes;
unordered_map<string, Cell*> name_cells;
unordered_map<string, Net*> name_nets;
unordered_map<string, IOPin*> name_iopins;
unordered_map<string, ViaType*> name_viatypes;
robin_hood::unordered_map<string, CellType*> name_celltypes;
robin_hood::unordered_map<string, Cell*> name_cells;
robin_hood::unordered_map<string, Net*> name_nets;
robin_hood::unordered_map<string, IOPin*> name_iopins;
robin_hood::unordered_map<string, ViaType*> name_viatypes;
vector<Layer> layers;
vector<Site> sites;

View File

@ -20,10 +20,10 @@ int EdgeTypes::getEdgeSpace(const int edge1, const int edge2) const {
#ifdef DEBUG
if (edge1 < 0 || edge1 >= (int)types.size()) {
printlog(LOG_ERROR, "invalid edge ID: %d", edge1);
logger.error("invalid edge ID: %d", edge1);
}
if (edge2 < 0 || edge2 >= (int)types.size()) {
printlog(LOG_ERROR, "invalid edge ID: %d", edge2);
logger.error("invalid edge ID: %d", edge2);
}
#endif
return distTable[edge1][edge2];

View File

@ -10,7 +10,7 @@ char Track::macro() const {
case 'v':
return 'X';
default:
printlog(LOG_ERROR, "track direction not recognized: %c", direction);
logger.error("track direction not recognized: %c", direction);
return '\0';
}
}

View File

@ -24,10 +24,10 @@ public:
void addLayer(const string& layer) { layers_.push_back(layer); }
const vector<string>& getLayer() const { return layers_; }
const int getFirstTrackLoc() const { return start; }
const int getLastTrackLoc() const { return start + (num - 1) * step; }
const int getTrackLoc(unsigned trackIndex) const { return start + trackIndex * step; }
const int getPitch() const { return step; }
int getFirstTrackLoc() const { return start; }
int getLastTrackLoc() const { return start + (num - 1) * step; }
int getTrackLoc(unsigned trackIndex) const { return start + trackIndex * step; }
int getPitch() const { return step; }
char macro() const;
unsigned numLayers() const { return layers_.size(); }

View File

@ -111,7 +111,7 @@ void Net::addPin(Pin* pin) {
void PowerNet::addRail(SNet* snet, int lx, int hx, int y) {
map<int, SNet*>::iterator rail = rails.find(y);
if (rail != rails.end() && rail->second != snet) {
printlog(LOG_ERROR, "rail %s already exists at y=%d , new rail %s is from %d to %d", rail->second->name.c_str(),
logger.error("rail %s already exists at y=%d , new rail %s is from %d to %d", rail->second->name.c_str(),
y, snet->name.c_str(),
lx,
hx);

View File

@ -48,7 +48,7 @@ void Pin::getPinCenter(int& x, int& y) {
x = iopin->x + (lx + hx) / 2;
y = iopin->y + (ly + hy) / 2;
} else {
printlog(LOG_ERROR, "invalid pin %s:%d", __FILE__, __LINE__);
logger.error("invalid pin %s:%d", __FILE__, __LINE__);
x = INT_MIN;
y = INT_MIN;
}

View File

@ -60,14 +60,14 @@ public:
~IOPin();
const string& netName() const { return _netName; }
const int width() const { return this->type->getW(); }
const int height() const { return this->type->getH(); }
const int lx() const { return x; }
const int ly() const { return y; }
const int hx() const { return x + width(); }
const int hy() const { return y + height(); }
const int cx() const { return x + width() / 2; }
const int cy() const { return y + height() / 2; }
int width() const { return this->type->getW(); }
int height() const { return this->type->getH(); }
int lx() const { return x; }
int ly() const { return y; }
int hx() const { return x + width(); }
int hy() const { return y + height(); }
int cx() const { return x + width() / 2; }
int cy() const { return y + height() / 2; }
void getBounds(int& lx, int& ly, int& hx, int& hy, int& rIndex) const;
int orient() { return _orient; }
};

View File

@ -10,8 +10,8 @@ public:
int nRows;
std::string format;
unordered_map<string, int> cellMap;
unordered_map<string, int> typeMap;
robin_hood::unordered_map<string, int> cellMap;
robin_hood::unordered_map<string, int> typeMap;
vector<string> cellName;
vector<int> cellX;
vector<int> cellY;
@ -22,7 +22,7 @@ public:
vector<int> typeHeight;
vector<char> typeFixed;
vector<vector<vector<int>>> typeShapes;
unordered_map<string, int> typePinMap;
robin_hood::unordered_map<string, int> typePinMap;
vector<int> typeNPins;
vector<vector<string>> typePinName;
vector<vector<char>> typePinDir;
@ -54,7 +54,7 @@ public:
int tileH;
double blockagePorosity;
vector<int> IOPinRouteLayer;
vector<pair<int, vector<int>>> routeBlkgs; // cellID, BlockedLayers
vector<pair<int, vector<int>>> routeBlkgs; // cellID, BlockedLayers
BookshelfData() {
nCells = 0;
@ -104,7 +104,7 @@ public:
typePinY[i][j] *= scale;
}
for (int j = 0; j < typeShapes[i].size(); j++) {
for (int k = 0; k < typeShapes[i][j].size(); k++){
for (int k = 0; k < typeShapes[i][j].size(); k++) {
typeShapes[i][j][k] *= scale;
}
}
@ -176,11 +176,11 @@ public:
}
siteWidth = gcd(sizes);
siteHeight = gcd(heights);
printlog(LOG_INFO, "estimate site size = %d x %d", siteWidth, siteHeight);
logger.info("estimate site size = %d x %d", siteWidth, siteHeight);
ii = heightSet.begin();
ie = heightSet.end();
for (; ii != ie; ++ii) {
printlog(LOG_INFO, "standard cell heights: %d rows", (*ii) / siteHeight);
logger.info("standard cell heights: %d rows", (*ii) / siteHeight);
}
}
int gcd(vector<int>& nums) {
@ -199,7 +199,7 @@ public:
bool factorValid = true;
for (int i = 0; i < (int)nums.size(); i++) {
int num = nums[i];
// printlog(LOG_INFO, "%d : %d", i, num);
// logger.info("%d : %d", i, num);
if (num % factor != 0) {
factorValid = false;
break;
@ -308,11 +308,11 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
directory = auxFile.substr(0, found);
directory += "/";
}
printlog(LOG_INFO, "dir = %s", directory.c_str());
logger.info("dir = %s", directory.c_str());
ifstream fs(auxFile.c_str());
if (!fs.good()) {
printlog(LOG_ERROR, "cannot open file: %s", auxFile.c_str());
logger.error("cannot open file: %s", auxFile.c_str());
return false;
}
@ -351,7 +351,7 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
} else if (ext == ".pl") {
filePl = directory + file;
} else {
printlog(LOG_ERROR, "unrecognized file extension: %s", ext.c_str());
logger.error("unrecognized file extension: %s", ext.c_str());
}
}
// step 1: read floorplan, rows from:
@ -404,7 +404,8 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
// this->scale = 1;
}
printlog(LOG_INFO, "parsing rows");
logger.info("parsing rows");
this->LefConvertFactor = 1; // suppose 1 in bookshelf
this->dieLX = INT_MAX;
this->dieLY = INT_MAX;
this->dieHX = INT_MIN;
@ -422,11 +423,15 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
this->dieHY = std::max(this->dieHY, row->y() + bsData.siteHeight);
// make sure the parsed results are the same as the estimated results
assert_msg(bsData.rowXStep[i] == bsData.siteWidth,
"Row %s rowXStep (%d) is not equal to siteWidth (%d)",
row->name().c_str(), bsData.rowXStep[i], bsData.siteWidth);
"Row %s rowXStep (%d) is not equal to siteWidth (%d)",
row->name().c_str(),
bsData.rowXStep[i],
bsData.siteWidth);
assert_msg(bsData.rowHeight[i] == bsData.siteHeight,
"Row %s rowYStep (%d) is not equal to siteHeight (%d)",
row->name().c_str(), bsData.rowHeight[i], bsData.siteHeight);
"Row %s rowYStep (%d) is not equal to siteHeight (%d)",
row->name().c_str(),
bsData.rowHeight[i],
bsData.siteHeight);
}
// parsing gcellgrid
@ -466,7 +471,7 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
// NOTE: ICCAD/DAC 2012 does not define trackPitch and the capacity of each GCell
// cannot be directly computed by tracks. We restore the bookshelf capacity value
// in Database::Others::capV(H) and will handle them in GRDatabase.
printlog(LOG_INFO, "parsing layers");
logger.info("parsing layers");
int defaultPitch = bsData.siteWidth;
int defaultWidth = bsData.siteWidth / 2;
int defaultSpace = bsData.siteWidth - defaultWidth;
@ -516,7 +521,7 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
}
}
printlog(LOG_INFO, "parsing celltype");
logger.info("parsing celltype");
const Layer& layer = this->layers[0];
for (int i = 0; i < bsData.nTypes; i++) {
CellType* celltype = this->addCellType(bsData.typeName[i], this->celltypes.size());
@ -533,7 +538,7 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
direction = 'o';
break;
default:
printlog(LOG_ERROR, "unknown pin direction: %c", bsData.typePinDir[i][j]);
logger.error("unknown pin direction: %c", bsData.typePinDir[i][j]);
break;
}
PinType* pintype = celltype->addPin(bsData.typePinName[i][j], direction, 's');
@ -549,7 +554,7 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
}
}
printlog(LOG_INFO, "parsing cells");
logger.info("parsing cells");
for (int i = 0; i < bsData.nCells; i++) {
int typeID = bsData.cellType[i];
if (typeID < 0) {
@ -576,7 +581,7 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
}
}
printlog(LOG_INFO, "parsing nets");
logger.info("parsing nets");
for (unsigned i = 0; i != bsData.nNets; ++i) {
Net* net = this->addNet(bsData.netName[i]);
for (unsigned j = 0; j != bsData.netCells[i].size(); ++j) {
@ -587,8 +592,7 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
pin = iopin->pin;
if (pin->is_connected) {
string netName(net->name);
printlog(
LOG_WARN, "IO Pin is re-connected: %s %s", netName.c_str(), bsData.cellName[cellID].c_str());
logger.warning("IO Pin is re-connected: %s %s", netName.c_str(), bsData.cellName[cellID].c_str());
}
iopin->is_connected = true;
} else {
@ -596,11 +600,10 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
pin = cell->pin(bsData.netPins[i][j]);
if (pin->is_connected) {
string netName(net->name);
printlog(LOG_WARN,
"Pin is re-connected: %s %s %d",
netName.c_str(),
bsData.cellName[cellID].c_str(),
bsData.netPins[i][j]);
logger.warning("Pin is re-connected: %s %s %d",
netName.c_str(),
bsData.cellName[cellID].c_str(),
bsData.netPins[i][j]);
}
cell->is_connected = true;
}
@ -635,22 +638,22 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
}
bsData.clearData();
printlog(LOG_INFO, "finish reading bookshelf.");
logger.info("finish reading bookshelf.");
return true;
}
bool Database::readBSNodes(const std::string& file) {
printlog(LOG_INFO, "reading nodes");
logger.info("reading nodes");
ifstream fs(file.c_str());
if (!fs.good()) {
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
logger.error("cannot open file: %s", file.c_str());
return false;
}
// int nNodes = 0;
// int nTerminals = 0;
vector<string> tokens;
while (readBSLine(fs, tokens)) {
// printlog(LOG_INFO, "%d : %s", i++, tokens[0].c_str());
// logger.info("%d : %s", i++, tokens[0].c_str());
if (tokens[0] == "UCLA") {
continue;
} else if (tokens[0] == "NumNodes") {
@ -668,7 +671,7 @@ bool Database::readBSNodes(const std::string& file) {
// cout << cName << '\t' << cType << endl;
}
if (cType == "terminal" && cWidth > 1 && cHeight > 1) {
// printlog(LOG_INFO, "read terminal");
// logger.info("read terminal");
cType = cName;
cFixed = true;
}
@ -678,7 +681,7 @@ bool Database::readBSNodes(const std::string& file) {
}
int typeID = -1;
if (cType == "terminal") {
// printlog(LOG_INFO, "read terminal");
// logger.info("read terminal");
typeID = -1;
cFixed = true;
} else if (bsData.typeMap.find(cType) == bsData.typeMap.end()) {
@ -716,10 +719,10 @@ bool Database::readBSNodes(const std::string& file) {
}
bool Database::readBSNets(const std::string& file) {
printlog(LOG_INFO, "reading net");
logger.info("reading net");
ifstream fs(file.c_str());
if (!fs.good()) {
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
logger.error("cannot open file: %s", file.c_str());
return false;
}
vector<string> tokens;
@ -727,13 +730,13 @@ bool Database::readBSNets(const std::string& file) {
if (tokens[0] == "UCLA") {
continue;
} else if (tokens[0] == "NumNets") {
// printlog(LOG_INFO, "#nets : %d", atoi(tokens[1].c_str()));
// logger.info("#nets : %d", atoi(tokens[1].c_str()));
int numNets = atoi(tokens[1].c_str());
bsData.netName.resize(numNets);
bsData.netCells.resize(numNets);
bsData.netPins.resize(numNets);
} else if (tokens[0] == "NumPins") {
// printlog(LOG_INFO, "#pins : %d", atoi(tokens[1].c_str()));
// logger.info("#pins : %d", atoi(tokens[1].c_str()));
} else if (tokens[0] == "NetDegree") {
int degree = atoi(tokens[1].c_str());
string nName = tokens[2];
@ -744,7 +747,7 @@ bool Database::readBSNets(const std::string& file) {
string cName = tokens[0];
if (bsData.cellMap.find(cName) == bsData.cellMap.end()) {
assert(false);
printlog(LOG_ERROR, "cell not found : %s", cName.c_str());
logger.error("cell not found : %s", cName.c_str());
return false;
}
@ -773,7 +776,7 @@ bool Database::readBSNets(const std::string& file) {
stringstream ss;
ss << bsData.typeNPins[typeID];
pinName = ss.str();
// printlog(LOG_INFO, "pinname = %s", pinName.c_str());
// logger.info("pinname = %s", pinName.c_str());
}
if (typeID >= 0) {
tpName.append(bsData.typeName[typeID]);
@ -813,10 +816,10 @@ bool Database::readBSNets(const std::string& file) {
}
bool Database::readBSScl(const std::string& file) {
printlog(LOG_INFO, "reading scl");
logger.info("reading scl");
ifstream fs(file.c_str());
if (!fs.good()) {
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
logger.error("cannot open file: %s", file.c_str());
return false;
}
vector<string> tokens;
@ -857,10 +860,10 @@ bool Database::readBSScl(const std::string& file) {
}
bool Database::readBSRoute(const std::string& file) {
printlog(LOG_INFO, "reading route");
logger.info("reading route");
ifstream fs(file.c_str());
if (!fs.good()) {
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
logger.error("cannot open file: %s", file.c_str());
return false;
}
vector<string> tokens;
@ -912,7 +915,7 @@ bool Database::readBSRoute(const std::string& file) {
} else if (status == ReadingPinLayer) {
std::string cName = tokens[0];
if (bsData.cellMap.find(cName) == bsData.cellMap.end()) {
printlog(LOG_ERROR, "pin not found : %s", cName.c_str());
logger.error("pin not found : %s", cName.c_str());
getchar();
}
int cellID = bsData.cellMap[cName];
@ -921,7 +924,7 @@ bool Database::readBSRoute(const std::string& file) {
} else if (status == ReadingBlockages) {
std::string cName = tokens[0];
if (bsData.cellMap.find(cName) == bsData.cellMap.end()) {
printlog(LOG_ERROR, "cell not found : %s", cName.c_str());
logger.error("cell not found : %s", cName.c_str());
getchar();
}
int cellID = bsData.cellMap[cName];
@ -952,10 +955,10 @@ bool Database::readBSRoute(const std::string& file) {
}
bool Database::readBSShapes(const std::string& file) {
printlog(LOG_INFO, "reading shapes");
logger.info("reading shapes");
ifstream fs(file.c_str());
if (!fs.good()) {
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
logger.error("cannot open file: %s", file.c_str());
return false;
}
vector<string> tokens;
@ -987,10 +990,10 @@ bool Database::readBSShapes(const std::string& file) {
}
bool Database::readBSWts(const std::string& file) {
printlog(LOG_INFO, "reading weights");
logger.info("reading weights");
ifstream fs(file.c_str());
if (!fs.good()) {
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
logger.error("cannot open file: %s", file.c_str());
return false;
}
vector<string> tokens;
@ -999,10 +1002,10 @@ bool Database::readBSWts(const std::string& file) {
}
bool Database::readBSPl(const std::string& file) {
printlog(LOG_INFO, "reading placement");
logger.info("reading placement");
ifstream fs(file.c_str());
if (!fs.good()) {
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
logger.error("cannot open file: %s", file.c_str());
return false;
}
vector<string> tokens;
@ -1010,10 +1013,10 @@ bool Database::readBSPl(const std::string& file) {
if (tokens[0] == "UCLA") {
continue;
} else if (tokens.size() >= 4) {
unordered_map<string, int>::iterator itr = bsData.cellMap.find(tokens[0]);
robin_hood::unordered_map<string, int>::iterator itr = bsData.cellMap.find(tokens[0]);
if (itr == bsData.cellMap.end()) {
assert(false);
printlog(LOG_ERROR, "cell not found: %s", tokens[0].c_str());
logger.error("cell not found: %s", tokens[0].c_str());
return false;
}
int cell = itr->second;
@ -1036,7 +1039,7 @@ bool Database::readBSPl(const std::string& file) {
bool Database::writeBSPl(const std::string& file) {
ofstream fs(file.c_str());
if (!fs.good()) {
printlog(LOG_ERROR, "cannot open file: %s", file.c_str());
logger.error("cannot open file: %s", file.c_str());
return false;
}
fs << "UCLA pl 1.0\n";

View File

@ -6,7 +6,7 @@ bool Database::readConstraints(const std::string& file) {
string buffer;
ifstream ifs(file.c_str());
if (!ifs.good()) {
printlog(LOG_ERROR, "cannot open constraint file: %s", file.c_str());
logger.error("cannot open constraint file: %s", file.c_str());
return false;
}
while (ifs >> buffer) {
@ -18,7 +18,7 @@ bool Database::readConstraints(const std::string& file) {
string value = buffer.substr(equal + 1, unit - equal - 1);
maxDensity = atof(value.c_str()) / 100.0;
} else {
printlog(LOG_WARN, "use input max util %f", maxDensity);
logger.warning("use input max util %f", maxDensity);
}
} else if (key == "maximum_movement") {
if (!maxDisp) {
@ -26,7 +26,7 @@ bool Database::readConstraints(const std::string& file) {
string value = buffer.substr(equal + 1, unit - equal - 1);
maxDisp = atof(value.c_str());
} else {
printlog(LOG_WARN, "use input max disp %f", maxDisp);
logger.warning("use input max disp %f", maxDisp);
}
}
}
@ -37,7 +37,7 @@ bool Database::readConstraints(const std::string& file) {
bool Database::readSize(const std::string& file) {
ifstream fs(file.c_str());
if (!fs.good()) {
printlog(LOG_ERROR, "cannot open size file: %s", file.c_str());
logger.error("cannot open size file: %s", file.c_str());
return false;
}
@ -54,12 +54,12 @@ bool Database::readSize(const std::string& file) {
}
}
if (siteW == 0 || siteH == 0) {
printlog(LOG_INFO, "not enough information to retrieve placement site size");
logger.info("not enough information to retrieve placement site size");
return false;
}
#ifndef NDEBUG
printlog(LOG_INFO, "reading %s", file.c_str());
logger.info("reading %s", file.c_str());
#endif
std::set<CellType*> sized;
@ -70,7 +70,7 @@ bool Database::readSize(const std::string& file) {
if (name != "") {
Cell* cell = getCell(name);
if (cell == NULL) {
printlog(LOG_ERROR, "cell not found : %s", name.c_str());
logger.error("cell not found : %s", name.c_str());
break;
} else {
CellType* celltype = cell->ctype();

View File

@ -146,12 +146,12 @@ inline void fastCopy(char* t, const char* s, size_t n);
bool Database::readLEF(const std::string& file) {
FILE* fp;
if (!(fp = fopen(file.c_str(), "r"))) {
printlog(LOG_ERROR, "Unable to open LEF file: %s", file.c_str());
logger.error("Unable to open LEF file: %s", file.c_str());
return false;
}
#ifndef NDEBUG
printlog(LOG_INFO, "reading %s", file.c_str());
logger.info("reading %s", file.c_str());
#endif
lefrSetUnitsCbk(readLefUnits);
@ -168,7 +168,7 @@ bool Database::readLEF(const std::string& file) {
lefrReset();
int res = lefrRead(fp, file.c_str(), (void*)this);
if (res) {
printlog(LOG_ERROR, "Error in reading LEF");
logger.error("Error in reading LEF");
return false;
}
lefrReleaseNResetMemory();
@ -183,12 +183,12 @@ bool Database::readLEF(const std::string& file) {
bool Database::readDEF(const std::string& file) {
FILE* fp;
if (!(fp = fopen(file.c_str(), "r"))) {
printlog(LOG_ERROR, "Unable to open DEF file: %s", file.c_str());
logger.error("Unable to open DEF file: %s", file.c_str());
return false;
}
#ifndef NDEBUG
printlog(LOG_INFO, "reading %s", file.c_str());
logger.info("reading %s", file.c_str());
#endif
defrSetDesignCbk(readDefDesign);
@ -222,7 +222,7 @@ bool Database::readDEF(const std::string& file) {
defrReset();
int res = defrRead(fp, file.c_str(), (void*)this, 1);
if (res) {
printlog(LOG_ERROR, "Error in reading DEF");
logger.error("Error in reading DEF");
return false;
}
defrReleaseNResetMemory();
@ -236,12 +236,12 @@ bool Database::readDEFPG(const std::string& file) {
string buffer;
ifstream ifs(file.c_str());
if (!ifs.good()) {
printlog(LOG_ERROR, "Unable to open DEF PG file: %s", file.c_str());
logger.error("Unable to open DEF PG file: %s", file.c_str());
return false;
}
#ifndef NDEBUG
printlog(LOG_INFO, "reading %s", file.c_str());
logger.info("reading %s", file.c_str());
#endif
Database* db = this;
@ -280,7 +280,7 @@ bool Database::readDEFPG(const std::string& file) {
ifs >> buffer >> buffer >> shape >> buffer >> fx >> fy >> buffer >> vianame;
ViaType* viatype = db->getViaType(vianame);
if (!viatype) {
printlog(LOG_ERROR, "Via type is not defined: %s", vianame.c_str());
logger.error("Via type is not defined: %s", vianame.c_str());
return false;
}
// snet->addVia(viatype, fx, fy);
@ -305,7 +305,7 @@ bool Database::readDEFPG(const std::string& file) {
// }
Layer* layer = db->getLayer(layername);
if (!layer) {
printlog(LOG_ERROR, "Layer is not defined: %s", layername.c_str());
logger.error("Layer is not defined: %s", layername.c_str());
return false;
}
// int lx, ly, hx, hy;
@ -334,7 +334,7 @@ bool Database::readDEFPG(const std::string& file) {
} else if (type == "GROUND") {
// snet->type = 'g';
} else {
printlog(LOG_ERROR, "unknown use: %s", type.c_str());
logger.error("unknown use: %s", type.c_str());
}
}
if (buffer == ";") {
@ -390,20 +390,20 @@ bool Database::writeComponents(ofstream& ofs) {
bool Database::writeICCAD2017(const string& inputDef, const string& outputDef) {
ifstream ifs(inputDef.c_str());
if (!ifs.good()) {
printlog(LOG_ERROR, "Unable to create/open DEF: %s", inputDef.c_str());
logger.error("Unable to create/open DEF: %s", inputDef.c_str());
return false;
}
#ifndef NDEBUG
printlog(LOG_INFO, "reading %s", inputDef.c_str());
logger.info("reading %s", inputDef.c_str());
#endif
ofstream ofs(outputDef.c_str());
if (!ofs.good()) {
printlog(LOG_ERROR, "Unable to create/open DEF: %s", outputDef.c_str());
logger.error("Unable to create/open DEF: %s", outputDef.c_str());
return false;
}
printlog(LOG_INFO, "writing %s", outputDef.c_str());
logger.info("writing %s", outputDef.c_str());
string line;
while (getline(ifs, line)) {
@ -431,10 +431,10 @@ bool Database::writeICCAD2017(const string& inputDef, const string& outputDef) {
bool Database::writeICCAD2017(const string& outputDef) {
ofstream ofs(outputDef.c_str(), ios::app);
if (!ofs.good()) {
printlog(LOG_ERROR, "Unable to create/open DEF: %s", outputDef.c_str());
logger.error("Unable to create/open DEF: %s", outputDef.c_str());
return false;
}
printlog(LOG_INFO, "writing %s", outputDef.c_str());
logger.info("writing %s", outputDef.c_str());
writeComponents(ofs); // just replace the information of components, while others keep remain.
@ -447,10 +447,10 @@ bool Database::writeICCAD2017(const string& outputDef) {
bool Database::writeDEF(const string& file) {
ofstream ofs(file.c_str());
if (!ofs.good()) {
printlog(LOG_ERROR, "Unable to create/open DEF: %s", file.c_str());
logger.error("Unable to create/open DEF: %s", file.c_str());
return false;
}
printlog(LOG_INFO, "writing %s", file.c_str());
logger.info("writing %s", file.c_str());
ofs << "VERSION 5.8 ;" << endl;
ofs << "DIVIDERCHAR \"/\" ;" << endl;
@ -534,7 +534,7 @@ bool Database::writeDEF(const string& file) {
ossr << "\t+ TYPE GUIDE ;\n";
break;
default:
printlog(LOG_ERROR, "region type not recognized: %c", region->type());
logger.error("region type not recognized: %c", region->type());
ossr << " ;\n";
break;
}
@ -562,7 +562,7 @@ bool Database::writeDEF(const string& file) {
ofs << "\n\t\t+ DIRECTION INOUT";
break;
default:
printlog(LOG_ERROR, "iopin direction not recognized: %c", iopin->type->direction());
logger.error("iopin direction not recognized: %c", iopin->type->direction());
break;
}
if (iopin->x != INT_MIN || iopin->y != INT_MIN) {
@ -662,7 +662,7 @@ int readLefProp(lefrCallbackType_e c, lefiProp* prop, lefiUserData ud) {
double microndist;
sstable >> type1 >> type2 >> type3;
if (type3 == "EXCEPTABUTTED") {
printlog(LOG_WARN, "ignore EXCEPTABUTTED between %s and %s", type1.c_str(), type2.c_str());
logger.warning("ignore EXCEPTABUTTED between %s and %s", type1.c_str(), type2.c_str());
sstable >> microndist;
} else {
microndist = stod(type3);
@ -717,7 +717,7 @@ int readLefLayer(lefrCallbackType_e c, lefiLayer* leflayer, lefiUserData ud) {
type = 'r';
} else if (!strcmp(leflayer->type(), "CUT")) {
if (db->layers.empty()) {
printlog(LOG_WARN, "remove cut layer %s below the first metal layer", name.c_str());
logger.warning("remove cut layer %s below the first metal layer", name.c_str());
return 0;
}
type = 'c';
@ -849,14 +849,14 @@ int readLefLayer(lefrCallbackType_e c, lefiLayer* leflayer, lefiUserData ud) {
if (leflayer->hasSpacingNumber()) {
switch (leflayer->numSpacing()) {
case 0:
printlog(LOG_WARN, "layer has no spacing: %s", name.c_str());
logger.warning("layer has no spacing: %s", name.c_str());
return 0;
case 1:
layer.spacing = leflayer->spacing(0);
return 0;
default:
layer.spacing = leflayer->spacing(0);
printlog(LOG_WARN, "layer has multiple spacing: %s", name.c_str());
logger.warning("layer has multiple spacing: %s", name.c_str());
return 0;
}
}
@ -877,7 +877,7 @@ int readLefVia(lefrCallbackType_e c, lefiVia* lvia, lefiUserData ud) {
string layername(lvia->lefiVia::layerName(i));
Layer* layer = db->getLayer(layername);
if (!layer) {
printlog(LOG_ERROR, "layer not found: %s", layername.c_str());
logger.error("layer not found: %s", layername.c_str());
}
for (int j = 0; j < lvia->lefiVia::numRects(i); ++j) {
via->addRect(*layer,
@ -954,7 +954,7 @@ int readLefPin(lefrCallbackType_e c, lefiPin* pin, lefiUserData ud) {
} else if (use == "SIGNAL") {
// Pin is used for regular net connectivity.
} else {
printlog(LOG_ERROR, "unknown use: %s.%s", celltype->name.c_str(), use.c_str());
logger.error("unknown use: %s.%s", celltype->name.c_str(), use.c_str());
}
}
@ -964,17 +964,16 @@ int readLefPin(lefrCallbackType_e c, lefiPin* pin, lefiUserData ud) {
direction = 'o';
} else if (sDir == "OUTPUT TRISTATE") {
direction = 'o';
printlog(
LOG_WARN, "treat pin %s.%s direction %s as OUTPUT", celltype->name.c_str(), name.c_str(), sDir.c_str());
logger.warning(
"treat pin %s.%s direction %s as OUTPUT", celltype->name.c_str(), name.c_str(), sDir.c_str());
} else if (sDir == "INPUT") {
direction = 'i';
} else if (sDir == "INOUT") {
if (name != "VDD" && name != "vdd" && name != "VSS" && name != "vss") {
printlog(
LOG_WARN, "unknown pin %s.%s direction: %s", celltype->name.c_str(), name.c_str(), sDir.c_str());
logger.warning("unknown pin %s.%s direction: %s", celltype->name.c_str(), name.c_str(), sDir.c_str());
}
} else {
printlog(LOG_ERROR, "unknown pin %s.%s direction: %s", celltype->name.c_str(), name.c_str(), sDir.c_str());
logger.error("unknown pin %s.%s direction: %s", celltype->name.c_str(), name.c_str(), sDir.c_str());
}
}
@ -990,7 +989,7 @@ int readLefPin(lefrCallbackType_e c, lefiPin* pin, lefiUserData ud) {
// }
if (pin->hasTaperRule()) {
printlog(LOG_WARN, "pin %s has taper rule %s", name.c_str(), pin->taperRule());
logger.warning("pin %s has taper rule %s", name.c_str(), pin->taperRule());
}
PinType* pintype = celltype->addPin(name, direction, type);
@ -1006,7 +1005,7 @@ int readLefPin(lefrCallbackType_e c, lefiPin* pin, lefiUserData ud) {
for (unsigned i = 0; i != (unsigned)geom->numItems(); ++i) {
switch (geom->itemType(i)) {
case lefiGeomUnknown:
printlog(LOG_WARN, "lefiGeomUnknown: %s.%s", celltype->name.c_str(), name.c_str());
logger.warning("lefiGeomUnknown: %s.%s", celltype->name.c_str(), name.c_str());
break;
case lefiGeomLayerE:
layer = db->getLayer(string(geom->getLayer(i)));
@ -1132,11 +1131,8 @@ int readLefPin(lefrCallbackType_e c, lefiPin* pin, lefiUserData ud) {
}
break;
default:
printlog(LOG_WARN,
"unknown lefiGeomEnum %u: %s.%s",
geom->itemType(i),
celltype->name.c_str(),
name.c_str());
logger.warning(
"unknown lefiGeomEnum %u: %s.%s", geom->itemType(i), celltype->name.c_str(), name.c_str());
break;
}
}
@ -1148,7 +1144,7 @@ int readLefSite(lefrCallbackType_e c, lefiSite* site, lefiUserData ud) {
int convertFactor = db->LefConvertFactor;
int width = lround(site->sizeX() * convertFactor);
int height = lround(site->sizeY() * convertFactor);
printlog(LOG_INFO, "site name: %s class: %s siteW: %d siteH: %d", site->name(), site->siteClass(), width, height);
logger.info("site name: %s class: %s siteW: %d siteH: %d", site->name(), site->siteClass(), width, height);
db->addSite(site->name(), site->siteClass(), width, height);
return 0;
}
@ -1171,7 +1167,7 @@ int readLefMacro(lefrCallbackType_e c, lefiMacro* macro, lefiUserData ud) {
celltype->cls = 'b';
} else {
celltype->cls = clsname[0];
printlog(LOG_WARN, "Class type is not defined: %s", clsname);
logger.warning("Class type is not defined: %s", clsname);
}
} else {
celltype->cls = 'c';
@ -1212,17 +1208,17 @@ int readLefMacro(lefrCallbackType_e c, lefiMacro* macro, lefiUserData ud) {
} else if (edgeside == "BOTTOM") {
static bool missBot = true;
if (missBot) {
printlog(LOG_WARN, "unknown edge side: %s", edgeside.c_str());
logger.warning("unknown edge side: %s", edgeside.c_str());
missBot = false;
}
} else if (edgeside == "TOP") {
static bool missTop = true;
if (missTop) {
printlog(LOG_WARN, "unknown edge side: %s", edgeside.c_str());
logger.warning("unknown edge side: %s", edgeside.c_str());
missTop = false;
}
} else {
printlog(LOG_WARN, "unknown edge side: %s", edgeside.c_str());
logger.warning("unknown edge side: %s", edgeside.c_str());
}
}
}
@ -1308,10 +1304,10 @@ int readDefTrack(defrCallbackType_e c, defiTrack* dtrack, defiUserData ud) {
} else if (layer->direction != track->direction) {
layer->nonPreferDirTrack = *track;
} else {
printlog(LOG_ERROR, "wrong definition of tracks for layer %s", layername.c_str());
logger.error("wrong definition of tracks for layer %s", layername.c_str());
}
} else {
printlog(LOG_ERROR, "layer name not found: %s", layername.c_str());
logger.error("layer name not found: %s", layername.c_str());
}
}
return 0;
@ -1356,7 +1352,7 @@ int readDefVia(defrCallbackType_e c, defiVia* dvia, defiUserData ud) {
if (layer) {
via->addRect(*layer, lx, ly, hx, hy);
} else {
printlog(LOG_INFO, "layer name not found: %s", dvialayer);
logger.info("layer name not found: %s", dvialayer);
}
}
@ -1431,7 +1427,7 @@ int readDefNdr(defrCallbackType_e c, defiNonDefault* nd, defiUserData ud) {
string vianame(nd->viaName(i));
ViaType* viatype = db->getViaType(vianame);
if (!viatype) {
printlog(LOG_WARN, "NDR via type not found: %s", vianame.c_str());
logger.warning("NDR via type not found: %s", vianame.c_str());
}
ndr->vias.push_back(viatype);
}
@ -1457,20 +1453,16 @@ int readDefComponent(defrCallbackType_e c, defiComponent* co, defiUserData ud) {
cell->fixed(false);
if (co->placementOrient() % 2 == 1) {
// 0:N, 1:W, 2:S, 3:E, 4:FN, 5:FW, 6:FS, 7:FE
printlog(LOG_WARN,
"Cell [%s]'s placementOrient [%d] is not supported.",
cell->name().c_str(),
co->placementOrient());
logger.warning(
"Cell [%s]'s placementOrient [%d] is not supported.", cell->name().c_str(), co->placementOrient());
}
} else if (co->isFixed()) {
cell->place(co->placementX(), co->placementY(), isFlipX(co->placementOrient()), isFlipY(co->placementOrient()));
cell->fixed(true);
if (co->placementOrient() % 2 == 1) {
// 0:N, 1:W, 2:S, 3:E, 4:FN, 5:FW, 6:FS, 7:FE
printlog(LOG_WARN,
"Cell [%s]'s placementOrient [%d] is not supported.",
cell->name().c_str(),
co->placementOrient());
logger.warning(
"Cell [%s]'s placementOrient [%d] is not supported.", cell->name().c_str(), co->placementOrient());
}
}
return 0;
@ -1489,11 +1481,11 @@ int readDefPin(defrCallbackType_e c, defiPin* dpin, defiUserData ud) {
// OUTPUT to the chip, input to external
direction = 'i';
} else {
printlog(LOG_WARN, "unknown pin signal direction: %s", dpin->direction());
logger.warning("unknown pin signal direction: %s", dpin->direction());
}
} else {
string pinName(dpin->pinName());
printlog(LOG_WARN, "Pin %s has no pin signal direction", pinName.c_str());
logger.warning("Pin %s has no pin signal direction", pinName.c_str());
}
IOPin* iopin = db->addIOPin(string(dpin->pinName()), string(dpin->netName()), direction);
@ -1524,7 +1516,7 @@ int readDefBlockage(defrCallbackType_e c, defiBlockage* dblk, defiUserData ud) {
string layername(dblk->layerName());
Layer* layer = db->getLayer(layername);
if (!layer) {
printlog(LOG_ERROR, "layer not found: %s", layername.c_str());
logger.error("layer not found: %s", layername.c_str());
return 1;
}
for (int i = 0; i < dblk->numRectangles(); ++i) {
@ -1556,7 +1548,7 @@ int readDefSNet(defrCallbackType_e c, defiNet* dnet, defiUserData ud) {
} else if (use == "GROUND") {
snet->type = 'g';
} else {
printlog(LOG_ERROR, "unknown use: %s", use.c_str());
logger.error("unknown use: %s", use.c_str());
}
}
@ -1590,7 +1582,7 @@ int readDefSNet(defrCallbackType_e c, defiNet* dnet, defiUserData ud) {
layername = dpath->getLayer();
layer = db->getLayer(layername);
if (!layer) {
printlog(LOG_ERROR, "Layer is not defined: %s", layername.c_str());
logger.error("Layer is not defined: %s", layername.c_str());
return false;
}
break;
@ -1598,7 +1590,7 @@ int readDefSNet(defrCallbackType_e c, defiNet* dnet, defiUserData ud) {
vianame = dpath->getVia();
viatype = db->getViaType(vianame);
if (!viatype) {
printlog(LOG_ERROR, "Via type is not defined: %s", vianame.c_str());
logger.error("Via type is not defined: %s", vianame.c_str());
return false;
}
snet->addVia(viatype, fx, fy);
@ -1689,12 +1681,12 @@ int readDefNet(defrCallbackType_e c, defiNet* dnet, defiUserData ud) {
string designrulename(dnet->nonDefaultRule());
ndr = db->getNDR(designrulename);
if (!ndr) {
printlog(LOG_WARN, "NDR rule is not defined: %s", designrulename.c_str());
logger.warning("NDR rule is not defined: %s", designrulename.c_str());
}
}
if ((unsigned)dnet->numConnections() == 0) {
string netName(dnet->name());
printlog(LOG_WARN, "Net %s is 0-Pin net. Ignore.", netName.c_str());
logger.warning("Net %s is 0-Pin net. Ignore.", netName.c_str());
return 0;
}
@ -1706,12 +1698,12 @@ int readDefNet(defrCallbackType_e c, defiNet* dnet, defiUserData ud) {
string iopinname(dnet->pin(i));
IOPin* iopin = db->getIOPin(iopinname);
if (!iopin) {
printlog(LOG_WARN, "IO pin is not defined: %s", iopinname.c_str());
logger.warning("IO pin is not defined: %s", iopinname.c_str());
}
pin = iopin->pin;
if (pin->is_connected) {
string netName(dnet->name());
printlog(LOG_WARN, "IO Pin is re-connected: %s %s", netName.c_str(), iopinname.c_str());
logger.warning("IO Pin is re-connected: %s %s", netName.c_str(), iopinname.c_str());
}
iopin->is_connected = true;
} else {
@ -1719,16 +1711,16 @@ int readDefNet(defrCallbackType_e c, defiNet* dnet, defiUserData ud) {
string pinname(dnet->pin(i));
Cell* cell = db->getCell(cellname);
if (!cell) {
printlog(LOG_WARN, "Cell is not defined: %s", cellname.c_str());
logger.warning("Cell is not defined: %s", cellname.c_str());
}
pin = cell->pin(pinname);
if (!pin) {
string netName(dnet->name());
printlog(LOG_WARN, "Pin is not defined: %s %s %s", netName.c_str(), cellname.c_str(), pinname.c_str());
logger.warning("Pin is not defined: %s %s %s", netName.c_str(), cellname.c_str(), pinname.c_str());
}
if (pin->is_connected) {
string netName(dnet->name());
printlog(LOG_WARN, "Pin is re-connected: %s %s %s", netName.c_str(), cellname.c_str(), pinname.c_str());
logger.warning("Pin is re-connected: %s %s %s", netName.c_str(), cellname.c_str(), pinname.c_str());
}
cell->is_connected = true;
}
@ -1818,10 +1810,10 @@ int readDefRegion(defrCallbackType_e c, defiRegion* dreg, defiUserData ud) {
} else if (!strcmp(dreg->type(), "GUIDE")) {
type = 'g';
} else {
printlog(LOG_WARN, "Unknown region type: %s", dreg->type());
logger.warning("Unknown region type: %s", dreg->type());
}
} else {
printlog(LOG_WARN, "Region is defined without type, use default region type = FENCE");
logger.warning("Region is defined without type, use default region type = FENCE");
}
Region* region = db->addRegion(string(dreg->name()), type);
@ -1833,7 +1825,7 @@ int readDefRegion(defrCallbackType_e c, defiRegion* dreg, defiUserData ud) {
}
//-----Group-----
//#define GROUP_MARKER 9999999 //first mark all group member cell with this marker, then replace the value in one scan
// #define GROUP_MARKER 9999999 //first mark all group member cell with this marker, then replace the value in one scan
int readDefGroupName(defrCallbackType_e c, const char* cl, defiUserData ud) { return 0; }
int readDefGroupMember(defrCallbackType_e c, const char* cl, defiUserData ud) {
@ -1847,7 +1839,7 @@ int readDefGroup(defrCallbackType_e c, defiGroup* dgp, defiUserData ud) {
string regionname(dgp->regionName());
Region* region = db->getRegion(regionname);
if (!region) {
printlog(LOG_WARN, "Region is not defined: %s", regionname.c_str());
logger.warning("Region is not defined: %s", regionname.c_str());
return 1;
}
region->members = db->regions[0]->members;

View File

@ -74,7 +74,7 @@ bool readVerilogLine(istream &is, vector<string> &tokens) {
bool Database::readVerilog(const std::string &file) {
ifstream fs(file.c_str());
if (!fs.good()) {
printlog(LOG_ERROR, "cannot open verilog file: %s", file.c_str());
logger.error("cannot open verilog file: %s", file.c_str());
return false;
}
@ -132,7 +132,7 @@ bool Database::readVerilog(const std::string &file) {
string pinName(tokens[i]);
IOPin *iopin = getIOPin(pinName);
if (!iopin) {
printlog(LOG_ERROR, "io pin not found: %s", pinName.c_str());
logger.error("io pin not found: %s", pinName.c_str());
}
Net *net = addNet(pinName);
net->addPin(iopin->pin);

View File

@ -65,24 +65,6 @@ double mem_use::get_peak() {
#endif
}
void printlog(int log_level, const char* format, ...) {
if (!verbose_parser_log) {
return;
}
if (log_level >= GLOBAL_LOG_LEVEL) {
std::string curr_log = tstamp.get_time_stamp();
if (log_level > LOG_INFO) {
curr_log += log_level_ANSI_color(log_level);
}
std::cout << curr_log;
va_list ap;
va_start(ap, format);
vfprintf(stdout, format, ap);
printf("\n");
fflush(stdout);
}
}
void assert_msg(bool condition, const char* format, ...) {
if (!condition) {
std::cerr << "Assertion failure: ";
@ -120,6 +102,7 @@ std::string log_level_ANSI_color(int& log_level) {
}
// C++ 20 only
// void PrintfLogger::setup_logger(argparse::ArgumentParser parser) {
// std::filesystem::path current_dir(std::filesystem::current_path());
// // std::filesystem::path result_dir(parser.get<std::string>("result_dir"));
@ -139,7 +122,6 @@ std::string log_level_ANSI_color(int& log_level) {
// if (write_log) {
// f = fopen(log_file_path.c_str(), "w");
// this->warning("Logging into file would affect the elapsed time. Disable it before submission.");
// if (f == NULL) {
// this->error("Cannot open logfile {}", log_file_path);
// exit(1);

View File

@ -28,7 +28,6 @@ extern bool verbose_parser_log;
// 1. Timer
class timer {
using clock = std::chrono::high_resolution_clock;
@ -54,18 +53,6 @@ public:
// 3. Easy print
// print(a, b, c)
inline void print() { std::cout << std::endl; }
template <typename T, typename... TAIL>
void print(const T& t, TAIL... tail) {
std::cout << t << ' ';
print(tail...);
}
// "printlog(LOG_LEVEL, a, b, c...)" puts a time stamp in beginning
// try to make code compatible with old printlog(int level, const char *format, ...)
void printlog(int level, const char* format, ...);
void assert_msg(bool condition, const char* format, ...);
std::string log_level_ANSI_color(int& log_level);
@ -75,6 +62,7 @@ std::string log_level_ANSI_color(int& log_level);
class PrintfLogger {
static constexpr bool write_log = false;
FILE* f;
bool tmp_verbose_parser_log = false;
public:
// void setup_logger(argparse::ArgumentParser parser);
@ -82,6 +70,20 @@ public:
if (f != NULL) fclose(f);
}
void enable_logger() {
tmp_verbose_parser_log = verbose_parser_log;
verbose_parser_log = true;
}
void disable_logger() {
tmp_verbose_parser_log = verbose_parser_log;
verbose_parser_log = false;
}
void reset_logger() {
verbose_parser_log = tmp_verbose_parser_log;
}
template <typename... Args>
void log(int log_level, const char* format, Args&&... args) {
if (!verbose_parser_log) {
@ -105,37 +107,89 @@ public:
}
}
void log(int log_level, const char* format) {
if (!verbose_parser_log) {
return;
}
if (log_level >= GLOBAL_LOG_LEVEL) {
std::string curr_log = tstamp.get_time_stamp();
if (log_level > LOG_INFO) {
curr_log += log_level_ANSI_color(log_level);
}
std::cout << curr_log;
puts(format);
fflush(stdout);
if (write_log) {
fprintf(f, "%s", curr_log.c_str());
fputs(format, f);
fflush(f);
}
}
}
template <typename... Args>
void debug(const char* format, Args&&... args) {
log(LOG_DEBUG, format, args...);
if (sizeof...(args) != 0) {
log(LOG_DEBUG, format, args...);
} else {
log(LOG_DEBUG, format);
}
};
template <typename... Args>
void verbose(const char* format, Args&&... args) {
log(LOG_VERBOSE, format, args...);
if (sizeof...(args) != 0) {
log(LOG_VERBOSE, format, args...);
} else {
log(LOG_VERBOSE, format);
}
};
template <typename... Args>
void info(const char* format, Args&&... args) {
log(LOG_INFO, format, args...);
if (sizeof...(args) != 0) {
log(LOG_INFO, format, args...);
} else {
log(LOG_INFO, format);
}
};
template <typename... Args>
void notice(const char* format, Args&&... args) {
log(LOG_NOTICE, format, args...);
if (sizeof...(args) != 0) {
log(LOG_NOTICE, format, args...);
} else {
log(LOG_NOTICE, format);
}
};
template <typename... Args>
void warning(const char* format, Args&&... args) {
log(LOG_WARN, format, args...);
if (sizeof...(args) != 0) {
log(LOG_WARN, format, args...);
} else {
log(LOG_WARN, format);
}
};
template <typename... Args>
void error(const char* format, Args&&... args) {
log(LOG_ERROR, format, args...);
if (sizeof...(args) != 0) {
log(LOG_ERROR, format, args...);
} else {
log(LOG_ERROR, format);
}
};
template <typename... Args>
void fatal(const char* format, Args&&... args) {
log(LOG_FATAL, format, args...);
if (sizeof...(args) != 0) {
log(LOG_FATAL, format, args...);
} else {
log(LOG_FATAL, format);
}
};
template <typename... Args>
void ok(const char* format, Args&&... args) {
log(LOG_OK, format, args...);
if (sizeof...(args) != 0) {
log(LOG_OK, format, args...);
} else {
log(LOG_OK, format);
}
};
};

View File

@ -1 +0,0 @@
from . import *

View File

@ -9,10 +9,10 @@
* except tiny modifications on preprocessing and postprocessing
*/
#include <ATen/cuda/CUDAContext.h>
#include <float.h>
#include <math.h>
#include <torch/extension.h>
#include <ATen/cuda/CUDAContext.h>
#include "cuda_runtime.h"
@ -210,15 +210,15 @@ void dct2dPostprocessCudaLauncher(
dim3 blockSize(TPB, TPB, 1);
auto stream = at::cuda::getCurrentCUDAStream();
dct2dPostprocess<T, ComplexType<T>><<<gridSize, blockSize, 0, stream>>>((ComplexType<T> *)x,
y,
M,
N,
M / 2,
N / 2,
(T)(2. / (M * N)),
(T)(4. / (M * N)),
(ComplexType<T> *)expkM,
(ComplexType<T> *)expkN);
y,
M,
N,
M / 2,
N / 2,
(T)(2. / (M * N)),
(T)(4. / (M * N)),
(ComplexType<T> *)expkM,
(ComplexType<T> *)expkN);
}
// idct2_fft2
@ -712,18 +712,12 @@ void dct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at
auto N = x.size(-1);
auto M = x.numel() / N;
AT_DISPATCH_ALL_TYPES(x.scalar_type(), "dct2_fft2_forward_cuda", [&] {
dct2dPreprocessCudaLauncher<scalar_t>(x.data_ptr<scalar_t>(), out.data_ptr<scalar_t>(), M, N);
dct2dPreprocessCudaLauncher<float>(x.data_ptr<float>(), out.data_ptr<float>(), M, N);
buf = at::view_as_real(at::fft_rfft2(out, c10::nullopt, {-2, -1}, "backward")).contiguous();
buf = at::view_as_real(at::fft_rfft2(out, c10::nullopt, {-2, -1}, "backward")).contiguous();
dct2dPostprocessCudaLauncher<scalar_t>(buf.data_ptr<scalar_t>(),
out.data_ptr<scalar_t>(),
M,
N,
expkM.data_ptr<scalar_t>(),
expkN.data_ptr<scalar_t>());
});
dct2dPostprocessCudaLauncher<float>(
buf.data_ptr<float>(), out.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
}
void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
@ -737,18 +731,12 @@ void idct2_fft2_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a
auto N = x.size(-1);
auto M = x.numel() / N;
AT_DISPATCH_ALL_TYPES(x.scalar_type(), "idct2_fft2_forward_cuda", [&] {
idct2_fft2PreprocessCudaLauncher<scalar_t>(x.data_ptr<scalar_t>(),
buf.data_ptr<scalar_t>(),
M,
N,
expkM.data_ptr<scalar_t>(),
expkN.data_ptr<scalar_t>());
idct2_fft2PreprocessCudaLauncher<float>(
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
idct2_fft2PostprocessCudaLauncher<scalar_t>(y.data_ptr<scalar_t>(), out.data_ptr<scalar_t>(), M, N);
});
idct2_fft2PostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
}
void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
@ -762,18 +750,12 @@ void idct_idxst_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a
auto N = x.size(-1);
auto M = x.numel() / N;
AT_DISPATCH_ALL_TYPES(x.scalar_type(), "idct_idxst_forward_cuda", [&] {
idct_idxstPreprocessCudaLauncher<scalar_t>(x.data_ptr<scalar_t>(),
buf.data_ptr<scalar_t>(),
M,
N,
expkM.data_ptr<scalar_t>(),
expkN.data_ptr<scalar_t>());
idct_idxstPreprocessCudaLauncher<float>(
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
idct_idxstPostprocessCudaLauncher<scalar_t>(y.data_ptr<scalar_t>(), out.data_ptr<scalar_t>(), M, N);
});
idct_idxstPostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
}
void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, at::Tensor out, at::Tensor buf) {
@ -787,16 +769,10 @@ void idxst_idct_forward_cuda(at::Tensor x, at::Tensor expkM, at::Tensor expkN, a
auto N = x.size(-1);
auto M = x.numel() / N;
AT_DISPATCH_ALL_TYPES(x.scalar_type(), "idxst_idct_forward_cuda", [&] {
idxst_idctPreprocessCudaLauncher<scalar_t>(x.data_ptr<scalar_t>(),
buf.data_ptr<scalar_t>(),
M,
N,
expkM.data_ptr<scalar_t>(),
expkN.data_ptr<scalar_t>());
idxst_idctPreprocessCudaLauncher<float>(
x.data_ptr<float>(), buf.data_ptr<float>(), M, N, expkM.data_ptr<float>(), expkN.data_ptr<float>());
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
auto y = at::fft_irfft2(at::view_as_complex(buf), {{M, N}}, {-2, -1}, "backward").contiguous();
idxst_idctPostprocessCudaLauncher<scalar_t>(y.data_ptr<scalar_t>(), out.data_ptr<scalar_t>(), M, N);
});
idxst_idctPostprocessCudaLauncher<float>(y.data_ptr<float>(), out.data_ptr<float>(), M, N);
}

View File

@ -15,7 +15,8 @@ torch::Tensor density_map_cuda_forward(torch::Tensor normalize_node_info,
torch::Tensor aux_mat,
int num_bin_x,
int num_bin_y,
int num_nodes);
int num_nodes,
bool deterministic);
torch::Tensor density_map_cuda_backward(torch::Tensor normalize_node_info,
torch::Tensor grad_mat,
@ -24,7 +25,8 @@ torch::Tensor density_map_cuda_backward(torch::Tensor normalize_node_info,
float grad_weight,
int num_bin_x,
int num_bin_y,
int num_nodes);
int num_nodes,
bool deterministic);
torch::Tensor density_map_cuda_forward_naive(torch::Tensor node_pos,
torch::Tensor node_size,
@ -37,7 +39,8 @@ torch::Tensor density_map_cuda_forward_naive(torch::Tensor node_pos,
float min_node_w,
float min_node_h,
float margin,
bool clamp_node);
bool clamp_node,
bool deterministic);
#define CHECK_CUDA(x) TORCH_CHECK(x.device().is_cuda(), #x " must be a CUDA tensor")
#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
@ -61,27 +64,22 @@ torch::Tensor density_map_normalize_node(torch::Tensor node_pos,
CHECK_INPUT(unit_len);
CHECK_INPUT(normalize_node_info);
return density_map_cuda_normalize_node(node_pos,
node_size,
node_weight,
expand_ratio,
unit_len,
normalize_node_info,
num_bin_x,
num_bin_y,
num_nodes);
return density_map_cuda_normalize_node(
node_pos, node_size, node_weight, expand_ratio, unit_len, normalize_node_info, num_bin_x, num_bin_y, num_nodes);
}
torch::Tensor density_map_forward(torch::Tensor normalize_node_info,
torch::Tensor sorted_node_map,
torch::Tensor aux_mat,
int num_bin_x,
int num_bin_y,
int num_nodes) {
int num_nodes,
bool deterministic) {
CHECK_INPUT(normalize_node_info);
CHECK_INPUT(sorted_node_map);
CHECK_INPUT(aux_mat);
return density_map_cuda_forward(normalize_node_info, sorted_node_map, aux_mat, num_bin_x, num_bin_y, num_nodes);
return density_map_cuda_forward(
normalize_node_info, sorted_node_map, aux_mat, num_bin_x, num_bin_y, num_nodes, deterministic);
}
torch::Tensor density_map_backward(torch::Tensor normalize_node_info,
@ -91,14 +89,22 @@ torch::Tensor density_map_backward(torch::Tensor normalize_node_info,
float grad_weight,
int num_bin_x,
int num_bin_y,
int num_nodes) {
int num_nodes,
bool deterministic) {
CHECK_INPUT(normalize_node_info);
CHECK_INPUT(grad_mat);
CHECK_INPUT(sorted_node_map);
CHECK_INPUT(node_grad);
return density_map_cuda_backward(
normalize_node_info, grad_mat, sorted_node_map, node_grad, grad_weight, num_bin_x, num_bin_y, num_nodes);
return density_map_cuda_backward(normalize_node_info,
grad_mat,
sorted_node_map,
node_grad,
grad_weight,
num_bin_x,
num_bin_y,
num_nodes,
deterministic);
}
torch::Tensor density_map_forward_naive(torch::Tensor node_pos,
@ -112,7 +118,8 @@ torch::Tensor density_map_forward_naive(torch::Tensor node_pos,
float min_node_w,
float min_node_h,
float margin,
bool clamp_node) {
bool clamp_node,
bool deterministic) {
CHECK_INPUT(node_pos);
CHECK_INPUT(node_size);
CHECK_INPUT(node_weight);
@ -130,12 +137,13 @@ torch::Tensor density_map_forward_naive(torch::Tensor node_pos,
min_node_w,
min_node_h,
margin,
clamp_node);
clamp_node,
deterministic);
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
m.def("pre_normalize", &density_map_normalize_node, "normalize bin size to 1");
m.def("forward", &density_map_forward, "get density map from node information");
m.def("forward_naive", &density_map_cuda_forward_naive, "calculate density map");
m.def("forward_naive", &density_map_forward_naive, "calculate density map");
m.def("backward", &density_map_backward, "calculate density gradient of each node");
}

View File

@ -1,59 +1,58 @@
#include <ATen/cuda/CUDAContext.h>
#include <cuda.h>
#include <cuda_runtime.h>
#include <torch/extension.h>
#include <ATen/cuda/CUDAContext.h>
#include <THC/THCAtomics.cuh>
#include <vector>
template <typename scalar_t>
__device__ scalar_t overlap(scalar_t x_l, scalar_t x_h, scalar_t bin_x_l) {
template <typename T>
__device__ T overlap(T x_l, T x_h, T bin_x_l) {
// bin_x_h == bin_x_l + 1
return min(x_h, bin_x_l + 1) - max(x_l, bin_x_l);
}
template <typename scalar_t>
__global__ void density_map_cuda_normalize_node_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_size,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> node_weight,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> expand_ratio,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> unit_len,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> normalize_node_info,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_size,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> node_weight,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> expand_ratio,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> unit_len,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> normalize_node_info,
int num_bin_x,
int num_bin_y,
int num_nodes) {
const int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_nodes) {
// normalize bin_size_x and bin_size_y to 1
normalize_node_info[i][0] = (node_pos[i][0] - node_size[i][0] / 2) / unit_len[0]; // x_l
normalize_node_info[i][1] = (node_pos[i][0] + node_size[i][0] / 2) / unit_len[0]; // x_h
normalize_node_info[i][2] = (node_pos[i][1] - node_size[i][1] / 2) / unit_len[1]; // y_l
normalize_node_info[i][3] = (node_pos[i][1] + node_size[i][1] / 2) / unit_len[1]; // y_h
normalize_node_info[i][4] = node_weight[i] * expand_ratio[i]; // weight
if (normalize_node_info[i][1] - normalize_node_info[i][0] < 0 ||
normalize_node_info[i][3] - normalize_node_info[i][2] < 0) {
normalize_node_info[i][3] - normalize_node_info[i][2] < 0 ||
(node_size[i][0] < 1e-6 && node_size[i][1] < 1e-6)) {
normalize_node_info[i][4] = -normalize_node_info[i][4]; // we should ignore node whose weight <= 0
}
}
}
template <typename scalar_t>
__global__ void __launch_bounds__(256, 4) density_map_cuda_forward_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> normalize_node_info,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> normalize_node_info,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> sorted_node_map,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> aux_mat,
float *aux_mat,
int num_nodes,
int num_bin_x,
int num_bin_y) {
const int index = blockIdx.x * blockDim.z + threadIdx.z;
if (index < num_nodes) {
const int i = (sorted_node_map[index] >= 0) ? sorted_node_map[index] : index;
const scalar_t weight = normalize_node_info[i][4];
const float weight = normalize_node_info[i][4];
if (weight > 0) {
const scalar_t x_l = normalize_node_info[i][0];
const scalar_t x_h = normalize_node_info[i][1];
const scalar_t y_l = normalize_node_info[i][2];
const scalar_t y_h = normalize_node_info[i][3];
const float x_l = normalize_node_info[i][0];
const float x_h = normalize_node_info[i][1];
const float y_l = normalize_node_info[i][2];
const float y_h = normalize_node_info[i][3];
int x_lf = lround(floor(x_l));
int x_hf = lround(floor(x_h));
int y_lf = lround(floor(y_l));
@ -64,25 +63,65 @@ __global__ void __launch_bounds__(256, 4) density_map_cuda_forward_kernel(
y_hf = min(y_hf, num_bin_y - 1);
for (int j = x_lf + threadIdx.y; j < x_hf + 1; j += blockDim.y) {
scalar_t bin_x_l = static_cast<scalar_t>(j);
scalar_t overlap_x = overlap(x_l, x_h, bin_x_l);
float bin_x_l = static_cast<float>(j);
float overlap_x = overlap(x_l, x_h, bin_x_l);
for (int k = y_lf + threadIdx.x; k < y_hf + 1; k += blockDim.x) {
scalar_t bin_y_l = static_cast<scalar_t>(k);
scalar_t overlap_y = overlap(y_l, y_h, bin_y_l);
scalar_t overlap_area = overlap_x * overlap_y;
gpuAtomicAdd(&aux_mat[j][k], weight * overlap_area);
float bin_y_l = static_cast<float>(k);
float overlap_y = overlap(y_l, y_h, bin_y_l);
float overlap_area = overlap_x * overlap_y;
atomicAdd(&aux_mat[j * num_bin_y + k], weight * overlap_area);
}
}
}
}
}
template <typename scalar_t>
__global__ void __launch_bounds__(256, 4) density_map_cuda_backward_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> normalize_node_info,
const scalar_t *grad_mat,
__global__ void __launch_bounds__(256, 4) density_map_cuda_deterministic_forward_kernel(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> normalize_node_info,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> sorted_node_map,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_grad,
unsigned long long *aux_mat,
int num_nodes,
int num_bin_x,
int num_bin_y,
unsigned long long scalar) {
const int index = blockIdx.x * blockDim.z + threadIdx.z;
if (index < num_nodes) {
const int i = (sorted_node_map[index] >= 0) ? sorted_node_map[index] : index;
const float weight = normalize_node_info[i][4];
if (weight > 0) {
const float x_l = normalize_node_info[i][0];
const float x_h = normalize_node_info[i][1];
const float y_l = normalize_node_info[i][2];
const float y_h = normalize_node_info[i][3];
int x_lf = lround(floor(x_l));
int x_hf = lround(floor(x_h));
int y_lf = lround(floor(y_l));
int y_hf = lround(floor(y_h));
x_lf = max(x_lf, 0);
x_hf = min(x_hf, num_bin_x - 1);
y_lf = max(y_lf, 0);
y_hf = min(y_hf, num_bin_y - 1);
for (int j = x_lf + threadIdx.y; j < x_hf + 1; j += blockDim.y) {
float bin_x_l = static_cast<float>(j);
float overlap_x = overlap(x_l, x_h, bin_x_l);
for (int k = y_lf + threadIdx.x; k < y_hf + 1; k += blockDim.x) {
float bin_y_l = static_cast<float>(k);
float overlap_y = overlap(y_l, y_h, bin_y_l);
float overlap_area = overlap_x * overlap_y;
atomicAdd(&aux_mat[j * num_bin_y + k],
static_cast<unsigned long long>(weight * overlap_area * scalar));
}
}
}
}
}
__global__ void __launch_bounds__(256, 4) density_map_cuda_backward_kernel(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> normalize_node_info,
const float *grad_mat,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> sorted_node_map,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_grad,
float grad_weight,
int num_bin_x,
int num_bin_y,
@ -90,12 +129,12 @@ __global__ void __launch_bounds__(256, 4) density_map_cuda_backward_kernel(
const int index = blockIdx.x * blockDim.z + threadIdx.z;
if (index < num_nodes) {
const int i = (sorted_node_map[index] >= 0) ? sorted_node_map[index] : index;
const scalar_t weight = normalize_node_info[i][4];
const float weight = normalize_node_info[i][4];
if (weight > 0) {
const scalar_t x_l = normalize_node_info[i][0];
const scalar_t x_h = normalize_node_info[i][1];
const scalar_t y_l = normalize_node_info[i][2];
const scalar_t y_h = normalize_node_info[i][3];
const float x_l = normalize_node_info[i][0];
const float x_h = normalize_node_info[i][1];
const float y_l = normalize_node_info[i][2];
const float y_h = normalize_node_info[i][3];
int x_lf = lround(floor(x_l));
int x_hf = lround(floor(x_h));
int y_lf = lround(floor(y_l));
@ -106,33 +145,31 @@ __global__ void __launch_bounds__(256, 4) density_map_cuda_backward_kernel(
y_hf = min(y_hf, num_bin_y - 1);
extern __shared__ unsigned char grad_xy[];
scalar_t *grad_x = (scalar_t *)grad_xy;
scalar_t *grad_y = grad_x + blockDim.z;
float *grad_x = (float *)grad_xy;
float *grad_y = grad_x + blockDim.z;
if (threadIdx.x == 0 && threadIdx.y == 0) {
grad_x[threadIdx.z] = grad_y[threadIdx.z] = 0;
}
__syncthreads();
scalar_t part_grad_x = 0;
scalar_t part_grad_y = 0;
float part_grad_x = 0;
float part_grad_y = 0;
for (int j = x_lf + threadIdx.y; j < x_hf + 1; j += blockDim.y) {
scalar_t bin_x_l = static_cast<scalar_t>(j);
scalar_t overlap_x = overlap(x_l, x_h, bin_x_l);
float bin_x_l = static_cast<float>(j);
float overlap_x = overlap(x_l, x_h, bin_x_l);
for (int k = y_lf + threadIdx.x; k < y_hf + 1; k += blockDim.x) {
scalar_t bin_y_l = static_cast<scalar_t>(k);
scalar_t overlap_y = overlap(y_l, y_h, bin_y_l);
scalar_t overlap_area = overlap_x * overlap_y;
scalar_t tmp_x = grad_mat[0 * num_bin_x * num_bin_y + j * num_bin_y + k];
scalar_t tmp_y = grad_mat[1 * num_bin_x * num_bin_y + j * num_bin_y + k];
// part_grad_x += overlap_area * grad_mat[0][j][k];
// part_grad_y += overlap_area * grad_mat[1][j][k];
float bin_y_l = static_cast<float>(k);
float overlap_y = overlap(y_l, y_h, bin_y_l);
float overlap_area = overlap_x * overlap_y;
float tmp_x = grad_mat[0 * num_bin_x * num_bin_y + j * num_bin_y + k];
float tmp_y = grad_mat[1 * num_bin_x * num_bin_y + j * num_bin_y + k];
part_grad_x += overlap_area * tmp_x;
part_grad_y += overlap_area * tmp_y;
}
}
gpuAtomicAdd(&grad_x[threadIdx.z], part_grad_x);
gpuAtomicAdd(&grad_y[threadIdx.z], part_grad_y);
atomicAdd(&grad_x[threadIdx.z], part_grad_x);
atomicAdd(&grad_y[threadIdx.z], part_grad_y);
__syncthreads();
if (threadIdx.x == 0 && threadIdx.y == 0) {
@ -143,6 +180,68 @@ __global__ void __launch_bounds__(256, 4) density_map_cuda_backward_kernel(
}
}
__global__ void density_map_cuda_deterministic_backward_kernel(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> normalize_node_info,
const float *grad_mat,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> sorted_node_map,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_grad,
float grad_weight,
int num_bin_x,
int num_bin_y,
int num_nodes) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
if (index < num_nodes) {
const int i = (sorted_node_map[index] >= 0) ? sorted_node_map[index] : index;
const float weight = normalize_node_info[i][4];
if (weight > 0) {
const float x_l = normalize_node_info[i][0];
const float x_h = normalize_node_info[i][1];
const float y_l = normalize_node_info[i][2];
const float y_h = normalize_node_info[i][3];
int x_lf = lround(floor(x_l));
int x_hf = lround(floor(x_h));
int y_lf = lround(floor(y_l));
int y_hf = lround(floor(y_h));
x_lf = max(x_lf, 0);
x_hf = min(x_hf, num_bin_x - 1);
y_lf = max(y_lf, 0);
y_hf = min(y_hf, num_bin_y - 1);
float gradX = 0;
float gradY = 0;
for (int j = x_lf; j < x_hf + 1; j++) {
float bin_x_l = static_cast<float>(j);
float overlap_x = overlap(x_l, x_h, bin_x_l);
for (int k = y_lf; k < y_hf + 1; k++) {
float bin_y_l = static_cast<float>(k);
float overlap_y = overlap(y_l, y_h, bin_y_l);
float overlap_area = overlap_x * overlap_y;
gradX += grad_mat[0 * num_bin_x * num_bin_y + j * num_bin_y + k] * overlap_area;
gradY += grad_mat[1 * num_bin_x * num_bin_y + j * num_bin_y + k] * overlap_area;
}
}
node_grad[i][0] = grad_weight * weight * gradX;
node_grad[i][1] = grad_weight * weight * gradY;
}
}
}
__global__ void copyFromFloatAuxMat(
unsigned long long *aux_mat_uint64, float *aux_mat, unsigned long long scalar, float inv_scalar, int num_bin) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_bin) {
aux_mat_uint64[i] = static_cast<unsigned long long>(aux_mat[i] * scalar);
}
}
__global__ void copyToFloatAuxMat(
unsigned long long *aux_mat_uint64, float *aux_mat, unsigned long long scalar, float inv_scalar, int num_bin) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_bin) {
aux_mat[i] = static_cast<float>(inv_scalar * aux_mat_uint64[i]);
}
}
torch::Tensor density_map_cuda_normalize_node(torch::Tensor node_pos,
torch::Tensor node_size,
torch::Tensor node_weight,
@ -158,18 +257,16 @@ torch::Tensor density_map_cuda_normalize_node(torch::Tensor node_pos,
const int threads = 128;
const int blocks = (num_nodes + threads - 1) / threads;
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "density_map_cuda_normalize_node", ([&] {
density_map_cuda_normalize_node_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
node_size.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
node_weight.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
expand_ratio.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
unit_len.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
normalize_node_info.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_bin_x,
num_bin_y,
num_nodes);
}));
density_map_cuda_normalize_node_kernel<<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
expand_ratio.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
unit_len.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
normalize_node_info.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
num_bin_x,
num_bin_y,
num_nodes);
return normalize_node_info;
}
@ -179,7 +276,8 @@ torch::Tensor density_map_cuda_forward(torch::Tensor normalize_node_info,
torch::Tensor aux_mat,
int num_bin_x,
int num_bin_y,
int num_nodes) {
int num_nodes,
bool deterministic) {
cudaSetDevice(normalize_node_info.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
@ -187,15 +285,69 @@ torch::Tensor density_map_cuda_forward(torch::Tensor normalize_node_info,
dim3 blockSize(2, 2, thread_count);
int block_count = (num_nodes - 1 + thread_count) / thread_count;
AT_DISPATCH_ALL_TYPES(normalize_node_info.scalar_type(), "density_map_cuda_forward", ([&] {
density_map_cuda_forward_kernel<scalar_t><<<block_count, blockSize, 0, stream>>>(
normalize_node_info.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
aux_mat.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_nodes,
num_bin_x,
num_bin_y);
}));
if (deterministic) {
// each bin size is pre-normalized to 1x1
// max_value_bits -> #bits of the maximum density == (num_bin_x * num_bin_y)
int max_value_bits = max(static_cast<int>(ceil(log2((num_bin_x + 0.1) * (num_bin_y + 0.1)))) + 1, 32);
int scalar_bits = max(64 - max_value_bits, 0);
unsigned long long scalar = (1UL << scalar_bits);
float inv_scalar = 1.0 / static_cast<float>(scalar);
int num_bin = num_bin_x * num_bin_y;
// use cache to save runtime
int cp_threads = 512;
int cp_blocks = (num_bin + cp_threads - 1) / cp_threads;
static unsigned long long *aux_mat_uint64_ptr = nullptr;
static int aux_mat_uint64_size = -1;
if (aux_mat_uint64_ptr == nullptr) {
aux_mat_uint64_size = num_bin;
cudaMalloc(&aux_mat_uint64_ptr, aux_mat_uint64_size * sizeof(unsigned long long));
} else if (num_bin != aux_mat_uint64_size) {
cudaFree(aux_mat_uint64_ptr);
aux_mat_uint64_ptr = nullptr;
aux_mat_uint64_size = num_bin;
cudaMalloc(&aux_mat_uint64_ptr, aux_mat_uint64_size * sizeof(unsigned long long));
}
copyFromFloatAuxMat<<<cp_blocks, cp_threads, 0, stream>>>(
aux_mat_uint64_ptr, aux_mat.data_ptr<float>(), scalar, inv_scalar, num_bin);
density_map_cuda_deterministic_forward_kernel<<<block_count, blockSize, 0, stream>>>(
normalize_node_info.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
aux_mat_uint64_ptr,
num_nodes,
num_bin_x,
num_bin_y,
scalar);
copyToFloatAuxMat<<<cp_blocks, cp_threads, 0, stream>>>(
aux_mat_uint64_ptr, aux_mat.data_ptr<float>(), scalar, inv_scalar, num_bin);
// without cache, need a lot of cudaMallocAsync...
// int cp_threads = 512;
// int cp_blocks = (num_bin + cp_threads - 1) / cp_threads;
// unsigned long long *aux_mat_uint64 = nullptr;
// cudaMallocAsync(&aux_mat_uint64, num_bin * sizeof(unsigned long long), stream);
// copyFromFloatAuxMat<<<cp_blocks, cp_threads, 0, stream>>>(
// aux_mat_uint64, aux_mat.data_ptr<float>(), scalar, inv_scalar, num_bin);
// density_map_cuda_deterministic_forward_kernel<<<block_count, blockSize, 0, stream>>>(
// normalize_node_info.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
// sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
// aux_mat_uint64,
// num_nodes,
// num_bin_x,
// num_bin_y,
// scalar);
// copyToFloatAuxMat<<<cp_blocks, cp_threads, 0, stream>>>(
// aux_mat_uint64, aux_mat.data_ptr<float>(), scalar, inv_scalar, num_bin);
// cudaFreeAsync(aux_mat_uint64, stream);
} else {
density_map_cuda_forward_kernel<<<block_count, blockSize, 0, stream>>>(
normalize_node_info.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
aux_mat.data_ptr<float>(),
num_nodes,
num_bin_x,
num_bin_y);
}
return aux_mat;
}
@ -207,27 +359,38 @@ torch::Tensor density_map_cuda_backward(torch::Tensor normalize_node_info,
float grad_weight,
int num_bin_x,
int num_bin_y,
int num_nodes) {
int num_nodes,
bool deterministic) {
cudaSetDevice(normalize_node_info.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
int thread_count = 64;
dim3 blockSize(2, 2, thread_count);
int block_count = (num_nodes - 1 + thread_count) / thread_count;
AT_DISPATCH_ALL_TYPES(normalize_node_info.scalar_type(), "density_map_cuda_backward", ([&] {
size_t shared_mem_size = sizeof(scalar_t) * thread_count * 2;
density_map_cuda_backward_kernel<scalar_t>
<<<block_count, blockSize, shared_mem_size, stream>>>(
normalize_node_info.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
grad_mat.data_ptr<scalar_t>(),
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
node_grad.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
grad_weight,
num_bin_x,
num_bin_y,
num_nodes);
}));
if (deterministic) {
int threads = 64;
int blocks = (num_nodes + threads - 1) / threads;
density_map_cuda_deterministic_backward_kernel<<<blocks, threads, 0, stream>>>(
normalize_node_info.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
grad_mat.data_ptr<float>(),
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
node_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
grad_weight,
num_bin_x,
num_bin_y,
num_nodes);
} else {
int thread_count = 64;
dim3 blockSize(2, 2, thread_count);
int block_count = (num_nodes - 1 + thread_count) / thread_count;
size_t shared_mem_size = sizeof(float) * thread_count * 2;
density_map_cuda_backward_kernel<<<block_count, blockSize, shared_mem_size, stream>>>(
normalize_node_info.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
grad_mat.data_ptr<float>(),
sorted_node_map.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
node_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
grad_weight,
num_bin_x,
num_bin_y,
num_nodes);
}
return node_grad;
}

View File

@ -1,18 +1,16 @@
#include <ATen/cuda/CUDAContext.h>
#include <cuda.h>
#include <cuda_runtime.h>
#include <torch/extension.h>
#include <ATen/cuda/CUDAContext.h>
#include <THC/THCAtomics.cuh>
#include <vector>
template <typename scalar_t>
__global__ void density_map_cuda_forward_naive_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_size,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> node_weight,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> unit_len,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> aux_mat,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_size,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> node_weight,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> unit_len,
float *aux_mat,
int num_bin_x,
int num_bin_y,
int num_nodes,
@ -22,23 +20,23 @@ __global__ void density_map_cuda_forward_naive_kernel(
bool clamp_node) {
const int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_nodes) {
scalar_t node_w = node_size[i][0];
scalar_t node_h = node_size[i][1];
scalar_t ratio = 1.0;
float node_w = node_size[i][0];
float node_h = node_size[i][1];
float ratio = 1.0;
if (clamp_node) {
const scalar_t node_area = node_w * node_h;
node_w = max(node_w, static_cast<scalar_t>(min_node_w));
node_h = max(node_h, static_cast<scalar_t>(min_node_h));
const float node_area = node_w * node_h;
node_w = max(node_w, static_cast<float>(min_node_w));
node_h = max(node_h, static_cast<float>(min_node_h));
ratio = node_area / (node_w * node_h);
}
const scalar_t mgn = static_cast<scalar_t>(margin);
const scalar_t num_bin_x_minus_mgn = static_cast<scalar_t>(num_bin_x) - mgn;
const scalar_t num_bin_y_minus_mgn = static_cast<scalar_t>(num_bin_y) - mgn;
const scalar_t small_mgn = static_cast<scalar_t>(margin * 0.1);
scalar_t x_l = max((node_pos[i][0] - node_w / 2) / unit_len[0], mgn);
scalar_t x_h = min((node_pos[i][0] + node_w / 2) / unit_len[0], num_bin_x_minus_mgn);
scalar_t y_l = max((node_pos[i][1] - node_h / 2) / unit_len[1], mgn);
scalar_t y_h = min((node_pos[i][1] + node_h / 2) / unit_len[1], num_bin_y_minus_mgn);
const float mgn = static_cast<float>(margin);
const float num_bin_x_minus_mgn = static_cast<float>(num_bin_x) - mgn;
const float num_bin_y_minus_mgn = static_cast<float>(num_bin_y) - mgn;
const float small_mgn = static_cast<float>(margin * 0.1);
float x_l = max((node_pos[i][0] - node_w / 2) / unit_len[0], mgn);
float x_h = min((node_pos[i][0] + node_w / 2) / unit_len[0], num_bin_x_minus_mgn);
float y_l = max((node_pos[i][1] - node_h / 2) / unit_len[1], mgn);
float y_h = min((node_pos[i][1] + node_h / 2) / unit_len[1], num_bin_y_minus_mgn);
x_l = min(x_l, num_bin_x_minus_mgn);
x_h = max(x_h, mgn);
y_l = min(y_l, num_bin_y_minus_mgn);
@ -46,7 +44,7 @@ __global__ void density_map_cuda_forward_naive_kernel(
if (x_h - x_l < small_mgn || y_h - y_l < small_mgn) {
return;
}
const scalar_t p_node_wght = node_weight[i] * ratio;
const float p_node_wght = node_weight[i] * ratio;
const int x_lf = lround(floor(x_l));
const int x_hf = lround(floor(x_h));
@ -54,28 +52,90 @@ __global__ void density_map_cuda_forward_naive_kernel(
const int y_hf = lround(floor(y_h));
for (int j = x_lf; j < x_hf + 1; j++) {
const scalar_t bin_x_l = j;
const scalar_t bin_x_h = j + 1;
scalar_t overlap_x = min(x_h, bin_x_h) - max(x_l, bin_x_l);
const float bin_x_l = j;
const float bin_x_h = j + 1;
float overlap_x = min(x_h, bin_x_h) - max(x_l, bin_x_l);
for (int k = y_lf; k < y_hf + 1; k++) {
const scalar_t bin_y_l = k;
const scalar_t bin_y_h = k + 1;
scalar_t overlap_y = min(y_h, bin_y_h) - max(y_l, bin_y_l);
scalar_t overlap_area = overlap_x * overlap_y;
gpuAtomicAdd(&aux_mat[j][k], p_node_wght * overlap_area);
const float bin_y_l = k;
const float bin_y_h = k + 1;
float overlap_y = min(y_h, bin_y_h) - max(y_l, bin_y_l);
float overlap_area = overlap_x * overlap_y;
atomicAdd(&aux_mat[j * num_bin_y + k], p_node_wght * overlap_area);
}
}
}
}
__global__ void density_map_cuda_deterministic_forward_naive_kernel(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_size,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> node_weight,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> unit_len,
unsigned long long *aux_mat,
int num_bin_x,
int num_bin_y,
int num_nodes,
float min_node_w,
float min_node_h,
float margin,
bool clamp_node,
unsigned long long scalar) {
const int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_nodes) {
float node_w = node_size[i][0];
float node_h = node_size[i][1];
float ratio = 1.0;
if (clamp_node) {
const float node_area = node_w * node_h;
node_w = max(node_w, static_cast<float>(min_node_w));
node_h = max(node_h, static_cast<float>(min_node_h));
ratio = node_area / (node_w * node_h);
}
const float mgn = static_cast<float>(margin);
const float num_bin_x_minus_mgn = static_cast<float>(num_bin_x) - mgn;
const float num_bin_y_minus_mgn = static_cast<float>(num_bin_y) - mgn;
const float small_mgn = static_cast<float>(margin * 0.1);
float x_l = max((node_pos[i][0] - node_w / 2) / unit_len[0], mgn);
float x_h = min((node_pos[i][0] + node_w / 2) / unit_len[0], num_bin_x_minus_mgn);
float y_l = max((node_pos[i][1] - node_h / 2) / unit_len[1], mgn);
float y_h = min((node_pos[i][1] + node_h / 2) / unit_len[1], num_bin_y_minus_mgn);
x_l = min(x_l, num_bin_x_minus_mgn);
x_h = max(x_h, mgn);
y_l = min(y_l, num_bin_y_minus_mgn);
y_h = max(y_h, mgn);
if (x_h - x_l < small_mgn || y_h - y_l < small_mgn) {
return;
}
const float p_node_wght = node_weight[i] * ratio;
const int x_lf = lround(floor(x_l));
const int x_hf = lround(floor(x_h));
const int y_lf = lround(floor(y_l));
const int y_hf = lround(floor(y_h));
for (int j = x_lf; j < x_hf + 1; j++) {
const float bin_x_l = j;
const float bin_x_h = j + 1;
float overlap_x = min(x_h, bin_x_h) - max(x_l, bin_x_l);
for (int k = y_lf; k < y_hf + 1; k++) {
const float bin_y_l = k;
const float bin_y_h = k + 1;
float overlap_y = min(y_h, bin_y_h) - max(y_l, bin_y_l);
float overlap_area = overlap_x * overlap_y;
atomicAdd(&aux_mat[j * num_bin_y + k],
static_cast<unsigned long long>(p_node_wght * overlap_area * scalar));
}
}
}
}
template <typename scalar_t>
__global__ void density_map_cuda_backward_naive_kernel(
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_size,
const torch::PackedTensorAccessor32<scalar_t, 3, torch::RestrictPtrTraits> grad_mat,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> unit_len,
const torch::PackedTensorAccessor32<scalar_t, 1, torch::RestrictPtrTraits> node_weight,
torch::PackedTensorAccessor32<scalar_t, 2, torch::RestrictPtrTraits> node_grad,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_size,
const torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> grad_mat,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> unit_len,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> node_weight,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_grad,
float grad_weight,
int num_bin_x,
int num_bin_y,
@ -86,23 +146,23 @@ __global__ void density_map_cuda_backward_naive_kernel(
bool clamp_node) {
const int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_nodes) {
scalar_t node_w = node_size[i][0];
scalar_t node_h = node_size[i][1];
scalar_t ratio = 1.0;
float node_w = node_size[i][0];
float node_h = node_size[i][1];
float ratio = 1.0;
if (clamp_node) {
const scalar_t node_area = node_w * node_h;
node_w = max(node_w, static_cast<scalar_t>(min_node_w));
node_h = max(node_h, static_cast<scalar_t>(min_node_h));
const float node_area = node_w * node_h;
node_w = max(node_w, static_cast<float>(min_node_w));
node_h = max(node_h, static_cast<float>(min_node_h));
ratio = node_area / (node_w * node_h);
}
const scalar_t mgn = static_cast<scalar_t>(margin);
const scalar_t num_bin_x_minus_mgn = static_cast<scalar_t>(num_bin_x) - mgn;
const scalar_t num_bin_y_minus_mgn = static_cast<scalar_t>(num_bin_y) - mgn;
const scalar_t small_mgn = static_cast<scalar_t>(margin * 0.1);
scalar_t x_l = max((node_pos[i][0] - node_w / 2) / unit_len[0], mgn);
scalar_t x_h = min((node_pos[i][0] + node_w / 2) / unit_len[0], num_bin_x_minus_mgn);
scalar_t y_l = max((node_pos[i][1] - node_h / 2) / unit_len[1], mgn);
scalar_t y_h = min((node_pos[i][1] + node_h / 2) / unit_len[1], num_bin_y_minus_mgn);
const float mgn = static_cast<float>(margin);
const float num_bin_x_minus_mgn = static_cast<float>(num_bin_x) - mgn;
const float num_bin_y_minus_mgn = static_cast<float>(num_bin_y) - mgn;
const float small_mgn = static_cast<float>(margin * 0.1);
float x_l = max((node_pos[i][0] - node_w / 2) / unit_len[0], mgn);
float x_h = min((node_pos[i][0] + node_w / 2) / unit_len[0], num_bin_x_minus_mgn);
float y_l = max((node_pos[i][1] - node_h / 2) / unit_len[1], mgn);
float y_h = min((node_pos[i][1] + node_h / 2) / unit_len[1], num_bin_y_minus_mgn);
x_l = min(x_l, num_bin_x_minus_mgn);
x_h = max(x_h, mgn);
y_l = min(y_l, num_bin_y_minus_mgn);
@ -116,17 +176,17 @@ __global__ void density_map_cuda_backward_naive_kernel(
const int y_lf = lround(floor(y_l));
const int y_hf = lround(floor(y_h));
scalar_t gradX = 0;
scalar_t gradY = 0;
float gradX = 0;
float gradY = 0;
for (int j = x_lf; j < x_hf + 1; j++) {
const scalar_t bin_x_l = j;
const scalar_t bin_x_h = j + 1;
scalar_t overlap_x = min(x_h, bin_x_h) - max(x_l, bin_x_l);
const float bin_x_l = j;
const float bin_x_h = j + 1;
float overlap_x = min(x_h, bin_x_h) - max(x_l, bin_x_l);
for (int k = y_lf; k < y_hf + 1; k++) {
const scalar_t bin_y_l = k;
const scalar_t bin_y_h = k + 1;
scalar_t overlap_y = min(y_h, bin_y_h) - max(y_l, bin_y_l);
scalar_t overlap_area = overlap_x * overlap_y;
const float bin_y_l = k;
const float bin_y_h = k + 1;
float overlap_y = min(y_h, bin_y_h) - max(y_l, bin_y_l);
float overlap_area = overlap_x * overlap_y;
gradX += grad_mat[0][j][k] * overlap_area;
gradY += grad_mat[1][j][k] * overlap_area;
}
@ -136,6 +196,22 @@ __global__ void density_map_cuda_backward_naive_kernel(
}
}
__global__ void copyFromFloatAuxMat2(
unsigned long long *aux_mat_uint64, float *aux_mat, unsigned long long scalar, float inv_scalar, int num_bin) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_bin) {
aux_mat_uint64[i] = static_cast<unsigned long long>(aux_mat[i] * scalar);
}
}
__global__ void copyToFloatAuxMat2(
unsigned long long *aux_mat_uint64, float *aux_mat, unsigned long long scalar, float inv_scalar, int num_bin) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_bin) {
aux_mat[i] = static_cast<float>(inv_scalar * aux_mat_uint64[i]);
}
}
torch::Tensor density_map_cuda_forward_naive(torch::Tensor node_pos,
torch::Tensor node_size,
torch::Tensor node_weight,
@ -147,28 +223,61 @@ torch::Tensor density_map_cuda_forward_naive(torch::Tensor node_pos,
float min_node_w,
float min_node_h,
float margin,
bool clamp_node) {
bool clamp_node,
bool deterministic) {
cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
const int threads = 64;
const int blocks = (num_nodes + threads - 1) / threads;
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "density_map_cuda_forward_naive", ([&] {
density_map_cuda_forward_naive_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
node_size.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
node_weight.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
unit_len.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
aux_mat.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
num_bin_x,
num_bin_y,
num_nodes,
min_node_w,
min_node_h,
margin,
clamp_node);
}));
if (deterministic) {
// each bin size is internally normalized to 1x1
// max_value_bits -> #bits of the maximum density == (num_bin_x * num_bin_y)
int max_value_bits = max(static_cast<int>(ceil(log2((num_bin_x + 0.1) * (num_bin_y + 0.1)))) + 1, 32);
int scalar_bits = max(64 - max_value_bits, 0);
unsigned long long scalar = (1UL << scalar_bits);
float inv_scalar = 1.0 / static_cast<float>(scalar);
int num_bin = num_bin_x * num_bin_y;
int cp_threads = 512;
int cp_blocks = (num_bin + cp_threads - 1) / cp_threads;
unsigned long long *aux_mat_uint64 = nullptr;
cudaMallocAsync(&aux_mat_uint64, num_bin * sizeof(unsigned long long), stream);
copyFromFloatAuxMat2<<<cp_blocks, cp_threads, 0, stream>>>(
aux_mat_uint64, aux_mat.data_ptr<float>(), scalar, inv_scalar, num_bin);
density_map_cuda_deterministic_forward_naive_kernel<<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
unit_len.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
aux_mat_uint64,
num_bin_x,
num_bin_y,
num_nodes,
min_node_w,
min_node_h,
margin,
clamp_node,
scalar);
copyToFloatAuxMat2<<<cp_blocks, cp_threads, 0, stream>>>(
aux_mat_uint64, aux_mat.data_ptr<float>(), scalar, inv_scalar, num_bin);
cudaFreeAsync(aux_mat_uint64, stream);
} else {
density_map_cuda_forward_naive_kernel<<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
unit_len.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
aux_mat.data_ptr<float>(),
num_bin_x,
num_bin_y,
num_nodes,
min_node_w,
min_node_h,
margin,
clamp_node);
}
return aux_mat;
}
@ -186,30 +295,29 @@ torch::Tensor density_map_cuda_backward(torch::Tensor node_pos,
float min_node_w,
float min_node_h,
float margin,
bool clamp_node) {
bool clamp_node,
bool deterministic) {
cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
const int threads = 64;
const int blocks = (num_nodes + threads - 1) / threads;
AT_DISPATCH_ALL_TYPES(node_pos.scalar_type(), "density_map_cuda_backward_naive", ([&] {
density_map_cuda_backward_naive_kernel<scalar_t><<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
node_size.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
grad_mat.packed_accessor32<scalar_t, 3, torch::RestrictPtrTraits>(),
unit_len.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
node_weight.packed_accessor32<scalar_t, 1, torch::RestrictPtrTraits>(),
node_grad.packed_accessor32<scalar_t, 2, torch::RestrictPtrTraits>(),
grad_weight,
num_bin_x,
num_bin_y,
num_nodes,
min_node_w,
min_node_h,
margin,
clamp_node);
}));
density_map_cuda_backward_naive_kernel<<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
grad_mat.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
unit_len.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
node_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
node_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
grad_weight,
num_bin_x,
num_bin_y,
num_nodes,
min_node_w,
min_node_h,
margin,
clamp_node);
return node_grad;
}

View File

@ -0,0 +1,37 @@
# Files
file(GLOB_RECURSE SRC_FILES_DP ${CMAKE_CURRENT_SOURCE_DIR}/check/*.cpp
${CMAKE_CURRENT_SOURCE_DIR}/dp/*.cpp
${CMAKE_CURRENT_SOURCE_DIR}/db/*.cpp
${CMAKE_CURRENT_SOURCE_DIR}/lg/*.cpp)
file(GLOB_RECURSE SRC_FILES_DP_CUDA ${CMAKE_CURRENT_SOURCE_DIR}/*.cu)
# OpenMP
find_package(OpenMP REQUIRED)
# CUDA DP Kernel
cuda_add_library(dp_cuda_tmp STATIC ${SRC_FILES_DP_CUDA})
set_target_properties(dp_cuda_tmp PROPERTIES CUDA_RESOLVE_DEVICE_SYMBOLS ON)
set_target_properties(dp_cuda_tmp PROPERTIES CUDA_SEPARABLE_COMPILATION ON)
set_target_properties(dp_cuda_tmp PROPERTIES POSITION_INDEPENDENT_CODE ON)
target_include_directories(dp_cuda_tmp PRIVATE ${PROJECT_SOURCE_DIR}/cpp_to_py ${TORCH_INCLUDE_DIRS} ${FLUTE_INCLUDE_DIR})
target_link_libraries(dp_cuda_tmp torch ${TORCH_PYTHON_LIBRARY} xplace_common flute OpenMP::OpenMP_CXX)
# CPU DP object
add_library(dp SHARED ${CMAKE_CURRENT_SOURCE_DIR}/../io_parser/gp/GPDatabase.cpp
${SRC_FILES_DP}
${SRC_FILES_DP_CUDA})
target_include_directories(dp PRIVATE ${PROJECT_SOURCE_DIR}/cpp_to_py ${TORCH_INCLUDE_DIRS} ${FLUTE_INCLUDE_DIR})
target_link_libraries(dp PRIVATE torch ${TORCH_PYTHON_LIBRARY} xplace_common flute dp_cuda_tmp pthread)
target_compile_options(dp PRIVATE -fPIC)
install(TARGETS dp DESTINATION ${XPLACE_LIB_DIR})
# For Pybind
add_pytorch_extension(gpudp PyBindCppMain.cpp
EXTRA_INCLUDE_DIRS ${PROJECT_SOURCE_DIR}/cpp_to_py ${FLUTE_INCLUDE_DIR}
EXTRA_LINK_LIBRARIES xplace_common flute io_parser dp)
install(TARGETS gpudp DESTINATION ${XPLACE_LIB_DIR})

View File

@ -0,0 +1,119 @@
#include "common/common.h"
#include "common/db/Database.h"
#include "gpudp/db/dp_torch.h"
namespace Xplace {
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
pybind11::class_<dp::DPTorchRawDB, std::shared_ptr<dp::DPTorchRawDB>>(m, "DPTorchRawDB")
.def(pybind11::init<torch::Tensor,
torch::Tensor,
torch::Tensor,
torch::Tensor,
torch::Tensor,
torch::Tensor,
torch::Tensor,
torch::Tensor,
torch::Tensor,
torch::Tensor,
torch::Tensor,
torch::Tensor,
torch::Tensor,
torch::Tensor,
float,
float,
float,
float,
int,
int,
float,
float>())
.def("check", &dp::DPTorchRawDB::check)
.def("scale", &dp::DPTorchRawDB::scale)
.def("commit", &dp::DPTorchRawDB::commit)
.def("rollback", &dp::DPTorchRawDB::rollback)
.def("commit_from", &dp::DPTorchRawDB::commit_from)
.def("get_curr_cposx", &dp::DPTorchRawDB::get_curr_cposx, py::return_value_policy::move)
.def("get_curr_cposy", &dp::DPTorchRawDB::get_curr_cposy, py::return_value_policy::move)
.def("get_curr_lposx", &dp::DPTorchRawDB::get_curr_lposx, py::return_value_policy::move)
.def("get_curr_lposy", &dp::DPTorchRawDB::get_curr_lposy, py::return_value_policy::move);
m.def("create_dp_rawdb",
[](torch::Tensor node_lpos_init_,
torch::Tensor node_size_,
torch::Tensor node_weight_,
torch::Tensor pin_rel_lpos_,
torch::Tensor pin_id2node_id_,
torch::Tensor pin_id2net_id_,
torch::Tensor node2pin_list_,
torch::Tensor node2pin_list_end_,
torch::Tensor hyperedge_list_,
torch::Tensor hyperedge_list_end_,
torch::Tensor net_mask_,
torch::Tensor node_id2region_id_,
torch::Tensor region_boxes_,
torch::Tensor region_boxes_end_,
float xl_,
float xh_,
float yl_,
float yh_,
int num_movable_nodes_,
int num_nodes_,
float site_width_,
float row_height_) {
return std::make_shared<dp::DPTorchRawDB>(node_lpos_init_,
node_size_,
node_weight_,
pin_rel_lpos_,
pin_id2node_id_,
pin_id2net_id_,
node2pin_list_,
node2pin_list_end_,
hyperedge_list_,
hyperedge_list_end_,
net_mask_,
node_id2region_id_,
region_boxes_,
region_boxes_end_,
xl_,
xh_,
yl_,
yh_,
num_movable_nodes_,
num_nodes_,
site_width_,
row_height_);
});
m.def("macroLegalization", [](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr, int num_bins_x, int num_bins_y) {
return dp::macroLegalization(*at_db_ptr, num_bins_x, num_bins_y);
});
m.def("abacusLegalization", [](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr, int num_bins_x, int num_bins_y) {
return dp::abacusLegalization(*at_db_ptr, num_bins_x, num_bins_y);
});
m.def("greedyLegalization", [](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr, int num_bins_x, int num_bins_y) {
return dp::greedyLegalization(*at_db_ptr, num_bins_x, num_bins_y);
});
m.def("kReorder",
[](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr, int num_bins_x, int num_bins_y, int K, int max_iters) {
return dp::kReorder(*at_db_ptr, num_bins_x, num_bins_y, K, max_iters);
});
m.def(
"globalSwap",
[](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr, int num_bins_x, int num_bins_y, int batch_size, int max_iters) {
return dp::globalSwap(*at_db_ptr, num_bins_x, num_bins_y, batch_size, max_iters);
});
m.def("independentSetMatching",
[](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr,
int num_bins_x,
int num_bins_y,
int batch_size,
int set_size,
int max_iters) {
return dp::independentSetMatching(*at_db_ptr, num_bins_x, num_bins_y, batch_size, set_size, max_iters);
});
m.def("legalityCheck", [](std::shared_ptr<dp::DPTorchRawDB> at_db_ptr, float scale_factor) {
return dp::legalityCheck(*at_db_ptr, scale_factor);
});
}
} // namespace Xplace

View File

@ -0,0 +1 @@
This GPU-accelerated detailed placer is adapted from [ABCDPlace](https://ieeexplore.ieee.org/document/8982049).

View File

@ -0,0 +1,383 @@
#include "common/common.h"
#include "gpudp/db/dp_torch.h"
#include "gpudp/lg/legalization_db.h"
namespace dp {
bool boundaryCheck(const float* x,
const float* y,
const float* node_size_x,
const float* node_size_y,
const float scale_factor,
float xl,
float yl,
float xh,
float yh,
int num_movable_nodes) {
// use scale factor to control the precision
float precision = (scale_factor == 1.0) ? 1e-6 : scale_factor * 0.1;
bool legal_flag = true;
// check node within boundary
for (int i = 0; i < num_movable_nodes; ++i) {
float node_xl = x[i];
float node_yl = y[i];
float node_xh = node_xl + node_size_x[i];
float node_yh = node_yl + node_size_y[i];
if (node_xl + precision < xl || node_xh > xh + precision || node_yl + precision < yl ||
node_yh > yh + precision) {
logger.error("node %d (%g, %g, %g, %g) out of boundary\n", i, node_xl, node_yl, node_xh, node_yh);
legal_flag = false;
}
}
return legal_flag;
}
bool siteAlignmentCheck(const float* x,
const float* y,
const float site_width,
const float row_height,
const float scale_factor,
const float xl,
const float yl,
int num_movable_nodes) {
// use scale factor to control the precision
float precision = (scale_factor == 1.0) ? 1e-6 : scale_factor * 0.1;
bool legal_flag = true;
// check row and site alignment
for (int i = 0; i < num_movable_nodes; ++i) {
float node_xl = x[i];
float node_yl = y[i];
float row_id_f = (node_yl - yl) / row_height;
int row_id = floorDiv(node_yl - yl, row_height);
float row_yl = yl + row_height * row_id;
float row_yh = row_yl + row_height;
if (std::abs(row_id_f - row_id) > precision) {
logger.error("node %d (%g, %g) failed to align to row %d (%g, %g), gap %g, yl %g, row_height %g",
i,
node_xl,
node_yl,
row_id,
row_yl,
row_yh,
std::abs(node_yl - row_yl),
yl,
row_height);
legal_flag = false;
}
float site_id_f = (node_xl - xl) / site_width;
int site_id = floorDiv(node_xl - xl, site_width);
if (std::abs(site_id_f - site_id) > precision) {
logger.error("node %d (%g, %g) failed to align to row %d (%g, %g) and site; xl %g, site_width %g",
i,
node_xl,
node_yl,
row_id,
row_yl,
row_yh,
xl,
site_width);
legal_flag = false;
}
}
return legal_flag;
}
bool fenceRegionCheck(const float* x,
const float* y,
const float* node_size_x,
const float* node_size_y,
const float* flat_region_boxes,
const int* flat_region_boxes_start,
const int* node2fence_region_map,
int num_movable_nodes,
int num_regions) {
bool legal_flag = true;
// check fence regions
for (int i = 0; i < num_movable_nodes; ++i) {
float node_xl = x[i];
float node_yl = y[i];
float node_xh = node_xl + node_size_x[i];
float node_yh = node_yl + node_size_y[i];
int region_id = node2fence_region_map[i];
if (region_id < num_regions) {
int box_bgn = flat_region_boxes_start[region_id];
int box_end = flat_region_boxes_start[region_id + 1];
float node_area = (node_xh - node_xl) * (node_yh - node_yl);
// I assume there is no overlap between boxes of a region
// otherwise, preprocessing is required
for (int box_id = box_bgn; box_id < box_end; ++box_id) {
int box_offset = box_id * 4;
float box_xl = flat_region_boxes[box_offset];
float box_xh = flat_region_boxes[box_offset + 1];
float box_yl = flat_region_boxes[box_offset + 2];
float box_yh = flat_region_boxes[box_offset + 3];
float dx = std::max(std::min(node_xh, box_xh) - std::max(node_xl, box_xl), (float)0);
float dy = std::max(std::min(node_yh, box_yh) - std::max(node_yl, box_yl), (float)0);
float overlap = dx * dy;
if (overlap > 0) {
node_area -= overlap;
}
}
if (node_area > 0) { // not consumed by boxes within a region
std::string fence_str = "";
for (int box_id = box_bgn; box_id < box_end; ++box_id) {
int box_offset = box_id * 4;
float box_xl = flat_region_boxes[box_offset];
float box_xh = flat_region_boxes[box_offset + 1];
float box_yl = flat_region_boxes[box_offset + 2];
float box_yh = flat_region_boxes[box_offset + 3];
fence_str += (" (" + std::to_string(box_xl) + ", " + std::to_string(box_yl) + ", " +
std::to_string(box_xh) + ", " + std::to_string(box_yh) + ")");
}
logger.error("node %d (%g, %g, %g, %g), out of fence region %d: %s",
i,
node_xl,
node_yl,
node_xh,
node_yh,
region_id,
fence_str.c_str());
legal_flag = false;
}
}
}
return legal_flag;
}
bool overlapCheck(const float* x,
const float* y,
const float* node_size_x,
const float* node_size_y,
float site_width,
float row_height,
float scale_factor,
float xl,
float yl,
float xh,
float yh,
int num_nodes,
int num_movable_nodes) {
bool legal_flag = true;
int num_rows = ceilDiv(yh - yl, row_height);
assert(num_rows > 0);
std::vector<std::vector<int> > row_nodes(num_rows);
// general to node and fixed boxes
auto getXL = [&](int id) { return x[id]; };
auto getYL = [&](int id) { return y[id]; };
auto getXH = [&](int id) { return x[id] + node_size_x[id]; };
auto getYH = [&](int id) { return y[id] + node_size_y[id]; };
auto getSiteXL = [&](float xx) { return int(floorDiv(xx - xl, site_width)); };
auto getSiteYL = [&](float yy) { return int(floorDiv(yy - yl, row_height)); };
auto getSiteXH = [&](float xx) { return int(ceilDiv(xx - xl, site_width)); };
auto getSiteYH = [&](float yy) { return int(ceilDiv(yy - yl, row_height)); };
// add a box to row
auto addBox2Row = [&](int id, float bxl, float byl, float bxh, float byh) {
int row_idxl = floorDiv(byl - yl, row_height);
int row_idxh = ceilDiv(byh - yl, row_height);
row_idxl = std::max(row_idxl, 0);
row_idxh = std::min(row_idxh, num_rows);
for (int row_id = row_idxl; row_id < row_idxh; ++row_id) {
float row_yl = yl + row_id * row_height;
float row_yh = row_yl + row_height;
if (byl < row_yh && byh > row_yl) // overlap with row
{
row_nodes[row_id].push_back(id);
}
}
};
// distribute movable cells to rows
for (int i = 0; i < num_nodes; ++i) {
float node_xl = x[i];
float node_yl = y[i];
float node_xh = node_xl + node_size_x[i];
float node_yh = node_yl + node_size_y[i];
addBox2Row(i, node_xl, node_yl, node_xh, node_yh);
}
// sort cells within rows
for (int i = 0; i < num_rows; ++i) {
auto& nodes_in_row = row_nodes.at(i);
// using left edge
std::sort(nodes_in_row.begin(), nodes_in_row.end(), [&](int node_id1, int node_id2) {
float x1 = getXL(node_id1);
float x2 = getXL(node_id2);
return x1 < x2 || (x1 == x2 && (node_id1 < node_id2));
});
// After sorting by left edge,
// there is a special case for fixed cells where
// one fixed cell is completely within another in a row.
// This will cause failure to detect some overlaps.
// We need to remove the "small" fixed cell that is inside another.
if (!nodes_in_row.empty()) {
std::vector<int> tmp_nodes;
tmp_nodes.reserve(nodes_in_row.size());
tmp_nodes.push_back(nodes_in_row.front());
for (int j = 1, je = nodes_in_row.size(); j < je; ++j) {
int node_id1 = nodes_in_row.at(j - 1);
int node_id2 = nodes_in_row.at(j);
// two fixed cells
if (node_id1 >= num_movable_nodes && node_id2 >= num_movable_nodes) {
float xh1 = getXH(node_id1);
float xh2 = getXH(node_id2);
if (xh1 < xh2) {
tmp_nodes.push_back(node_id2);
}
} else {
tmp_nodes.push_back(node_id2);
}
}
nodes_in_row.swap(tmp_nodes);
}
}
// check overlap
// use scale factor to control the precision
// auto scaleBack2Integer = [&](float value) {
// return (scale_factor == 1.0) ? value : std::round(value / scale_factor);
// };
for (int i = 0; i < num_rows; ++i) {
for (unsigned int j = 0; j < row_nodes.at(i).size(); ++j) {
if (j > 0) {
int node_id = row_nodes[i][j];
int prev_node_id = row_nodes[i][j - 1];
if (node_id < num_movable_nodes || prev_node_id < num_movable_nodes) // ignore two fixed nodes
{
float prev_xl = getXL(prev_node_id);
float prev_yl = getYL(prev_node_id);
float prev_xh = getXH(prev_node_id);
float prev_yh = getYH(prev_node_id);
float cur_xl = getXL(node_id);
float cur_yl = getYL(node_id);
float cur_xh = getXH(node_id);
float cur_yh = getYH(node_id);
int prev_site_xl = getSiteXL(prev_xl);
int prev_site_xh = getSiteXH(prev_xh);
int cur_site_xl = getSiteXL(cur_xl);
int cur_site_xh = getSiteXH(cur_xh);
// detect overlap
if (prev_site_xh > cur_site_xl) {
logger.error(
"row %d (%g, %g), overlap node %d (%g, %g, %g, %g) with "
"node %d (%g, %g, %g, %g) site (%d, %d), gap %g",
i,
yl + i * row_height,
yl + (i + 1) * row_height,
prev_node_id,
prev_xl,
prev_yl,
prev_xh,
prev_yh,
node_id,
cur_xl,
cur_yl,
cur_xh,
cur_yh,
cur_site_xl,
cur_site_xh,
prev_xh - cur_xl);
legal_flag = false;
}
}
}
}
}
return legal_flag;
}
bool legalityCheckKernelCPU(const float* x,
const float* y,
const float* node_size_x,
const float* node_size_y,
const float* flat_region_boxes,
const int* flat_region_boxes_start,
const int* node2fence_region_map,
float xl,
float yl,
float xh,
float yh,
float site_width,
float row_height,
int num_nodes,
int num_movable_nodes,
int num_regions,
float scale_factor) {
bool legal_flag = true;
int num_rows = ceil((yh - yl) / row_height);
assert(num_rows > 0);
fflush(stdout);
std::vector<std::vector<int> > row_nodes(num_rows);
// check node within boundary
if (!boundaryCheck(x, y, node_size_x, node_size_y, scale_factor, xl, yl, xh, yh, num_movable_nodes)) {
legal_flag = false;
std::cerr << "boundary check error!" << std::endl;
}
// check row and site alignment
if (!siteAlignmentCheck(x, y, site_width, row_height, scale_factor, xl, yl, num_movable_nodes)) {
legal_flag = false;
std::cerr << "site alignment check error!" << std::endl;
}
if (!overlapCheck(
x, y, node_size_x, node_size_y, site_width, row_height, scale_factor, xl, yl, xh, yh, num_nodes, num_movable_nodes)) {
legal_flag = false;
std::cerr << "overlap check error!" << std::endl;
}
// check fence regions
if (!fenceRegionCheck(x,
y,
node_size_x,
node_size_y,
flat_region_boxes,
flat_region_boxes_start,
node2fence_region_map,
num_movable_nodes,
num_regions)) {
legal_flag = false;
std::cerr << "fence region check error!" << std::endl;
}
if (!legal_flag) {
logger.error("placement legality check error!");
}
return legal_flag;
}
bool legalityCheck(DPTorchRawDB& at_db, float scale_factor) {
return legalityCheckKernelCPU(at_db.x.cpu().data_ptr<float>(),
at_db.y.cpu().data_ptr<float>(),
at_db.node_size_x.cpu().data_ptr<float>(),
at_db.node_size_y.cpu().data_ptr<float>(),
at_db.flat_region_boxes.cpu().data_ptr<float>(),
at_db.flat_region_boxes_start.cpu().data_ptr<int>(),
at_db.node2fence_region_map.cpu().data_ptr<int>(),
at_db.xl,
at_db.yl,
at_db.xh,
at_db.yh,
at_db.site_width,
at_db.row_height,
at_db.num_nodes,
at_db.num_movable_nodes,
at_db.num_regions,
scale_factor);
}
} // namespace dp

View File

@ -0,0 +1,164 @@
#include "gpudp/db/dp_torch.h"
#include "common/common.h"
#include "common/db/Database.h"
namespace dp {
DPTorchRawDB::DPTorchRawDB(torch::Tensor node_lpos_init_,
torch::Tensor node_size_,
torch::Tensor node_weight_,
torch::Tensor pin_rel_lpos_,
torch::Tensor pin_id2node_id_,
torch::Tensor pin_id2net_id_,
torch::Tensor node2pin_list_,
torch::Tensor node2pin_list_end_,
torch::Tensor hyperedge_list_,
torch::Tensor hyperedge_list_end_,
torch::Tensor net_mask_,
torch::Tensor node_id2region_id_,
torch::Tensor region_boxes_,
torch::Tensor region_boxes_end_,
float xl_,
float xh_,
float yl_,
float yh_,
int num_movable_nodes_,
int num_nodes_,
float site_width_,
float row_height_) {
node_lpos_init = node_lpos_init_;
node_size = node_size_;
pin_rel_lpos = pin_rel_lpos_;
node_size_x = node_size.index({"...", 0}).clone().contiguous();
node_size_y = node_size.index({"...", 1}).clone().contiguous();
init_x = node_lpos_init.index({"...", 0}).clone().contiguous();
init_y = node_lpos_init.index({"...", 1}).clone().contiguous();
pin_offset_x = pin_rel_lpos.index({"...", 0}).clone().contiguous();
pin_offset_y = pin_rel_lpos.index({"...", 1}).clone().contiguous();
x = init_x.clone().contiguous();
y = init_y.clone().contiguous();
num_nodes = num_nodes_;
num_pins = pin_id2node_id_.size(0);
num_nets = hyperedge_list_end_.size(0);
num_regions = region_boxes_end_.size(0);
num_movable_nodes = num_movable_nodes_;
flat_node2pin_start_map =
torch::cat({torch::zeros({1}, torch::dtype(torch::kInt32).device(torch::Device(node_size.device()))),
node2pin_list_end_},
0)
.contiguous();
flat_node2pin_map = node2pin_list_;
pin2node_map = pin_id2node_id_;
flat_net2pin_start_map =
torch::cat({torch::zeros({1}, torch::dtype(torch::kInt32).device(torch::Device(node_size.device()))),
hyperedge_list_end_},
0)
.contiguous();
flat_net2pin_map = hyperedge_list_;
pin2net_map = pin_id2net_id_;
flat_region_boxes_start =
torch::cat({torch::zeros({1}, torch::dtype(torch::kInt32).device(torch::Device(node_size.device()))),
region_boxes_end_},
0)
.contiguous();
flat_region_boxes = region_boxes_.flatten().contiguous().clone();
node2fence_region_map = node_id2region_id_;
net_mask = net_mask_;
node_weight = node_weight_;
site_width = site_width_;
row_height = row_height_;
xl = xl_;
xh = xh_;
yl = yl_;
yh = yh_;
num_sites_x = std::round((xh - xl) / site_width);
num_sites_y = std::round((yh - yl) / row_height);
num_threads = std::max(db::setting.numThreads, 1);
}
bool DPTorchRawDB::check(float scale_factor) {
// NOTE: if tensors are on GPU, legalityCheck would copy large data from GPU to CPU
return legalityCheck(*this, scale_factor);
}
void DPTorchRawDB::scale(float scale_factor, bool use_round) {
pin_rel_lpos.mul_(scale_factor);
if (use_round) {
node_size.mul_(scale_factor).round_();
node_lpos_init.mul_(scale_factor).round_();
x.mul_(scale_factor).round_();
y.mul_(scale_factor).round_();
flat_region_boxes.mul_(scale_factor).round_();
site_width = round(site_width * scale_factor);
row_height = round(row_height * scale_factor);
xl = round(xl * scale_factor);
xh = round(xh * scale_factor);
yl = round(yl * scale_factor);
yh = round(yh * scale_factor);
} else {
node_size.mul_(scale_factor);
node_lpos_init.mul_(scale_factor);
x.mul_(scale_factor);
y.mul_(scale_factor);
flat_region_boxes.mul_(scale_factor);
site_width = site_width * scale_factor;
row_height = row_height * scale_factor;
xl = xl * scale_factor;
xh = xh * scale_factor;
yl = yl * scale_factor;
yh = yh * scale_factor;
}
}
void DPTorchRawDB::commit() {
// commit cached pos to original pos
init_x.index({torch::indexing::Slice(0, num_movable_nodes)})
.data()
.copy_(x.index({torch::indexing::Slice(0, num_movable_nodes)}));
init_y.index({torch::indexing::Slice(0, num_movable_nodes)})
.data()
.copy_(y.index({torch::indexing::Slice(0, num_movable_nodes)}));
}
void DPTorchRawDB::rollback() {
// rollback cached pos to original pos
x.index({torch::indexing::Slice(0, num_movable_nodes)})
.data()
.copy_(init_x.index({torch::indexing::Slice(0, num_movable_nodes)}));
y.index({torch::indexing::Slice(0, num_movable_nodes)})
.data()
.copy_(init_y.index({torch::indexing::Slice(0, num_movable_nodes)}));
}
void DPTorchRawDB::commit_from(torch::Tensor x_, torch::Tensor y_) {
// commit external pos to original pos
init_x.index({torch::indexing::Slice(0, num_movable_nodes)})
.data()
.copy_(x_.index({torch::indexing::Slice(0, num_movable_nodes)}));
init_y.index({torch::indexing::Slice(0, num_movable_nodes)})
.data()
.copy_(y_.index({torch::indexing::Slice(0, num_movable_nodes)}));
x.index({torch::indexing::Slice(0, num_movable_nodes)})
.data()
.copy_(x_.index({torch::indexing::Slice(0, num_movable_nodes)}));
y.index({torch::indexing::Slice(0, num_movable_nodes)})
.data()
.copy_(y_.index({torch::indexing::Slice(0, num_movable_nodes)}));
}
torch::Tensor DPTorchRawDB::get_curr_cposx() { return x + node_size_x / 2; }
torch::Tensor DPTorchRawDB::get_curr_cposy() { return y + node_size_y / 2; }
torch::Tensor DPTorchRawDB::get_curr_lposx() { return x; }
torch::Tensor DPTorchRawDB::get_curr_lposy() { return y; }
} // namespace dp

View File

@ -0,0 +1,114 @@
#pragma once
#include "common/common.h"
#include "common/db/Database.h"
namespace dp {
class DPTorchRawDB {
public:
DPTorchRawDB(torch::Tensor node_lpos_init_,
torch::Tensor node_size_,
torch::Tensor node_weight_,
torch::Tensor pin_rel_lpos_,
torch::Tensor pin_id2node_id_,
torch::Tensor pin_id2net_id_,
torch::Tensor node2pin_list_,
torch::Tensor node2pin_list_end_,
torch::Tensor hyperedge_list_,
torch::Tensor hyperedge_list_end_,
torch::Tensor net_mask_,
torch::Tensor node_id2region_id_,
torch::Tensor region_boxes_,
torch::Tensor region_boxes_end_,
float xl_,
float xh_,
float yl_,
float yh_,
int num_movable_nodes_,
int num_nodes_,
float site_width_,
float row_height_);
bool check(float scale_factor);
void scale(float scale_factor, bool use_round);
void commit();
void rollback();
void commit_from(torch::Tensor x_, torch::Tensor y_);
torch::Tensor get_curr_cposx();
torch::Tensor get_curr_cposy();
torch::Tensor get_curr_lposx();
torch::Tensor get_curr_lposy();
public:
/* node info */
// for backup
torch::Tensor node_lpos_init;
torch::Tensor node_size;
torch::Tensor pin_rel_lpos;
torch::Tensor node_weight;
torch::Tensor init_x; // original pos (keep it const except committing)
torch::Tensor init_y; // original pos (keep it const except committing)
torch::Tensor x; // mutable/cached pos (current)
torch::Tensor y; // mutable/cached pos (current)
torch::Tensor node_size_x;
torch::Tensor node_size_y;
/* pin info */
torch::Tensor pin_offset_x;
torch::Tensor pin_offset_y;
torch::Tensor flat_node2pin_start_map;
torch::Tensor flat_node2pin_map;
torch::Tensor pin2node_map;
/* net info */
torch::Tensor flat_net2pin_start_map;
torch::Tensor flat_net2pin_map;
torch::Tensor pin2net_map;
torch::Tensor net_mask;
/* fence info */
torch::Tensor flat_region_boxes_start;
torch::Tensor flat_region_boxes;
torch::Tensor node2fence_region_map;
/* chip info */
float xl;
float yl;
float xh;
float yh;
/* row info */
int num_sites_x;
int num_sites_y;
int num_pins;
int num_nets;
int num_nodes;
int num_movable_nodes;
int num_regions;
float site_width;
float row_height;
int num_threads;
};
/* API for python */
// Legalization
bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y);
void abacusLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y);
void greedyLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y);
// Detailed Placement
void kReorder(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int K, int max_iters);
void globalSwap(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int max_iters);
void independentSetMatching(
DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int set_size, int max_iters);
// Legality Check
bool legalityCheck(DPTorchRawDB& at_db, float scale_factor);
} // namespace dp

View File

@ -0,0 +1,39 @@
#include <cuda.h>
#include <cuda_runtime.h>
#include "common/common.h"
#include "gpudp/db/dp_torch.h"
#include "detailed_place_db.cuh"
namespace dp {
__global__ void compute_total_hpwl_kernel(DetailedPlaceData db, const float* xx, const float* yy, double* net_hpwls) {
for (int i = blockIdx.x * blockDim.x + threadIdx.x; i < db.num_nets; i += blockDim.x * gridDim.x) {
net_hpwls[i] = double(db.compute_net_hpwl(i, xx, yy));
}
}
float compute_total_hpwl(const DetailedPlaceData& db, const float* xx, const float* yy, double* net_hpwls) {
compute_total_hpwl_kernel<<<ceilDiv(db.num_nets, 512), 512>>>(db, xx, yy, net_hpwls);
// auto hpwl = thrust::reduce(thrust::device, net_hpwls, net_hpwls+db.num_nets);
double* d_out = NULL;
// Determine temporary device storage requirements
void* d_temp_storage = NULL;
size_t temp_storage_bytes = 0;
cub::DeviceReduce::Sum(d_temp_storage, temp_storage_bytes, net_hpwls, d_out, db.num_nets);
// Allocate temporary storage
checkCuda(cudaMalloc(&d_temp_storage, temp_storage_bytes));
checkCuda(cudaMalloc(&d_out, sizeof(double)));
// Run sum-reduction
cub::DeviceReduce::Sum(d_temp_storage, temp_storage_bytes, net_hpwls, d_out, db.num_nets);
// copy d_out to hpwl
double hpwl = 0;
checkCuda(cudaMemcpy(&hpwl, d_out, sizeof(double), cudaMemcpyDeviceToHost));
cudaFree(d_temp_storage);
cudaFree(d_out);
return float(hpwl);
}
} // namespace dp

View File

@ -0,0 +1,24 @@
#include "common/common.h"
#include "gpudp/db/dp_torch.h"
namespace dp {
void kReorderCUDA(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int K, int max_iters);
void globalSwapCUDA(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int max_iters);
void independentSetMatchingCUDA(
DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int set_size, int max_iters);
void kReorder(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int K, int max_iters) {
kReorderCUDA(at_db, num_bins_x, num_bins_y, K, max_iters);
}
void globalSwap(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int max_iters) {
globalSwapCUDA(at_db, num_bins_x, num_bins_y, batch_size, max_iters);
}
void independentSetMatching(
DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int set_size, int max_iters) {
independentSetMatchingCUDA(at_db, num_bins_x, num_bins_y, batch_size, set_size, max_iters);
}
} // namespace dp

View File

@ -0,0 +1,417 @@
#pragma once
#include <cuda.h>
#include <cuda_runtime.h>
#include "common/common.h"
#include "common/db/Database.h"
#include "gpudp/db/dp_torch.h"
#include "pitch_nested_vector.cuh"
#include "utils.cuh"
namespace dp {
inline __host__ __device__ int floorDiv(float a, float b, float rtol = 1e-4) { return floor((a + rtol * b) / b); }
inline __host__ __device__ int ceilDiv(float a, float b, float rtol = 1e-4) { return ceil((a - rtol * b) / b); }
inline __host__ __device__ int roundDiv(float a, float b) { return round(a / b); }
template <typename T>
struct Space {
T xl;
T xh;
};
template <typename T>
struct Box {
T xl;
T yl;
T xh;
T yh;
__host__ __device__ Box() {
xl = cuda::numeric_limits<T>::max();
yl = cuda::numeric_limits<T>::max();
xh = cuda::numeric_limits<T>::lowest();
yh = cuda::numeric_limits<T>::lowest();
}
__host__ __device__ Box(T xxl, T yyl, T xxh, T yyh) : xl(xxl), yl(yyl), xh(xxh), yh(yyh) {}
__host__ __device__ T center_x() const { return (xl + xh) / 2; }
__host__ __device__ T center_y() const { return (yl + yh) / 2; }
};
struct RowMapIndex {
int row_id;
int sub_id;
};
struct BinMapIndex {
int bin_id;
int sub_id;
};
class DetailedPlaceData {
public:
DetailedPlaceData() {}
DetailedPlaceData(DPTorchRawDB& at_db)
: x(at_db.x.data_ptr<float>()),
y(at_db.y.data_ptr<float>()),
init_x(at_db.init_x.data_ptr<float>()),
init_y(at_db.init_y.data_ptr<float>()),
node_size_x(at_db.node_size_x.data_ptr<float>()),
node_size_y(at_db.node_size_y.data_ptr<float>()),
pin_offset_x(at_db.pin_offset_x.data_ptr<float>()),
pin_offset_y(at_db.pin_offset_y.data_ptr<float>()),
flat_node2pin_start_map(at_db.flat_node2pin_start_map.data_ptr<int>()),
flat_node2pin_map(at_db.flat_node2pin_map.data_ptr<int>()),
pin2node_map(at_db.pin2node_map.data_ptr<int>()),
flat_net2pin_start_map(at_db.flat_net2pin_start_map.data_ptr<int>()),
flat_net2pin_map(at_db.flat_net2pin_map.data_ptr<int>()),
pin2net_map(at_db.pin2net_map.data_ptr<int>()),
flat_region_boxes_start(at_db.flat_region_boxes_start.data_ptr<int>()),
flat_region_boxes(at_db.flat_region_boxes.data_ptr<float>()),
node2fence_region_map(at_db.node2fence_region_map.data_ptr<int>()),
net_mask(at_db.net_mask.data_ptr<bool>()),
node_weight(at_db.node_weight.data_ptr<float>()),
xl(at_db.xl),
xh(at_db.xh),
yl(at_db.yl),
yh(at_db.yh),
row_height(at_db.row_height),
site_width(at_db.site_width),
num_sites_x(at_db.num_sites_x),
num_sites_y(at_db.num_sites_y),
num_threads(at_db.num_threads),
num_nodes(at_db.num_nodes),
num_movable_nodes(at_db.num_movable_nodes),
num_nets(at_db.num_nets),
num_pins(at_db.num_pins),
num_regions(at_db.num_regions) {}
public:
typedef float type;
float* x;
float* y;
const float* init_x;
const float* init_y;
const float* node_size_x;
const float* node_size_y;
const float* pin_offset_x;
const float* pin_offset_y;
const int* flat_node2pin_start_map;
const int* flat_node2pin_map;
const int* pin2node_map;
const int* flat_net2pin_start_map;
const int* flat_net2pin_map;
const int* pin2net_map;
const int* flat_region_boxes_start;
const float* flat_region_boxes;
const int* node2fence_region_map;
const bool* net_mask;
const float* node_weight;
/* chip info */
float xl;
float yl;
float xh;
float yh;
/* row info */
int num_sites_x;
int num_sites_y;
float row_height;
float site_width;
int num_nets;
int num_movable_nodes;
int num_nodes;
int num_pins;
int num_regions;
int num_threads;
int num_bins_x;
int num_bins_y;
float bin_size_x;
float bin_size_y;
public:
void set_num_bins(int num_bins_x_, int num_bins_y_) {
num_bins_x = num_bins_x_;
num_bins_y = num_bins_y_;
bin_size_x = (xh - xl) / num_bins_x_;
bin_size_y = (yh - yl) / num_bins_y_;
}
inline __device__ int pos2site_x(float xx) const {
return min(max((int)floorDiv((xx - xl), site_width), 0), num_sites_x - 1);
}
inline __device__ int pos2site_y(float yy) const {
return min(max((int)floorDiv((yy - yl), row_height), 0), num_sites_y - 1);
}
inline __device__ int pos2site_ub_x(float xx) const {
return min(max(ceilDiv((xx - xl), site_width), 1), num_sites_x);
}
inline __device__ int pos2site_ub_y(float yy) const {
return min(max(ceilDiv((yy - yl), row_height), 1), num_sites_y);
}
inline __device__ int pos2bin_x(float xx) const {
int bx = floorDiv((xx - xl), bin_size_x);
bx = max(bx, 0);
bx = min(bx, num_bins_x - 1);
return bx;
}
inline __device__ int pos2bin_y(float yy) const {
int by = floorDiv((yy - yl), bin_size_y);
by = max(by, 0);
by = min(by, num_bins_y - 1);
return by;
}
inline __device__ void shift_box_to_layout(Box<float>& box) const {
box.xl = max(box.xl, xl);
box.xl = min(box.xl, xh);
box.xh = max(box.xh, xl);
box.xh = min(box.xh, xh);
box.yl = max(box.yl, yl);
box.yl = min(box.yl, yh);
box.yh = max(box.yh, yl);
box.yh = min(box.yh, yh);
}
inline __device__ float align2site(float xx) const {
return (int)floorDiv((xx - xl), site_width) * site_width + xl;
}
inline __device__ Space<float> align2site(Space<float> space) const {
space.xl = ceilDiv((space.xl - xl), site_width) * site_width + xl;
space.xh = floorDiv((space.xh - xl), site_width) * site_width + xl;
return space;
}
__device__ Box<float> compute_optimal_region(int node_id, const float* xx, const float* yy) const {
Box<float> box(xh, yh, xl, yl);
for (int node2pin_id = flat_node2pin_start_map[node_id]; node2pin_id < flat_node2pin_start_map[node_id + 1];
++node2pin_id) {
int node_pin_id = flat_node2pin_map[node2pin_id];
int net_id = pin2net_map[node_pin_id];
if (net_mask[net_id]) {
for (int net2pin_id = flat_net2pin_start_map[net_id]; net2pin_id < flat_net2pin_start_map[net_id + 1];
++net2pin_id) {
int net_pin_id = flat_net2pin_map[net2pin_id];
int other_node_id = pin2node_map[net_pin_id];
if (node_id != other_node_id) {
box.xl = min(box.xl, xx[other_node_id] + pin_offset_x[net_pin_id]);
box.xh = max(box.xh, xx[other_node_id] + pin_offset_x[net_pin_id]);
box.yl = min(box.yl, yy[other_node_id] + pin_offset_y[net_pin_id]);
box.yh = max(box.yh, yy[other_node_id] + pin_offset_y[net_pin_id]);
}
}
}
}
shift_box_to_layout(box);
return box;
}
__device__ float compute_net_hpwl(int net_id, const float* xx, const float* yy) const {
Box<float> box(xh, yh, xl, yl);
for (int net2pin_id = flat_net2pin_start_map[net_id]; net2pin_id < flat_net2pin_start_map[net_id + 1];
++net2pin_id) {
int net_pin_id = flat_net2pin_map[net2pin_id];
int other_node_id = pin2node_map[net_pin_id];
box.xl = min(box.xl, xx[other_node_id] + pin_offset_x[net_pin_id]);
box.xh = max(box.xh, xx[other_node_id] + pin_offset_x[net_pin_id]);
box.yl = min(box.yl, yy[other_node_id] + pin_offset_y[net_pin_id]);
box.yh = max(box.yh, yy[other_node_id] + pin_offset_y[net_pin_id]);
}
if (box.xl == xh || box.yl == yh) {
return (float)0;
}
return (box.xh - box.xl) + (box.yh - box.yl);
}
// __device__ float compute_total_hpwl() const {
// float total_hpwl = 0;
// for (int net_id = 0; net_id < num_nets; ++net_id) {
// total_hpwl += compute_net_hpwl(net_id, x, y);
// }
// return total_hpwl;
// }
__device__ bool inside_fence(int node_id, float xx, float yy) const {
float node_xl = xx;
float node_yl = yy;
float node_xh = node_xl + node_size_x[node_id];
float node_yh = node_yl + node_size_y[node_id];
bool legal_flag = true;
int region_id = node2fence_region_map[node_id];
if (region_id < num_regions) {
int box_bgn = flat_region_boxes_start[region_id];
int box_end = flat_region_boxes_start[region_id + 1];
float node_area = (node_xh - node_xl) * (node_yh - node_yl);
// assume there is no overlap between boxes of a region
// otherwise, preprocessing is required
for (int box_id = box_bgn; box_id < box_end; ++box_id) {
int box_offset = box_id * 4;
float box_xl = flat_region_boxes[box_offset];
float box_xh = flat_region_boxes[box_offset + 1];
float box_yl = flat_region_boxes[box_offset + 2];
float box_yh = flat_region_boxes[box_offset + 3];
float dx = max(min(node_xh, box_xh) - max(node_xl, box_xl), (float)0);
float dy = max(min(node_yh, box_yh) - max(node_yl, box_yl), (float)0);
float overlap = dx * dy;
if (overlap > 0) {
node_area -= overlap;
}
}
if (node_area > 0) {
// not consumed by boxes within a region
legal_flag = false;
}
}
return legal_flag;
}
void make_row2node_map(const float* host_x,
const float* host_y,
const float* host_node_size_x,
const float* host_node_size_y,
int host_num_nodes,
std::vector<std::vector<int>>& row2node_map) {
// distribute cells to rows
for (int i = 0; i < host_num_nodes; ++i) {
float node_yl = host_y[i];
float node_yh = node_yl + host_node_size_y[i];
int row_idxl = floorDiv(node_yl - yl, row_height);
int row_idxh = ceilDiv(node_yh - yl, row_height);
row_idxl = max(row_idxl, 0);
row_idxh = min(row_idxh, num_sites_y);
for (int row_id = row_idxl; row_id < row_idxh; ++row_id) {
float row_yl = yl + row_id * row_height;
float row_yh = row_yl + row_height;
if (node_yl < row_yh && node_yh > row_yl) // overlap with row
{
row2node_map[row_id].push_back(i);
}
}
}
// sort cells within rows
#pragma omp parallel for num_threads(num_threads) schedule(dynamic, 1)
for (int i = 0; i < num_sites_y; ++i) {
auto& row2nodes = row2node_map[i];
// sort cells within rows according to left edges
std::sort(row2nodes.begin(), row2nodes.end(), [&](int node_id1, int node_id2) {
float x1 = host_x[node_id1];
float x2 = host_x[node_id2];
return x1 < x2 || (x1 == x2 && node_id1 < node_id2);
});
if (!row2nodes.empty()) {
std::vector<int> tmp_nodes;
tmp_nodes.reserve(row2nodes.size());
tmp_nodes.push_back(row2nodes.front());
for (int j = 1, je = row2nodes.size(); j < je; ++j) {
int node_id1 = row2nodes.at(j - 1);
int node_id2 = row2nodes.at(j);
// two fixed cells
if (node_id1 >= num_movable_nodes && node_id2 >= num_movable_nodes) {
float xl1 = host_x[node_id1];
float xl2 = host_x[node_id2];
float width1 = host_node_size_x[node_id1];
float width2 = host_node_size_x[node_id2];
float xh1 = xl1 + width1;
float xh2 = xl2 + width2;
// only collect node_id2 if its right edge is righter than node_id1
if (xh1 < xh2) {
tmp_nodes.push_back(node_id2);
}
} else {
tmp_nodes.push_back(node_id2);
}
}
row2nodes.swap(tmp_nodes);
// sort according to center
std::sort(row2nodes.begin(), row2nodes.end(), [&](int node_id1, int node_id2) {
float x1 = host_x[node_id1] + host_node_size_x[node_id1] / 2;
float x2 = host_x[node_id2] + host_node_size_x[node_id2] / 2;
return x1 < x2 || (x1 == x2 && node_id1 < node_id2);
});
}
}
}
void make_row2node_map_with_spaces(const float* host_x,
const float* host_y,
const float* host_node_size_x,
const float* host_node_size_y,
std::vector<std::vector<int>>& row2node_map,
std::vector<RowMapIndex>& node2row_map,
std::vector<Space<float>>& spaces) {
make_row2node_map(host_x, host_y, host_node_size_x, host_node_size_y, num_nodes + 2, row2node_map);
// construct node2row_map
for (int i = 0; i < num_sites_y; ++i) {
for (unsigned int j = 0; j < row2node_map[i].size(); ++j) {
int node_id = row2node_map[i][j];
if (node_id < num_movable_nodes) {
RowMapIndex& row_id = node2row_map[node_id];
row_id.row_id = i;
row_id.sub_id = j;
}
}
}
// construct spaces
for (int i = 0; i < num_sites_y; ++i) {
for (unsigned int j = 0; j < row2node_map[i].size(); ++j) {
int node_id = row2node_map[i][j];
if (node_id < num_movable_nodes) {
assert(j);
int left_node_id = row2node_map[i][j - 1];
spaces[node_id].xl = host_x[left_node_id] + host_node_size_x[left_node_id];
assert(j + 1 < row2node_map[i].size());
int right_node_id = row2node_map[i][j + 1];
spaces[node_id].xh = host_x[right_node_id];
}
}
}
}
void make_bin2node_map(const float* host_x,
const float* host_y,
const float* host_node_size_x,
const float* host_node_size_y,
std::vector<std::vector<int>>& bin2node_map,
std::vector<BinMapIndex>& node2bin_map) {
// construct bin2node_map
for (int i = 0; i < num_movable_nodes; ++i) {
int node_id = i;
float node_x = host_x[node_id] + host_node_size_x[node_id] / 2;
float node_y = host_y[node_id] + host_node_size_y[node_id] / 2;
int bx = min(max((int)floorDiv(node_x - xl, bin_size_x), 0), num_bins_x - 1);
int by = min(max((int)floorDiv(node_y - yl, bin_size_y), 0), num_bins_y - 1);
int bin_id = bx * num_bins_y + by;
int sub_id = bin2node_map.at(bin_id).size();
bin2node_map.at(bin_id).push_back(node_id);
}
for (int bin_id = 0; bin_id < bin2node_map.size(); ++bin_id) {
for (int sub_id = 0; sub_id < bin2node_map[bin_id].size(); ++sub_id) {
int node_id = bin2node_map[bin_id][sub_id];
BinMapIndex& bm_idx = node2bin_map.at(node_id);
bm_idx.bin_id = bin_id;
bm_idx.sub_id = sub_id;
}
}
}
};
float compute_total_hpwl(const DetailedPlaceData& db, const float* xx, const float* yy, double* net_hpwls);
} // namespace dp

File diff suppressed because it is too large Load Diff

View File

@ -0,0 +1,323 @@
#include <curand.h>
#include <curand_kernel.h>
#include "detailed_place_db.cuh"
#include "gpudp/dp/ism/apply_solution.cuh"
#include "gpudp/dp/ism/auction.cuh"
#include "gpudp/dp/ism/collect_independent_sets.cuh"
#include "gpudp/dp/ism/cost_matrix_construction.cuh"
#include "gpudp/dp/ism/cpu_state.cuh"
#include "gpudp/dp/ism/maximal_independent_set.cuh"
#include "gpudp/dp/ism/shuffle.cuh"
namespace dp {
#define DETERMINISTIC
#define NUM_NODE_SIZES 64 ///< number of different cell sizes
struct SizedBinIndex {
int size_id;
int bin_id;
};
template <typename T>
struct IndependentSetMatchingState {
typedef T type;
typedef int cost_type;
int* ordered_nodes = nullptr;
Space<T>* spaces = nullptr; ///< array of cell spaces, each cell only consider the space on its left side except
///< for the left and right boundary
int num_node_sizes; ///< number of cell sizes considered
int* independent_sets = nullptr; ///< independent sets, length of batch_size*set_size
int* independent_set_sizes = nullptr; ///< size of each independent set
int* selected_maximal_independent_set = nullptr; ///< storing the selected maximum independent set
int* select_scratch = nullptr; ///< temporary storage for selection kernel
int num_selected; ///< maximum independent set size
int* device_num_selected; ///< maximum independent set size
double* net_hpwls; ///< HPWL for each net, use integer to get consistent values
int* selected_markers = nullptr; ///< must be int for cub to compute prefix sum
unsigned char* dependent_markers = nullptr;
int* independent_set_empty_flag = nullptr; ///< a stopping flag for maximum independent set
int num_independent_sets; ///< host copy
cost_type* cost_matrices = nullptr; ///< cost matrices batch_size*set_size*set_size
cost_type* cost_matrices_copy = nullptr; ///< temporary copy of cost matrices
int* solutions = nullptr; ///< batch_size*set_size
char* auction_scratch = nullptr; ///< temporary memory for auction solver
char* stop_flags = nullptr; ///< record stopping status from auction solver
T* orig_x = nullptr; ///< original locations of cells for applying solutions
T* orig_y = nullptr;
cost_type* orig_costs = nullptr; ///< original costs
cost_type* solution_costs = nullptr; ///< solution costs
Space<T>* orig_spaces = nullptr; ///< original spaces of cells for apply solutions
int batch_size; ///< pre-allocated number of independent sets
int set_size;
int cost_matrix_size; ///< set_size*set_size
int num_bins; ///< num_bins_x*num_bins_y
int* device_num_moved; ///< device copy
int num_moved; ///< host copy, number of moved cells
int large_number; ///< a large number
float auction_max_eps; ///< maximum epsilon for auction solver
float auction_min_eps; ///< minimum epsilon for auction solver
float auction_factor; ///< decay factor for auction epsilon
int auction_max_iterations; ///< maximum iteration
T skip_threshold; ///< ignore connections if cells are far apart
};
template <typename T>
__global__ void iota(T* a, int n) {
for (int i = blockIdx.x * blockDim.x + threadIdx.x; i < n; i += blockDim.x * gridDim.x) {
a[i] = i;
}
}
__global__ void cost_matrix_init(int* cost_matrix, int set_size) {
for (int i = blockIdx.x; i < set_size; i += gridDim.x) {
for (int j = threadIdx.x; j < set_size; j += blockDim.x) {
cost_matrix[i * set_size + j] = (i == j) ? 0 : cuda::numeric_limits<int>::max();
}
}
}
template <typename T>
__global__ void print_global(T* a, int n) {
unsigned int tid = threadIdx.x;
unsigned int bid = blockIdx.x;
if (tid == 0 && bid == 0) {
printf("[%d]\n", n);
for (int i = 0; i < n; ++i) {
printf("%g ", (double)a[i]);
}
printf("\n");
}
}
template <typename T>
__global__ void print_cost_matrix(const T* cost_matrix, int set_size, bool major) {
unsigned int tid = threadIdx.x;
unsigned int bid = blockIdx.x;
if (tid == 0 && bid == 0) {
printf("[%dx%d]\n", set_size, set_size);
for (int r = 0; r < set_size; ++r) {
for (int c = 0; c < set_size; ++c) {
if (major) // column major
{
printf("%g ", (double)cost_matrix[c * set_size + r]);
} else {
printf("%g ", (double)cost_matrix[r * set_size + c]);
}
}
printf("\n");
}
printf("\n");
}
}
template <typename T>
__global__ void print_solution(const T* solution, int n) {
unsigned int tid = threadIdx.x;
unsigned int bid = blockIdx.x;
if (tid == 0 && bid == 0) {
printf("[%d]\n", n);
for (int i = 0; i < n; ++i) {
printf("%g ", (double)solution[i]);
}
printf("\n");
}
}
void construct_spaces(DetailedPlaceData& db,
const float* host_x,
const float* host_y,
const float* host_node_size_x,
const float* host_node_size_y,
std::vector<Space<float>>& host_spaces,
int num_threads) {
std::vector<std::vector<int> > row2node_map(db.num_sites_y);
db.make_row2node_map(host_x, host_y, host_node_size_x, host_node_size_y, db.num_nodes, row2node_map);
// construct spaces
host_spaces.resize(db.num_movable_nodes);
for (int i = 0; i < db.num_sites_y; ++i) {
for (unsigned int j = 0; j < row2node_map[i].size(); ++j) {
auto const& row2nodes = row2node_map[i];
int node_id = row2nodes[j];
auto& space = host_spaces[node_id];
if (node_id < db.num_movable_nodes) {
auto left_bound = db.xl;
if (j) {
left_bound = host_x[node_id];
}
space.xl = ceilDiv(left_bound - db.xl, db.site_width) * db.site_width + db.xl;
auto right_bound = db.xh;
if (j + 1 < row2nodes.size()) {
int right_node_id = row2nodes[j + 1];
right_bound = min(right_bound, host_x[right_node_id]);
}
space.xh = std::floor(right_bound);
space.xh = floorDiv(space.xh - db.xl, db.site_width) * db.site_width + db.xl;
}
}
}
}
void independentSetMatchingCUDA(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y, int batch_size, int set_size, int max_iters) {
cudaSetDevice(at_db.node_size_x.get_device());
DetailedPlaceData db(at_db);
db.set_num_bins(num_bins_x, num_bins_y);
// fix random seed
std::srand(1000);
IndependentSetMatchingState<float> state;
// initialize host database
DetailedPlaceCPUDB<float> host_db;
init_cpu_db(db, host_db);
state.batch_size = batch_size;
state.set_size = set_size;
state.cost_matrix_size = state.set_size * state.set_size;
state.num_bins = db.num_bins_x * db.num_bins_y;
state.num_moved = 0;
state.large_number = ((db.xh - db.xl) + (db.yh - db.yl)) * set_size;
state.skip_threshold = ((db.xh - db.xl) + (db.yh - db.yl)) * 0.01;
state.auction_max_eps = 10.0;
state.auction_min_eps = 1.0;
state.auction_factor = 0.1;
state.auction_max_iterations = 9999;
checkCuda(cudaMemcpy(host_db.x.data(), db.x, sizeof(float) * db.num_nodes, cudaMemcpyDeviceToHost));
checkCuda(cudaMemcpy(host_db.y.data(), db.y, sizeof(float) * db.num_nodes, cudaMemcpyDeviceToHost));
std::vector<Space<float>> host_spaces(db.num_movable_nodes);
construct_spaces(db,
host_db.x.data(),
host_db.y.data(),
host_db.node_size_x.data(),
host_db.node_size_y.data(),
host_spaces,
db.num_threads);
// initialize cuda state
allocateCopyCuda(state.spaces, host_spaces.data(), db.num_movable_nodes);
allocateCuda(state.ordered_nodes, db.num_movable_nodes, int);
iota<<<ceilDiv(db.num_movable_nodes, 512), 512>>>(state.ordered_nodes, db.num_movable_nodes);
allocateCuda(state.independent_sets, state.batch_size * state.set_size, int);
allocateCuda(state.independent_set_sizes, state.batch_size, int);
allocateCuda(state.selected_maximal_independent_set, db.num_movable_nodes, int);
allocateCuda(state.select_scratch, db.num_movable_nodes, int);
allocateCuda(state.device_num_selected, 1, int);
allocateCuda(state.orig_x, state.batch_size * state.set_size, float);
allocateCuda(state.orig_y, state.batch_size * state.set_size, float);
allocateCuda(state.orig_spaces, state.batch_size * state.set_size, Space<float>);
allocateCuda(state.selected_markers, db.num_nodes, int);
allocateCuda(state.dependent_markers, db.num_nodes, unsigned char);
allocateCuda(state.independent_set_empty_flag, 1, int);
allocateCuda(state.cost_matrices,
state.batch_size * state.set_size * state.set_size,
typename IndependentSetMatchingState<float>::cost_type);
allocateCuda(state.cost_matrices_copy,
state.batch_size * state.set_size * state.set_size,
typename IndependentSetMatchingState<float>::cost_type);
allocateCuda(state.solutions, state.batch_size * state.set_size, int);
allocateCuda(
state.orig_costs, state.batch_size * state.set_size, typename IndependentSetMatchingState<float>::cost_type);
allocateCuda(state.solution_costs,
state.batch_size * state.set_size,
typename IndependentSetMatchingState<float>::cost_type);
allocateCuda(state.net_hpwls, db.num_nets, typename std::remove_pointer<decltype(state.net_hpwls)>::type);
allocateCopyCuda(state.device_num_moved, &state.num_moved, 1);
init_auction<float>(state.batch_size, state.set_size, state.auction_scratch, state.stop_flags);
Shuffler<int, unsigned int> shuffler(2023ULL, state.ordered_nodes, db.num_movable_nodes);
// initialize host state
IndependentSetMatchingCPUState<float> host_state;
init_cpu_state(db, state, host_state);
// initialize kmeans state
KMeansState<float> kmeans_state;
init_kmeans(db, state, kmeans_state);
std::vector<float> hpwls(max_iters + 1);
hpwls[0] = compute_total_hpwl(db, db.x, db.y, state.net_hpwls);
logger.info("initial hpwl %g", hpwls[0]);
for (int iter = 0; iter < max_iters; ++iter) {
shuffler();
checkCuda(cudaDeviceSynchronize());
maximal_independent_set(db, state);
checkCuda(cudaDeviceSynchronize());
collect_independent_sets(db, state, kmeans_state, host_db, host_state);
checkCuda(cudaDeviceSynchronize());
cost_matrix_construction(db, state);
checkCuda(cudaDeviceSynchronize());
// solve independent sets
// print_cost_matrix<<<1, 1>>>(state.cost_matrices + state.cost_matrix_size*3, state.set_size, 0);
linear_assignment_auction(state.cost_matrices,
state.solutions,
state.num_independent_sets,
state.set_size,
state.auction_scratch,
state.stop_flags,
state.auction_max_eps,
state.auction_min_eps,
state.auction_factor,
state.auction_max_iterations);
checkCuda(cudaDeviceSynchronize());
// print_solution<<<1, 1>>>(state.solutions + state.set_size*3, state.set_size);
// apply solutions
apply_solution(db, state);
checkCuda(cudaDeviceSynchronize());
hpwls[iter + 1] = compute_total_hpwl(db, db.x, db.y, state.net_hpwls);
if ((iter % (max(max_iters / 10, 1))) == 0 || iter + 1 == max_iters) {
logger.info("iteration %d, target hpwl %g, delta %g(%g%%), %d independent sets, moved %g%% cells",
iter,
hpwls[iter + 1],
hpwls[iter + 1] - hpwls[0],
(hpwls[iter + 1] - hpwls[0]) / hpwls[0] * 100,
state.num_independent_sets,
state.num_moved / (double)db.num_movable_nodes * 100);
}
}
// destroy state
cudaFree(state.spaces);
cudaFree(state.ordered_nodes);
cudaFree(state.independent_sets);
cudaFree(state.independent_set_sizes);
cudaFree(state.selected_maximal_independent_set);
cudaFree(state.select_scratch);
cudaFree(state.device_num_selected);
cudaFree(state.net_hpwls);
cudaFree(state.cost_matrices);
cudaFree(state.cost_matrices_copy);
cudaFree(state.solutions);
cudaFree(state.orig_costs);
cudaFree(state.solution_costs);
cudaFree(state.orig_x);
cudaFree(state.orig_y);
cudaFree(state.orig_spaces);
cudaFree(state.selected_markers);
cudaFree(state.dependent_markers);
cudaFree(state.independent_set_empty_flag);
cudaFree(state.device_num_moved);
destroy_auction(state.auction_scratch, state.stop_flags);
destroy_kmeans(kmeans_state);
}
} // namespace dp

View File

@ -0,0 +1,15 @@
#pragma once
#include "gpudp/dp/detailed_place_db.cuh"
namespace dp {
template <typename T>
__host__ __device__ bool adjust_pos(T& x, T width, const Space<T>& space) {
// the order is very tricky for numerical stability
x = min(x, space.xh - width);
x = max(x, space.xl);
return width + space.xl <= space.xh;
}
} // namespace dp

View File

@ -0,0 +1,281 @@
#pragma once
#include "adjust_pos.cuh"
#include "gpudp/dp/detailed_place_db.cuh"
namespace dp {
template <typename T>
__global__ void copy_orig_cost_kernel(const T* cost_matrices, const char* stop_flags, const int set_size, T* costs) {
int i = blockIdx.x; // set
if (stop_flags[i]) {
auto cost_matrix = cost_matrices + i * set_size * set_size;
auto cost = costs + i * set_size;
for (int j = threadIdx.x; j < set_size; j += blockDim.x) {
cost[j] = cost_matrix[j * set_size + j];
}
}
}
template <typename T>
__global__ void copy_solution_cost_kernel(
const T* cost_matrices, const char* stop_flags, const int* solutions, const int set_size, T* costs) {
int i = blockIdx.x; // set
if (stop_flags[i]) {
auto cost_matrix = cost_matrices + i * set_size * set_size;
auto cost = costs + i * set_size;
auto solution = solutions + i * set_size;
for (int j = threadIdx.x; j < set_size; j += blockDim.x) {
int sol_k = solution[j];
cost[j] = cost_matrix[j * set_size + sol_k];
}
}
}
template <typename T, int BlockDim>
__global__ void block_reduce_sum(T* costs, const char* stop_flags, int batch_size, int set_size) {
int bid = blockIdx.x; // set
int tid = threadIdx.x;
if (stop_flags[bid]) {
// Specialize BlockReduce for a 1D block of BlockDim threads on type int
typedef cub::BlockReduce<T, BlockDim> BlockReduce;
// Allocate shared memory for BlockReduce
__shared__ typename BlockReduce::TempStorage temp_storage;
// Obtain a segment of consecutive items that are blocked across threads
int thread_data[1];
thread_data[0] = costs[bid * set_size + tid];
__syncthreads();
// Compute the block-wide sum for thread0
T aggregate = BlockReduce(temp_storage).Sum(thread_data);
__syncthreads();
if (tid == 0) {
costs[bid * set_size] = aggregate;
}
}
}
template <typename T>
void compute_costs(const char* stop_flags, const int batch_size, const int set_size, T* costs) {
switch (set_size) {
case 2:
block_reduce_sum<T, 2><<<batch_size, 2>>>(costs, stop_flags, batch_size, set_size);
break;
case 4:
block_reduce_sum<T, 4><<<batch_size, 4>>>(costs, stop_flags, batch_size, set_size);
break;
case 8:
block_reduce_sum<T, 8><<<batch_size, 8>>>(costs, stop_flags, batch_size, set_size);
break;
case 16:
block_reduce_sum<T, 16><<<batch_size, 16>>>(costs, stop_flags, batch_size, set_size);
break;
case 32:
block_reduce_sum<T, 32><<<batch_size, 32>>>(costs, stop_flags, batch_size, set_size);
break;
case 64:
block_reduce_sum<T, 64><<<batch_size, 64>>>(costs, stop_flags, batch_size, set_size);
break;
case 128:
block_reduce_sum<T, 128><<<batch_size, 128>>>(costs, stop_flags, batch_size, set_size);
break;
case 256:
block_reduce_sum<T, 256><<<batch_size, 256>>>(costs, stop_flags, batch_size, set_size);
break;
case 512:
block_reduce_sum<T, 512><<<batch_size, 512>>>(costs, stop_flags, batch_size, set_size);
break;
case 1024:
block_reduce_sum<T, 1024><<<batch_size, 1024>>>(costs, stop_flags, batch_size, set_size);
break;
default:
assert_msg(0, "unsupported set size %d", set_size);
}
}
template <typename T>
__global__ void print_copy_costs_kernel(const T* costs, int batch_size, int set_size) {
if (blockIdx.x == 0 && threadIdx.x == 0) {
for (int i = 0; i < batch_size; ++i) {
printf("[%d] orig/solution_costs ", i);
for (int j = 0; j < set_size; ++j) {
printf("%g ", (float)costs[i * set_size + j]);
}
printf("\n");
}
}
}
template <typename T>
void compute_orig_cost(
const T* cost_matrices, const char* stop_flags, const int batch_size, const int set_size, T* costs) {
copy_orig_cost_kernel<<<batch_size, set_size>>>(cost_matrices, stop_flags, set_size, costs);
compute_costs(stop_flags, batch_size, set_size, costs);
}
template <typename T>
void compute_solution_cost(const T* cost_matrices,
const int* solutions,
const char* stop_flags,
const int batch_size,
const int set_size,
T* costs) {
copy_solution_cost_kernel<<<batch_size, set_size>>>(cost_matrices, stop_flags, solutions, set_size, costs);
// print_copy_costs_kernel<<<1, 1>>>(costs, batch_size, set_size);
compute_costs(stop_flags, batch_size, set_size, costs);
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void store_orig_pos_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
int i = blockIdx.x; // set
const int* __restrict__ independent_set = state.independent_sets + i * state.set_size;
auto orig_x = state.orig_x + i * state.set_size;
auto orig_y = state.orig_y + i * state.set_size;
auto orig_spaces = state.orig_spaces + i * state.set_size;
for (int j = threadIdx.x; j < state.set_size; j += blockDim.x) {
int node_id = independent_set[j];
if (node_id < db.num_movable_nodes) {
assert(node_id >= 0);
orig_x[j] = db.x[node_id];
orig_y[j] = db.y[node_id];
orig_spaces[j] = state.spaces[node_id];
}
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void move_nodes_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
int i = blockIdx.x; // set
int idx = i * state.set_size;
if (state.stop_flags[i]) {
// encourage movement
if (state.orig_costs[idx] <= state.solution_costs[idx]) {
const int* __restrict__ independent_set = state.independent_sets + i * state.set_size;
const int* __restrict__ solution = state.solutions + i * state.set_size;
const typename IndependentSetMatchingStateType::type* __restrict__ orig_x =
state.orig_x + i * state.set_size;
const typename IndependentSetMatchingStateType::type* __restrict__ orig_y =
state.orig_y + i * state.set_size;
const Space<typename IndependentSetMatchingStateType::type>* __restrict__ orig_spaces =
state.orig_spaces + i * state.set_size;
for (int j = threadIdx.x; j < state.set_size; j += blockDim.x) {
int node_id = independent_set[j];
int sol_k = solution[j];
if (node_id < db.num_movable_nodes) {
auto node_width = db.node_size_x[node_id];
auto& x = db.x[node_id];
auto& y = db.y[node_id];
auto& space = state.spaces[node_id];
if (j != sol_k) {
atomicAdd(state.device_num_moved, 1);
auto const& orig_space = orig_spaces[sol_k];
x = orig_x[sol_k];
bool ret = adjust_pos(x, node_width, orig_space);
assert(ret);
y = orig_y[sol_k];
space = orig_space;
}
}
}
}
}
}
template <typename IndependentSetMatchingStateType>
__global__ void print_orig_and_solution_costs_kernel(IndependentSetMatchingStateType state) {
if (blockIdx.x == 0 && threadIdx.x == 0) {
for (int i = 0; i < state.num_independent_sets; ++i) {
int stop = state.stop_flags[i];
printf("[%d] orig_costs %g, solution_costs %g, delta %g, stop_flag %d\n",
i,
(float)state.orig_costs[i * state.set_size],
(float)state.solution_costs[i * state.set_size],
(float)(state.solution_costs[i * state.set_size] - state.orig_costs[i * state.set_size]),
stop);
}
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void check_hpwl_kernel(DetailedPlaceDBType db,
IndependentSetMatchingStateType state,
const int* independent_set) {
if (blockIdx.x == 0 && threadIdx.x == 0) {
for (int i = 0; i < state.set_size; ++i) {
int node_id = independent_set[i];
if (node_id < db.num_movable_nodes) {
printf("node %d (%g, %g)", node_id, db.x[node_id], db.y[node_id]);
typename DetailedPlaceDBType::type target_hpwl = 0;
for (int node2pin_id = db.flat_node2pin_start_map[node_id];
node2pin_id < db.flat_node2pin_start_map[node_id + 1];
++node2pin_id) {
int node_pin_id = db.flat_node2pin_map[node2pin_id];
int net_id = db.pin2net_map[node_pin_id];
if (db.net_mask[net_id]) {
{
Box<typename DetailedPlaceDBType::type> box;
box.xl = db.xh;
box.yl = db.yh;
box.xh = db.xl;
box.yh = db.yl;
for (int net2pin_id = db.flat_net2pin_start_map[net_id];
net2pin_id < db.flat_net2pin_start_map[net_id + 1];
++net2pin_id) {
int net_pin_id = db.flat_net2pin_map[net2pin_id];
int other_node_id = db.pin2node_map[net_pin_id];
auto xxl = db.x[other_node_id] + db.pin_offset_x[net_pin_id];
auto yyl = db.y[other_node_id] + db.pin_offset_y[net_pin_id];
box.xl = min(box.xl, xxl);
box.xh = max(box.xh, xxl);
box.yl = min(box.yl, yyl);
box.yh = max(box.yh, yyl);
}
typename DetailedPlaceDBType::type hpwl = box.xh - box.xl + box.yh - box.yl;
target_hpwl += hpwl;
printf(", net %d hpwl %g", net_id, (double)hpwl);
}
{
auto const& box = state.net_boxes[net_id];
typename DetailedPlaceDBType::type xxl = db.x[node_id] + db.pin_offset_x[node_pin_id];
typename DetailedPlaceDBType::type yyl = db.y[node_id] + db.pin_offset_y[node_pin_id];
typename DetailedPlaceDBType::type bxl = min(box.xl, xxl);
typename DetailedPlaceDBType::type bxh = max(box.xh, xxl);
typename DetailedPlaceDBType::type byl = min(box.yl, yyl);
typename DetailedPlaceDBType::type byh = max(box.yh, yyl);
typename DetailedPlaceDBType::type hpwl = (bxh - bxl) + (byh - byl);
printf(" (%g)", hpwl);
}
}
}
printf(", total hpwl %g\n", (double)target_hpwl);
}
}
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
void apply_solution(DetailedPlaceDBType& db, IndependentSetMatchingStateType& state) {
compute_orig_cost(
state.cost_matrices, state.stop_flags, state.num_independent_sets, state.set_size, state.orig_costs);
compute_solution_cost(state.cost_matrices,
state.solutions,
state.stop_flags,
state.num_independent_sets,
state.set_size,
state.solution_costs);
store_orig_pos_kernel<<<state.num_independent_sets, state.set_size>>>(db, state);
move_nodes_kernel<<<state.num_independent_sets, state.set_size>>>(db, state);
checkCuda(cudaMemcpy(&state.num_moved, state.device_num_moved, sizeof(int), cudaMemcpyDeviceToHost));
}
} // namespace dp

View File

@ -0,0 +1,202 @@
#pragma once
#include "gpudp/dp/detailed_place_db.cuh"
namespace dp {
#define BIG_NEGATIVE -9999999
#define MAX_MINIBATCH 64
template <typename T>
inline void init_auction(const int num_graphs, const int num_nodes, char*& scratch, char*& stop_flags) {
checkCuda(cudaMalloc(
&scratch,
num_graphs * (3 * num_nodes + 1) * sizeof(int) + num_graphs * (num_nodes * num_nodes + num_nodes) * sizeof(T)));
allocateCuda(stop_flags, num_graphs, char);
}
inline void destroy_auction(char* scratch, char* stop_flags) {
cudaFree(scratch);
cudaFree(stop_flags);
}
template <typename T>
__global__ void __launch_bounds__(256, 4) linear_assignment_auction_kernel(const int num_nodes,
const T* __restrict__ data_ptr,
int* person2item_ptr,
int* item2person_ptr,
T* bids_ptr,
T* prices_ptr,
int* sbids_ptr,
char* stop_flag_ptr,
const float auction_max_eps,
const float auction_min_eps,
const float auction_factor,
const int max_iterations) {
const int batch_id = blockIdx.x;
const int node_id = threadIdx.x;
__shared__ float auction_eps;
__shared__ int num_iteration;
__shared__ int num_assigned;
extern __shared__ T s_prices[];
if (node_id == 0) {
auction_eps = auction_max_eps;
num_iteration = 0;
}
const T* __restrict__ data = data_ptr + batch_id * num_nodes * num_nodes;
int* person2item = person2item_ptr + batch_id * num_nodes;
int* item2person = item2person_ptr + batch_id * num_nodes;
T* bids = bids_ptr + batch_id * num_nodes * num_nodes;
int* sbids = sbids_ptr + batch_id * num_nodes;
T* prices = prices_ptr + batch_id * num_nodes;
char* stop_flag = stop_flag_ptr + batch_id;
__syncthreads();
while (auction_eps >= auction_min_eps && num_iteration < max_iterations) {
// clear num_assigned
if (node_id == 0) {
num_assigned = 0;
}
// pre-init
for (int i = node_id; i < num_nodes; i += blockDim.x) {
person2item[i] = -1;
item2person[i] = -1;
}
__syncthreads();
// start iterative solving
while (num_assigned < num_nodes && num_iteration < max_iterations) {
// phase 1: init bid and bids
for (int i = node_id; i < num_nodes; i += blockDim.x) {
sbids[i] = 0;
}
for (int i = node_id; i < num_nodes * num_nodes; i += blockDim.x) {
bids[i] = 0;
}
// preload price
s_prices[node_id] = prices[node_id];
__syncthreads();
// phase 2: bidding
if (person2item[node_id] == -1) {
T top1_val = BIG_NEGATIVE;
T top2_val = BIG_NEGATIVE;
int top1_col;
T tmp_val;
for (int col = 0; col < num_nodes; col++) {
tmp_val = data[node_id * num_nodes + col];
if (tmp_val < 0) {
continue;
}
tmp_val = tmp_val - s_prices[col];
if (tmp_val >= top1_val) {
top2_val = top1_val;
top1_col = col;
top1_val = tmp_val;
} else if (tmp_val > top2_val) {
top2_val = tmp_val;
}
}
if (top2_val == BIG_NEGATIVE) {
top2_val = top1_val;
}
T bid = top1_val - top2_val + auction_eps;
bids[num_nodes * top1_col + node_id] = bid;
atomicMax(sbids + top1_col, 1);
}
__syncthreads();
// phase 3 : assignment
if (sbids[node_id] != 0) {
T high_bid = 0;
int high_bidder = -1;
T tmp_bid = -1;
for (int i = 0; i < num_nodes; i++) {
tmp_bid = bids[node_id * num_nodes + i];
if (tmp_bid > high_bid) {
high_bid = tmp_bid;
high_bidder = i;
}
}
int current_person = item2person[node_id];
if (current_person >= 0) {
person2item[current_person] = -1;
} else {
atomicAdd(&num_assigned, 1);
}
prices[node_id] += high_bid;
person2item[high_bidder] = node_id;
item2person[node_id] = high_bidder;
}
__syncthreads();
// update iteration
if (node_id == 0) {
num_iteration++;
}
__syncthreads();
}
// scale auction_eps
if (node_id == 0) {
auction_eps *= auction_factor;
}
__syncthreads();
}
__syncthreads();
// report whether finish solving
if (node_id == 0) {
*stop_flag = (num_assigned == num_nodes);
}
}
template <typename T>
void linear_assignment_auction(const T* cost_matrics,
int* solutions,
const int num_graphs,
const int num_nodes,
char* scratch,
char* stop_flags,
const float auction_max_eps,
const float auction_min_eps,
const float auction_factor,
const int max_iterations) {
// get pointers from scratch, size of scratch: num_graphs * (4*num_nodes + num_nodes*num_nodes) * 4 bytes
int* person2item = (int*)scratch;
int* item2person = person2item + num_graphs * num_nodes;
int* sbids = item2person + num_graphs * num_nodes;
T* prices = (T*)(sbids + num_graphs * num_nodes);
T* bids = prices + num_graphs * num_nodes;
// init
cudaMemsetAsync(prices, 0, num_graphs * num_nodes * sizeof(T));
// launch solver
linear_assignment_auction_kernel<T><<<num_graphs, num_nodes, num_nodes * sizeof(T)>>>(num_nodes,
cost_matrics,
person2item,
item2person,
bids,
prices,
sbids,
stop_flags,
auction_max_eps,
auction_min_eps,
auction_factor,
max_iterations);
cudaDeviceSynchronize();
// copy solutions
cudaMemcpy(solutions, person2item, num_graphs * num_nodes * sizeof(int), cudaMemcpyDeviceToDevice);
}
} // namespace dp

View File

@ -0,0 +1,372 @@
#pragma once
#include "gpudp/dp/detailed_place_db.cuh"
#include "cpu_state.cuh"
namespace dp {
#define DETERMINISTIC
template <typename T>
struct KMeansState {
#ifdef DETERMINISTIC
typedef long long int coordinate_type;
#else
typedef T coordinate_type;
#endif
coordinate_type* centers_x; // To ensure determinism, use fixed point numbers
coordinate_type* centers_y;
T* weights;
int* partition_sizes;
int* node2centers_map;
int num_seeds;
#ifdef DETERMINISTIC
static constexpr T scale = 16384;
#else
static constexpr T scale = 1;
#endif
};
/// @brief A wrapper for atomicAdd
/// As CUDA atomicAdd does not support for long long int, using unsigned long long int is equivalent.
template <typename T>
inline __device__ T atomicAddWrapper(T* address, T value) {
return atomicAdd(address, value);
}
/// @brief Template specialization for long long int
template <>
inline __device__ long long int atomicAddWrapper<long long int>(long long int* address, long long int value) {
return atomicAdd((unsigned long long int*)address, (unsigned long long int)value);
}
template <typename T>
__global__ void fill_array_kernel(T* array, int n, T v) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < n) {
array[i] = v;
}
}
template <typename T>
inline void fill_array(T* array, int n, T v) {
fill_array_kernel<<<ceilDiv(n, 512), 512>>>(array, n, v);
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
void init_kmeans(const DetailedPlaceDBType& db,
const IndependentSetMatchingStateType& state,
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
typedef typename DetailedPlaceDBType::type T;
allocateCuda(kmeans_state.centers_x,
state.batch_size,
typename KMeansState<typename DetailedPlaceDBType::type>::coordinate_type);
allocateCuda(kmeans_state.centers_y,
state.batch_size,
typename KMeansState<typename DetailedPlaceDBType::type>::coordinate_type);
allocateCuda(kmeans_state.weights, state.batch_size, T);
allocateCuda(kmeans_state.partition_sizes, state.batch_size, int);
allocateCuda(kmeans_state.node2centers_map, db.num_movable_nodes, int);
}
template <typename T>
void destroy_kmeans(KMeansState<T>& kmeans_state) {
cudaFree(kmeans_state.centers_x);
cudaFree(kmeans_state.centers_y);
cudaFree(kmeans_state.weights);
cudaFree(kmeans_state.partition_sizes);
cudaFree(kmeans_state.node2centers_map);
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
void prepare_kmeans(const DetailedPlaceDBType& db,
const IndependentSetMatchingStateType& state,
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
// need at least 1 seed; otherwise, it will cause problem in later kernels
kmeans_state.num_seeds = max(min(state.num_selected / state.set_size, state.batch_size), 1);
// set weights to 1.0
fill_array(kmeans_state.weights, kmeans_state.num_seeds, (typename DetailedPlaceDBType::type)1.0);
}
template <typename T>
__inline__ __device__ T kmeans_distance(T node_x, T node_y, T center_x, T center_y) {
T distance = fabs(node_x - center_x) + fabs(node_y - center_y);
return distance;
}
template <typename T>
struct ItemWithIndex {
T value;
int index;
};
template <typename T>
struct ReduceMinOP {
__host__ __device__ ItemWithIndex<T> operator()(const ItemWithIndex<T>& a, const ItemWithIndex<T>& b) const {
return (a.value < b.value) ? a : b;
}
};
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType, int ThreadsPerBlock = 128>
__global__ void kmeans_find_centers_kernel(DetailedPlaceDBType db,
IndependentSetMatchingStateType state,
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
assert(blockIdx.x < state.num_selected);
int node_id = state.selected_maximal_independent_set[blockIdx.x];
assert(node_id < db.num_movable_nodes);
auto node_x = db.x[node_id];
auto node_y = db.y[node_id];
typedef cub::BlockReduce<ItemWithIndex<typename DetailedPlaceDBType::type>, ThreadsPerBlock> BlockReduce;
__shared__ typename BlockReduce::TempStorage temp_storage;
ItemWithIndex<typename DetailedPlaceDBType::type> thread_data;
thread_data.value = cuda::numeric_limits<typename DetailedPlaceDBType::type>::max();
thread_data.index = cuda::numeric_limits<int>::max();
for (int center_id = threadIdx.x; center_id < kmeans_state.num_seeds; center_id += ThreadsPerBlock) {
assert(center_id < kmeans_state.num_seeds);
// scale back to floating point numbers
typename DetailedPlaceDBType::type center_x =
kmeans_state.centers_x[center_id] / KMeansState<typename DetailedPlaceDBType::type>::scale;
typename DetailedPlaceDBType::type center_y =
kmeans_state.centers_y[center_id] / KMeansState<typename DetailedPlaceDBType::type>::scale;
typename DetailedPlaceDBType::type weight = kmeans_state.weights[center_id];
typename DetailedPlaceDBType::type distance = kmeans_distance(node_x, node_y, center_x, center_y) * weight;
if (distance < thread_data.value) {
thread_data.value = distance;
thread_data.index = center_id;
}
}
if (threadIdx.x < kmeans_state.num_seeds) {
assert(thread_data.index < kmeans_state.num_seeds);
}
__syncthreads();
// Compute the block-wide max for thread0
ItemWithIndex<typename DetailedPlaceDBType::type> aggregate =
BlockReduce(temp_storage)
.Reduce(thread_data, ReduceMinOP<typename DetailedPlaceDBType::type>(), kmeans_state.num_seeds);
__syncthreads();
if (threadIdx.x == 0) {
assert(blockIdx.x < state.num_selected);
kmeans_state.node2centers_map[blockIdx.x] = aggregate.index;
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void init_kmeans_seeds_kernel(DetailedPlaceDBType db,
IndependentSetMatchingStateType state,
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < kmeans_state.num_seeds) {
assert(db.num_movable_nodes - i - 1 < db.num_movable_nodes && db.num_movable_nodes - i - 1 >= 0);
int random_number = state.ordered_nodes[db.num_movable_nodes - i - 1];
random_number = random_number % state.num_selected;
int node_id = state.selected_maximal_independent_set[random_number];
assert(node_id < db.num_movable_nodes);
// scale up for fixed point numbers
kmeans_state.centers_x[i] = db.x[node_id] * KMeansState<typename DetailedPlaceDBType::type>::scale;
kmeans_state.centers_y[i] = db.y[node_id] * KMeansState<typename DetailedPlaceDBType::type>::scale;
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
void init_kmeans_seeds(const DetailedPlaceDBType& db,
IndependentSetMatchingStateType& state,
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
init_kmeans_seeds_kernel<<<ceilDiv(kmeans_state.num_seeds, 256), 256>>>(db, state, kmeans_state);
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void reset_kmeans_partition_sizes_kernel(DetailedPlaceDBType db,
IndependentSetMatchingStateType state,
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < kmeans_state.num_seeds) {
kmeans_state.partition_sizes[i] = 0;
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void compute_kmeans_partition_sizes_kernel(DetailedPlaceDBType db,
IndependentSetMatchingStateType state,
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < state.num_selected) {
int center_id = kmeans_state.node2centers_map[i];
assert(center_id < kmeans_state.num_seeds);
atomicAdd(kmeans_state.partition_sizes + center_id, 1);
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void reset_kmeans_centers_kernel(DetailedPlaceDBType db,
IndependentSetMatchingStateType state,
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < kmeans_state.num_seeds) {
if (kmeans_state.partition_sizes[i]) {
kmeans_state.centers_x[i] = 0;
kmeans_state.centers_y[i] = 0;
}
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void compute_kmeans_centers_sum_kernel(DetailedPlaceDBType db,
IndependentSetMatchingStateType state,
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < state.num_selected) {
int node_id = state.selected_maximal_independent_set[i];
int center_id = kmeans_state.node2centers_map[i];
assert(center_id < kmeans_state.num_seeds);
assert(node_id < db.num_movable_nodes);
// scale up for fixed point numbers
atomicAddWrapper<typename KMeansState<typename DetailedPlaceDBType::type>::coordinate_type>(
kmeans_state.centers_x + center_id, db.x[node_id] * KMeansState<typename DetailedPlaceDBType::type>::scale);
atomicAddWrapper<typename KMeansState<typename DetailedPlaceDBType::type>::coordinate_type>(
kmeans_state.centers_y + center_id, db.y[node_id] * KMeansState<typename DetailedPlaceDBType::type>::scale);
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void compute_kmeans_centers_div_kernel(DetailedPlaceDBType db,
IndependentSetMatchingStateType state,
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < kmeans_state.num_seeds) {
int s = kmeans_state.partition_sizes[i];
if (s) {
kmeans_state.centers_x[i] /= s;
kmeans_state.centers_y[i] /= s;
}
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
void kmeans_update_centers(const DetailedPlaceDBType& db,
IndependentSetMatchingStateType& state,
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
// reset partition_sizes to 0
reset_kmeans_partition_sizes_kernel<<<ceilDiv(kmeans_state.num_seeds, 256), 256>>>(db, state, kmeans_state);
// compute partition sizes
compute_kmeans_partition_sizes_kernel<<<ceilDiv(state.num_selected, 256), 256>>>(db, state, kmeans_state);
// reset kmeans centers to 0
reset_kmeans_centers_kernel<<<ceilDiv(kmeans_state.num_seeds, 256), 256>>>(db, state, kmeans_state);
// compute kmeans centers sum
compute_kmeans_centers_sum_kernel<<<ceilDiv(state.num_selected, 256), 256>>>(db, state, kmeans_state);
// compute kmeans centers div
compute_kmeans_centers_div_kernel<<<ceilDiv(kmeans_state.num_seeds, 256), 256>>>(db, state, kmeans_state);
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void compute_kmeans_weights_kernel(DetailedPlaceDBType db,
IndependentSetMatchingStateType state,
KMeansState<typename DetailedPlaceDBType::type> kmeans_state) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < kmeans_state.num_seeds) {
int s = kmeans_state.partition_sizes[i];
auto& w = kmeans_state.weights[i];
if (s > state.set_size) {
auto ratio = s / (typename DetailedPlaceDBType::type)state.set_size;
ratio = 1.0 + 0.5 * log(ratio);
w *= ratio;
}
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
void kmeans_update_weights(const DetailedPlaceDBType& db,
IndependentSetMatchingStateType& state,
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
compute_kmeans_weights_kernel<<<ceilDiv(kmeans_state.num_seeds, 256), 256>>>(db, state, kmeans_state);
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
void kmeans_collect_sets_cuda2cpu(const DetailedPlaceDBType& db,
IndependentSetMatchingStateType& state,
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
std::vector<int> selected_nodes(state.num_selected);
checkCuda(cudaMemcpy(selected_nodes.data(),
state.selected_maximal_independent_set,
sizeof(int) * state.num_selected,
cudaMemcpyDeviceToHost));
std::vector<int> node2centers_map(state.num_selected);
checkCuda(cudaMemcpy(node2centers_map.data(),
kmeans_state.node2centers_map,
sizeof(int) * state.num_selected,
cudaMemcpyDeviceToHost));
std::vector<int> flat_independent_sets(state.batch_size * state.set_size, std::numeric_limits<int>::max());
std::vector<int> independent_set_sizes(state.batch_size, 0);
// directly use flat array
for (int i = 0; i < state.num_selected; ++i) {
int node_id = selected_nodes.at(i);
int center_id = node2centers_map.at(i);
int& size = independent_set_sizes.at(center_id);
if (size < state.set_size) {
flat_independent_sets.at(center_id * state.set_size + size) = node_id;
++size;
}
}
checkCuda(cudaMemcpy(state.independent_sets,
flat_independent_sets.data(),
sizeof(int) * state.batch_size * state.set_size,
cudaMemcpyHostToDevice));
checkCuda(cudaMemcpy(state.independent_set_sizes,
independent_set_sizes.data(),
sizeof(int) * state.batch_size,
cudaMemcpyHostToDevice));
// statistics
logger.debug(
"from %d nodes, collect %d sets, avg %d nodes, min/max %d/%d nodes",
state.num_selected,
state.num_independent_sets,
std::accumulate(independent_set_sizes.begin(), independent_set_sizes.begin() + state.num_independent_sets, 0) /
state.num_independent_sets,
*std::min_element(independent_set_sizes.begin(), independent_set_sizes.begin() + state.num_independent_sets),
*std::max_element(independent_set_sizes.begin(), independent_set_sizes.begin() + state.num_independent_sets));
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
void partition_kmeans(const DetailedPlaceDBType& db,
IndependentSetMatchingStateType& state,
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state) {
prepare_kmeans(db, state, kmeans_state);
init_kmeans_seeds(db, state, kmeans_state);
for (int iter = 0; iter < 2; ++iter) {
// for each node, find centers
kmeans_find_centers_kernel<DetailedPlaceDBType, IndependentSetMatchingStateType, 256>
<<<state.num_selected, 256>>>(db, state, kmeans_state);
// for each center, adjust itself
kmeans_update_centers(db, state, kmeans_state);
// for each partition, update weight
kmeans_update_weights(db, state, kmeans_state);
}
state.num_independent_sets = kmeans_state.num_seeds;
kmeans_collect_sets_cuda2cpu(db, state, kmeans_state);
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
void collect_independent_sets(const DetailedPlaceDBType& db,
IndependentSetMatchingStateType& state,
KMeansState<typename DetailedPlaceDBType::type>& kmeans_state,
DetailedPlaceCPUDB<typename DetailedPlaceDBType::type>& host_db,
IndependentSetMatchingCPUState<typename DetailedPlaceDBType::type>& host_state) {
partition_kmeans(db, state, kmeans_state);
}
} // namespace dp

View File

@ -0,0 +1,231 @@
#pragma once
#include "adjust_pos.cuh"
#include "gpudp/dp/detailed_place_db.cuh"
#include "reduce_min.cuh"
namespace dp {
#define MAX_NODE_DEGREE 32
template <typename T>
struct SharedBox {
T xl;
T yl;
T xh;
T yh;
};
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void print_net_boxes_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
if (blockIdx.x == 0 && threadIdx.x == 0) {
for (int node_id = 0; node_id < db.num_movable_nodes; ++node_id) {
if (state.selected_markers[node_id]) {
int node2pin_id = db.flat_node2pin_start_map[node_id];
const int node2pin_id_end = db.flat_node2pin_start_map[node_id + 1];
for (; node2pin_id < node2pin_id_end; ++node2pin_id) {
int node_pin_id = db.flat_node2pin_map[node2pin_id];
int net_id = db.pin2net_map[node_pin_id];
auto const& box = state.net_boxes[net_id];
printf("node %d: net %d (%g, %g, %g, %g)\n", node_id, net_id, box.xl, box.yl, box.xh, box.yh);
}
}
}
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void compute_cost_matrix_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
int i = blockIdx.y; // set
int j = blockIdx.x; // node in set
const int* __restrict__ independent_set = state.independent_sets + i * state.set_size;
auto cost_matrix = state.cost_matrices + i * state.cost_matrix_size + j * state.set_size;
__shared__ int node_id;
__shared__ typename DetailedPlaceDBType::type node_width;
__shared__ SharedBox<typename DetailedPlaceDBType::type> net_boxes[MAX_NODE_DEGREE];
__shared__ int node2pin_id_bgn;
__shared__ int node2pin_id_end;
if (threadIdx.x == 0) {
node_id = independent_set[j];
node_width = cuda::numeric_limits<typename DetailedPlaceDBType::type>::max();
if (node_id < db.num_movable_nodes) {
node_width = db.node_size_x[node_id];
node2pin_id_bgn = db.flat_node2pin_start_map[node_id];
node2pin_id_end = db.flat_node2pin_start_map[node_id + 1];
node2pin_id_end = min(node2pin_id_bgn + MAX_NODE_DEGREE, node2pin_id_end);
int idx = 0;
for (int node2pin_id = node2pin_id_bgn; node2pin_id < node2pin_id_end; ++node2pin_id, ++idx) {
int node_pin_id = db.flat_node2pin_map[node2pin_id];
int net_id = db.pin2net_map[node_pin_id];
auto& box = net_boxes[idx];
box.xl = db.xh;
box.yl = db.yh;
box.xh = db.xl;
box.yh = db.yl;
if (db.net_mask[net_id]) {
int net2pin_id_bgn = db.flat_net2pin_start_map[net_id];
int net2pin_id_end = db.flat_net2pin_start_map[net_id + 1];
for (int net2pin_id = net2pin_id_bgn; net2pin_id < net2pin_id_end; ++net2pin_id) {
int net_pin_id = db.flat_net2pin_map[net2pin_id];
int other_node_id = db.pin2node_map[net_pin_id];
if (other_node_id != node_id) {
typename DetailedPlaceDBType::type xxl = db.x[other_node_id] + db.pin_offset_x[net_pin_id];
typename DetailedPlaceDBType::type yyl = db.y[other_node_id] + db.pin_offset_y[net_pin_id];
box.xl = min(box.xl, xxl);
box.xh = max(box.xh, xxl);
box.yl = min(box.yl, yyl);
box.yh = max(box.yh, yyl);
}
}
}
}
}
}
__syncthreads();
for (int k = threadIdx.x; k < state.set_size; k += blockDim.x) // pos in set
{
int pos_id = independent_set[k];
auto& cost = cost_matrix[k]; // row major
if (node_id < db.num_movable_nodes && pos_id < db.num_movable_nodes) {
typename DetailedPlaceDBType::type target_x = db.x[pos_id];
typename DetailedPlaceDBType::type target_y = db.y[pos_id];
auto const& target_space = state.spaces[pos_id];
int target_hpwl = 0;
if (adjust_pos(target_x, node_width, target_space)) {
// consider FENCE region
if (db.num_regions && !db.inside_fence(node_id, target_x, target_y)) {
cost = BIG_NEGATIVE; // as a marker for post processing
} else {
int idx = 0;
for (int node2pin_id = node2pin_id_bgn; node2pin_id < node2pin_id_end; ++node2pin_id, ++idx) {
int node_pin_id = db.flat_node2pin_map[node2pin_id];
int net_id = db.pin2net_map[node_pin_id];
auto const& box = net_boxes[idx];
if (db.net_mask[net_id]) {
typename DetailedPlaceDBType::type xxl = target_x + db.pin_offset_x[node_pin_id];
typename DetailedPlaceDBType::type yyl = target_y + db.pin_offset_y[node_pin_id];
typename DetailedPlaceDBType::type bxl = min(box.xl, xxl);
typename DetailedPlaceDBType::type bxh = max(box.xh, xxl);
typename DetailedPlaceDBType::type byl = min(box.yl, yyl);
typename DetailedPlaceDBType::type byh = max(box.yh, yyl);
target_hpwl += (bxh - bxl) + (byh - byl);
}
}
// target_hpwl = target_hpwl*db.row_height + (abs(target_x-node_x) + abs(target_y-node_y));
// row major
cost = target_hpwl;
}
} else {
cost = BIG_NEGATIVE; // as a marker for post processing
}
} else {
// cost = state.large_number*(j != k);
cost = BIG_NEGATIVE; // as a marker for post processing
}
}
}
/// @brief change from minimization problem for maximization problem with non-negative edge weights
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void postprocess_cost_matrix_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
int i = blockIdx.y; // set
int j = blockIdx.x; // node in set
const int* __restrict__ independent_set = state.independent_sets + i * state.set_size;
auto cost_matrix = state.cost_matrices + i * state.cost_matrix_size + j * state.set_size;
auto max_cost = state.cost_matrices_copy[i * state.cost_matrix_size];
for (int k = threadIdx.x; k < state.set_size; k += blockDim.x) // pos in set
{
int node_id = independent_set[j];
int pos_id = independent_set[k];
auto& cost = cost_matrix[k]; // row major
if (node_id < db.num_movable_nodes && pos_id < db.num_movable_nodes) {
if (cost >= 0) {
cost = max_cost - cost;
}
// cost < 0 is already assigned to negative
} else if (j == k) {
cost = max_cost; // dummy cells or positions
}
// j != k is already assigned to negative
}
}
template <typename T>
__global__ void print_cost_matrix_kernel(const T* cost_matrix, int set_size) {
unsigned int tid = threadIdx.x;
unsigned int bid = blockIdx.x;
if (tid == 0 && bid == 0) {
printf("[%dx%d]\n", set_size, set_size);
for (int r = 0; r < set_size; ++r) {
for (int c = 0; c < set_size; ++c) {
auto cost = cost_matrix[r * set_size + c];
if (cost == BIG_NEGATIVE) {
printf("X ");
} else {
printf("%g ", (double)cost);
}
}
printf("\n");
}
printf("\n");
}
}
template <typename IndependentSetMatchingStateType>
__global__ void print_max_cost_kernel(IndependentSetMatchingStateType state) {
unsigned int tid = threadIdx.x;
unsigned int bid = blockIdx.x;
if (tid == 0 && bid == 0) {
printf("[%d]\n", state.num_independent_sets);
for (int i = 0; i < state.num_independent_sets; ++i) {
printf("%g ", (double)state.cost_matrices_copy[i * state.cost_matrix_size]);
}
printf("\n");
}
}
template <typename IndependentSetMatchingStateType>
__global__ void check_cost_matrices_kernel(IndependentSetMatchingStateType state) {
unsigned int tid = threadIdx.x;
unsigned int bid = blockIdx.x;
if (tid == 0 && bid == 0) {
for (int i = 0; i < state.num_independent_sets; ++i) {
for (int j = 0; j < state.cost_matrix_size; ++j) {
auto cost = state.cost_matrices[i * state.cost_matrix_size + j];
assert(cost == cuda::numeric_limits<typename IndependentSetMatchingStateType::cost_type>::lowest() ||
cost >= 0);
}
}
}
}
template <typename T>
struct CompareCost {
__host__ __device__ bool operator()(T cost1, T cost2) const { return cost1 > cost2; }
};
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
void cost_matrix_construction(const DetailedPlaceDBType& db, IndependentSetMatchingStateType& state) {
dim3 grid(state.set_size, state.num_independent_sets, 1);
compute_cost_matrix_kernel<<<grid, state.set_size>>>(db, state);
checkCuda(cudaMemcpy(state.cost_matrices_copy,
state.cost_matrices,
sizeof(typename IndependentSetMatchingStateType::cost_type) * state.num_independent_sets *
state.cost_matrix_size,
cudaMemcpyDeviceToDevice));
typename IndependentSetMatchingStateType::cost_type ref = 0;
reduce_2d(state.cost_matrices_copy,
state.num_independent_sets,
state.cost_matrix_size,
ref,
CompareCost<typename IndependentSetMatchingStateType::cost_type>());
postprocess_cost_matrix_kernel<<<grid, state.set_size>>>(db, state);
}
} // namespace dp

View File

@ -0,0 +1,92 @@
#pragma once
#include "gpudp/dp/detailed_place_db.cuh"
#include "diamond_search.h"
namespace dp {
template <typename T>
struct DetailedPlaceCPUDB {
typedef T type;
int num_movable_nodes;
int num_bins_x;
int num_bins_y;
T bin_size_x;
T bin_size_y;
T xl, yl, xh, yh;
std::vector<T> node_size_x;
std::vector<T> node_size_y;
std::vector<T> x;
std::vector<T> y;
};
template <typename T>
struct IndependentSetMatchingCPUState {
typedef T type;
int batch_size;
int set_size;
int grid_size;
int max_diamond_search_sequence;
int num_independent_sets;
std::vector<std::vector<int>> independent_sets;
std::vector<int> flat_independent_sets; ///< flat version of storage
std::vector<int> independent_set_sizes; ///< size of each set
std::vector<int> selected_nodes;
std::vector<unsigned char> selected_markers;
std::vector<int> ordered_nodes;
std::vector<BinMapIndex> node2bin_map;
std::vector<std::vector<int>>
bin2node_map; ///< the first dimension is size, all the cells are categorized by width
std::vector<GridIndex<int>> search_grids;
};
inline int ceil_power2(int v) { return (1 << (int)ceil(log2((float)v))); }
template <typename DetailedPlaceDBType>
void init_cpu_db(const DetailedPlaceDBType& db, DetailedPlaceCPUDB<typename DetailedPlaceDBType::type>& host_db) {
host_db.num_movable_nodes = db.num_movable_nodes;
host_db.num_bins_x = db.num_bins_x;
host_db.num_bins_y = db.num_bins_y;
host_db.bin_size_x = db.bin_size_x;
host_db.bin_size_y = db.bin_size_y;
host_db.xl = db.xl;
host_db.yl = db.yl;
host_db.xh = db.xh;
host_db.yh = db.yh;
host_db.node_size_x.resize(db.num_nodes);
checkCuda(cudaMemcpy(host_db.node_size_x.data(),
db.node_size_x,
sizeof(typename DetailedPlaceDBType::type) * db.num_nodes,
cudaMemcpyDeviceToHost));
host_db.node_size_y.resize(db.num_nodes);
checkCuda(cudaMemcpy(host_db.node_size_y.data(),
db.node_size_y,
sizeof(typename DetailedPlaceDBType::type) * db.num_nodes,
cudaMemcpyDeviceToHost));
host_db.x.resize(db.num_nodes);
host_db.y.resize(db.num_nodes);
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
void init_cpu_state(const DetailedPlaceDBType& db,
const IndependentSetMatchingStateType& state,
IndependentSetMatchingCPUState<typename DetailedPlaceDBType::type>& host_state) {
host_state.batch_size = state.batch_size;
host_state.set_size = state.set_size;
host_state.grid_size = ceil_power2(std::max(db.num_bins_x, db.num_bins_y) / 8);
host_state.max_diamond_search_sequence = host_state.grid_size * host_state.grid_size / 2;
logger.info("diamond search grid size %d, sequence length %d",
host_state.grid_size,
host_state.max_diamond_search_sequence);
host_state.selected_nodes.reserve(db.num_movable_nodes);
host_state.selected_markers.assign(db.num_movable_nodes, 1);
host_state.ordered_nodes.resize(db.num_movable_nodes);
host_state.search_grids = diamond_search_sequence(host_state.grid_size, host_state.grid_size);
host_state.independent_sets.resize(state.batch_size, std::vector<int>(state.set_size));
host_state.flat_independent_sets.resize(state.batch_size * state.set_size);
host_state.independent_set_sizes.resize(state.batch_size);
host_state.node2bin_map.resize(db.num_movable_nodes);
host_state.bin2node_map.resize(db.num_bins_x * db.num_bins_y);
}
} // namespace dp

View File

@ -0,0 +1,121 @@
#pragma once
#include <algorithm>
#include <cmath>
#include <cstdio>
#include <cstdlib>
#include <limits>
#include <numeric>
#include <type_traits>
#include <vector>
namespace dp {
/// @brief grid index
template <typename T>
struct GridIndex {
T ir; ///< row index
T ic; ///< column index
GridIndex() : ir(std::numeric_limits<T>::max()), ic(std::numeric_limits<T>::max()) {}
GridIndex(T r, T c) : ir(r), ic(c) {}
T manhattan_distance(const GridIndex& rhs) const { return fabs(ir - rhs.ir) + fabs(ic - rhs.ic); }
double angle(const GridIndex& rhs) const { return atan2(ic - rhs.ic, ir - rhs.ir); }
};
/// @brief compare grid (row index, column index) by its manhattan distance to a
/// target grid
template <typename T>
struct CompareGridByDistance2Target {
GridIndex<T> target;
CompareGridByDistance2Target(const GridIndex<T>& g) : target(g) {}
bool operator()(const GridIndex<T>& g1, const GridIndex<T>& g2) const {
T d1 = g1.manhattan_distance(target);
double angle1 = g1.angle(target);
T d2 = g2.manhattan_distance(target);
double angle2 = g2.angle(target);
return d1 < d2 || (d1 == d2 && (angle1 < angle2));
}
};
/// @brief kernel to generate the sequence for diamond search
/// @tparam the template must be a signed integer
/// @param num_rows number of rows
/// @param num_cols number of columns
/// @return the sequence in order from small distance to the center grid to
/// large
template <typename T>
std::vector<GridIndex<T> > diamond_search_sequence_kernel(T num_rows, T num_cols) {
//// 2D grid map in row major
//// each element is the (row index, column index)
// std::vector<GridIndex<T> > grid_map (num_rows*num_cols, GridIndex<T>(0,
// 0)); for (T ir = 0; ir < num_rows; ++ir)
//{
// for (T ic = 0; ic < num_cols; ++ic)
// {
// grid_map[ir*num_cols+ic] = GridIndex<T>(-(T)num_rows/2+ir,
// -(T)num_cols/2+ic);
// }
//}
//// sort from small distance to large
// std::sort(grid_map.begin(), grid_map.end(),
// CompareGridByDistance2Target<T>(GridIndex<T>(0, 0)));
// directly generate diamond shape grids
// in clock-wise direction
// the sequence covers the following shape
// 1
// 111
// 11111
// 111
// 1
std::vector<GridIndex<T> > grid_map;
grid_map.reserve(num_rows * num_cols / 2);
T max_sum = std::min(num_rows, num_cols) / 2;
grid_map.push_back(GridIndex<T>(0, 0));
for (T sum = 1; sum <= max_sum; ++sum) {
// y > 0, x [-sum, sum]
for (T ir = -sum; ir < sum; ++ir) {
grid_map.push_back(GridIndex<T>(ir, sum - std::abs(ir)));
}
// y < 0, x [sum, -sum]
for (T ir = sum; ir > -sum; --ir) {
grid_map.push_back(GridIndex<T>(ir, -(sum - std::abs(ir))));
}
}
return grid_map;
}
/// @brief top API to generate the sequence for diamond search
/// @param num_rows number of rows
/// @param num_cols number of columns
/// @return the sequence in order from small distance to the center grid to
/// large
template <typename T>
std::vector<GridIndex<typename std::make_signed<T>::type> > diamond_search_sequence(T num_rows, T num_cols) {
return diamond_search_sequence_kernel<typename std::make_signed<T>::type>(num_rows, num_cols);
}
template <typename T>
void diamond_search_print(const std::vector<GridIndex<T> >& grid_sequence) {
unsigned int sum = 0;
unsigned int count = 0;
GridIndex<T> target(0, 0);
printf("[0] ");
for (typename std::vector<GridIndex<T> >::const_iterator it = grid_sequence.begin(); it != grid_sequence.end();
++it, ++count) {
T distance = it->manhattan_distance(target);
if (sum != distance) {
sum = distance;
printf("\n[%u] ", count);
}
printf("(%d,%d) ", it->ir, it->ic);
}
printf("\n");
}
} // namespace dp

View File

@ -0,0 +1,179 @@
#pragma once
#include "gpudp/dp/detailed_place_db.cuh"
namespace dp {
__global__ void collect_kernel(const int* d_flags, int* d_sums, int* d_results, const int length) {
const int tid = blockIdx.x * blockDim.x + threadIdx.x;
if (d_flags[tid] == 1 && tid < length) {
d_results[d_sums[tid]] = tid;
}
}
template <typename T, typename V>
__global__ void select_kernel_add(const T* a, const V* b, int* c) {
if (blockIdx.x == 0 && threadIdx.x == 0) {
*c = (int)(*a) + (int)(*b);
}
}
void select(const int* d_flags, int* d_results, const int length, int* scratch, int* num_collected) {
size_t temp_storage_bytes = 0;
void* d_temp_storage = NULL; // need this NULL pointer to get temp_storage_bytes
int* prefix_sum = scratch;
checkCuda(cub::DeviceScan::ExclusiveSum(d_temp_storage, temp_storage_bytes, d_flags, prefix_sum, length));
// Run exclusive prefix sum
checkCuda(cub::DeviceScan::ExclusiveSum((void*)d_results, temp_storage_bytes, d_flags, prefix_sum, length));
// cudaDeviceSynchronize();
select_kernel_add<<<1, 1>>>(prefix_sum + (length - 1), d_flags + (length - 1), num_collected);
collect_kernel<<<(length + 256 - 1) / 256, 256>>>(d_flags, prefix_sum, d_results, length);
cudaDeviceSynchronize();
}
/// @brief for each node, check its first level neighbors, if they are selected, mark itself as dependent
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__device__ void mark_dependent_nodes_self(const DetailedPlaceDBType& db,
IndependentSetMatchingStateType& state,
int node_id) {
if (state.selected_markers[node_id]) {
state.dependent_markers[node_id] = 1;
return;
}
typename DetailedPlaceDBType::type node_xl = db.x[node_id];
typename DetailedPlaceDBType::type node_yl = db.y[node_id];
// in case all nets are masked
int node2pin_start = db.flat_node2pin_start_map[node_id];
int node2pin_end = db.flat_node2pin_start_map[node_id + 1];
for (int node2pin_id = node2pin_start; node2pin_id < node2pin_end; ++node2pin_id) {
int node_pin_id = db.flat_node2pin_map[node2pin_id];
int net_id = db.pin2net_map[node_pin_id];
if (db.net_mask[net_id]) {
int net2pin_start = db.flat_net2pin_start_map[net_id];
int net2pin_end = db.flat_net2pin_start_map[net_id + 1];
for (int net2pin_id = net2pin_start; net2pin_id < net2pin_end; ++net2pin_id) {
int net_pin_id = db.flat_net2pin_map[net2pin_id];
int other_node_id = db.pin2node_map[net_pin_id];
typename DetailedPlaceDBType::type other_node_xl = db.x[other_node_id];
typename DetailedPlaceDBType::type other_node_yl = db.y[other_node_id];
if (std::abs(node_xl - other_node_xl) + std::abs(node_yl - other_node_yl) < state.skip_threshold) {
if (other_node_id < db.num_movable_nodes && state.selected_markers[other_node_id]) {
state.dependent_markers[node_id] = 1;
return;
}
}
}
}
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void maximal_independent_set_kernel(DetailedPlaceDBType db,
IndependentSetMatchingStateType state,
int* empty) {
const int from = blockIdx.x * blockDim.x + threadIdx.x;
const int incr = gridDim.x * blockDim.x;
// do
//{
// empty = true;
for (int node_id = from; node_id < db.num_movable_nodes; node_id += incr) {
if (!state.dependent_markers[node_id]) {
if (*empty) {
atomicExch(empty, false);
}
// empty = false;
bool min_node_flag = true;
{
typename DetailedPlaceDBType::type node_xl = db.x[node_id];
typename DetailedPlaceDBType::type node_yl = db.y[node_id];
int node_rank = state.ordered_nodes[node_id];
// in case all nets are masked
int node2pin_start = db.flat_node2pin_start_map[node_id];
int node2pin_end = db.flat_node2pin_start_map[node_id + 1];
for (int node2pin_id = node2pin_start; node2pin_id < node2pin_end; ++node2pin_id) {
int node_pin_id = db.flat_node2pin_map[node2pin_id];
int net_id = db.pin2net_map[node_pin_id];
if (db.net_mask[net_id]) {
int net2pin_start = db.flat_net2pin_start_map[net_id];
int net2pin_end = db.flat_net2pin_start_map[net_id + 1];
for (int net2pin_id = net2pin_start; net2pin_id < net2pin_end; ++net2pin_id) {
int net_pin_id = db.flat_net2pin_map[net2pin_id];
int other_node_id = db.pin2node_map[net_pin_id];
typename DetailedPlaceDBType::type other_node_xl = db.x[other_node_id];
typename DetailedPlaceDBType::type other_node_yl = db.y[other_node_id];
typename DetailedPlaceDBType::type distance =
abs(node_xl - other_node_xl) + abs(node_yl - other_node_yl);
if (other_node_id < db.num_movable_nodes && (distance < state.skip_threshold) &&
(state.selected_markers[other_node_id] ||
(state.dependent_markers[other_node_id] == 0 &&
state.ordered_nodes[other_node_id] < node_rank))) {
min_node_flag = false;
break;
}
}
if (!min_node_flag) {
break;
}
}
}
}
if (min_node_flag) {
state.selected_markers[node_id] = 1;
}
}
}
//} while (!empty);
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void mark_dependent_nodes_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
for (int node_id = blockIdx.x * blockDim.x + threadIdx.x; node_id < db.num_movable_nodes;
node_id += blockDim.x * gridDim.x) {
if (!state.dependent_markers[node_id]) {
mark_dependent_nodes_self(db, state, node_id);
}
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
__global__ void init_markers_kernel(DetailedPlaceDBType db, IndependentSetMatchingStateType state) {
for (int i = blockIdx.x * blockDim.x + threadIdx.x; i < db.num_nodes; i += blockDim.x * gridDim.x) {
state.selected_markers[i] = 0;
// make sure multi-row height cells are not selected
state.dependent_markers[i] = (db.node_size_y[i] > db.row_height);
}
}
template <typename DetailedPlaceDBType, typename IndependentSetMatchingStateType>
void maximal_independent_set(DetailedPlaceDBType const& db, IndependentSetMatchingStateType& state) {
// if dependent_markers is 1, it means "cannot be selected"
// if selected_markers is 1, it means "already selected"
init_markers_kernel<<<ceilDiv(db.num_nodes, 256), 256>>>(db, state);
int host_empty;
int iteration = 0;
do {
host_empty = true;
checkCuda(cudaMemcpy(state.independent_set_empty_flag, &host_empty, sizeof(int), cudaMemcpyHostToDevice));
maximal_independent_set_kernel<<<ceilDiv(db.num_movable_nodes, 256), 256>>>(
db, state, state.independent_set_empty_flag);
mark_dependent_nodes_kernel<<<ceilDiv(db.num_movable_nodes, 256), 256>>>(db, state);
checkCuda(cudaMemcpy(&host_empty, state.independent_set_empty_flag, sizeof(int), cudaMemcpyDeviceToHost));
++iteration;
} while (!host_empty && iteration < 10);
select(state.selected_markers,
state.selected_maximal_independent_set,
db.num_movable_nodes,
state.select_scratch,
state.device_num_selected);
checkCuda(cudaMemcpy(&state.num_selected, state.device_num_selected, sizeof(int), cudaMemcpyDeviceToHost));
}
} // namespace dp

View File

@ -0,0 +1,155 @@
#pragma once
#include "gpudp/dp/utils.cuh"
namespace dp {
template <typename T, typename V>
__device__ void warpReduce(/*volatile*/ T* sdata, int tid, V comp) {
sdata[tid] = sdata[tid + 32 * !comp(sdata[tid], sdata[tid + 32])];
sdata[tid] = sdata[tid + 16 * !comp(sdata[tid], sdata[tid + 16])];
sdata[tid] = sdata[tid + 8 * !comp(sdata[tid], sdata[tid + 8])];
sdata[tid] = sdata[tid + 4 * !comp(sdata[tid], sdata[tid + 4])];
sdata[tid] = sdata[tid + 2 * !comp(sdata[tid], sdata[tid + 2])];
sdata[tid] = sdata[tid + 1 * !comp(sdata[tid], sdata[tid + 1])];
}
/**
* 优化:解决了 reduce3 中存在的多余同步操作(每个warp默认自动同步)。
* globalInputData 输入数据,位于全局内存
* globalOutputData 输出数据,位于全局内存
* n length of array
* ref reference value
* comp compare function which returns the target element
*/
template <typename T, typename V, unsigned int BlockSize = 256>
__global__ void reduce4(T* globalInputData, T* globalOutputData, int n, T ref, V comp) {
__shared__ T sdata[BlockSize];
// 坐标索引
int tid = threadIdx.x;
int index = blockIdx.x * (blockDim.x * 2) + threadIdx.x;
int indexWithOffset = index + blockDim.x;
if (index >= n)
sdata[tid] = ref;
else if (indexWithOffset >= n)
sdata[tid] = globalInputData[index];
else {
// printf("tid = %d, index = %d, indexWithOffset = %d, index+blockDim.x*!comp(globalInputData[index],
// globalInputData[indexWithOffset]) = %d\n",
// tid, index, indexWithOffset, index+blockDim.x*!comp(globalInputData[index],
// globalInputData[indexWithOffset])
// );
sdata[tid] = (comp(globalInputData[index], globalInputData[indexWithOffset]))
? globalInputData[index]
: globalInputData[indexWithOffset];
}
__syncthreads();
// 在共享内存中对每一个块进行规约计算
for (int s = blockDim.x / 2; s > 0; s >>= 1) {
if (tid < s) {
sdata[tid] = sdata[tid + s * !comp(sdata[tid], sdata[tid + s])];
}
__syncthreads();
}
// if (tid < 32)
//{
// warpReduce(sdata, tid, comp);
//}
// 把计算结果从共享内存写回全局内存
if (tid == 0) {
globalOutputData[blockIdx.x] = sdata[0];
}
}
template <typename T, typename V, unsigned int BlockSize = 256>
void reduce(T* fMatrix_Device, int iMatrixSize, const T& ref, const V& comp) {
for (int i = 1, iNum = iMatrixSize; i < iMatrixSize; i = 2 * i * BlockSize) {
int iBlockNum = (iNum + (2 * BlockSize) - 1) / (2 * BlockSize);
reduce4<T, V, BlockSize><<<iBlockNum, BlockSize>>>(fMatrix_Device, fMatrix_Device, iNum, ref, comp);
iNum = iBlockNum;
}
}
template <typename T, typename V, unsigned int BlockSize = 256>
void reduce(T* fMatrix_Device, int iMatrixSize, const T& ref, const V& comp, cudaStream_t& stream) {
for (int i = 1, iNum = iMatrixSize; i < iMatrixSize; i = 2 * i * BlockSize) {
int iBlockNum = (iNum + (2 * BlockSize) - 1) / (2 * BlockSize);
reduce4<T, V, BlockSize><<<iBlockNum, BlockSize, 0, stream>>>(fMatrix_Device, fMatrix_Device, iNum, ref, comp);
iNum = iBlockNum;
}
}
/**
* improvement: resolved redundant synchronization in reduce3, i.e., synchronize each warp
* globalInputData input data, located in global memory
* globalOutputData output data, located in global memory
* nc number of initial columns before reduction
* n number of columns
* ref reference value
* comp compare function which returns the target element
*/
template <typename T, typename V, unsigned int BlockSize = 256>
__global__ void reduce4_2d(T* globalInputData, T* globalOutputData, int nc, int n, T ref, V comp) {
__shared__ T sdata[BlockSize];
// compute indices
int tid = threadIdx.x;
int yOffset = blockIdx.y * nc;
int index = yOffset + blockIdx.x * (blockDim.x * 2) + threadIdx.x;
int indexWithOffset = index + blockDim.x;
if (index >= yOffset + n)
sdata[tid] = ref;
else if (indexWithOffset >= yOffset + n)
sdata[tid] = globalInputData[index];
else {
// printf("tid = %d, index = %d, indexWithOffset = %d, index+blockDim.x*!comp(globalInputData[index],
// globalInputData[indexWithOffset]) = %d\n",
// tid, index, indexWithOffset, index+blockDim.x*!comp(globalInputData[index],
// globalInputData[indexWithOffset])
// );
sdata[tid] = (comp(globalInputData[index], globalInputData[indexWithOffset]))
? globalInputData[index]
: globalInputData[indexWithOffset];
}
__syncthreads();
// reduction for data in shared memory
for (int s = blockDim.x / 2; s > 0; s >>= 1) {
if (tid < s) {
sdata[tid] = sdata[tid + s * !comp(sdata[tid], sdata[tid + s])];
}
__syncthreads();
}
// if (tid < 32)
//{
// warpReduce(sdata, tid, comp);
//}
// write data back from shared memory to global memory
if (tid == 0) {
globalOutputData[yOffset + blockIdx.x] = sdata[0];
// printf("globalOutputData[%d] = %g\n", blockIdx.y*nc + blockIdx.x, sdata[0].cost);
}
}
template <typename T, typename V, unsigned int BlockSize = 256>
void reduce_2d(T* fMatrix_Device, int m, int n, const T& ref, const V& comp) {
for (int i = 1, iNum = n; i < n; i = 2 * i * BlockSize) {
int iBlockNum = (iNum + (2 * BlockSize) - 1) / (2 * BlockSize);
dim3 grid(iBlockNum, m, 1);
reduce4_2d<T, V, BlockSize><<<grid, BlockSize>>>(fMatrix_Device, fMatrix_Device, n, iNum, ref, comp);
iNum = iBlockNum;
}
}
} // namespace dp

View File

@ -0,0 +1,108 @@
#pragma once
#include <curand.h>
#include <curand_kernel.h>
#include "gpudp/dp/utils.cuh"
namespace dp {
template <typename T, typename V>
__global__ void print_shuffle(const T* values, const V* keys, int n) {
if (blockIdx.x == 0 && threadIdx.x == 0) {
printf("values[%d]\n", n);
for (int i = 0; i < n; ++i) {
printf("%d ", int(values[i]));
}
printf("\n");
printf("keys[%d]\n", n);
for (int i = 0; i < n; ++i) {
printf("%d ", int(keys[i]));
}
printf("\n");
}
}
/// @brief A shuffler that can be repeatedly called.
/// @tparam T value type
/// @tparam V key type
template <typename T, typename V>
class Shuffler {
public:
/// @brief constructor
/// @param seed random seed
/// @param values data array that will be manipulated
/// @param n length of array
Shuffler(size_t seed, T* values, int n) {
/* Create pseudo-random number generator */
checkCurand(curandCreateGenerator(&m_gen, CURAND_RNG_PSEUDO_DEFAULT));
/* Set seed */
checkCurand(curandSetPseudoRandomGeneratorSeed(m_gen, seed));
m_values_in = values;
allocateCuda(m_keys_in, n, V);
allocateCuda(m_keys_out, n, V);
allocateCuda(m_values_out, n, T);
m_temp_storage = NULL;
m_temp_storage_bytes = 0;
m_num_items = n;
}
/// @brief destructor
~Shuffler() {
if (m_temp_storage) {
cudaFree(m_temp_storage);
}
cudaFree(m_keys_in);
cudaFree(m_keys_out);
cudaFree(m_values_out);
checkCurand(curandDestroyGenerator(m_gen));
}
/// @brief top API to shuffle data. It can be called repeatedly.
void operator()() {
/* Generate n floats on device */
checkCurand(curandGenerate(m_gen, m_keys_in, m_num_items));
// Determine temporary device storage requirements
void* d_temp_storage = NULL;
size_t temp_storage_bytes = 0;
cub::DeviceRadixSort::SortPairs(
d_temp_storage, temp_storage_bytes, m_keys_in, m_keys_out, m_values_in, m_values_out, m_num_items);
// Allocate temporary storage
// re-allocate if different size
if (m_temp_storage_bytes != temp_storage_bytes) {
if (m_temp_storage_bytes) {
cudaFree(m_temp_storage);
m_temp_storage = NULL;
}
m_temp_storage_bytes = temp_storage_bytes;
logger.debug("allocate %lu bytes in shuffler for length %d*(%d+%d)",
m_temp_storage_bytes,
m_num_items,
sizeof(T),
sizeof(V));
checkCuda(cudaMalloc(&m_temp_storage, m_temp_storage_bytes));
}
// Run sorting operation
cub::DeviceRadixSort::SortPairs(
m_temp_storage, m_temp_storage_bytes, m_keys_in, m_keys_out, m_values_in, m_values_out, m_num_items);
// copy back to m_values_in, not necessary
// As m_values_in corresponds to external data, copying back the output can allow in-place manipulation
checkCuda(cudaMemcpy(m_values_in, m_values_out, sizeof(T) * m_num_items, cudaMemcpyDeviceToDevice));
}
protected:
curandGenerator_t m_gen; ///< random number generator
V* m_keys_in; ///< on device, to store real key
T* m_values_in; ///< on device, to store real data
V* m_keys_out; ///< on device, a buffer
T* m_values_out; ///< on device, a buffer
void* m_temp_storage; ///< on device, temporary storage for sorting
size_t m_temp_storage_bytes; ///< number of bytes for m_temp_storage
int m_num_items; ///< length of array
};
} // namespace dp

File diff suppressed because it is too large Load Diff

View File

@ -0,0 +1,108 @@
#pragma once
#include <cuda.h>
#include <cuda_runtime.h>
#include "utils.cuh"
template <typename T>
struct PitchNestedVector {
T* flat_element_map; ///< allocate on device, length of size1*size2
unsigned int* dim2_sizes; ///< sizes of dimension 2
unsigned int size1; ///< length in dimension 1
unsigned int size2; ///< maximum length in dimension 2
unsigned int num_elements; ///< total number of elements
/// @brief constructor
__host__ PitchNestedVector() : flat_element_map(nullptr), dim2_sizes(nullptr), size1(0), size2(0) {}
/// @brief initialization
__host__ void initialize(const std::vector<std::vector<T> >& nested_map) {
// construct flat map on host
unsigned int max_num_elements = 0;
num_elements = 0;
for (typename std::vector<std::vector<T> >::const_iterator it = nested_map.begin(); it != nested_map.end();
++it) {
max_num_elements = max(max_num_elements, (unsigned int)it->size());
num_elements += it->size();
}
std::vector<T> host_flat_element_map(nested_map.size() * max_num_elements, std::numeric_limits<T>::max());
std::vector<unsigned int> host_dim2_sizes(nested_map.size());
for (unsigned int i = 0; i < nested_map.size(); ++i) {
const std::vector<T>& vec = nested_map[i];
std::copy(vec.begin(), vec.end(), host_flat_element_map.begin() + max_num_elements * i);
host_dim2_sizes[i] = vec.size();
}
// copy to device
size1 = nested_map.size();
size2 = max_num_elements;
allocateCopyCuda(flat_element_map, host_flat_element_map.data(), host_flat_element_map.size());
allocateCopyCuda(dim2_sizes, host_dim2_sizes.data(), host_dim2_sizes.size());
}
__host__ void destroy() {
if (flat_element_map) {
cudaFree(flat_element_map);
cudaFree(dim2_sizes);
}
}
/// @brief access element
inline __device__ const T& operator()(unsigned int i, unsigned int j) const {
#ifdef DEBUG
if (!(i < size1 && j < size(i))) {
printf("%u < %u && %u < %u\n", i, size1, j, size(i));
}
#endif
assert(i < size1 && j < size(i));
return flat_element_map[i * size2 + j];
}
/// @brief access element
inline __device__ T& operator()(unsigned int i, unsigned int j) {
#ifdef DEBUG
if (!(i < size1 && j < size(i))) {
printf("%u < %u && %u < %u\n", i, size1, j, size(i));
}
#endif
assert(i < size1 && j < size(i));
return flat_element_map[i * size2 + j];
}
/// @brief access each row
inline __device__ const T* operator()(unsigned int i) const {
#ifdef DEBUG
if (!(i < size1)) {
printf("%u < %u\n", i, size1);
}
#endif
assert(i < size1);
return flat_element_map + i * size2;
}
/// @brief access each row
inline __device__ T* operator()(unsigned int i) {
#ifdef DEBUG
if (!(i < size1)) {
printf("%u < %u\n", i, size1);
}
#endif
assert(i < size1);
return flat_element_map + i * size2;
}
/// @brief length of each row
inline __device__ unsigned int size(unsigned int i) const {
#ifdef DEBUG
if (!(i < size1)) {
printf("%u < %u\n", i, size1);
}
#endif
assert(i < size1);
return dim2_sizes[i];
}
/// @brief total number of elements
inline __device__ unsigned int size() const { return num_elements; }
};

View File

@ -0,0 +1,279 @@
#pragma once
#include <cuda.h>
#include <cuda_runtime.h>
#include <float.h>
#include <limits.h>
#include "cub/cub.cuh"
#define checkCuda(expression) \
{ \
cudaError_t status = (expression); \
if (status != cudaSuccess) { \
printf("CUDA Runtime Error: %s at %s:%d\n", cudaGetErrorString(expression), __FILE__, __LINE__); \
std::exit(EXIT_FAILURE); \
} \
}
#define checkCurand(expression) \
{ \
curandStatus_t status = (expression); \
if (status != CURAND_STATUS_SUCCESS) { \
printf("Curand Error at %s:%d\n", __FILE__, __LINE__); \
std::exit(EXIT_FAILURE); \
} \
}
#define allocateCuda(var, size, type) \
{ \
cudaError_t status = cudaMalloc(&(var), (size) * sizeof(type)); \
if (status != cudaSuccess) { \
printf("cudaMalloc failed for " #var " at %s:%d\n", __FILE__, __LINE__); \
} \
}
#define allocateCopyCuda(var, rhs, size) \
{ \
allocateCuda(var, size, decltype(*rhs)); \
checkCuda(cudaMemcpy(var, rhs, sizeof(decltype(*rhs)) * (size), cudaMemcpyHostToDevice)); \
}
#define allocateCopyCpu(var, rhs, size, T) \
{ \
var = (T*)malloc(sizeof(T) * (size)); \
checkCuda(cudaMemcpy((void*)var, (void*)rhs, sizeof(T) * (size), cudaMemcpyDeviceToHost)); \
}
// For cuda::numeric_limits
namespace cuda { // namespace cuda
template <typename T>
struct numeric_limits_base {
typedef T type;
};
template <typename T>
struct numeric_limits : public numeric_limits_base<T> {};
template <>
struct numeric_limits<char> : public numeric_limits_base<char> {
/** The minimum finite value, or for floating types with
denormalization, the minimum positive normalized value. */
__host__ __device__ static constexpr type min() noexcept { return CHAR_MIN; }
/** The maximum finite value. */
__host__ __device__ static constexpr type max() noexcept { return CHAR_MAX; }
/** A finite value x such that there is no other finite value y
* where y < x. */
__host__ __device__ static constexpr type lowest() noexcept { return CHAR_MIN; }
/** A the machine epsilon. */
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
};
template <>
struct numeric_limits<unsigned char> : public numeric_limits_base<unsigned char> {
/** The minimum finite value, or for floating types with
denormalization, the minimum positive normalized value. */
__host__ __device__ static constexpr type min() noexcept { return 0; }
/** The maximum finite value. */
__host__ __device__ static constexpr type max() noexcept { return UCHAR_MAX; }
/** A finite value x such that there is no other finite value y
* where y < x. */
__host__ __device__ static constexpr type lowest() noexcept { return 0; }
/** A the machine epsilon. */
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
};
template <>
struct numeric_limits<short> : public numeric_limits_base<short> {
/** The minimum finite value, or for floating types with
denormalization, the minimum positive normalized value. */
__host__ __device__ static constexpr type min() noexcept { return SHRT_MIN; }
/** The maximum finite value. */
__host__ __device__ static constexpr type max() noexcept { return SHRT_MAX; }
/** A finite value x such that there is no other finite value y
* where y < x. */
__host__ __device__ static constexpr type lowest() noexcept { return SHRT_MIN; }
/** A the machine epsilon. */
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
};
template <>
struct numeric_limits<unsigned short> : public numeric_limits_base<unsigned short> {
/** The minimum finite value, or for floating types with
denormalization, the minimum positive normalized value. */
__host__ __device__ static constexpr type min() noexcept { return 0; }
/** The maximum finite value. */
__host__ __device__ static constexpr type max() noexcept { return USHRT_MAX; }
/** A finite value x such that there is no other finite value y
* where y < x. */
__host__ __device__ static constexpr type lowest() noexcept { return 0; }
/** A the machine epsilon. */
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
};
template <>
struct numeric_limits<int> : public numeric_limits_base<int> {
/** The minimum finite value, or for floating types with
denormalization, the minimum positive normalized value. */
__host__ __device__ static constexpr type min() noexcept { return INT_MIN; }
/** The maximum finite value. */
__host__ __device__ static constexpr type max() noexcept { return INT_MAX; }
/** A finite value x such that there is no other finite value y
* where y < x. */
__host__ __device__ static constexpr type lowest() noexcept { return INT_MIN; }
/** A the machine epsilon. */
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
};
template <>
struct numeric_limits<unsigned int> : public numeric_limits_base<unsigned int> {
/** The minimum finite value, or for floating types with
denormalization, the minimum positive normalized value. */
__host__ __device__ static constexpr type min() noexcept { return 0; }
/** The maximum finite value. */
__host__ __device__ static constexpr type max() noexcept { return UINT_MAX; }
/** A finite value x such that there is no other finite value y
* where y < x. */
__host__ __device__ static constexpr type lowest() noexcept { return 0; }
/** A the machine epsilon. */
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
};
template <>
struct numeric_limits<long> : public numeric_limits_base<long> {
/** The minimum finite value, or for floating types with
denormalization, the minimum positive normalized value. */
__host__ __device__ static constexpr type min() noexcept { return LONG_MIN; }
/** The maximum finite value. */
__host__ __device__ static constexpr type max() noexcept { return LONG_MAX; }
/** A finite value x such that there is no other finite value y
* where y < x. */
__host__ __device__ static constexpr type lowest() noexcept { return LONG_MIN; }
/** A the machine epsilon. */
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
};
template <>
struct numeric_limits<unsigned long> : public numeric_limits_base<unsigned long> {
/** The minimum finite value, or for floating types with
denormalization, the minimum positive normalized value. */
__host__ __device__ static constexpr type min() noexcept { return 0; }
/** The maximum finite value. */
__host__ __device__ static constexpr type max() noexcept { return ULONG_MAX; }
/** A finite value x such that there is no other finite value y
* where y < x. */
__host__ __device__ static constexpr type lowest() noexcept { return 0; }
/** A the machine epsilon. */
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
};
template <>
struct numeric_limits<long long> : public numeric_limits_base<long long> {
/** The minimum finite value, or for floating types with
denormalization, the minimum positive normalized value. */
__host__ __device__ static constexpr type min() noexcept { return LLONG_MIN; }
/** The maximum finite value. */
__host__ __device__ static constexpr type max() noexcept { return LLONG_MAX; }
/** A finite value x such that there is no other finite value y
* where y < x. */
__host__ __device__ static constexpr type lowest() noexcept { return LLONG_MIN; }
/** A the machine epsilon. */
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
};
template <>
struct numeric_limits<unsigned long long> : public numeric_limits_base<unsigned long long> {
/** The minimum finite value, or for floating types with
denormalization, the minimum positive normalized value. */
__host__ __device__ static constexpr type min() noexcept { return 0; }
/** The maximum finite value. */
__host__ __device__ static constexpr type max() noexcept { return ULLONG_MAX; }
/** A finite value x such that there is no other finite value y
* where y < x. */
__host__ __device__ static constexpr type lowest() noexcept { return 0; }
/** A the machine epsilon. */
__host__ __device__ static constexpr type epsilon() noexcept { return 0; }
};
template <>
struct numeric_limits<float> : public numeric_limits_base<float> {
/** The minimum finite value, or for floating types with
denormalization, the minimum positive normalized value. */
__host__ __device__ static constexpr type min() noexcept { return FLT_MIN; }
/** The maximum finite value. */
__host__ __device__ static constexpr type max() noexcept { return FLT_MAX; }
/** A finite value x such that there is no other finite value y
* where y < x. */
__host__ __device__ static constexpr type lowest() noexcept { return -FLT_MAX; }
/** A the machine epsilon. */
__host__ __device__ static constexpr type epsilon() noexcept { return FLT_EPSILON; }
};
template <>
struct numeric_limits<double> : public numeric_limits_base<double> {
/** The minimum finite value, or for floating types with
denormalization, the minimum positive normalized value. */
__host__ __device__ static constexpr type min() noexcept { return DBL_MIN; }
/** The maximum finite value. */
__host__ __device__ static constexpr type max() noexcept { return DBL_MAX; }
/** A finite value x such that there is no other finite value y
* where y < x. */
__host__ __device__ static constexpr type lowest() noexcept { return -DBL_MAX; }
/** A the machine epsilon. */
__host__ __device__ static constexpr type epsilon() noexcept { return DBL_EPSILON; }
};
template <>
struct numeric_limits<long double> : public numeric_limits_base<long double> {
/** The minimum finite value, or for floating types with
denormalization, the minimum positive normalized value. */
__host__ __device__ static constexpr type min() noexcept { return LDBL_MIN; }
/** The maximum finite value. */
__host__ __device__ static constexpr type max() noexcept { return LDBL_MAX; }
/** A finite value x such that there is no other finite value y
* where y < x. */
__host__ __device__ static constexpr type lowest() noexcept { return -LDBL_MAX; }
/** A the machine epsilon. */
__host__ __device__ static constexpr type epsilon() noexcept { return LDBL_EPSILON; }
};
} // namespace cuda

View File

@ -0,0 +1,429 @@
#include "gpudp/lg/legalization_db.h"
namespace dp {
struct AbacusCluster {
int prev_cluster_id; ///< previous cluster, set to INT_MIN if the cluster is
///< invalid
int next_cluster_id; ///< next cluster, set to INT_MIN if the cluster is
///< invalid
int bgn_row_node_id; ///< id of first node in the row
int end_row_node_id; ///< id of last node in the row
float e; ///< weight of displacement in the objective
float q; ///< x = q/e
float w; ///< width
float x; ///< optimal location
/// @return whether this is a valid cluster
bool valid() const { return prev_cluster_id != INT_MIN && next_cluster_id != INT_MIN; }
};
/// @brief helper function for distributing cells to rows
/// sort cells within a row and clean overlapping fixed cells
void sortNodesInRow(const float* host_x,
const float* host_y,
const float* host_node_size_x,
const float* host_node_size_y,
int num_movable_nodes,
std::vector<int>& nodes_in_row) {
// sort cells within rows according to left edges
std::sort(nodes_in_row.begin(), nodes_in_row.end(), [&](int node_id1, int node_id2) {
float x1 = host_x[node_id1];
float x2 = host_x[node_id2];
// put larger width front will help remove
// overlapping fixed cells, especially when
// x1 == x2, then we need the wider one comes first
float w1 = host_node_size_x[node_id1];
float w2 = host_node_size_x[node_id2];
return x1 < x2 || (x1 == x2 && (w1 > w2 || (w1 == w2 && node_id1 < node_id2)));
});
// After sorting by left edge,
// there is a special case for fixed cells where
// one fixed cell is completely within another in a row.
// This will cause failure to detect some overlaps.
// We need to remove the "small" fixed cell that is inside another.
if (!nodes_in_row.empty()) {
std::vector<int> tmp_nodes;
tmp_nodes.reserve(nodes_in_row.size());
tmp_nodes.push_back(nodes_in_row.front());
int j_1 = 0;
for (int j = 1, je = nodes_in_row.size(); j < je; ++j) {
int node_id1 = nodes_in_row.at(j_1);
int node_id2 = nodes_in_row.at(j);
// two fixed cells
if (node_id1 >= num_movable_nodes && node_id2 >= num_movable_nodes) {
float xl1 = host_x[node_id1];
float xl2 = host_x[node_id2];
float width1 = host_node_size_x[node_id1];
float width2 = host_node_size_x[node_id2];
float xh1 = xl1 + width1;
float xh2 = xl2 + width2;
// only collect node_id2 if its right edge is righter than node_id1
if (xh1 < xh2) {
tmp_nodes.push_back(node_id2);
j_1 = j;
}
} else {
tmp_nodes.push_back(node_id2);
j_1 = j;
}
}
nodes_in_row.swap(tmp_nodes);
// sort according to center
std::sort(nodes_in_row.begin(), nodes_in_row.end(), [&](int node_id1, int node_id2) {
float x1 = host_x[node_id1] + host_node_size_x[node_id1] / 2;
float x2 = host_x[node_id2] + host_node_size_x[node_id2] / 2;
return x1 < x2 || (x1 == x2 && node_id1 < node_id2);
});
for (int j = 1, je = nodes_in_row.size(); j < je; ++j) {
int node_id1 = nodes_in_row.at(j - 1);
int node_id2 = nodes_in_row.at(j);
float xl1 = host_x[node_id1];
float xl2 = host_x[node_id2];
float width1 = host_node_size_x[node_id1];
float width2 = host_node_size_x[node_id2];
float xh1 = xl1 + width1;
float xh2 = xl2 + width2;
float yl1 = host_y[node_id1];
float yl2 = host_y[node_id2];
float yh1 = yl1 + host_node_size_y[node_id1];
float yh2 = yl2 + host_node_size_y[node_id2];
assert_msg(xl1 < xl2 && xh1 < xh2,
"node %d (%g, %g, %g, %g) overlaps with node %d (%g, %g, %g, %g)",
node_id1,
xl1,
yl1,
xh1,
yh1,
node_id2,
xl2,
yl2,
xh2,
yh2);
}
}
}
void distributeMovableAndFixedCells2Bins(const float* x,
const float* y,
const float* node_size_x,
const float* node_size_y,
float bin_size_x,
float bin_size_y,
float xl,
float yl,
float xh,
float yh,
float site_width,
int num_bins_x,
int num_bins_y,
int num_nodes,
int num_movable_nodes,
std::vector<std::vector<int>>& bin_cells) {
for (int i = 0; i < num_nodes; i += 1) {
if (i < num_movable_nodes && roundDiv(node_size_y[i], bin_size_y) <= 1) {
// single-row movable nodes only distribute to one bin
int bin_id_x = (x[i] + node_size_x[i] / 2 - xl) / bin_size_x;
int bin_id_y = (y[i] + node_size_y[i] / 2 - yl) / bin_size_y;
bin_id_x = std::min(std::max(bin_id_x, 0), num_bins_x - 1);
bin_id_y = std::min(std::max(bin_id_y, 0), num_bins_y - 1);
int bin_id = bin_id_x * num_bins_y + bin_id_y;
bin_cells[bin_id].push_back(i);
} else {
// fixed nodes may distribute to multiple bins
int node_id = i;
int bin_id_xl = std::max((x[node_id] - xl) / bin_size_x, (float)0);
int bin_id_xh = std::min((int)ceil((x[node_id] + node_size_x[node_id] - xl) / bin_size_x), num_bins_x);
int bin_id_yl = std::max((y[node_id] - yl) / bin_size_y, (float)0);
int bin_id_yh = std::min((int)ceil((y[node_id] + node_size_y[node_id] - yl) / bin_size_y), num_bins_y);
for (int bin_id_x = bin_id_xl; bin_id_x < bin_id_xh; ++bin_id_x) {
for (int bin_id_y = bin_id_yl; bin_id_y < bin_id_yh; ++bin_id_y) {
int bin_id = bin_id_x * num_bins_y + bin_id_y;
bin_cells[bin_id].push_back(node_id);
}
}
}
}
}
/// @param row_nodes node indices in this row
/// @param clusters pre-allocated clusters in this row with the same length as
/// that of row_nodes
/// @param num_row_nodes length of row_nodes
/// @return true if succeed, otherwise false
bool abacusPlaceRowCPU(const float* init_x,
const float* node_size_x,
const float* node_size_y,
float* x,
float row_height,
float xl,
float xh,
int num_nodes,
int num_movable_nodes,
int* row_nodes,
AbacusCluster* clusters,
int num_row_nodes) {
// a very large number
float M = std::pow(10, ceilDiv(std::log((xh - xl) * num_row_nodes), log(10)));
bool ret_flag = true;
// merge two clusters
// the second cluster will be invalid
auto merge_cluster = [&](int dst_cluster_id, int src_cluster_id) {
assert(dst_cluster_id < num_row_nodes);
AbacusCluster& dst_cluster = clusters[dst_cluster_id];
assert(src_cluster_id < num_row_nodes);
AbacusCluster& src_cluster = clusters[src_cluster_id];
assert(dst_cluster.valid() && src_cluster.valid());
for (int i = dst_cluster_id + 1; i < src_cluster_id; ++i) {
assert(!clusters[i].valid());
}
dst_cluster.end_row_node_id = src_cluster.end_row_node_id;
assert(dst_cluster.e < M && src_cluster.e < M);
dst_cluster.e += src_cluster.e;
dst_cluster.q += src_cluster.q - src_cluster.e * dst_cluster.w;
dst_cluster.w += src_cluster.w;
// update linked list
if (src_cluster.next_cluster_id < num_row_nodes) {
clusters[src_cluster.next_cluster_id].prev_cluster_id = dst_cluster_id;
}
dst_cluster.next_cluster_id = src_cluster.next_cluster_id;
src_cluster.prev_cluster_id = std::numeric_limits<int>::min();
src_cluster.next_cluster_id = std::numeric_limits<int>::min();
};
// collapse clusters between [0, cluster_id]
// compute the locations and merge clusters
auto collapse = [&](int cluster_id, float range_xl, float range_xh) {
int cur_cluster_id = cluster_id;
assert(cur_cluster_id < num_row_nodes);
int prev_cluster_id = clusters[cur_cluster_id].prev_cluster_id;
AbacusCluster* cluster = nullptr;
AbacusCluster* prev_cluster = nullptr;
while (true) {
assert(cur_cluster_id < num_row_nodes);
cluster = &clusters[cur_cluster_id];
cluster->x = cluster->q / cluster->e;
// make sure cluster >= range_xl, so fixed nodes will not be moved
// in illegal case, cluster+w > range_xh may occur, but it is OK.
// We can collect failed clusters later
cluster->x = std::max(std::min(cluster->x, range_xh - cluster->w), range_xl);
assert(cluster->x >= range_xl && cluster->x + cluster->w <= range_xh);
prev_cluster_id = cluster->prev_cluster_id;
if (prev_cluster_id >= 0) {
prev_cluster = &clusters[prev_cluster_id];
if (prev_cluster->x + prev_cluster->w > cluster->x) {
merge_cluster(prev_cluster_id, cur_cluster_id);
cur_cluster_id = prev_cluster_id;
} else {
break;
}
} else {
break;
}
}
};
// initial cluster has only one cell
for (int i = 0; i < num_row_nodes; ++i) {
int node_id = row_nodes[i];
AbacusCluster& cluster = clusters[i];
cluster.prev_cluster_id = i - 1;
cluster.next_cluster_id = i + 1;
cluster.bgn_row_node_id = i;
cluster.end_row_node_id = i;
cluster.e = (node_id < num_movable_nodes && node_size_y[node_id] <= row_height) ? 1.0 : M;
cluster.q = cluster.e * init_x[node_id];
cluster.w = node_size_x[node_id];
// this is required since we also include fixed nodes
cluster.x = (node_id < num_movable_nodes && node_size_y[node_id] > row_height) ? x[node_id] : init_x[node_id];
}
// kernel algorithm for placeRow
float range_xl = xl;
float range_xh = xh;
for (int j = 0; j < num_row_nodes; ++j) {
const AbacusCluster& next_cluster = clusters[j];
if (next_cluster.e >= M) // fixed node
{
range_xh = std::min(next_cluster.x, range_xh);
break;
} else {
assert(std::abs(node_size_y[row_nodes[j]] - row_height) < 1e-6);
}
}
for (int i = 0; i < num_row_nodes; ++i) {
const AbacusCluster& cluster = clusters[i];
if (cluster.e < M) {
assert(std::abs(node_size_y[row_nodes[i]] - row_height) < 1e-6);
collapse(i, range_xl, range_xh);
} else // set range xl/xh according to fixed nodes
{
range_xl = cluster.x + cluster.w;
range_xh = xh;
for (int j = i + 1; j < num_row_nodes; ++j) {
const AbacusCluster& next_cluster = clusters[j];
if (next_cluster.e >= M) // fixed node
{
range_xh = std::min(next_cluster.x, range_xh);
break;
}
}
}
}
// apply solution
for (int i = 0; i < num_row_nodes; ++i) {
if (clusters[i].valid()) {
const AbacusCluster& cluster = clusters[i];
float xc = cluster.x;
for (int j = cluster.bgn_row_node_id; j <= cluster.end_row_node_id; ++j) {
int node_id = row_nodes[j];
if (node_id < num_movable_nodes && std::abs(node_size_y[node_id] - row_height) < 1e-6) {
x[node_id] = xc;
} else if (xc != x[node_id]) {
if (node_id < num_movable_nodes)
logger.warning(
"multi-row node %d tends to move from %.12f to "
"%.12f, ignored",
node_id,
x[node_id],
xc);
else
logger.warning(
"fixed node %d tends to move from %.12f to %.12f, ignored", node_id, x[node_id], xc);
ret_flag = false;
}
xc += node_size_x[node_id];
}
}
}
return ret_flag;
}
void abacusLegalizeRow(const float* init_x,
const float* node_size_x,
const float* node_size_y,
float* x,
float* y,
float xl,
float xh,
float bin_size_x,
float bin_size_y,
int num_bins_x,
int num_bins_y,
int num_nodes,
int num_movable_nodes,
std::vector<std::vector<int>>& bin_cells,
std::vector<std::vector<AbacusCluster>>& bin_clusters) {
for (unsigned int i = 0; i < bin_cells.size(); i += 1) {
auto& row2nodes = bin_cells.at(i);
// sort bin cells from left to right
sortNodesInRow(x, y, node_size_x, node_size_y, num_movable_nodes, row2nodes);
auto& clusters = bin_clusters.at(i);
int num_row_nodes = row2nodes.size();
int bin_id_x = i / num_bins_y;
// int bin_id_y = i-bin_id_x*num_bins_y;
float bin_xl = xl + bin_size_x * bin_id_x;
float bin_xh = std::min(bin_xl + bin_size_x, xh);
abacusPlaceRowCPU(init_x,
node_size_x,
node_size_y,
x,
bin_size_y, // must be equal to row_height
bin_xl,
bin_xh,
num_nodes,
num_movable_nodes,
row2nodes.data(),
clusters.data(),
num_row_nodes);
}
float displace = 0;
for (int i = 0; i < num_movable_nodes; ++i) {
displace += fabs(x[i] - init_x[i]);
}
logger.debug("average displace = %g", displace / num_movable_nodes);
}
void abacusLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
LegalizationData db(at_db);
db.set_num_bins(num_bins_x, num_bins_y);
// adjust bin sizes
float bin_size_x = (db.xh - db.xl) / num_bins_x;
float bin_size_y = db.row_height;
num_bins_y = ceilDiv(db.yh - db.yl, bin_size_y);
// include both movable and fixed nodes
std::vector<std::vector<int>> bin_cells(num_bins_x * num_bins_y);
// distribute cells to bins
distributeMovableAndFixedCells2Bins(db.x,
db.y,
db.node_size_x,
db.node_size_y,
bin_size_x,
bin_size_y,
db.xl,
db.yl,
db.xh,
db.yh,
db.site_width,
num_bins_x,
num_bins_y,
db.num_nodes,
db.num_movable_nodes,
bin_cells);
std::vector<std::vector<AbacusCluster>> bin_clusters(num_bins_x * num_bins_y);
for (unsigned int i = 0; i < bin_cells.size(); ++i) {
bin_clusters[i].resize(bin_cells[i].size());
}
abacusLegalizeRow(db.init_x,
db.node_size_x,
db.node_size_y,
db.x,
db.y,
db.xl,
db.xh,
bin_size_x,
bin_size_y,
num_bins_x,
num_bins_y,
db.num_nodes,
db.num_movable_nodes,
bin_cells,
bin_clusters);
// need to align nodes to sites
// this also considers cell width which is not integral times of site_width
for (auto const& cells : bin_cells) {
float xxl = db.xl;
for (auto node_id : cells) {
if (node_id < db.num_movable_nodes) {
db.x[node_id] = std::max(std::min(db.x[node_id], db.xh - db.node_size_x[node_id]), xxl);
db.x[node_id] = floorDiv(db.x[node_id] - db.xl, db.site_width) * db.site_width + db.xl;
xxl = db.x[node_id] + db.node_size_x[node_id];
} else if (node_id < db.num_nodes) {
xxl = ceilDiv(db.x[node_id] + db.node_size_x[node_id] - db.xl, db.site_width) * db.site_width + db.xl;
}
}
}
}
} // namespace dp

View File

@ -0,0 +1,709 @@
#include "gpudp/lg/legalization_db.h"
namespace dp {
template <typename T>
struct Interval {
T xl;
T xh;
Interval(T l, T h) : xl(l), xh(h) {}
void intersect(T rhs_xl, T rhs_xh) {
xl = std::max(xl, rhs_xl);
xh = std::min(xh, rhs_xh);
}
};
template <typename T>
struct Blank {
T xl;
T yl;
T xh;
T yh;
void intersect(const Blank& rhs) {
xl = std::max(xl, rhs.xl);
xh = std::min(xh, rhs.xh);
yl = std::max(yl, rhs.yl);
yh = std::min(yh, rhs.yh);
}
};
void distributeCells2Bins(const LegalizationData& db,
const float* x,
const float* y,
const float* node_size_x,
const float* node_size_y,
float bin_size_x,
float bin_size_y,
float xl,
float yl,
float xh,
float yh,
int num_bins_x,
int num_bins_y,
int num_nodes,
int num_movable_nodes,
std::vector<std::vector<int>>& bin_cells) {
// do not handle large macros
// one cell cannot be distributed to one bin
for (int i = 0; i < num_movable_nodes; i += 1) {
if (!db.is_dummy_fixed(i)) {
int bin_id_x = (x[i] + node_size_x[i] / 2 - xl) / bin_size_x;
int bin_id_y = (y[i] + node_size_y[i] / 2 - yl) / bin_size_y;
bin_id_x = std::min(std::max(bin_id_x, 0), num_bins_x - 1);
bin_id_y = std::min(std::max(bin_id_y, 0), num_bins_y - 1);
int bin_id = bin_id_x * num_bins_y + bin_id_y;
bin_cells[bin_id].push_back(i);
}
}
}
void distributeFixedCells2Bins(const LegalizationData& db,
const float* x,
const float* y,
const float* node_size_x,
const float* node_size_y,
float bin_size_x,
float bin_size_y,
float xl,
float yl,
float xh,
float yh,
int num_bins_x,
int num_bins_y,
int num_nodes,
int num_movable_nodes,
std::vector<std::vector<int>>& bin_cells) {
// one cell can be assigned to multiple bins
for (int i = 0; i < num_nodes; i += 1) {
if (db.is_dummy_fixed(i) || i >= num_movable_nodes) {
int node_id = i;
int bin_id_xl = std::max((int)floorDiv(x[node_id] - xl, bin_size_x), 0);
int bin_id_xh = std::min((int)ceilDiv((x[node_id] + node_size_x[node_id] - xl), bin_size_x), num_bins_x);
int bin_id_yl = std::max((int)floorDiv(y[node_id] - yl, bin_size_y), 0);
int bin_id_yh = std::min((int)ceilDiv((y[node_id] + node_size_y[node_id] - yl), bin_size_y), num_bins_y);
for (int bin_id_x = bin_id_xl; bin_id_x < bin_id_xh; ++bin_id_x) {
for (int bin_id_y = bin_id_yl; bin_id_y < bin_id_yh; ++bin_id_y) {
int bin_id = bin_id_x * num_bins_y + bin_id_y;
bin_cells[bin_id].push_back(node_id);
}
}
}
}
}
void distributeBlanks2Bins(const float* x,
const float* y,
const float* node_size_x,
const float* node_size_y,
const std::vector<std::vector<int>>& bin_fixed_cells,
float bin_size_x,
float bin_size_y,
float blank_bin_size_y,
float xl,
float yl,
float xh,
float yh,
float site_width,
float row_height,
int num_bins_x,
int num_bins_y,
int blank_num_bins_y,
std::vector<std::vector<Blank<float>>>& bin_blanks) {
for (int i = 0; i < num_bins_x * num_bins_y; i += 1) {
int bin_id_x = i / num_bins_y;
int bin_id_y = i - bin_id_x * num_bins_y;
int blank_num_bins_per_bin = roundDiv(bin_size_y, blank_bin_size_y);
int blank_bin_id_yl = bin_id_y * blank_num_bins_per_bin;
int blank_bin_id_yh = std::min(blank_bin_id_yl + blank_num_bins_per_bin, blank_num_bins_y);
for (int blank_bin_id_y = blank_bin_id_yl; blank_bin_id_y < blank_bin_id_yh; ++blank_bin_id_y) {
float bin_xl = xl + bin_id_x * bin_size_x;
float bin_xh = std::min(bin_xl + bin_size_x, xh);
float bin_yl = yl + blank_bin_id_y * blank_bin_size_y;
float bin_yh = std::min(bin_yl + blank_bin_size_y, yh);
int blank_bin_id = bin_id_x * blank_num_bins_y + blank_bin_id_y;
for (float by = bin_yl; by < bin_yh; by += row_height) {
Blank<float> blank;
blank.xl = floorDiv((bin_xl - xl), site_width) * site_width + xl; // align blanks to sites
blank.xh = floorDiv((bin_xh - xl), site_width) * site_width + xl; // align blanks to sites
blank.yl = by;
blank.yh = by + row_height;
bin_blanks.at(blank_bin_id).push_back(blank);
}
const std::vector<int>& cells = bin_fixed_cells.at(i);
std::vector<Blank<float>>& blanks = bin_blanks.at(blank_bin_id);
for (unsigned int bi = 0; bi < blanks.size(); ++bi) {
Blank<float>& blank = blanks.at(bi);
for (unsigned int ci = 0; ci < cells.size(); ++ci) {
int node_id = cells.at(ci);
float node_xl = x[node_id];
float node_yl = y[node_id];
float node_xh = node_xl + node_size_x[node_id];
float node_yh = node_yl + node_size_y[node_id];
if (node_yh > blank.yl && node_yl < blank.yh && node_xh > blank.xl &&
node_xl < blank.xh) // overlap
{
if (node_xl <= blank.xl && node_xh >= blank.xh) // erase
{
bin_blanks.at(blank_bin_id).erase(bin_blanks.at(blank_bin_id).begin() + bi);
--bi;
break;
} else if (node_xl <= blank.xl) { // one blank
blank.xl = ceilDiv((node_xh - xl), site_width) * site_width + xl; // align blanks to sites
} else if (node_xh >= blank.xh) { // one blank
blank.xh = floorDiv((node_xl - xl), site_width) * site_width + xl; // align blanks to sites
} else { // two blanks
Blank<float> new_blank = blank;
blank.xh = floorDiv((node_xl - xl), site_width) * site_width + xl; // align blanks to sites
new_blank.xl =
floorDiv((node_xh - xl), site_width) * site_width + xl; // align blanks to sites
bin_blanks.at(blank_bin_id).insert(bin_blanks.at(blank_bin_id).begin() + bi + 1, new_blank);
--bi;
break;
}
}
}
}
}
}
}
void legalizeBin(
const float* init_x,
const float* init_y,
const float* node_size_x,
const float* node_size_y,
std::vector<std::vector<Blank<float>>>& bin_blanks, // blanks in each bin, sorted from low to high, left to right
std::vector<std::vector<int>>& bin_cells, // unplaced cells in each bin
float* x,
float* y,
int num_bins_x,
int num_bins_y,
int blank_num_bins_y,
float bin_size_x,
float bin_size_y,
float blank_bin_size_y,
float site_width,
float row_height,
float xl,
float yl,
float xh,
float yh,
float alpha, // a parameter to tune anchor initial locations and current locations
float beta, // a parameter to tune space reserving
bool lr_flag, // from left to right
int* num_unplaced_cells) {
for (int i = 0; i < num_bins_x * num_bins_y; i += 1) {
int bin_id_x = i / num_bins_y;
int bin_id_y = i - bin_id_x * num_bins_y;
int blank_num_bins_per_bin = roundDiv(bin_size_y, blank_bin_size_y);
int blank_bin_id_yl = bin_id_y * blank_num_bins_per_bin;
int blank_bin_id_yh = std::min(blank_bin_id_yl + blank_num_bins_per_bin, blank_num_bins_y);
// cells in this bin
std::vector<int>& cells = bin_cells.at(i);
// sort cells according to width
if (lr_flag) {
std::sort(cells.begin(), cells.end(), [&](int i, int j) -> bool {
float wi = -1000 * (init_x[i] + node_size_x[i] / 2) + node_size_x[i] + node_size_y[i];
float wj = -1000 * (init_x[j] + node_size_x[j] / 2) + node_size_x[j] + node_size_y[j];
return wi < wj || (wi == wj && (init_y[i] > init_y[j] || (init_y[i] == init_y[j] && i < j)));
});
} else {
std::sort(cells.begin(), cells.end(), [&](int i, int j) -> bool {
float wi = 1000 * (init_x[i] + node_size_x[i] / 2) + node_size_x[i] + node_size_y[i];
float wj = 1000 * (init_x[j] + node_size_x[j] / 2) + node_size_x[j] + node_size_y[j];
return wi < wj || (wi == wj && (init_y[i] < init_y[j] || (init_y[i] == init_y[j] && i < j)));
});
}
for (int ci = bin_cells.at(i).size() - 1; ci >= 0; --ci) {
int node_id = cells.at(ci);
// align to site
float init_xl =
floorDiv(((alpha * init_x[node_id] + (1 - alpha) * x[node_id]) - xl), site_width) * site_width + xl;
float init_yl = (alpha * init_y[node_id] + (1 - alpha) * y[node_id]);
float width = ceilDiv(node_size_x[node_id], site_width) * site_width;
float height = node_size_y[node_id];
int num_node_rows = ceilDiv(height, row_height); // may take multiple rows
int blank_index_offset[num_node_rows];
std::fill(blank_index_offset, blank_index_offset + num_node_rows, 0);
int blank_initial_bin_id_y = floorDiv((init_yl - yl), blank_bin_size_y);
blank_initial_bin_id_y = std::min(blank_bin_id_yh - 1, std::max(blank_bin_id_yl, blank_initial_bin_id_y));
int blank_bin_id_dist_y = std::max(blank_initial_bin_id_y + 1, blank_bin_id_yh - blank_initial_bin_id_y);
int best_blank_bin_id_y = -1;
int best_blank_bi[num_node_rows];
std::fill(best_blank_bi, best_blank_bi + num_node_rows, -1);
float best_cost = xh - xl + yh - yl;
float best_xl = -1;
float best_yl = -1;
for (int bin_id_offset_y = 0; abs(bin_id_offset_y) < blank_bin_id_dist_y;
bin_id_offset_y = (bin_id_offset_y > 0) ? -bin_id_offset_y : -(bin_id_offset_y - 1)) {
int blank_bin_id_y = blank_initial_bin_id_y + bin_id_offset_y;
if (blank_bin_id_y < blank_bin_id_yl || blank_bin_id_y + num_node_rows > blank_bin_id_yh) {
continue;
}
int blank_bin_id = bin_id_x * blank_num_bins_y + blank_bin_id_y;
// blanks in this bin
const std::vector<Blank<float>>& blanks = bin_blanks.at(blank_bin_id);
int row_best_blank_bi[num_node_rows];
std::fill(row_best_blank_bi, row_best_blank_bi + num_node_rows, -1);
float row_best_cost = xh - xl + yh - yl;
float row_best_xl = -1;
float row_best_yl = -1;
bool search_flag = true;
for (unsigned int bi = 0; search_flag && bi < bin_blanks.at(blank_bin_id).size(); ++bi) {
const Blank<float>& blank = blanks[bi];
// for multi-row height cells, check blanks in upper rows
// find blanks with maximum intersection
blank_index_offset[0] = bi;
std::fill(blank_index_offset + 1, blank_index_offset + num_node_rows, -1);
while (true) {
Interval<float> intersect_blank(blank.xl, blank.xh);
for (int row_offset = 1; row_offset < num_node_rows; ++row_offset) {
int next_blank_bin_id_y = blank_bin_id_y + row_offset;
int next_blank_bin_id = bin_id_x * blank_num_bins_y + next_blank_bin_id_y;
unsigned int next_bi = blank_index_offset[row_offset] + 1;
for (; next_bi < bin_blanks.at(next_blank_bin_id).size(); ++next_bi) {
const Blank<float>& next_blank = bin_blanks.at(next_blank_bin_id)[next_bi];
Interval<float> intersect_blank_tmp = intersect_blank;
intersect_blank_tmp.intersect(next_blank.xl, next_blank.xh);
if (intersect_blank_tmp.xh - intersect_blank_tmp.xl >= width) {
intersect_blank = intersect_blank_tmp;
blank_index_offset[row_offset] = next_bi;
break;
}
}
if (next_bi == bin_blanks.at(next_blank_bin_id).size()) // not found
{
intersect_blank.xl = intersect_blank.xh = 0;
break;
}
}
float intersect_blank_width = intersect_blank.xh - intersect_blank.xl;
if (intersect_blank_width >= width) {
// compute displacement
float target_xl = init_xl;
float target_yl = blank.yl;
// alow tolerance to avoid more dead space
float beta = 4;
float tolerance = std::min(beta * width, intersect_blank_width / beta);
if (target_xl <= intersect_blank.xl + tolerance) {
target_xl = intersect_blank.xl;
} else if (target_xl + width >= intersect_blank.xh - tolerance) {
target_xl = (intersect_blank.xh - width);
}
float cost = fabs(target_xl - init_xl) + fabs(target_yl - init_yl);
// update best cost
if (cost < row_best_cost) {
std::copy(blank_index_offset, blank_index_offset + num_node_rows, row_best_blank_bi);
row_best_cost = cost;
row_best_xl = target_xl;
row_best_yl = target_yl;
} else { // early exit since we iterate within rows from left to right
search_flag = false;
}
} else { // not found
break;
}
if (num_node_rows < 2) { // for single-row height cells
break;
}
}
}
if (row_best_cost < best_cost) {
best_blank_bin_id_y = blank_bin_id_y;
std::copy(row_best_blank_bi, row_best_blank_bi + num_node_rows, best_blank_bi);
best_cost = row_best_cost;
best_xl = row_best_xl;
best_yl = row_best_yl;
} else if (best_cost + row_height < bin_id_offset_y * row_height) {
break; // early exit since we iterate from close row to far-away row
}
}
// found blank
if (best_blank_bin_id_y >= 0) {
x[node_id] = best_xl;
y[node_id] = best_yl;
// update cell position and blank
for (int row_offset = 0; row_offset < num_node_rows; ++row_offset) {
assert(best_blank_bi[row_offset] >= 0);
// blanks in this bin
int best_blank_bin_id = bin_id_x * blank_num_bins_y + best_blank_bin_id_y + row_offset;
std::vector<Blank<float>>& blanks = bin_blanks.at(best_blank_bin_id);
Blank<float>& blank = blanks.at(best_blank_bi[row_offset]);
assert(best_xl >= blank.xl && best_xl + width <= blank.xh);
assert(best_yl + row_height * row_offset == blank.yl);
if (best_xl == blank.xl) {
// update blank
blank.xl += width;
if (floorDiv((blank.xl - xl), site_width) * site_width != blank.xl - xl) {
logger.debug("1. move node %d from %g to %g, blank (%g, %g)",
node_id,
x[node_id],
blank.xl,
blank.xl,
blank.xh);
}
if (blank.xl >= blank.xh) {
bin_blanks.at(best_blank_bin_id)
.erase(bin_blanks.at(best_blank_bin_id).begin() + best_blank_bi[row_offset]);
}
} else if (best_xl + width == blank.xh) {
// update blank
blank.xh -= width;
if (floorDiv((blank.xh - xl), site_width) * site_width != blank.xh - xl) {
logger.debug("2. move node %d from %g to %g, blank (%g, %g)",
node_id,
x[node_id],
blank.xh - width,
blank.xl,
blank.xh);
}
if (blank.xl >= blank.xh) {
bin_blanks.at(best_blank_bin_id)
.erase(bin_blanks.at(best_blank_bin_id).begin() + best_blank_bi[row_offset]);
}
} else {
// need to update current blank and insert one more blank
Blank<float> new_blank;
new_blank.xl = best_xl + width;
new_blank.xh = blank.xh;
new_blank.yl = blank.yl;
new_blank.yh = blank.yh;
blank.xh = best_xl;
if (floorDiv((blank.xl - xl), site_width) * site_width != blank.xl - xl ||
floorDiv((blank.xh - xl), site_width) * site_width != blank.xh - xl ||
floorDiv((new_blank.xl - xl), site_width) * site_width != new_blank.xl - xl ||
floorDiv((new_blank.xh - xl), site_width) * site_width != new_blank.xh - xl) {
logger.debug("3. move node %d from %g to %g, blank (%g, %g), new_blank (%g, %g)",
node_id,
x[node_id],
init_xl,
blank.xl,
blank.xh,
new_blank.xl,
new_blank.xh);
}
bin_blanks.at(best_blank_bin_id)
.insert(bin_blanks.at(best_blank_bin_id).begin() + best_blank_bi[row_offset] + 1,
new_blank);
}
}
// remove from cells
bin_cells.at(i).erase(bin_cells.at(i).begin() + ci);
}
}
*num_unplaced_cells += bin_cells.at(i).size();
}
}
template <typename T>
void resizeBinObjects(std::vector<std::vector<T>>& bin_objs, int num_bins_x, int num_bins_y) {
bin_objs.resize(num_bins_x * num_bins_y);
}
template <typename T>
void countBinObjects(const std::vector<std::vector<T>>& bin_objs) {
int count = 0;
for (unsigned int i = 0; i < bin_objs.size(); ++i) {
count += bin_objs.at(i).size();
}
}
void mergeBinBlanks(const std::vector<std::vector<Blank<float>>>& src_bin_blanks,
int src_num_bins_x,
int src_num_bins_y, // dimensions for the src
std::vector<std::vector<Blank<float>>>& dst_bin_blanks,
int dst_num_bins_x,
int dst_num_bins_y, // dimensions for the dst
int scale_ratio_x, // roughly src_num_bins_x/dst_num_bins_x
float min_blank_width // minimum blank width to consider
) {
for (int i = 0; i < dst_num_bins_x * dst_num_bins_y; i += 1) {
// assume src_num_bins_y == dst_num_bins_y
int dst_bin_id_x = i / dst_num_bins_y;
int dst_bin_id_y = i - dst_bin_id_x * dst_num_bins_y;
int src_bin_id_x_bgn = dst_bin_id_x * scale_ratio_x;
int src_bin_id_x_end = std::min(src_bin_id_x_bgn + scale_ratio_x, src_num_bins_x);
std::vector<Blank<float>>& dst_bin_blank = dst_bin_blanks.at(i);
for (int ix = src_bin_id_x_bgn; ix < src_bin_id_x_end; ++ix) {
int iy = dst_bin_id_y; // same as src_bin_id_y
int src_bin_id = ix * src_num_bins_y + iy;
const std::vector<Blank<float>>& src_bin_blank = src_bin_blanks.at(src_bin_id);
int offset = 0;
if (!dst_bin_blank.empty() && !src_bin_blank.empty()) {
const Blank<float>& first_blank = src_bin_blank.at(0);
Blank<float>& last_blank = dst_bin_blank.at(dst_bin_blank.size() - 1);
if (last_blank.yl == first_blank.yl && last_blank.xh == first_blank.xl) {
last_blank.xh = first_blank.xh;
offset = 1;
}
}
for (unsigned int k = offset; k < src_bin_blank.size(); ++k) {
const Blank<float>& blank = src_bin_blank.at(k);
// prune small blanks
if (blank.xh - blank.xl >= min_blank_width) {
dst_bin_blanks.at(i).push_back(blank);
}
}
}
}
}
void mergeBinCells(
const std::vector<std::vector<int>>& src_bin_cells,
int src_num_bins_x,
int src_num_bins_y, // dimensions for the src
std::vector<std::vector<int>>& dst_bin_cells,
int dst_num_bins_x,
int dst_num_bins_y, // dimensions for the dst
int scale_ratio_x,
int scale_ratio_y // roughly src_num_bins_x/dst_num_bins_x, but may not be exactly the same due to even/odd numbers
) {
for (int i = 0; i < dst_num_bins_x * dst_num_bins_y; i += 1) {
int dst_bin_id_x = i / dst_num_bins_y;
int dst_bin_id_y = i - dst_bin_id_x * dst_num_bins_y;
int src_bin_id_x_bgn = dst_bin_id_x * scale_ratio_x;
int src_bin_id_y_bgn = dst_bin_id_y * scale_ratio_y;
int src_bin_id_x_end = std::min(src_bin_id_x_bgn + scale_ratio_x, src_num_bins_x);
int src_bin_id_y_end = std::min(src_bin_id_y_bgn + scale_ratio_y, src_num_bins_y);
for (int ix = src_bin_id_x_bgn; ix < src_bin_id_x_end; ++ix) {
for (int iy = src_bin_id_y_bgn; iy < src_bin_id_y_end; ++iy) {
int src_bin_id = ix * src_num_bins_y + iy;
const std::vector<int>& src_bin_cell = src_bin_cells.at(src_bin_id);
dst_bin_cells.at(i).insert(dst_bin_cells.at(i).end(), src_bin_cell.begin(), src_bin_cell.end());
}
}
}
}
void minNodeSize(const std::vector<std::vector<int>>& bin_cells,
const float* node_size_x,
const float* node_size_y,
float site_width,
float row_height,
int num_bins_x,
int num_bins_y,
int* min_node_size_x) {
for (int i = 0; i < num_bins_x * num_bins_y; i += 1) {
const std::vector<int>& cells = bin_cells.at(i);
float min_size_x = std::numeric_limits<int>::max();
for (unsigned int k = 0; k < cells.size(); ++k) {
int node_id = cells.at(k);
min_size_x = std::min(min_size_x, node_size_x[node_id]);
}
if (min_size_x != std::numeric_limits<int>::max()) {
*min_node_size_x = std::min(*min_node_size_x, (int)ceilDiv(min_size_x, site_width));
}
}
}
void greedyLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
LegalizationData db(at_db);
db.set_num_bins(num_bins_x, num_bins_y);
// first from right to left
// then from left to right
for (int i = 0; i < 2; ++i) {
num_bins_x = 1;
num_bins_y = 1;
// adjust bin sizes
float bin_size_x = (db.xh - db.xl) / static_cast<float>(num_bins_x);
float bin_size_y = (db.yh - db.yl) / static_cast<float>(num_bins_y);
bin_size_y = std::max((float)(ceilDiv(bin_size_y, db.row_height) * db.row_height), db.row_height);
num_bins_y = ceilDiv((db.yh - db.yl), bin_size_y);
// bin dimension in y direction for blanks is different from that for cells
float blank_bin_size_y = db.row_height;
int blank_num_bins_y = floorDiv((db.yh - db.yl), blank_bin_size_y);
logger.debug("%s blank_num_bins_y = %d", "Standard cell legalization", blank_num_bins_y);
// allocate bin cells
std::vector<std::vector<int>> bin_cells(num_bins_x * num_bins_y);
std::vector<std::vector<int>> bin_cells_copy(num_bins_x * num_bins_y);
// distribute cells to bins
distributeCells2Bins(db,
db.x,
db.y,
db.node_size_x,
db.node_size_y,
bin_size_x,
bin_size_y,
db.xl,
db.yl,
db.xh,
db.yh,
num_bins_x,
num_bins_y,
db.num_nodes,
db.num_movable_nodes,
bin_cells);
// allocate bin fixed cells
std::vector<std::vector<int>> bin_fixed_cells(num_bins_x * num_bins_y);
// distribute fixed cells to bins
distributeFixedCells2Bins(db,
db.init_x,
db.init_y,
db.node_size_x,
db.node_size_y,
bin_size_x,
bin_size_y,
db.xl,
db.yl,
db.xh,
db.yh,
num_bins_x,
num_bins_y,
db.num_nodes,
db.num_movable_nodes,
bin_fixed_cells);
// allocate bin blanks
std::vector<std::vector<Blank<float>>> bin_blanks(num_bins_x * blank_num_bins_y);
std::vector<std::vector<Blank<float>>> bin_blanks_copy(num_bins_x * blank_num_bins_y);
// distribute blanks to bins
distributeBlanks2Bins(db.init_x,
db.init_y,
db.node_size_x,
db.node_size_y,
bin_fixed_cells,
bin_size_x,
bin_size_y,
blank_bin_size_y,
db.xl,
db.yl,
db.xh,
db.yh,
db.site_width,
db.row_height,
num_bins_x,
num_bins_y,
blank_num_bins_y,
bin_blanks);
int num_unplaced_cells_host;
// minimum width in sites
int min_unplaced_node_size_x_host;
int num_iters = floor(log((float)std::min(num_bins_x, num_bins_y)) / log(2.0)) + 1;
for (int iter = 0; iter < num_iters; ++iter) {
logger.debug(
"%s iteration %d with %dx%d bins", "Standard cell legalization", iter, num_bins_x, num_bins_y);
num_unplaced_cells_host = 0;
logger.debug("%s #bin_blanks", "Standard cell legalization");
countBinObjects(bin_blanks);
legalizeBin(db.init_x,
db.init_y,
db.node_size_x,
db.node_size_y,
bin_blanks, // blanks in each bin, sorted from low to high, left to right
bin_cells, // unplaced cells in each bin
db.x,
db.y,
num_bins_x,
num_bins_y,
blank_num_bins_y,
bin_size_x,
bin_size_y,
blank_bin_size_y,
db.site_width,
db.row_height,
db.xl,
db.yl,
db.xh,
db.yh,
0.5,
4.0,
i % 2,
&num_unplaced_cells_host);
logger.debug("%s num_unplaced_cells = %d", "Standard cell legalization", num_unplaced_cells_host);
if (num_unplaced_cells_host == 0 || iter + 1 == num_iters) {
break;
}
// compute minimum size of unplaced cells
min_unplaced_node_size_x_host = floorDiv((db.xh - db.xl), db.site_width);
minNodeSize(bin_cells,
db.node_size_x,
db.node_size_y,
db.site_width,
db.row_height,
num_bins_x,
num_bins_y,
&min_unplaced_node_size_x_host);
logger.debug("%s minimum unplaced node_size_x %d sites",
"Standard cell legalization",
min_unplaced_node_size_x_host);
// ceil(num_bins_x/2), ceil(num_bins_y/2)
int dst_num_bins_x = (num_bins_x >> 1) + (num_bins_x & 1);
int dst_num_bins_y = (num_bins_y >> 1) + (num_bins_y & 1);
int scale_ratio_x = (num_bins_x == dst_num_bins_x) ? 1 : num_bins_x / dst_num_bins_x;
int scale_ratio_y = (num_bins_y == dst_num_bins_y) ? 1 : num_bins_y / dst_num_bins_y;
resizeBinObjects(bin_cells_copy, dst_num_bins_x, dst_num_bins_y);
mergeBinCells(bin_cells,
num_bins_x,
num_bins_y, // dimensions for the src
bin_cells_copy, // ceil(src_num_bins_x/2) * ceil(src_num_bins_y/2)
dst_num_bins_x,
dst_num_bins_y,
scale_ratio_x,
scale_ratio_y);
resizeBinObjects(bin_blanks_copy, dst_num_bins_x, blank_num_bins_y);
mergeBinBlanks(bin_blanks,
num_bins_x,
blank_num_bins_y, // dimensions for the src
bin_blanks_copy, // ceil(src_num_bins_x/2) * ceil(src_num_bins_y/2)
dst_num_bins_x,
blank_num_bins_y,
scale_ratio_x,
min_unplaced_node_size_x_host * db.site_width);
// update bin dimensions
num_bins_x = dst_num_bins_x;
num_bins_y = dst_num_bins_y;
bin_size_x = bin_size_x * 2;
bin_size_y = bin_size_y * 2;
std::swap(bin_cells, bin_cells_copy);
std::swap(bin_blanks, bin_blanks_copy);
}
}
}
} // namespace dp

View File

@ -0,0 +1,381 @@
#pragma once
#include <vector>
#include "gpudp/dp/ism/diamond_search.h"
#include "gpudp/lg/legalization_db.h"
namespace dp {
/// @brief A class models Hannan grids.
class HannanGrids {
public:
HannanGrids(const float* x,
const float* y,
const float* width,
const float* height,
std::size_t n,
const float xl,
const float yl,
const float xh,
const float yh,
const float spacing_x,
const float spacing_y) {
build(x, y, width, height, n, xl, yl, xh, yh, spacing_x, spacing_y);
}
std::size_t dim_x() const { return m_coordx.size(); }
std::size_t dim_y() const { return m_coordy.size(); }
/// @brief query x index in log(n) time complexity
std::size_t grid_x(float x) const {
auto it = std::lower_bound(m_coordx.begin(), m_coordx.end(), x);
std::size_t ix = std::min((std::size_t)std::distance(m_coordx.begin(), it), dim_x() - 1);
float gxl = m_coordx[ix];
if (gxl > x && ix) {
ix -= 1;
}
return ix;
}
/// @brief query y index in log(n) time complexity
std::size_t grid_y(float y) const {
auto it = std::lower_bound(m_coordy.begin(), m_coordy.end(), y);
std::size_t iy = std::min((std::size_t)std::distance(m_coordy.begin(), it), dim_y() - 1);
float gyl = m_coordy[iy];
if (gyl > y && iy) {
iy -= 1;
}
return iy;
}
/// @brief get x coordinate of a grid
float coord_x(std::size_t ix) const { return m_coordx[ix]; }
/// @brief get y coordinate of a grid
float coord_y(std::size_t iy) const { return m_coordy[iy]; }
/// @brief check whether a grid overlaps with a rectangle.
/// Touching is not considered as overlap.
bool overlap(std::size_t ix, std::size_t iy, float xl, float yl, float xh, float yh) const {
float gxl = m_coordx[ix];
float gxh = (ix + 1 == dim_x()) ? std::numeric_limits<float>::max() : m_coordx[ix + 1];
float gyl = m_coordy[iy];
float gyh = (iy + 1 == dim_y()) ? std::numeric_limits<float>::max() : m_coordy[iy + 1];
return std::max(gxl, xl) < std::min(gxh, xh) && std::max(gyl, yl) < std::min(gyh, yh);
}
protected:
/// @brief build grids from rectangles and boundaries
void build(const float* x,
const float* y,
const float* width,
const float* height,
std::size_t n,
const float xl,
const float yl,
const float xh,
const float yh,
const float spacing_x,
const float spacing_y) {
// collect all scan lines
m_coordx.reserve((n << 1) + 2);
m_coordy.reserve((n << 1) + 2);
m_coordx.push_back(xl);
m_coordx.push_back(xh);
m_coordy.push_back(yl);
m_coordy.push_back(yh);
for (std::size_t i = 0; i < n; ++i) {
m_coordx.push_back(x[i]);
m_coordx.push_back(x[i] + width[i]);
m_coordy.push_back(y[i]);
m_coordy.push_back(y[i] + height[i]);
}
// sort and make them unique
std::sort(m_coordx.begin(), m_coordx.end());
std::sort(m_coordy.begin(), m_coordy.end());
m_coordx.resize(std::distance(m_coordx.begin(), std::unique(m_coordx.begin(), m_coordx.end())));
m_coordy.resize(std::distance(m_coordy.begin(), std::unique(m_coordy.begin(), m_coordy.end())));
// in case some grids are too large
// add more scan lines with step size spacing_x and spacing_y
for (std::size_t i = 1, ie = m_coordx.size(); i < ie; ++i) {
float gxl = m_coordx[i - 1];
float gxh = m_coordx[i];
for (float xl = gxl + spacing_x; xl < gxh; xl += spacing_x) {
m_coordx.push_back(xl);
}
}
for (std::size_t i = 1, ie = m_coordy.size(); i < ie; ++i) {
float gyl = m_coordy[i - 1];
float gyh = m_coordy[i];
for (float yl = gyl + spacing_y; yl < gyh; yl += spacing_y) {
m_coordy.push_back(yl);
}
}
// they should already be unique
std::sort(m_coordx.begin(), m_coordx.end());
std::sort(m_coordy.begin(), m_coordy.end());
}
std::vector<float> m_coordx; ///< coordinates of grid lines in x direction
std::vector<float> m_coordy; ///< coordinates of grid lines in y direction
};
/// @brief A class models binary maps on Hannan grids.
class HannanGridMap : public HannanGrids {
public:
HannanGridMap(const float* x,
const float* y,
const float* width,
const float* height,
std::size_t n,
const float xl,
const float yl,
const float xh,
const float yh,
const float spacing_x,
const float spacing_y)
: HannanGrids(x, y, width, height, n, xl, yl, xh, yh, spacing_x, spacing_y) {
// construct 2D binary map
m_map.assign(this->dim_x() * this->dim_y(), 0);
}
/// @brief set an entry in grid map
void set(std::size_t ix, std::size_t iy, bool value) { m_map[ix * this->dim_y() + iy] = value; }
/// @brief get an entry in grid map
bool at(std::size_t ix, std::size_t iy) const { return m_map[ix * this->dim_y() + iy]; }
/// @brief check whether a rectangle overlaps with any grid in the map
bool overlap(float xl, float yl, float xh, float yh) const {
std::size_t ixl = this->grid_x(xl);
std::size_t ixh = this->grid_x(xh) + 1;
std::size_t iyl = this->grid_y(yl);
std::size_t iyh = this->grid_y(yh) + 1;
for (std::size_t ix = ixl; ix < ixh; ++ix) {
for (std::size_t iy = iyl; iy < iyh; ++iy) {
if (this->HannanGrids::overlap(ix, iy, xl, yl, xh, yh) && this->at(ix, iy)) {
return true;
}
}
}
return false;
}
/// @brief add a rectangle to the grid map
void add(float xl, float yl, float xh, float yh) {
std::size_t ixl = this->grid_x(xl);
std::size_t ixh = this->grid_x(xh) + 1;
std::size_t iyl = this->grid_y(yl);
std::size_t iyh = this->grid_y(yh) + 1;
for (std::size_t ix = ixl; ix < ixh; ++ix) {
for (std::size_t iy = iyl; iy < iyh; ++iy) {
if (this->HannanGrids::overlap(ix, iy, xl, yl, xh, yh)) {
this->set(ix, iy, 1);
}
}
}
}
protected:
std::vector<unsigned char> m_map; ///< 2D map indicating whether a grid is taken or not
};
/// @brief A greedy macro legalization algorithm manipulating on Hannan grids.
/// The procedure of the algorithm is as follows.
/// For each macro:
/// Perfrom spiral/diamond search to the locations;
/// Find the first one with minimum displacement;
/// Update the grid map;
/// If the layout is very tight, it may not be able to find a solution.
/// @return true if all macros legalized
bool hannanLegalize(LegalizationData& db,
std::vector<int>& macros,
const std::vector<int>& fixed_macros,
int max_iters) {
logger.info("Legalize movable macros on Hannan grids");
// count number of failures to control the order
std::vector<int> failure_counts(db.num_movable_nodes, 0);
std::vector<float> x(db.num_movable_nodes, 0);
std::vector<float> y(db.num_movable_nodes, 0);
bool legal = true;
for (int iter = 0; iter < max_iters; ++iter) {
logger.info("round %d", iter);
// copy location to working array
for (auto node_id : macros) {
x[node_id] = db.x[node_id];
y[node_id] = db.y[node_id];
}
// sort from left to right, large to small
std::sort(macros.begin(), macros.end(), [&](int node_id1, int node_id2) {
int factor1 = (1 + failure_counts[node_id1]);
int factor2 = (1 + failure_counts[node_id2]);
float a1 = db.node_size_x[node_id1] * db.node_size_y[node_id1]; // * factor1;
float a2 = db.node_size_x[node_id2] * db.node_size_y[node_id2]; // * factor2;
float x1 = x[node_id1] / factor1;
float x2 = x[node_id2] / factor2;
float y1 = y[node_id1] / factor1;
float y2 = y[node_id2] / factor2;
// return a1 > a2 || (a1 == a2 && (x1 < x2 || (x1 == x2 && (y1 < y2 || (y1 == y2 && node_id1 <
// node_id2))))); return x1 < x2 || (x1 == x2 && (a1 > a2 || (a1 == a2 && (y1 < y2 || (y1 == y2 && node_id1
// < node_id2)))));
return x1 < x2 || (x1 == x2 && (y1 < y2 || (y1 == y2 && (a1 > a2 || (a1 == a2 && node_id1 < node_id2)))));
});
float spacing_x = std::numeric_limits<float>::max();
float spacing_y = std::numeric_limits<float>::max();
for (auto node_id : macros) {
spacing_x = std::min(spacing_x, db.node_size_x[node_id]);
spacing_y = std::min(spacing_y, db.node_size_y[node_id]);
}
// make sure the grid is not too small
spacing_x = std::max(spacing_x, (db.xh - db.xl) / db.num_bins_x);
spacing_y = std::max(spacing_y, (db.yh - db.yl) / db.num_bins_y);
logger.debug("maximum grid spacing %gx%g, equivalent to %dx%d bins",
(double)spacing_x,
(double)spacing_y,
(int)((db.xh - db.xl) / spacing_x),
(int)((db.yh - db.yl) / spacing_y));
// construct hannan grid map for fixed macros
// collect fixed and dummy fixed nodes
std::vector<float> vx;
std::vector<float> vy;
std::vector<float> node_size_x;
std::vector<float> node_size_y;
vx.reserve(db.num_nodes);
vy.reserve(db.num_nodes);
node_size_x.reserve(db.num_nodes);
node_size_y.reserve(db.num_nodes);
for (auto node_id : fixed_macros) {
vx.push_back(db.x[node_id]);
vy.push_back(db.y[node_id]);
node_size_x.push_back(db.node_size_x[node_id]);
node_size_y.push_back(db.node_size_y[node_id]);
}
for (auto node_id : macros) {
vx.push_back(x[node_id]);
vy.push_back(y[node_id]);
node_size_x.push_back(db.node_size_x[node_id]);
node_size_y.push_back(db.node_size_y[node_id]);
}
HannanGridMap grid_map(vx.data(),
vy.data(),
node_size_x.data(),
node_size_y.data(),
vx.size(),
db.xl,
db.yl,
db.xh,
db.yh,
spacing_x,
spacing_y);
// the right and top boundary should always be occupied
for (std::size_t ix = 0; ix < grid_map.dim_x(); ++ix) {
grid_map.set(ix, grid_map.dim_y() - 1, 1);
}
for (std::size_t iy = 0; iy < grid_map.dim_y(); ++iy) {
grid_map.set(grid_map.dim_x() - 1, iy, 1);
}
// set fixed nodes to occupy the grid map
for (auto node_id : fixed_macros) {
float xl = db.init_x[node_id];
float xh = xl + db.node_size_x[node_id];
float yl = db.init_y[node_id];
float yh = yl + db.node_size_y[node_id];
std::size_t ixl = grid_map.grid_x(xl);
std::size_t ixh = grid_map.grid_x(xh);
std::size_t iyl = grid_map.grid_y(yl);
std::size_t iyh = grid_map.grid_y(yh);
for (std::size_t ix = ixl; ix <= ixh; ++ix) {
for (std::size_t iy = iyl; iy <= iyh; ++iy) {
if (grid_map.HannanGrids::overlap(ix, iy, xl, yl, xh, yh)) {
grid_map.set(ix, iy, 1);
}
}
}
}
auto search_grids = diamond_search_sequence(grid_map.dim_y(), grid_map.dim_x());
logger.debug("Construct %lux%lu Hannan grids, diamond search sequence %lu",
grid_map.dim_x(),
grid_map.dim_y(),
search_grids.size());
legal = true;
for (auto node_id : macros) {
float node_x = x[node_id];
float node_y = y[node_id];
float width = db.node_size_x[node_id];
float height = db.node_size_y[node_id];
std::size_t init_ix = grid_map.grid_x(node_x);
std::size_t init_iy = grid_map.grid_y(node_y);
bool found = false;
for (auto grid_offset : search_grids) {
std::size_t ix = init_ix + grid_offset.ic;
std::size_t iy = init_iy + grid_offset.ir;
// valid grid
if (ix < grid_map.dim_x() && iy < grid_map.dim_y()) {
float xl = grid_map.coord_x(ix);
float yl = grid_map.coord_y(iy);
if (grid_offset.ic == 0 && grid_offset.ir == 0) {
assert_msg(xl == node_x, "%g != %g", xl, node_x);
assert_msg(yl == node_y, "%g != %g", yl, node_y);
}
// make sure the coordinates are aligned to row and site
float aligned_xl = db.align2site(xl, width);
float aligned_yl = db.align2row(yl, height);
if (aligned_xl < xl) {
xl = aligned_xl + db.site_width;
}
if (aligned_yl < yl) {
yl = aligned_yl + db.row_height;
}
float xh = xl + width;
float yh = yl + height;
if (!grid_map.overlap(xl, yl, xh, yh)) {
x[node_id] = xl;
y[node_id] = yl;
grid_map.add(xl, yl, xh, yh);
found = true;
break;
}
}
}
if (!found) {
logger.error("failed to find legal position for macro %d (%g, %g, %g, %g)",
node_id,
node_x,
node_y,
node_x + width,
node_y + height);
failure_counts[node_id] += 1;
legal = false;
}
}
if (legal) {
break;
}
}
// copy solutions back
for (auto node_id : macros) {
db.x[node_id] = x[node_id];
db.y[node_id] = y[node_id];
}
return legal;
}
} // namespace dp

View File

@ -0,0 +1,165 @@
#pragma once
#include "common/common.h"
#include "common/db/Database.h"
#include "gpudp/db/dp_torch.h"
namespace dp {
inline int floorDiv(float a, float b, float rtol = 1e-4) { return std::floor((a + rtol * b) / b); }
inline int ceilDiv(float a, float b, float rtol = 1e-4) { return std::ceil((a - rtol * b) / b); }
inline int roundDiv(float a, float b) { return std::round(a / b); }
template <typename T>
struct Space {
T xl;
T xh;
};
struct RowMapIndex {
int row_id;
int sub_id;
};
struct BinMapIndex {
int bin_id;
int sub_id;
};
struct Box {
float xl;
float yl;
float xh;
float yh;
Box() {
xl = std::numeric_limits<float>::max();
yl = std::numeric_limits<float>::max();
xh = std::numeric_limits<float>::lowest();
yh = std::numeric_limits<float>::lowest();
}
Box(float xxl, float yyl, float xxh, float yyh) : xl(xxl), yl(yyl), xh(xxh), yh(yyh) {}
float center_x() const { return (xl + xh) / 2; }
float center_y() const { return (yl + yh) / 2; }
float width() const { return (xh - xl); }
float height() const { return (yh - yl); }
float area() const { return (xh - xl) * (yh - yl); }
};
class LegalizationData {
public:
LegalizationData() {}
LegalizationData(DPTorchRawDB& at_db)
: x(at_db.x.data_ptr<float>()),
y(at_db.y.data_ptr<float>()),
init_x(at_db.init_x.data_ptr<float>()),
init_y(at_db.init_y.data_ptr<float>()),
node_size_x(at_db.node_size_x.data_ptr<float>()),
node_size_y(at_db.node_size_y.data_ptr<float>()),
pin_offset_x(at_db.pin_offset_x.data_ptr<float>()),
pin_offset_y(at_db.pin_offset_y.data_ptr<float>()),
flat_node2pin_start_map(at_db.flat_node2pin_start_map.data_ptr<int>()),
flat_node2pin_map(at_db.flat_node2pin_map.data_ptr<int>()),
pin2node_map(at_db.pin2node_map.data_ptr<int>()),
flat_net2pin_start_map(at_db.flat_net2pin_start_map.data_ptr<int>()),
flat_net2pin_map(at_db.flat_net2pin_map.data_ptr<int>()),
pin2net_map(at_db.pin2net_map.data_ptr<int>()),
flat_region_boxes_start(at_db.flat_region_boxes_start.data_ptr<int>()),
flat_region_boxes(at_db.flat_region_boxes.data_ptr<float>()),
node2fence_region_map(at_db.node2fence_region_map.data_ptr<int>()),
net_mask(at_db.net_mask.data_ptr<bool>()),
node_weight(at_db.node_weight.data_ptr<float>()),
xl(at_db.xl),
xh(at_db.xh),
yl(at_db.yl),
yh(at_db.yh),
row_height(at_db.row_height),
site_width(at_db.site_width),
num_sites_x(at_db.num_sites_x),
num_sites_y(at_db.num_sites_y),
num_threads(at_db.num_threads),
num_nodes(at_db.num_nodes),
num_movable_nodes(at_db.num_movable_nodes),
num_nets(at_db.num_nets),
num_pins(at_db.num_pins),
num_regions(at_db.num_regions) {}
public:
float* x; // new pos x, need to be checked their legality
float* y; // new pos y, need to be checked their legality
const float* init_x; // original pos x
const float* init_y; // original pos y
const float* node_size_x;
const float* node_size_y;
const float* pin_offset_x;
const float* pin_offset_y;
const int* flat_node2pin_start_map;
const int* flat_node2pin_map;
const int* pin2node_map;
const int* flat_net2pin_start_map;
const int* flat_net2pin_map;
const int* pin2net_map;
const int* flat_region_boxes_start;
const float* flat_region_boxes;
const int* node2fence_region_map;
const bool* net_mask;
const float* node_weight;
/* chip info */
float xl;
float yl;
float xh;
float yh;
/* row info */
int num_sites_x;
int num_sites_y;
float row_height;
float site_width;
int num_nets;
int num_movable_nodes;
int num_nodes;
int num_pins;
int num_regions;
int num_threads;
int num_bins_x;
int num_bins_y;
float bin_size_x;
float bin_size_y;
public:
void set_num_bins(int num_bins_x_, int num_bins_y_) {
num_bins_x = num_bins_x_;
num_bins_y = num_bins_y_;
bin_size_x = (xh - xl) / num_bins_x_;
bin_size_y = (yh - yl) / num_bins_y_;
}
inline bool is_dummy_fixed(int node_id) const {
// DUMMY_FIXED_NUM_ROWS == 2
return (node_id < num_movable_nodes && node_size_y[node_id] > (row_height * 2));
}
inline float align2row(float y, float height) const {
float yy = std::max(std::min(y, yh - height), yl);
yy = floorDiv(yy - yl, row_height) * row_height + yl;
return yy;
}
inline float align2site(float x, float width) const {
float xx = std::max(std::min(x, xh - width), xl);
xx = floorDiv(xx - xl, site_width) * site_width + xl;
return xx;
}
};
} // namespace dp

View File

@ -0,0 +1,377 @@
#include "gpudp/lg/hannan_legalize.h"
#include "gpudp/lg/legalization_db.h"
namespace dp {
/// @brief The macro legalization follows the way of floorplanning,
/// because macros have quite different sizes.
bool check_macro_legality(LegalizationData& db, const std::vector<int>& macros, bool fast_check) {
// check legality between movable and fixed macros
// for debug only, so it is slow
auto checkOverlap2Nodes = [&](int i,
int node_id1,
float xl1,
float yl1,
float width1,
float height1,
int j,
int node_id2,
float xl2,
float yl2,
float width2,
float height2) {
float xh1 = xl1 + width1;
float yh1 = yl1 + height1;
float xh2 = xl2 + width2;
float yh2 = yl2 + height2;
if (std::min(xh1, xh2) > std::max(xl1, xl2) && std::min(yh1, yh2) > std::max(yl1, yl2)) {
logger.error(
"macro %d (%g, %g, %g, %g) var %d overlaps with macro %d "
"(%g, %g, %g, %g) var %d, fixed: %d",
node_id1,
xl1,
yl1,
xh1,
yh1,
i,
node_id2,
xl2,
yl2,
xh2,
yh2,
j,
(int)(node_id2 >= db.num_movable_nodes));
return true;
}
return false;
};
bool legal = true;
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
int node_id1 = macros[i];
float xl1 = db.x[node_id1];
float yl1 = db.y[node_id1];
float width1 = db.node_size_x[node_id1];
float height1 = db.node_size_y[node_id1];
// constraints with other macros
for (unsigned int j = i + 1; j < ie; ++j) {
int node_id2 = macros[j];
float xl2 = db.x[node_id2];
float yl2 = db.y[node_id2];
float width2 = db.node_size_x[node_id2];
float height2 = db.node_size_y[node_id2];
bool overlap =
checkOverlap2Nodes(i, node_id1, xl1, yl1, width1, height1, j, node_id2, xl2, yl2, width2, height2);
if (overlap) {
legal = false;
if (fast_check) {
return legal;
}
}
}
// constraints with fixed macros
// when considering fixed macros, there is no guarantee to find legal
// solution with current ad-hoc constraint graphs
for (int j = db.num_movable_nodes; j < db.num_nodes; ++j) {
int node_id2 = j;
float xl2 = db.init_x[node_id2];
float yl2 = db.init_y[node_id2];
float width2 = db.node_size_x[node_id2];
float height2 = db.node_size_y[node_id2];
bool overlap =
checkOverlap2Nodes(i, node_id1, xl1, yl1, width1, height1, j, node_id2, xl2, yl2, width2, height2);
if (overlap) {
legal = false;
if (fast_check) {
return legal;
}
}
}
}
if (legal) {
logger.debug("Macro legality check PASSED");
} else {
logger.error("Macro legality check FAILED");
}
return legal;
}
struct MacroLegalizeStats {
float total_displace;
float max_displace;
float total_weighted_displace; ///< displacement weighted by macro area ratio to
///< average macro area
float max_weighted_displace;
// float average_macro_area;
};
MacroLegalizeStats compute_displace(const LegalizationData& db, const std::vector<int>& macros) {
MacroLegalizeStats stats;
stats.total_displace = 0;
stats.max_displace = 0;
stats.total_weighted_displace = 0;
stats.max_weighted_displace = 0;
// stats.average_macro_area = 0;
// for (auto node_id : macros)
//{
// stats.average_macro_area += db.node_size_x[node_id] *
// db.node_size_y[node_id];
//}
// stats.average_macro_area /= macros.size();
for (auto node_id : macros) {
float displace = std::abs(db.init_x[node_id] - db.x[node_id]) + std::abs(db.init_y[node_id] - db.y[node_id]);
stats.total_displace += displace;
stats.max_displace = std::max(stats.max_displace, displace);
displace *= db.node_weight[node_id];
stats.total_weighted_displace += displace;
stats.max_weighted_displace = std::max(stats.max_weighted_displace, displace);
}
return stats;
}
/// @brief Rough legalize some special macros
/// 1. macros that form small clusters overlapping with each other
/// 2. macros blocked by big ones
/// All the other macros are regarded as fixed.
/// @param small_clusters_flag controls whether to perform the legalization for
/// 1
/// @param blocked_macros_flag controls whether to perform the legalization for
/// 2
bool roughLegalize(LegalizationData& db,
const std::vector<int>& macros,
const std::vector<int>& fixed_macros,
bool small_clusters_flag,
bool blocked_macros_flag) {
std::vector<unsigned char> markers(db.num_nodes, false);
std::vector<int> macros_for_rough_legalize;
std::vector<int> fixed_macros_for_rough_legalize;
// collect small clusters
if (small_clusters_flag) {
std::vector<std::vector<int> > clusters(macros.size());
float cluster_area_ratio = 2;
float cluster_overlap_ratio = 0.5;
unsigned int cluster_macro_numbers_threshold = 2;
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
int node_id1 = macros[i];
Box box1(db.x[node_id1],
db.y[node_id1],
db.x[node_id1] + db.node_size_x[node_id1],
db.y[node_id1] + db.node_size_y[node_id1]);
float a1 = box1.area();
clusters.at(i).push_back(node_id1);
for (unsigned int j = i + 1; j < ie; ++j) {
int node_id2 = macros[j];
Box box2(db.x[node_id2],
db.y[node_id2],
db.x[node_id2] + db.node_size_x[node_id2],
db.y[node_id2] + db.node_size_y[node_id2]);
float a2 = box2.area();
if (a1 >= a2 / cluster_area_ratio && a1 <= a2 * cluster_area_ratio) {
float overlap = std::max((float)0, std::min(box1.xh, box2.xh) - std::max(box1.xl, box2.xl)) *
std::max((float)0, std::min(box1.yh, box2.yh) - std::max(box1.yl, box2.yl));
if (overlap >= std::min(a1, a2) * cluster_overlap_ratio) {
clusters.at(i).push_back(node_id2);
}
}
}
}
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
if (clusters.at(i).size() >= cluster_macro_numbers_threshold) {
markers.at(macros.at(i)) = true;
}
}
}
// collect small macros blocked by large ones
// If a small macro is blocked by two big macros, it is easier to move the
// small one around. We detect such blocks by checking whether the macro is
// blocked from left, right, bottom, top 4 directions. Any macro with (left,
// right) or (bottom, top) blocked will be collected.
if (blocked_macros_flag) {
float blocked_macros_area_ratio = 10; // the area ratio of macros to be regarded as large
float blocked_macros_direct_threshold = 0.9; // determine the direction blocked
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
int node_id1 = macros[i];
if (!markers[node_id1]) {
Box box1(db.x[node_id1],
db.y[node_id1],
db.x[node_id1] + db.node_size_x[node_id1],
db.y[node_id1] + db.node_size_y[node_id1]);
float a1 = box1.area();
std::array<unsigned char, 4> intersect_directs; // from L, R, B, float
// direction, the box
// is overlapped
intersect_directs.fill(0);
for (unsigned int j = 0; j < ie; ++j) {
int node_id2 = macros[j];
if (i != j && !markers[node_id2]) {
Box box2(db.x[node_id2],
db.y[node_id2],
db.x[node_id2] + db.node_size_x[node_id2],
db.y[node_id2] + db.node_size_y[node_id2]);
float a2 = box2.area();
if (a1 * blocked_macros_area_ratio < a2) {
Box intersect_box(std::max(box1.xl, box2.xl),
std::max(box1.yl, box2.yl),
std::min(box1.xh, box2.xh),
std::min(box1.yh, box2.yh));
if (intersect_box.xl < intersect_box.xh && intersect_box.yl < intersect_box.yh) {
if (intersect_box.height() > box1.height() * blocked_macros_direct_threshold) {
if (box2.xl <= box1.xl) {
intersect_directs[0] = 1; // xl
}
if (box2.xh >= box1.xh) {
intersect_directs[1] = 1; // xh
}
}
if (intersect_box.width() > box1.width() * blocked_macros_direct_threshold) {
if (box2.yl <= box1.yl) {
intersect_directs[2] = 1; // yl
}
if (box2.yh >= box1.yh) {
intersect_directs[3] = 1; // yh
}
}
}
}
if ((intersect_directs[0] && intersect_directs[1]) ||
(intersect_directs[2] && intersect_directs[3])) {
markers[node_id1] = true;
logger.debug("collect %d", node_id1);
break;
}
}
}
}
}
}
fixed_macros_for_rough_legalize = fixed_macros;
for (auto node_id : macros) {
if (markers[node_id]) {
macros_for_rough_legalize.push_back(node_id);
} else {
fixed_macros_for_rough_legalize.push_back(node_id);
}
}
logger.info("Rough legalize small clusters with %lu macros", macros_for_rough_legalize.size());
return hannanLegalize(db, macros_for_rough_legalize, fixed_macros_for_rough_legalize, 1);
}
bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
LegalizationData db(at_db);
db.set_num_bins(num_bins_x, num_bins_y);
// collect macros
std::vector<int> macros;
for (int i = 0; i < db.num_movable_nodes; ++i) {
if (db.is_dummy_fixed(i)) {
// in some extreme case, some macros with 0 area should be ignored
float area = db.node_size_x[i] * db.node_size_y[i];
if (area > 0) {
macros.push_back(i);
}
}
}
logger.info("Macro legalization: regard %lu cells as dummy fixed (movable macros)", macros.size());
// in case there is no movable macros
if (macros.empty()) {
return true;
}
// fixed macros
std::vector<int> fixed_macros;
fixed_macros.reserve(db.num_nodes - db.num_movable_nodes);
for (int i = db.num_movable_nodes; i < db.num_nodes; ++i) {
// in some extreme case, some fixed macros with 0 area should be ignored
float area = db.node_size_x[i] * db.node_size_y[i];
if (area > 0) {
fixed_macros.push_back(i);
}
}
// store the best legalization solution found
std::vector<float> best_x(macros.size());
std::vector<float> best_y(macros.size());
MacroLegalizeStats best_displace;
best_displace.total_displace = std::numeric_limits<float>::max();
best_displace.max_displace = std::numeric_limits<float>::max();
best_displace.total_weighted_displace = std::numeric_limits<float>::max();
best_displace.max_weighted_displace = std::numeric_limits<float>::max();
// update current best solution
auto update_best = [&](bool legal, const MacroLegalizeStats& displace) {
if (legal && displace.total_displace < best_displace.total_displace) {
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
int macro_id = macros[i];
best_x[i] = db.x[macro_id];
best_y[i] = db.y[macro_id];
}
best_displace = displace;
}
};
// first round rough legalization with Hannan grid for clusters
bool small_clusters_flag = true;
bool blocked_macros_flag = false;
roughLegalize(db, macros, fixed_macros, small_clusters_flag, blocked_macros_flag);
auto displace = compute_displace(db, macros);
logger.info("Macro displacement total %g, max %g, weighted total %g, max %g",
displace.total_displace,
displace.max_displace,
displace.total_weighted_displace,
displace.max_weighted_displace);
bool legal = check_macro_legality(db, macros, true);
// try Hannan grid legalization if still not legal
if (!legal) {
legal = hannanLegalize(db, macros, fixed_macros, 10);
auto displace = compute_displace(db, macros);
logger.info("Macro displacement total %g, max %g, weighted total %g, max %g",
displace.total_displace,
displace.max_displace,
displace.total_weighted_displace,
displace.max_weighted_displace);
legal = check_macro_legality(db, macros, true);
update_best(legal, displace);
// apply best solution
if (best_displace.total_displace < std::numeric_limits<float>::max()) {
logger.info(
"use best macro displacement total %g, max %g, weighted "
"total %g, max %g",
best_displace.total_displace,
best_displace.max_displace,
best_displace.total_weighted_displace,
best_displace.max_weighted_displace);
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
int macro_id = macros[i];
db.x[macro_id] = best_x[i];
db.y[macro_id] = best_y[i];
}
}
}
logger.info("Align macros to site and rows");
// align the lower left corner to row and site
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
int node_id = macros[i];
db.x[node_id] = db.align2site(db.x[node_id], db.node_size_x[node_id]);
db.y[node_id] = db.align2row(db.y[node_id], db.node_size_y[node_id]);
}
legal = check_macro_legality(db, macros, false);
return legal;
}
} // namespace dp

View File

@ -0,0 +1,69 @@
# Files
file(GLOB_RECURSE SRC_FILES_GGR ${CMAKE_CURRENT_SOURCE_DIR}/db/*.cpp
${CMAKE_CURRENT_SOURCE_DIR}/gr/*.cpp)
file(GLOB_RECURSE SRC_FILES_GGR_CUDA ${CMAKE_CURRENT_SOURCE_DIR}/*.cu)
# CUDA GGR Kernel
cuda_add_library(ggr_cuda_tmp STATIC ${SRC_FILES_GGR_CUDA})
set_target_properties(ggr_cuda_tmp PROPERTIES CUDA_RESOLVE_DEVICE_SYMBOLS ON)
set_target_properties(ggr_cuda_tmp PROPERTIES CUDA_SEPARABLE_COMPILATION ON)
set_target_properties(ggr_cuda_tmp PROPERTIES POSITION_INDEPENDENT_CODE ON)
target_include_directories(ggr_cuda_tmp PRIVATE ${PROJECT_SOURCE_DIR}/cpp_to_py ${TORCH_INCLUDE_DIRS} ${FLUTE_INCLUDE_DIR})
target_link_libraries(ggr_cuda_tmp torch ${TORCH_PYTHON_LIBRARY} xplace_common flute)
# target_compile_options(ggr_cuda_tmp PRIVATE "$<$<COMPILE_LANGUAGE:CUDA>:SHELL:-use_fast_math>") # not work...
# CPU GGR object
add_library(ggr SHARED ${CMAKE_CURRENT_SOURCE_DIR}/../io_parser/gp/GPDatabase.cpp
${SRC_FILES_GGR}
${SRC_FILES_GGR_CUDA})
target_include_directories(ggr PRIVATE ${PROJECT_SOURCE_DIR}/cpp_to_py ${TORCH_INCLUDE_DIRS} ${FLUTE_INCLUDE_DIR})
target_link_libraries(ggr PRIVATE torch ${TORCH_PYTHON_LIBRARY} xplace_common flute ggr_cuda_tmp pthread)
target_compile_options(ggr PRIVATE -fPIC)
install(TARGETS ggr DESTINATION ${XPLACE_LIB_DIR})
# For Pybind
add_pytorch_extension(gpugr PyBindCppMain.cpp
EXTRA_INCLUDE_DIRS ${PROJECT_SOURCE_DIR}/cpp_to_py ${FLUTE_INCLUDE_DIR}
EXTRA_LINK_LIBRARIES xplace_common flute io_parser ggr)
install(TARGETS gpugr DESTINATION ${XPLACE_LIB_DIR})
################ For debug only ##################
# set(CMAKE_BUILD_TYPE Release)
# set(CMAKE_CXX_STANDARD 17)
# set(CMAKE_CUDA17_EXTENSION_COMPILE_OPTION "-std=c++17")
# set(CMAKE_CUDA_FLAGS "${CMAKE_CUDA_FLAGS} -arch sm_86 --extended-lambda --use_fast_math ")
# project(ggr LANGUAGES C CXX CUDA)
# file(GLOB_RECURSE SRC_FILES_GR2 ${CMAKE_CURRENT_SOURCE_DIR}/*.cpp)
# file(GLOB_RECURSE SRC_FILES_GR2_CUDA ${CMAKE_CURRENT_SOURCE_DIR}/*.cu)
# add_executable(gpugr_cpp PyBindCppMain.cpp
# ${CMAKE_CURRENT_SOURCE_DIR}/../io_parser/gp/GPDatabase.cpp
# ${SRC_FILES_GR2}
# ${SRC_FILES_GR2_CUDA})
# set_target_properties(gpugr_cpp PROPERTIES CUDA_RESOLVE_DEVICE_SYMBOLS ON)
# set_target_properties(gpugr_cpp PROPERTIES CUDA_SEPARABLE_COMPILATION ON)
# set_target_properties(gpugr_cpp PROPERTIES LINK_FLAGS "-Wl,--whole-archive -rdynamic -lpthread -Wl,--no-whole-archive")
# target_include_directories(
# gpugr_cpp PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/.. ${PROJECT_SOURCE_DIR}/cpp_to_py ${FLUTE_INCLUDE_DIR} ${TORCH_INCLUDE_DIRS})
# target_link_libraries(
# gpugr_cpp PRIVATE torch ${TORCH_PYTHON_LIBRARY} flute xplace_common)
# target_compile_options(gpugr_cpp PRIVATE $<$<COMPILE_LANGUAGE:CUDA>:
# -arch=sm_86
# --use_fast_math
# -std=c++17
# >)
# install(TARGETS
# gpugr_cpp
# DESTINATION ${XPLACE_LIB_DIR})

View File

@ -0,0 +1,80 @@
#include "common/common.h"
#include "common/db/Database.h"
#include "gpugr/db/GRDatabase.h"
#include "gpugr/gr/RouteForce.h"
#include "flute.h"
namespace Xplace {
bool loadGRParams(const pybind11::dict& kwargs) {
gr::grSetting.reset();
// ----- design related options -----
if (kwargs.contains("device_id")) {
gr::grSetting.deviceId = kwargs["device_id"].cast<int>();
}
if (kwargs.contains("route_xSize")) {
gr::grSetting.routeXSize = kwargs["route_xSize"].cast<int>();
}
if (kwargs.contains("route_ySize")) {
gr::grSetting.routeYSize = kwargs["route_ySize"].cast<int>();
}
if (kwargs.contains("csrn_scale")) {
gr::grSetting.csrnScale = kwargs["csrn_scale"].cast<int>();
}
if (kwargs.contains("rrrIters")) {
gr::grSetting.rrrIters = kwargs["rrrIters"].cast<int>();
}
if (kwargs.contains("route_guide")) {
gr::grSetting.routeGuideFile = kwargs["route_guide"].cast<std::string>();
}
return true;
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
pybind11::class_<gr::GRDatabase, std::shared_ptr<gr::GRDatabase>>(m, "GRDatabase")
.def(pybind11::init<std::shared_ptr<db::Database>, std::shared_ptr<gp::GPDatabase>>())
.def("report_gr_stat", &gr::GRDatabase::reportGRStat)
.def("setup_capacity", &gr::GRDatabase::setupCapacity)
.def("setup_wiredist", &gr::GRDatabase::setupWireDist)
.def("setup_obs", &gr::GRDatabase::setupObs)
.def("setup_grnets", &gr::GRDatabase::setupGrNets);
pybind11::class_<gr::RouteForce, std::shared_ptr<gr::RouteForce>>(m, "RouteForce")
.def(pybind11::init<std::shared_ptr<gr::GRDatabase>>())
.def("run_ggr", &gr::RouteForce::run_ggr)
.def("num_ovfl_nets", &gr::RouteForce::getNumOvflNets)
.def("gcell_steps", &gr::RouteForce::getGcellStep)
.def("microns", &gr::RouteForce::getMicrons)
.def("layer_pitch", &gr::RouteForce::getLayerPitch)
.def("layer_width", &gr::RouteForce::getLayerWidth)
.def("dmd_map", &gr::RouteForce::getDemandMap, py::return_value_policy::move)
.def("cap_map", &gr::RouteForce::getCapacityMap, py::return_value_policy::move)
.def("route_grad", &gr::RouteForce::calcRouteGrad, py::return_value_policy::move)
.def("filler_route_grad", &gr::RouteForce::calcFillerRouteGrad, py::return_value_policy::move)
.def("pseudo_grad", &gr::RouteForce::calcPseudoPinGrad, py::return_value_policy::move)
.def("inflate_ratio", &gr::RouteForce::calcNodeInflateRatio, py::return_value_policy::move)
.def("inflate_pin_rel_cpos", &gr::RouteForce::calcInflatedPinRelCpos, py::return_value_policy::move);
m.def("create_grdatabase", [](std::shared_ptr<db::Database> rawdb, std::shared_ptr<gp::GPDatabase> gpdb) {
logger.enable_logger();
std::shared_ptr<gr::GRDatabase> grdb = std::make_shared<gr::GRDatabase>(rawdb, gpdb);
logger.reset_logger();
return grdb;
});
m.def("create_routeforce", [](std::shared_ptr<gr::GRDatabase> grdb) {
logger.enable_logger();
std::shared_ptr<gr::RouteForce> routeforce = std::make_shared<gr::RouteForce>(grdb);
logger.reset_logger();
return routeforce;
});
m.def("load_gr_params", &loadGRParams, "Parse input args to DB and return graph information");
m.def("read_flute", &Flute::readLUT, "Read Flute LUT");
}
} // namespace Xplace

132
cpp_to_py/gpugr/README.md Normal file
View File

@ -0,0 +1,132 @@
# GGR: Superfast Full-Scale GPU-Accelerated Global Routing
GGR is a superfast full-scale GPU-accelerated global router developed by the research team supervised by Prof. Evangeline F. Y. Young and Prof. Martin D.F. Wong at The Chinese University of Hong Kong (CUHK). It includes an an efficient and high quality Z-shape pattern routing and a GPU-accelerated maze router GAMER.
More details are in the following paper:
Shiju Lin, Jinwei Liu, Evangeline F.Y. Young and Martin D.F. Wong. "[GAMER: GPU-Accelerated Maze Routing](https://ieeexplore.ieee.org/document/9799536)". In IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems, vol. 42, no. 2, pp. 583-593, Feb. 2023.
Shiju Lin and Martin D. F. Wong. "[Superfast Full-Scale CPU-Accelerated Global Routing](https://doi.org/10.1145/3508352.3549474)". In Proceedings of the 41st IEEE/ACM International Conference on Computer-Aided Design (ICCAD '22). Association for Computing Machinery, New York, NY, USA, Article 51, 1–8.
**GGR is integrated in Xplace now!**
## Notes and Limitations
- We defaultly use `N=0` (without CUDA graph optimization). When `N >= 1`, unknown error would occur during terminating the Python program. To enable CUDA graph optimization, please set `use_tf = true` in `cpp_to_py/gpugr/gr/MazeRoute.cu` and re-compile the project.
- GGR is **deterministic** when `N=0` but not test in `N >= 1`.
- We currently only support LEF/DEF format.
- The runtime of this version is a little bit slower than the version described in the GGR paper because we use a stronger but slower parser and make the algorithm deterministic yet robust while sacrificing the runtime.
## Parameters
Please refer to `cpp_to_py/gpugr/PyBindCppMain.cpp`.
- `device_id`: GPU id.
- `route_xSize` / `route_ySize`: given GR GridGraph size. If the GridGraph size is `0`, use the GridGraph definition in DEF file. If the GridGraph size is `0` and there is no definition in DEF, we defaultly set it as `512`.
- `rrrIters`: the number of rip-up and re-route iterations (maze route). If `rrrIters = 0`, perform pattern route only.
- `csrn_scale`: the size of coarsen grid in maze routing.
- `route_guide`: the file name of output route guide.
## Citation
If you find **GGR** useful in your research, please consider to cite:
```bibtex
@inproceedings{lin2022ggr,
author = {Lin, Shiju and Wong, Martin D. F.},
booktitle = {Proceedings of the 41st IEEE/ACM International Conference on Computer-Aided Design},
title = {Superfast Full-Scale CPU-Accelerated Global Routing},
year = {2022},
}
@article{lin2023gamer,
author={Lin, Shiju and Liu, Jinwei and Young, Evangeline F. Y. and Wong, Martin D. F.},
journal={IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems},
title={GAMER: GPU-Accelerated Maze Routing},
year={2023},
```
## Example
```python
# main_test_gr.py
import torch
import time
from utils import IOParser
from cpp_to_py import gpugr
from src import Flute
from src.core.route_force import calc_gr_wl_via, estimate_num_shorts
num_threads = 20
gpu_id = 0
Flute.register(num_threads)
torch.cuda.synchronize("cuda:{}".format(gpu_id))
# 1) Benchmark Setting
root = "your_path"
design_name = "ispd19_test9"
params = {
"benchmark": "iccad2019",
"lef": "%s/%s/%s.input.lef" % (root, design_name, design_name),
"def": "%s/%s/%s.input.def" % (root, design_name, design_name),
"design_name": design_name,
}
route_guide_file = "test.guide"
# 2) LEF/DEF Parser
print("--- Start GR ---")
start_gr_time = time.time()
parser = IOParser()
rawdb, gpdb = parser.read(
params, verbose_log=True, lite_mode=True, random_place=False, num_threads=num_threads
)
# 3) Construct Global Routing Database and Run GGR
gpugr.load_gr_params(
{
"device_id": gpu_id,
"route_xSize": 0,
"route_ySize": 0,
"rrrIters": 1,
"route_guide": route_guide_file,
}
)
grdb = gpugr.create_grdatabase(rawdb, gpdb)
routeforce = gpugr.create_routeforce(grdb)
routeforce.run_ggr()
end_gr_time = time.time()
print("--- End GR ---")
# 4) Report Global Routing Statistics
skip_m1_route = True
m1direction = gpdb.m1direction() # 0 for H, 1 for V, metal1's layer idx is 0
hId = 1 if m1direction else 0
vId = 0 if m1direction else 1
if skip_m1_route:
hId = hId + 2 if hId == 0 else hId
vId = vId + 2 if vId == 0 else vId
dmd_map, wire_dmd_map, via_dmd_map = routeforce.dmd_map()
cap_map: torch.Tensor = routeforce.cap_map()
cg_mapH = dmd_map[hId::2].sum(dim=0) / cap_map[hId::2].sum(dim=0)
cg_mapV = dmd_map[vId::2].sum(dim=0) / cap_map[vId::2].sum(dim=0)
cg_mapHV = torch.stack((cg_mapH, cg_mapV))
cg_mapHV = torch.where(cg_mapHV > 1, cg_mapHV - 1, 0)
numOvflNets = routeforce.num_ovfl_nets()
gr_wirelength, gr_numVias = calc_gr_wl_via(grdb, routeforce)
gr_numShorts = estimate_num_shorts(routeforce, gpdb, cap_map, wire_dmd_map, via_dmd_map)
gr_time = end_gr_time - start_gr_time
print(
"#OvflNets: %d, GR WL: %d, GR #Vias: %d, #EstShorts: %d | GR Time: %.4f"
% (numOvflNets, gr_wirelength, gr_numVias, gr_numShorts, gr_time)
)
```
## Contact
[Shiju Lin](https://appsrv.cse.cuhk.edu.hk/~sjlin/) (sjlin@cse.cuhk.edu.hk) and [Lixin Liu](https://liulixinkerry.github.io/) (lxliu@cse.cuhk.edu.hk)
## License
GGR is an open source project licensed under a BSD 3-Clause License that can be found in the [LICENSE](../../LICENSE) file.

File diff suppressed because it is too large Load Diff

View File

@ -0,0 +1,106 @@
#pragma once
#include "GRSetting.h"
#include "GrNet.h"
#include "common/common.h"
#include "common/db/Database.h"
#include "io_parser/gp/GPDatabase.h"
namespace gr {
enum class AggrParaRunSpace { DEFAULT, LARGER_WIDTH, LARGER_LENGTH };
class RectOnLayer {
public:
int layer = -1;
int lx, ly, hx, hy;
RectOnLayer() {}
RectOnLayer(const int layer_, const int lx_, const int ly_, const int hx_, const int hy_)
: layer(layer_), lx(lx_), ly(ly_), hx(hx_), hy(hy_) {}
int getDirRange(unsigned i) { return (i == 0) ? hx - lx : hy - ly; }
};
// class GRNet {
// std::vector<std::vector<std::tuple<int, int, int>>> global_pins;
// std::vector<std::vector<RectOnLayer>> pins; // GCell locations
// };
class GRDatabase {
public:
db::Database& rawdb;
gp::GPDatabase& gpdb;
// GR Obstacle
std::vector<RectOnLayer> fixObs;
std::vector<RectOnLayer> movObs; // for movable nodes
std::vector<std::vector<int>> tracks;
// Metal Layer
std::vector<int> layerWidth;
std::vector<int> layerPitch;
std::vector<int> defaultSpacing;
std::vector<int> maxEOLSpacingVec; // from all spacing types
std::vector<int> maxEOLWidthVec; // from all spacing types
// Design
int nLayers;
int xSize;
int ySize;
int nMaxGrid;
int gridGraphSize;
int m1direction; // layer 0 direction, 'v': 1, 'h': 0
double m2pitch;
int microns;
int mainGcellStepX, mainGcellStepY;
std::vector<std::vector<int>> gridlines;
std::vector<std::vector<int>> gridCenters;
int ISPD19 = 0;
int ISPD18 = 0;
int METAL5 = 0;
int csrnScale = 0;
int cgxsize, cgysize;
// GR variables
std::vector<float> capacity, wireDist; // init once
std::vector<float> fixTmpUsage, fixTmpLength; // init once
std::vector<float> movTmpUsage, movTmpLength; // update dynamically
std::vector<float> fixedUsage, fixedLength; // variables for GR
std::vector<GrNet> grNets;
std::vector<int> gpdbPinId2gbPinId;
GRDatabase(std::shared_ptr<db::Database> rawdb_, std::shared_ptr<gp::GPDatabase> gpdb_);
~GRDatabase();
void setupCapacity();
void setupCapacityBookshelf();
void setupWireDist();
void addFixObs();
void addMovObs();
void updateUsageLength();
void setupObs();
void setupGrNets();
void resetGrNetsRoute();
void resetGrNets() { grNets.clear(); };
std::pair<int, int> reportGRStat();
void writeGuides(std::string outputFile);
int encodeId(int l, int x, int y);
tuple<int, int, int, int> getOrientOffset(int orient, int lx, int ly, int hx, int hy);
int getEOLSpace(int width, int l);
int getParallelRunSpace(int l, int width, int length);
utils::PointT<int> getObsMargin(RectOnLayer box, AggrParaRunSpace aggr);
utils::IntervalT<int> rangeSearchTracks(const utils::IntervalT<int>& locRange, int layerIdx);
void markObs(std::vector<RectOnLayer>& allObs, std::vector<float>& wireUsage, std::vector<float>& wireTotalLength);
void markObsBookShelf(std::vector<RectOnLayer>& allObs,
std::vector<float>& wireUsage,
std::vector<float>& wireTotalLength);
void addCellObs(std::vector<RectOnLayer>& allObs, db::Cell* cell);
};
} // namespace gr

View File

@ -0,0 +1,19 @@
#include "GRSetting.h"
namespace gr {
void GRSetting::reset() {
deviceId = 0;
routeXSize = 0;
routeYSize = 0;
csrnScale = 0;
rrrIters = 0;
routeGuideFile = "";
}
GRSetting grSetting;
} // namespace gr

View File

@ -0,0 +1,25 @@
#pragma once
#include <string>
namespace gr {
class GRSetting {
public:
// 1. SystemSetting
int deviceId = 0;
// 2. Gridgraph setting
int routeXSize = 0;
int routeYSize = 0;
int csrnScale = 0;
// 3. The number of Rip-up and Reroute iterations (if 0, only PR is invoked)
int rrrIters = 0;
std::string routeGuideFile = "";
void reset();
};
extern GRSetting grSetting;
} // namespace gr

View File

@ -0,0 +1,20 @@
#include "GrNet.h"
namespace gr {
bool GrNet::needToRoute() {
std::unordered_map<int, int> cnt;
for (auto e : pins) {
for (auto f : e) {
cnt[f]++;
}
}
for (auto e : pins) {
for (auto f : e) {
if (cnt[f] == pins.size()) return false;
}
}
return true;
}
}

View File

@ -0,0 +1,36 @@
#pragma once
#include <vector>
#include <unordered_map>
namespace gr {
class GrNet {
public:
void setPins(const std::vector<std::vector<int>>& p) { pins = p; }
void setBoundingBox(int lx, int ly, int ux, int uy) {
lowerx = lx;
lowery = ly;
upperx = ux;
uppery = uy;
}
void setWires(const std::vector<int>& w) { wires = w; }
void setVias(const std::vector<int>& v) { vias = v; }
void resetRoute() { wires.clear(), vias.clear(); }
void addVias();
bool needToRoute();
void setNoRoute() { noroute = 1; }
int area() { return (upperx - lowerx + 1) * (uppery - lowery + 1); }
int hpwl() { return upperx + uppery - lowerx - lowery; }
const std::vector<int>& getWires() { return wires; }
const std::vector<int>& getVias() { return vias; }
const std::vector<std::vector<int>>& getPins() { return pins; }
int lowerx, lowery, upperx, uppery, noroute = 0;
std::vector<int> points;
std::vector<int> pin2gbpinId;
std::vector<std::vector<int>> pin2gpdbPinIds;
private:
std::vector<std::vector<int>> pins;
std::vector<int> wires, vias;
};
} // namespace gr

View File

@ -0,0 +1,962 @@
#include "GPURouter.h"
#include "InCellUsage.cuh"
#include <cstdio>
namespace gr {
constexpr int MAX_ROUTE_LEN_PER_PIN = 130; // too large may exceed the maximum GPU memory
constexpr int INF = 10000000;
constexpr int MAX_COST = 10000000;
#define BLOCK_SIZE 512
#define BLOCK_NUMBER(n) (((n) + (BLOCK_SIZE) - 1) / BLOCK_SIZE)
__managed__ int STAMP = 0, wireLen, viaLen;
void GPURouter::initialize(int device_id, int layer, int x, int y, int N_, int cgxsize_, int cgysize_, int direction, int csrn_scale) {
gpuMR.startGPU(device_id, layer, cgxsize_, cgysize_);
DEVICE_ID = device_id;
DIRECTION = direction;
COARSENING_SCALE = csrn_scale;
cgxsize = cgxsize_;
cgysize = cgysize_;
LAYER = layer;
N = N_;
X = x;
Y = y;
int gridGraphSize = LAYER * N * N;
cudaMalloc(&dist, (MAX_BATCH_SIZE + 6) * gridGraphSize * sizeof(int));
cudaMalloc(&prev, (MAX_BATCH_SIZE + 6) * gridGraphSize * sizeof(int));
cudaMalloc(&capacity, gridGraphSize * sizeof(float));
cudaMalloc(&wireDist, gridGraphSize * sizeof(float));
cudaMalloc(&fixedLength, gridGraphSize * sizeof(float));
cudaMalloc(&fixed, gridGraphSize * sizeof(float));
cudaMalloc(&wires, gridGraphSize * sizeof(int));
cudaMalloc(&vias, gridGraphSize * sizeof(int));
cudaMemset(wires, 0, sizeof(int) * gridGraphSize);
cudaMemset(vias, 0, sizeof(int) * gridGraphSize);
cudaMalloc(&modifiedWire, gridGraphSize * sizeof(int));
cudaMalloc(&modifiedVia, gridGraphSize * sizeof(int));
cudaMalloc(&viaCost, gridGraphSize * sizeof(dtype));
cudaMalloc(&cost, gridGraphSize * sizeof(dtype));
cudaMalloc(&costSum, gridGraphSize * sizeof(int64_t));
cudaMalloc(&cell_resource, gridGraphSize * sizeof(float));
cudaMalloc(&isOverflowWire, gridGraphSize * sizeof(int));
cudaMalloc(&isOverflowVia, gridGraphSize * sizeof(int));
cudaMalloc(&unitShortCostDiscounted, LAYER * sizeof(float));
cudaMallocManaged(&allpins, MAX_BATCH_SIZE * MAX_PIN_SIZE_PER_NET * sizeof(int));
}
GPURouter::~GPURouter() {
gpuMR.endGPU();
cudaFree(dist);
cudaFree(prev);
cudaFree(capacity);
cudaFree(wireDist);
cudaFree(fixedLength);
cudaFree(fixed);
cudaFree(wires);
cudaFree(vias);
cudaFree(modifiedWire);
cudaFree(modifiedVia);
cudaFree(viaCost);
cudaFree(cost);
cudaFree(costSum);
cudaFree(cell_resource);
cudaFree(isOverflowWire);
cudaFree(isOverflowVia);
cudaFree(unitShortCostDiscounted);
cudaFree(allpins);
if(pins != nullptr) cudaFree(pins);
if(pinNum != nullptr) cudaFree(pinNum);
if(pinNumOffset != nullptr) cudaFree(pinNumOffset);
if(routes != nullptr) cudaFree(routes);
if(routesOffset != nullptr) cudaFree(routesOffset);
if(isOverflowNet != nullptr) cudaFree(isOverflowNet);
if(points != nullptr) cudaFree(points);
if(gbpoints != nullptr) cudaFree(gbpoints);
if(gbpinRoutes != nullptr) cudaFree(gbpinRoutes);
if(gbpin2netId != nullptr) cudaFree(gbpin2netId);
if(plPinId2gbPinId != nullptr) cudaFree(plPinId2gbPinId);
if(routesOffsetCPU != nullptr) { delete[] routesOffsetCPU; }
if(pinNumCPU != nullptr) { delete[] pinNumCPU; }
}
void GPURouter::setUnitViaMultiplier(float value) {
unitViaMultiplier = value;
}
void GPURouter::setUnitVioCost(vector<float>& values, float discount) {
//printf("??? setUnitVioCost %.2f\n", discount);
float temp[100];
for(int i = 0; i < LAYER; i++)
temp[i] = values[i] * discount;
cudaMemcpy(unitShortCostDiscounted, temp, LAYER * sizeof(float), cudaMemcpyHostToDevice);
}
void GPURouter::setLogisticSlope(float value) {
logisticSlope = value;
}
void GPURouter::setUnitViaCost(float value) {
unitViaCost = value;
}
void GPURouter::setMap(const vector<float> &cap, const vector<float> &wir, const vector<float> &fixedL, const vector<float> &fix) {
int gridGraphSize = LAYER * N * N;
auto copy = [&] (const vector<float> &vec, float *target) {
cudaMemcpy(target, vec.data(), gridGraphSize * sizeof(float), cudaMemcpyHostToDevice);
};
copy(cap, capacity);
copy(wir, wireDist);
copy(fixedL, fixedLength);
copy(fix, fixed);
}
__global__ void calculateCellResource(float *cell_resource, int *wires, float *fixed, int *vias, const float *capacity, int N, int LAYER, int tot) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if(idx >= tot) return;
cell_resource[idx] = cellResource(idx, wires, fixed, vias, capacity, N, LAYER);
}
__global__ void calculateCoarseCost(float *cell_resource, int *cost, int *wires, float *fixed, int *vias, float *capacity, int N, int xsize, int ysize, int X, int Y, int LAYER, int DIRECTION, int COARSENING_SCALE) {
int layer = blockIdx.x / xsize, x = blockIdx.x % xsize, y = threadIdx.x;
if(layer == 0) {
cost[blockIdx.x * blockDim.x + threadIdx.x] = 10000000;
return;
}
int minx = x * COARSENING_SCALE, maxx = min(X - 1, x * COARSENING_SCALE + COARSENING_SCALE - 1);
int miny = y * COARSENING_SCALE, maxy = min(Y - 1, y * COARSENING_SCALE + COARSENING_SCALE - 1);
float ans = 0;
if(DIRECTION ^ (layer & 1)) {
if(y + 1 < ysize) {
float sum = 0;
for(int i = minx; i <= maxx; i++)
for(int j = miny; j <= maxy; j++) {
//if(layer == 1 && x == 3 && y == 18)
// printf("GPU %d %d: %.2lf\n", i, j, cellResource(layer * N * N + i * N + j, wires, fixed, vias, capacity, N, LAYER));
sum += cell_resource[layer * N * N + i * N + j];
}
ans += 1.0 * COARSENING_SCALE / max(0.1, sum / (maxx - minx + 1) / (maxy - miny + 1));
//if(layer == 1 && x == 3 && y == 18)
// printf("sum: %.2lf\n", sum);
sum = 0;
for(int i = minx; i <= maxx; i++)
for(int j = miny + COARSENING_SCALE; j <= min(Y - 1, maxy + COARSENING_SCALE); j++) {
//if(layer == 1 && x == 3 && y == 18)
// printf("GPU %d %d: %.2lf\n", i, j, cellResource(layer * N * N + i * N + j, wires, fixed, vias, capacity, N, LAYER));
sum += cell_resource[layer * N * N + i * N + j];
}
ans += 1.0 * COARSENING_SCALE / max(0.1, sum / (maxx - minx + 1) / (min(Y - 1, maxy + COARSENING_SCALE) - miny - COARSENING_SCALE + 1));
//if(layer == 1 && x == 3 && y == 18)
// printf("sum: %.2lf\n", sum);
cost[layer * xsize * ysize + x * ysize + y] = 100 * ans;
}
} else {
if(x + 1 < xsize) {
float sum = 0;
for(int i = minx; i <= maxx; i++)
for(int j = miny; j <= maxy; j++)
sum += cell_resource[layer * N * N + j * N + i];
ans += 1.0 * COARSENING_SCALE / max(0.1, sum / (maxx - minx + 1) / (maxy - miny + 1));
sum = 0;
for(int i = minx + COARSENING_SCALE; i <= min(X - 1, maxx + COARSENING_SCALE); i++)
for(int j = miny; j <= maxy; j++)
sum += cell_resource[layer * N * N + j * N + i];
ans += 1.0 * COARSENING_SCALE / max(0.1, sum / (min(X - 1, maxx + COARSENING_SCALE) - minx - COARSENING_SCALE + 1) / (maxy - miny + 1));
cost[layer * xsize * ysize + x * ysize + y] = 100 * ans;
}
}
}
__global__ void calculateCoarseVia(float *cell_resource, int *coarseVia, int *wires, float *fixed, int *vias, float *capacity, int N, int LAYER, int xsize, int ysize, int X, int Y, int DIRECTION, int COARSENING_SCALE) {
int layer = blockIdx.x / xsize, x = blockIdx.x % xsize, y = threadIdx.x;
int minx = x * COARSENING_SCALE, maxx = min(X - 1, x * COARSENING_SCALE + COARSENING_SCALE - 1);
int miny = y * COARSENING_SCALE, maxy = min(Y - 1, y * COARSENING_SCALE + COARSENING_SCALE - 1);
if(layer + 1 < LAYER) {
float sum = 0, ans = 0;
for(int i = minx; i <= maxx; i++)
for(int j = miny; j <= maxy; j++)
if((layer & 1) ^ DIRECTION)
sum += cell_resource[layer * N * N + i * N + j];
else
sum += cell_resource[layer * N * N + j * N + i];
ans += 1.0 / max(0.1, sum / (maxx - minx + 1) / (maxy - miny + 1));
sum = 0;
for(int i = minx; i <= maxx; i++)
for(int j = miny; j <= maxy; j++)
if(!(layer & 1) ^ DIRECTION)
sum += cell_resource[(layer + 1) * N * N + i * N + j];
else
sum += cell_resource[(layer + 1) * N * N + j * N + i];
ans += 1.0 / max(0.1, sum / (maxx - minx + 1) / (maxy - miny + 1));
coarseVia[layer * xsize * ysize + x * ysize + y] = 100 * ans;
}
}
__global__ void initMap(dtype *dist, int *prev, int total, int N) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if(idx < N) {
dist[idx] = INF;
prev[idx] = idx % total;
//for(int i = 0; i < N; i++)
// dist[idx + i * total] = INF, prev[idx + i * total] = idx;
}
}
/*
__managed__ float minval, maxval = 0;
__managed__ unsigned long long allroutecost = 0;
#define debug -1
__global__ void traceBack(int *modifiedWire, int *modifiedVia, dtype *dist, int *prev, int *wires, int *vias, int *pins, int *routes, int *routesOffset, int *cudaPos, int netId, int N, int flag) {
routes += routesOffset[netId];
dtype minDist = INF;
int p = -1, lef = -1, rig = -1, finished = 1;
for(int i = 0, cur = 1; i < pins[0]; i++) {
//if(netId == debug || debug == -2)
// printf("pin %d\n", i);
for(int j = 1; j <= pins[cur]; j++) {
int pos = pins[cur + j];
//if(netId == debug || debug == -2)
// printf("access point %d=(%d,%d,%d) dist: %d\n", cudaPos[pos], cudaPos[pos] / N / N, cudaPos[pos] % (N * N) / N, cudaPos[pos] % N, dist[pos]);
if(dist[pos] == INF) finished = 0;
if(dist[pos] > 0 && dist[pos] < minDist) {
minDist = dist[pos];
p = pos;
lef = cur + 1;
rig = cur + pins[cur];
}
}
cur += pins[cur] + 1;
}
if(flag && finished == 0)
printf("REAL ERROR: DISCONNECTED NET%d\n", netId);
if(p == -1) {
//printf("WARNING: No pin is connected in this round. NET ID: %d\n", netId);
return;
}
//if(netId == debug)
// printf("Start tracing result: %d %d %d\n", p / N / N, p % (N * N) / N, p % N);
maxval = max(maxval, 1.0 * dist[p]);
int expected = p;
while(dist[expected] > 0)
expected = prev[expected];
while(dist[p] > 0) {
//if(netId == debug || debug == -2)
// printf("%d %d (%d, %d, %d): %d; prev: %d %d (%d, %d, %d): %d\n", p, cudaPos[p], cudaPos[p] / N / N, cudaPos[p] % (N * N) / N, cudaPos[p] % N, dist[p], prev[p], cudaPos[prev[p]], cudaPos[prev[p]] / N / N, cudaPos[prev[p]] % (N * N) / N, cudaPos[prev[p]] % N, dist[prev[p]]);
int minp = cudaPos[p], maxp = cudaPos[prev[p]], pre = prev[p];
if(minp > maxp) {
int temp = minp;
minp = maxp;
maxp = temp;
}
int lmin = minp / N / N, lmax = maxp / N / N, x = minp % (N * N) / N, y = minp % N;
if(lmin != lmax) {
for(int i = lmin; i <= lmax; i++) {
int idx = i * N * N + ((i - lmin) % 2 ? y * N + x : x * N + y);
if(i < lmax) {
atomicAdd(vias + idx, 1);
routes[routes[0]++] = idx;
routes[routes[0]++] = -1;
modifiedVia[idx] = STAMP;
}
//if(idx != cudaPos[pre])
// dist[idx] = 0;
}
} else {
routes[routes[0]++] = minp;
routes[routes[0]++] = maxp - minp;
for(int i = minp; i <= maxp; i++) {
if(i < maxp) {
atomicAdd(wires + i, 1);
modifiedWire[i] = STAMP;
}
//if(i != cudaPos[pre])
// dist[i] = 0;
}
}
//if(dist[pre] + sum != distp)
// printf("ERROR: INCONSISTENT COST; net %d sum %d\n", netId, sum);
dist[p] = 0;
p = pre;
}
if(expected != p)
printf("netid %d: TRACEBACK ERROR\n", netId);
//printf("Expected tracing result: %d %d %d\n", expected / N / N, expected % (N * N) / N, expected % N);
//if(netId == debug)
// printf("Final tracing result: %d %d %d\n", p / N / N, p % (N * N) / N, p % N);
for(int i = lef; i <= rig; i++) {
dist[pins[i]] = 0;
//if(debug == netId)
// printf("setting 0 distance: %d=%d,%d,%d\n", pins[i], pins[i] / N / N, pins[i] / N % N, pins[i] % N);
}
if(routes[0] > routesOffset[netId + 1] - routesOffset[netId])
printf("%d **ERROR: ROUTE_LEN_PER_PIN INSUFFICENT; \n", routes[0]);
}
*/
__global__ void calculateWireCost(dtype *cost, float *wireDist, float *fixed, float *fixedLength, int *wires, int *vias, float *capacity, float *unitShortCost, float logisticSlope, int N, int LAYER) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if(!(idx < LAYER * N * N && idx % N + 1 < N)) return;
if(capacity[idx] < 0.01) {
cost[idx] = INF;
return;
}
int expectedLen = (fixed[idx] * fixedLength[idx] + wires[idx] * wireDist[idx]) / capacity[idx];
float remain = capacity[idx] - (fixed[idx] + wires[idx] + 1 + twoCellsViaUsage(idx, vias, N, LAYER));
//if(idx == 2 * N * N + 63)
// for(int i = 0; i < 9; i++)
// printf("%d unit %.2lf\n", i, unitShortCost[i]);
float result = wireDist[idx] + expectedLen / (1.0 + exp(logisticSlope * remain)) * unitShortCost[idx / N / N];
int result_r = static_cast<int>(result);
cost[idx] = (result_r > MAX_COST ? MAX_COST : result_r);
}
__global__ void calculateViaCost(int *wires, float *fixed, float *capacity, dtype *viaCost, float unitViaMultiplier, float unitViaCost, float logisticSlope, int N, int LAYER) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
//if(idx >= viaLen) return;
//idx = ids[idx];
//if(modified[idx] != STAMP) return;
if(idx >= (LAYER - 1) * N * N) return;
int layer = idx / N / N + 1, y = idx % (N * N) / N, x = idx % N;
float result = unitViaCost * (unitViaMultiplier + inCellViaCost(idx, wires, fixed, capacity, logisticSlope, N) + inCellViaCost(layer * N * N + x * N + y, wires, fixed, capacity, logisticSlope, N));
//result /= 100;
int result_r = static_cast<int>(result);
viaCost[idx] = (result_r > MAX_COST ? MAX_COST : result_r);
}
__global__ void setStartCells(dtype *dist, int *pins, int N, int T) {
pins += blockIdx.x * N + 2;
dist += blockIdx.x * T;
for(int i = 1; i <= pins[0]; i++) {
dist[pins[i]] = 0;
}
}
__global__ void markUnrouteUsage(int *pins, int *vias, int cnt) {
int cur = blockIdx.x * blockDim.x + threadIdx.x;
if(cur < cnt) atomicAdd(vias + pins[cur], 1);
}
__global__ void markOverflowWires(const float *capacity, int *wires, int *vias, float *fixed, int *isOverflow, int N, int LAYER) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if(idx < LAYER * N * N && idx % N + 1 < N)
isOverflow[idx] = (wires[idx] + fixed[idx] + twoCellsViaUsage(idx, vias, N, LAYER) > capacity[idx]);
}
__global__ void markOverflowVias(const float *capacity, int *wires, int *vias, float *fixed, int *isOverflow, int N, int LAYER) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if(idx >= (LAYER - 1) * N * N) return;
int layer = idx / N / N + 1, y = idx % (N * N) / N, x = idx % N;
int upper_idx = layer * N * N + x * N + y;
isOverflow[idx] = (inCellUsedArea(idx, wires, fixed, N) > capacity[idx] || inCellUsedArea(upper_idx, wires, fixed, N) > capacity[upper_idx]);
}
__global__ void markOverflowNets(int *isOverflowVia, int *isOverflowWire, int *isOverflowNet, int *routes, int *routesOffset, int NET_NUM) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if(idx >= NET_NUM) return;
routes += routesOffset[idx];
isOverflowNet[idx] = 0;
if(routes[0] == -1) {
// printf("resolving failed net from pattern routing. netId: %d\n", idx);
isOverflowNet[idx] = 1;
return;
}
//if(routes[0] == 0) return;
for(int i = 1; i < routes[0]; i += 2) if(routes[i + 1] == -1) {
if(isOverflowVia[routes[i]]) {
isOverflowNet[idx] = 1;
return;
}
} else {
for(int j = 0; j < routes[i + 1]; j++)
if(isOverflowWire[routes[i] + j]) {
isOverflowNet[idx] = 1;
return;
}
}
}
__global__ void commit(int *routes, int *routesOffset, int *wires, int *vias, int NET_NUM) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if(idx >= NET_NUM) return;
routes += routesOffset[idx];
for(int i = 1; i < routes[0]; i += 2) if(routes[i + 1] > 0) {
for(int j = 0; j < routes[i + 1]; j++)
atomicAdd(wires + routes[i] + j, 1);
} else
atomicAdd(vias + routes[i], 1);
}
__global__ void ripupOverflowNets(int *isOverflowNet, int *routes, int *routesOffset, int *wires, int *vias, int NET_NUM) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if(idx >= NET_NUM || isOverflowNet[idx] == 0) return;
routes += routesOffset[idx];
for(int i = 1; i < routes[0]; i += 2) if(routes[i + 1] > 0) {
for(int j = 0; j < routes[i + 1]; j++)
atomicAdd(wires + routes[i] + j, -1);
} else
atomicAdd(vias + routes[i], -1);
routes[0] = 1;
}
__global__ void getWires(int N, int LAYER, int *wires, int *ids) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
int layer = idx / (N * N), x = idx / N % N, y = idx % N;
// bool needsUpdate = false;
for(int dl = -1; dl <= 1; dl++) if(0 <= layer + dl && layer + dl < LAYER)
for(int dx = -1; dx <= 1; dx++) if(0 <= x + dx && x + dx < N)
for(int dy = -1; dy <= 1; dy++) if(0 <= y + dy && y + dy < N)
if(wires[(layer + dl) * N * N + (x + dx) * N + (y + dy)] == STAMP) {
ids[atomicAdd(&wireLen, 1)] = idx;
return;
}
}
__global__ void getVias(int N, int LAYER, int *vias, int *ids) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
int layer = idx / (N * N), x = idx / N % N, y = idx % N;
// bool needsUpdate = false;
for(int dl = -1; dl <= 1; dl++) if(0 <= layer + dl && layer + dl < LAYER)
for(int dx = -1; dx <= 1; dx++) if(0 <= x + dx && x + dx < N)
for(int dy = -1; dy <= 1; dy++) if(0 <= y + dy && y + dy < N)
if(vias[(layer + dl) * N * N + (x + dx) * N + (y + dy)] == STAMP) {
ids[atomicAdd(&viaLen, 1)] = idx;
return;
}
}
__global__ void calculateCostSum(int LAYER, int N, int *cost, int64_t *costSum) {
extern __shared__ int64_t sum[];
int a = threadIdx.x << 1, b = threadIdx.x << 1 | 1;
for(int i = 1; i < LAYER; i++) {
sum[a] = cost[i * N * N + blockIdx.x * N + N - 1 - a];
sum[b] = cost[i * N * N + blockIdx.x * N + N - 1 - b];
__syncthreads();
for(int d = 0; (1 << d) < N; d++) {
if(a >> d & 1)
sum[a] += sum[(a >> d << d) - 1];
if(b >> d & 1)
sum[b] += sum[(b >> d << d) - 1];
__syncthreads();
}
costSum[i * N * N + blockIdx.x * N + N - 1 - a] = sum[a];
costSum[i * N * N + blockIdx.x * N + N - 1 - b] = sum[b];
__syncthreads();
}
}
__global__ void printWires(int *wires, int LAYER, int N) {
for(int i = 0; i < LAYER; i++)
for(int j = 0; j < N; j++)
for(int k = 0; k < N; k++)
if(wires[i * N * N + j * N + k]) printf("%d %d ", i * N * N + j * N + k, wires[i * N * N + j * N + k]);
}
__global__ void output(int *wires, int *vias, int LAYER, int X, int Y, int N, int DIRECTION) {
for(int i = 2; i < 3; i++)
for(int j = 0; j < X; j++)
for(int k = 0; k < Y; k++)
if((i & 1) ^ DIRECTION) {
if(k + 1 < Y) printf("%d %d %d: %d\n", i, j, k, wires[i * N * N + j * N + k]);
} else {
if(j + 1 < X) printf("%d %d %d: %d\n", i, j, k, wires[i * N * N + j * N + k]);
}
/*if((i & 1) ^ DIRECTION)
for(int j = 0; j < X; j++)
for(int k = 0; k < Y - 1; k++)
printf("%d %d %d: %d\n", i, j, k, wires[i * N * N + j * N + k]);
else
for(int j = 0; j < Y; j++)
for(int k = 0; k < X - 1; k++)
printf("%d %d %d: %d\n", i, j, k, wires[i * N * N + j * N + k]);*/
}
/*
__global__ void generateBatch(int n, int *minx, int *maxx, int *miny, int *maxy, int *res, int *siz, int *used, int *flag) {
//printf("thid %d %d\n", threadIdx.x, n);
extern __shared__ int selected[];
for(int i = threadIdx.x; i < n; i += blockDim.x) used[i] = 0;
if(threadIdx.x == 0) selected[1] = 0;
__syncthreads();
while(selected[1] < n) {
int cur = selected[1];
if(threadIdx.x == 0) selected[0] = n;
__syncthreads();
for(int i = threadIdx.x; i < n; i += blockDim.x) {
if(used[i] == 0) atomicMin(selected, i);
flag[i] = 0;
}
__syncthreads();
#define allowed_overlap 2
while(selected[0] < n) {
int t = selected[0];
__syncthreads();
if(threadIdx.x == 0) res[cur++] = t, used[t] = 1, selected[0] = n;//, printf("%d ", t);
__syncthreads();
for(int offset = 0; offset < n; offset += blockDim.x) {
int i = threadIdx.x + offset;
if(i < n) {
if(used[i] == 0 && flag[i] == 0) {
if(maxx[i] + allowed_overlap < minx[t] ||
maxx[t] + allowed_overlap < minx[i] ||
maxy[i] + allowed_overlap < miny[t] ||
maxy[t] + allowed_overlap < miny[i]) atomicMin(selected, i);
else
flag[i] = 1;
}
}
__syncthreads();
if(selected[0] < n) break;
}
}
//if(threadIdx.x == 0) printf("\n");
if(threadIdx.x == 0)
siz[++siz[0]] = cur - selected[1], selected[1] = cur;
__syncthreads();
}
}
*/
__global__ void generateBatch(int n, int *minx, int *maxx, int *miny, int *maxy, int *res, int *siz, int *used) {
extern __shared__ int shared[];
if(threadIdx.x == 0) shared[0] = 0;
__syncthreads();
while(shared[0] < n) {
if(threadIdx.x == 0) shared[2] = 0;
__syncthreads();
for(int i = 0; i < n; i++) if(used[i] == 0) {
if(threadIdx.x == 0) shared[1] = 0;
__syncthreads();
#define allowed_overlap 2
if(threadIdx.x < shared[2]) {
if (maxx[i] + allowed_overlap < minx[res[threadIdx.x + shared[0]]] ||
maxx[res[threadIdx.x + shared[0]]] + allowed_overlap < minx[i] ||
maxy[i] + allowed_overlap < miny[res[threadIdx.x + shared[0]]] ||
maxy[res[threadIdx.x + shared[0]]] + allowed_overlap < miny[i]) {}
else
shared[1] = 1;
}
__syncthreads();
if(threadIdx.x == 0 && shared[1] == 0)
res[shared[0] + shared[2]++] = i, used[i] = 1;
__syncthreads();
}
if(threadIdx.x == 0) {
siz[++siz[0]] = shared[2], shared[0] += shared[2];
if(shared[2] > blockDim.x) printf("ERROR in kernel\n");
}
__syncthreads();
}
}
void GPURouter::route(vector<GrNet> &nets, int iter) {
logger.info("GPU Routing start... DIRECTION: %d", DIRECTION);
double prtime = 0, prpreparetime = 0, batchgentime = 0, timer2 = 0;
vector<int> netsToRoute;
if(iter > 0) {
logger.info("Maze Routing...");
ripupOverflowNets<<<BLOCK_NUMBER(NET_NUM), BLOCK_SIZE>>> (isOverflowNet, routes, routesOffset, wires, vias, NET_NUM);
cudaDeviceSynchronize();
for(size_t netId = 0; netId < nets.size(); netId++)
if(isOverflowNet[netId] && !nets[netId].noroute) {
netsToRoute.emplace_back(netId);
// if(nets[netId].noroute) printf("ERROR: net %d is noroute but overflow\n", netId);
}
} else {
logger.info("Pattern Routing...");
// int cnt = 0;
for(int i = 0; i < nets.size(); i++)
if(!nets[i].noroute) netsToRoute.emplace_back(i);
/*else {
auto &temp = nets[i].getPins();
for(auto e : temp)
for(auto f : e)
allpins[cnt++] = f;
if(cnt > MAX_BATCH_SIZE * MAX_PIN_SIZE_PER_NET) std::cerr << "ERROR!\n";
}
markUnrouteUsage<<<BLOCK_NUMBER(cnt), BLOCK_SIZE>>> (allpins, vias, cnt);
cudaDeviceSynchronize();*/
}
vector<int> batchSizes;
{
double t = clock();
constexpr bool check_vis_correctness = false;
std::vector<int> s = netsToRoute;
int margin = 0; // NOTE: margin == 0 is okay for PR, but do not check for MR
//int margin = iter ? 0 : 2;
std::sort(s.begin(), s.end(), [&] (int l, int r) {
int area_l = nets[l].area();
int area_r = nets[r].area();
if (area_l == area_r) {
return l > r;
}
return area_l > area_r;
//return nets[l].area() * 1.0 / nets[l].getPins().size() > nets[r].area() * 1.0 / nets[r].getPins().size();
});
// std::sort(s.begin(), s.end(), [&] (int l, int r) {
// int hpwl_l = nets[l].hpwl();
// int hpwl_r = nets[r].hpwl();
// if (hpwl_l == hpwl_r) return l > r;
// return hpwl_l < hpwl_r;
// });
//for(int i = 0; i < s.size(); i++)
// printf("%d, [%d, %d] [%d, %d]\n", s[i], nets[s[i]].lowerx, nets[s[i]].upperx, nets[s[i]].lowery, nets[s[i]].uppery);
/*int n = netsToRoute.size();
nt *minx, *maxx, *miny, *maxy, *res, *siz, *used, *flag;
cudaMallocManaged(&minx, n * sizeof(int));
cudaMallocManaged(&maxx, n * sizeof(int));
cudaMallocManaged(&miny, n * sizeof(int));
cudaMallocManaged(&maxy, n * sizeof(int));
cudaMallocManaged(&res, n * sizeof(int));
cudaMallocManaged(&siz, (n + 1) * sizeof(int));
cudaMalloc(&used, n * sizeof(int));
cudaMalloc(&flag, n * sizeof(int));
siz[0] = 0;
for(int i = 0; i < n; i++) {
auto &net = nets[s[i]];
minx[i] = net.lowerx;
maxx[i] = net.upperx;
miny[i] = net.lowery;
maxy[i] = net.uppery;
}
generateBatch<<<1, 1024, 3 * sizeof(int)>>> (n, minx, maxx, miny, maxy, res, siz, used);
cudaDeviceSynchronize();
for(int i = 0; i < n; i++)
netsToRoute[i] = s[res[i]];
for(int i = 1; i <= siz[0]; i++)
batchSizes.emplace_back(siz[i]);*/
/*int startpos = 0;
for(auto e : batchSizes) {
for(int i = startpos; i < startpos + e; i++)
for(int j= startpos; j < i; j++) {
int n1 = netsToRoute[i], n2 = netsToRoute[j];
if(maxx[n1] + 2 < minx[n2] ||
maxx[n2] + 2 < minx[n1] ||
maxy[n1] + 2 < miny[n2] ||
maxy[n2] + 2 < miny[n2]) {}
else
printf("ERROR: overlap %d %d!\n", i, j);
}
startpos += e;
}*/
//static std::vector<std::vector<int>> vis(2000, std::vector<int> (2000, 0));
/*RangedBitset<1600> test;
test.set(10, 128, 1);
printf("%d\n", test.check_all(9, 128, 1));
printf("%d\n", test.check_all(13, 128, 1));
printf("%d\n", test.check_all(9, 128, 0));
printf("%d\n", test.check_all(13, 128, 0));
exit(0);*/
// const int LEN = 10;
if (vis.size() == 0) {
vis.resize(2000, std::vector<short>(2000, 0));
visLL.resize(2000, std::vector<short>(2000, 0));
visRR.resize(2000, std::vector<short>(2000, 0));
}
auto noConflict = [&] (int netId) {
const auto &net = nets[netId];
//int blockL = net.lowery / LEN, blockR = net.uppery / LEN;
//if(blockL == blockR) {
// for(int i = net.lowerx; i <= net.upperx; i++)
// for(int j = net.lowery; j <= net.uppery; j++)
// if(vis[i][j]) return false;
for(int i = max(0, net.lowerx - margin); i <= net.upperx + margin; i++)
for(int j = max(0, net.lowery - margin); j <= net.uppery + margin; j++)
if(vis[i][j]) return false;
/*} else {
for(int i = net.lowerx; i <= net.upperx; i++) {
if(visLL[i][net.lowery] || visRR[i][net.uppery]) return false;
for(int j = blockL + 1; j < blockR; j++)
if(visLL[i][j * LEN]) return false;
}
}*/
return true;
};
auto insert = [&] (int netId) {
//double t = clock();
const auto &net = nets[netId];
int xl = max(0, net.lowerx - margin), xr = net.upperx + margin;
int yl = max(0, net.lowery - margin), yr = net.uppery + margin;
//int blockL = yl / LEN - (yl % LEN == 0), blockR = yr / LEN + (yr % LEN == 0);
for(int i = xl; i <= xr; i++) {
for(int j = yl; j <= yr; j++) {
if (check_vis_correctness) {
vis[i][j] += 1;
} else {
vis[i][j] = 1;
}
}
/*for(int j = yl; j <= (blockR - 1) * LEN; j++)
visLL[i][j] = 1;
for(int j = (blockL + 1) * LEN; j <= yr; j++)
visRR[i][j] = 1;*/
}
//modify_cnt += clock() - t;
};
auto remove = [&] (int netId) {
const auto &net = nets[netId];
int xl = max(0, net.lowerx - margin), xr = net.upperx + margin;
int yl = max(0, net.lowery - margin), yr = net.uppery + margin;
//int blockL = yl / LEN - (yl % LEN == 0), blockR = yr / LEN + (yr % LEN == 0);
for(int i = xl; i <= xr; i++) {
for(int j = yl; j <= yr; j++) {
if (check_vis_correctness) {
vis[i][j] -= 1;
} else {
vis[i][j] = 0;
}
}
/*for(int j = yl; j <= (blockR - 1) * LEN; j++)
visLL[i][j] = 0;
for(int j = (blockL + 1) * LEN; j <= yr; j++)
visRR[i][j] = 0;*/
}
};
auto checkVis = [&] () {
for (int i = 0; i < 2000; i++) {
for (int j = 0; j < 2000; j++) {
if (vis[i][j] > 1) {
std::cout << "ERROR in batch generation" << std::endl;
return;
}
}
}
};
netsToRoute.clear();
int lastUnroute = 0;
// FIXME: if batch generation allows bbox overlap, data race would happen in PR kernel
while(netsToRoute.size() < s.size()) {
int sz = netsToRoute.size();
// int last = lastUnroute;
int cnt = 0;
for(size_t i = lastUnroute; i < s.size(); i++) if(s[i] != -1) {
bool no_conflict = 1;
no_conflict = noConflict(s[i]);
// older version, allow bbox overlap but much faster
// if(nets[s[i]].area() * 2 < netsToRoute.size() - sz)
// no_conflict = noConflict(s[i]);
// else for(size_t j = sz; j < netsToRoute.size(); j++) {
// const auto &a = nets[s[i]];
// const auto &b = nets[netsToRoute[j]];
// if(!(a.upperx + margin < b.lowerx || b.upperx + margin < a.lowerx ||
// a.uppery + margin < b.lowery || b.uppery + margin < a.lowery)) {
// no_conflict = 0;
// break;
// }
// }
if(no_conflict)
netsToRoute.emplace_back(s[i]), insert(s[i]), s[i] = -1, cnt = 0;
else
cnt++;
//if(cnt >= 100) break;
if(iter && netsToRoute.size() - sz == MAX_BATCH_SIZE) break;
if(!iter && netsToRoute.size() - sz == 250) break;
}
while(lastUnroute < s.size() && s[lastUnroute] == -1) lastUnroute++;
if (check_vis_correctness) checkVis();
for(int i = sz; i < netsToRoute.size(); i++)
remove(netsToRoute[i]);
batchSizes.emplace_back(netsToRoute.size() - sz);
}
batchgentime += clock() - t;
logger.info("INFO: Batch Generation Time %.4f", batchgentime / CLOCKS_PER_SEC);
}
reverse(batchSizes.begin(), batchSizes.end());
reverse(netsToRoute.begin(), netsToRoute.end());
logger.info("number of batches: %d; number of nets: %d", batchSizes.size(), netsToRoute.size());
int startpos = 0;
int sumofpins = 0;
// int lowestpins = 0;
// int batch_cnt = 0;
double mrtime = 0, costtime = 0;
for(auto batchSize : batchSizes) {
calculateWireCost<<<BLOCK_NUMBER(LAYER * N * N), BLOCK_SIZE>>> (cost, wireDist, fixed, fixedLength, wires, vias, capacity, unitShortCostDiscounted, logisticSlope, N, LAYER);
calculateViaCost<<<BLOCK_NUMBER((LAYER - 1) * N * N), BLOCK_SIZE>>> (wires, fixed, capacity, viaCost, unitViaMultiplier, unitViaCost, logisticSlope, N, LAYER);
calculateCostSum<<<N, N / 2, N * sizeof(int64_t)>>> (LAYER, N, cost, costSum);
if(iter == 0) {
int offset = batchSize;
int gbPinOffset = batchSize;
if(points == nullptr) {
cudaMallocManaged(&points, 20000000 * sizeof(int));
cudaMallocManaged(&gbpoints, 10000000 * sizeof(int));
}
// double prepare_part_time = 0;
double prepare_detailed_time = 0;
double t = clock();
//if(startpos == 11394)
//logger.info("net id %d", netsToRoute[startpos]);
//netsToRoute[startpos] = 8026;
for(int i = 0; i < batchSize; i++) {
int netId = netsToRoute[startpos + i];
points[i] = offset;
gbpoints[i] = gbPinOffset;
offset += prepare(prepare_detailed_time, nets[netId], points + offset, routesOffsetCPU[netId], gbpoints + gbPinOffset, gbPinOffset, X, Y, N, LAYER, DIRECTION);
}
if(offset > 20000000 || gbPinOffset > 10000000)
logger.error("ERROR offset %d %d", offset, gbPinOffset);
timer2 += prepare_detailed_time;
prpreparetime += clock() - t;
t = clock();
//cudaMemcpy(cuda_points, points, offset * sizeof(int), cudaMemcpyHostToDevice);
patternRoute(points, batchSize, costSum, viaCost, dist, prev, wires, vias, routes, gbpoints, gbpinRoutes, X, Y, N, LAYER, DIRECTION);
prtime += clock() - t;
} else {
calculateCellResource<<<BLOCK_NUMBER(LAYER * N * N), BLOCK_SIZE>>> (cell_resource, wires, fixed, vias, capacity, N, LAYER, LAYER * N * N);
calculateCoarseCost<<<LAYER * cgxsize, cgysize>>> (cell_resource, gpuMR.cost, wires, fixed, vias, capacity, N, cgxsize, cgysize, X, Y, LAYER, DIRECTION, COARSENING_SCALE);
calculateCoarseVia<<<(LAYER - 1) * cgxsize, cgysize>>> (cell_resource, gpuMR.via, wires, fixed, vias, capacity, N, LAYER, cgxsize, cgysize, X, Y, DIRECTION, COARSENING_SCALE);
gpuMR.run(DIRECTION, iter);
int pin10 = 0;
for(int i = 0; i < batchSize; i++) {
int netId = netsToRoute[startpos + i], cur = 0, *pins = allpins + i * MAX_PIN_SIZE_PER_NET;
auto& vec = nets[netId].getPins();
pins[cur++] = routesOffsetCPU[netId];
pins[cur++] = vec.size();
sumofpins += vec.size();
if(vec.size() <= 10) pin10++;
for(auto a : vec) {
pins[cur++] = a.size();
for(auto b : a) {
pins[cur++] = b;
//if(b / N / N >= 5)
//std::cerr << "upper lower pins exist. " << b / N / N << std::endl;
}
}
if(cur >= MAX_PIN_SIZE_PER_NET) {
std::cerr << "ERROR: NOT ENOUGH FOR PINS cur: " << cur << " MAX_PIN_SIZE_PER_NET: " << MAX_PIN_SIZE_PER_NET << std::endl;
exit(-1);
}
}
//printf("%d / %d = %.2lf\n", pin10, batchSize, pin10 * 1.0 / batchSize);
initMap<<<BLOCK_NUMBER(batchSize * LAYER * N * N), BLOCK_SIZE>>> (dist, prev, N * N * LAYER, batchSize * N * N * LAYER);
setStartCells<<<batchSize, 1>>> (dist, allpins, MAX_PIN_SIZE_PER_NET, N * N * LAYER);
double t = clock();
gpuMR.getResults(costtime, costSum, allpins, dist, prev, cost, viaCost, wires, vias, routes, N, COARSENING_SCALE, DIRECTION, batchSize, netsToRoute[startpos]);
cudaDeviceSynchronize();
mrtime += (clock() - t) * 1.0 / CLOCKS_PER_SEC;
}
//std::cerr << "pins: " << sumofpins << ' ' << 1.0 * sumofpins / netsToRoute.size() << std::endl;
startpos += batchSize;
}
if(!iter) {
logger.info("INFO: PR Prepare Time %.4f", prpreparetime / CLOCKS_PER_SEC);
logger.info("INFO: PR Kernel Time %.4f", prtime / CLOCKS_PER_SEC);
} else {
logger.info("INFO: MR Func time %.4f", mrtime);
logger.info("INFO: MR Cost calc time %.4f", costtime / CLOCKS_PER_SEC);
}
//output<<<1, 1>>> (wires, vias, LAYER, X, Y, N, DIRECTION);
//gpuMR.query();
//if(iter == db::setting.rrrIterLimit - 1) {
if(1) {
markOverflowWires<<<BLOCK_NUMBER(LAYER * N * N), BLOCK_SIZE>>> (capacity, wires, vias, fixed, isOverflowWire, N, LAYER);
markOverflowVias<<<BLOCK_NUMBER((LAYER - 1) * N * N), BLOCK_SIZE>>> (capacity, wires, vias, fixed, isOverflowVia, N, LAYER);
markOverflowNets<<<BLOCK_NUMBER(NET_NUM), BLOCK_SIZE>>> (isOverflowVia, isOverflowWire, isOverflowNet, routes, routesOffset, NET_NUM);
cudaDeviceSynchronize();
int cnt = 0;
for(size_t netId = 0; netId < nets.size(); netId++)
if(isOverflowNet[netId]) {
cnt++;
if(nets[netId].noroute) printf("ERROR: net %zu is noroute but overflow\n", netId);
}
logger.info("Final Overflow Net Number: %d", cnt);
numOvflNets = cnt;
}
}
void GPURouter::setFromNets(vector<GrNet> &nets, int numPlPin_) {
NET_NUM = nets.size();
pinNumCPU = new int[NET_NUM];
routesOffsetCPU = new int[NET_NUM + 1];
routesOffsetCPU[0] = 0;
for(int i = 0; i < NET_NUM; i++) {
pinNumCPU[i] = nets[i].getPins().size();
routesOffsetCPU[i + 1] = pinNumCPU[i] * MAX_ROUTE_LEN_PER_PIN + routesOffsetCPU[i];
}
cudaMallocManaged(&isOverflowNet, NET_NUM * sizeof(int));
cudaMalloc(&routes, routesOffsetCPU[NET_NUM] * sizeof(int));
cudaMemset(routes, 0, routesOffsetCPU[NET_NUM] * sizeof(int));
logger.info("total routes: %d", routesOffsetCPU[NET_NUM]);
cudaMalloc(&routesOffset, (NET_NUM + 1) * sizeof(int));
cudaMemcpy(routesOffset, routesOffsetCPU, sizeof(int) * (NET_NUM + 1), cudaMemcpyHostToDevice);
for (int netId = nets.size() - 1; netId >= 0; netId--) {
auto& lastGrNet = nets[netId];
if (lastGrNet.pin2gbpinId.size() > 0) {
numGbPin = lastGrNet.pin2gbpinId[lastGrNet.pin2gbpinId.size() - 1] + 1;
break;
}
}
// numRoutes, routeId, routeId, routeId, routeId, numVias
cudaMalloc(&gbpinRoutes, 6 * numGbPin * sizeof(int));
cudaMemset(gbpinRoutes, 0, 6 * numGbPin * sizeof(int));
// number of pins in placement database
numPlPin = numPlPin_;
std::vector<int> gbpin2netIdCPU(numGbPin);
std::vector<int> plPinId2gbPinIdCPU(numPlPin, -1);
for (int netId = 0; netId < nets.size(); netId ++) {
for (int pinId = 0; pinId < nets[netId].getPins().size(); pinId++) {
int gbpinId = nets[netId].pin2gbpinId[pinId];
gbpin2netIdCPU[gbpinId] = netId;
std::vector<int>& gpdbPinIds = nets[netId].pin2gpdbPinIds[pinId];
for (int gpdbPinId : gpdbPinIds) {
plPinId2gbPinIdCPU[gpdbPinId] = gbpinId;
}
}
}
cudaMalloc(&gbpin2netId, numGbPin * sizeof(int));
cudaMemcpy(gbpin2netId, gbpin2netIdCPU.data(), numGbPin * sizeof(int), cudaMemcpyHostToDevice);
cudaMalloc(&plPinId2gbPinId, numPlPin * sizeof(int));
cudaMemcpy(plPinId2gbPinId, plPinId2gbPinIdCPU.data(), numPlPin * sizeof(int), cudaMemcpyHostToDevice);
cudaDeviceSynchronize();
}
void GPURouter::setToNets(vector<GrNet> &nets) {
int *routesCPU = new int[routesOffsetCPU[NET_NUM]];
int mx = 0;
cudaMemcpy(routesCPU, routes, sizeof(int) * routesOffsetCPU[NET_NUM], cudaMemcpyDeviceToHost);
int num_net_use_too_many_route = 0;
for(size_t netId = 0; netId < nets.size(); netId++) {
vector<int> wires, vias;
int *routesSub = routesCPU + routesOffsetCPU[netId];
mx = max(mx, routesSub[0] / pinNumCPU[netId]);
if(routesSub[0] > routesOffsetCPU[netId + 1] - routesOffsetCPU[netId]) {
num_net_use_too_many_route++;
// std::cerr << "ERROR: too many routesSub! Please set MAX_ROUTE_LEN_PER_PIN larger than " << routesSub[0] / pinNumCPU[netId] << std::endl;
}
for(int i = 1; i < routesSub[0]; i += 2) {
if (routesSub[i + 1] > 0) {
wires.emplace_back(routesSub[i]);
wires.emplace_back(routesSub[i + 1]);
} else if (routesSub[i + 1] == -1) {
vias.emplace_back(routesSub[i]);
}
}
nets[netId].setWires(wires);
nets[netId].setVias(vias);
//nets[netId].useExtraVias();
};
logger.info("max routes[0] = %d", mx);
if (num_net_use_too_many_route) {
std::cerr << "ERROR: there are " << num_net_use_too_many_route << " nets use too many route segments!";
std::cerr << " Please set MAX_ROUTE_LEN_PER_PIN (" << MAX_ROUTE_LEN_PER_PIN << ") larger than " << mx << std::endl;
}
delete[] routesCPU;
}
} // namespace gr

View File

@ -0,0 +1,116 @@
#pragma once
#include "MazeRoute.h"
#include "PatternRoute.h"
#include "common/common.h"
#include "gpugr/db/GrNet.h"
namespace gr {
typedef int dtype;
class GPURouter {
public:
GPURouter(){};
GPURouter(
int device_id, int layer, int x, int y, int N_, int cgxsize_, int cgysize_, int direction, int csrn_scale) {
initialize(device_id, layer, x, y, N_, cgxsize_, cgysize_, direction, csrn_scale);
}
~GPURouter();
void initialize(
int device_id, int layer, int x, int y, int N_, int cgxsize_, int cgysize_, int direction, int csrn_scale);
void setMap(const vector<float> &cap,
const vector<float> &wir,
const vector<float> &fixedL,
const vector<float> &fix);
void setFromNets(vector<GrNet> &nets, int numPlPin_);
void setToNets(vector<GrNet> &nets);
void route(vector<GrNet> &nets, int iterleft);
void setUnitViaMultiplier(float w);
void setUnitVioCost(vector<float>& cost, float discount);
void setLogisticSlope(float value);
void setUnitViaCost(float value);
void query();
public:
std::tuple<torch::Tensor, torch::Tensor, torch::Tensor> getDemandMap();
torch::Tensor getCapacityMap();
torch::Tensor calcRouteGrad(torch::Tensor mask_map,
torch::Tensor wire_dmd_map_2d,
torch::Tensor via_dmd_map_2d,
torch::Tensor cap_map_2d,
torch::Tensor dist_weights,
torch::Tensor wirelength_weights,
torch::Tensor route_gradmat,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
float grad_weight,
float unit_wire_cost,
float unit_via_cost,
int num_nodes);
torch::Tensor calcFillerRouteGrad(torch::Tensor filler_pos,
torch::Tensor filler_size,
torch::Tensor filler_weight,
torch::Tensor expand_ratio,
torch::Tensor grad_mat,
float grad_weight,
float unit_len_x,
float unit_len_y,
int num_bin_x,
int num_bin_y,
int num_fillers);
torch::Tensor calcPseudoPinGrad(torch::Tensor node_pos, torch::Tensor pseudo_pin_pos, float gamma);
torch::Tensor calcNodeInflateRatio(torch::Tensor node_pos,
torch::Tensor node_size,
torch::Tensor node_weight,
torch::Tensor expand_ratio,
torch::Tensor inflate_mat,
float grad_weight,
float unit_len_x,
float unit_len_y,
int num_bin_x,
int num_bin_y,
bool use_weighted_inflation);
torch::Tensor calcInflatedPinRelCpos(torch::Tensor node_inflate_ratio,
torch::Tensor old_pin_rel_cpos,
torch::Tensor pin_id2node_id,
int num_movable_nodes);
int getNumOvflNets() { return numOvflNets; }
private:
GPUMazeRouter gpuMR;
// routes:
// (x, y): starting point x, length |y|; negative y implies vias
int DEVICE_ID;
int LAYER, N, X, Y, NET_NUM, DIRECTION;
int COARSENING_SCALE;
int cgxsize, cgysize;
int *pinNum = nullptr, *pinNumOffset = nullptr, *pins = nullptr;
int *routes = nullptr, *routesOffset = nullptr, *routesOffsetCPU = nullptr, *pinNumCPU = nullptr;
int *allpins;
int *points = nullptr, *gbpoints = nullptr;
int *gbpinRoutes = nullptr, *gbpin2netId = nullptr, *plPinId2gbPinId = nullptr;
float *capacity, *wireDist, *fixedLength, *fixed;
int *wires, *vias, *prev;
int *isOverflowWire, *isOverflowVia, *isOverflowNet = nullptr;
int *boundaries, *isLocked;
int *wiresCPU, *viasCPU;
int *cudaIndex, *cudaCostIndex;
int *modifiedVia, *modifiedWire, *viasToBeUpdated, *wiresToBeUpdated;
dtype *dist, *cost, *viaCost;
int64_t *costSum;
float *unitShortCostDiscounted, unitViaCost, unitViaMultiplier = 1, logisticSlope = 1, *cell_resource;
int numGbPin, numPlPin;
int numOvflNets = 0;
const int MAX_BATCH_SIZE = 100, MAX_PIN_SIZE_PER_NET = 500000;
std::vector<std::vector<short>> vis, visLL, visRR;
};
} // namespace gr

View File

@ -0,0 +1,577 @@
#include "GPURouter.h"
#include "InCellUsage.cuh"
namespace gr {
#define BLOCK_SIZE 512
#define BLOCK_NUMBER(n) (((n) + (BLOCK_SIZE) - 1) / BLOCK_SIZE)
__device__ void inline cudaSwapInt(int &a, int &b) {
int c(a);
a = b;
b = c;
}
__device__ float overlap(float x_l, float x_h, float bin_x_l) {
// bin_x_h == bin_x_l + 1
return min(x_h, bin_x_l + 1) - max(x_l, bin_x_l);
}
__global__ void calc_node_grad_deterministic_cuda_kernel(
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_grad,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_grad,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> node2pin_list,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> node2pin_list_end,
int num_nodes) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int i = index >> 1; // node index
if (i < num_nodes) {
const int c = index & 1; // channel index
int64_t start_idx = 0;
if (i != 0) {
start_idx = node2pin_list_end[i - 1];
}
int64_t end_idx = node2pin_list_end[i];
if (end_idx != start_idx) {
node_grad[i][c] += pin_grad[node2pin_list[start_idx]][c];
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
node_grad[i][c] += pin_grad[node2pin_list[idx]][c];
}
}
}
}
__global__ void getDmdTensor(
torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> dmdMap,
torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> wireDmdMap,
torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> viaDmdMap,
const float *capacity, int *wires, int *vias, float *fixed, int N, int LAYER, int xSize, int ySize, int DIRECTION) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if(idx < LAYER * N * N && idx % N + 1 < N) {
int layer = idx / N / N, x = idx / N % N, y = idx % N;
if (!(layer & 1) ^ DIRECTION) cudaSwapInt(x, y);
if (layer < LAYER && x < xSize && y < ySize) {
float wireDmd = wires[idx] + fixed[idx];
float viaDmd = twoCellsViaUsage(idx, vias, N, LAYER);
dmdMap[layer][x][y] = wireDmd + viaDmd;
wireDmdMap[layer][x][y] = wireDmd;
viaDmdMap[layer][x][y] = viaDmd;
}
}
}
__global__ void getCapTensor(
torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> capMap,
const float *capacity, int *wires, int *vias, float *fixed, int N, int LAYER, int xSize, int ySize, int DIRECTION) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if(idx < LAYER * N * N && idx % N + 1 < N) {
int layer = idx / N / N, x = idx / N % N, y = idx % N;
if (!(layer & 1) ^ DIRECTION) cudaSwapInt(x, y);
if (layer < LAYER && x < xSize && y < ySize) {
capMap[layer][x][y] = capacity[idx];
}
}
}
__global__ void compGcellRouteForce(
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> gbpin_grad,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> mask_map,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> wire_dmd_map_2d,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> via_dmd_map_2d,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> cap_map_2d,
torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> dist_weights,
torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> wirelength_weights,
torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> route_gradmat,
float grad_weight, float unit_wire_cost, float unit_via_cost,
int *gbpinRoutes, int *gbpin2netId, int *routes, int *routesOffset,
int numGbPin, int N, int LAYER, int xSize, int ySize, int DIRECTION
) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
if (idx < numGbPin) {
int netId = gbpin2netId[idx];
routes += routesOffset[netId];
if (routes[0] == -1) {
// FIXME: failed PR, use floating cost instead of int cost
return;
}
gbpinRoutes += idx * 6;
int numGbpinRoutes = gbpinRoutes[0];
// int numGbpinVias = gbpinRoutes[5]; // TODO: consider #Vias in route grad
for (int i = 1; i < 1 + numGbpinRoutes; i++) {
int routeId = gbpinRoutes[i];
bool reverseRoute = false;
if (routeId < 0) {
// for a segment (lx, hx), if reverseRoute is true, the gbpin is located at hx, else at lx
routeId = -routeId;
reverseRoute = true;
}
int p = routes[routeId];
int l = p / N / N, x = p % (N * N) / N, y = p % N;
if (!(l & 1) ^ DIRECTION) cudaSwapInt(x, y);
int lx = x, hx = x, ly = y, hy = y;
if ((l & 1) ^ DIRECTION) {
hy += routes[routeId + 1];
} else {
hx += routes[routeId + 1];
}
if (lx != hx) {
// x direction route segement
float grad = 0, total_weight = 0;
for (int j = lx; j <= hx; j++) {
float cur_dist_weight;
if (reverseRoute) {
cur_dist_weight = dist_weights[hx - j];
} else {
cur_dist_weight = dist_weights[j - lx];
}
// TODO: 1) should we also consider X direction?
// 2) consider via cost?
float cost = unit_wire_cost / min(cap_map_2d[j][ly], 0.2);
grad += cost * route_gradmat[1][j][ly] * mask_map[j][ly] * cur_dist_weight;
total_weight += cur_dist_weight;
}
gbpin_grad[idx][1] = grad_weight * grad / total_weight * wirelength_weights[hx - lx + 1];
} else if (ly != hy) {
// y direction route segement
float grad = 0, total_weight = 0;
for (int j = ly; j<= hy; j++) {
float cur_dist_weight;
if (reverseRoute) {
cur_dist_weight = dist_weights[hy - j];
} else {
cur_dist_weight = dist_weights[j - ly];
}
float cost = unit_wire_cost / min(cap_map_2d[lx][j], 0.2);
grad += cost * route_gradmat[0][lx][j] * mask_map[lx][j] * cur_dist_weight;
total_weight += cur_dist_weight;
}
gbpin_grad[idx][0] = grad_weight * grad / total_weight * wirelength_weights[hy - ly + 1];
}
}
}
}
__global__ void assignRouteForceToPlPin(
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> plpin_grad,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> gbpin_grad,
int *plPinId2gbPinId, int numPlPin
) {
int plPinId = blockIdx.x * blockDim.x + threadIdx.x;
if (plPinId < numPlPin) {
int gbPinId = plPinId2gbPinId[plPinId];
plpin_grad[plPinId][0] = gbpin_grad[gbPinId][0];
plpin_grad[plPinId][1] = gbpin_grad[gbPinId][1];
}
}
__global__ void fillerRouteForce(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> filler_pos,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> filler_size,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> filler_weight,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> expand_ratio,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> filler_grad,
const float *grad_mat,
float grad_weight,
float unit_len_x,
float unit_len_y,
int num_bin_x,
int num_bin_y,
int num_fillers
) {
const int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_fillers) {
float weight = filler_weight[i];
if (weight > 0) {
const float x_l = (filler_pos[i][0] - filler_size[i][0] / 2) / unit_len_x;
const float x_h = (filler_pos[i][0] + filler_size[i][0] / 2) / unit_len_x;
const float y_l = (filler_pos[i][1] - filler_size[i][1] / 2) / unit_len_y;
const float y_h = (filler_pos[i][1] + filler_size[i][1] / 2) / unit_len_y;
if (x_h - x_l < 0 || y_h - y_l < 0) return;
weight *= expand_ratio[i];
int x_lf = lround(floor(x_l));
int x_hf = lround(floor(x_h));
int y_lf = lround(floor(y_l));
int y_hf = lround(floor(y_h));
x_lf = max(x_lf, 0);
x_hf = min(x_hf, num_bin_x - 1);
y_lf = max(y_lf, 0);
y_hf = min(y_hf, num_bin_y - 1);
float gradX = 0;
float gradY = 0;
for (int j = x_lf; j < x_hf + 1; j++) {
float bin_x_l = static_cast<float>(j);
float overlap_x = overlap(x_l, x_h, bin_x_l);
for (int k = y_lf; k < y_hf + 1; k++) {
float bin_y_l = static_cast<float>(k);
float overlap_y = overlap(y_l, y_h, bin_y_l);
float overlap_area = overlap_x * overlap_y;
gradX += grad_mat[0 * num_bin_x * num_bin_y + j * num_bin_y + k] * overlap_area;
gradY += grad_mat[1 * num_bin_x * num_bin_y + j * num_bin_y + k] * overlap_area;
}
}
filler_grad[i][0] = grad_weight * weight * gradX;
filler_grad[i][1] = grad_weight * weight * gradY;
}
}
}
__global__ void inflateNodeRatioWeighted(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_size,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> node_weight,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> expand_ratio,
const torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> inflate_mat,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_inflate_ratio,
float grad_weight,
float unit_len_x,
float unit_len_y,
int num_bin_x,
int num_bin_y,
int num_nodes
) {
const int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_nodes) {
float weight = node_weight[i] * expand_ratio[i];
if (weight > 0) {
const float x_l = (node_pos[i][0] - node_size[i][0] / 2) / unit_len_x;
const float x_h = (node_pos[i][0] + node_size[i][0] / 2) / unit_len_x;
const float y_l = (node_pos[i][1] - node_size[i][1] / 2) / unit_len_y;
const float y_h = (node_pos[i][1] + node_size[i][1] / 2) / unit_len_y;
if (x_h - x_l < 0 || y_h - y_l < 0) return;
int x_lf = lround(floor(x_l));
int x_hf = lround(floor(x_h));
int y_lf = lround(floor(y_l));
int y_hf = lround(floor(y_h));
x_lf = max(x_lf, 0);
x_hf = min(x_hf, num_bin_x - 1);
y_lf = max(y_lf, 0);
y_hf = min(y_hf, num_bin_y - 1);
const float node_area = (x_h - x_l) * (y_h - y_l);
float inflate_x = 0, inflate_y = 0;
for (int j = x_lf; j < x_hf + 1; j++) {
float bin_x_l = static_cast<float>(j);
float overlap_x = overlap(x_l, x_h, bin_x_l);
for (int k = y_lf; k < y_hf + 1; k++) {
float bin_y_l = static_cast<float>(k);
float overlap_y = overlap(y_l, y_h, bin_y_l);
float overlap_area_ratio = overlap_x * overlap_y / node_area;
inflate_x += overlap_area_ratio * inflate_mat[0][j][k];
inflate_y += overlap_area_ratio * inflate_mat[1][j][k];
}
}
node_inflate_ratio[i][0] = grad_weight * weight * inflate_x;
node_inflate_ratio[i][1] = grad_weight * weight * inflate_y;
}
}
}
__global__ void inflateNodeRatioMax(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_size,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> node_weight,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> expand_ratio,
const torch::PackedTensorAccessor32<float, 3, torch::RestrictPtrTraits> inflate_mat,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_inflate_ratio,
float grad_weight,
float unit_len_x,
float unit_len_y,
int num_bin_x,
int num_bin_y,
int num_nodes
) {
const int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_nodes) {
float weight = node_weight[i] * expand_ratio[i];
if (weight > 0) {
const float x_l = (node_pos[i][0] - node_size[i][0] / 2) / unit_len_x;
const float x_h = (node_pos[i][0] + node_size[i][0] / 2) / unit_len_x;
const float y_l = (node_pos[i][1] - node_size[i][1] / 2) / unit_len_y;
const float y_h = (node_pos[i][1] + node_size[i][1] / 2) / unit_len_y;
if (x_h - x_l < 0 || y_h - y_l < 0) return;
int x_lf = lround(floor(x_l));
int x_hf = lround(floor(x_h));
int y_lf = lround(floor(y_l));
int y_hf = lround(floor(y_h));
x_lf = max(x_lf, 0);
x_hf = min(x_hf, num_bin_x - 1);
y_lf = max(y_lf, 0);
y_hf = min(y_hf, num_bin_y - 1);
float inflate_x = 0, inflate_y = 0;
for (int j = x_lf; j < x_hf + 1; j++) {
for (int k = y_lf; k < y_hf + 1; k++) {
inflate_x = max(inflate_x, inflate_mat[0][j][k]);
inflate_y = max(inflate_y, inflate_mat[1][j][k]);
}
}
node_inflate_ratio[i][0] = grad_weight * weight * inflate_x;
node_inflate_ratio[i][1] = grad_weight * weight * inflate_y;
}
}
}
__global__ void inflatePinRelCpos(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_inflate_ratio,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> old_pin_rel_cpos,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> pin_id2node_id,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> new_pin_rel_cpos,
int num_movable_nodes,
int num_pins
) {
const int i = blockIdx.x * blockDim.x + threadIdx.x;
if (i < num_pins) {
int64_t node_id = pin_id2node_id[i];
if (node_id < num_movable_nodes) {
new_pin_rel_cpos[i][0] = old_pin_rel_cpos[i][0] * node_inflate_ratio[node_id][0];
new_pin_rel_cpos[i][1] = old_pin_rel_cpos[i][1] * node_inflate_ratio[node_id][1];
}
}
}
__global__ void pseudoPinForce(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pseudo_pin_pos,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_grad,
int num_nodes,
float inv_gamma) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int pin_id = index >> 1; // net index
if (pin_id < num_nodes) {
const int c = index & 1; // channel index
float x1 = node_pos[pin_id][c];
float x2 = pseudo_pin_pos[pin_id][c];
float x_max = max(x1, x2);
float x_min = min(x1, x2);
float sum_x_exp_x = 0;
float sum_x_exp_nx = 0;
float sum_exp_x = 0;
float sum_exp_nx = 0;
for (int i = 0; i < 2; i++) {
float cur_x = i == 0 ? x1 : x2;
float recenter_exp_x = exp((cur_x - x_max) * inv_gamma);
float recenter_exp_nx = exp((x_min - cur_x) * inv_gamma);
sum_x_exp_x += cur_x * recenter_exp_x;
sum_x_exp_nx += cur_x * recenter_exp_nx;
sum_exp_x += recenter_exp_x;
sum_exp_nx += recenter_exp_nx;
}
float inv_sum_exp_x = 1 / sum_exp_x;
float inv_sum_exp_nx = 1 / sum_exp_nx;
float s_x = sum_x_exp_x * inv_sum_exp_x;
float ns_nx = sum_x_exp_nx * inv_sum_exp_nx;
float x_coeff = inv_gamma * inv_sum_exp_x;
float nx_coeff = -inv_gamma * inv_sum_exp_nx;
float grad_const = (1 - inv_gamma * s_x) * inv_sum_exp_x;
float grad_nconst = (1 + inv_gamma * ns_nx) * inv_sum_exp_nx;
// calc x1 (original pin)'s gradient
float recenter_exp_x = exp((x1 - x_max) * inv_gamma);
float recenter_exp_nx = exp((x_min - x1) * inv_gamma);
node_grad[pin_id][c] = (grad_const + x_coeff * x1) * recenter_exp_x -
(grad_nconst + nx_coeff * x1) * recenter_exp_nx;
}
}
std::tuple<torch::Tensor, torch::Tensor, torch::Tensor> GPURouter::getDemandMap() {
int xSize = X;
int ySize = Y;
torch::Tensor dmdMap = torch::zeros({LAYER, xSize, ySize},
torch::dtype(torch::kFloat32).device(torch::Device(torch::kCUDA, DEVICE_ID)));
torch::Tensor wireDmdMap = torch::zeros({LAYER, xSize, ySize},
torch::dtype(torch::kFloat32).device(torch::Device(torch::kCUDA, DEVICE_ID)));
torch::Tensor viaDmdMap = torch::zeros({LAYER, xSize, ySize},
torch::dtype(torch::kFloat32).device(torch::Device(torch::kCUDA, DEVICE_ID)));
getDmdTensor<<<BLOCK_NUMBER(LAYER * N * N), BLOCK_SIZE>>>(
dmdMap.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
wireDmdMap.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
viaDmdMap.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
capacity, wires, vias, fixed, N, LAYER, xSize, ySize, DIRECTION);
return {dmdMap, wireDmdMap, viaDmdMap};
}
torch::Tensor GPURouter::getCapacityMap() {
int xSize = X;
int ySize = Y;
torch::Tensor capMap = torch::zeros({LAYER, xSize, ySize},
torch::dtype(torch::kFloat32).device(torch::Device(torch::kCUDA, DEVICE_ID)));
getCapTensor<<<BLOCK_NUMBER(LAYER * N * N), BLOCK_SIZE>>>(
capMap.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
capacity, wires, vias, fixed, N, LAYER, xSize, ySize, DIRECTION);
return capMap;
}
torch::Tensor GPURouter::calcRouteGrad(torch::Tensor mask_map,
torch::Tensor wire_dmd_map_2d,
torch::Tensor via_dmd_map_2d,
torch::Tensor cap_map_2d,
torch::Tensor dist_weights,
torch::Tensor wirelength_weights,
torch::Tensor route_gradmat,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
float grad_weight,
float unit_wire_cost,
float unit_via_cost,
int num_nodes) {
// 1. compute demand map and capacity map
// 2. compute route_gradmat for each gcell (use torchDCT)
// 3. compute route force for each global pin
torch::Tensor gbpin_grad = torch::zeros({numGbPin, 2},
torch::dtype(torch::kFloat32).device(torch::Device(torch::kCUDA, DEVICE_ID)));
compGcellRouteForce<<<BLOCK_NUMBER(numGbPin), BLOCK_SIZE>>>(
gbpin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
mask_map.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
wire_dmd_map_2d.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
via_dmd_map_2d.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
cap_map_2d.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
dist_weights.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
wirelength_weights.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
route_gradmat.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
grad_weight, unit_wire_cost, unit_via_cost,
gbpinRoutes, gbpin2netId, routes, routesOffset,
numGbPin, N, LAYER, X, Y, DIRECTION
);
// 4. assign route force to placement pins (node's pins)
torch::Tensor plpin_grad = torch::zeros({numPlPin, 2},
torch::dtype(torch::kFloat32).device(torch::Device(torch::kCUDA, DEVICE_ID)));
assignRouteForceToPlPin<<<BLOCK_NUMBER(numPlPin), BLOCK_SIZE>>>(
plpin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
gbpin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
plPinId2gbPinId, numPlPin
);
// 5. calc node grad (use deterministic summation)
auto node_grad = torch::zeros({num_nodes, 2}, torch::dtype(plpin_grad.dtype()).device(plpin_grad.device()));
const int threads = 128;
const int blocks = (num_nodes * 2 + threads - 1) / threads;
calc_node_grad_deterministic_cuda_kernel<<<blocks, threads, 0>>>(
node_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
plpin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node2pin_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
node2pin_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
num_nodes);
return node_grad;
}
torch::Tensor GPURouter::calcFillerRouteGrad(torch::Tensor filler_pos,
torch::Tensor filler_size,
torch::Tensor filler_weight,
torch::Tensor expand_ratio,
torch::Tensor grad_mat,
float grad_weight,
float unit_len_x,
float unit_len_y,
int num_bin_x,
int num_bin_y,
int num_fillers) {
int threads = 64;
int blocks = (num_fillers + threads - 1) / threads;
auto filler_grad = torch::zeros({num_fillers, 2}, torch::dtype(filler_pos.dtype()).device(filler_pos.device()));
fillerRouteForce<<<blocks, threads, 0>>>(
filler_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
filler_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
filler_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
expand_ratio.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
filler_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
grad_mat.data_ptr<float>(),
grad_weight, unit_len_x, unit_len_y, num_bin_x, num_bin_y, num_fillers
);
return filler_grad;
}
torch::Tensor GPURouter::calcPseudoPinGrad(torch::Tensor node_pos, torch::Tensor pseudo_pin_pos, float gamma) {
const auto num_nodes = node_pos.size(0);
const int threads = 128;
const int blocks = (num_nodes * 2 + threads - 1) / threads;
float inv_gamma = 1 / gamma;
auto node_grad = torch::zeros({num_nodes, 2}, torch::dtype(node_pos.dtype()).device(node_pos.device()));
pseudoPinForce<<<blocks, threads, 0>>>(
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
pseudo_pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
num_nodes, inv_gamma
);
return node_grad;
}
torch::Tensor GPURouter::calcNodeInflateRatio(torch::Tensor node_pos,
torch::Tensor node_size,
torch::Tensor node_weight,
torch::Tensor expand_ratio,
torch::Tensor inflate_mat,
float grad_weight,
float unit_len_x,
float unit_len_y,
int num_bin_x,
int num_bin_y,
bool use_weighted_inflation) {
const auto num_nodes = node_pos.size(0);
const int threads = 128;
const int blocks = (num_nodes + threads - 1) / threads;
auto node_inflate_ratio = torch::ones({num_nodes, 2}, torch::dtype(node_pos.dtype()).device(node_pos.device()));
if (use_weighted_inflation) {
inflateNodeRatioWeighted<<<blocks, threads, 0>>>(
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
expand_ratio.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
inflate_mat.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
node_inflate_ratio.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
grad_weight, unit_len_x, unit_len_y, num_bin_x, num_bin_y, num_nodes
);
} else {
inflateNodeRatioMax<<<blocks, threads, 0>>>(
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node_size.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
expand_ratio.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
inflate_mat.packed_accessor32<float, 3, torch::RestrictPtrTraits>(),
node_inflate_ratio.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
grad_weight, unit_len_x, unit_len_y, num_bin_x, num_bin_y, num_nodes
);
}
return node_inflate_ratio;
}
torch::Tensor GPURouter::calcInflatedPinRelCpos(torch::Tensor node_inflate_ratio,
torch::Tensor old_pin_rel_cpos,
torch::Tensor pin_id2node_id,
int num_movable_nodes) {
const auto num_pins = old_pin_rel_cpos.size(0);
const int threads = 128;
const int blocks = (num_pins + threads - 1) / threads;
auto new_pin_rel_cpos = old_pin_rel_cpos.clone();
inflatePinRelCpos<<<blocks, threads, 0>>>(
node_inflate_ratio.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
old_pin_rel_cpos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
pin_id2node_id.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
new_pin_rel_cpos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
num_movable_nodes, num_pins
);
return new_pin_rel_cpos;
}
} // namespace gr

View File

@ -0,0 +1,41 @@
#pragma once
#include "common/common.h"
namespace gr {
__device__ __forceinline__ float myExp(float x) { return (1 << min(30, static_cast<int>(x))); }
__device__ __forceinline__ int inCellViaUsage(int idx, int *vias, int N, int LAYER) {
int layer = idx / N / N - 1, y = idx / N % N, x = idx % N;
int ans = 0;
if (layer + 2 < LAYER) ans += vias[idx]; // a via from botLayer to currlayer
if (layer >= 0) ans += vias[layer * N * N + x * N + y]; // a via from currlayer to topLayer
return ans;
}
__device__ __forceinline__ float twoCellsViaUsage(int idx, int *vias, int N, int LAYER) {
return sqrt(0.5 * (inCellViaUsage(idx, vias, N, LAYER) + inCellViaUsage(idx + 1, vias, N, LAYER))) * 1.5;
}
__device__ __forceinline__ float inCellUsedArea(int idx, int *wires, float *fixed, int N) {
float ans = 0;
if (idx % N > 0) ans += fixed[idx - 1] + wires[idx - 1];
if (idx % N + 1 < N) ans += fixed[idx] + wires[idx];
return ans / 2;
}
__device__ __forceinline__ float inCellViaCost(
int idx, int *wires, float *fixed, const float *capacity, float logisticSlope, int N) {
return 1.0 / (1.0 + myExp(logisticSlope * (capacity[idx] - inCellUsedArea(idx, wires, fixed, N))));
}
__device__ __forceinline__ float cellResource(
int idx, int *wires, float *fixed, int *vias, const float *capacity, int N, int LAYER) {
float ans = wires[idx] + fixed[idx];
if (idx % N) ans += wires[idx - 1] + fixed[idx - 1];
ans /= 2;
ans += sqrt(1.0 * inCellViaUsage(idx, vias, N, LAYER)) * 1.5;
return capacity[idx] - ans;
}
} // namespace gr

File diff suppressed because it is too large Load Diff

View File

@ -0,0 +1,50 @@
#pragma once
#include <algorithm>
#include <cassert>
#include <cstdio>
#include <iostream>
#include <vector>
namespace gr {
extern const int MAX_LAYER_NUM, MAX_PIN_NUM;
class GPUMazeRouter {
public:
void startGPU(int device_id, int layer, int x, int y);
void endGPU();
void query();
void run(int DIRECTION, int iterleft);
void getResults(double &t,
int64_t *costSum,
int *pins,
int *dist,
int *fgprev,
int *wireCost,
int *viaCost,
int *wires,
int *vias,
int *routes,
int N,
int SCALE,
int DIRECTION,
int routesOffset,
int netId);
const static int MAX_TOT_PIN_NUM = 1000000, MAX_NUM_NET = 100;
int MAX_TURN_NUM = 10;
int LAYER, X, Y, NX, NY; // 0: x-1,x,x+1... 1: y-1,y,y+1...
int *costMap, *viaMap, *markMap, *cudaRoutedPin;
int *cost, *via, *costL, *costR, *cudaMap, *reset;
int *cudaPrev;
const int MAX_PIN_SIZE_PER_NET = 500000, MAX_PIN_NUM = 10000;
int firstTime = 0; // varible to determine whether the first run of TF
};
extern int counter1, counter2, counter3, counter4;
extern double time1;
} // namespace gr

View File

@ -0,0 +1,552 @@
#include "PatternRoute.h"
#include <iostream>
#include <set>
#include "common/db/Database.h"
#include "common/utils/robin_hood.h"
#include "flute.h"
namespace gr {
using namespace Flute;
void prepareSingeNet(gr::GrNet &grNet, int routesOffset, int X, int Y, int N, int LAYER, int DIRECTION) {
const std::vector<std::vector<int>> &pins = grNet.getPins();
std::vector<int> &points = grNet.points;
points.clear();
robin_hood::unordered_map<int, std::vector<int>> loc2Pins;
// double startTimer = clock();
std::vector<int> xpos(pins.size()), ypos(pins.size());
for (int i = 0; i < pins.size(); i++) {
int layer = pins[i][0] / N / N, _x = pins[i][0] / N % N, _y = pins[i][0] % N;
if (!(layer & 1) ^ DIRECTION) std::swap(_x, _y);
xpos[i] = _x;
ypos[i] = _y;
loc2Pins[_x * N + _y].emplace_back(layer);
}
std::sort(xpos.begin(), xpos.end());
std::sort(ypos.begin(), ypos.end());
xpos.erase(std::unique(xpos.begin(), xpos.end()), xpos.end());
ypos.erase(std::unique(ypos.begin(), ypos.end()), ypos.end());
int degree = loc2Pins.size(), cur = 0;
if (degree == 0) std::cerr << "ERROR: degree 0" << std::endl;
// const int MAX_DEGREE = 100000;
// if (degree > MAX_DEGREE) std::cerr << "Not Enough X and Y in Pattern Routing" << std::endl;
int x[degree * 4], y[degree * 4];
for (auto e : loc2Pins) x[cur] = e.first / N, y[cur] = e.first % N, cur++;
Tree flutetree = flute(degree, x, y, 3);
robin_hood::unordered_map<int, int> loc2node, node2loc;
std::set<int> locations;
int node_cnt = 0;
for (int i = 0; i < degree * 2 - 2; i++) locations.insert(flutetree.branch[i].x * N + flutetree.branch[i].y);
for (auto e : locations) node2loc[loc2node[e] = node_cnt++] = e;
std::vector<robin_hood::unordered_set<int>> graph(node_cnt);
std::vector<std::vector<int>> cntx(xpos.size(), std::vector<int>(ypos.size(), 0));
std::vector<std::vector<int>> cnty(xpos.size(), std::vector<int>(ypos.size(), 0));
std::vector<std::vector<int>> idx(xpos.size(), std::vector<int>(ypos.size(), -1));
for (auto e : loc2node) {
int x = std::lower_bound(xpos.begin(), xpos.end(), e.first / N) - xpos.begin();
int y = std::lower_bound(ypos.begin(), ypos.end(), e.first % N) - ypos.begin();
// printf("%d %d -> %d\n", x, y, e.second);
idx[x][y] = e.second;
}
for (int i = 0; i < degree * 2 - 2; i++) {
Branch &branch1 = flutetree.branch[i], &branch2 = flutetree.branch[branch1.n];
int id1 = loc2node[branch1.x * N + branch1.y], id2 = loc2node[branch2.x * N + branch2.y];
if (id1 == id2) continue;
int x1 = node2loc[id1] / N, y1 = node2loc[id1] % N;
int x2 = node2loc[id2] / N, y2 = node2loc[id2] % N;
// printf("%d %d %d %d\n", x1, y1, x2, y2);
x1 = std::lower_bound(xpos.begin(), xpos.end(), x1) - xpos.begin();
x2 = std::lower_bound(xpos.begin(), xpos.end(), x2) - xpos.begin();
y1 = std::lower_bound(ypos.begin(), ypos.end(), y1) - ypos.begin();
y2 = std::lower_bound(ypos.begin(), ypos.end(), y2) - ypos.begin();
// printf("%d %d %d %d\n", x1, y1, x2, y2);
if (x1 != x2 && y1 != y2) {
graph[id1].insert(id2), graph[id2].insert(id1);
/* if(locations.count(x1 * N + y2) || locations.count(x2 * N + y1))
std::cerr << "ERROR & ERROR: BAD FLUTE RESULTS\n";
for(int i = std::min(x1, x2); i <= std::max(x1, x2); i++)
if(locations.count(i * N + y1) || locations.count(i * N + y2))
std::cerr << "ERROR: BAD FLUTE RESULTS\n";
for(int i = std::min(y1, y2); i <= std::max(y1, y2); i++)
if(locations.count(x1 * N + i) || locations.count(x2 * N + i))
std::cerr << "ERROR: BAD FLUTE RESULTS\n";
*/
} else {
if (x1 == x2)
for (int t = std::min(y1, y2); t < std::max(y1, y2); t++) cnty[x1][t]++;
else
for (int t = std::min(x1, x2); t < std::max(x1, x2); t++) cntx[t][y2]++;
}
}
free(flutetree.branch);
// printf("cnt = %d, %d %d\n", cntx[0][0], idx[0][0], idx[1][0]);
for (int i = 0; i < xpos.size(); i++) {
int last = -1;
for (int j = 0; j < (int)ypos.size(); j++) {
if (j && cnty[i][j - 1] == 0) last = -1;
int cur = -1;
if (idx[i][j] >= 0) cur = idx[i][j];
if (cur >= 0 && last >= 0 && cur != last) graph[cur].insert(last), graph[last].insert(cur);
if (cur >= 0) last = cur;
}
}
for (int i = 0; i < ypos.size(); i++) {
int last = -1;
for (int j = 0; j < (int)xpos.size(); j++) {
if (j && cntx[j - 1][i] == 0) last = -1;
int cur = -1;
if (idx[j][i] >= 0) cur = idx[j][i];
if (cur >= 0 && last >= 0 && cur != last) graph[cur].insert(last), graph[last].insert(cur);
// printf("i=%d, j=%d, cur=%d,last=%d\n", i, j, cur, last);
if (cur >= 0) last = cur;
}
}
std::vector<int> vis(node_cnt, 0);
for (int i = 0; i < node_cnt; i++) {
if (graph[i].size() > 4) {
std::cerr << "ERROR in FLUTE Results\n";
exit(-1);
}
}
points.emplace_back(routesOffset);
points.emplace_back(node_cnt);
int len = 0;
// for each point in points
// points[0]: location
// points[1, 2]: min and max layer
// points[3, 4, 5, 6] children locations
std::function<void(int)> dfs = [&](int x) {
vis[x] = 1;
int startlen = len;
points.emplace_back(node2loc[x]);
len++;
if (loc2Pins.count(node2loc[x])) {
auto temp = loc2Pins[node2loc[x]];
points.emplace_back(*std::min_element(temp.begin(), temp.end()));
points.emplace_back(*std::max_element(temp.begin(), temp.end()));
len += 2;
} else {
points.emplace_back(-1);
points.emplace_back(-1);
len += 2;
}
for (auto e : graph[x]) {
if (!vis[e]) {
points.emplace_back(node2loc[e]);
len++;
}
}
if (len - startlen > 6) {
printf(" %d ERROR in len\n", (int)graph[x].size());
}
while (len % 6 != 0) {
points.emplace_back(-1);
len++;
}
for (auto e : graph[x]) {
if (!vis[e]) dfs(e);
}
};
dfs(0);
if (len != 6 * node_cnt) {
for (int i = 0; i < pins.size(); i++) {
int layer = pins[i][0] / N / N, _x = pins[i][0] / N % N, _y = pins[i][0] % N;
if (!(layer & 1) ^ DIRECTION) std::swap(_x, _y);
printf("(%d, %d)\n", _x, _y);
}
for (auto e : loc2node) printf("(%d, %d)_%d ", e.first / N, e.first % N, e.second);
puts("");
for (int i = 0; i < node_cnt; i++)
for (auto e : graph[i]) printf("E(%d, %d) ", i, e);
puts("");
std::cerr << "ERROR in pattern routing preparation" << std::endl;
}
}
void prepareGrNets(std::vector<gr::GrNet> &grNets,
std::vector<int> &netsToRoute,
std::vector<int> &batchSizes,
std::vector<std::vector<int>> &points_cpu_vec,
std::vector<std::tuple<int, int, int>> &batchId2vec_info,
int *routesOffsetCPU,
int X,
int Y,
int N,
int LAYER,
int DIRECTION) {
// FIXME: the vanilla version of FLUTE cannot support multi-threads. If need MT, please
// change the FLUTE to the version in https://github.com/The-OpenROAD-Project-Attic/flute3
int totalBs = 0;
int numThreads = 1; // multi-thread is not supported by FLUTE
for (int batchSize : batchSizes) {
totalBs += batchSize;
}
auto thread_func = [&](int threadIdx) {
for (int i = threadIdx; i < totalBs; i += numThreads) {
int netId = netsToRoute[i];
prepareSingeNet(grNets[netId], routesOffsetCPU[netId], X, Y, N, LAYER, DIRECTION);
}
};
std::thread threads[numThreads];
for (int j = 0; j < numThreads; j++) {
threads[j] = std::thread(thread_func, j);
}
for (auto &t : threads) {
t.join();
}
logger.info("Finish Flute %d");
points_cpu_vec.clear();
batchId2vec_info.clear();
batchId2vec_info.resize(batchSizes.size());
int startpos = 0;
constexpr int MAX_POINTS_SIZE = 20000000;
points_cpu_vec.push_back(std::vector<int>());
points_cpu_vec.back().reserve(MAX_POINTS_SIZE);
for (int batchId = 0; batchId < batchSizes.size(); batchId++) {
int batchSize = batchSizes[batchId];
if (batchSize == 0) continue;
int offset = batchSize;
std::vector<int> curBatch_points(batchSize, -1);
for (int i = 0; i < batchSize; i++) {
// the first batchSize elements indicate the offset
curBatch_points[i] = offset;
int netId = netsToRoute[startpos + i];
offset += grNets[netId].points.size();
// std::cout << batchSize << " " << i << " BigVecId " << points_cpu_vec.size() << " " <<
// curBatch_points.size() << " " << points_cpu_vec.back().size() << " " << grNets[netId].getPins().size() <<
// " " << grNets[netId].points.size() << std::endl;
curBatch_points.insert(curBatch_points.end(),
std::make_move_iterator(grNets[netId].points.begin()),
std::make_move_iterator(grNets[netId].points.end()));
}
int startIdx, endIdx, inBigVecId;
if (points_cpu_vec.back().size() + curBatch_points.size() < MAX_POINTS_SIZE) {
startIdx = points_cpu_vec.back().size();
endIdx = startIdx + curBatch_points.size();
auto &tmp = points_cpu_vec.back();
tmp.insert(tmp.end(),
std::make_move_iterator(curBatch_points.begin()),
std::make_move_iterator(curBatch_points.end()));
inBigVecId = points_cpu_vec.size() - 1;
} else {
startIdx = 0;
endIdx = curBatch_points.size();
points_cpu_vec.emplace_back(std::move(curBatch_points));
points_cpu_vec.back().reserve(MAX_POINTS_SIZE);
inBigVecId = points_cpu_vec.size() - 1;
}
batchId2vec_info[batchId] = {inBigVecId, startIdx, endIdx};
startpos += batchSize;
}
logger.info("#BigVec %d", points_cpu_vec.size());
}
int prepare(double &count,
gr::GrNet &grNet,
int *points,
int routesOffset,
int *gbpoints,
int &gbPinOffset,
int X,
int Y,
int N,
int LAYER,
int DIRECTION) {
auto &pins = grNet.getPins();
// std::map<int, std::vector<int>> loc2Pins;
robin_hood::unordered_map<int, std::vector<int>> loc2Pins;
robin_hood::unordered_map<int, int> loc2pinIds;
// double startTimer = clock();
std::vector<int> xpos(pins.size()), ypos(pins.size());
for (int i = 0; i < pins.size(); i++) {
int layer = pins[i][0] / N / N, _x = pins[i][0] / N % N, _y = pins[i][0] % N;
if (!(layer & 1) ^ DIRECTION) std::swap(_x, _y);
xpos[i] = _x;
ypos[i] = _y;
auto& loc2PinsLayerVec = loc2Pins[_x * N + _y];
loc2PinsLayerVec.emplace_back(layer);
loc2pinIds[_x * N + _y] = i;
// if(pins[i].size() > 1) {
// int layer = pins[i][pins[i].size() - 1] / N / N;
// loc2PinsLayerVec.emplace_back(layer);
// }
}
std::sort(xpos.begin(), xpos.end());
std::sort(ypos.begin(), ypos.end());
xpos.erase(unique(xpos.begin(), xpos.end()), xpos.end());
ypos.erase(unique(ypos.begin(), ypos.end()), ypos.end());
int degree = loc2Pins.size(), cur = 0;
if (degree == 0) std::cerr << "ERROR: degree 0" << std::endl;
constexpr int MAX_DEGREE = 100000;
if (degree > MAX_DEGREE) std::cerr << "Not Enough X and Y in Pattern Routing" << std::endl;
int x[degree * 4], y[degree * 4];
for (auto e : loc2Pins) x[cur] = e.first / N, y[cur] = e.first % N, cur++;
Tree flutetree = flute(degree, x, y, 3);
// count += clock() - startTimer;
robin_hood::unordered_map<int, int> loc2node, node2loc;
std::set<int> locations;
int node_cnt = 0;
for (int i = 0; i < degree * 2 - 2; i++) locations.insert(flutetree.branch[i].x * N + flutetree.branch[i].y);
for (auto e : locations) {
node2loc[loc2node[e] = node_cnt++] = e;
if (!loc2pinIds.contains(e)) {
// e is not a real pin position but is a pseudo pin generated by RSMT
loc2pinIds[e] = -1;
}
}
// std::vector<std::set<int>> graph(node_cnt);
std::vector<robin_hood::unordered_set<int>> graph(node_cnt);
std::vector<std::vector<int>> cntx(xpos.size(), std::vector<int>(ypos.size(), 0));
std::vector<std::vector<int>> cnty(xpos.size(), std::vector<int>(ypos.size(), 0));
std::vector<std::vector<int>> idx(xpos.size(), std::vector<int>(ypos.size(), -1));
for (auto e : loc2node) {
int x = lower_bound(xpos.begin(), xpos.end(), e.first / N) - xpos.begin();
int y = lower_bound(ypos.begin(), ypos.end(), e.first % N) - ypos.begin();
// printf("%d %d -> %d\n", x, y, e.second);
idx[x][y] = e.second;
}
for (int i = 0; i < degree * 2 - 2; i++) {
Branch &branch1 = flutetree.branch[i], &branch2 = flutetree.branch[branch1.n];
int id1 = loc2node[branch1.x * N + branch1.y], id2 = loc2node[branch2.x * N + branch2.y];
if (id1 == id2) continue;
int x1 = node2loc[id1] / N, y1 = node2loc[id1] % N;
int x2 = node2loc[id2] / N, y2 = node2loc[id2] % N;
// printf("%d %d %d %d\n", x1, y1, x2, y2);
x1 = lower_bound(xpos.begin(), xpos.end(), x1) - xpos.begin();
x2 = lower_bound(xpos.begin(), xpos.end(), x2) - xpos.begin();
y1 = lower_bound(ypos.begin(), ypos.end(), y1) - ypos.begin();
y2 = lower_bound(ypos.begin(), ypos.end(), y2) - ypos.begin();
// printf("%d %d %d %d\n", x1, y1, x2, y2);
if (x1 != x2 && y1 != y2) {
graph[id1].insert(id2), graph[id2].insert(id1);
/* if(locations.count(x1 * N + y2) || locations.count(x2 * N + y1))
std::cerr << "ERROR & ERROR: BAD FLUTE RESULTS\n";
for(int i = min(x1, x2); i <= max(x1, x2); i++)
if(locations.count(i * N + y1) || locations.count(i * N + y2))
std::cerr << "ERROR: BAD FLUTE RESULTS\n";
for(int i = min(y1, y2); i <= max(y1, y2); i++)
if(locations.count(x1 * N + i) || locations.count(x2 * N + i))
std::cerr << "ERROR: BAD FLUTE RESULTS\n";
*/
} else {
if (x1 == x2)
for (int t = min(y1, y2); t < max(y1, y2); t++) cnty[x1][t]++;
else
for (int t = min(x1, x2); t < max(x1, x2); t++) cntx[t][y2]++;
}
}
free(flutetree.branch);
// NOTE: When a_x < b_x < c_x and a_y == b_y == c_y, FLUTE may report two edges A-B, A-C,
// here we fix it to A-B, B-C
// printf("cnt = %d, %d %d\n", cntx[0][0], idx[0][0], idx[1][0]);
for (int i = 0; i < xpos.size(); i++) {
int last = -1;
for (int j = 0; j < (int)ypos.size(); j++) {
if (j && cnty[i][j - 1] == 0) last = -1;
int cur = -1;
if (idx[i][j] >= 0) cur = idx[i][j];
if (cur >= 0 && last >= 0 && cur != last) graph[cur].insert(last), graph[last].insert(cur);
if (cur >= 0) last = cur;
}
}
for (int i = 0; i < ypos.size(); i++) {
int last = -1;
for (int j = 0; j < (int)xpos.size(); j++) {
if (j && cntx[j - 1][i] == 0) last = -1;
int cur = -1;
if (idx[j][i] >= 0) cur = idx[j][i];
if (cur >= 0 && last >= 0 && cur != last) graph[cur].insert(last), graph[last].insert(cur);
// printf("i=%d, j=%d, cur=%d,last=%d\n", i, j, cur, last);
if (cur >= 0) last = cur;
}
}
// NOTE: Fix corner cases, graph[i] includes 4 straight edges and >= 1 bevel edges
for (int i = 0; i < node_cnt; i++) {
if (graph[i].size() > 4) {
std::vector<int> movedIds;
int thisX = node2loc[i] / N, thisY = node2loc[i] % N;
for (auto childId : graph[i]) {
int childX = node2loc[childId] / N, childY = node2loc[childId] % N;
if (childX != thisX && childY != thisY) {
movedIds.emplace_back(childId);
}
}
for (auto childId : movedIds) {
std::queue<int> q;
std::vector<bool> possibleSet(node_cnt, false);
q.push(i);
while (q.size() > 0) {
int cur = q.front();
q.pop();
if (possibleSet[cur]) continue;
possibleSet[cur] = true;
for (auto c : graph[cur]) {
if (!possibleSet[c] && c != childId) {
q.push(c);
}
}
}
if (graph[i].size() == 4) break;
int childX = node2loc[childId] / N, childY = node2loc[childId] % N;
int minDist = std::numeric_limits<int>::max();
int new_i = -1;
for (int j = 0; j < node_cnt; j++) {
if (!possibleSet[j]) continue;
if (j == i || j == childId) continue;
if (graph[j].size() < 4) {
int tarX = node2loc[j] / N, tarY = node2loc[j] % N;
int dist = std::abs(tarX - childX) + std::abs(tarY - childY);
if (dist == 0) continue;
if (dist < minDist) {
minDist = dist;
new_i = j;
}
}
}
if (new_i == -1) {
continue;
}
graph[i].erase(childId);
graph[childId].erase(i);
graph[childId].insert(new_i);
graph[new_i].insert(childId);
}
}
}
for (int i = 0; i < node_cnt; i++) {
if (graph[i].size() > 4) {
std::cerr << "ERROR in FLUTE Results\n";
printf("(%d %d) Childs: ", node2loc[i] / N, node2loc[i] % N);
for (auto e : graph[i]) {
printf("(%d %d) ", node2loc[e] / N, node2loc[e] % N);
}
printf("\nAll pts: ");
for (int j = 0; j < node_cnt; j++) {
printf("(%d %d) ", node2loc[j] / N, node2loc[j] % N);
}
std::cout << std::endl;
exit(-1);
}
}
std::vector<int> vis(node_cnt, 0);
points[0] = routesOffset;
points[1] = node_cnt;
int len = 0;
points += 2;
// points[0]: location
// points[1, 2]: min and max layer
// points[3, 4, 5] children locations
std::function<void(int)> dfs = [&](int x) {
vis[x] = 1;
int startlen = len;
// points[0]: location
int loc = node2loc[x];
points[len++] = loc;
if (loc2Pins.count(loc)) {
// points[1, 2]: min and max layer
auto temp = loc2Pins[loc];
points[len++] = *std::min_element(temp.begin(), temp.end());
points[len++] = *std::max_element(temp.begin(), temp.end());
} else
points[len++] = -1, points[len++] = -1;
// points[3, 4, 5] children locations
for (auto e : graph[x])
if (!vis[e]) points[len++] = node2loc[e];
if (len - startlen > 6) printf(" %d ERROR in len\n", (int)graph[x].size());
while (len % 6 != 0) points[len++] = -1;
for (auto e : graph[x])
if (!vis[e]) dfs(e);
};
dfs(0);
if (len != 6 * node_cnt) {
printf("len: %d node_cnt: %d\n", len, node_cnt);
for (int i = 0; i < pins.size(); i++) {
int layer = pins[i][0] / N / N, _x = pins[i][0] / N % N, _y = pins[i][0] % N;
if (!(layer & 1) ^ DIRECTION) std::swap(_x, _y);
printf("(%d, %d)\n", _x, _y);
}
for (auto e : loc2node) printf("(%d, %d)_%d ", e.first / N, e.first % N, e.second);
puts("");
for (int i = 0; i < node_cnt; i++)
for (auto e : graph[i]) printf("E(%d, %d) ", i, e);
puts("");
std::cerr << "ERROR in pattern routing preparation" << std::endl;
}
// rewrite points child
robin_hood::unordered_map<int, int> loc2point_id;
for (int i = 0; i < node_cnt; i++) {
loc2point_id[points[i * 6]] = i * 6;
}
for (int i = 0; i < node_cnt; i++) {
for (int j = 3; j < 6; j++) {
if (points[i * 6 + j] == -1) continue;
points[i * 6 + j] = loc2point_id[points[i * 6 + j]];
}
}
// for (int i = 0; i < node_cnt; i++) {
// int x = points[i * 6] / N, y = points[i * 6] % N;
// if (x > grNet.upperx || x < grNet.lowerx || y > grNet.uppery || y < grNet.lowery) {
// printf("pin: (%d, %d) out of boundary of net_bbox: (%d, %d, %d, %d)\n",
// x, y, grNet.lowerx, grNet.lowery, grNet.upperx, grNet.uppery);
// }
// }
// gbpoints
for (int i = 0; i < node_cnt; i++) {
int pinId = loc2pinIds[points[i * 6]];
if (pinId == -1) {
gbpoints[i] = -1;
} else {
gbpoints[i] = grNet.pin2gbpinId[pinId];
}
}
gbPinOffset += node_cnt;
// if (node_cnt > pins.size() && node_cnt > 10) {
// std::cout << "node_cnt " << node_cnt << ", #gbpins " << pins.size() << std::endl;
// for (int i = 0; i < node_cnt; i++) {
// std::cout << points[i * 6] << " ";
// }
// std::cout << std::endl;
// for (int i = 0; i < node_cnt; i++) {
// std::cout << gbpoints[i] << " ";
// }
// std::cout << std::endl;
// for (int i = 0; i < node_cnt; i++) {
// int loc = points[i * 6];
// if (loc2pinIds[loc] != -1) {
// int pinid = loc2pinIds[loc];
// int layer = pins[pinid][0] / N / N, _x = pins[pinid][0] / N % N, _y = pins[pinid][0] % N;
// std::cout << _x * N + _y << " ";
// } else {
// std::cout << "xxxxxx" << " ";
// }
// }
// std::cout << std::endl;
// exit(0);
// }
return len + 2;
}
} // namespace gr

File diff suppressed because it is too large Load Diff

View File

@ -0,0 +1,49 @@
#pragma once
#include "common/common.h"
#include "gpugr/db/GrNet.h"
namespace gr {
int prepare(double &count,
gr::GrNet &grNet,
int *points,
int routesOffset,
int *gbpoints,
int &gbPinOffset,
int X,
int Y,
int N,
int LAYER,
int DIRECTION);
void prepareSingeNet(gr::GrNet &grNet, int routesOffset, int X, int Y, int N, int LAYER, int DIRECTION);
void prepareGrNets(std::vector<gr::GrNet> &grNets,
std::vector<int> &netsToRoute,
std::vector<int> &batchSizes,
std::vector<std::vector<int>> &points_cpu_vec,
std::vector<std::tuple<int, int, int>> &batchId2vec_info,
int *routesOffsetCPU,
int X,
int Y,
int N,
int LAYER,
int DIRECTION);
void patternRoute(int *points,
int batchSize,
int64_t *wireCostSum,
int *viaCost,
int *map,
int *prev,
int *wires,
int *vias,
int *routes,
int *gbpoints,
int *gbpinRoutes,
int X,
int Y,
int N,
int LAYER,
int DIRECTION);
} // namespace gr

View File

@ -0,0 +1,170 @@
#include "RouteForce.h"
namespace gr {
RouteForce::RouteForce(std::shared_ptr<gr::GRDatabase> grdb_) : grdb(*grdb_) {}
void RouteForce::run_ggr() {
logger.enable_logger();
utils::timer T_total;
T_total.start();
// we only need the PR segment, our current data structure unsupport MR route force
// if rrrIters > 0, this router can only be used for congestion map computation or solution evaluation
int runMazeRouteTimes = grSetting.rrrIters;
// Parameters
int rrrIterLimit = 1 + runMazeRouteTimes;
double _unitWireCostRaw = 0.5 * grdb.microns / grdb.m2pitch;
double _unitViaCostRaw = 4;
double _unitViaCost = _unitViaCostRaw / _unitWireCostRaw * grdb.microns;
double _unitShortVioCostRaw = 500;
double rrrInitVioCostDiscount = 0.1;
router.initialize(grSetting.deviceId,
grdb.nLayers,
grdb.xSize,
grdb.ySize,
grdb.nMaxGrid,
grdb.cgxsize,
grdb.cgysize,
grdb.m1direction,
grdb.csrnScale);
router.setMap(grdb.capacity, grdb.wireDist, grdb.fixedLength, grdb.fixedUsage);
std::vector<float> _unitShortVioCost(grdb.nLayers), _unitShortVioCostDiscounted(grdb.nLayers);
router.setFromNets(grdb.grNets, grdb.gpdb.getPins().size());
router.setUnitViaCost(_unitViaCost);
for (int i = 0; i < grdb.nLayers; ++i) {
_unitShortVioCost[i] =
_unitShortVioCostRaw * grdb.layerWidth[i] * grdb.microns / grdb.m2pitch / grdb.m2pitch / _unitWireCostRaw;
}
double tot_time = 0;
for (int iter = 0; iter < rrrIterLimit; iter++) {
router.setLogisticSlope(1 << iter);
router.setUnitVioCost(_unitShortVioCost, 0.1);
if (iter == 0) {
router.setUnitViaMultiplier(1);
} else {
router.setUnitViaMultiplier(max(100 / pow(5, iter - 1), 4.0));
router.setUnitVioCost(_unitShortVioCost,
rrrInitVioCostDiscount + (1.0 - rrrInitVioCostDiscount) / (rrrIterLimit - 1) * iter);
}
utils::timer T;
T.start();
router.route(grdb.grNets, iter);
tot_time += T.elapsed();
logger.info("##### GPU Routing Iter: %d Time: %.4f #####", iter, T.elapsed());
// break;
}
router.setToNets(grdb.grNets);
logger.info("Total GPU Routing time: %.4f", tot_time);
if (grSetting.routeGuideFile != "") {
grdb.writeGuides(grSetting.routeGuideFile);
}
logger.info("Total GPU GR Time: %.4f", T_total.elapsed());
logger.reset_logger();
}
std::tuple<torch::Tensor, torch::Tensor, torch::Tensor> RouteForce::getDemandMap() { return router.getDemandMap(); }
torch::Tensor RouteForce::getCapacityMap() { return router.getCapacityMap(); }
torch::Tensor RouteForce::calcRouteGrad(torch::Tensor mask_map,
torch::Tensor wire_dmd_map_2d,
torch::Tensor via_dmd_map_2d,
torch::Tensor cap_map_2d,
torch::Tensor dist_weights,
torch::Tensor wirelength_weights,
torch::Tensor route_gradmat,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
float grad_weight,
float unit_wire_cost,
float unit_via_cost,
int num_nodes) {
return router.calcRouteGrad(mask_map,
wire_dmd_map_2d,
via_dmd_map_2d,
cap_map_2d,
dist_weights,
wirelength_weights,
route_gradmat,
node2pin_list,
node2pin_list_end,
grad_weight,
unit_wire_cost,
unit_via_cost,
num_nodes);
};
torch::Tensor RouteForce::calcFillerRouteGrad(torch::Tensor filler_pos,
torch::Tensor filler_size,
torch::Tensor filler_weight,
torch::Tensor expand_ratio,
torch::Tensor grad_mat,
float grad_weight,
float unit_len_x,
float unit_len_y,
int num_bin_x,
int num_bin_y,
int num_fillers) {
return router.calcFillerRouteGrad(filler_pos,
filler_size,
filler_weight,
expand_ratio,
grad_mat,
grad_weight,
unit_len_x,
unit_len_y,
num_bin_x,
num_bin_y,
num_fillers);
}
torch::Tensor RouteForce::calcPseudoPinGrad(torch::Tensor node_pos, torch::Tensor pseudo_pin_pos, float gamma) {
return router.calcPseudoPinGrad(node_pos, pseudo_pin_pos, gamma);
}
torch::Tensor RouteForce::calcNodeInflateRatio(torch::Tensor node_pos,
torch::Tensor node_size,
torch::Tensor node_weight,
torch::Tensor expand_ratio,
torch::Tensor inflate_mat,
float grad_weight,
float unit_len_x,
float unit_len_y,
int num_bin_x,
int num_bin_y,
bool use_weighted_inflation) {
return router.calcNodeInflateRatio(node_pos,
node_size,
node_weight,
expand_ratio,
inflate_mat,
grad_weight,
unit_len_x,
unit_len_y,
num_bin_x,
num_bin_y,
use_weighted_inflation);
}
torch::Tensor RouteForce::calcInflatedPinRelCpos(torch::Tensor node_inflate_ratio,
torch::Tensor old_pin_rel_cpos,
torch::Tensor pin_id2node_id,
int num_movable_conn_nodes) {
return router.calcInflatedPinRelCpos(node_inflate_ratio, old_pin_rel_cpos, pin_id2node_id, num_movable_conn_nodes);
}
int RouteForce::getNumOvflNets() { return router.getNumOvflNets(); }
int RouteForce::getMicrons() { return grdb.microns; }
std::tuple<int, int> RouteForce::getGcellStep() { return {grdb.mainGcellStepX, grdb.mainGcellStepY}; }
std::vector<int> RouteForce::getLayerPitch() { return grdb.layerPitch; }
std::vector<int> RouteForce::getLayerWidth() { return grdb.layerWidth; }
} // namespace gr

View File

@ -0,0 +1,68 @@
#pragma once
#include "common/common.h"
#include "common/db/Database.h"
#include "gpugr/db/GRDatabase.h"
#include "gpugr/gr/GPURouter.h"
namespace gr {
class RouteForce {
public:
RouteForce(std::shared_ptr<gr::GRDatabase> grdb_);
void run_ggr();
public:
std::tuple<torch::Tensor, torch::Tensor, torch::Tensor> getDemandMap();
torch::Tensor getCapacityMap();
torch::Tensor calcRouteGrad(torch::Tensor mask_map,
torch::Tensor wire_dmd_map_2d,
torch::Tensor via_dmd_map_2d,
torch::Tensor cap_map_2d,
torch::Tensor dist_weights,
torch::Tensor wirelength_weights,
torch::Tensor route_gradmat,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
float grad_weight,
float unit_wire_cost,
float unit_via_cost,
int num_nodes);
torch::Tensor calcFillerRouteGrad(torch::Tensor filler_pos,
torch::Tensor filler_size,
torch::Tensor filler_weight,
torch::Tensor expand_ratio,
torch::Tensor grad_mat,
float grad_weight,
float unit_len_x,
float unit_len_y,
int num_bin_x,
int num_bin_y,
int num_fillers);
torch::Tensor calcPseudoPinGrad(torch::Tensor node_pos, torch::Tensor pseudo_pin_pos, float gamma);
torch::Tensor calcNodeInflateRatio(torch::Tensor node_pos,
torch::Tensor node_size,
torch::Tensor node_weight,
torch::Tensor expand_ratio,
torch::Tensor inflate_mat,
float grad_weight,
float unit_len_x,
float unit_len_y,
int num_bin_x,
int num_bin_y,
bool use_weighted_inflation);
torch::Tensor calcInflatedPinRelCpos(torch::Tensor node_inflate_ratio,
torch::Tensor old_pin_rel_cpos,
torch::Tensor pin_id2node_id,
int num_movable_nodes);
int getNumOvflNets();
int getMicrons();
std::tuple<int, int> getGcellStep();
std::vector<int> getLayerPitch();
std::vector<int> getLayerWidth();
private:
gr::GRDatabase& grdb;
gr::GPURouter router;
};
} // namespace gr

View File

@ -0,0 +1,78 @@
#pragma once
#include "../task.hpp"
/**
@file critical.hpp
@brief critical include file
*/
namespace tf {
// ----------------------------------------------------------------------------
// CriticalSection
// ----------------------------------------------------------------------------
/**
@class CriticalSection
@brief class to create a critical region of limited workers to run tasks
tf::CriticalSection is a warpper over tf::Semaphore and is specialized for
limiting the maximum concurrency over a set of tasks.
A critical section starts with an initial count representing that limit.
When a task is added to the critical section,
the task acquires and releases the semaphore internal to the critical section.
This design avoids explicit call of tf::Task::acquire and tf::Task::release.
The following example creates a critical section of one worker and adds
the five tasks to the critical section.
@code{.cpp}
tf::Executor executor(8); // create an executor of 8 workers
tf::Taskflow taskflow;
// create a critical section of 1 worker
tf::CriticalSection critical_section(1);
tf::Task A = taskflow.emplace([](){ std::cout << "A" << std::endl; });
tf::Task B = taskflow.emplace([](){ std::cout << "B" << std::endl; });
tf::Task C = taskflow.emplace([](){ std::cout << "C" << std::endl; });
tf::Task D = taskflow.emplace([](){ std::cout << "D" << std::endl; });
tf::Task E = taskflow.emplace([](){ std::cout << "E" << std::endl; });
critical_section.add(A, B, C, D, E);
executor.run(taskflow).wait();
@endcode
*/
class CriticalSection : public Semaphore {
public:
/**
@brief constructs a critical region of a limited number of workers
*/
explicit CriticalSection(int max_workers = 1);
/**
@brief adds a task into the critical region
*/
template <typename... Tasks>
void add(Tasks...tasks);
};
inline CriticalSection::CriticalSection(int max_workers) :
Semaphore {max_workers} {
}
template <typename... Tasks>
void CriticalSection::add(Tasks... tasks) {
(tasks.acquire(*this), ...);
(tasks.release(*this), ...);
}
} // end of namespace tf. ---------------------------------------------------

View File

@ -0,0 +1,203 @@
// reference:
// - gomp: https://github.com/gcc-mirror/gcc/blob/master/libgomp/iter.c
// - komp: https://github.com/llvm-mirror/openmp/blob/master/runtime/src/kmp_dispatch.cpp
#pragma once
#include "../executor.hpp"
namespace tf {
// ----------------------------------------------------------------------------
// default parallel for
// ----------------------------------------------------------------------------
// Function: for_each
template <typename B, typename E, typename C>
Task FlowBuilder::for_each(B&& beg, E&& end, C c) {
using I = stateful_iterator_t<B, E>;
using namespace std::string_literals;
Task task = emplace(
[b=std::forward<B>(beg), e=std::forward<E>(end), c] (Subflow& sf) mutable {
// fetch the stateful values
I beg = b;
I end = e;
if(beg == end) {
return;
}
size_t chunk_size = 1;
size_t W = sf._executor.num_workers();
size_t N = std::distance(beg, end);
// only myself - no need to spawn another graph
if(W <= 1 || N <= chunk_size) {
std::for_each(beg, end, c);
return;
}
if(N < W) {
W = N;
}
std::atomic<size_t> next(0);
for(size_t w=0; w<W; w++) {
//sf.emplace([&next, beg, N, chunk_size, W, c] () mutable {
sf.silent_async([&next, beg, N, chunk_size, W, c] () mutable {
size_t z = 0;
size_t p1 = 2 * W * (chunk_size + 1);
double p2 = 0.5 / static_cast<double>(W);
size_t s0 = next.load(std::memory_order_relaxed);
while(s0 < N) {
size_t r = N - s0;
// fine-grained
if(r < p1) {
while(1) {
s0 = next.fetch_add(chunk_size, std::memory_order_relaxed);
if(s0 >= N) {
return;
}
size_t e0 = (chunk_size <= (N - s0)) ? s0 + chunk_size : N;
std::advance(beg, s0-z);
for(size_t x=s0; x<e0; x++) {
c(*beg++);
}
z = e0;
}
break;
}
// coarse-grained
else {
size_t q = static_cast<size_t>(p2 * r);
if(q < chunk_size) {
q = chunk_size;
}
size_t e0 = (q <= r) ? s0 + q : N;
if(next.compare_exchange_strong(s0, e0, std::memory_order_relaxed,
std::memory_order_relaxed)) {
std::advance(beg, s0-z);
for(size_t x = s0; x< e0; x++) {
c(*beg++);
}
z = e0;
s0 = next.load(std::memory_order_relaxed);
}
}
}
//}).name("pfg_"s + std::to_string(w));
});
}
sf.join();
});
return task;
}
// Function: for_each_index
template <typename B, typename E, typename S, typename C>
Task FlowBuilder::for_each_index(B&& beg, E&& end, S&& inc, C c){
using I = stateful_index_t<B, E, S>;
using namespace std::string_literals;
Task task = emplace(
[b=std::forward<B>(beg), e=std::forward<E>(end), a=std::forward<S>(inc), c]
(Subflow& sf) mutable {
// fetch the iterator values
I beg = b;
I end = e;
I inc = a;
if(is_range_invalid(beg, end, inc)) {
TF_THROW("invalid range [", beg, ", ", end, ") with step size ", inc);
}
size_t chunk_size = 1;
size_t W = sf._executor.num_workers();
size_t N = distance(beg, end, inc);
// only myself - no need to spawn another graph
if(W <= 1 || N <= chunk_size) {
for(size_t x=0; x<N; x++, beg+=inc) {
c(beg);
}
return;
}
if(N < W) {
W = N;
}
std::atomic<size_t> next(0);
for(size_t w=0; w<W; w++) {
//sf.emplace([&next, beg, inc, N, chunk_size, W, c] () mutable {
sf.silent_async([&next, beg, inc, N, chunk_size, W, c] () mutable {
size_t p1 = 2 * W * (chunk_size + 1);
double p2 = 0.5 / static_cast<double>(W);
size_t s0 = next.load(std::memory_order_relaxed);
while(s0 < N) {
size_t r = N - s0;
// find-grained
if(r < p1) {
while(1) {
s0 = next.fetch_add(chunk_size, std::memory_order_relaxed);
if(s0 >= N) {
return;
}
size_t e0 = (chunk_size <= (N - s0)) ? s0 + chunk_size : N;
auto s = static_cast<I>(s0) * inc + beg;
for(size_t x=s0; x<e0; x++, s+=inc) {
c(s);
}
}
break;
}
// coarse-grained
else {
size_t q = static_cast<size_t>(p2 * r);
if(q < chunk_size) {
q = chunk_size;
}
size_t e0 = (q <= r) ? s0 + q : N;
if(next.compare_exchange_strong(s0, e0, std::memory_order_relaxed,
std::memory_order_relaxed)) {
auto s = static_cast<I>(s0) * inc + beg;
for(size_t x=s0; x<e0; x++, s+= inc) {
c(s);
}
s0 = next.load(std::memory_order_relaxed);
}
}
}
//}).name("pfg_"s + std::to_string(w));
});
}
sf.join();
});
return task;
}
} // end of namespace tf -----------------------------------------------------

View File

@ -0,0 +1,262 @@
#pragma once
#include "../executor.hpp"
namespace tf {
// ----------------------------------------------------------------------------
// default reduction
// ----------------------------------------------------------------------------
template <typename B, typename E, typename T, typename O>
Task FlowBuilder::reduce(B&& beg, E&& end, T& init, O bop) {
using I = stateful_iterator_t<B, E>;
using namespace std::string_literals;
Task task = emplace(
[b=std::forward<B>(beg), e=std::forward<E>(end), &r=init, bop]
(Subflow& sf) mutable {
// fetch the iterator values
I beg = b;
I end = e;
if(beg == end) {
return;
}
//size_t C = (c == 0) ? 1 : c;
size_t C = 1;
size_t W = sf._executor.num_workers();
size_t N = std::distance(beg, end);
// only myself - no need to spawn another graph
if(W <= 1 || N <= C) {
for(; beg!=end; r = bop(r, *beg++));
return;
}
if(N < W) {
W = N;
}
std::mutex mutex;
std::atomic<size_t> next(0);
for(size_t w=0; w<W; w++) {
if(w*2 >= N) {
break;
}
//sf.emplace([&mutex, &next, &r, beg, N, W, o, C] () mutable {
sf.silent_async([&mutex, &next, &r, beg, N, W, bop, C] () mutable {
size_t s0 = next.fetch_add(2, std::memory_order_relaxed);
if(s0 >= N) {
return;
}
std::advance(beg, s0);
if(N - s0 == 1) {
std::lock_guard<std::mutex> lock(mutex);
r = bop(r, *beg);
return;
}
auto beg1 = beg++;
auto beg2 = beg++;
T sum = bop(*beg1, *beg2);
size_t z = s0 + 2;
size_t p1 = 2 * W * (C + 1);
double p2 = 0.5 / static_cast<double>(W);
s0 = next.load(std::memory_order_relaxed);
while(s0 < N) {
size_t r = N - s0;
// fine-grained
if(r < p1) {
while(1) {
s0 = next.fetch_add(C, std::memory_order_relaxed);
if(s0 >= N) {
break;
}
size_t e0 = (C <= (N - s0)) ? s0 + C : N;
std::advance(beg, s0-z);
for(size_t x=s0; x<e0; x++, beg++) {
sum = bop(sum, *beg);
}
z = e0;
}
break;
}
// coarse-grained
else {
size_t q = static_cast<size_t>(p2 * r);
if(q < C) {
q = C;
}
size_t e0 = (q <= r) ? s0 + q : N;
if(next.compare_exchange_strong(s0, e0, std::memory_order_relaxed,
std::memory_order_relaxed)) {
std::advance(beg, s0-z);
for(size_t x = s0; x<e0; x++, beg++) {
sum = bop(sum, *beg);
}
z = e0;
s0 = next.load(std::memory_order_relaxed);
}
}
}
std::lock_guard<std::mutex> lock(mutex);
r = bop(r, sum);
//}).name("prg_"s + std::to_string(w));
});
}
sf.join();
});
return task;
}
// ----------------------------------------------------------------------------
// default transform and reduction
// ----------------------------------------------------------------------------
template <typename B, typename E, typename T, typename BOP, typename UOP>
Task FlowBuilder::transform_reduce(
B&& beg, E&& end, T& init, BOP bop, UOP uop
) {
using I = stateful_iterator_t<B, E>;
using namespace std::string_literals;
Task task = emplace(
[b=std::forward<B>(beg), e=std::forward<E>(end), &r=init, bop, uop]
(Subflow& sf) mutable {
// fetch the iterator values
I beg = b;
I end = e;
if(beg == end) {
return;
}
//size_t C = (c == 0) ? 1 : c;
size_t C = 1;
size_t W = sf._executor.num_workers();
size_t N = std::distance(beg, end);
// only myself - no need to spawn another graph
if(W <= 1 || N <= C) {
for(; beg!=end; r = bop(r, uop(*beg++)));
return;
}
if(N < W) {
W = N;
}
std::mutex mutex;
std::atomic<size_t> next(0);
for(size_t w=0; w<W; w++) {
if(w*2 >= N) {
break;
}
//sf.emplace([&mutex, &next, &r, beg, N, W, bop, uop, C] () mutable {
sf.silent_async([&mutex, &next, &r, beg, N, W, bop, uop, C] () mutable {
size_t s0 = next.fetch_add(2, std::memory_order_relaxed);
if(s0 >= N) {
return;
}
std::advance(beg, s0);
if(N - s0 == 1) {
std::lock_guard<std::mutex> lock(mutex);
r = bop(r, uop(*beg));
return;
}
auto beg1 = beg++;
auto beg2 = beg++;
T sum = bop(uop(*beg1), uop(*beg2));
size_t z = s0 + 2;
size_t p1 = 2 * W * (C + 1);
double p2 = 0.5 / static_cast<double>(W);
s0 = next.load(std::memory_order_relaxed);
while(s0 < N) {
size_t r = N - s0;
// fine-grained
if(r < p1) {
while(1) {
s0 = next.fetch_add(C, std::memory_order_relaxed);
if(s0 >= N) {
break;
}
size_t e0 = (C <= (N - s0)) ? s0 + C : N;
std::advance(beg, s0-z);
for(size_t x=s0; x<e0; x++, beg++) {
sum = bop(sum, uop(*beg));
}
z = e0;
}
break;
}
// coarse-grained
else {
size_t q = static_cast<size_t>(p2 * r);
if(q < C) {
q = C;
}
size_t e0 = (q <= r) ? s0 + q : N;
if(next.compare_exchange_strong(s0, e0, std::memory_order_relaxed,
std::memory_order_relaxed)) {
std::advance(beg, s0-z);
for(size_t x = s0; x<e0; x++, beg++) {
sum = bop(sum, uop(*beg));
}
z = e0;
s0 = next.load(std::memory_order_relaxed);
}
}
}
std::lock_guard<std::mutex> lock(mutex);
r = bop(r, sum);
//}).name("prg_"s + std::to_string(w));
});
}
sf.join();
});
return task;
}
} // end of namespace tf -----------------------------------------------------

View File

@ -0,0 +1,479 @@
#pragma once
#include "../executor.hpp"
namespace tf {
// threshold whether or not to perform parallel sort
template <typename I>
constexpr size_t parallel_sort_cutoff() {
//using value_type = std::decay_t<decltype(*std::declval<I>())>;
using value_type = typename std::iterator_traits<I>::value_type;
constexpr size_t object_size = sizeof(value_type);
if constexpr(std::is_same_v<value_type, std::string>) {
return 128;
}
else {
if constexpr(object_size < 16) return 4096;
else if constexpr(object_size < 32) return 2048;
else if constexpr(object_size < 64) return 1024;
else if constexpr(object_size < 128) return 768;
else if constexpr(object_size < 256) return 512;
else if constexpr(object_size < 512) return 256;
else return 128;
}
}
// ----------------------------------------------------------------------------
// pattern-defeating quick sort (pdqsort)
// ----------------------------------------------------------------------------
// Sorts [begin, end) using insertion sort with the given comparison function.
template<typename RandItr, typename Compare>
void insertion_sort(RandItr begin, RandItr end, Compare comp) {
using T = typename std::iterator_traits<RandItr>::value_type;
if (begin == end) {
return;
}
for (RandItr cur = begin + 1; cur != end; ++cur) {
RandItr shift = cur;
RandItr shift_1 = cur - 1;
// Compare first to avoid 2 moves for an element
// already positioned correctly.
if (comp(*shift, *shift_1)) {
T tmp = std::move(*shift);
do {
*shift-- = std::move(*shift_1);
}while (shift != begin && comp(tmp, *--shift_1));
*shift = std::move(tmp);
}
}
}
// Sorts [begin, end) using insertion sort with the given comparison function.
// Assumes *(begin - 1) is an element smaller than or equal to any element
// in [begin, end).
template<typename RandItr, typename Compare>
void unguarded_insertion_sort(RandItr begin, RandItr end, Compare comp) {
using T = typename std::iterator_traits<RandItr>::value_type;
if (begin == end) {
return;
}
for (RandItr cur = begin + 1; cur != end; ++cur) {
RandItr shift = cur;
RandItr shift_1 = cur - 1;
// Compare first so we can avoid 2 moves
// for an element already positioned correctly.
if (comp(*shift, *shift_1)) {
T tmp = std::move(*shift);
do {
*shift-- = std::move(*shift_1);
}while (comp(tmp, *--shift_1));
*shift = std::move(tmp);
}
}
}
// Attempts to use insertion sort on [begin, end).
// Will return false if more than
// partial_insertion_sort_limit elements were moved,
// and abort sorting. Otherwise it will successfully sort and return true.
template<typename RandItr, typename Compare>
bool partial_insertion_sort(RandItr begin, RandItr end, Compare comp) {
using T = typename std::iterator_traits<RandItr>::value_type;
using D = typename std::iterator_traits<RandItr>::difference_type;
// When we detect an already sorted partition, attempt an insertion sort
// that allows this amount of element moves before giving up.
constexpr auto partial_insertion_sort_limit = D{8};
if (begin == end) return true;
auto limit = D{0};
for (RandItr cur = begin + 1; cur != end; ++cur) {
if (limit > partial_insertion_sort_limit) {
return false;
}
RandItr shift = cur;
RandItr shift_1 = cur - 1;
// Compare first so we can avoid 2 moves
// for an element already positioned correctly.
if (comp(*shift, *shift_1)) {
T tmp = std::move(*shift);
do {
*shift-- = std::move(*shift_1);
}while (shift != begin && comp(tmp, *--shift_1));
*shift = std::move(tmp);
limit += cur - shift;
}
}
return true;
}
// Partitions [begin, end) around pivot *begin using comparison function comp.
// Elements equal to the pivot are put in the right-hand partition.
// Returns the position of the pivot after partitioning and whether the passed
// sequence already was correctly partitioned.
// Assumes the pivot is a median of at least 3 elements and that [begin, end)
// is at least insertion_sort_threshold long.
template<typename Iter, typename Compare>
std::pair<Iter, bool> partition_right(Iter begin, Iter end, Compare comp) {
using T = typename std::iterator_traits<Iter>::value_type;
// Move pivot into local for speed.
T pivot(std::move(*begin));
Iter first = begin;
Iter last = end;
// Find the first element greater than or equal than the pivot
// (the median of 3 guarantees/ this exists).
while (comp(*++first, pivot));
// Find the first element strictly smaller than the pivot.
// We have to guard this search if there was no element before *first.
if (first - 1 == begin) while (first < last && !comp(*--last, pivot));
else while (!comp(*--last, pivot));
// If the first pair of elements that should be swapped to partition
// are the same element, the passed in sequence already was correctly
// partitioned.
bool already_partitioned = first >= last;
// Keep swapping pairs of elements that are on the wrong side of the pivot.
// Previously swapped pairs guard the searches,
// which is why the first iteration is special-cased above.
while (first < last) {
std::iter_swap(first, last);
while (comp(*++first, pivot));
while (!comp(*--last, pivot));
}
// Put the pivot in the right place.
Iter pivot_pos = first - 1;
*begin = std::move(*pivot_pos);
*pivot_pos = std::move(pivot);
return std::make_pair(pivot_pos, already_partitioned);
}
// Similar function to the one above, except elements equal to the pivot
// are put to the left of the pivot and it doesn't check or return
// if the passed sequence already was partitioned.
// Since this is rarely used (the many equal case),
// and in that case pdqsort already has O(n) performance,
// no block quicksort is applied here for simplicity.
template<typename RandItr, typename Compare>
RandItr partition_left(RandItr begin, RandItr end, Compare comp) {
using T = typename std::iterator_traits<RandItr>::value_type;
T pivot(std::move(*begin));
RandItr first = begin;
RandItr last = end;
while (comp(pivot, *--last));
if (last + 1 == end) {
while (first < last && !comp(pivot, *++first));
}
else {
while (!comp(pivot, *++first));
}
while (first < last) {
std::iter_swap(first, last);
while (comp(pivot, *--last));
while (!comp(pivot, *++first));
}
RandItr pivot_pos = last;
*begin = std::move(*pivot_pos);
*pivot_pos = std::move(pivot);
return pivot_pos;
}
template<typename Iter, typename Compare>
void parallel_pdqsort(
tf::Subflow& sf,
Iter begin, Iter end, Compare comp,
int bad_allowed, bool leftmost = true
) {
// Partitions below this size are sorted sequentially
constexpr auto cutoff = parallel_sort_cutoff<Iter>();
// Partitions below this size are sorted using insertion sort
constexpr auto insertion_sort_threshold = 24;
// Partitions above this size use Tukey's ninther to select the pivot.
constexpr auto ninther_threshold = 128;
//using diff_t = typename std::iterator_traits<Iter>::difference_type;
// Use a while loop for tail recursion elimination.
while (true) {
//diff_t size = end - begin;
size_t size = end - begin;
if(size <= cutoff) {
std::sort(begin, end, comp);
return;
}
//// Insertion sort is faster for small arrays.
//if (size < insertion_sort_threshold) {
// if (leftmost) {
// insertion_sort(begin, end, comp);
// }
// else {
// unguarded_insertion_sort(begin, end, comp);
// }
// return;
//}
// Choose pivot as median of 3 or pseudomedian of 9.
//diff_t s2 = size / 2;
size_t s2 = size >> 1;
if (size > ninther_threshold) {
sort3(begin, begin + s2, end - 1, comp);
sort3(begin + 1, begin + (s2 - 1), end - 2, comp);
sort3(begin + 2, begin + (s2 + 1), end - 3, comp);
sort3(begin + (s2 - 1), begin + s2, begin + (s2 + 1), comp);
std::iter_swap(begin, begin + s2);
}
else {
sort3(begin + s2, begin, end - 1, comp);
}
// If *(begin - 1) is the end of the right partition
// of a previous partition operation, there is no element in [begin, end)
// that is smaller than *(begin - 1).
// Then if our pivot compares equal to *(begin - 1) we change strategy,
// putting equal elements in the left partition,
// greater elements in the right partition.
// We do not have to recurse on the left partition,
// since it's sorted (all equal).
if (!leftmost && !comp(*(begin - 1), *begin)) {
begin = partition_left(begin, end, comp) + 1;
continue;
}
// Partition and get results.
auto pair = partition_right(begin, end, comp);
auto pivot_pos = pair.first;
auto already_partitioned = pair.second;
// Check for a highly unbalanced partition.
//diff_t l_size = pivot_pos - begin;
//diff_t r_size = end - (pivot_pos + 1);
size_t l_size = pivot_pos - begin;
size_t r_size = end - (pivot_pos + 1);
bool highly_unbalanced = l_size < size / 8 || r_size < size / 8;
// If we got a highly unbalanced partition we shuffle elements
// to break many patterns.
if (highly_unbalanced) {
// If we had too many bad partitions, switch to heapsort
// to guarantee O(n log n).
if (--bad_allowed == 0) {
std::make_heap(begin, end, comp);
std::sort_heap(begin, end, comp);
return;
}
if (l_size >= insertion_sort_threshold) {
std::iter_swap(begin, begin + l_size / 4);
std::iter_swap(pivot_pos - 1, pivot_pos - l_size / 4);
if (l_size > ninther_threshold) {
std::iter_swap(begin + 1, begin + (l_size / 4 + 1));
std::iter_swap(begin + 2, begin + (l_size / 4 + 2));
std::iter_swap(pivot_pos - 2, pivot_pos - (l_size / 4 + 1));
std::iter_swap(pivot_pos - 3, pivot_pos - (l_size / 4 + 2));
}
}
if (r_size >= insertion_sort_threshold) {
std::iter_swap(pivot_pos + 1, pivot_pos + (1 + r_size / 4));
std::iter_swap(end - 1, end - r_size / 4);
if (r_size > ninther_threshold) {
std::iter_swap(pivot_pos + 2, pivot_pos + (2 + r_size / 4));
std::iter_swap(pivot_pos + 3, pivot_pos + (3 + r_size / 4));
std::iter_swap(end - 2, end - (1 + r_size / 4));
std::iter_swap(end - 3, end - (2 + r_size / 4));
}
}
}
// decently balanced
else {
// sequence try to use insertion sort.
if (already_partitioned &&
partial_insertion_sort(begin, pivot_pos, comp) &&
partial_insertion_sort(pivot_pos + 1, end, comp)
) {
return;
}
}
// Sort the left partition first using recursion and
// do tail recursion elimination for the right-hand partition.
sf.silent_async(
[&sf, begin, pivot_pos, comp, bad_allowed, leftmost] () mutable {
parallel_pdqsort(sf, begin, pivot_pos, comp, bad_allowed, leftmost);
}
);
begin = pivot_pos + 1;
leftmost = false;
}
}
// ----------------------------------------------------------------------------
// 3-way quick sort
// ----------------------------------------------------------------------------
// 3-way quick sort
template <typename RandItr, typename C>
void parallel_3wqsort(tf::Subflow& sf, RandItr first, RandItr last, C compare) {
using namespace std::string_literals;
constexpr auto cutoff = parallel_sort_cutoff<RandItr>();
sort_partition:
if(static_cast<size_t>(last - first) < cutoff) {
std::sort(first, last+1, compare);
return;
}
auto m = pseudo_median_of_nine(first, last, compare);
if(m != first) {
std::iter_swap(first, m);
}
auto l = first;
auto r = last;
auto f = std::next(first, 1);
bool is_swapped_l = false;
bool is_swapped_r = false;
while(f <= r) {
if(compare(*f, *l)) {
is_swapped_l = true;
std::iter_swap(l, f);
l++;
f++;
}
else if(compare(*l, *f)) {
is_swapped_r = true;
std::iter_swap(r, f);
r--;
}
else {
f++;
}
}
if(l - first > 1 && is_swapped_l) {
//sf.emplace([&](tf::Subflow& sfl) mutable {
// parallel_3wqsort(sfl, first, l-1, compare);
//});
sf.silent_async([&sf, first, l, &compare] () mutable {
parallel_3wqsort(sf, first, l-1, compare);
});
}
if(last - r > 1 && is_swapped_r) {
//sf.emplace([&](tf::Subflow& sfr) mutable {
// parallel_3wqsort(sfr, r+1, last, compare);
//});
//sf.silent_async([&sf, r, last, &compare] () mutable {
// parallel_3wqsort(sf, r+1, last, compare);
//});
first = r+1;
goto sort_partition;
}
//sf.join();
}
// ----------------------------------------------------------------------------
// tf::Taskflow::sort
// ----------------------------------------------------------------------------
// Function: sort
template <typename B, typename E, typename C>
Task FlowBuilder::sort(B&& beg, E&& end, C cmp) {
using I = stateful_iterator_t<B, E>;
Task task = emplace(
[b=std::forward<B>(beg), e=std::forward<E>(end), cmp] (Subflow& sf) mutable {
// fetch the iterator values
I beg = b;
I end = e;
if(beg == end) {
return;
}
size_t W = sf._executor.num_workers();
size_t N = std::distance(beg, end);
// only myself - no need to spawn another graph
if(W <= 1 || N <= parallel_sort_cutoff<I>()) {
std::sort(beg, end, cmp);
return;
}
//parallel_3wqsort(sf, beg, end-1, c);
parallel_pdqsort(sf, beg, end, cmp, log2(end - beg));
sf.join();
});
return task;
}
// Function: sort
template <typename B, typename E>
Task FlowBuilder::sort(B&& beg, E&& end) {
using I = stateful_iterator_t<B, E>;
//using value_type = std::decay_t<decltype(*std::declval<I>())>;
using value_type = typename std::iterator_traits<I>::value_type;
return sort(
std::forward<B>(beg), std::forward<E>(end), std::less<value_type>{}
);
}
} // namespace tf ------------------------------------------------------------

View File

@ -0,0 +1,50 @@
#pragma once
namespace tf {
// taskflow
class AsyncTopology;
class Node;
class Graph;
class FlowBuilder;
class Semaphore;
class Subflow;
class Task;
class TaskView;
class Taskflow;
class Topology;
class TopologyBase;
class Executor;
class WorkerView;
class ObserverInterface;
class ChromeTracingObserver;
class TFProfObserver;
class TFProfManager;
template <typename T>
class Future;
// cudaFlow
class cudaNode;
class cudaGraph;
class cudaTask;
class cudaFlow;
class cudaFlowCapturer;
class cudaFlowCapturerBase;
class cudaCapturingBase;
class cudaLinearCapturing;
class cudaSequentialCapturing;
class cudaRoundRobinCapturing;
// syclFlow
class syclNode;
class syclGraph;
class syclTask;
class syclFlow;
} // end of namespace tf -----------------------------------------------------

View File

@ -0,0 +1,8 @@
#pragma once
#define TF_ENABLE_PROFILER "TF_ENABLE_PROFILER"
namespace tf {
} // end of namespace tf -----------------------------------------------------

View File

@ -0,0 +1,26 @@
#pragma once
#include <iostream>
#include <sstream>
#include <exception>
#include "../utility/stream.hpp"
namespace tf {
// Procedure: throw_se
// Throws the system error under a given error code.
template <typename... ArgsT>
//void throw_se(const char* fname, const size_t line, Error::Code c, ArgsT&&... args) {
void throw_re(const char* fname, const size_t line, ArgsT&&... args) {
std::ostringstream oss;
oss << "[" << fname << ":" << line << "] ";
//ostreamize(oss, std::forward<ArgsT>(args)...);
(oss << ... << args);
throw std::runtime_error(oss.str());
}
} // ------------------------------------------------------------------------
#define TF_THROW(...) tf::throw_re(__FILE__, __LINE__, __VA_ARGS__);

File diff suppressed because it is too large Load Diff

View File

@ -0,0 +1,754 @@
#pragma once
#include "task.hpp"
/**
@file flow_builder.hpp
@brief flow builder include file
*/
namespace tf {
/**
@class FlowBuilder
@brief building methods of a task dependency graph
*/
class FlowBuilder {
friend class Executor;
public:
/**
@brief creates a static task
@tparam C callable type constructible from std::function<void()>
@param callable callable to construct a static task
@return a tf::Task handle
The following example creates a static task.
@code{.cpp}
tf::Task static_task = taskflow.emplace([](){});
@endcode
Please refer to @ref StaticTasking for details.
*/
template <typename C,
std::enable_if_t<is_static_task_v<C>, void>* = nullptr
>
Task emplace(C&& callable);
/**
@brief creates a dynamic task
@tparam C callable type constructible from std::function<void(tf::Subflow&)>
@param callable callable to construct a dynamic task
@return a tf::Task handle
The following example creates a dynamic task (tf::Subflow)
that spawns two static tasks.
@code{.cpp}
tf::Task dynamic_task = taskflow.emplace([](tf::Subflow& sf){
tf::Task static_task1 = sf.emplace([](){});
tf::Task static_task2 = sf.emplace([](){});
});
@endcode
Please refer to @ref DynamicTasking for details.
*/
template <typename C,
std::enable_if_t<is_dynamic_task_v<C>, void>* = nullptr
>
Task emplace(C&& callable);
/**
@brief creates a condition task
@tparam C callable type constructible from std::function<int()>
@param callable callable to construct a condition task
@return a tf::Task handle
The following example creates an if-else block using one condition task
and three static tasks.
@code{.cpp}
tf::Taskflow taskflow;
auto [init, cond, yes, no] = taskflow.emplace(
[] () { },
[] () { return 0; },
[] () { std::cout << "yes\n"; },
[] () { std::cout << "no\n"; }
);
// executes yes if cond returns 0, or no if cond returns 1
cond.precede(yes, no);
cond.succeed(init);
@endcode
Please refer to @ref ConditionalTasking for details.
*/
template <typename C,
std::enable_if_t<is_condition_task_v<C>, void>* = nullptr
>
Task emplace(C&& callable);
/**
@brief creates multiple tasks from a list of callable objects
@tparam C callable types
@param callables one or multiple callable objects constructible from each task category
@return a tf::Task handle
The method returns a tuple of tasks each corresponding to the given
callable target. You can use structured binding to get the return tasks
one by one.
The following example creates four static tasks and assign them to
@c A, @c B, @c C, and @c D using structured binding.
@code{.cpp}
auto [A, B, C, D] = taskflow.emplace(
[] () { std::cout << "A"; },
[] () { std::cout << "B"; },
[] () { std::cout << "C"; },
[] () { std::cout << "D"; }
);
@endcode
*/
template <typename... C, std::enable_if_t<(sizeof...(C)>1), void>* = nullptr>
auto emplace(C&&... callables);
/**
@brief removes a task from a taskflow
@param task task to remove
Removes a task and its input and output dependencies from this graph.
If the task does not belong to this graph, nothing will happen.
*/
void erase(Task task);
/**
@brief creates a module task from a taskflow
@param taskflow a taskflow object for the module
@return a tf::Task handle
Please refer to @ref ComposableTasking for details.
*/
Task composed_of(Taskflow& taskflow);
/**
@brief creates a placeholder task
@return a tf::Task handle
A placeholder task maps to a node in the taskflow graph, but
it does not have any callable work assigned yet.
A placeholder task is different from an empty task handle that
does not point to any node in a graph.
@code{.cpp}
// create a placeholder task with no callable target assigned
tf::Task placeholder = taskflow.placeholder();
assert(placeholder.empty() == false && placeholder.has_work() == false);
// create an empty task handle
tf::Task task;
assert(task.empty() == true);
// assign the task handle to the placeholder task
task = placeholder;
assert(task.empty() == false && task.has_work() == false);
@endcode
*/
Task placeholder();
/**
@brief creates a %cudaFlow task on the caller's GPU device context
@tparam C callable type constructible from @c std::function<void(tf::cudaFlow&)>
@return a tf::Task handle
This method is equivalent to calling tf::FlowBuilder::emplace_on(callable, d)
where @c d is the caller's device context.
The following example creates a %cudaFlow of two kernel tasks, @c task1 and
@c task2, where @c task1 runs before @c task2.
@code{.cpp}
taskflow.emplace([&](tf::cudaFlow& cf){
// create two kernel tasks
tf::cudaTask task1 = cf.kernel(grid1, block1, shm1, kernel1, args1);
tf::cudaTask task2 = cf.kernel(grid2, block2, shm2, kernel2, args2);
// kernel1 runs before kernel2
task1.precede(task2);
});
@endcode
Please refer to @ref GPUTaskingcudaFlow and @ref GPUTaskingcudaFlowCapturer
for details.
*/
template <typename C,
std::enable_if_t<is_cudaflow_task_v<C>, void>* = nullptr
>
Task emplace(C&& callable);
/**
@brief creates a %cudaFlow task on the given device
@tparam C callable type constructible from std::function<void(tf::cudaFlow&)>
@tparam D device type, either @c int or @c std::ref<int> (stateful)
@return a tf::Task handle
The following example creates a %cudaFlow of two kernel tasks, @c task1 and
@c task2 on GPU @c 2, where @c task1 runs before @c task2
@code{.cpp}
taskflow.emplace_on([&](tf::cudaFlow& cf){
// create two kernel tasks
tf::cudaTask task1 = cf.kernel(grid1, block1, shm1, kernel1, args1);
tf::cudaTask task2 = cf.kernel(grid2, block2, shm2, kernel2, args2);
// kernel1 runs before kernel2
task1.precede(task2);
}, 2);
@endcode
*/
template <typename C, typename D,
std::enable_if_t<is_cudaflow_task_v<C>, void>* = nullptr
>
Task emplace_on(C&& callable, D&& device);
/**
@brief creates a %syclFlow task on the default queue
@tparam C callable type constructible from std::function<void(tf::syclFlow&)>
@param callable a callable that takes a referenced tf::syclFlow object
@return a tf::Task handle
The following example creates a %syclFlow on the default queue to submit
two kernel tasks, @c task1 and @c task2, where @c task1 runs before @c task2.
@code{.cpp}
taskflow.emplace([&](tf::syclFlow& cf){
// create two single-thread kernel tasks
tf::syclTask task1 = cf.single_task([](){});
tf::syclTask task2 = cf.single_task([](){});
// kernel1 runs before kernel2
task1.precede(task2);
});
@endcode
*/
template <typename C, std::enable_if_t<is_syclflow_task_v<C>, void>* = nullptr>
Task emplace(C&& callable);
/**
@brief creates a %syclFlow task on the given queue
@tparam C callable type constructible from std::function<void(tf::syclFlow&)>
@tparam Q queue type
@param callable a callable that takes a referenced tf::syclFlow object
@param queue a queue of type sycl::queue
@return a tf::Task handle
The following example creates a %syclFlow on the given queue to submit
two kernel tasks, @c task1 and @c task2, where @c task1 runs before @c task2.
@code{.cpp}
taskflow.emplace_on([&](tf::syclFlow& cf){
// create two single-thread kernel tasks
tf::syclTask task1 = cf.single_task([](){});
tf::syclTask task2 = cf.single_task([](){});
// kernel1 runs before kernel2
task1.precede(task2);
}, queue);
@endcode
*/
template <typename C, typename Q,
std::enable_if_t<is_syclflow_task_v<C>, void>* = nullptr
>
Task emplace_on(C&& callable, Q&& queue);
/**
@brief adds adjacent dependency links to a linear list of tasks
@param tasks a vector of tasks
*/
void linearize(std::vector<Task>& tasks);
/**
@brief adds adjacent dependency links to a linear list of tasks
@param tasks an initializer list of tasks
*/
void linearize(std::initializer_list<Task> tasks);
// ------------------------------------------------------------------------
// parallel iterations
// ------------------------------------------------------------------------
/**
@brief constructs a STL-styled parallel-for task
@tparam B beginning iterator type
@tparam E ending iterator type
@tparam C callable type
@param first iterator to the beginning (inclusive)
@param last iterator to the end (exclusive)
@param callable a callable object to apply to the dereferenced iterator
@return a tf::Task handle
The task spawns a subflow that applies the callable object to each object obtained by dereferencing every iterator in the range <tt>[first, last)</tt>.
This method is equivalent to the parallel execution of the following loop:
@code{.cpp}
for(auto itr=first; itr!=last; itr++) {
callable(*itr);
}
@endcode
Arguments templated to enable stateful passing using std::reference_wrapper.
The callable needs to take a single argument of
the dereferenced iterator type.
Please refer to @ref ParallelIterations for details.
*/
template <typename B, typename E, typename C>
Task for_each(B&& first, E&& last, C callable);
/**
@brief constructs an index-based parallel-for task
@tparam B beginning index type (must be integral)
@tparam E ending index type (must be integral)
@tparam S step type (must be integral)
@tparam C callable type
@param first index of the beginning (inclusive)
@param last index of the end (exclusive)
@param step step size
@param callable a callable object to apply to each valid index
@return a tf::Task handle
The task spawns a subflow that applies the callable object to each index in the range <tt>[first, last)</tt> with the step size.
This method is equivalent to the parallel execution of the following loop:
@code{.cpp}
// case 1: step size is positive
for(auto i=first; i<last; i+=step) {
callable(i);
}
// case 2: step size is negative
for(auto i=first, i>last; i+=step) {
callable(i);
}
@endcode
Arguments are templated to enable stateful passing using std::reference_wrapper.
The callable needs to take a single argument of the integral index type.
Please refer to @ref ParallelIterations for details.
*/
template <typename B, typename E, typename S, typename C>
Task for_each_index(B&& first, E&& last, S&& step, C callable);
// ------------------------------------------------------------------------
// reduction
// ------------------------------------------------------------------------
/**
@brief constructs a STL-styled parallel-reduce task
@tparam B beginning iterator type
@tparam E ending iterator type
@tparam T result type
@tparam O binary reducer type
@param first iterator to the beginning (inclusive)
@param last iterator to the end (exclusive)
@param init initial value of the reduction and the storage for the reduced result
@param bop binary operator that will be applied
@return a tf::Task handle
The task spawns a subflow to perform parallel reduction over @c init and the elements in the range <tt>[first, last)</tt>. The reduced result is store in @c init.
This method is equivalent to the parallel execution of the following loop:
@code{.cpp}
for(auto itr=first; itr!=last; itr++) {
init = bop(init, *itr);
}
@endcode
Arguments are templated to enable stateful passing using std::reference_wrapper.
Please refer to @ref ParallelReduction for details.
*/
template <typename B, typename E, typename T, typename O>
Task reduce(B&& first, E&& last, T& init, O bop);
// ------------------------------------------------------------------------
// transfrom and reduction
// ------------------------------------------------------------------------
/**
@brief constructs a STL-styled parallel transform-reduce task
@tparam B beginning iterator type
@tparam E ending iterator type
@tparam T result type
@tparam BOP binary reducer type
@tparam UOP unary transformion type
@param first iterator to the beginning (inclusive)
@param last iterator to the end (exclusive)
@param init initial value of the reduction and the storage for the reduced result
@param bop binary operator that will be applied in unspecified order to the results of @c uop
@param uop unary operator that will be applied to transform each element in the range to the result type
@return a tf::Task handle
The task spawns a subflow to perform parallel reduction over @c init and the transformed elements in the range <tt>[first, last)</tt>.
The reduced result is store in @c init.
This method is equivalent to the parallel execution of the following loop:
@code{.cpp}
for(auto itr=first; itr!=last; itr++) {
init = bop(init, uop(*itr));
}
@endcode
Arguments are templated to enable stateful passing using std::reference_wrapper.
Please refer to @ref ParallelReduction for details.
*/
template <typename B, typename E, typename T, typename BOP, typename UOP>
Task transform_reduce(B&& first, E&& last, T& init, BOP bop, UOP uop);
// ------------------------------------------------------------------------
// sort
// ------------------------------------------------------------------------
/**
@brief constructs a dynamic task to perform STL-styled parallel sort
@tparam B beginning iterator type (random-accessible)
@tparam E ending iterator type (random-accessible)
@tparam C comparator type
@param first iterator to the beginning (inclusive)
@param last iterator to the end (exclusive)
@param cmp comparison function object
The task spawns a subflow to parallelly sort elements in the range
<tt>[first, last)</tt>.
Arguments are templated to enable stateful passing using std::reference_wrapper.
Please refer to @ref ParallelSort for details.
*/
template <typename B, typename E, typename C>
Task sort(B&& first, E&& last, C cmp);
/**
@brief constructs a dynamic task to perform STL-styled parallel sort using
the @c std::less<T> comparator, where @c T is the element type
@tparam B beginning iterator type (random-accessible)
@tparam E ending iterator type (random-accessible)
@param first iterator to the beginning (inclusive)
@param last iterator to the end (exclusive)
The task spawns a subflow to parallelly sort elements in the range
<tt>[first, last)</tt> using the @c std::less<T> comparator,
where @c T is the dereferenced iterator type.
Arguments are templated to enable stateful passing using std::reference_wrapper.
Please refer to @ref ParallelSort for details.
*/
template <typename B, typename E>
Task sort(B&& first, E&& last);
protected:
/**
@brief constructs a flow builder with a graph
*/
FlowBuilder(Graph& graph);
/**
@brief associated graph object
*/
Graph& _graph;
private:
template <typename L>
void _linearize(L&);
};
// Constructor
inline FlowBuilder::FlowBuilder(Graph& graph) :
_graph {graph} {
}
// Function: emplace
template <typename C, std::enable_if_t<is_static_task_v<C>, void>*>
Task FlowBuilder::emplace(C&& c) {
return Task(_graph.emplace_back(
std::in_place_type_t<Node::Static>{}, std::forward<C>(c)
));
}
// Function: emplace
template <typename C, std::enable_if_t<is_dynamic_task_v<C>, void>*>
Task FlowBuilder::emplace(C&& c) {
return Task(_graph.emplace_back(
std::in_place_type_t<Node::Dynamic>{}, std::forward<C>(c)
));
}
// Function: emplace
template <typename C, std::enable_if_t<is_condition_task_v<C>, void>*>
Task FlowBuilder::emplace(C&& c) {
return Task(_graph.emplace_back(
std::in_place_type_t<Node::Condition>{}, std::forward<C>(c)
));
}
// Function: emplace
template <typename... C, std::enable_if_t<(sizeof...(C)>1), void>*>
auto FlowBuilder::emplace(C&&... cs) {
return std::make_tuple(emplace(std::forward<C>(cs))...);
}
// Function: erase
inline void FlowBuilder::erase(Task task) {
if (!task._node) {
return;
}
task.for_each_dependent([&] (Task dependent) {
auto& S = dependent._node->_successors;
if(auto I = std::find(S.begin(), S.end(), task._node); I != S.end()) {
S.erase(I);
}
});
task.for_each_successor([&] (Task dependent) {
auto& D = dependent._node->_dependents;
if(auto I = std::find(D.begin(), D.end(), task._node); I != D.end()) {
D.erase(I);
}
});
_graph.erase(task._node);
}
// Function: composed_of
inline Task FlowBuilder::composed_of(Taskflow& taskflow) {
auto node = _graph.emplace_back(
std::in_place_type_t<Node::Module>{}, &taskflow
);
return Task(node);
}
// Function: placeholder
inline Task FlowBuilder::placeholder() {
auto node = _graph.emplace_back();
return Task(node);
}
// Procedure: _linearize
template <typename L>
void FlowBuilder::_linearize(L& keys) {
auto itr = keys.begin();
auto end = keys.end();
if(itr == end) {
return;
}
auto nxt = itr;
for(++nxt; nxt != end; ++nxt, ++itr) {
itr->_node->_precede(nxt->_node);
}
}
// Procedure: linearize
inline void FlowBuilder::linearize(std::vector<Task>& keys) {
_linearize(keys);
}
// Procedure: linearize
inline void FlowBuilder::linearize(std::initializer_list<Task> keys) {
_linearize(keys);
}
// ----------------------------------------------------------------------------
/**
@class Subflow
@brief class to construct a subflow graph from the execution of a dynamic task
By default, a subflow automatically @em joins its parent node.
You may explicitly join or detach a subflow by calling tf::Subflow::join
or tf::Subflow::detach, respectively.
The following example creates a taskflow graph that spawns a subflow from
the execution of task @c B, and the subflow contains three tasks, @c B1,
@c B2, and @c B3, where @c B3 runs after @c B1 and @c B2.
@code{.cpp}
// create three regular tasks
tf::Task A = taskflow.emplace([](){}).name("A");
tf::Task C = taskflow.emplace([](){}).name("C");
tf::Task D = taskflow.emplace([](){}).name("D");
// create a subflow graph (dynamic tasking)
tf::Task B = taskflow.emplace([] (tf::Subflow& subflow) {
tf::Task B1 = subflow.emplace([](){}).name("B1");
tf::Task B2 = subflow.emplace([](){}).name("B2");
tf::Task B3 = subflow.emplace([](){}).name("B3");
B1.precede(B3);
B2.precede(B3);
}).name("B");
A.precede(B); // B runs after A
A.precede(C); // C runs after A
B.precede(D); // D runs after B
C.precede(D); // D runs after C
@endcode
*/
class Subflow : public FlowBuilder {
friend class Executor;
friend class FlowBuilder;
public:
/**
@brief enables the subflow to join its parent task
Performs an immediate action to join the subflow. Once the subflow is joined,
it is considered finished and you may not modify the subflow anymore.
*/
void join();
/**
@brief enables the subflow to detach from its parent task
Performs an immediate action to detach the subflow. Once the subflow is detached,
it is considered finished and you may not modify the subflow anymore.
*/
void detach();
/**
@brief queries if the subflow is joinable
When a subflow is joined or detached, it becomes not joinable.
*/
bool joinable() const;
/**
@brief runs a given function asynchronously
@tparam F callable type
@tparam ArgsT parameter types
@param f callable object to call
@param args parameters to pass to the callable
@return a tf::Future that will holds the result of the execution
This method is thread-safe and can be called by multiple tasks in the
subflow at the same time.
The difference to tf::Executor::async is that the created asynchronous task
pertains to the subflow.
When the subflow joins, all asynchronous tasks created from the subflow
are guaranteed to finish before the join.
For example:
@code{.cpp}
std::atomic<int> counter(0);
taskflow.empalce([&](tf::Subflow& sf){
for(int i=0; i<100; i++) {
sf.async([&](){ counter++; });
}
sf.join();
assert(counter == 100);
});
@endcode
You cannot create asynchronous tasks from a detached subflow.
Doing this results in undefined behavior.
*/
template <typename F, typename... ArgsT>
auto async(F&& f, ArgsT&&... args);
/**
@brief similar to tf::Subflow::async but did not return a future object
*/
template <typename F, typename... ArgsT>
void silent_async(F&& f, ArgsT&&... args);
private:
Subflow(Executor&, Node*, Graph&);
Executor& _executor;
Node* _parent;
bool _joinable {true};
};
// Constructor
inline Subflow::Subflow(Executor& executor, Node* parent, Graph& graph) :
FlowBuilder {graph},
_executor {executor},
_parent {parent} {
}
// Function: joined
inline bool Subflow::joinable() const {
return _joinable;
}
} // end of namespace tf. ---------------------------------------------------

View File

@ -0,0 +1,572 @@
#pragma once
#include "../utility/iterator.hpp"
#include "../utility/object_pool.hpp"
#include "../utility/traits.hpp"
#include "../utility/singleton.hpp"
#include "../utility/os.hpp"
#include "../utility/math.hpp"
#include "../utility/small_vector.hpp"
#include "../utility/serializer.hpp"
#include "error.hpp"
#include "declarations.hpp"
#include "semaphore.hpp"
#include "environment.hpp"
#include "topology.hpp"
namespace tf {
// ----------------------------------------------------------------------------
// Class: CustomGraphBase
// ----------------------------------------------------------------------------
class CustomGraphBase {
public:
virtual void dump(std::ostream&, const void*, const std::string&) const = 0;
virtual ~CustomGraphBase() = default;
};
// ----------------------------------------------------------------------------
// Class: Graph
// ----------------------------------------------------------------------------
class Graph {
friend class Node;
friend class Taskflow;
friend class Executor;
friend class Sanitizer;
public:
Graph() = default;
Graph(const Graph&) = delete;
Graph(Graph&&);
~Graph();
Graph& operator = (const Graph&) = delete;
Graph& operator = (Graph&&);
void clear();
void clear_detached();
void merge(Graph&&);
bool empty() const;
size_t size() const;
template <typename ...Args>
Node* emplace_back(Args&& ...);
Node* emplace_back();
void erase(Node*);
private:
std::vector<Node*> _nodes;
};
// ----------------------------------------------------------------------------
// Class: Node
class Node {
friend class Graph;
friend class Task;
friend class TaskView;
friend class Taskflow;
friend class Executor;
friend class FlowBuilder;
friend class Subflow;
friend class Sanitizer;
TF_ENABLE_POOLABLE_ON_THIS;
// state bit flag
constexpr static int BRANCHED = 0x1;
constexpr static int DETACHED = 0x2;
constexpr static int ACQUIRED = 0x4;
constexpr static int READY = 0x8;
// static work handle
struct Static {
template <typename C>
Static(C&&);
std::function<void()> work;
};
// dynamic work handle
struct Dynamic {
template <typename C>
Dynamic(C&&);
std::function<void(Subflow&)> work;
Graph subgraph;
};
// condition work handle
struct Condition {
template <typename C>
Condition(C&&);
std::function<int()> work;
};
// module work handle
struct Module {
template <typename T>
Module(T&&);
Taskflow* module {nullptr};
};
// Async work
struct Async {
template <typename T>
Async(T&&, std::shared_ptr<AsyncTopology>);
std::function<void(bool)> work;
std::shared_ptr<AsyncTopology> topology;
};
// Silent async work
struct SilentAsync {
template <typename C>
SilentAsync(C&&);
std::function<void()> work;
};
// cudaFlow work handle
struct cudaFlow {
template <typename C, typename G>
cudaFlow(C&& c, G&& g);
std::function<void(Executor&, Node*)> work;
std::unique_ptr<CustomGraphBase> graph;
};
// syclFlow work handle
struct syclFlow {
template <typename C, typename G>
syclFlow(C&& c, G&& g);
std::function<void(Executor&, Node*)> work;
std::unique_ptr<CustomGraphBase> graph;
};
using handle_t = std::variant<
std::monostate, // placeholder
Static, // static tasking
Dynamic, // dynamic tasking
Condition, // conditional tasking
Module, // composable tasking
Async, // async tasking
SilentAsync, // async tasking (no future)
cudaFlow, // cudaFlow
syclFlow // syclFlow
>;
struct Semaphores {
std::vector<Semaphore*> to_acquire;
std::vector<Semaphore*> to_release;
};
public:
// variant index
constexpr static auto PLACEHOLDER = get_index_v<std::monostate, handle_t>;
constexpr static auto STATIC = get_index_v<Static, handle_t>;
constexpr static auto DYNAMIC = get_index_v<Dynamic, handle_t>;
constexpr static auto CONDITION = get_index_v<Condition, handle_t>;
constexpr static auto MODULE = get_index_v<Module, handle_t>;
constexpr static auto ASYNC = get_index_v<Async, handle_t>;
constexpr static auto SILENT_ASYNC = get_index_v<SilentAsync, handle_t>;
constexpr static auto CUDAFLOW = get_index_v<cudaFlow, handle_t>;
constexpr static auto SYCLFLOW = get_index_v<syclFlow, handle_t>;
template <typename... Args>
Node(Args&&... args);
~Node();
size_t num_successors() const;
size_t num_dependents() const;
size_t num_strong_dependents() const;
size_t num_weak_dependents() const;
const std::string& name() const;
private:
std::string _name;
void* _data {nullptr};
handle_t _handle;
SmallVector<Node*> _successors;
SmallVector<Node*> _dependents;
std::unique_ptr<Semaphores> _semaphores;
Topology* _topology {nullptr};
Node* _parent {nullptr};
std::atomic<int> _state {0};
std::atomic<size_t> _join_counter {0};
void _precede(Node*);
void _set_up_join_counter();
bool _has_state(int) const;
bool _is_cancelled() const;
bool _acquire_all(std::vector<Node*>&);
std::vector<Node*> _release_all();
};
// ----------------------------------------------------------------------------
// Node Object Pool
// ----------------------------------------------------------------------------
inline ObjectPool<Node> node_pool;
// ----------------------------------------------------------------------------
// Definition for Node::Static
// ----------------------------------------------------------------------------
// Constructor
template <typename C>
Node::Static::Static(C&& c) : work {std::forward<C>(c)} {
}
// ----------------------------------------------------------------------------
// Definition for Node::Dynamic
// ----------------------------------------------------------------------------
// Constructor
template <typename C>
Node::Dynamic::Dynamic(C&& c) : work {std::forward<C>(c)} {
}
// ----------------------------------------------------------------------------
// Definition for Node::Condition
// ----------------------------------------------------------------------------
// Constructor
template <typename C>
Node::Condition::Condition(C&& c) : work {std::forward<C>(c)} {
}
// ----------------------------------------------------------------------------
// Definition for Node::cudaFlow
// ----------------------------------------------------------------------------
template <typename C, typename G>
Node::cudaFlow::cudaFlow(C&& c, G&& g) :
work {std::forward<C>(c)},
graph {std::forward<G>(g)} {
}
// ----------------------------------------------------------------------------
// Definition for Node::syclFlow
// ----------------------------------------------------------------------------
template <typename C, typename G>
Node::syclFlow::syclFlow(C&& c, G&& g) :
work {std::forward<C>(c)},
graph {std::forward<G>(g)} {
}
// ----------------------------------------------------------------------------
// Definition for Node::Module
// ----------------------------------------------------------------------------
// Constructor
template <typename T>
Node::Module::Module(T&& tf) : module {tf} {
}
// ----------------------------------------------------------------------------
// Definition for Node::Async
// ----------------------------------------------------------------------------
// Constructor
template <typename C>
Node::Async::Async(C&& c, std::shared_ptr<AsyncTopology>tpg) :
work {std::forward<C>(c)},
topology {std::move(tpg)} {
}
// ----------------------------------------------------------------------------
// Definition for Node::SilentAsync
// ----------------------------------------------------------------------------
// Constructor
template <typename C>
Node::SilentAsync::SilentAsync(C&& c) :
work {std::forward<C>(c)} {
}
// ----------------------------------------------------------------------------
// Definition for Node
// ----------------------------------------------------------------------------
// Constructor
template <typename... Args>
Node::Node(Args&&... args): _handle{std::forward<Args>(args)...} {
}
// Destructor
inline Node::~Node() {
// this is to avoid stack overflow
if(_handle.index() == DYNAMIC) {
auto& subgraph = std::get<Dynamic>(_handle).subgraph;
std::vector<Node*> nodes;
nodes.reserve(subgraph.size());
std::move(
subgraph._nodes.begin(), subgraph._nodes.end(), std::back_inserter(nodes)
);
subgraph._nodes.clear();
size_t i = 0;
while(i < nodes.size()) {
if(nodes[i]->_handle.index() == DYNAMIC) {
auto& sbg = std::get<Dynamic>(nodes[i]->_handle).subgraph;
std::move(
sbg._nodes.begin(), sbg._nodes.end(), std::back_inserter(nodes)
);
sbg._nodes.clear();
}
++i;
}
//auto& np = Graph::_node_pool();
for(i=0; i<nodes.size(); ++i) {
node_pool.recycle(nodes[i]);
}
}
}
// Procedure: _precede
inline void Node::_precede(Node* v) {
_successors.push_back(v);
v->_dependents.push_back(this);
}
// Function: num_successors
inline size_t Node::num_successors() const {
return _successors.size();
}
// Function: dependents
inline size_t Node::num_dependents() const {
return _dependents.size();
}
// Function: num_weak_dependents
inline size_t Node::num_weak_dependents() const {
size_t n = 0;
for(size_t i=0; i<_dependents.size(); i++) {
if(_dependents[i]->_handle.index() == Node::CONDITION) {
n++;
}
}
return n;
}
// Function: num_strong_dependents
inline size_t Node::num_strong_dependents() const {
size_t n = 0;
for(size_t i=0; i<_dependents.size(); i++) {
if(_dependents[i]->_handle.index() != Node::CONDITION) {
n++;
}
}
return n;
}
// Function: name
inline const std::string& Node::name() const {
return _name;
}
// Function: _is_cancelled
inline bool Node::_is_cancelled() const {
if(_handle.index() == Node::ASYNC) {
auto& h = std::get<Node::Async>(_handle);
if(h.topology && h.topology->_is_cancelled) {
return true;
}
}
// async tasks spawned from subflow does not have topology
return _topology && _topology->_is_cancelled;
}
// Procedure: _set_up_join_counter
inline void Node::_set_up_join_counter() {
size_t c = 0;
for(auto p : _dependents) {
if(p->_handle.index() == Node::CONDITION) {
//_set_state(Node::BRANCHED);
_state.fetch_or(Node::BRANCHED, std::memory_order_relaxed);
}
else {
c++;
}
}
_join_counter.store(c, std::memory_order_release);
}
// Function: _acquire_all
inline bool Node::_acquire_all(std::vector<Node*>& nodes) {
auto& to_acquire = _semaphores->to_acquire;
for(size_t i = 0; i < to_acquire.size(); ++i) {
if(!to_acquire[i]->_try_acquire_or_wait(this)) {
for(size_t j = 1; j <= i; ++j) {
auto r = to_acquire[i-j]->_release();
nodes.insert(end(nodes), begin(r), end(r));
}
return false;
}
}
return true;
}
// Function: _release_all
inline std::vector<Node*> Node::_release_all() {
auto& to_release = _semaphores->to_release;
std::vector<Node*> nodes;
for(const auto& sem : to_release) {
auto r = sem->_release();
nodes.insert(end(nodes), begin(r), end(r));
}
return nodes;
}
// ----------------------------------------------------------------------------
// Graph definition
// ----------------------------------------------------------------------------
//// Function: _node_pool
//inline ObjectPool<Node>& Graph::_node_pool() {
// static ObjectPool<Node> pool;
// return pool;
//}
// Destructor
inline Graph::~Graph() {
clear();
}
// Move constructor
inline Graph::Graph(Graph&& other) :
_nodes {std::move(other._nodes)} {
}
// Move assignment
inline Graph& Graph::operator = (Graph&& other) {
clear();
_nodes = std::move(other._nodes);
return *this;
}
// Procedure: clear
inline void Graph::clear() {
for(auto node : _nodes) {
node_pool.recycle(node);
}
_nodes.clear();
}
// Procedure: clear_detached
inline void Graph::clear_detached() {
auto mid = std::partition(_nodes.begin(), _nodes.end(), [] (Node* node) {
return !(node->_state.load(std::memory_order_relaxed) & Node::DETACHED);
});
for(auto itr = mid; itr != _nodes.end(); ++itr) {
node_pool.recycle(*itr);
}
_nodes.resize(std::distance(_nodes.begin(), mid));
}
// Procedure: merge
inline void Graph::merge(Graph&& g) {
for(auto n : g._nodes) {
_nodes.push_back(n);
}
g._nodes.clear();
}
// Function: size
inline size_t Graph::size() const {
return _nodes.size();
}
// Function: empty
inline bool Graph::empty() const {
return _nodes.empty();
}
// Function: emplace_back
// create a node from a give argument; constructor is called if necessary
template <typename ...ArgsT>
Node* Graph::emplace_back(ArgsT&&... args) {
_nodes.push_back(node_pool.animate(std::forward<ArgsT>(args)...));
return _nodes.back();
}
// Function: emplace_back
// create a node from a give argument; constructor is called if necessary
inline Node* Graph::emplace_back() {
_nodes.push_back(node_pool.animate());
return _nodes.back();
}
// Function: erase
inline void Graph::erase(Node* node) {
if(auto I = std::find(_nodes.begin(), _nodes.end(), node); I != _nodes.end()) {
_nodes.erase(I);
node_pool.recycle(node);
}
}
} // end of namespace tf. ---------------------------------------------------

View File

@ -0,0 +1,267 @@
// 2019/02/09 - created by Tsung-Wei Huang
// - modified the event count from Eigen
#pragma once
#include <iostream>
#include <vector>
#include <cstdlib>
#include <cstdio>
#include <atomic>
#include <memory>
#include <deque>
#include <mutex>
#include <condition_variable>
#include <thread>
#include <algorithm>
#include <numeric>
#include <cassert>
// This file is part of Eigen, a lightweight C++ template library
// for linear algebra.
//
// Copyright (C) 2016 Dmitry Vyukov <dvyukov@google.com>
//
// This Source Code Form is subject to the terms of the Mozilla
// Public License v. 2.0. If a copy of the MPL was not distributed
// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
namespace tf {
// Notifier allows to wait for arbitrary predicates in non-blocking
// algorithms. Think of condition variable, but wait predicate does not need to
// be protected by a mutex. Usage:
// Waiting thread does:
//
// if (predicate)
// return act();
// Notifier::Waiter& w = waiters[my_index];
// ec.prepare_wait(&w);
// if (predicate) {
// ec.cancel_wait(&w);
// return act();
// }
// ec.commit_wait(&w);
//
// Notifying thread does:
//
// predicate = true;
// ec.notify(true);
//
// notify is cheap if there are no waiting threads. prepare_wait/commit_wait are not
// cheap, but they are executed only if the preceeding predicate check has
// failed.
//
// Algorihtm outline:
// There are two main variables: predicate (managed by user) and _state.
// Operation closely resembles Dekker mutual algorithm:
// https://en.wikipedia.org/wiki/Dekker%27s_algorithm
// Waiting thread sets _state then checks predicate, Notifying thread sets
// predicate then checks _state. Due to seq_cst fences in between these
// operations it is guaranteed than either waiter will see predicate change
// and won't block, or notifying thread will see _state change and will unblock
// the waiter, or both. But it can't happen that both threads don't see each
// other changes, which would lead to deadlock.
class Notifier {
friend class Executor;
public:
struct Waiter {
std::atomic<Waiter*> next;
std::mutex mu;
std::condition_variable cv;
uint64_t epoch;
unsigned state;
enum {
kNotSignaled,
kWaiting,
kSignaled,
};
};
explicit Notifier(size_t N) : _waiters{N} {
assert(_waiters.size() < (1 << kWaiterBits) - 1);
// Initialize epoch to something close to overflow to test overflow.
_state = kStackMask | (kEpochMask - kEpochInc * _waiters.size() * 2);
}
~Notifier() {
// Ensure there are no waiters.
assert((_state.load() & (kStackMask | kWaiterMask)) == kStackMask);
}
// prepare_wait prepares for waiting.
// After calling this function the thread must re-check the wait predicate
// and call either cancel_wait or commit_wait passing the same Waiter object.
void prepare_wait(Waiter* w) {
w->epoch = _state.fetch_add(kWaiterInc, std::memory_order_relaxed);
std::atomic_thread_fence(std::memory_order_seq_cst);
}
// commit_wait commits waiting.
void commit_wait(Waiter* w) {
w->state = Waiter::kNotSignaled;
// Modification epoch of this waiter.
uint64_t epoch =
(w->epoch & kEpochMask) +
(((w->epoch & kWaiterMask) >> kWaiterShift) << kEpochShift);
uint64_t state = _state.load(std::memory_order_seq_cst);
for (;;) {
if (int64_t((state & kEpochMask) - epoch) < 0) {
// The preceeding waiter has not decided on its fate. Wait until it
// calls either cancel_wait or commit_wait, or is notified.
std::this_thread::yield();
state = _state.load(std::memory_order_seq_cst);
continue;
}
// We've already been notified.
if (int64_t((state & kEpochMask) - epoch) > 0) return;
// Remove this thread from prewait counter and add it to the waiter list.
assert((state & kWaiterMask) != 0);
uint64_t newstate = state - kWaiterInc + kEpochInc;
//newstate = (newstate & ~kStackMask) | (w - &_waiters[0]);
newstate = static_cast<uint64_t>((newstate & ~kStackMask) | static_cast<uint64_t>(w - &_waiters[0]));
if ((state & kStackMask) == kStackMask)
w->next.store(nullptr, std::memory_order_relaxed);
else
w->next.store(&_waiters[state & kStackMask], std::memory_order_relaxed);
if (_state.compare_exchange_weak(state, newstate,
std::memory_order_release))
break;
}
_park(w);
}
// cancel_wait cancels effects of the previous prepare_wait call.
void cancel_wait(Waiter* w) {
uint64_t epoch =
(w->epoch & kEpochMask) +
(((w->epoch & kWaiterMask) >> kWaiterShift) << kEpochShift);
uint64_t state = _state.load(std::memory_order_relaxed);
for (;;) {
if (int64_t((state & kEpochMask) - epoch) < 0) {
// The preceeding waiter has not decided on its fate. Wait until it
// calls either cancel_wait or commit_wait, or is notified.
std::this_thread::yield();
state = _state.load(std::memory_order_relaxed);
continue;
}
// We've already been notified.
if (int64_t((state & kEpochMask) - epoch) > 0) return;
// Remove this thread from prewait counter.
assert((state & kWaiterMask) != 0);
if (_state.compare_exchange_weak(state, state - kWaiterInc + kEpochInc,
std::memory_order_relaxed))
return;
}
}
// notify wakes one or all waiting threads.
// Must be called after changing the associated wait predicate.
void notify(bool all) {
std::atomic_thread_fence(std::memory_order_seq_cst);
uint64_t state = _state.load(std::memory_order_acquire);
for (;;) {
// Easy case: no waiters.
if ((state & kStackMask) == kStackMask && (state & kWaiterMask) == 0)
return;
uint64_t waiters = (state & kWaiterMask) >> kWaiterShift;
uint64_t newstate;
if (all) {
// Reset prewait counter and empty wait list.
newstate = (state & kEpochMask) + (kEpochInc * waiters) + kStackMask;
} else if (waiters) {
// There is a thread in pre-wait state, unblock it.
newstate = state + kEpochInc - kWaiterInc;
} else {
// Pop a waiter from list and unpark it.
Waiter* w = &_waiters[state & kStackMask];
Waiter* wnext = w->next.load(std::memory_order_relaxed);
uint64_t next = kStackMask;
//if (wnext != nullptr) next = wnext - &_waiters[0];
if (wnext != nullptr) next = static_cast<uint64_t>(wnext - &_waiters[0]);
// Note: we don't add kEpochInc here. ABA problem on the lock-free stack
// can't happen because a waiter is re-pushed onto the stack only after
// it was in the pre-wait state which inevitably leads to epoch
// increment.
newstate = (state & kEpochMask) + next;
}
if (_state.compare_exchange_weak(state, newstate,
std::memory_order_acquire)) {
if (!all && waiters) return; // unblocked pre-wait thread
if ((state & kStackMask) == kStackMask) return;
Waiter* w = &_waiters[state & kStackMask];
if (!all) w->next.store(nullptr, std::memory_order_relaxed);
_unpark(w);
return;
}
}
}
// notify n workers
void notify_n(size_t n) {
if(n >= _waiters.size()) {
notify(true);
}
else {
for(size_t k=0; k<n; ++k) {
notify(false);
}
}
}
size_t size() const {
return _waiters.size();
}
private:
// State_ layout:
// - low kStackBits is a stack of waiters committed wait.
// - next kWaiterBits is count of waiters in prewait state.
// - next kEpochBits is modification counter.
static const uint64_t kStackBits = 16;
static const uint64_t kStackMask = (1ull << kStackBits) - 1;
static const uint64_t kWaiterBits = 16;
static const uint64_t kWaiterShift = 16;
static const uint64_t kWaiterMask = ((1ull << kWaiterBits) - 1)
<< kWaiterShift;
static const uint64_t kWaiterInc = 1ull << kWaiterBits;
static const uint64_t kEpochBits = 32;
static const uint64_t kEpochShift = 32;
static const uint64_t kEpochMask = ((1ull << kEpochBits) - 1) << kEpochShift;
static const uint64_t kEpochInc = 1ull << kEpochShift;
std::atomic<uint64_t> _state;
std::vector<Waiter> _waiters;
void _park(Waiter* w) {
std::unique_lock<std::mutex> lock(w->mu);
while (w->state != Waiter::kSignaled) {
w->state = Waiter::kWaiting;
w->cv.wait(lock);
}
}
void _unpark(Waiter* waiters) {
Waiter* next = nullptr;
for (Waiter* w = waiters; w; w = next) {
next = w->next.load(std::memory_order_relaxed);
unsigned state;
{
std::unique_lock<std::mutex> lock(w->mu);
state = w->state;
w->state = Waiter::kSignaled;
}
// Avoid notifying if it wasn't waiting.
if (state == Waiter::kWaiting) w->cv.notify_one();
}
}
};
} // namespace tf ------------------------------------------------------------

View File

@ -0,0 +1,735 @@
#pragma once
#include "task.hpp"
#include "worker.hpp"
/**
@file observer.hpp
@brief observer include file
*/
namespace tf {
// ----------------------------------------------------------------------------
// timeline data structure
// ----------------------------------------------------------------------------
/**
@brief default time point type of observers
*/
using observer_stamp_t = std::chrono::time_point<std::chrono::steady_clock>;
/**
@private
*/
struct Segment {
std::string name;
TaskType type;
observer_stamp_t beg;
observer_stamp_t end;
template <typename Archiver>
auto save(Archiver& ar) const {
return ar(name, type, beg, end);
}
template <typename Archiver>
auto load(Archiver& ar) {
return ar(name, type, beg, end);
}
Segment() = default;
Segment(
const std::string& n, TaskType t, observer_stamp_t b, observer_stamp_t e
) : name {n}, type {t}, beg {b}, end {e} {
}
auto span() const {
return end-beg;
}
};
/**
@private
*/
struct Timeline {
size_t uid;
observer_stamp_t origin;
std::vector<std::vector<std::vector<Segment>>> segments;
Timeline() = default;
Timeline(const Timeline& rhs) = delete;
Timeline(Timeline&& rhs) = default;
Timeline& operator = (const Timeline& rhs) = delete;
Timeline& operator = (Timeline&& rhs) = default;
template <typename Archiver>
auto save(Archiver& ar) const {
return ar(uid, origin, segments);
}
template <typename Archiver>
auto load(Archiver& ar) {
return ar(uid, origin, segments);
}
};
/**
@private
*/
struct ProfileData {
std::vector<Timeline> timelines;
ProfileData() = default;
ProfileData(const ProfileData& rhs) = delete;
ProfileData(ProfileData&& rhs) = default;
ProfileData& operator = (const ProfileData& rhs) = delete;
ProfileData& operator = (ProfileData&&) = default;
template <typename Archiver>
auto save(Archiver& ar) const {
return ar(timelines);
}
template <typename Archiver>
auto load(Archiver& ar) {
return ar(timelines);
}
};
// ----------------------------------------------------------------------------
// observer interface
// ----------------------------------------------------------------------------
/**
@class: ObserverInterface
@brief The interface class for creating an executor observer.
The tf::ObserverInterface class let users define custom methods to monitor
the behaviors of an executor. This is particularly useful when you want to
inspect the performance of an executor and visualize when each thread
participates in the execution of a task.
To prevent users from direct access to the internal threads and tasks,
tf::ObserverInterface provides immutable wrappers,
tf::WorkerView and tf::TaskView, over workers and tasks.
Please refer to tf::WorkerView and tf::TaskView for details.
Example usage:
@code{.cpp}
struct MyObserver : public tf::ObserverInterface {
MyObserver(const std::string& name) {
std::cout << "constructing observer " << name << '\n';
}
void set_up(size_t num_workers) override final {
std::cout << "setting up observer with " << num_workers << " workers\n";
}
void on_entry(WorkerView w, tf::TaskView tv) override final {
std::ostringstream oss;
oss << "worker " << w.id() << " ready to run " << tv.name() << '\n';
std::cout << oss.str();
}
void on_exit(WorkerView w, tf::TaskView tv) override final {
std::ostringstream oss;
oss << "worker " << w.id() << " finished running " << tv.name() << '\n';
std::cout << oss.str();
}
};
tf::Taskflow taskflow;
tf::Executor executor;
// insert tasks into taskflow
// ...
// create a custom observer
std::shared_ptr<MyObserver> observer = executor.make_observer<MyObserver>("MyObserver");
// run the taskflow
executor.run(taskflow).wait();
@endcode
*/
class ObserverInterface {
friend class Executor;
public:
/**
@brief virtual destructor
*/
virtual ~ObserverInterface() = default;
/**
@brief constructor-like method to call when the executor observer is fully created
@param num_workers the number of the worker threads in the executor
*/
virtual void set_up(size_t num_workers) = 0;
/**
@brief method to call before a worker thread executes a closure
@param w an immutable view of this worker thread
@param task_view a constant wrapper object to the task
*/
virtual void on_entry(WorkerView w, TaskView task_view) = 0;
/**
@brief method to call after a worker thread executed a closure
@param w an immutable view of this worker thread
@param task_view a constant wrapper object to the task
*/
virtual void on_exit(WorkerView w, TaskView task_view) = 0;
};
// ----------------------------------------------------------------------------
// ChromeObserver definition
// ----------------------------------------------------------------------------
/**
@class: ChromeObserver
@brief observer interface based on Chrome tracing format
A tf::ChromeObserver inherits tf::ObserverInterface and defines methods to dump
the observed thread activities into a format that can be visualized through
@ChromeTracing.
@code{.cpp}
tf::Taskflow taskflow;
tf::Executor executor;
// insert tasks into taskflow
// ...
// create a custom observer
std::shared_ptr<tf::ChromeObserver> observer = executor.make_observer<tf::ChromeObserver>();
// run the taskflow
executor.run(taskflow).wait();
// dump the thread activities to a chrome-tracing format.
observer->dump(std::cout);
@endcode
*/
class ChromeObserver : public ObserverInterface {
friend class Executor;
// data structure to record each task execution
struct Segment {
std::string name;
observer_stamp_t beg;
observer_stamp_t end;
Segment(
const std::string& n,
observer_stamp_t b,
observer_stamp_t e
);
};
// data structure to store the entire execution timeline
struct Timeline {
observer_stamp_t origin;
std::vector<std::vector<Segment>> segments;
std::vector<std::stack<observer_stamp_t>> stacks;
};
public:
/**
@brief dumps the timelines into a @ChromeTracing format through
an output stream
*/
void dump(std::ostream& ostream) const;
/**
@brief dumps the timelines into a @ChromeTracing format
*/
inline std::string dump() const;
/**
@brief clears the timeline data
*/
inline void clear();
/**
@brief queries the number of tasks observed
*/
inline size_t num_tasks() const;
private:
inline void set_up(size_t num_workers) override final;
inline void on_entry(WorkerView w, TaskView task_view) override final;
inline void on_exit(WorkerView w, TaskView task_view) override final;
Timeline _timeline;
};
// constructor
inline ChromeObserver::Segment::Segment(
const std::string& n, observer_stamp_t b, observer_stamp_t e
) :
name {n}, beg {b}, end {e} {
}
// Procedure: set_up
inline void ChromeObserver::set_up(size_t num_workers) {
_timeline.segments.resize(num_workers);
_timeline.stacks.resize(num_workers);
for(size_t w=0; w<num_workers; ++w) {
_timeline.segments[w].reserve(32);
}
_timeline.origin = observer_stamp_t::clock::now();
}
// Procedure: on_entry
inline void ChromeObserver::on_entry(WorkerView wv, TaskView) {
_timeline.stacks[wv.id()].push(observer_stamp_t::clock::now());
}
// Procedure: on_exit
inline void ChromeObserver::on_exit(WorkerView wv, TaskView tv) {
size_t w = wv.id();
assert(!_timeline.stacks[w].empty());
auto beg = _timeline.stacks[w].top();
_timeline.stacks[w].pop();
_timeline.segments[w].emplace_back(
tv.name(), beg, observer_stamp_t::clock::now()
);
}
// Function: clear
inline void ChromeObserver::clear() {
for(size_t w=0; w<_timeline.segments.size(); ++w) {
_timeline.segments[w].clear();
while(!_timeline.stacks[w].empty()) {
_timeline.stacks[w].pop();
}
}
}
// Procedure: dump
inline void ChromeObserver::dump(std::ostream& os) const {
size_t first;
for(first = 0; first<_timeline.segments.size(); ++first) {
if(_timeline.segments[first].size() > 0) {
break;
}
}
os << '[';
for(size_t w=first; w<_timeline.segments.size(); w++) {
if(w != first && _timeline.segments[w].size() > 0) {
os << ',';
}
for(size_t i=0; i<_timeline.segments[w].size(); i++) {
os << '{'
<< "\"cat\":\"ChromeObserver\",";
// name field
os << "\"name\":\"";
if(_timeline.segments[w][i].name.empty()) {
os << w << '_' << i;
}
else {
os << _timeline.segments[w][i].name;
}
os << "\",";
// segment field
os << "\"ph\":\"X\","
<< "\"pid\":1,"
<< "\"tid\":" << w << ','
<< "\"ts\":" << std::chrono::duration_cast<std::chrono::microseconds>(
_timeline.segments[w][i].beg - _timeline.origin
).count() << ','
<< "\"dur\":" << std::chrono::duration_cast<std::chrono::microseconds>(
_timeline.segments[w][i].end - _timeline.segments[w][i].beg
).count();
if(i != _timeline.segments[w].size() - 1) {
os << "},";
}
else {
os << '}';
}
}
}
os << "]\n";
}
// Function: dump
inline std::string ChromeObserver::dump() const {
std::ostringstream oss;
dump(oss);
return oss.str();
}
// Function: num_tasks
inline size_t ChromeObserver::num_tasks() const {
return std::accumulate(
_timeline.segments.begin(), _timeline.segments.end(), size_t{0},
[](size_t sum, const auto& exe){
return sum + exe.size();
}
);
}
// ----------------------------------------------------------------------------
// TFProfObserver definition
// ----------------------------------------------------------------------------
/**
@class TFProfObserver
@brief observer interface based on the built-in taskflow profiler format
A tf::TFProfObserver inherits tf::ObserverInterface and defines methods to dump
the observed thread activities into a format that can be visualized through
@TFProf.
@code{.cpp}
tf::Taskflow taskflow;
tf::Executor executor;
// insert tasks into taskflow
// ...
// create a custom observer
std::shared_ptr<tf::TFProfObserver> observer = executor.make_observer<tf::TFProfObserver>();
// run the taskflow
executor.run(taskflow).wait();
// dump the thread activities to Taskflow Profiler format.
observer->dump(std::cout);
@endcode
We recommend using our @TFProf python script to observe thread activities
instead of the raw function call.
The script will turn on environment variables needed for observing all executors
in a taskflow program and dump the result to a valid, clean JSON file
compatible with the format of @TFProf.
*/
class TFProfObserver : public ObserverInterface {
friend class Executor;
friend class TFProfManager;
public:
/**
@brief dumps the timelines into a @TFProf format through
an output stream
*/
void dump(std::ostream& ostream) const;
/**
@brief dumps the timelines into a JSON string
*/
std::string dump() const;
/**
@brief clears the timeline data
*/
void clear();
/**
@brief queries the number of tasks observed
*/
size_t num_tasks() const;
private:
Timeline _timeline;
std::vector<std::stack<observer_stamp_t>> _stacks;
inline void set_up(size_t num_workers) override final;
inline void on_entry(WorkerView, TaskView) override final;
inline void on_exit(WorkerView, TaskView) override final;
};
// Procedure: set_up
inline void TFProfObserver::set_up(size_t num_workers) {
_timeline.uid = unique_id<size_t>();
_timeline.origin = observer_stamp_t::clock::now();
_timeline.segments.resize(num_workers);
_stacks.resize(num_workers);
}
// Procedure: on_entry
inline void TFProfObserver::on_entry(WorkerView wv, TaskView) {
_stacks[wv.id()].push(observer_stamp_t::clock::now());
}
// Procedure: on_exit
inline void TFProfObserver::on_exit(WorkerView wv, TaskView tv) {
size_t w = wv.id();
assert(!_stacks[w].empty());
if(_stacks[w].size() > _timeline.segments[w].size()) {
_timeline.segments[w].resize(_stacks[w].size());
}
auto beg = _stacks[w].top();
_stacks[w].pop();
_timeline.segments[w][_stacks[w].size()].emplace_back(
tv.name(), tv.type(), beg, observer_stamp_t::clock::now()
);
}
// Function: clear
inline void TFProfObserver::clear() {
for(size_t w=0; w<_timeline.segments.size(); ++w) {
for(size_t l=0; l<_timeline.segments[w].size(); ++l) {
_timeline.segments[w][l].clear();
}
while(!_stacks[w].empty()) {
_stacks[w].pop();
}
}
}
// Procedure: dump
inline void TFProfObserver::dump(std::ostream& os) const {
size_t first;
for(first = 0; first<_timeline.segments.size(); ++first) {
if(_timeline.segments[first].size() > 0) {
break;
}
}
// not timeline data to dump
if(first == _timeline.segments.size()) {
os << "{}\n";
return;
}
os << "{\"executor\":\"" << _timeline.uid << "\",\"data\":[";
bool comma = false;
for(size_t w=first; w<_timeline.segments.size(); w++) {
for(size_t l=0; l<_timeline.segments[w].size(); l++) {
if(_timeline.segments[w][l].empty()) {
continue;
}
if(comma) {
os << ',';
}
else {
comma = true;
}
os << "{\"worker\":" << w << ",\"level\":" << l << ",\"data\":[";
for(size_t i=0; i<_timeline.segments[w][l].size(); ++i) {
const auto& s = _timeline.segments[w][l][i];
if(i) os << ',';
// span
os << "{\"span\":["
<< std::chrono::duration_cast<std::chrono::microseconds>(
s.beg - _timeline.origin
).count() << ","
<< std::chrono::duration_cast<std::chrono::microseconds>(
s.end - _timeline.origin
).count() << "],";
// name
os << "\"name\":\"";
if(s.name.empty()) {
os << w << '_' << i;
}
else {
os << s.name;
}
os << "\",";
// category "type": "Condition Task",
os << "\"type\":\"" << to_string(s.type) << "\"";
os << "}";
}
os << "]}";
}
}
os << "]}\n";
}
// Function: dump
inline std::string TFProfObserver::dump() const {
std::ostringstream oss;
dump(oss);
return oss.str();
}
// Function: num_tasks
inline size_t TFProfObserver::num_tasks() const {
return std::accumulate(
_timeline.segments.begin(), _timeline.segments.end(), size_t{0},
[](size_t sum, const auto& exe){
return sum + exe.size();
}
);
}
// ----------------------------------------------------------------------------
// TFProfManager
// ----------------------------------------------------------------------------
/**
@private
*/
class TFProfManager {
friend class Executor;
public:
~TFProfManager();
TFProfManager(const TFProfManager&) = delete;
TFProfManager& operator=(const TFProfManager&) = delete;
static TFProfManager& get();
void dump(std::ostream& ostream) const;
private:
const std::string _fpath;
std::mutex _mutex;
std::vector<std::shared_ptr<TFProfObserver>> _observers;
TFProfManager();
void _manage(std::shared_ptr<TFProfObserver> observer);
};
// constructor
inline TFProfManager::TFProfManager() :
_fpath {get_env(TF_ENABLE_PROFILER)} {
}
// Procedure: manage
inline void TFProfManager::_manage(std::shared_ptr<TFProfObserver> observer) {
std::lock_guard lock(_mutex);
_observers.push_back(std::move(observer));
}
// Procedure: dump
inline void TFProfManager::dump(std::ostream& os) const {
for(size_t i=0; i<_observers.size(); ++i) {
if(i) os << ',';
_observers[i]->dump(os);
}
}
// Destructor
inline TFProfManager::~TFProfManager() {
std::ofstream ofs(_fpath);
if(ofs) {
// .tfp
if(_fpath.rfind(".tfp") != std::string::npos) {
ProfileData data;
data.timelines.reserve(_observers.size());
for(size_t i=0; i<_observers.size(); ++i) {
data.timelines.push_back(std::move(_observers[i]->_timeline));
}
Serializer<std::ofstream> serializer(ofs);
serializer(data);
}
// .json
else {
ofs << "[\n";
for(size_t i=0; i<_observers.size(); ++i) {
if(i) ofs << ',';
_observers[i]->dump(ofs);
}
ofs << "]\n";
}
}
}
// Function: get
inline TFProfManager& TFProfManager::get() {
static TFProfManager mgr;
return mgr;
}
// ----------------------------------------------------------------------------
// Identifier for Each Built-in Observer
// ----------------------------------------------------------------------------
/** @enum ObserverType
@brief enumeration of all observer types
*/
enum class ObserverType : int {
TFPROF = 0,
CHROME,
UNDEFINED
};
/**
@brief convert an observer type to a human-readable string
*/
inline const char* to_string(ObserverType type) {
switch(type) {
case ObserverType::TFPROF: return "tfprof";
case ObserverType::CHROME: return "chrome";
default: return "undefined";
}
}
} // end of namespace tf -----------------------------------------------------

View File

@ -0,0 +1,125 @@
#pragma once
#include <vector>
#include <mutex>
#include "declarations.hpp"
/**
@file semaphore.hpp
@brief semaphore include file
*/
namespace tf {
// ----------------------------------------------------------------------------
// Semaphore
// ----------------------------------------------------------------------------
/**
@class Semaphore
@brief class to create a semophore object for building a concurrency constraint
A semaphore creates a constraint that limits the maximum concurrency,
i.e., the number of workers, in a set of tasks.
You can let a task acquire/release one or multiple semaphores before/after
executing its work.
A task can acquire and release a semaphore,
or just acquire or just release it.
A tf::Semaphore object starts with an initial count.
As long as that count is above 0, tasks can acquire the semaphore and do
their work.
If the count is 0 or less, a task trying to acquire the semaphore will not run
but goes to a waiting list of that semaphore.
When the semaphore is released by another task,
it reschedules all tasks on that waiting list.
@code{.cpp}
tf::Executor executor(8); // create an executor of 8 workers
tf::Taskflow taskflow;
tf::Semaphore semaphore(1); // create a semaphore with initial count 1
std::vector<tf::Task> tasks {
taskflow.emplace([](){ std::cout << "A" << std::endl; }),
taskflow.emplace([](){ std::cout << "B" << std::endl; }),
taskflow.emplace([](){ std::cout << "C" << std::endl; }),
taskflow.emplace([](){ std::cout << "D" << std::endl; }),
taskflow.emplace([](){ std::cout << "E" << std::endl; })
};
for(auto & task : tasks) { // each task acquires and release the semaphore
task.acquire(semaphore);
task.release(semaphore);
}
executor.run(taskflow).wait();
@endcode
The above example creates five tasks with no dependencies between them.
Under normal circumstances, the five tasks would be executed concurrently.
However, this example has a semaphore with initial count 1,
and all tasks need to acquire that semaphore before running and release that
semaphore after they are done.
This organization limits the number of concurrently running tasks to only one.
*/
class Semaphore {
friend class Node;
public:
/**
@brief constructs a semaphore with the given counter
*/
explicit Semaphore(int max_workers);
/**
@brief queries the counter value (not thread-safe during the run)
*/
int count() const;
private:
std::mutex _mtx;
int _counter;
std::vector<Node*> _waiters;
bool _try_acquire_or_wait(Node*);
std::vector<Node*> _release();
};
inline Semaphore::Semaphore(int max_workers) :
_counter(max_workers) {
}
inline bool Semaphore::_try_acquire_or_wait(Node* me) {
std::lock_guard<std::mutex> lock(_mtx);
if(_counter > 0) {
--_counter;
return true;
}
else {
_waiters.push_back(me);
return false;
}
}
inline std::vector<Node*> Semaphore::_release() {
std::lock_guard<std::mutex> lock(_mtx);
++_counter;
std::vector<Node*> r{std::move(_waiters)};
return r;
}
inline int Semaphore::count() const {
return _counter;
}
} // end of namespace tf. ---------------------------------------------------

View File

@ -0,0 +1,713 @@
#pragma once
#include "graph.hpp"
/**
@file task.hpp
@brief task include file
*/
namespace tf {
// ----------------------------------------------------------------------------
// Task Types
// ----------------------------------------------------------------------------
/**
@enum TaskType
@brief enumeration of all task types
*/
enum class TaskType : int {
/** @brief placeholder task type */
PLACEHOLDER = 0,
/** @brief cudaFlow task type */
CUDAFLOW,
/** @brief syclFlow task type */
SYCLFLOW,
/** @brief static task type */
STATIC,
/** @brief dynamic (subflow) task type */
DYNAMIC,
/** @brief condition task type */
CONDITION,
/** @brief module task type */
MODULE,
/** @brief asynchronous task type */
ASYNC,
/** @brief undefined task type (for internal use only) */
UNDEFINED
};
/**
@brief array of all task types (used for iterating task types)
*/
inline constexpr std::array<TaskType, 8> TASK_TYPES = {
TaskType::PLACEHOLDER,
TaskType::CUDAFLOW,
TaskType::SYCLFLOW,
TaskType::STATIC,
TaskType::DYNAMIC,
TaskType::CONDITION,
TaskType::MODULE,
TaskType::ASYNC
};
/**
@brief convert a task type to a human-readable string
*/
inline const char* to_string(TaskType type) {
const char* val;
switch(type) {
case TaskType::PLACEHOLDER: val = "placeholder"; break;
case TaskType::CUDAFLOW: val = "cudaflow"; break;
case TaskType::SYCLFLOW: val = "syclflow"; break;
case TaskType::STATIC: val = "static"; break;
case TaskType::DYNAMIC: val = "subflow"; break;
case TaskType::CONDITION: val = "condition"; break;
case TaskType::MODULE: val = "module"; break;
case TaskType::ASYNC: val = "async"; break;
default: val = "undefined"; break;
}
return val;
}
// ----------------------------------------------------------------------------
// Task Traits
// ----------------------------------------------------------------------------
/**
@brief determines if a callable is a static task
A static task is a callable object constructible from std::function<void()>.
*/
template <typename C>
constexpr bool is_static_task_v = std::is_invocable_r_v<void, C> &&
!std::is_invocable_r_v<int, C>;
/**
@brief determines if a callable is a dynamic task
A dynamic task is a callable object constructible from std::function<void(Subflow&)>.
*/
template <typename C>
constexpr bool is_dynamic_task_v = std::is_invocable_r_v<void, C, Subflow&>;
/**
@brief determines if a callable is a condition task
A condition task is a callable object constructible from std::function<int()>.
*/
template <typename C>
constexpr bool is_condition_task_v = std::is_invocable_r_v<int, C>;
/**
@brief determines if a callable is a %cudaFlow task
A cudaFlow task is a callable object constructible from
std::function<void(tf::cudaFlow&)> or std::function<void(tf::cudaFlowCapturer&)>.
*/
template <typename C>
constexpr bool is_cudaflow_task_v = std::is_invocable_r_v<void, C, cudaFlow&> ||
std::is_invocable_r_v<void, C, cudaFlowCapturer&>;
/**
@brief determines if a callable is a %syclFlow task
A syclFlow task is a callable object constructible from
std::function<void(tf::syclFlow&)>.
*/
template <typename C>
constexpr bool is_syclflow_task_v = std::is_invocable_r_v<void, C, syclFlow&>;
// ----------------------------------------------------------------------------
// Task
// ----------------------------------------------------------------------------
/**
@class Task
@brief handle to a node in a task dependency graph
A Task is handle to manipulate a node in a taskflow graph.
It provides a set of methods for users to access and modify the attributes of
the associated graph node without directly touching internal node data.
*/
class Task {
friend class FlowBuilder;
friend class Taskflow;
friend class TaskView;
public:
/**
@brief constructs an empty task
*/
Task() = default;
/**
@brief constructs the task with the copy of the other task
*/
Task(const Task& other);
/**
@brief replaces the contents with a copy of the other task
*/
Task& operator = (const Task&);
/**
@brief replaces the contents with a null pointer
*/
Task& operator = (std::nullptr_t);
/**
@brief compares if two tasks are associated with the same graph node
*/
bool operator == (const Task& rhs) const;
/**
@brief compares if two tasks are not associated with the same graph node
*/
bool operator != (const Task& rhs) const;
/**
@brief queries the name of the task
*/
const std::string& name() const;
/**
@brief queries the number of successors of the task
*/
size_t num_successors() const;
/**
@brief queries the number of predecessors of the task
*/
size_t num_dependents() const;
/**
@brief queries the number of strong dependents of the task
*/
size_t num_strong_dependents() const;
/**
@brief queries the number of weak dependents of the task
*/
size_t num_weak_dependents() const;
/**
@brief assigns a name to the task
@param name a @std_string acceptable string
@return @c *this
*/
Task& name(const std::string& name);
/**
@brief assigns a callable
@tparam C callable type
@param callable callable to construct one of the static, dynamic, condition, and cudaFlow tasks
@return @c *this
*/
template <typename C>
Task& work(C&& callable);
/**
@brief creates a module task from a taskflow
@param taskflow a taskflow object for the module
@return @c *this
*/
Task& composed_of(Taskflow& taskflow);
/**
@brief adds precedence links from this to other tasks
@tparam Ts parameter pack
@param tasks one or multiple tasks
@return @c *this
*/
template <typename... Ts>
Task& precede(Ts&&... tasks);
/**
@brief adds precedence links from other tasks to this
@tparam Ts parameter pack
@param tasks one or multiple tasks
@return @c *this
*/
template <typename... Ts>
Task& succeed(Ts&&... tasks);
/**
@brief makes the task release this semaphore
*/
Task& release(Semaphore& semaphore);
/**
@brief makes the task acquire this semaphore
*/
Task& acquire(Semaphore& semaphore);
/**
@brief assigns pointer to user data
@param data pointer to user data
@return @c *this
*/
Task& data(void* data);
/**
@brief resets the task handle to null
*/
void reset();
/**
@brief resets the associated work to a placeholder
*/
void reset_work();
/**
@brief queries if the task handle points to a task node
*/
bool empty() const;
/**
@brief queries if the task has a work assigned
*/
bool has_work() const;
/**
@brief applies an visitor callable to each successor of the task
*/
template <typename V>
void for_each_successor(V&& visitor) const;
/**
@brief applies an visitor callable to each dependents of the task
*/
template <typename V>
void for_each_dependent(V&& visitor) const;
/**
@brief obtains a hash value of the underlying node
*/
size_t hash_value() const;
/**
@brief returns the task type
*/
TaskType type() const;
/**
@brief dumps the task through an output stream
*/
void dump(std::ostream& ostream) const;
/**
@brief queries pointer to user data
*/
void* data() const;
private:
Task(Node*);
Node* _node {nullptr};
};
// Constructor
inline Task::Task(Node* node) : _node {node} {
}
// Constructor
inline Task::Task(const Task& rhs) : _node {rhs._node} {
}
// Function: precede
template <typename... Ts>
Task& Task::precede(Ts&&... tasks) {
(_node->_precede(tasks._node), ...);
//_precede(std::forward<Ts>(tasks)...);
return *this;
}
// Function: succeed
template <typename... Ts>
Task& Task::succeed(Ts&&... tasks) {
(tasks._node->_precede(_node), ...);
//_succeed(std::forward<Ts>(tasks)...);
return *this;
}
// Function: composed_of
inline Task& Task::composed_of(Taskflow& tf) {
_node->_handle.emplace<Node::Module>(&tf);
return *this;
}
// Operator =
inline Task& Task::operator = (const Task& rhs) {
_node = rhs._node;
return *this;
}
// Operator =
inline Task& Task::operator = (std::nullptr_t ptr) {
_node = ptr;
return *this;
}
// Operator ==
inline bool Task::operator == (const Task& rhs) const {
return _node == rhs._node;
}
// Operator !=
inline bool Task::operator != (const Task& rhs) const {
return _node != rhs._node;
}
// Function: name
inline Task& Task::name(const std::string& name) {
_node->_name = name;
return *this;
}
// Function: acquire
inline Task& Task::acquire(Semaphore& s) {
if(!_node->_semaphores) {
//_node->_semaphores.emplace();
_node->_semaphores = std::make_unique<Node::Semaphores>();
}
_node->_semaphores->to_acquire.push_back(&s);
return *this;
}
// Function: release
inline Task& Task::release(Semaphore& s) {
if(!_node->_semaphores) {
//_node->_semaphores.emplace();
_node->_semaphores = std::make_unique<Node::Semaphores>();
}
_node->_semaphores->to_release.push_back(&s);
return *this;
}
// Procedure: reset
inline void Task::reset() {
_node = nullptr;
}
// Procedure: reset_work
inline void Task::reset_work() {
_node->_handle.emplace<std::monostate>();
}
// Function: name
inline const std::string& Task::name() const {
return _node->_name;
}
// Function: num_dependents
inline size_t Task::num_dependents() const {
return _node->num_dependents();
}
// Function: num_strong_dependents
inline size_t Task::num_strong_dependents() const {
return _node->num_strong_dependents();
}
// Function: num_weak_dependents
inline size_t Task::num_weak_dependents() const {
return _node->num_weak_dependents();
}
// Function: num_successors
inline size_t Task::num_successors() const {
return _node->num_successors();
}
// Function: empty
inline bool Task::empty() const {
return _node == nullptr;
}
// Function: has_work
inline bool Task::has_work() const {
return _node ? _node->_handle.index() != 0 : false;
}
// Function: task_type
inline TaskType Task::type() const {
switch(_node->_handle.index()) {
case Node::PLACEHOLDER: return TaskType::PLACEHOLDER;
case Node::STATIC: return TaskType::STATIC;
case Node::DYNAMIC: return TaskType::DYNAMIC;
case Node::CONDITION: return TaskType::CONDITION;
case Node::MODULE: return TaskType::MODULE;
case Node::ASYNC: return TaskType::ASYNC;
case Node::SILENT_ASYNC: return TaskType::ASYNC;
case Node::CUDAFLOW: return TaskType::CUDAFLOW;
case Node::SYCLFLOW: return TaskType::SYCLFLOW;
default: return TaskType::UNDEFINED;
}
}
// Function: for_each_successor
template <typename V>
void Task::for_each_successor(V&& visitor) const {
for(size_t i=0; i<_node->_successors.size(); ++i) {
visitor(Task(_node->_successors[i]));
}
}
// Function: for_each_dependent
template <typename V>
void Task::for_each_dependent(V&& visitor) const {
for(size_t i=0; i<_node->_dependents.size(); ++i) {
visitor(Task(_node->_dependents[i]));
}
}
// Function: hash_value
inline size_t Task::hash_value() const {
return std::hash<Node*>{}(_node);
}
// Procedure: dump
inline void Task::dump(std::ostream& os) const {
os << "task ";
if(name().empty()) os << _node;
else os << name();
os << " [type=" << to_string(type()) << ']';
}
// Function: work
template <typename C>
Task& Task::work(C&& c) {
if constexpr(is_static_task_v<C>) {
_node->_handle.emplace<Node::Static>(std::forward<C>(c));
}
else if constexpr(is_dynamic_task_v<C>) {
_node->_handle.emplace<Node::Dynamic>(std::forward<C>(c));
}
else if constexpr(is_condition_task_v<C>) {
_node->_handle.emplace<Node::Condition>(std::forward<C>(c));
}
else if constexpr(is_cudaflow_task_v<C>) {
_node->_handle.emplace<Node::cudaFlow>(std::forward<C>(c));
}
else {
static_assert(dependent_false_v<C>, "invalid task callable");
}
return *this;
}
// Function: name
inline void* Task::data() const {
return _node->_data;
}
// Function: name
inline Task& Task::data(void* data) {
_node->_data = data;
return *this;
}
// ----------------------------------------------------------------------------
// global ostream
// ----------------------------------------------------------------------------
/**
@brief overload of ostream inserter operator for cudaTask
*/
inline std::ostream& operator << (std::ostream& os, const Task& task) {
task.dump(os);
return os;
}
// ----------------------------------------------------------------------------
/**
@class TaskView
@brief class to access task information from the observer interface
*/
class TaskView {
friend class Executor;
public:
/**
@brief queries the name of the task
*/
const std::string& name() const;
/**
@brief queries the number of successors of the task
*/
size_t num_successors() const;
/**
@brief queries the number of predecessors of the task
*/
size_t num_dependents() const;
/**
@brief queries the number of strong dependents of the task
*/
size_t num_strong_dependents() const;
/**
@brief queries the number of weak dependents of the task
*/
size_t num_weak_dependents() const;
/**
@brief applies an visitor callable to each successor of the task
*/
template <typename V>
void for_each_successor(V&& visitor) const;
/**
@brief applies an visitor callable to each dependents of the task
*/
template <typename V>
void for_each_dependent(V&& visitor) const;
/**
@brief queries the task type
*/
TaskType type() const;
/**
@brief obtains a hash value of the underlying node
*/
size_t hash_value() const;
private:
TaskView(const Node&);
TaskView(const TaskView&) = default;
const Node& _node;
};
// Constructor
inline TaskView::TaskView(const Node& node) : _node {node} {
}
// Function: name
inline const std::string& TaskView::name() const {
return _node._name;
}
// Function: num_dependents
inline size_t TaskView::num_dependents() const {
return _node.num_dependents();
}
// Function: num_strong_dependents
inline size_t TaskView::num_strong_dependents() const {
return _node.num_strong_dependents();
}
// Function: num_weak_dependents
inline size_t TaskView::num_weak_dependents() const {
return _node.num_weak_dependents();
}
// Function: num_successors
inline size_t TaskView::num_successors() const {
return _node.num_successors();
}
// Function: type
inline TaskType TaskView::type() const {
switch(_node._handle.index()) {
case Node::PLACEHOLDER: return TaskType::PLACEHOLDER;
case Node::STATIC: return TaskType::STATIC;
case Node::DYNAMIC: return TaskType::DYNAMIC;
case Node::CONDITION: return TaskType::CONDITION;
case Node::MODULE: return TaskType::MODULE;
case Node::ASYNC: return TaskType::ASYNC;
case Node::SILENT_ASYNC: return TaskType::ASYNC;
case Node::CUDAFLOW: return TaskType::CUDAFLOW;
case Node::SYCLFLOW: return TaskType::SYCLFLOW;
default: return TaskType::UNDEFINED;
}
}
// Function: hash_value
inline size_t TaskView::hash_value() const {
return std::hash<const Node*>{}(&_node);
}
// Function: for_each_successor
template <typename V>
void TaskView::for_each_successor(V&& visitor) const {
for(size_t i=0; i<_node._successors.size(); ++i) {
visitor(TaskView(_node._successors[i]));
}
}
// Function: for_each_dependent
template <typename V>
void TaskView::for_each_dependent(V&& visitor) const {
for(size_t i=0; i<_node._dependents.size(); ++i) {
visitor(TaskView(_node._dependents[i]));
}
}
} // end of namespace tf. ---------------------------------------------------
namespace std {
/**
@struct hash
@brief hash specialization for std::hash<tf::Task>
*/
template <>
struct hash<tf::Task> {
auto operator() (const tf::Task& task) const noexcept {
return task.hash_value();
}
};
/**
@struct hash
@brief hash specialization for std::hash<tf::TaskView>
*/
template <>
struct hash<tf::TaskView> {
auto operator() (const tf::TaskView& task_view) const noexcept {
return task_view.hash_value();
}
};
} // end of namespace std ----------------------------------------------------

View File

@ -0,0 +1,540 @@
#pragma once
#include "flow_builder.hpp"
/**
@file core/taskflow.hpp
@brief taskflow include file
*/
namespace tf {
// ----------------------------------------------------------------------------
/**
@class Taskflow
@brief main entry to create a task dependency graph
A %taskflow manages a task dependency graph where each task represents a
callable object (e.g., @std_lambda, @std_function) and an edge represents a
dependency between two tasks. A task is one of the following types:
1. static task : the callable constructible from
@c std::function<void()>
2. dynamic task : the callable constructible from
@c std::function<void(tf::Subflow&)>
3. condition task: the callable constructible from
@c std::function<int()>
4. module task : the task constructed from tf::Taskflow::composed_of
5. %cudaFlow task: the callable constructible from
@c std::function<void(tf::cudaFlow&)> or
@c std::function<void(tf::cudaFlowCapturer&)>
6. %syclFlow task: the callable constructible from
@c std::function<void(tf::syclFlow&)>
Each task is a basic computation unit and is run by one worker thread
from an executor.
The following example creates a simple taskflow graph of four static tasks,
@c A, @c B, @c C, and @c D, where
@c A runs before @c B and @c C and
@c D runs after @c B and @c C.
@code{.cpp}
tf::Executor executor;
tf::Taskflow taskflow("simple");
tf::Task A = taskflow.emplace([](){ std::cout << "TaskA\n"; });
tf::Task B = taskflow.emplace([](){ std::cout << "TaskB\n"; });
tf::Task C = taskflow.emplace([](){ std::cout << "TaskC\n"; });
tf::Task D = taskflow.emplace([](){ std::cout << "TaskD\n"; });
A.precede(B, C); // A runs before B and C
D.succeed(B, C); // D runs after B and C
executor.run(taskflow).wait();
@endcode
Please refer to @ref Cookbook to learn more about each task type
and how to submit a taskflow to an executor.
*/
class Taskflow : public FlowBuilder {
friend class Topology;
friend class Executor;
friend class FlowBuilder;
struct Dumper {
std::stack<const Taskflow*> stack;
std::unordered_set<const Taskflow*> visited;
};
public:
/**
@brief constructs a taskflow with the given name
*/
Taskflow(const std::string& name);
/**
@brief constructs a taskflow
*/
Taskflow();
/**
@brief constructs a taskflow from a moved taskflow
Move a running taskflow can result in undefined behavior.
You should only move a taskflow to another if it is not being run by
an executor.
*/
Taskflow(Taskflow&& rhs);
/**
@brief move assignment operator
Move a running taskflow can result in undefined behavior.
You should only move a taskflow to another if it is not being run by
an executor.
*/
Taskflow& operator = (Taskflow&& rhs);
/**
@brief default destructor
When the destructor is called, all tasks and their associated data
(e.g., captured data) will be destroyed.
It is your responsibility to ensure all submitted execution of this
taskflow have completed before destroying it.
*/
~Taskflow() = default;
/**
@brief dumps the taskflow to a DOT format through a std::ostream target
*/
void dump(std::ostream& ostream) const;
/**
@brief dumps the taskflow to a std::string of DOT format
*/
std::string dump() const;
/**
@brief queries the number of tasks
*/
size_t num_tasks() const;
/**
@brief queries the emptiness of the taskflow
*/
bool empty() const;
/**
@brief assigns a name to the taskflow
*/
void name(const std::string&);
/**
@brief queries the name of the taskflow
*/
const std::string& name() const ;
/**
@brief clears the associated task dependency graph
When you clear a taskflow, all tasks and their associated data
(e.g., captured data) will be destroyed.
You should never clean a taskflow while it is being run by an executor.
*/
void clear();
/**
@brief applies a visitor to each task in the taskflow
A visitor is a callable that takes an argument of type tf::Task
and returns nothing. The following example iterates each task in a
taskflow and prints its name:
@code{.cpp}
taskflow.for_each_task([](tf::Task task){
std::cout << task.name() << '\n';
});
@endcode
*/
template <typename V>
void for_each_task(V&& visitor) const;
private:
mutable std::mutex _mutex;
std::string _name;
Graph _graph;
std::queue<std::shared_ptr<Topology>> _topologies;
std::optional<std::list<Taskflow>::iterator> _satellite;
void _dump(std::ostream&, const Taskflow*) const;
void _dump(std::ostream&, const Node*, Dumper&) const;
void _dump(std::ostream&, const Graph&, Dumper&) const;
};
// Constructor
inline Taskflow::Taskflow(const std::string& name) :
FlowBuilder {_graph},
_name {name} {
}
// Constructor
inline Taskflow::Taskflow() : FlowBuilder{_graph} {
}
// Move constructor
inline Taskflow::Taskflow(Taskflow&& rhs) : FlowBuilder{_graph} {
std::scoped_lock<std::mutex> lock(rhs._mutex);
_name = std::move(rhs._name);
_graph = std::move(rhs._graph);
_topologies = std::move(rhs._topologies);
_satellite = rhs._satellite;
rhs._satellite.reset();
}
// Move assignment
inline Taskflow& Taskflow::operator = (Taskflow&& rhs) {
if(this != &rhs) {
std::scoped_lock<std::mutex, std::mutex> lock(_mutex, rhs._mutex);
_name = std::move(rhs._name);
_graph = std::move(rhs._graph);
_topologies = std::move(rhs._topologies);
_satellite = rhs._satellite;
rhs._satellite.reset();
}
return *this;
}
// Procedure:
inline void Taskflow::clear() {
_graph.clear();
}
// Function: num_tasks
inline size_t Taskflow::num_tasks() const {
return _graph.size();
}
// Function: empty
inline bool Taskflow::empty() const {
return _graph.empty();
}
// Function: name
inline void Taskflow::name(const std::string &name) {
_name = name;
}
// Function: name
inline const std::string& Taskflow::name() const {
return _name;
}
// Function: for_each_task
template <typename V>
void Taskflow::for_each_task(V&& visitor) const {
for(size_t i=0; i<_graph._nodes.size(); ++i) {
visitor(Task(_graph._nodes[i]));
}
}
// Procedure: dump
inline std::string Taskflow::dump() const {
std::ostringstream oss;
dump(oss);
return oss.str();
}
// Function: dump
inline void Taskflow::dump(std::ostream& os) const {
os << "digraph Taskflow {\n";
_dump(os, this);
os << "}\n";
}
// Procedure: _dump
inline void Taskflow::_dump(std::ostream& os, const Taskflow* top) const {
Dumper dumper;
dumper.stack.push(top);
dumper.visited.insert(top);
while(!dumper.stack.empty()) {
auto f = dumper.stack.top();
dumper.stack.pop();
os << "subgraph cluster_p" << f << " {\nlabel=\"Taskflow: ";
if(f->_name.empty()) os << 'p' << f;
else os << f->_name;
os << "\";\n";
_dump(os, f->_graph, dumper);
os << "}\n";
}
}
// Procedure: _dump
inline void Taskflow::_dump(
std::ostream& os, const Node* node, Dumper& dumper
) const {
os << 'p' << node << "[label=\"";
if(node->_name.empty()) os << 'p' << node;
else os << node->_name;
os << "\" ";
// shape for node
switch(node->_handle.index()) {
case Node::CONDITION:
os << "shape=diamond color=black fillcolor=aquamarine style=filled";
break;
case Node::CUDAFLOW:
os << " style=\"filled\""
<< " color=\"black\" fillcolor=\"purple\""
<< " fontcolor=\"white\""
<< " shape=\"folder\"";
break;
case Node::SYCLFLOW:
os << " style=\"filled\""
<< " color=\"black\" fillcolor=\"red\""
<< " fontcolor=\"white\""
<< " shape=\"folder\"";
break;
default:
break;
}
os << "];\n";
for(size_t s=0; s<node->_successors.size(); ++s) {
if(node->_handle.index() == Node::CONDITION) {
// case edge is dashed
os << 'p' << node << " -> p" << node->_successors[s]
<< " [style=dashed label=\"" << s << "\"];\n";
}
else {
os << 'p' << node << " -> p" << node->_successors[s] << ";\n";
}
}
// subflow join node
if(node->_parent && node->_successors.size() == 0) {
os << 'p' << node << " -> p" << node->_parent << ";\n";
}
switch(node->_handle.index()) {
case Node::DYNAMIC: {
auto& sbg = std::get<Node::Dynamic>(node->_handle).subgraph;
if(!sbg.empty()) {
os << "subgraph cluster_p" << node << " {\nlabel=\"Subflow: ";
if(node->_name.empty()) os << 'p' << node;
else os << node->_name;
os << "\";\n" << "color=blue\n";
_dump(os, sbg, dumper);
os << "}\n";
}
}
break;
case Node::CUDAFLOW: {
std::get<Node::cudaFlow>(node->_handle).graph->dump(
os, node, node->_name
);
}
break;
case Node::SYCLFLOW: {
std::get<Node::syclFlow>(node->_handle).graph->dump(
os, node, node->_name
);
}
break;
default:
break;
}
}
// Procedure: _dump
inline void Taskflow::_dump(
std::ostream& os, const Graph& graph, Dumper& dumper
) const {
for(const auto& n : graph._nodes) {
// regular task
if(n->_handle.index() != Node::MODULE) {
_dump(os, n, dumper);
}
// module task
else {
auto module = std::get<Node::Module>(n->_handle).module;
os << 'p' << n << "[shape=box3d, color=blue, label=\"";
if(n->_name.empty()) os << n;
else os << n->_name;
os << " [Taskflow: ";
if(module->_name.empty()) os << 'p' << module;
else os << module->_name;
os << "]\"];\n";
if(dumper.visited.find(module) == dumper.visited.end()) {
dumper.visited.insert(module);
dumper.stack.push(module);
}
for(const auto s : n->_successors) {
os << 'p' << n << "->" << 'p' << s << ";\n";
}
}
}
}
// ----------------------------------------------------------------------------
// class definition: Future
// ----------------------------------------------------------------------------
/**
@class Future
@brief class to access the result of task execution
tf::Future is a derived class from std::future that will eventually hold the
execution result of a submitted taskflow (e.g., tf::Executor::run)
or an asynchronous task (e.g., tf::Executor::async).
In addition to base methods of std::future,
you can call tf::Future::cancel to cancel the execution of the running taskflow
associated with this future object.
The following example cancels a submission of a taskflow that contains
1000 tasks each running one second.
@code{.cpp}
tf::Executor executor;
tf::Taskflow taskflow;
for(int i=0; i<1000; i++) {
taskflow.emplace([](){
std::this_thread::sleep_for(std::chrono::seconds(1));
});
}
// submit the taskflow
tf::Future fu = executor.run(taskflow);
// request to cancel the submitted execution above
fu.cancel();
// wait until the cancellation finishes
fu.get();
@endcode
*/
template <typename T>
class Future : public std::future<T> {
friend class Executor;
friend class Subflow;
using handle_t = std::variant<
std::monostate, std::weak_ptr<Topology>, std::weak_ptr<AsyncTopology>
>;
// variant index
constexpr static auto ASYNC = get_index_v<std::weak_ptr<AsyncTopology>, handle_t>;
constexpr static auto TASKFLOW = get_index_v<std::weak_ptr<Topology>, handle_t>;
public:
/**
@brief default constructor
*/
Future() = default;
/**
@brief disabled copy constructor
*/
Future(const Future&) = delete;
/**
@brief default move constructor
*/
Future(Future&&) = default;
/**
@brief disabled copy assignment
*/
Future& operator = (const Future&) = delete;
/**
@brief default move assignment
*/
Future& operator = (Future&&) = default;
/**
@brief cancels the execution of the running taskflow associated with
this future object
@return @c true if the execution can be cancelled or
@c false if the execution has already completed
*/
bool cancel();
private:
handle_t _handle;
template <typename P>
Future(std::future<T>&&, P&&);
};
template <typename T>
template <typename P>
Future<T>::Future(std::future<T>&& fu, P&& p) :
std::future<T> {std::move(fu)},
_handle {std::forward<P>(p)} {
}
// Function: cancel
template <typename T>
bool Future<T>::cancel() {
return std::visit([](auto&& arg){
using P = std::decay_t<decltype(arg)>;
if constexpr(std::is_same_v<P, std::monostate>) {
return false;
}
else {
auto ptr = arg.lock();
if(ptr) {
ptr->_is_cancelled = true;
return true;
}
return false;
}
}, _handle);
}
} // end of namespace tf. ---------------------------------------------------

View File

@ -0,0 +1,61 @@
#pragma once
namespace tf {
// ----------------------------------------------------------------------------
// class: TopologyBase
class TopologyBase {
friend class Executor;
friend class Node;
template <typename T>
friend class Future;
protected:
std::atomic<bool> _is_cancelled { false };
};
// ----------------------------------------------------------------------------
// class: AsyncTopology
class AsyncTopology : public TopologyBase {
};
// ----------------------------------------------------------------------------
// class: Topology
class Topology : public TopologyBase {
friend class Executor;
public:
template <typename P, typename C>
Topology(Taskflow&, P&&, C&&);
private:
Taskflow& _taskflow;
std::promise<void> _promise;
std::vector<Node*> _sources;
std::function<bool()> _pred;
std::function<void()> _call;
std::atomic<size_t> _join_counter {0};
};
// Constructor
template <typename P, typename C>
Topology::Topology(Taskflow& tf, P&& p, C&& c):
_taskflow(tf),
_pred {std::forward<P>(p)},
_call {std::forward<C>(c)} {
}
} // end of namespace tf. ----------------------------------------------------

View File

@ -0,0 +1,248 @@
#pragma once
#include <atomic>
#include <vector>
#include <cassert>
#include <cstdint>
#include <cstddef>
#include <cstdlib>
namespace tf {
/**
@class: TaskQueue
@tparam T data type (must be a pointer)
@brief Lock-free unbounded single-producer multiple-consumer queue.
This class implements the work stealing queue described in the paper,
"Correct and Efficient Work-Stealing for Weak Memory Models,"
available at https://www.di.ens.fr/~zappa/readings/ppopp13.pdf.
Only the queue owner can perform pop and push operations,
while others can steal data from the queue.
*/
template <typename T>
class TaskQueue {
static_assert(std::is_pointer_v<T>, "T must be a pointer type");
struct Array {
int64_t C;
int64_t M;
std::atomic<T>* S;
explicit Array(int64_t c) :
C {c},
M {c-1},
S {new std::atomic<T>[static_cast<size_t>(C)]} {
}
~Array() {
delete [] S;
}
int64_t capacity() const noexcept {
return C;
}
void push(int64_t i, T o) noexcept {
S[i & M].store(o, std::memory_order_relaxed);
}
T pop(int64_t i) noexcept {
return S[i & M].load(std::memory_order_relaxed);
}
Array* resize(int64_t b, int64_t t) {
Array* ptr = new Array {2*C};
for(int64_t i=t; i!=b; ++i) {
ptr->push(i, pop(i));
}
return ptr;
}
};
std::atomic<int64_t> _top;
std::atomic<int64_t> _bottom;
std::atomic<Array*> _array;
std::vector<Array*> _garbage;
public:
/**
@brief constructs the queue with a given capacity
@param capacity the capacity of the queue (must be power of 2)
*/
explicit TaskQueue(int64_t capacity = 1024);
/**
@brief destructs the queue
*/
~TaskQueue();
/**
@brief queries if the queue is empty at the time of this call
*/
bool empty() const noexcept;
/**
@brief queries the number of items at the time of this call
*/
size_t size() const noexcept;
/**
@brief queries the capacity of the queue
*/
int64_t capacity() const noexcept;
/**
@brief inserts an item to the queue
Only the owner thread can insert an item to the queue.
The operation can trigger the queue to resize its capacity
if more space is required.
@tparam O data type
@param item the item to perfect-forward to the queue
*/
void push(T item);
/**
@brief pops out an item from the queue
Only the owner thread can pop out an item from the queue.
The return can be a nullptr if this operation failed (empty queue).
*/
T pop();
/**
@brief steals an item from the queue
Any threads can try to steal an item from the queue.
The return can be a nullptr if this operation failed (not necessary empty).
*/
T steal();
};
// Constructor
template <typename T>
TaskQueue<T>::TaskQueue(int64_t c) {
assert(c && (!(c & (c-1))));
_top.store(0, std::memory_order_relaxed);
_bottom.store(0, std::memory_order_relaxed);
_array.store(new Array{c}, std::memory_order_relaxed);
_garbage.reserve(32);
}
// Destructor
template <typename T>
TaskQueue<T>::~TaskQueue() {
for(auto a : _garbage) {
delete a;
}
delete _array.load();
}
// Function: empty
template <typename T>
bool TaskQueue<T>::empty() const noexcept {
int64_t b = _bottom.load(std::memory_order_relaxed);
int64_t t = _top.load(std::memory_order_relaxed);
return b <= t;
}
// Function: size
template <typename T>
size_t TaskQueue<T>::size() const noexcept {
int64_t b = _bottom.load(std::memory_order_relaxed);
int64_t t = _top.load(std::memory_order_relaxed);
return static_cast<size_t>(b >= t ? b - t : 0);
}
// Function: push
template <typename T>
void TaskQueue<T>::push(T o) {
int64_t b = _bottom.load(std::memory_order_relaxed);
int64_t t = _top.load(std::memory_order_acquire);
Array* a = _array.load(std::memory_order_relaxed);
// queue is full
if(a->capacity() - 1 < (b - t)) {
Array* tmp = a->resize(b, t);
_garbage.push_back(a);
std::swap(a, tmp);
_array.store(a, std::memory_order_release);
// Note: the original paper using relaxed causes t-san to complain
//_array.store(a, std::memory_order_relaxed);
}
a->push(b, o);
std::atomic_thread_fence(std::memory_order_release);
_bottom.store(b + 1, std::memory_order_relaxed);
}
// Function: pop
template <typename T>
T TaskQueue<T>::pop() {
int64_t b = _bottom.load(std::memory_order_relaxed) - 1;
Array* a = _array.load(std::memory_order_relaxed);
_bottom.store(b, std::memory_order_relaxed);
std::atomic_thread_fence(std::memory_order_seq_cst);
int64_t t = _top.load(std::memory_order_relaxed);
T item {nullptr};
if(t <= b) {
item = a->pop(b);
if(t == b) {
// the last item just got stolen
if(!_top.compare_exchange_strong(t, t+1,
std::memory_order_seq_cst,
std::memory_order_relaxed)) {
item = nullptr;
}
_bottom.store(b + 1, std::memory_order_relaxed);
}
}
else {
_bottom.store(b + 1, std::memory_order_relaxed);
}
return item;
}
// Function: steal
template <typename T>
T TaskQueue<T>::steal() {
int64_t t = _top.load(std::memory_order_acquire);
std::atomic_thread_fence(std::memory_order_seq_cst);
int64_t b = _bottom.load(std::memory_order_acquire);
T item {nullptr};
if(t < b) {
Array* a = _array.load(std::memory_order_consume);
item = a->pop(t);
if(!_top.compare_exchange_strong(t, t+1,
std::memory_order_seq_cst,
std::memory_order_relaxed)) {
return nullptr;
}
}
return item;
}
// Function: capacity
template <typename T>
int64_t TaskQueue<T>::capacity() const noexcept {
return _array.load(std::memory_order_relaxed)->capacity();
}
} // end of namespace tf -----------------------------------------------------

View File

@ -0,0 +1,127 @@
#pragma once
#include "declarations.hpp"
#include "tsq.hpp"
#include "notifier.hpp"
/**
@file worker.hpp
@brief worker include file
*/
namespace tf {
/**
@private
*/
class Worker {
friend class Executor;
friend class WorkerView;
private:
size_t _id;
size_t _vtm;
Executor* _executor;
Notifier::Waiter* _waiter;
std::default_random_engine _rdgen { std::random_device{}() };
TaskQueue<Node*> _wsq;
};
/**
@private
*/
struct PerThreadWorker {
Worker* worker;
PerThreadWorker() : worker {nullptr} {}
PerThreadWorker(const PerThreadWorker&) = delete;
PerThreadWorker(PerThreadWorker&&) = delete;
PerThreadWorker& operator = (const PerThreadWorker&) = delete;
PerThreadWorker& operator = (PerThreadWorker&&) = delete;
};
/**
@private
*/
inline PerThreadWorker& this_worker() {
thread_local PerThreadWorker worker;
return worker;
}
// ----------------------------------------------------------------------------
// Class Definition: WorkerView
// ----------------------------------------------------------------------------
/**
@class WorkerView
@brief class to create an immutable view of a worker in an executor
An executor keeps a set of internal worker threads to run tasks.
A worker view provides users an immutable interface to observe
when a worker runs a task, and the view object is only accessible
from an observer derived from tf::ObserverInterface.
*/
class WorkerView {
friend class Executor;
public:
/**
@brief queries the worker id associated with the executor
A worker id is a unsigned integer in the range <tt>[0, N)</tt>,
where @c N is the number of workers spawned at the construction
time of the executor.
*/
size_t id() const;
/**
@brief queries the size of the queue (i.e., number of pending tasks to
run) associated with the worker
*/
size_t queue_size() const;
/**
@brief queries the current capacity of the queue
*/
size_t queue_capacity() const;
private:
WorkerView(const Worker&);
WorkerView(const WorkerView&) = default;
const Worker& _worker;
};
// Constructor
inline WorkerView::WorkerView(const Worker& w) : _worker{w} {
}
// function: id
inline size_t WorkerView::id() const {
return _worker._id;
}
// Function: queue_size
inline size_t WorkerView::queue_size() const {
return _worker._wsq.size();
}
// Function: queue_capacity
inline size_t WorkerView::queue_capacity() const {
return static_cast<size_t>(_worker._wsq.capacity());
}
} // end of namespact tf -----------------------------------------------------

View File

@ -0,0 +1,486 @@
#pragma once
#include "../cuda_flow.hpp"
#include "../cuda_capturer.hpp"
#include "../cuda_meta.hpp"
/**
@file cuda_find.hpp
@brief cuda find algorithms include file
*/
namespace tf::detail {
/** @private */
template <typename T>
struct cudaFindPair {
T key;
unsigned index;
__device__ operator unsigned () const { return index; }
};
/** @private */
template <typename P, typename I, typename U>
void cuda_find_if_loop(P&& p, I input, unsigned count, unsigned* idx, U pred) {
if(count == 0) {
cuda_single_task(p, [=] __device__ () { *idx = 0; });
return;
}
using E = std::decay_t<P>;
auto B = (count + E::nv - 1) / E::nv;
// set the index to the maximum
cuda_single_task(p, [=] __device__ () { *idx = count; });
// launch the kernel to atomic-find the minimum
cuda_kernel<<<B, E::nt, 0, p.stream()>>>([=] __device__ (auto tid, auto bid) {
__shared__ unsigned shm_id;
if(!tid) {
shm_id = count;
}
__syncthreads();
auto tile = cuda_get_tile(bid, E::nv, count);
auto x = cuda_mem_to_reg_strided<E::nt, E::vt>(
input + tile.begin, tid, tile.count()
);
auto id = count;
for(unsigned i=0; i<E::vt; i++) {
auto j = E::nt*i + tid;
if(j < tile.count() && pred(x[i])) {
id = j + tile.begin;
break;
}
}
// Note: the reduce version is not faster though
// reduce to a scalar per block.
//__shared__ typename cudaBlockReduce<E::nt, unsigned>::Storage shm;
//id = cudaBlockReduce<E::nt, unsigned>()(
// tid,
// id,
// shm,
// (tile.count() < E::nt ? tile.count() : E::nt),
// cuda_minimum<unsigned>{},
// false
//);
// only need the minimum id
atomicMin(&shm_id, id);
__syncthreads();
// reduce all to the global memory
if(!tid) {
atomicMin(idx, shm_id);
//atomicMin(idx, id);
}
});
}
/** @private */
template <typename P, typename I, typename O>
void cuda_min_element_loop(
P&& p, I input, unsigned count, unsigned* idx, O op, void* ptr
) {
if(count == 0) {
cuda_single_task(p, [=] __device__ () { *idx = 0; });
return;
}
using T = cudaFindPair<typename std::iterator_traits<I>::value_type>;
cuda_uninitialized_reduce_loop(p,
cuda_make_load_iterator<T>([=]__device__(auto i){
return T{*(input+i), i};
}),
count,
idx,
[=] __device__ (const auto& a, const auto& b) {
return op(a.key, b.key) ? a : b;
},
ptr
);
}
/** @private */
template <typename P, typename I, typename O>
void cuda_max_element_loop(
P&& p, I input, unsigned count, unsigned* idx, O op, void* ptr
) {
if(count == 0) {
cuda_single_task(p, [=] __device__ () { *idx = 0; });
return;
}
using T = cudaFindPair<typename std::iterator_traits<I>::value_type>;
cuda_uninitialized_reduce_loop(p,
cuda_make_load_iterator<T>([=]__device__(auto i){
return T{*(input+i), i};
}),
count,
idx,
[=] __device__ (const auto& a, const auto& b) {
return op(a.key, b.key) ? b : a;
},
ptr
);
}
} // end of namespace tf::detail ---------------------------------------------
namespace tf {
// ----------------------------------------------------------------------------
// cuda_find_if
// ----------------------------------------------------------------------------
/**
@brief finds the index of the first element that satisfies the given criteria
@tparam P execution policy type
@tparam I input iterator type
@tparam U unary operator type
@param p execution policy
@param first iterator to the beginning of the range
@param last iterator to the end of the range
@param idx pointer to the index of the found element
@param op unary operator which returns @c true for the required element
The function launches kernels asynchronously to find the index @c idx of the
first element in the range <tt>[first, last)</tt>
such that <tt>op(*(first+idx))</tt> is true.
This is equivalent to the parallel execution of the following loop:
@code{.cpp}
unsigned idx = 0;
for(; first != last; ++first, ++idx) {
if (p(*first)) {
return idx;
}
}
return idx;
@endcode
*/
template <typename P, typename I, typename U>
void cuda_find_if(
P&& p, I first, I last, unsigned* idx, U op
) {
detail::cuda_find_if_loop(p, first, std::distance(first, last), idx, op);
}
// ----------------------------------------------------------------------------
// cudaFlow
// ----------------------------------------------------------------------------
// Function: find_if
template <typename I, typename U>
cudaTask cudaFlow::find_if(I first, I last, unsigned* idx, U op) {
return capture([=](cudaFlowCapturer& cap){
cap.make_optimizer<cudaLinearCapturing>();
cap.find_if(first, last, idx, op);
});
}
// Function: find_if
template <typename I, typename U>
void cudaFlow::find_if(cudaTask task, I first, I last, unsigned* idx, U op) {
capture(task, [=](cudaFlowCapturer& cap){
cap.make_optimizer<cudaLinearCapturing>();
cap.find_if(first, last, idx, op);
});
}
// ----------------------------------------------------------------------------
// cudaFlowCapturer
// ----------------------------------------------------------------------------
// Function: find_if
template <typename I, typename U>
cudaTask cudaFlowCapturer::find_if(I first, I last, unsigned* idx, U op) {
return on([=](cudaStream_t stream) mutable {
cudaDefaultExecutionPolicy p(stream);
cuda_find_if(p, first, last, idx, op);
});
}
// Function: find_if
template <typename I, typename U>
void cudaFlowCapturer::find_if(
cudaTask task, I first, I last, unsigned* idx, U op
) {
on(task, [=](cudaStream_t stream) mutable {
cudaDefaultExecutionPolicy p(stream);
cuda_find_if(p, first, last, idx, op);
});
}
// ----------------------------------------------------------------------------
// cuda_min_element
// ----------------------------------------------------------------------------
/**
@brief queries the buffer size in bytes needed to call tf::cuda_min_element
@tparam P execution policy type
@tparam T value type
@param count number of elements to search
The function is used to decide the buffer size in bytes for calling
tf::cuda_min_element.
*/
template <typename P, typename T>
unsigned cuda_min_element_buffer_size(unsigned count) {
return cuda_reduce_buffer_size<P, detail::cudaFindPair<T>>(count);
}
/**
@brief finds the index of the minimum element in a range
@tparam P execution policy type
@tparam I input iterator type
@tparam O comparator type
@param p execution policy object
@param first iterator to the beginning of the range
@param last iterator to the end of the range
@param idx solution index of the minimum element
@param op comparison function object
@param buf pointer to the buffer
The function launches kernels asynchronously to find
the smallest element in the range <tt>[first, last)</tt>
using the given comparator @c op.
You need to provide a buffer that holds at least
tf::cuda_min_element_buffer_size bytes for internal use.
The function is equivalent to a parallel execution of the following loop:
@code{.cpp}
if(first == last) {
return 0;
}
auto smallest = first;
for (++first; first != last; ++first) {
if (op(*first, *smallest)) {
smallest = first;
}
}
return std::distance(first, smallest);
@endcode
*/
template <typename P, typename I, typename O>
void cuda_min_element(P&& p, I first, I last, unsigned* idx, O op, void* buf) {
detail::cuda_min_element_loop(
p, first, std::distance(first, last), idx, op, buf
);
}
// ----------------------------------------------------------------------------
// cudaFlowCapturer::min_element
// ----------------------------------------------------------------------------
// Function: min_element
template <typename I, typename O>
cudaTask cudaFlowCapturer::min_element(I first, I last, unsigned* idx, O op) {
using T = typename std::iterator_traits<I>::value_type;
auto bufsz = cuda_min_element_buffer_size<cudaDefaultExecutionPolicy, T>(
std::distance(first, last)
);
return on([=, buf=MoC{cudaScopedDeviceMemory<std::byte>(bufsz)}]
(cudaStream_t stream) mutable {
cudaDefaultExecutionPolicy p(stream);
cuda_min_element(p, first, last, idx, op, buf.get().data());
});
}
// Function: min_element
template <typename I, typename O>
void cudaFlowCapturer::min_element(
cudaTask task, I first, I last, unsigned* idx, O op
) {
using T = typename std::iterator_traits<I>::value_type;
auto bufsz = cuda_min_element_buffer_size<cudaDefaultExecutionPolicy, T>(
std::distance(first, last)
);
on(task, [=, buf=MoC{cudaScopedDeviceMemory<std::byte>(bufsz)}]
(cudaStream_t stream) mutable {
cudaDefaultExecutionPolicy p(stream);
cuda_min_element(p, first, last, idx, op, buf.get().data());
});
}
// ----------------------------------------------------------------------------
// cudaFlow::min_element
// ----------------------------------------------------------------------------
// Function: min_element
template <typename I, typename O>
cudaTask cudaFlow::min_element(I first, I last, unsigned* idx, O op) {
return capture([=](cudaFlowCapturer& cap){
cap.make_optimizer<cudaLinearCapturing>();
cap.min_element(first, last, idx, op);
});
}
// Function: min_element
template <typename I, typename O>
void cudaFlow::min_element(
cudaTask task, I first, I last, unsigned* idx, O op
) {
capture(task, [=](cudaFlowCapturer& cap){
cap.make_optimizer<cudaLinearCapturing>();
cap.min_element(first, last, idx, op);
});
}
// ----------------------------------------------------------------------------
// cuda_max_element
// ----------------------------------------------------------------------------
/**
@brief queries the buffer size in bytes needed to call tf::cuda_max_element
@tparam P execution policy type
@tparam T value type
@param count number of elements to search
The function is used to decide the buffer size in bytes for calling
tf::cuda_max_element.
*/
template <typename P, typename T>
unsigned cuda_max_element_buffer_size(unsigned count) {
return cuda_reduce_buffer_size<P, detail::cudaFindPair<T>>(count);
}
/**
@brief finds the index of the maximum element in a range
@tparam P execution policy type
@tparam I input iterator type
@tparam O comparator type
@param p execution policy object
@param first iterator to the beginning of the range
@param last iterator to the end of the range
@param idx solution index of the maximum element
@param op comparison function object
@param buf pointer to the buffer
The function launches kernels asynchronously to find
the largest element in the range <tt>[first, last)</tt>
using the given comparator @c op.
You need to provide a buffer that holds at least
tf::cuda_max_element_buffer_size bytes for internal use.
The function is equivalent to a parallel execution of the following loop:
@code{.cpp}
if(first == last) {
return 0;
}
auto largest = first;
for (++first; first != last; ++first) {
if (op(*largest, *first)) {
largest = first;
}
}
return std::distance(first, largest);
@endcode
*/
template <typename P, typename I, typename O>
void cuda_max_element(P&& p, I first, I last, unsigned* idx, O op, void* buf) {
detail::cuda_max_element_loop(
p, first, std::distance(first, last), idx, op, buf
);
}
// ----------------------------------------------------------------------------
// cudaFlowCapturer::max_element
// ----------------------------------------------------------------------------
// Function: max_element
template <typename I, typename O>
cudaTask cudaFlowCapturer::max_element(I first, I last, unsigned* idx, O op) {
using T = typename std::iterator_traits<I>::value_type;
auto bufsz = cuda_max_element_buffer_size<cudaDefaultExecutionPolicy, T>(
std::distance(first, last)
);
return on([=, buf=MoC{cudaScopedDeviceMemory<std::byte>(bufsz)}]
(cudaStream_t stream) mutable {
cudaDefaultExecutionPolicy p(stream);
cuda_max_element(p, first, last, idx, op, buf.get().data());
});
}
// Function: max_element
template <typename I, typename O>
void cudaFlowCapturer::max_element(
cudaTask task, I first, I last, unsigned* idx, O op
) {
using T = typename std::iterator_traits<I>::value_type;
auto bufsz = cuda_max_element_buffer_size<cudaDefaultExecutionPolicy, T>(
std::distance(first, last)
);
on(task, [=, buf=MoC{cudaScopedDeviceMemory<std::byte>(bufsz)}]
(cudaStream_t stream) mutable {
cudaDefaultExecutionPolicy p(stream);
cuda_max_element(p, first, last, idx, op, buf.get().data());
});
}
// ----------------------------------------------------------------------------
// cudaFlow::max_element
// ----------------------------------------------------------------------------
// Function: max_element
template <typename I, typename O>
cudaTask cudaFlow::max_element(I first, I last, unsigned* idx, O op) {
return capture([=](cudaFlowCapturer& cap){
cap.make_optimizer<cudaLinearCapturing>();
cap.max_element(first, last, idx, op);
});
}
// Function: max_element
template <typename I, typename O>
void cudaFlow::max_element(
cudaTask task, I first, I last, unsigned* idx, O op
) {
capture(task, [=](cudaFlowCapturer& cap){
cap.make_optimizer<cudaLinearCapturing>();
cap.max_element(first, last, idx, op);
});
}
} // end of namespace tf -----------------------------------------------------

View File

@ -0,0 +1,283 @@
#pragma once
#include "../cuda_flow.hpp"
#include "../cuda_capturer.hpp"
#include "../cuda_meta.hpp"
/**
@file cuda_for_each.hpp
@brief cuda parallel-iteration algorithms include file
*/
namespace tf {
namespace detail {
/** @private */
template <typename P, typename I, typename C>
void cuda_for_each_loop(P&& p, I first, unsigned count, C c) {
using E = std::decay_t<P>;
unsigned B = (count + E::nv - 1) / E::nv;
cuda_kernel<<<B, E::nt, 0, p.stream()>>>(
[=] __device__ (auto tid, auto bid) {
auto tile = cuda_get_tile(bid, E::nv, count);
cuda_strided_iterate<E::nt, E::vt>([=](auto, auto j) {
c(*(first + tile.begin + j));
}, tid, tile.count());
});
}
/** @private */
template <typename P, typename I, typename C>
void cuda_for_each_index_loop(
P&& p, I first, I inc, unsigned count, C c
) {
using E = std::decay_t<P>;
unsigned B = (count + E::nv - 1) / E::nv;
cuda_kernel<<<B, E::nt, 0, p.stream()>>>(
[=]__device__(auto tid, auto bid) {
auto tile = cuda_get_tile(bid, E::nv, count);
cuda_strided_iterate<E::nt, E::vt>([=]__device__(auto, auto j) {
c(first + inc*(tile.begin+j));
}, tid, tile.count());
});
}
} // end of namespace detail -------------------------------------------------
// ----------------------------------------------------------------------------
// cuda standard algorithms: single_task/for_each/for_each_index
// ----------------------------------------------------------------------------
/**
@brief runs a callable asynchronously using one kernel thread
@tparam P execution policy type
@tparam C closure type
@param p execution policy
@param c closure to run by one kernel thread
The function launches a single kernel thread to run the given callable
through the stream in the execution policy object.
*/
template <typename P, typename C>
void cuda_single_task(P&& p, C c) {
cuda_kernel<<<1, 1, 0, p.stream()>>>(
[=]__device__(auto, auto) mutable { c(); }
);
}
/**
@brief performs asynchronous parallel iterations over a range of items
@tparam P execution policy type
@tparam I input iterator type
@tparam C unary operator type
@param p execution policy object
@param first iterator to the beginning of the range
@param last iterator to the end of the range
@param c unary operator to apply to each dereferenced iterator
This function is equivalent to a parallel execution of the following loop
on a GPU:
@code{.cpp}
for(auto itr = first; itr != last; itr++) {
c(*itr);
}
@endcode
*/
template <typename P, typename I, typename C>
void cuda_for_each(P&& p, I first, I last, C c) {
unsigned count = std::distance(first, last);
if(count == 0) {
return;
}
detail::cuda_for_each_loop(p, first, count, c);
}
/**
@brief performs asynchronous parallel iterations over
an index-based range of items
@tparam P execution policy type
@tparam I input index type
@tparam C unary operator type
@param p execution policy object
@param first index to the beginning of the range
@param last index to the end of the range
@param inc step size between successive iterations
@param c unary operator to apply to each index
This function is equivalent to a parallel execution of
the following loop on a GPU:
@code{.cpp}
// step is positive [first, last)
for(auto i=first; i<last; i+=step) {
c(i);
}
// step is negative [first, last)
for(auto i=first; i>last; i+=step) {
c(i);
}
@endcode
*/
template <typename P, typename I, typename C>
void cuda_for_each_index(P&& p, I first, I last, I inc, C c) {
if(is_range_invalid(first, last, inc)) {
TF_THROW("invalid range [", first, ", ", last, ") with inc size ", inc);
}
unsigned count = distance(first, last, inc);
if(count == 0) {
return;
}
detail::cuda_for_each_index_loop(p, first, inc, count, c);
}
// ----------------------------------------------------------------------------
// single_task
// ----------------------------------------------------------------------------
/** @private */
template <typename C>
__global__ void cuda_single_task(C callable) {
callable();
}
// ----------------------------------------------------------------------------
// cudaFlow
// ----------------------------------------------------------------------------
// Function: single_task
template <typename C>
cudaTask cudaFlow::single_task(C c) {
return kernel(1, 1, 0, cuda_single_task<C>, c);
}
// Function: single_task
template <typename C>
void cudaFlow::single_task(cudaTask task, C c) {
return kernel(task, 1, 1, 0, cuda_single_task<C>, c);
}
// Function: for_each
template <typename I, typename C>
cudaTask cudaFlow::for_each(I first, I last, C c) {
return capture([=](cudaFlowCapturer& cap) mutable {
cap.make_optimizer<cudaLinearCapturing>();
cap.for_each(first, last, c);
});
}
// Function: for_each_index
template <typename I, typename C>
cudaTask cudaFlow::for_each_index(I first, I last, I inc, C c) {
return capture([=](cudaFlowCapturer& cap) mutable {
cap.make_optimizer<cudaLinearCapturing>();
cap.for_each_index(first, last, inc, c);
});
}
// Function: for_each
template <typename I, typename C>
void cudaFlow::for_each(cudaTask task, I first, I last, C c) {
capture(task, [=](cudaFlowCapturer& cap) mutable {
cap.make_optimizer<cudaLinearCapturing>();
cap.for_each(first, last, c);
});
}
// Function: for_each_index
template <typename I, typename C>
void cudaFlow::for_each_index(cudaTask task, I first, I last, I inc, C c) {
capture(task, [=](cudaFlowCapturer& cap) mutable {
cap.make_optimizer<cudaLinearCapturing>();
cap.for_each_index(first, last, inc, c);
});
}
// ----------------------------------------------------------------------------
// cudaFlowCapturer
// ----------------------------------------------------------------------------
// Function: for_each
template <typename I, typename C>
cudaTask cudaFlowCapturer::for_each(I first, I last, C c) {
return on([=](cudaStream_t stream) mutable {
cudaDefaultExecutionPolicy p(stream);
cuda_for_each(p, first, last, c);
});
}
// Function: for_each_index
template <typename I, typename C>
cudaTask cudaFlowCapturer::for_each_index(I beg, I end, I inc, C c) {
return on([=] (cudaStream_t stream) mutable {
cudaDefaultExecutionPolicy p(stream);
cuda_for_each_index(p, beg, end, inc, c);
});
}
// Function: for_each
template <typename I, typename C>
void cudaFlowCapturer::for_each(cudaTask task, I first, I last, C c) {
on(task, [=](cudaStream_t stream) mutable {
cudaDefaultExecutionPolicy p(stream);
cuda_for_each(p, first, last, c);
});
}
// Function: for_each_index
template <typename I, typename C>
void cudaFlowCapturer::for_each_index(
cudaTask task, I beg, I end, I inc, C c
) {
on(task, [=] (cudaStream_t stream) mutable {
cudaDefaultExecutionPolicy p(stream);
cuda_for_each_index(p, beg, end, inc, c);
});
}
// Function: single_task
template <typename C>
cudaTask cudaFlowCapturer::single_task(C callable) {
return on([=] (cudaStream_t stream) mutable {
cudaDefaultExecutionPolicy p(stream);
cuda_single_task(p, callable);
});
}
// Function: single_task
template <typename C>
void cudaFlowCapturer::single_task(cudaTask task, C callable) {
on(task, [=] (cudaStream_t stream) mutable {
cudaDefaultExecutionPolicy p(stream);
cuda_single_task(p, callable);
});
}
} // end of namespace tf -----------------------------------------------------

View File

@ -0,0 +1,57 @@
#pragma once
#include "../cuda_error.hpp"
namespace tf {
// ----------------------------------------------------------------------------
// row-major matrix multiplication
// ----------------------------------------------------------------------------
template <typename T>
__global__ void cuda_matmul(
const T* A,
const T* B,
T* C,
size_t M,
size_t K,
size_t N
) {
__shared__ T A_tile[32][32];
__shared__ T B_tile[32][32];
size_t x = blockIdx.x * blockDim.x + threadIdx.x;
size_t y = blockIdx.y * blockDim.y + threadIdx.y;
T res = 0;
for(size_t k = 0; k < K; k += 32) {
if((threadIdx.x + k) < K && y < M) {
A_tile[threadIdx.y][threadIdx.x] = A[y * K + threadIdx.x + k];
}
else{
A_tile[threadIdx.y][threadIdx.x] = 0;
}
if((threadIdx.y + k) < K && x < N) {
B_tile[threadIdx.y][threadIdx.x] = B[(threadIdx.y + k) * N + x];
}
else{
B_tile[threadIdx.y][threadIdx.x] = 0;
}
__syncthreads();
for(size_t i = 0; i < 32; ++i) {
res += A_tile[threadIdx.y][i] * B_tile[i][threadIdx.x];
}
__syncthreads();
}
if(x < N && y < M) {
C[y * N + x] = res;
}
}
} // end of namespace tf ---------------------------------------------------------

Some files were not shown because too many files have changed in this diff Show More