diff --git a/cpp_to_py/common/db/Database.cpp b/cpp_to_py/common/db/Database.cpp index dda5b5f..e764480 100644 --- a/cpp_to_py/common/db/Database.cpp +++ b/cpp_to_py/common/db/Database.cpp @@ -20,7 +20,7 @@ Database::~Database() { void Database::load() { // ----- design related options ----- - if (setting.BookshelfAux != "" && setting.BookshelfPl != "") { + if (setting.BookshelfAux != "") { setting.Format = "bookshelf"; readBSAux(setting.BookshelfAux, setting.BookshelfPl); } diff --git a/cpp_to_py/common/io/file_bkshf_db.cpp b/cpp_to_py/common/io/file_bkshf_db.cpp index 61638c2..a23f00b 100644 --- a/cpp_to_py/common/io/file_bkshf_db.cpp +++ b/cpp_to_py/common/io/file_bkshf_db.cpp @@ -342,6 +342,12 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile) readBSLine(fs, tokens); fs.close(); + bool includePl = true; + if (plFile == "") { + logger.warning("No pl file specified. Try to find pl in aux file."); + includePl = false; + } + std::string fileNodes; std::string fileNets; std::string fileScl; @@ -376,6 +382,10 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile) logger.error("unrecognized file extension: %s", ext.c_str()); } } + if (includePl) { + logger.info("pl file %s is given.", plFile.c_str()); + filePl = plFile; + } // step 1: read floorplan, rows from: // scl - rows // step 2: read cell types from: @@ -399,16 +409,12 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile) readBSRoute(fileRoute); readBSShapes(fileShapes); readBSWts(fileWts); - // readBSPl ( filePl ); - readBSPl(plFile); + readBSPl(filePl); readBSScl(fileScl); } else if (bsData.format == "ispd2005") { readBSNets(fileNets); - // readBSRoute ( fileRoute ); - // readBSShapes( fileShapes); readBSWts(fileWts); - // readBSPl ( filePl ); - readBSPl(plFile); + readBSPl(filePl); readBSScl(fileScl); } diff --git a/cpp_to_py/common/io/file_lefdef_db.cpp b/cpp_to_py/common/io/file_lefdef_db.cpp index a756ed1..233f7ab 100644 --- a/cpp_to_py/common/io/file_lefdef_db.cpp +++ b/cpp_to_py/common/io/file_lefdef_db.cpp @@ -1477,10 +1477,10 @@ int readDefComponent(defrCallbackType_e c, defiComponent* co, defiUserData ud) { cell->unplace(); } else if (co->isPlaced()) { cell->place(co->placementX(), co->placementY(), co->placementOrient()); - if (celltype->cls == "CORE") { + if (celltype->cls == "CORE" || celltype->cls == "BLOCK") { cell->fixed(false); } else { - // Set all non-CORE cells as fixed cells + // Set all non-CORE yet non-BLOCK cells as fixed cells cell->fixed(true); } if (co->placementOrient() % 2 == 1) { diff --git a/cpp_to_py/gpudp/PyBindCppMain.cpp b/cpp_to_py/gpudp/PyBindCppMain.cpp index 0162fc7..68ed5c2 100644 --- a/cpp_to_py/gpudp/PyBindCppMain.cpp +++ b/cpp_to_py/gpudp/PyBindCppMain.cpp @@ -20,6 +20,7 @@ PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) { torch::Tensor, torch::Tensor, torch::Tensor, + torch::Tensor, float, float, float, @@ -34,6 +35,7 @@ PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) { .def("commit", &dp::DPTorchRawDB::commit) .def("rollback", &dp::DPTorchRawDB::rollback) .def("commit_from", &dp::DPTorchRawDB::commit_from) + .def("commit_from_partial", &dp::DPTorchRawDB::commit_from_partial) .def("get_curr_cposx", &dp::DPTorchRawDB::get_curr_cposx, py::return_value_policy::move) .def("get_curr_cposy", &dp::DPTorchRawDB::get_curr_cposy, py::return_value_policy::move) .def("get_curr_lposx", &dp::DPTorchRawDB::get_curr_lposx, py::return_value_policy::move) @@ -43,6 +45,7 @@ PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) { [](torch::Tensor node_lpos_init_, torch::Tensor node_size_, torch::Tensor node_weight_, + torch::Tensor is_macro_, torch::Tensor pin_rel_lpos_, torch::Tensor pin_id2node_id_, torch::Tensor pin_id2net_id_, @@ -66,6 +69,7 @@ PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) { return std::make_shared(node_lpos_init_, node_size_, node_weight_, + is_macro_, pin_rel_lpos_, pin_id2node_id_, pin_id2net_id_, diff --git a/cpp_to_py/gpudp/db/dp_torch.cpp b/cpp_to_py/gpudp/db/dp_torch.cpp index 5f89236..25f4441 100644 --- a/cpp_to_py/gpudp/db/dp_torch.cpp +++ b/cpp_to_py/gpudp/db/dp_torch.cpp @@ -8,6 +8,7 @@ namespace dp { DPTorchRawDB::DPTorchRawDB(torch::Tensor node_lpos_init_, torch::Tensor node_size_, torch::Tensor node_weight_, + torch::Tensor is_macro_, torch::Tensor pin_rel_lpos_, torch::Tensor pin_id2node_id_, torch::Tensor pin_id2net_id_, @@ -74,6 +75,7 @@ DPTorchRawDB::DPTorchRawDB(torch::Tensor node_lpos_init_, net_mask = net_mask_; node_weight = node_weight_; + is_macro = is_macro_; site_width = site_width_; row_height = row_height_; @@ -180,6 +182,16 @@ void DPTorchRawDB::commit_from(torch::Tensor x_, torch::Tensor y_) { .copy_(y_.index({torch::indexing::Slice(0, num_movable_nodes)})); } +void DPTorchRawDB::commit_from_partial(torch::Tensor x_, torch::Tensor y_) { + // commit external pos to new pos + x.index({torch::indexing::Slice(0, num_movable_nodes)}) + .data() + .copy_(x_.index({torch::indexing::Slice(0, num_movable_nodes)})); + y.index({torch::indexing::Slice(0, num_movable_nodes)}) + .data() + .copy_(y_.index({torch::indexing::Slice(0, num_movable_nodes)})); +} + torch::Tensor DPTorchRawDB::get_curr_cposx() { return x + node_size_x / 2; } torch::Tensor DPTorchRawDB::get_curr_cposy() { return y + node_size_y / 2; } torch::Tensor DPTorchRawDB::get_curr_lposx() { return x; } diff --git a/cpp_to_py/gpudp/db/dp_torch.h b/cpp_to_py/gpudp/db/dp_torch.h index 18551b4..774eebf 100644 --- a/cpp_to_py/gpudp/db/dp_torch.h +++ b/cpp_to_py/gpudp/db/dp_torch.h @@ -10,6 +10,7 @@ public: DPTorchRawDB(torch::Tensor node_lpos_init_, torch::Tensor node_size_, torch::Tensor node_weight_, + torch::Tensor is_macro_, torch::Tensor pin_rel_lpos_, torch::Tensor pin_id2node_id_, torch::Tensor pin_id2net_id_, @@ -35,6 +36,7 @@ public: void commit(); void rollback(); void commit_from(torch::Tensor x_, torch::Tensor y_); + void commit_from_partial(torch::Tensor x_, torch::Tensor y_); torch::Tensor get_curr_cposx(); torch::Tensor get_curr_cposy(); torch::Tensor get_curr_lposx(); @@ -48,6 +50,7 @@ public: torch::Tensor pin_rel_lpos; torch::Tensor node_weight; + torch::Tensor is_macro; torch::Tensor init_x; // original pos (keep it const except committing) torch::Tensor init_y; // original pos (keep it const except committing) diff --git a/cpp_to_py/gpudp/lg/greedy_legalize.cpp b/cpp_to_py/gpudp/lg/greedy_legalize.cpp index 8b5ee39..4140b6e 100644 --- a/cpp_to_py/gpudp/lg/greedy_legalize.cpp +++ b/cpp_to_py/gpudp/lg/greedy_legalize.cpp @@ -55,7 +55,7 @@ void distributeCells2Bins(const LegalizationData& db, num_legalized_nodes = num_movable_nodes; } for (int i = 0; i < num_legalized_nodes; i += 1) { - if (!db.is_dummy_fixed(i)) { + if (!db.is_mov_macro(i)) { int bin_id_x = (x[i] + node_size_x[i] / 2 - xl) / bin_size_x; int bin_id_y = (y[i] + node_size_y[i] / 2 - yl) / bin_size_y; @@ -87,7 +87,7 @@ void distributeFixedCells2Bins(const LegalizationData& db, std::vector>& bin_cells) { // one cell can be assigned to multiple bins for (int i = 0; i < num_nodes; i += 1) { - if (db.is_dummy_fixed(i) || i >= num_movable_nodes) { + if (db.is_mov_macro(i) || i >= num_movable_nodes) { int node_id = i; int bin_id_xl = std::max((int)floorDiv(x[node_id] - xl, bin_size_x, 0), 0); int bin_id_xh = std::min((int)ceilDiv((x[node_id] + node_size_x[node_id] - xl), bin_size_x, 0), num_bins_x); diff --git a/cpp_to_py/gpudp/lg/legalization_db.h b/cpp_to_py/gpudp/lg/legalization_db.h index 6339f5e..e9ff209 100644 --- a/cpp_to_py/gpudp/lg/legalization_db.h +++ b/cpp_to_py/gpudp/lg/legalization_db.h @@ -71,6 +71,7 @@ public: node2fence_region_map(at_db.node2fence_region_map.data_ptr()), net_mask(at_db.net_mask.data_ptr()), node_weight(at_db.node_weight.data_ptr()), + is_macro(at_db.is_macro.data_ptr()), xl(at_db.xl), xh(at_db.xh), yl(at_db.yl), @@ -95,6 +96,9 @@ public: const float* node_size_x; const float* node_size_y; + const float* node_weight; + const bool* is_macro; + const float* pin_offset_x; const float* pin_offset_y; @@ -111,7 +115,6 @@ public: const int* node2fence_region_map; const bool* net_mask; - const float* node_weight; /* chip info */ float xl; @@ -146,9 +149,8 @@ public: bin_size_x = (xh - xl) / num_bins_x_; bin_size_y = (yh - yl) / num_bins_y_; } - inline bool is_dummy_fixed(int node_id) const { - // DUMMY_FIXED_NUM_ROWS == 2 - return (node_id < num_movable_nodes && node_size_y[node_id] > (row_height * 2)); + inline bool is_mov_macro(int node_id) const { + return (node_id < num_movable_nodes && is_macro[node_id]); } inline float align2row(float y, float height) const { diff --git a/cpp_to_py/gpudp/lg/macro_legalize.cpp b/cpp_to_py/gpudp/lg/macro_legalize.cpp index 160eba2..f00ae39 100644 --- a/cpp_to_py/gpudp/lg/macro_legalize.cpp +++ b/cpp_to_py/gpudp/lg/macro_legalize.cpp @@ -24,7 +24,7 @@ bool check_macro_legality(LegalizationData& db, const std::vector& macros, float yh1 = yl1 + height1; float xh2 = xl2 + width2; float yh2 = yl2 + height2; - if (std::min(xh1, xh2) > std::max(xl1, xl2) && std::min(yh1, yh2) > std::max(yl1, yl2)) { + if (std::min(xh1, xh2) - std::max(xl1, xl2) > 1e-3 && std::min(yh1, yh2) - std::max(yl1, yl2) > 1e-3) { logger.error( "macro %d (%g, %g, %g, %g) var %d overlaps with macro %d " "(%g, %g, %g, %g) var %d, fixed: %d", @@ -273,7 +273,7 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) { // collect macros std::vector macros; for (int i = 0; i < db.num_movable_nodes; ++i) { - if (db.is_dummy_fixed(i)) { + if (db.is_mov_macro(i)) { // in some extreme case, some macros with 0 area should be ignored float area = db.node_size_x[i] * db.node_size_y[i]; if (area > 0) { @@ -281,7 +281,7 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) { } } } - logger.info("Macro legalization: regard %lu cells as dummy fixed (movable macros)", macros.size()); + logger.info("Macro legalization: regard %lu cells as movable macros", macros.size()); // in case there is no movable macros if (macros.empty()) { @@ -320,10 +320,7 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) { } }; - // first round rough legalization with Hannan grid for clusters - bool small_clusters_flag = true; - bool blocked_macros_flag = false; - roughLegalize(db, macros, fixed_macros, small_clusters_flag, blocked_macros_flag); + // 1) LP legalization, check displacement and legality auto displace = compute_displace(db, macros); logger.info("Macro displacement total %g, max %g, weighted total %g, max %g", displace.total_displace, @@ -331,9 +328,27 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) { displace.total_weighted_displace, displace.max_weighted_displace); bool legal = check_macro_legality(db, macros, true); + update_best(legal, displace); - // try Hannan grid legalization if still not legal + // 2) rough legalization with Hannan grid for clusters if (!legal) { + logger.warning("LP not legal, try roughLegalize."); + bool small_clusters_flag = true; + bool blocked_macros_flag = false; + roughLegalize(db, macros, fixed_macros, small_clusters_flag, blocked_macros_flag); + auto displace = compute_displace(db, macros); + logger.info("Macro displacement total %g, max %g, weighted total %g, max %g", + displace.total_displace, + displace.max_displace, + displace.total_weighted_displace, + displace.max_weighted_displace); + legal = check_macro_legality(db, macros, true); + update_best(legal, displace); + } + + // 3) try Hannan grid legalization if still not legal + if (!legal) { + logger.warning("Not legal, try hannanLegalize."); legal = hannanLegalize(db, macros, fixed_macros, 10); auto displace = compute_displace(db, macros); logger.info("Macro displacement total %g, max %g, weighted total %g, max %g", @@ -361,15 +376,22 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) { } } - logger.info("Align macros to site and rows"); - // align the lower left corner to row and site - for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) { - int node_id = macros[i]; - db.x[node_id] = db.align2site(db.x[node_id], db.node_size_x[node_id]); - db.y[node_id] = db.align2row(db.y[node_id], db.node_size_y[node_id]); - } + if (legal) { + logger.info("Align macros to site and rows"); + // align the lower left corner to row and site + for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) { + int node_id = macros[i]; + db.x[node_id] = db.align2site(db.x[node_id], db.node_size_x[node_id]); + db.y[node_id] = db.align2row(db.y[node_id], db.node_size_y[node_id]); + } - legal = check_macro_legality(db, macros, false); + legal = check_macro_legality(db, macros, false); + if (!legal) { + logger.error("Macro legalization failed after aligning to site and row"); + } + } else { + logger.error("Macro legalization failed"); + } return legal; } diff --git a/cpp_to_py/wa_wirelength_hpwl_cuda/wa_wirelength_hpwl_cuda.cpp b/cpp_to_py/wa_wirelength_hpwl_cuda/wa_wirelength_hpwl_cuda.cpp index 1965db2..aaf08d6 100644 --- a/cpp_to_py/wa_wirelength_hpwl_cuda/wa_wirelength_hpwl_cuda.cpp +++ b/cpp_to_py/wa_wirelength_hpwl_cuda/wa_wirelength_hpwl_cuda.cpp @@ -8,27 +8,6 @@ torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos, torch::Tensor net_mask, torch::Tensor hpwl_scale); -std::vector wa_wirelength_cuda(torch::Tensor node_pos, - torch::Tensor pin_id2node_id, - torch::Tensor pin_rel_cpos, - torch::Tensor node2pin_list, - torch::Tensor node2pin_list_end, - torch::Tensor hyperedge_list, - torch::Tensor hyperedge_list_end, - torch::Tensor net_mask, - float gamma, - bool deterministic); - -std::vector wa_wirelength_hpwl_cuda(torch::Tensor node_pos, - torch::Tensor pin_id2node_id, - torch::Tensor pin_rel_cpos, - torch::Tensor node2pin_list, - torch::Tensor node2pin_list_end, - torch::Tensor hyperedge_list, - torch::Tensor hyperedge_list_end, - torch::Tensor net_mask, - float gamma, - bool deterministic); std::vector wa_wirelength_masked_scale_hpwl_cuda(torch::Tensor node_pos, torch::Tensor pin_id2node_id, @@ -66,68 +45,6 @@ torch::Tensor masked_scale_hpwl_sum(torch::Tensor node_pos, node_pos, pin_id2node_id, pin_rel_cpos, hyperedge_list, hyperedge_list_end, net_mask, hpwl_scale); } -std::vector wa_wirelength(torch::Tensor node_pos, - torch::Tensor pin_id2node_id, - torch::Tensor pin_rel_cpos, - torch::Tensor node2pin_list, - torch::Tensor node2pin_list_end, - torch::Tensor hyperedge_list, - torch::Tensor hyperedge_list_end, - torch::Tensor net_mask, - float gamma, - bool deterministic) { - CHECK_INPUT(node_pos); - CHECK_INPUT(pin_id2node_id); - CHECK_INPUT(pin_rel_cpos); - CHECK_INPUT(node2pin_list); - CHECK_INPUT(node2pin_list_end); - CHECK_INPUT(hyperedge_list); - CHECK_INPUT(hyperedge_list_end); - CHECK_INPUT(net_mask); - - return wa_wirelength_cuda(node_pos, - pin_id2node_id, - pin_rel_cpos, - node2pin_list, - node2pin_list_end, - hyperedge_list, - hyperedge_list_end, - net_mask, - gamma, - deterministic); -} - -std::vector wa_wirelength_hpwl(torch::Tensor node_pos, - torch::Tensor pin_id2node_id, - torch::Tensor pin_rel_cpos, - torch::Tensor node2pin_list, - torch::Tensor node2pin_list_end, - torch::Tensor hyperedge_list, - torch::Tensor hyperedge_list_end, - torch::Tensor net_mask, - float gamma, - bool deterministic) { - CHECK_INPUT(node_pos); - CHECK_INPUT(pin_id2node_id); - CHECK_INPUT(pin_rel_cpos); - CHECK_INPUT(node2pin_list); - CHECK_INPUT(node2pin_list_end); - CHECK_INPUT(hyperedge_list); - CHECK_INPUT(hyperedge_list_end); - CHECK_INPUT(net_mask); - - return wa_wirelength_hpwl_cuda(node_pos, - pin_id2node_id, - pin_rel_cpos, - node2pin_list, - node2pin_list_end, - hyperedge_list, - hyperedge_list_end, - net_mask, - gamma, - deterministic); -} - std::vector wa_wirelength_masked_scale_hpwl(torch::Tensor node_pos, torch::Tensor pin_id2node_id, torch::Tensor pin_rel_cpos, @@ -164,8 +81,6 @@ std::vector wa_wirelength_masked_scale_hpwl(torch::Tensor node_po PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) { m.def("masked_scale_hpwl_sum", &masked_scale_hpwl_sum, "calculate the sum of scaled HPWL"); - m.def("merged_forward_backward", &wa_wirelength, "calculate WA wirelength and pin grad"); - m.def("merged_forward_backward_with_hpwl", &wa_wirelength_hpwl, "calculate WA wirelength, pin grad and hpwl"); m.def("merged_forward_backward_with_masked_scale_hpwl", &wa_wirelength_masked_scale_hpwl, "calculate WA wirelength, pin grad and the scaled hpwl"); diff --git a/cpp_to_py/wa_wirelength_hpwl_cuda/wa_wirelength_hpwl_cuda_kernel.cu b/cpp_to_py/wa_wirelength_hpwl_cuda/wa_wirelength_hpwl_cuda_kernel.cu index 6e18839..2a605ce 100644 --- a/cpp_to_py/wa_wirelength_hpwl_cuda/wa_wirelength_hpwl_cuda_kernel.cu +++ b/cpp_to_py/wa_wirelength_hpwl_cuda/wa_wirelength_hpwl_cuda_kernel.cu @@ -74,134 +74,6 @@ __global__ void masked_scale_hpwl_cuda_kernel( } } -__global__ void wa_wirelength_kernel( - const torch::PackedTensorAccessor32 pin_pos, - const torch::PackedTensorAccessor32 hyperedge_list, - const torch::PackedTensorAccessor32 hyperedge_list_end, - const torch::PackedTensorAccessor32 net_mask, - torch::PackedTensorAccessor32 partial_wa_wl, - torch::PackedTensorAccessor32 pin_grad, - int num_nets, - float inv_gamma) { - const int index = blockIdx.x * blockDim.x + threadIdx.x; - const int i = index >> 1; // net index - if (i < num_nets && net_mask[i]) { - const int c = index & 1; // channel index - int64_t start_idx = 0; - if (i != 0) { - start_idx = hyperedge_list_end[i - 1]; - } - int64_t end_idx = hyperedge_list_end[i]; - if (end_idx != start_idx) { - int64_t pin_id = hyperedge_list[start_idx]; - float x_min = pin_pos[pin_id][c]; - float x_max = pin_pos[pin_id][c]; - for (int64_t idx = start_idx + 1; idx < end_idx; idx++) { - float xx = pin_pos[hyperedge_list[idx]][c]; - x_min = min(xx, x_min); - x_max = max(xx, x_max); - } - - float xexp_x_sum = 0; - float xexp_nx_sum = 0; - float exp_x_sum = 0; - float exp_nx_sum = 0; - - for (int64_t idx = start_idx; idx < end_idx; idx++) { - float xx = pin_pos[hyperedge_list[idx]][c]; - float exp_x = exp((xx - x_max) * inv_gamma); - float exp_nx = exp((x_min - xx) * inv_gamma); - - xexp_x_sum += xx * exp_x; - xexp_nx_sum += xx * exp_nx; - exp_x_sum += exp_x; - exp_nx_sum += exp_nx; - } - - float wl = xexp_x_sum / exp_x_sum - xexp_nx_sum / exp_nx_sum; - partial_wa_wl[i][c] = wl; - - float b_x = inv_gamma / (exp_x_sum); - float a_x = (1.0 - b_x * xexp_x_sum) / exp_x_sum; - float b_nx = -inv_gamma / (exp_nx_sum); - float a_nx = (1.0 - b_nx * xexp_nx_sum) / exp_nx_sum; - - for (int64_t idx = start_idx; idx < end_idx; idx++) { - float xx = pin_pos[hyperedge_list[idx]][c]; - float exp_x = exp((xx - x_max) * inv_gamma); - float exp_nx = exp((x_min - xx) * inv_gamma); - - pin_grad[hyperedge_list[idx]][c] = (a_x + b_x * xx) * exp_x - (a_nx + b_nx * xx) * exp_nx; - } - } - } -} - -__global__ void wa_wirelength_hpwl_kernel( - const torch::PackedTensorAccessor32 pin_pos, - const torch::PackedTensorAccessor32 hyperedge_list, - const torch::PackedTensorAccessor32 hyperedge_list_end, - const torch::PackedTensorAccessor32 net_mask, - torch::PackedTensorAccessor32 partial_wa_wl, - torch::PackedTensorAccessor32 partial_hpwl, - torch::PackedTensorAccessor32 pin_grad, - int num_nets, - float inv_gamma) { - const int index = blockIdx.x * blockDim.x + threadIdx.x; - const int i = index >> 1; // net index - if (i < num_nets && net_mask[i]) { - const int c = index & 1; // channel index - int64_t start_idx = 0; - if (i != 0) { - start_idx = hyperedge_list_end[i - 1]; - } - int64_t end_idx = hyperedge_list_end[i]; - if (end_idx != start_idx) { - int64_t pin_id = hyperedge_list[start_idx]; - float x_min = pin_pos[pin_id][c]; - float x_max = pin_pos[pin_id][c]; - for (int64_t idx = start_idx + 1; idx < end_idx; idx++) { - float xx = pin_pos[hyperedge_list[idx]][c]; - x_min = min(xx, x_min); - x_max = max(xx, x_max); - } - partial_hpwl[i][c] = abs(x_max - x_min); - - float xexp_x_sum = 0; - float xexp_nx_sum = 0; - float exp_x_sum = 0; - float exp_nx_sum = 0; - - for (int64_t idx = start_idx; idx < end_idx; idx++) { - float xx = pin_pos[hyperedge_list[idx]][c]; - float exp_x = exp((xx - x_max) * inv_gamma); - float exp_nx = exp((x_min - xx) * inv_gamma); - - xexp_x_sum += xx * exp_x; - xexp_nx_sum += xx * exp_nx; - exp_x_sum += exp_x; - exp_nx_sum += exp_nx; - } - - float wl = xexp_x_sum / exp_x_sum - xexp_nx_sum / exp_nx_sum; - partial_wa_wl[i][c] = wl; - - float b_x = inv_gamma / (exp_x_sum); - float a_x = (1.0 - b_x * xexp_x_sum) / exp_x_sum; - float b_nx = -inv_gamma / (exp_nx_sum); - float a_nx = (1.0 - b_nx * xexp_nx_sum) / exp_nx_sum; - - for (int64_t idx = start_idx; idx < end_idx; idx++) { - float xx = pin_pos[hyperedge_list[idx]][c]; - float exp_x = exp((xx - x_max) * inv_gamma); - float exp_nx = exp((x_min - xx) * inv_gamma); - - pin_grad[hyperedge_list[idx]][c] = (a_x + b_x * xx) * exp_x - (a_nx + b_nx * xx) * exp_nx; - } - } - } -} - __global__ void wa_wirelength_masked_scale_hpwl_kernel( const torch::PackedTensorAccessor32 pin_pos, const torch::PackedTensorAccessor32 hyperedge_list, @@ -296,57 +168,6 @@ void calc_node_grad_cuda(torch::Tensor node_grad, } } -std::vector wa_wirelength_cuda(torch::Tensor node_pos, - torch::Tensor pin_id2node_id, - torch::Tensor pin_rel_cpos, - torch::Tensor node2pin_list, - torch::Tensor node2pin_list_end, - torch::Tensor hyperedge_list, - torch::Tensor hyperedge_list_end, - torch::Tensor net_mask, - float gamma, - bool deterministic) { - cudaSetDevice(node_pos.get_device()); - auto stream = at::cuda::getCurrentCUDAStream(); - - const auto num_nodes = node_pos.size(0); - const auto num_pins = pin_id2node_id.size(0); - const auto num_nets = hyperedge_list_end.size(0); - const auto num_channels = 2; // x, y - - auto pin_pos = pin_rel_cpos.clone(); // pin - auto partial_wa_wl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device())); - auto pin_grad = torch::zeros({num_pins, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device())); - - const int threads = 128; - const int blocks = (num_pins * 2 + threads - 1) / threads; - - node_pos_to_pin_pos_cuda_kernel<<>>( - node_pos.packed_accessor32(), - pin_id2node_id.packed_accessor32(), - pin_pos.packed_accessor32(), - num_pins); - - const int threads2 = 128; - const int blocks2 = (num_nets * 2 + threads2 - 1) / threads2; - - float inv_gamma = 1 / gamma; - wa_wirelength_kernel<<>>( - pin_pos.packed_accessor32(), - hyperedge_list.packed_accessor32(), - hyperedge_list_end.packed_accessor32(), - net_mask.packed_accessor32(), - partial_wa_wl.packed_accessor32(), - pin_grad.packed_accessor32(), - num_nets, - inv_gamma); - - auto node_grad = torch::zeros({num_nodes, num_channels}, torch::dtype(pin_grad.dtype()).device(pin_grad.device())); - calc_node_grad_cuda( - node_grad, pin_id2node_id, pin_grad, node2pin_list, node2pin_list_end, num_nodes, deterministic); - - return {partial_wa_wl, node_grad}; -} torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos, torch::Tensor pin_id2node_id, @@ -391,60 +212,6 @@ torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos, return total_hpwl; } -std::vector wa_wirelength_hpwl_cuda(torch::Tensor node_pos, - torch::Tensor pin_id2node_id, - torch::Tensor pin_rel_cpos, - torch::Tensor node2pin_list, - torch::Tensor node2pin_list_end, - torch::Tensor hyperedge_list, - torch::Tensor hyperedge_list_end, - torch::Tensor net_mask, - float gamma, - bool deterministic) { - cudaSetDevice(node_pos.get_device()); - auto stream = at::cuda::getCurrentCUDAStream(); - - const auto num_nodes = node_pos.size(0); - const auto num_pins = pin_id2node_id.size(0); - const auto num_nets = hyperedge_list_end.size(0); - const auto num_channels = 2; // x, y - - auto pin_pos = pin_rel_cpos.clone(); // pin - auto partial_wa_wl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device())); - auto partial_hpwl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device())); - auto pin_grad = torch::zeros({num_pins, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device())); - - const int threads = 128; - const int blocks = (num_pins * 2 + threads - 1) / threads; - - node_pos_to_pin_pos_cuda_kernel<<>>( - node_pos.packed_accessor32(), - pin_id2node_id.packed_accessor32(), - pin_pos.packed_accessor32(), - num_pins); - - const int threads2 = 128; - const int blocks2 = (num_nets * 2 + threads2 - 1) / threads2; - - float inv_gamma = 1 / gamma; - wa_wirelength_hpwl_kernel<<>>( - pin_pos.packed_accessor32(), - hyperedge_list.packed_accessor32(), - hyperedge_list_end.packed_accessor32(), - net_mask.packed_accessor32(), - partial_wa_wl.packed_accessor32(), - partial_hpwl.packed_accessor32(), - pin_grad.packed_accessor32(), - num_nets, - inv_gamma); - - auto node_grad = torch::zeros({num_nodes, num_channels}, torch::dtype(pin_grad.dtype()).device(pin_grad.device())); - calc_node_grad_cuda( - node_grad, pin_id2node_id, pin_grad, node2pin_list, node2pin_list_end, num_nodes, deterministic); - - return {partial_wa_wl, node_grad, partial_hpwl}; -} - std::vector wa_wirelength_masked_scale_hpwl_cuda(torch::Tensor node_pos, torch::Tensor pin_id2node_id, torch::Tensor pin_rel_cpos, diff --git a/data/download_data.sh b/data/download_data.sh index 46ec9e7..fdaed6b 100755 --- a/data/download_data.sh +++ b/data/download_data.sh @@ -7,6 +7,18 @@ tar xvzf ispd2005.tar.gz rm -rf ispd2005.tar.gz mv ispd2005/ raw/ +echo "=== Downloading ispd2006" +wget --no-check-certificate "https://mycuhk-my.sharepoint.com/:u:/g/personal/1155136644_link_cuhk_edu_hk/EYBauANXekFAn1nlKRtec8YBecrfXNmocajWhqNfKWhRvA?e=wL1Z5z&download=1" -O ispd2006.tar.gz +tar xvzf ispd2006.tar.gz +rm -rf ispd2006.tar.gz +mv ispd2006/ raw/ + +echo "=== Downloading mms" +wget --no-check-certificate "https://mycuhk-my.sharepoint.com/:u:/g/personal/1155136644_link_cuhk_edu_hk/EQpwvzotaWBGlIm9zpOfIL4B_DRFb5jNOQ5mKHLd7yrByw?e=ovrb3e&download=1" -O mms.tar.gz +tar xvzf mms.tar.gz +rm -rf mms.tar.gz +mv mms/ raw/ + echo "=== Downloading ispd2015 ===" wget --no-check-certificate "https://mycuhk-my.sharepoint.com/:u:/g/personal/1155136644_link_cuhk_edu_hk/Ea4YjKNvi-9CnekS41Pw-GgBEhIRNnp6AhMDU9_xElLjNA?e=YSUMhQ&download=1" -O ispd2015.tar.gz tar xvzf ispd2015.tar.gz diff --git a/main.py b/main.py index a86a9dc..23c5f5e 100644 --- a/main.py +++ b/main.py @@ -22,21 +22,19 @@ def get_option(): parser.add_argument('--wa_coeff', type=float, default=4.0, help='wa coeff') parser.add_argument('--num_bin_x', type=int, default=512, help='#binX for density function') parser.add_argument('--num_bin_y', type=int, default=512, help='#binY for density function') - parser.add_argument('--threshold', type=float, default=4.0, help='normalized node area threshold for using naive mode') parser.add_argument('--density_weight', type=float, default=8e-5, help='the weight of density loss') parser.add_argument('--density_weight_coef', type=float, default=1.05, help='the ratio of density_weight') - parser.add_argument('--use_init_density_weight', type=str2bool, default=True, help='enable dynamic initialization of density_weight') parser.add_argument('--target_density', type=float, default=1.0, help='placement target density') parser.add_argument('--use_filler', type=str2bool, default=True, help='placement filler') parser.add_argument('--noise_ratio', type=float, default=0.025, help='noise ratio for initialization') parser.add_argument('--ignore_net_degree', type=int, default=100, help='threshold of net degree to ignore in wirelength calculation') - parser.add_argument('--scale_design', type=str2bool, default=False, help='normalize die area') parser.add_argument('--use_eplace_nesterov', type=str2bool, default=True, help='enable eplace nesterov optimizer') parser.add_argument('--clamp_node', type=str2bool, default=True, help='enable eplace node clamp trick') parser.add_argument('--use_precond', type=str2bool, default=True, help='apply precond') parser.add_argument('--stop_overflow', type=float, default=0.07, help='stop overflow in scheduler') parser.add_argument('--enable_skip_update', type=str2bool, default=True, help='enable skip update') - parser.add_argument("--loss_type", type=str, default="direct", help="loss type") + parser.add_argument('--enable_sample_force', type=str2bool, default=True, help='enable sample force') + parser.add_argument("--mixed_size", type=str2bool, default=False, help="enable mixed size placement") # global routing params parser.add_argument('--use_cell_inflate', type=str2bool, default=False, help='use cell inflation') diff --git a/src/__init__.py b/src/__init__.py index bba3f17..323def4 100644 --- a/src/__init__.py +++ b/src/__init__.py @@ -5,6 +5,6 @@ from .evaluator import * from .initializer import * from .nesterov_optimizer import NesterovOptimizer from .param_scheduler import ParamScheduler -from .detail_placement import detail_placement_main +from .detail_placement import detail_placement_main, macro_legalization_main from .run_placement_nesterov import run_placement_main_nesterov from .run_placement import run_placement_main \ No newline at end of file diff --git a/src/calculator.py b/src/calculator.py index 2c1659c..36aceb4 100644 --- a/src/calculator.py +++ b/src/calculator.py @@ -1,16 +1,6 @@ import torch from .param_scheduler import ParamScheduler -from .core import merged_wl_loss_grad, WAWirelengthLoss, WAWirelengthLossAndHPWL - - -def calc_loss(wl_loss, density_loss, ps, args): - if args.loss_type == "weighted_sum": - loss = (wl_loss + ps.density_weight * density_loss) / (1 + ps.density_weight) - elif args.loss_type == "direct": - loss = wl_loss + ps.density_weight * density_loss - else: - raise NotImplementedError("Loss type not defined") - return loss +from .core import merged_wl_loss_grad def apply_precond(mov_node_pos: torch.Tensor, ps: ParamScheduler, args): @@ -38,6 +28,8 @@ def calc_obj_and_grad( mov_node_pos = constraint_fn(mov_node_pos) conn_node_pos = mov_node_pos[mov_lhs:mov_rhs, ...] conn_node_pos = torch.cat([conn_node_pos, conn_fix_node_pos], dim=0) + + assert merged_forward_backward if merged_forward_backward: if mov_node_pos.grad is not None: mov_node_pos.grad.zero_() @@ -81,62 +73,11 @@ def calc_obj_and_grad( ) mov_node_pos.grad += node_grad_by_density * ps.density_weight + if ps.zero_macro_grad: + mov_node_pos.grad[mov_lhs:mov_rhs].masked_fill_(data.is_mov_macro[mov_lhs:mov_rhs].unsqueeze(1), 0) grad = apply_precond(mov_node_pos, ps, args) loss = wl_loss + ps.density_weight * density_loss - else: - if mov_node_pos.grad is not None: - mov_node_pos.grad.zero_() - else: - mov_node_pos.grad = torch.zeros_like(mov_node_pos).detach() - wl_loss = WAWirelengthLoss.apply( - conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos, - data.node2pin_list, data.node2pin_list_end, - data.hyperedge_list, data.hyperedge_list_end, data.net_mask, - ps.wa_coeff, args.deterministic - ) - density_loss, _ = density_map_layer( - mov_node_pos, mov_node_size, init_density_map, calc_overflow=False - ) - loss = calc_loss(wl_loss, density_loss, ps, args) - loss.backward() - grad = apply_precond(mov_node_pos, ps, args) + return loss, grad - -def calc_grad( - optimizer: torch.optim.Optimizer, mov_node_pos: torch.Tensor, wl_loss, density_loss -): - optimizer.zero_grad(set_to_none=False) - wl_loss.backward(retain_graph=True) - wl_grad = mov_node_pos.grad.detach().clone() - optimizer.zero_grad(set_to_none=False) - density_loss.backward(retain_graph=True) - density_grad = mov_node_pos.grad.detach().clone() - optimizer.zero_grad(set_to_none=False) - return wl_grad, density_grad - - -def fast_optimization( - mov_node_pos, trunc_node_pos_fn, mov_lhs, mov_rhs, conn_fix_node_pos, - density_map_layer, mov_node_size, init_density_map, ps, data, args -): - mov_node_pos = trunc_node_pos_fn(mov_node_pos) - conn_node_pos = mov_node_pos[mov_lhs:mov_rhs, ...] - conn_node_pos = torch.cat( - [conn_node_pos, conn_fix_node_pos], dim=0 - ) - wl_loss, hpwl = WAWirelengthLossAndHPWL.apply( - conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos, - data.node2pin_list, data.node2pin_list_end, - data.hyperedge_list, data.hyperedge_list_end, data.net_mask, - ps.wa_coeff, data.hpwl_scale, args.deterministic - ) - density_loss, overflow = density_map_layer( - mov_node_pos, mov_node_size, init_density_map - ) - loss = calc_loss(wl_loss, density_loss, ps, args) - loss.backward() - apply_precond(mov_node_pos, ps, args) - # calculate objective (hpwl, overflow) - return hpwl.detach(), overflow.detach(), mov_node_pos \ No newline at end of file diff --git a/src/core/__init__.py b/src/core/__init__.py index 5f03a5a..063944d 100644 --- a/src/core/__init__.py +++ b/src/core/__init__.py @@ -1,4 +1,4 @@ from .flute import Flute, get_flute_wl from .electronic_density_layer import ElectronicDensityLayer -from .wa_wirelength_hpwl import WAWirelengthLossAndHPWL, WAWirelengthLoss, masked_scale_hpwl, merged_wl_loss_grad +from .wa_wirelength_hpwl import masked_scale_hpwl, merged_wl_loss_grad from .route_force import get_route_force, run_gr_and_fft, run_gr_and_fft_main, route_inflation, route_inflation_roll_back \ No newline at end of file diff --git a/src/core/electronic_density_layer.py b/src/core/electronic_density_layer.py index 2641c3b..617db83 100644 --- a/src/core/electronic_density_layer.py +++ b/src/core/electronic_density_layer.py @@ -262,35 +262,6 @@ class ElectronicDensityLayer(torch.nn.Module): return node_weight - def get_density_map_naive( - self, - node_pos, - node_size, - init_density_map=None, - ): - node_weight = node_size.new_ones(node_pos.shape[0]) - if init_density_map is None: - node_pos.new_zeros(self.num_bin_x, self.num_bin_y) - aux_mat = init_density_map.clone() - num_nodes = node_pos.shape[0] - density_map = density_map_cuda.forward_naive( - node_pos, - node_size, - node_weight, - self.unit_len, - aux_mat, - self.num_bin_x, - self.num_bin_y, - num_nodes, - -1.0, - -1.0, - 1e-4, - False, - self.deterministic, - ) - - return density_map - def direct_calc_overflow( self, node_pos, diff --git a/src/core/macro_legalization.py b/src/core/macro_legalization.py new file mode 100644 index 0000000..0c96dad --- /dev/null +++ b/src/core/macro_legalization.py @@ -0,0 +1,947 @@ +# We follow the work [1] and [2] to implement the macro legalizer. +# [1] Cong, Jason, and Min Xie. "A robust mixed-size legalization and detailed placement algorithm." IEEE TCAD 2008. +# [2] Moffitt, M. D., Ng, A. N., Markov, I. L., & Pollack, M. E. "Constraint-driven floorplan repair." ACM TODAES 2008. + +# NOTE: Known Issue: Adding graph edge constraint in LP is very slow when num_macros is large. +# Because pulp lib use Python OrderDict to store all constraints, when #Constraints +# is large, the performance is bad. +# Re-write the LP in C++ may achieve some speed up. +# TODO: 1) macro spreading for routability optimization +# 2) graph pruning for speed up + +import numpy as np +import numba as nb +import logging +import pulp as pl +import igraph as ig + + +pulp_logger = logging.getLogger('pulp') +pulp_logger.setLevel(logging.INFO) +use_numba_parallel = False + +@nb.jit(cache=True, parallel=use_numba_parallel) +def check_macro_legality(macro_pos, macro_size, macro_fixed, die_info, check_all=True): + num_macros = macro_pos.shape[0] + # overlap = np.zeros((num_macros, num_macros), dtype=np.bool8) + legal = True + for i in nb.prange(num_macros): + lx_i = macro_pos[i][0] - macro_size[i][0] / 2 + ly_i = macro_pos[i][1] - macro_size[i][1] / 2 + hx_i = macro_pos[i][0] + macro_size[i][0] / 2 + hy_i = macro_pos[i][1] + macro_size[i][1] / 2 + for j in nb.prange(num_macros): + if i >= j: + continue + lx_j = macro_pos[j][0] - macro_size[j][0] / 2 + ly_j = macro_pos[j][1] - macro_size[j][1] / 2 + hx_j = macro_pos[j][0] + macro_size[j][0] / 2 + hy_j = macro_pos[j][1] + macro_size[j][1] / 2 + if min(hx_i, hx_j) - max(lx_i, lx_j) > 1e-3 and min(hy_i, hy_j) - max(ly_i, ly_j) > 1e-3: + # overlap[i][j] = True + # overlap[j][i] = True + legal = False + if not check_all and not legal: + return legal + print("Macro", i, "and Macro", j, "Overlap.") + + return legal + + + +@nb.jit(nopython=True, nogil=True, cache=True, parallel=use_numba_parallel) +def constraint_graph_construction( + macro_pos, macro_size, macro_fixed, die_info, prune=True +): + num_macros = macro_pos.shape[0] + macro_lpos = macro_pos - macro_size / 2 + edge_type = np.zeros((num_macros, num_macros), dtype=np.int8) # 0: x, 1: y, -1: None + edge_dist_x = np.zeros((num_macros, num_macros), dtype=np.float32) + edge_dist_y = np.zeros((num_macros, num_macros), dtype=np.float32) + edge_type[:, :] = -1 + for i in nb.prange(num_macros): + for j in nb.prange(num_macros): + if i >= j: + continue + # Detect x/y order + if macro_pos[i][0] <= macro_pos[j][0]: + # x_i -> x_j + x_order = 0 + else: + # x_j -> x_i + x_order = 1 + if macro_pos[i][1] <= macro_pos[j][1]: + # y_i -> y_j + y_order = 0 + else: + # y_j -> y_i + y_order = 1 + + # Calculate displacement + lx_i = macro_pos[i][0] - macro_size[i][0] / 2 + lx_j = macro_pos[j][0] - macro_size[j][0] / 2 + hx_i = macro_pos[i][0] + macro_size[i][0] / 2 + hx_j = macro_pos[j][0] + macro_size[j][0] / 2 + if x_order == 0: + dist_x = lx_j - hx_i + else: + dist_x = lx_i - hx_j + + ly_i = macro_pos[i][1] - macro_size[i][1] / 2 + ly_j = macro_pos[j][1] - macro_size[j][1] / 2 + hy_i = macro_pos[i][1] + macro_size[i][1] / 2 + hy_j = macro_pos[j][1] + macro_size[j][1] / 2 + if y_order == 0: + dist_y = ly_j - hy_i + else: + dist_y = ly_i - hy_j + + # dist martix is undirected and symmetric + edge_dist_x[i][j] = dist_x + edge_dist_x[j][i] = dist_x + edge_dist_y[i][j] = dist_y + edge_dist_y[j][i] = dist_y + + # Determine the edge type (horizontal or vertical) + if dist_x >= 0 and dist_y >= 0: + # non-overlap + edge_type[i][j] = 0 if dist_x >= dist_y else 1 + elif dist_x >= 0 and dist_y < 0: + # y projection overlap + edge_type[i][j] = 0 + elif dist_x < 0 and dist_y >= 0: + # x projection overlap + edge_type[i][j] = 1 + elif dist_x < 0 and dist_y < 0: + # overlap + edge_type[i][j] = 0 if dist_x >= dist_y else 1 + + # Prune edges between objects without x/y projection overlap + if prune: + if edge_type[i][j] == 0 and not (ly_i <= hy_j and ly_j <= hy_i): + edge_type[i][j] = -1 + if edge_type[i][j] == 1 and not (lx_i <= hx_j and lx_j <= hx_i): + edge_type[i][j] = -1 + + # Make sure edge orders + if edge_type[i][j] == 0 and x_order == 1: + edge_type[i][j] = -1 + edge_type[j][i] = 0 + elif edge_type[i][j] == 1 and y_order == 1: + edge_type[i][j] = -1 + edge_type[j][i] = 1 + + return edge_type, edge_dist_x, edge_dist_y + + +@nb.jit(nopython=True, nogil=True, cache=True, parallel=use_numba_parallel) +def initialize_xy_adj_weight( + edge_type, adj_matrix, weight_matrix, macro_size, num_macros, num_nodes, s_id, t_id +): + adj_matrix[s_id, :num_macros, :] = 1 + adj_matrix[:num_macros, t_id, :] = 1 + + for i in nb.prange(num_nodes): + for j in nb.prange(num_nodes): + if i == j: + continue + if i < num_macros and j < num_macros: + if edge_type[i][j] == 0: + # horizontal + adj_matrix[i][j][0] = 1 + elif edge_type[i][j] == 1: + adj_matrix[i][j][1] = 1 + if i >= num_macros: + weight_matrix[i][j][0] = np.divide(macro_size[j][0], 2) + weight_matrix[i][j][1] = np.divide(macro_size[j][1], 2) + elif j >= num_macros: + weight_matrix[i][j][0] = np.divide(macro_size[i][0], 2) + weight_matrix[i][j][1] = np.divide(macro_size[i][1], 2) + else: + weight_matrix[i][j][0] = np.divide(np.add(macro_size[i][0], macro_size[j][0]), 2) + weight_matrix[i][j][1] = np.divide(np.add(macro_size[i][1], macro_size[j][1]), 2) + + +@nb.jit(nopython=True, nogil=True, cache=True) +def compute_L_value( + macro_pos, macro_fixed, axis, topo_order_out, L, affected_L, adj_matrix, weight_matrix, die_ll, s_id, num_macros +): + for i in topo_order_out: + if affected_L[i, axis] == 0: + continue + if i < num_macros: + if macro_fixed[i]: + L[i, axis] = macro_pos[i, axis] + continue + if i == s_id: + L[i, axis] = die_ll[axis] + else: + is_preds = adj_matrix[:, i, axis] + for j, is_pred in enumerate(is_preds): + if is_pred == 0 or i == j: + continue + # j -> i + L[i, axis] = max(L[j, axis] + weight_matrix[j][i][axis], L[i, axis]) + + +@nb.jit(nopython=True, nogil=True, cache=True) +def compute_R_value( + macro_pos, macro_fixed, axis, topo_order_in, R, affected_R, adj_matrix, weight_matrix, die_ur, t_id, num_macros +): + for i in topo_order_in: + if affected_R[i, axis] == 0: + continue + if i < num_macros: + if macro_fixed[i]: + R[i, axis] = macro_pos[i, axis] + continue + if i == t_id: + R[i, axis] = die_ur[axis] + else: + is_succs = adj_matrix[i, :, axis] + for j, is_succ in enumerate(is_succs): + if is_succ == 0 or i == j: + continue + # i -> j + R[i, axis] = min(R[j, axis] - weight_matrix[i][j][axis], R[i, axis]) + + +def propagate_L_R( + g, macro_pos, macro_fixed, affected_L, affected_R, adj_matrix, weight_matrix, + L, R, die_ll, die_ur, s_id, t_id, num_macros, edges_pair=None +): + topo_order_out_X = nb.typed.List(g[0].topological_sorting(mode="out")) + topo_order_in_X = nb.typed.List(g[0].topological_sorting(mode="in")) + topo_order_out_Y = nb.typed.List(g[1].topological_sorting(mode="out")) + topo_order_in_Y = nb.typed.List(g[1].topological_sorting(mode="in")) + + if edges_pair: + axis_del, (u_del, v_del), axis_add, (u_add, v_add) = edges_pair + affected_L[:, :] = False + affected_R[:, :] = False + # In the old graph, u may not be topologically <= v + bfs_order, _, _ = g[axis_del].bfs(u_del, mode='out') + affected_L[np.array(bfs_order), axis_del] = True + bfs_order, _, _ = g[axis_del].bfs(v_del, mode='out') + affected_L[np.array(bfs_order), axis_del] = True + bfs_order, _, _ = g[axis_del].bfs(u_del, mode='in') + affected_R[np.array(bfs_order), axis_del] = True + bfs_order, _, _ = g[axis_del].bfs(v_del, mode='in') + affected_R[np.array(bfs_order), axis_del] = True + # In the new graph, u -> v, so u should be topologically <= v + bfs_order, _, _ = g[axis_add].bfs(u_add, mode='out') + affected_L[np.array(bfs_order), axis_add] = True + bfs_order, _, _ = g[axis_add].bfs(v_add, mode='in') + affected_R[np.array(bfs_order), axis_add] = True + + L[affected_L] = -np.inf + R[affected_R] = np.inf + compute_L_value(macro_pos, macro_fixed, 0, topo_order_out_X, L, affected_L, + adj_matrix, weight_matrix, die_ll, s_id, num_macros) + compute_R_value(macro_pos, macro_fixed, 0, topo_order_in_X, R, affected_R, + adj_matrix, weight_matrix, die_ur, t_id, num_macros) + compute_L_value(macro_pos, macro_fixed, 1, topo_order_out_Y, L, affected_L, + adj_matrix, weight_matrix, die_ll, s_id, num_macros) + compute_R_value(macro_pos, macro_fixed, 1, topo_order_in_Y, R, affected_R, + adj_matrix, weight_matrix, die_ur, t_id, num_macros) + + +@nb.jit(nopython=True, nogil=True, cache=True, parallel=use_numba_parallel) +def compute_edge_slack( + adj_matrix, weight_matrix, L, R, num_nodes, slack_matrix_e, +): + for i in nb.prange(num_nodes): + for j in nb.prange(num_nodes): + if adj_matrix[i][j][0] == 1: + slack_matrix_e[i][j][0] = R[j][0] - L[i][0] - weight_matrix[i][j][0] + if adj_matrix[i][j][1] == 1: + slack_matrix_e[i][j][1] = R[j][1] - L[i][1] - weight_matrix[i][j][1] + + +def slack_info( + adj_matrix, weight_matrix, L, R, num_macros, num_nodes, slack_v, slack_matrix_e, + update_edge_slack=True, +): + # Node Slack + slack_v[:num_macros, :] = R[:num_macros, :] - L[:num_macros, :] + x_nslack = np.minimum(slack_v[:, 0], 0) + y_nslack = np.minimum(slack_v[:, 1], 0) + x_tns, x_wns, nonzero_x = np.sum(x_nslack), np.min(x_nslack), np.sum(x_nslack < 0) + y_tns, y_wns, nonzero_y = np.sum(y_nslack), np.min(y_nslack), np.sum(y_nslack < 0) + info_v = (x_tns, x_wns, nonzero_x, y_tns, y_wns, nonzero_y) + if update_edge_slack: + slack_matrix_e[:,:,:] = 0 + compute_edge_slack(adj_matrix, weight_matrix, L, R, num_nodes, slack_matrix_e) + x_nslack = np.minimum(slack_matrix_e[:, :, 0], 0) + y_nslack = np.minimum(slack_matrix_e[:, :, 1], 0) + x_tns, x_wns, nonzero_x = np.sum(x_nslack), np.min(x_nslack), np.sum(x_nslack < 0) + y_tns, y_wns, nonzero_y = np.sum(y_nslack), np.min(y_nslack), np.sum(y_nslack < 0) + info_e = (x_tns, x_wns, nonzero_x, y_tns, y_wns, nonzero_y) + return info_v, info_e + return info_v, None + + +@nb.jit(nopython=True, nogil=True, cache=True) +def mark_edge_to_move(adj_matrix, weight_matrix, slack_v, i, L, R, s_id, t_id, macro_pos): + edges_pair = [] + x_slack = slack_v[i, 0] + y_slack = slack_v[i, 1] + if (x_slack >= 0 and y_slack >= 0) or (x_slack < 0 and y_slack < 0): + return edges_pair + if x_slack >= 0 and y_slack < 0: + # need to handle in g[1] + axis = 1 + else: + # need to handle in g[0] + axis = 0 + o_axis = 0 if axis == 1 else 1 + is_preds = adj_matrix[:, i, axis] + for j, is_pred in enumerate(is_preds): + # j -> i + if is_pred == 0 or i == j: + continue + if j == s_id or j == t_id: + continue + if L[i][axis] == L[j][axis] + weight_matrix[j][i][axis]: + edge_ij = False + edge_ji = False + if L[i][o_axis] + weight_matrix[i][j][o_axis] <= R[j][o_axis]: + edge_ij = True + is_succs_j = adj_matrix[j, :, o_axis] + for k, is_succ in enumerate(is_succs_j): + # i -> j -> k + if is_succ == 0 or k == j: + continue + if L[i][o_axis] + weight_matrix[i][j][o_axis] + weight_matrix[j][k][o_axis] > R[k][o_axis]: + edge_ij = False + break + if L[j][o_axis] + weight_matrix[j][k][o_axis] > R[k][o_axis]: + edge_ij = False + break + if L[j][o_axis] + weight_matrix[j][i][o_axis] <= R[i][o_axis]: + edge_ji = True + is_succs_i = adj_matrix[i, :, o_axis] + for k, is_succ in enumerate(is_succs_i): + # j -> i -> k + if is_succ == 0 or k == i: + continue + if L[j][o_axis] + weight_matrix[j][i][o_axis] + weight_matrix[i][k][o_axis] > R[k][o_axis]: + edge_ij = False + break + if L[i][o_axis] + weight_matrix[i][k][o_axis] > R[k][o_axis]: + edge_ij = False + break + if edge_ij and edge_ji: + if macro_pos[i][o_axis] <= macro_pos[j][o_axis]: + # i -> j + edges_pair.append((axis, (j, i), o_axis, (i, j))) + else: + # j -> i + edges_pair.append((axis, (j, i), o_axis, (j, i))) + elif edge_ij: + edges_pair.append((axis, (j, i), o_axis, (i, j))) + elif edge_ji: + edges_pair.append((axis, (j, i), o_axis, (j, i))) + else: + edges_pair.clear() + + return edges_pair + + +def longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur, logger, naive=False, prune=False): + edge_type, _, _ = constraint_graph_construction(macro_pos, macro_size, macro_fixed, die_info, prune=prune) + if naive: + return edge_type, None, None + logger.debug("Finish Graph construction.") + die_lx, die_hx, die_ly, die_hy = die_info + num_macros = macro_pos.shape[0] + logger.debug("Longest Path Refinement #Macros: %d #FixedMacros: %d" % (num_macros, macro_fixed.sum())) + num_nodes = num_macros + 2 # including source and target + s_id = num_macros + t_id = num_macros + 1 + dtype = macro_size.dtype + + adj_matrix = np.zeros((num_nodes, num_nodes, 2), dtype=np.int8) + weight_matrix = np.full((num_nodes, num_nodes, 2), -1, dtype=dtype) + + initialize_xy_adj_weight(edge_type, adj_matrix, weight_matrix, macro_size, + num_macros, num_nodes, s_id, t_id) + + edge_type[:, :] = -1 # not used anymore + + g_x = ig.Graph.Adjacency(adj_matrix[:,:,0], mode= "directed") + g_y = ig.Graph.Adjacency(adj_matrix[:,:,1], mode= "directed") + g = [g_x, g_y] + # Use negative weight to find longest path + if not g[0].is_dag(): + logger.warning("g_x is not a DAG.") + if not g[1].is_dag(): + logger.warning("g_y is not a DAG.") + + # Calcuate x_L, x_R, y_L and y_R + L = np.full((num_nodes, 2), -np.inf, dtype=dtype) + R = np.full((num_nodes, 2), np.inf, dtype=dtype) + affected_L = np.ones((num_nodes, 2), dtype=np.bool8) + affected_R = np.ones((num_nodes, 2), dtype=np.bool8) + # Propagate all nodes' L and R + propagate_L_R(g, macro_pos, macro_fixed, affected_L, affected_R, adj_matrix, weight_matrix, + L, R, die_ll, die_ur, s_id, t_id, num_macros) + # Calculate Node and Edge Slacks + slack_v = np.zeros((num_macros, 2), dtype=dtype) + slack_matrix_e = np.zeros((num_nodes, num_nodes, 2), dtype=dtype) + info_v, info_e = slack_info( + adj_matrix, weight_matrix, L, R, num_macros, num_nodes, slack_v, slack_matrix_e) + logger.debug("Before longest path refinement:") + logger.debug(" Node X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Node Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_v) + logger.debug(" Edge X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Edge Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_e) + + # plot_negative_slack_macro(macro_pos, macro_size, macro_fixed, die_info, slack_v) + + macro_area = np.prod(macro_size, axis=1) + num_trials = 0 + num_movement = 0 + while np.minimum(slack_v, 0).sum() < 0: + if num_trials == 5: + logger.error("Cannot fix longest path after %d trials." % num_trials) + break + macro_order = list(range(num_macros)) + sum_slack = slack_v.sum(axis=1) + macro_order.sort(key=lambda x: (macro_area[x], -sum_slack[x], macro_pos[x][0], macro_pos[x][1], x)) + logger.debug("--- Trial %d ---" % num_trials) + num_trials += 1 + for i in macro_order: + # Mark edges to move + edges_pair = mark_edge_to_move(adj_matrix, weight_matrix, slack_v, i, L, R, s_id, t_id, macro_pos) + # Move selected edges + for axis_del, (u_del, v_del), axis_add, (u_add, v_add) in edges_pair: + g[axis_del].delete_edges([(u_del, v_del)]) + g[axis_add].add_edges([(u_add, v_add)]) + assert adj_matrix[u_del, v_del, axis_del] == 1 + assert adj_matrix[u_add, v_add, axis_add] == 0 + adj_matrix[u_del, v_del, axis_del] = 0 + adj_matrix[u_add, v_add, axis_add] = 1 + logger.debug("Move %d: G_%d (%d, %d) -> G_%d (%d, %d)." % ( + num_movement, axis_del, u_del, v_del, axis_add, u_add, v_add)) + + edges_pair = (axis_del, (u_del, v_del), axis_add, (u_add, v_add)) + # edges_pair = None # Debug only + propagate_L_R(g, macro_pos, macro_fixed, affected_L, affected_R, adj_matrix, weight_matrix, + L, R, die_ll, die_ur, s_id, t_id, num_macros, edges_pair=edges_pair) + + info_v, _ = slack_info(adj_matrix, weight_matrix, L, R, num_macros, num_nodes, slack_v, None, + update_edge_slack=False) + logger.debug(" Updated Node X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Node Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_v) + num_movement += 1 + # plot_negative_slack_macro(macro_pos, macro_size, macro_fixed, die_info, slack_v) + + info_v, info_e = slack_info( + adj_matrix, weight_matrix, L, R, num_macros, num_nodes, slack_v, slack_matrix_e) + logger.debug("Finish longest path refinement:") + logger.debug(" Node X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Node Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_v) + logger.debug(" Edge X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Edge Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_e) + + assert (adj_matrix[:num_macros,:num_macros] == 1).all(axis=2).sum() == 0 + edge_type[:, :] = -1 + edge_type[adj_matrix[:num_macros,:num_macros,0] == 1] = 0 + edge_type[adj_matrix[:num_macros,:num_macros,1] == 1] = 1 + + return edge_type, g_x, g_y + + +def basic_variable(macro_pos, macro_size, macro_fixed, die_info): + num_macros = macro_pos.shape[0] + die_lx, die_hx, die_ly, die_hy = die_info + x_set, y_set, dx_set, dy_set = [], [], [], [] + for i in range(num_macros): + if not macro_fixed[i]: + x_set.append(pl.LpVariable( + "x_%d" % i, macro_size[i][0] / 2, die_hx - macro_size[i][0] / 2 + )) + y_set.append(pl.LpVariable( + "y_%d" % i, macro_size[i][1] / 2, die_hy - macro_size[i][1] / 2 + )) + dx_set.append(pl.LpVariable("d_x_%d" % i, 0, die_hx)) + dy_set.append(pl.LpVariable("d_y_%d" % i, 0, die_hy)) + else: + x_value = macro_pos[i][0] + y_value = macro_pos[i][1] + x_set.append(pl.LpVariable("x_%d" % i, x_value, x_value)) + y_set.append(pl.LpVariable("y_%d" % i, y_value, y_value)) + dx_set.append(pl.LpVariable("d_x_%d" % i, 0, 0)) + dy_set.append(pl.LpVariable("d_y_%d" % i, 0, 0)) + + for i in range(num_macros): + x_value = macro_pos[i][0] + y_value = macro_pos[i][1] + x_set[i].setInitialValue(x_value) + y_set[i].setInitialValue(y_value) + dx_set[i].setInitialValue(0) + dy_set[i].setInitialValue(0) + return x_set, y_set, dx_set, dy_set + + +def macro_legalization_xy(args, logger, macro_pos, macro_size, macro_fixed, macro_weights, die_info, die_ll, die_ur, + num_items=None, lpbackend=None, naive=False, prune=False, edge_type=None): + logger.info("Start macro_legalization_xy...") + if num_items is not None: + macro_pos_cache = np.copy(macro_pos) + macro_pos = macro_pos[:num_items] + macro_size = macro_size[:num_items] + macro_fixed = macro_fixed[:num_items] + + num_macros = macro_pos.shape[0] + if edge_type is None: + edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur, + logger, naive=naive, prune=prune) + logger.debug("Finish Graph X construction.") + + prob_x = pl.LpProblem("MacroLegalizationX", pl.LpMinimize) + prob_y = pl.LpProblem("MacroLegalizationY", pl.LpMinimize) + x_set, y_set, dx_set, dy_set = basic_variable(macro_pos, macro_size, macro_fixed, die_info) + prob_x += ( + pl.lpSum([macro_weights[i, 0] * dx_set[i] for i in range(num_macros)]), + "Sum_of_Total_displacement", + ) + prob_y += ( + pl.lpSum([macro_weights[i, 1] * dy_set[i] for i in range(num_macros)]), + "Sum_of_Total_displacement", + ) + for i in range(num_macros): + ori_x = macro_pos[i][0] + ori_y = macro_pos[i][1] + prob_x += ( + x_set[i] - ori_x <= dx_set[i], + "Displacement_x_%d" % i, + ) + prob_x += ( + x_set[i] - ori_x >= -dx_set[i], + "NegDisplacement_x_%d" % i, + ) + prob_y += ( + y_set[i] - ori_y <= dy_set[i], + "Displacement_y_%d" % i, + ) + prob_y += ( + y_set[i] - ori_y >= -dy_set[i], + "NegDisplacement_y_%d" % i, + ) + + # 2) Graph Version X: + dist_x = (macro_size[:,0] + macro_size[:,0].reshape(-1,1)) / 2 + for i in range(num_macros): + for j in range(num_macros): + if edge_type[i][j] == -1 or i == j: + continue + if macro_fixed[i] and macro_fixed[j]: + continue + if edge_type[i][j] == 0: + prob_x += ( + x_set[i] + dist_x[i][j] <= x_set[j], + "Horizontal_%d_%d" % (i, j), + ) + + # Write LP for debugging + # prob_x.writeLP("MacroLegalization.lp") + + # Solve by pl + logger.debug("Start solving...") + prob_x.solve(lpbackend) + + pl_status_x = pl.LpStatus[prob_x.status] + solve_success = pl_status_x == "Optimal" + displacement_x = pl.value(prob_x.objective) + + # Commit Solver Results + macro_pos_new = np.copy(macro_pos) + for v in prob_x.variables(): + if str(v.name).startswith("x_"): + macro_pos_new[int(str(v.name).split("_")[1])][0] = float(v.varValue) + macro_pos = macro_pos_new + + # 3) Graph Version Y: Need to update graph edges since placement is changed + edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur, + logger, naive=naive, prune=prune) + logger.debug("Finish Graph Y construction.") + dist_y = (macro_size[:,1] + macro_size[:,1].reshape(-1,1)) / 2 + for i in range(num_macros): + for j in range(num_macros): + if edge_type[i][j] == -1 or i == j: + continue + if macro_fixed[i] and macro_fixed[j]: + continue + if edge_type[i][j] == 1: + prob_y += ( + y_set[i] + dist_y[i][j] <= y_set[j], + "Vertical_%d_%d" % (i, j), + ) + + # Write LP for debugging + # prob_y.writeLP("MacroLegalization.lp") + + # Solve by pl + logger.debug("Start solving...") + prob_y.solve(lpbackend) + + pl_status_y = pl.LpStatus[prob_y.status] + solve_success = (pl_status_y == "Optimal") and solve_success + displacement_y = pl.value(prob_y.objective) + + # Commit Solver Results + macro_pos_new = np.copy(macro_pos) + for v in prob_y.variables(): + if str(v.name).startswith("y_"): + macro_pos_new[int(str(v.name).split("_")[1])][1] = float(v.varValue) + + logger.info( + "X Status: %s, DisplaceX = %.2f | Y Status: %s, DisplaceY = %.2f | Total Displacement = %.2f" % ( + pl_status_x, displacement_x, pl_status_y, displacement_y, displacement_x + displacement_y + )) + + # 4) Iterative Legalization: + if num_items is not None: + macro_pos_cache[:num_items] = macro_pos_new[:num_items] + logger.info("#Macros: %d, #Macros in step: %d" % (macro_pos_cache.shape[0], macro_pos_new.shape[0])) + macro_pos_new = macro_pos_cache + + return macro_pos_new, solve_success, displacement_x + displacement_y + + +def macro_legalization_mix(args, logger, macro_pos, macro_size, macro_fixed, macro_weights, die_info, die_ll, die_ur, + num_items=None, lpbackend=None, naive=False, prune=False, edge_type=None): + logger.info("Start macro_legalization_mix...") + if num_items is not None: + macro_pos_cache = np.copy(macro_pos) + macro_pos = macro_pos[:num_items] + macro_size = macro_size[:num_items] + macro_fixed = macro_fixed[:num_items] + + num_macros = macro_pos.shape[0] + if edge_type is None: + edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, + die_ll, die_ur, logger, naive=naive, prune=prune) + logger.debug("Finish Graph construction.") + + prob = pl.LpProblem("MacroLegalization", pl.LpMinimize) + logger.debug("Setup lp variables") + x_set, y_set, dx_set, dy_set = basic_variable(macro_pos, macro_size, macro_fixed, die_info) + logger.debug("Setup lp objectives") + prob += ( + pl.lpSum([macro_weights[i, 0] * dx_set[i] + macro_weights[i, 1] * dy_set[i] for i in range(num_macros)]), + "Sum_of_Total_displacement", + ) + logger.debug("Setup lp displacement constrains") + for i in range(num_macros): + ori_x = macro_pos[i][0] + ori_y = macro_pos[i][1] + prob += ( + x_set[i] - ori_x <= dx_set[i], + "Displacement_x_%d" % i, + ) + prob += ( + x_set[i] - ori_x >= -dx_set[i], + "NegDisplacement_x_%d" % i, + ) + prob += ( + y_set[i] - ori_y <= dy_set[i], + "Displacement_y_%d" % i, + ) + prob += ( + y_set[i] - ori_y >= -dy_set[i], + "NegDisplacement_y_%d" % i, + ) + logger.debug("Setup lp edge constraints") + dist_x = (macro_size[:,0] + macro_size[:,0].reshape(-1,1)) / 2 + dist_y = (macro_size[:,1] + macro_size[:,1].reshape(-1,1)) / 2 + for i in range(num_macros): + for j in range(num_macros): + if edge_type[i][j] == -1 or i == j: + continue + if macro_fixed[i] and macro_fixed[j]: + continue + if edge_type[i][j] == 0: + prob += ( + x_set[i] + dist_x[i][j] <= x_set[j], + "Horizontal_%d_%d" % (i, j), + ) + elif edge_type[i][j] == 1: + prob += ( + y_set[i] + dist_y[i][j] <= y_set[j], + "Vertical_%d_%d" % (i, j), + ) + # Write LP for debugging + # prob.writeLP("MacroLegalization.lp") + + # Solve by pl + logger.debug("Start solving...") + prob.solve(lpbackend) + + solve_success = pl.LpStatus[prob.status] == "Optimal" + displacement = pl.value(prob.objective) + logger.info("Status: %s, Total Displacement of MacroLegalization = %.2f" % (pl.LpStatus[prob.status], displacement)) + + # Commit Solver Results + macro_pos_new = np.copy(macro_pos) + for v in prob.variables(): + if str(v.name).startswith("x_"): + macro_pos_new[int(str(v.name).split("_")[1])][0] = float(v.varValue) + if str(v.name).startswith("y_"): + macro_pos_new[int(str(v.name).split("_")[1])][1] = float(v.varValue) + + if num_items is not None: + macro_pos_cache[:num_items] = macro_pos_new[:num_items] + logger.info("#Macros: %d, #Macros in step: %d" % (macro_pos_cache.shape[0], macro_pos_new.shape[0])) + macro_pos_new = macro_pos_cache + + return macro_pos_new, solve_success, displacement + + +def macro_legalization_ilp(args, logger, macro_pos, macro_size, macro_fixed, macro_weights, die_info, die_ll, die_ur, + num_items=None, lpbackend=None, edge_type=None): + logger.info("Start macro_legalization_ilp...") + if num_items is not None: + macro_pos_cache = np.copy(macro_pos) + macro_pos = macro_pos[:num_items] + macro_size = macro_size[:num_items] + macro_fixed = macro_fixed[:num_items] + num_macros = macro_pos.shape[0] + + prob = pl.LpProblem("MacroLegalization", pl.LpMinimize) + die_lx, die_hx, die_ly, die_hy = die_info + x_set, y_set, dx_set, dy_set = basic_variable(macro_pos, macro_size, macro_fixed, die_info) + prob += ( + pl.lpSum([macro_weights[i, 0] * dx_set[i] + macro_weights[i, 1] * dy_set[i] for i in range(num_macros)]), + "Sum_of_Total_displacement", + ) + for i in range(num_macros): + ori_x = macro_pos[i][0] + ori_y = macro_pos[i][1] + prob += ( + x_set[i] - ori_x <= dx_set[i], + "Displacement_x_%d" % i, + ) + prob += ( + x_set[i] - ori_x >= -dx_set[i], + "NegDisplacement_x_%d" % i, + ) + prob += ( + y_set[i] - ori_y <= dy_set[i], + "Displacement_y_%d" % i, + ) + prob += ( + y_set[i] - ori_y >= -dy_set[i], + "NegDisplacement_y_%d" % i, + ) + + # Naive Binary Variables Version: + dist_x = (macro_size[:,0] + macro_size[:,0].reshape(-1,1)) / 2 + dist_y = (macro_size[:,1] + macro_size[:,1].reshape(-1,1)) / 2 + choices = pl.LpVariable.dicts("Choice", (range(num_macros), range(num_macros), range(2)), 0, 1, cat=pl.const.LpInteger) + for i in range(num_macros): + for j in range(num_macros): + if i >= j: + continue + if macro_fixed[i] and macro_fixed[j]: + continue + prob += ( + x_set[i] + dist_x[i][j] <= x_set[j] + die_hx * (choices[i][j][0] + choices[i][j][1]), + "XLhs_%d_%d" % (i, j), + ) + prob += ( + x_set[i] - dist_x[i][j] >= x_set[j] - die_hx * (1 + choices[i][j][0] - choices[i][j][1]), + "XRhs_%d_%d" % (i, j), + ) + prob += ( + y_set[i] + dist_y[i][j] <= y_set[j] + die_hy * (1 - choices[i][j][0] + choices[i][j][1]), + "YLhs_%d_%d" % (i, j), + ) + prob += ( + y_set[i] - dist_y[i][j] >= y_set[j] - die_hy * (2 - choices[i][j][0] - choices[i][j][1]), + "YRhs_%d_%d" % (i, j), + ) + + # Write LP for debugging + # prob.writeLP("MacroLegalization.lp") + + # Solve by pl + logger.debug("Start solving...") + prob.solve(lpbackend) + + solve_success = pl.LpStatus[prob.status] == "Optimal" + displacement = pl.value(prob.objective) + logger.info("Status: %s, Total Displacement of MacroLegalization = %.2f" % (pl.LpStatus[prob.status], displacement)) + + # Commit Solver Results + macro_pos_new = np.copy(macro_pos) + for v in prob.variables(): + if str(v.name).startswith("x_"): + macro_pos_new[int(str(v.name).split("_")[1])][0] = float(v.varValue) + if str(v.name).startswith("y_"): + macro_pos_new[int(str(v.name).split("_")[1])][1] = float(v.varValue) + + if num_items is not None: + macro_pos_cache[:num_items] = macro_pos_new[:num_items] + logger.info("#Macros: %d, #Macros in step: %d" % (macro_pos_cache.shape[0], macro_pos_new.shape[0])) + macro_pos_new = macro_pos_cache + + return macro_pos_new, solve_success, displacement + + +def macro_rough_align(macro_pos, macro_size, macro_fixed, die_ll, die_ur, inv_scalar): + is_mov = np.logical_not(macro_fixed) + + macro_pos_lb = macro_size / 2 + die_ll + 1e-4 + macro_pos_ub = die_ur - macro_size / 2 + die_ll - 1e-4 + mov_macro_pos_lb = macro_pos_lb[is_mov] + mov_macro_pos_ub = macro_pos_ub[is_mov] + macro_pos[is_mov] = macro_pos[is_mov].clip(min=mov_macro_pos_lb, max=mov_macro_pos_ub) + + macro_lpos = macro_pos - macro_size / 2 + macro_lpos = np.multiply(macro_lpos, inv_scalar, out=macro_lpos) + macro_lpos = np.round_(macro_lpos, out=macro_lpos) + macro_lpos = np.divide(macro_lpos, inv_scalar, out=macro_lpos) + + macro_pos_ = macro_lpos + macro_size / 2 + macro_pos[is_mov] = macro_pos_[is_mov] + + return macro_pos + + +def plot_macros(macro_pos, macro_size, macro_fixed, die_info, given_colors=None, img_path=None): + import matplotlib as mpl + import matplotlib.pyplot as plt + from matplotlib.collections import PatchCollection + from matplotlib.patches import Rectangle + base_x = 8 + die_lx, die_hx, die_ly, die_hy = die_info + base_y = base_x / die_hx * die_hy + fig, ax = plt.subplots(1, figsize=(base_x, base_y)) + die_rects = [Rectangle((die_lx, die_ly), die_hx - die_lx, die_hy - die_ly)] + pc = PatchCollection(die_rects, facecolor='w', alpha=0.5, edgecolor='black') + ax.add_collection(pc) + all_color = plt.get_cmap('Set2').colors + for i in range(macro_pos.shape[0]): + x, y = macro_pos[i] + w, h = macro_size[i] + color_idx = given_colors[i] if given_colors is not None else 0 + color = all_color[color_idx] + rect = Rectangle((x - w / 2, y - h / 2), w, h, facecolor=color, alpha=0.5, edgecolor='black') + ax.add_patch(rect) + ax.text(x, y, "%d" % i, fontsize=14) + + plt.xlim(-die_hx * 0.02, die_hx * 1.02) + plt.ylim(-die_hy * 0.02, die_hy * 1.02) + ax.axis('off') + ax.get_xaxis().set_visible(False) + ax.get_yaxis().set_visible(False) + plt.tight_layout() + img_path = img_path if img_path is not None else "test.png" + plt.savefig(img_path, bbox_inches='tight') + plt.close() + + +def plot_negative_slack_macro(macro_pos, macro_size, macro_fixed, die_info, slack_v, img_path=None): + num_macros = slack_v.shape[0] + x_nslack = np.minimum(slack_v[:, 0], 0) + y_nslack = np.minimum(slack_v[:, 1], 0) + print(x_nslack) + print(y_nslack) + unique_values = np.sort(np.unique(x_nslack + y_nslack))[::-1] + given_colors = np.zeros(num_macros, dtype=np.int8) + for color_idx, value in enumerate(unique_values): + given_colors[(x_nslack + y_nslack) == value] = color_idx + plot_macros(macro_pos, macro_size, macro_fixed, die_info, given_colors=given_colors, img_path=img_path) + + +def macro_legalization_multi(macro_info, args, logger): + macro_pos, macro_size, macro_fixed, macro_weights, die_ll, die_ur, die_info, inv_scalar = macro_info + macro_pos = macro_pos.cpu().numpy() + macro_size = macro_size.cpu().numpy() + macro_fixed = macro_fixed.cpu().numpy() + macro_weights = macro_weights.cpu().numpy() + die_ll = die_ll.cpu().numpy() + die_ur = die_ur.cpu().numpy() + die_info = die_info.cpu().numpy() + inv_scalar = inv_scalar.cpu().numpy() + + # plot_macros(macro_pos, macro_size, macro_fixed, die_info, img_path="legalized_before.png") + nb.set_num_threads(args.num_threads) + + if check_macro_legality(macro_pos, macro_size, macro_fixed, die_info, check_all=False): + # Macros are legal, skip legalization + return macro_pos, True + + macro_pos = macro_rough_align(macro_pos, macro_size, macro_fixed, die_ll, die_ur, inv_scalar) + + def macro_lg_handler( + ml_func, method_name, solver, best_result, *func_args, max_times=1, timeLimit=None, **ml_func_kwargs + ): + total_displacement = 0 + for i in range(max_times): + solver.timeLimit = (i + 1) * 20 if timeLimit is None else timeLimit + logger.info("Use cbc to solve LP. TimeLimit = %ds." % solver.timeLimit) + macro_pos_tmp, solve_success, displacement = ml_func(*func_args, lpbackend=solver, **ml_func_kwargs) + if not check_macro_legality(macro_pos_tmp, macro_size, macro_fixed, die_info): + # update macro_pos to macro_pos_tmp + func_args = (*func_args[:2], macro_pos_tmp, *func_args[3:]) + ml_func_kwargs["edge_type"] = None + solve_success = False + total_displacement += displacement + if solve_success: + break + if not solve_success and not best_result[1] and method_name != "ilp": + best_result = (macro_pos_tmp, False, total_displacement, method_name) + if solve_success and (not best_result[1] or total_displacement < best_result[2]): + best_result = (macro_pos_tmp, True, total_displacement, method_name) + return best_result + + # best_result: (macro_pos, solve_success, displacement, method_name) + best_result = (None, False, float('inf'), None) + func_args = (args, logger, macro_pos, macro_size, macro_fixed, macro_weights, die_info, die_ll, die_ur) + solver = pl.PULP_CBC_CMD(msg=0, timeLimit=20, threads=args.num_threads) + if macro_pos.shape[0] > 500: + logger.info("Too many macros, try pruned graph version first.") + edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur, + logger, naive=False, prune=True) + best_result = macro_lg_handler(macro_legalization_mix, "mix", solver, best_result, *func_args, max_times=1, edge_type=edge_type, prune=True) + best_result = macro_lg_handler(macro_legalization_xy, "xy", solver, best_result, *func_args, max_times=3, edge_type=edge_type, prune=True) + + if not best_result[1]: + edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur, + logger, naive=False, prune=False) + best_result = macro_lg_handler(macro_legalization_mix, "mix", solver, best_result, *func_args, max_times=1, edge_type=edge_type) + best_result = macro_lg_handler(macro_legalization_xy, "xy", solver, best_result, *func_args, max_times=3, edge_type=edge_type) + + if not best_result[1]: + logger.warning("Both LPs are infeasible. Try ILP version.") + macro_lg_handler(macro_legalization_ilp, "ilp", solver, best_result, *func_args, timeLimit=120) + if not best_result[1]: + logger.error("ILP is infeasible.") + + # commit the result no matter it is legal or not + macro_pos, solve_success, displacement, method_name = best_result + # plot_macros(macro_pos, macro_size, macro_fixed, die_info, img_path="legalized_after.png") + if solve_success: + logger.info("Macro Legalization Success. Select macro legalization [%s]. Displacement = %.2f" % (method_name, displacement)) + + return macro_pos, solve_success + + +# For debug only +# numba_logger = logging.getLogger('numba') +# numba_logger.setLevel(logging.WARNING) +# import random +# import torch +# import time +# random.seed(0) +# class Args: +# def __init__(self) -> None: +# self.num_threads = 20 +# args = Args() +# logger = logging.getLogger(__name__) +# logging.basicConfig(encoding='utf-8', level=logging.DEBUG, format='%(asctime)s %(message)s') +# if __name__ == "__main__": +# macro_info = torch.load("macro_info.pt") +# logger.info("Start...") +# start_time = time.time() +# macro_pos, solve_success = macro_legalization_multi(macro_info, args, logger) +# logger.info("Macro legalization time: %.2f" % (time.time() - start_time)) diff --git a/src/core/wa_wirelength_hpwl.py b/src/core/wa_wirelength_hpwl.py index 3b36550..422b301 100644 --- a/src/core/wa_wirelength_hpwl.py +++ b/src/core/wa_wirelength_hpwl.py @@ -7,84 +7,6 @@ class HPWLCache: hpwl_cache = HPWLCache() -class WAWirelengthLossAndHPWL(torch.autograd.Function): - @staticmethod - def forward( - ctx, - node_pos, - pin_id2node_id, - pin_rel_cpos, - node2pin_list, - node2pin_list_end, - hyperedge_list, - hyperedge_list_end, - net_mask, - gamma, - hpwl_scale, - deterministic, - ): - - ( - partial_wa_wl, - node_grad, - partial_hpwl, - ) = wa_wirelength_hpwl_cuda.merged_forward_backward_with_hpwl( - node_pos, - pin_id2node_id, - pin_rel_cpos, - node2pin_list, - node2pin_list_end, - hyperedge_list, - hyperedge_list_end, - net_mask, - gamma, - deterministic, - ) - sum_hpwl = torch.round(partial_hpwl * hpwl_scale).sum() - ctx.save_for_backward(node_grad) - return torch.sum(partial_wa_wl), sum_hpwl - - @staticmethod - def backward(ctx, wa_grad_out, hpwl_grad_out): - node_grad = ctx.saved_tensors[0] - return (node_grad * wa_grad_out,) + (None,) * 10 - - -class WAWirelengthLoss(torch.autograd.Function): - @staticmethod - def forward( - ctx, - node_pos, - pin_id2node_id, - pin_rel_cpos, - node2pin_list, - node2pin_list_end, - hyperedge_list, - hyperedge_list_end, - net_mask, - gamma, - deterministic, - ): - partial_wa_wl, node_grad = wa_wirelength_hpwl_cuda.merged_forward_backward( - node_pos, - pin_id2node_id, - pin_rel_cpos, - node2pin_list, - node2pin_list_end, - hyperedge_list, - hyperedge_list_end, - net_mask, - gamma, - deterministic, - ) - ctx.save_for_backward(node_grad) - return torch.sum(partial_wa_wl) - - @staticmethod - def backward(ctx, wa_grad_out): - node_grad = ctx.saved_tensors[0] - return (node_grad * wa_grad_out,) + (None,) * 9 - def merged_wl_loss_grad( node_pos, diff --git a/src/database.py b/src/database.py index 42ee2f4..40148ad 100644 --- a/src/database.py +++ b/src/database.py @@ -317,6 +317,21 @@ class PlaceData(object): if hasattr(self, "__num_fillers__"): return self.__num_fillers__ + @property + def num_macros(self): + if hasattr(self, "__num_macros__"): + return self.__num_macros__ + + @property + def num_movable_macros(self): + if hasattr(self, "__num_movable_macros__"): + return self.__num_movable_macros__ + + @property + def num_fixed_macros(self): + if hasattr(self, "__num_fixed_macros__"): + return self.__num_fixed_macros__ + @property def num_bin_x(self): if hasattr(self, "__num_bin_x__"): @@ -540,7 +555,7 @@ class PlaceData(object): self.pin_rel_cpos /= scalar_at self.pin_rel_lpos /= scalar_at self.pin_size /= scalar_at - self.__die_scale__ *= self.site_width + self.__die_scale__ *= scalar_at return self def prescale(self): @@ -591,10 +606,28 @@ class PlaceData(object): self.net_mask = torch.logical_and( self.net_to_num_pins <= args.ignore_net_degree, self.net_to_num_pins >= 2 ) # 0: ignore, 1: consider in wirelength calculation - # obj related + # macros -> all mov nodes has ultra-large areas with >= 3 row height and all fixed nodes + # But nodes with zero width or zero height are not considered as macros mov_lhs, mov_rhs = self.movable_index - mov_cell_area = torch.prod(self.node_size[mov_lhs:mov_rhs, ...], 1) - self.__total_mov_area_without_filler__ = torch.sum(mov_cell_area).item() + num_movable_nodes = mov_rhs - mov_lhs + self.is_macro: torch.Tensor = self.node_size[:,1] * self.die_scale[1] / self.row_height > 2.01 + mov_node_area = torch.prod(self.node_size[mov_lhs:mov_rhs, ...], 1) + mov_node_area_order = torch.argsort(mov_node_area) + macro_area_threshold = 10 * torch.mean(mov_node_area[ + mov_node_area_order[:int(num_movable_nodes * 0.999)] + ]) + self.is_macro.logical_and_((self.node_area > macro_area_threshold).squeeze(1)) + self.is_macro[mov_rhs:] = True + self.is_macro.logical_and_((self.node_size * self.die_scale > 1e-4).all(dim=1)) + + self.is_mov_macro = self.is_macro.clone() + self.is_mov_macro[mov_rhs:] = False + + self.__num_macros__ = torch.sum(self.is_macro).item() + self.__num_movable_macros__ = torch.sum(self.is_mov_macro).item() + self.__num_fixed_macros__ = self.__num_macros__ - self.__num_movable_macros__ + # obj related + self.__total_mov_area_without_filler__ = torch.sum(mov_node_area).item() self.__bin_area__ = torch.prod(self.unit_len).item() return self @@ -630,42 +663,46 @@ class PlaceData(object): # init_density_map are all normalized to (0.0, 1.0) fixed_node_area = ori_dmap * self.bin_area placeable_area = die_area - fixed_node_area - if True: - mov_cell_area = torch.prod(mov_node_size, 1) - num_movable_nodes = mov_rhs - mov_lhs - mov_node_xsize_order = torch.argsort(mov_node_size[:, 0]) - filler_size_x = torch.mean( - mov_node_size[:, 0][ - mov_node_xsize_order[ - int(num_movable_nodes * 0.05) : int( - num_movable_nodes * 0.95 - ) - ] - ] - ) - filler_size_y = self.site_height / self.die_scale[1] - total_filler_area = max( - args.target_density * placeable_area - torch.sum(mov_cell_area), - 0.0, - ) - single_filler_size = torch.tensor( - [filler_size_x, filler_size_y], - device=mov_node_size.device, - dtype=mov_node_size.dtype, - ) - self.__num_fillers__ = int( - torch.round(total_filler_area / (filler_size_x * filler_size_y)) - ) + mov_node_area = torch.prod(mov_node_size, 1).sum() + mov_macro_area = self.node_area[self.is_mov_macro].sum() + mov_stdcell_area = mov_node_area - mov_macro_area + stdcell_placeable_area = placeable_area - mov_macro_area + stdcell_util = mov_stdcell_area / stdcell_placeable_area + if stdcell_util.item() > args.target_density: + logger.warning("Stdcell util %.2f is larger than target density %.2f. Increase target density to %.2f." % ( + stdcell_util, args.target_density, stdcell_util)) + args.target_density = stdcell_util + if self.is_mov_macro.sum().item() <= 1: + mov_node_xsize = mov_node_size[:, 0] else: - mov_cell_area = torch.prod(mov_node_size, 1) - total_filler_area = max( - args.target_density * placeable_area - - torch.sum(mov_cell_area).item(), - 0.0, - ) - single_filler_area = torch.mean(mov_cell_area) - single_filler_size = single_filler_area.sqrt().repeat(2) - self.__num_fillers__ = int(total_filler_area / single_filler_area) + mov_node_xsize = mov_node_size[:, 0][ + torch.logical_not(self.is_mov_macro[mov_lhs:mov_rhs]) + ].contiguous() + num_movable_nodes = mov_node_xsize.shape[0] + mov_node_xsize_order = torch.argsort(mov_node_xsize) + filler_size_x = torch.mean( + mov_node_xsize[ + mov_node_xsize_order[ + int(num_movable_nodes * 0.05) : int( + num_movable_nodes * 0.95 + ) + ] + ] + ) + filler_size_y = self.site_height / self.die_scale[1] + total_filler_area = max( + args.target_density * stdcell_placeable_area - mov_stdcell_area, + 0.0, + ) + single_filler_size = torch.tensor( + [filler_size_x, filler_size_y], + device=mov_node_size.device, + dtype=mov_node_size.dtype, + ) + self.__num_fillers__ = int( + torch.round(total_filler_area / (filler_size_x * filler_size_y)) + ) + if self.num_fillers > 0: self.filler_size = single_filler_size.repeat(self.num_fillers, 1) logger.info( @@ -689,18 +726,24 @@ class PlaceData(object): args.use_filler = False die_area, placeable_area = die_area.item(), placeable_area.item() - fixed_node_area, mov_cell_area = fixed_node_area.item(), torch.sum(mov_cell_area).item() + fixed_node_area, mov_node_area = fixed_node_area.item(), mov_node_area.item() + mov_macro_area, mov_stdcell_area = mov_macro_area.item(), mov_stdcell_area.item() total_filler_area = float(total_filler_area) logger.info( - "DieArea: %.3E FixArea: %.3E (%.1f%%) PlaceableArea: %.3E (%.1f%%) MovArea: %.3E (%.1f%%) FillerArea: %.3E (%.1f%%)" + "DieArea: %.3E FixArea: %.3E (%.1f%%) PlaceableArea: %.3E (%.1f%%) MovArea: %.3E (%.1f%%) FillerArea: %.3E (%.1f%%) " + "MovMacroArea: %.3E (%.1f%%) MovStdCellArea: %.3E (%.1f%%)" % ( die_area, fixed_node_area, fixed_node_area / die_area * 100, placeable_area, placeable_area / die_area * 100, - mov_cell_area, mov_cell_area / die_area * 100, + mov_node_area, mov_node_area / die_area * 100, total_filler_area, total_filler_area / die_area * 100, + mov_macro_area, mov_macro_area / die_area * 100, + mov_stdcell_area, mov_stdcell_area / die_area * 100, ) ) + if mov_node_area > placeable_area: + logger.warning("MovArea > PlaceableArea. Not enough olaceblae area to place all movable nodes.") return self @@ -769,6 +812,11 @@ class PlaceData(object): num_fltiopin, ) ) + content += ( + "#Macros = %d, #MovMacros = %d, #FixMacros = %d\n" % ( + self.num_macros, self.num_movable_macros, self.num_fixed_macros + ) + ) content += "Core Info " + str([i for i in self.die_info.cpu().numpy()]) + "\n" content += "Site Width = %d, Row Height = %d\n" % ( self.site_width, @@ -783,6 +831,12 @@ class PlaceData(object): content += "target density = %.2f\n" % (args.target_density) content += "===================" logger.info(content) + args.include_macros = True if self.num_movable_macros > 10 else False + if args.include_macros and not args.mixed_size: + logger.warning("Detect many macros. Suggest to turn on mixed_size in cmd args.") + if self.num_movable_macros == 0 and args.mixed_size: + logger.warning("#MovMacros is 0. Turn off mixed size mode.") + args.mixed_size = False return self def preprocess(self): @@ -790,8 +844,8 @@ class PlaceData(object): self.backup_ori_var() self.preshift() self.prescale_by_site_width() - if args.scale_design: - self.prescale() + # if args.scale_design: + # self.prescale() self.pre_compute_var() self.init_fence_region() self.logging_statistics() @@ -803,6 +857,21 @@ class PlaceData(object): self.compute_sorted_node_map() return self + def get_filler_pos(self): + if self.num_fillers > 0: + if self.enable_fence: + raise NotImplementedError("We haven't yet supported fence region.") + else: + filler_pos = torch.rand( + (self.num_fillers, 2), + dtype=self.node_size.dtype, + device=self.node_size.device, + ) + scale = self.die_ur - self.die_ll + shift = self.die_ll + filler_pos = filler_pos * scale + shift + return filler_pos + def get_mov_node_info(self, init_method="randn_center"): args = self.__args__ mov_lhs, mov_rhs = self.movable_index @@ -813,19 +882,15 @@ class PlaceData(object): scale = (self.die_ur - self.die_ll) * 0.001 loc = (self.die_ur + self.die_ll) * 0.5 mov_node_pos = torch.randn_like(mov_node_pos) * scale + loc + elif init_method == "randn_center_lxly": + # TODO: Mixed-size placement is very sensitive to the initial location. + # An elegant yet effective initialization may be needed. + scale = (self.die_ur - self.die_ll) * 0.001 + loc = (self.die_ur + self.die_ll) * 0.5 + mov_node_pos = torch.randn_like(mov_node_pos) * scale + loc + mov_node_size / 2 if self.num_fillers > 0: - if self.enable_fence: - raise NotImplementedError("We haven't yet supported fence region.") - else: - filler_pos = torch.rand( - (self.num_fillers, 2), - dtype=mov_node_size.dtype, - device=mov_node_size.device, - ) - scale = self.die_ur - self.die_ll - shift = self.die_ll - filler_pos = filler_pos * scale + shift + filler_pos = self.get_filler_pos() mov_node_pos = torch.cat([mov_node_pos, filler_pos], dim=0) mov_node_size = torch.cat([mov_node_size, self.filler_size], dim=0) @@ -844,6 +909,10 @@ class PlaceData(object): expand_ratio = mov_node_area / clamp_mov_node_area mov_node_size = clamp_mov_node_size + if args.target_density < 1.0: + expand_ratio[mov_lhs:mov_rhs].masked_fill_( + self.is_mov_macro[mov_lhs:mov_rhs], args.target_density) + return mov_node_pos, mov_node_size, expand_ratio def write_pl(self, node_pos, gp_prefix): diff --git a/src/detail_placement.py b/src/detail_placement.py index efe1920..8f1eeba 100644 --- a/src/detail_placement.py +++ b/src/detail_placement.py @@ -1,10 +1,10 @@ import torch from .database import PlaceData from .evaluator import get_obj_hpwl +from .core.macro_legalization import macro_legalization_multi from utils.visualization import draw_fig_with_cairo_cpp from cpp_to_py import gpudp, routedp import numba as nb -import numpy as np import os import time @@ -13,6 +13,7 @@ class PreprocessDatabaseCache: def __init__(self) -> None: self.node_size = None self.node_weight = None + self.is_macro = None self.pin_id2node_id = None self.node2pin_list = None self.node2pin_list_end = None @@ -20,6 +21,7 @@ class PreprocessDatabaseCache: def reset(self): self.node_size = None self.node_weight = None + self.is_macro = None self.pin_id2node_id = None self.node2pin_list = None self.node2pin_list_end = None @@ -76,10 +78,11 @@ def rearrange_dpdb_node_info(node_pos: torch.Tensor, data: PlaceData): if preprocess_db_cache.node_size is not None: node_size = preprocess_db_cache.node_size.clone() node_weight = preprocess_db_cache.node_weight + is_macro = preprocess_db_cache.is_macro pin_id2node_id = preprocess_db_cache.pin_id2node_id node2pin_list = preprocess_db_cache.node2pin_list node2pin_list_end = preprocess_db_cache.node2pin_list_end - return node_lpos, node_size, node_weight, pin_id2node_id, node2pin_list, node2pin_list_end + return node_lpos, node_size, node_weight, is_macro, pin_id2node_id, node2pin_list, node2pin_list_end pin_id2node_id: torch.Tensor = data.pin_id2node_id.clone().int().cpu().numpy() @@ -100,6 +103,14 @@ def rearrange_dpdb_node_info(node_pos: torch.Tensor, data: PlaceData): node_weight_ori[blkg_rhs:floatiopin_rhs] ), dim=0) + is_macro = torch.cat(( + data.is_macro[:fix_rhs], + data.is_macro[iopin_rhs:blkg_rhs], + data.is_macro[floatiopin_rhs:floatfix_rhs], + data.is_macro[fix_rhs:iopin_rhs], + data.is_macro[blkg_rhs:floatiopin_rhs] + ), dim=0) + old_node2pin_list_end: torch.Tensor = data.node2pin_list_end.int() old_node2pin_list: torch.Tensor = data.node2pin_list.int() @@ -137,18 +148,19 @@ def rearrange_dpdb_node_info(node_pos: torch.Tensor, data: PlaceData): if preprocess_db_cache.node_size is None: preprocess_db_cache.node_size = node_size preprocess_db_cache.node_weight = node_weight + preprocess_db_cache.is_macro = is_macro preprocess_db_cache.pin_id2node_id = pin_id2node_id preprocess_db_cache.node2pin_list = node2pin_list preprocess_db_cache.node2pin_list_end = node2pin_list_end - return node_lpos, node_size, node_weight, pin_id2node_id, node2pin_list, node2pin_list_end + return node_lpos, node_size, node_weight, is_macro, pin_id2node_id, node2pin_list, node2pin_list_end def setup_detailed_rawdb( node_pos: torch.Tensor, use_cpu_db_: bool, data: PlaceData, args, logger, after_lg=True ): curr_site_width = 1.0 # prescale_by_site_width - node_lpos, node_size, node_weight, pin_id2node_id, node2pin_list, node2pin_list_end = rearrange_dpdb_node_info( + node_lpos, node_size, node_weight, is_macro, pin_id2node_id, node2pin_list, node2pin_list_end = rearrange_dpdb_node_info( node_pos, data ) if after_lg: @@ -164,22 +176,9 @@ def setup_detailed_rawdb( mov_lhs, mov_rhs = data.movable_index conn_mov_lhs, conn_mov_rhs = data.movable_connected_index - if args.scale_design: - # scale back - die_scale = data.die_scale / data.site_width # assume site width == 1 in dp - node_lpos = node_lpos * die_scale - node_size = node_size * die_scale - pin_rel_lpos = data.pin_rel_lpos * die_scale - die_info = (data.die_info.reshape(2, 2).t() * die_scale).t().reshape(-1) - region_boxes = ( - (data.region_boxes.reshape(-1, 2, 2).permute(0, 2, 1) * die_scale) - .permute(0, 2, 1) - .reshape(-1, 4) - ) # [:, 0] -> lx, [:, 1] -> hx, [:, 2] -> ly, [:, 3] -> hy - else: - pin_rel_lpos = data.pin_rel_lpos - die_info = data.die_info - region_boxes = data.region_boxes + pin_rel_lpos = data.pin_rel_lpos + die_info = data.die_info + region_boxes = data.region_boxes _, floatmov_rhs, _ = data.node_type_indices[1] _, fix_rhs, _ = data.node_type_indices[2] @@ -206,6 +205,7 @@ def setup_detailed_rawdb( node_lpos.cpu(), node_size.cpu(), node_weight.cpu(), + is_macro.cpu(), pin_rel_lpos.cpu(), pin_id2node_id.cpu(), data.pin_id2net_id.int().cpu(), @@ -232,6 +232,7 @@ def setup_detailed_rawdb( node_lpos, node_size, node_weight, + is_macro.cpu(), pin_rel_lpos, pin_id2node_id, data.pin_id2net_id.int(), @@ -304,17 +305,73 @@ def commit_to_node_pos(node_pos: torch.Tensor, data:PlaceData, dp_rawdb): return node_pos +def run_macro_legalization(node_pos, data: PlaceData, lg_rawdb, args, logger): + if data.num_movable_macros == 0 or (data.num_movable_macros == 1 and data.num_fixed_macros == 0): + return True + if data.num_fixed_macros != 0: + logger.warning("Including fixed macros. LP formula may be infeasible.") + # Pre-process: get macro info + mov_lhs, mov_rhs = data.movable_index + macro_size = data.node_size[data.is_macro].contiguous() + macro_pos = node_pos[data.is_macro].detach().contiguous() + macro_fixed = (torch.cat([ + torch.zeros(mov_rhs - mov_lhs, dtype=torch.bool, device=node_pos.device), + torch.ones(node_pos.shape[0] - mov_rhs, dtype=torch.bool, device=node_pos.device) + ])[data.is_macro]).contiguous() + macro_weights = torch.ones_like(macro_pos) + # macro_id = torch.zeros_like(data.is_macro, dtype=torch.long) + # macro_id[data.is_macro] = torch.cumsum(data.is_macro, 0)[data.is_macro] + inv_scalar = torch.tensor( + [round(1.0 / get_ori_scale_factor(data))], dtype=torch.float32, device=node_pos.device + ) + macro_info = ( + macro_pos, macro_size, macro_fixed, macro_weights, + data.die_ll, data.die_ur, data.die_info, inv_scalar + ) + # Macro legalization + macro_pos, solve_success = macro_legalization_multi(macro_info, args, logger) + node_pos_cache = node_pos.clone() + node_pos_cache[data.is_macro] = torch.from_numpy(macro_pos).to(node_pos.device) + node_pos[mov_lhs:mov_rhs] = node_pos_cache[mov_lhs:mov_rhs] + # Post-process: round to integer and commit to lg_rawdb + node_lpos, _, _, _, _, _, _ = rearrange_dpdb_node_info(node_pos, data) + _, floatmov_rhs, _ = data.node_type_indices[1] + node_lpos[:floatmov_rhs][data.is_macro[:floatmov_rhs]] = node_lpos[:floatmov_rhs][ + data.is_macro[:floatmov_rhs]].contiguous().mul_(inv_scalar).round_().div_(inv_scalar) + + lg_rawdb.commit_from_partial(node_lpos[:, 0], node_lpos[:, 1]) + + return solve_success + + +def macro_legalization_main(node_pos: torch.Tensor, data: PlaceData, args, logger, lg_rawdb=None): + if lg_rawdb is None: + lg_rawdb = setup_detailed_rawdb(node_pos, True, data, args, logger, after_lg=False) + logger.info("Start running Macro Legalization... #Macros: %d, #MovMacros: %d." % ( + data.num_macros, data.num_movable_macros + )) + ml_time = time.time() + run_macro_legalization(node_pos, data, lg_rawdb, args, logger) + # align to site/row and check + if gpudp.macroLegalization(lg_rawdb, data.num_bin_x, data.num_bin_y): + lg_rawdb.commit() + logger.info("Check Pass in Macro Legalization") + else: + logger.error("Check failed in Macro Legalization.") + # Commit result + commit_to_node_pos(node_pos, data, lg_rawdb) + torch.cuda.synchronize(node_pos.device) + logger.info("***** Finish Macro Legalization, HPWL: %.4E Time: %.4f *****" % ( + get_obj_hpwl(node_pos, data, args).item(), time.time() - ml_time + )) + + def run_lg(node_pos: torch.Tensor, data: PlaceData, args, logger): # CPU legalization lg_rawdb = setup_detailed_rawdb(node_pos, True, data, args, logger, after_lg=False) # run LG - logger.info("Start running Macro Legalization...") - ml_time = time.time() - num_bins_x, num_bins_y = data.num_bin_x, data.num_bin_y - if gpudp.macroLegalization(lg_rawdb, num_bins_x, num_bins_y): - lg_rawdb.commit() - logger.info("Finish Macro Legalization. Time: %.4f" % (time.time() - ml_time)) + macro_legalization_main(node_pos, data, args, logger, lg_rawdb) total_cell_area = torch.sum(torch.prod(data.node_size, 1)).item() die_area = torch.prod(data.die_ur - data.die_ll).item() @@ -346,8 +403,6 @@ def run_lg(node_pos: torch.Tensor, data: PlaceData, args, logger): # # Commit result # commit_to_node_pos(node_pos, data, lg_rawdb) # torch.cuda.synchronize(node_pos.device) - # if args.scale_design: - # node_pos /= data.die_scale # info = (-1, 0, data.design_name) # draw_fig_with_cairo_cpp(node_pos, data.node_size, data, info, args, base_size=4096) @@ -368,9 +423,6 @@ def run_lg(node_pos: torch.Tensor, data: PlaceData, args, logger): get_obj_hpwl(node_pos, data, args).item(), time.time() - gl_time )) - if args.scale_design: - node_pos /= data.die_scale - del lg_rawdb return node_pos @@ -449,9 +501,6 @@ def run_dp(node_pos: torch.Tensor, data: PlaceData, args, logger): dp_handler(gpudp.globalSwap, "Global Swap", num_bins_x // 2, num_bins_y // 2, gs_bs, gs_iter) dp_handler(gpudp.kReorder, "K-Reorder 2", num_bins_x, num_bins_y, kr_K, kr_iter) - if args.scale_design: - node_pos /= data.die_scale - del dp_rawdb return node_pos @@ -479,12 +528,6 @@ def run_dp_route_opt(node_pos: torch.Tensor, gpdb, rawdb, ps, data: PlaceData, a node_lpos[:floatmov_rhs].mul_(inv_scalar).round_().div_(inv_scalar) node_size = data.node_size.cpu() - if args.scale_design: - # scale back - die_scale = data.die_scale / data.site_width # assume site width == 1 in dp - node_lpos = node_lpos * die_scale - node_size = node_size * die_scale - die_info = (data.die_info.reshape(2, 2).t() * die_scale).t().reshape(-1) site_width = 1.0 row_height = data.row_height / data.site_width die_info = data.die_info.cpu() diff --git a/src/evaluator.py b/src/evaluator.py index 5e0e241..ff6785c 100644 --- a/src/evaluator.py +++ b/src/evaluator.py @@ -1,7 +1,7 @@ import torch from .database import PlaceData from .core import masked_scale_hpwl -from cpp_to_py import hpwl_cuda +from cpp_to_py import hpwl_cuda, density_map_cuda def get_hpwl(data, pos): # CUDA only @@ -23,23 +23,39 @@ def get_obj_hpwl(node_pos, data: PlaceData, args): hpwl = torch.sum(get_hpwl(data, pin_pos.detach())) return hpwl -def get_obj_overflow(node_pos, density_map_layer, init_density_map, data: PlaceData, args): +def get_obj_overflow(node_pos, init_density_map, ps, data: PlaceData, args): mov_lhs, mov_rhs = data.movable_index - density_map = density_map_layer.get_density_map_naive( - node_pos[mov_lhs:mov_rhs], data.node_size[mov_lhs:mov_rhs], init_density_map + node_pos = node_pos[mov_lhs:mov_rhs] + node_size = data.node_size[mov_lhs:mov_rhs] + if ps.zero_macro_grad: + node_pos = node_pos[ + torch.logical_not(data.is_mov_macro[mov_lhs:mov_rhs]) + ].contiguous() + node_size = node_size[ + torch.logical_not(data.is_mov_macro[mov_lhs:mov_rhs]) + ].contiguous() + node_weight = node_size.new_ones(node_pos.shape[0]) + + if init_density_map is None: + init_density_map = node_pos.new_zeros(data.num_bin_x, data.num_bin_y) + aux_mat = init_density_map.clone() + density_map = density_map_cuda.forward_naive( + node_pos, node_size, node_weight, data.unit_len, aux_mat, data.num_bin_x, + data.num_bin_y, node_pos.shape[0], -1.0, -1.0, 1e-4, False, args.deterministic, ) + with torch.no_grad(): overflow_sum = ((density_map - args.target_density) * data.bin_area).clamp_(min=0.0).sum() overflow = overflow_sum / data.total_mov_area_without_filler return overflow -def evaluate_placement(node_pos, density_map_layer, init_density_map, data: PlaceData, args): +def evaluate_placement(node_pos, init_density_map, ps, data: PlaceData, args): # NOTE: since some nets are masked in global placement, hpwl may # underestimate, this function return the exact value of hpwl # Original overflow calculation uses the clamp node size (expand ratio), # this function uses the exact node size to evaluate the overflow hpwl = get_obj_hpwl(node_pos, data, args) - overflow = get_obj_overflow(node_pos, density_map_layer, init_density_map, data, args) + overflow = get_obj_overflow(node_pos, init_density_map, ps, data, args) return hpwl, overflow def fast_evaluator( diff --git a/src/initializer.py b/src/initializer.py index 52b5d1f..f06b286 100644 --- a/src/initializer.py +++ b/src/initializer.py @@ -1,24 +1,28 @@ import torch from .database import PlaceData from cpp_to_py import density_map_cuda -from .core import WAWirelengthLossAndHPWL -from .calculator import calc_grad +from .core import merged_wl_loss_grad -def get_init_density_map(rawdb, gpdb, data: PlaceData, args, logger): +def get_init_density_map(rawdb, gpdb, data: PlaceData, args, logger, ps=None): lhs, rhs = data.fixed_index device = data.node_size.get_device() dtype = data.node_size.dtype zeros_density_map = torch.zeros( (data.num_bin_x, data.num_bin_y), device=device, dtype=dtype, ) - if lhs == rhs: + if lhs == rhs and (ps is None or not ps.zero_macro_grad): data.init_density_map = zeros_density_map return zeros_density_map # get fix nodes which are located inside die node_pos = data.node_pos[lhs:rhs] node_size = data.node_size[lhs:rhs] node_weight = node_size.new_ones(node_size.shape[0]) + if ps is not None and ps.zero_macro_grad: + # compute the mov + fixed macro density map + node_pos = torch.cat([data.node_pos[data.is_mov_macro].contiguous(), node_pos]) + node_size = torch.cat([data.node_size[data.is_mov_macro].contiguous(), node_size]) + node_weight = node_size.new_ones(node_size.shape[0]) init_density_map = density_map_cuda.forward_naive( node_pos, node_size, node_weight, data.unit_len, zeros_density_map, data.num_bin_x, data.num_bin_y, node_pos.shape[0], -1.0, -1.0, 1e-4, False, @@ -105,22 +109,27 @@ def init_params( conn_node_pos = torch.cat( [conn_node_pos, conn_fix_node_pos], dim=0 ) - wl_loss, hpwl = WAWirelengthLossAndHPWL.apply( + _, conn_node_grad = merged_wl_loss_grad( conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos, data.node2pin_list, data.node2pin_list_end, data.hyperedge_list, data.hyperedge_list_end, data.net_mask, - ps.wa_coeff, data.hpwl_scale, args.deterministic + data.hpwl_scale, ps.wa_coeff, args.deterministic ) - density_loss, overflow = density_map_layer( + wl_grad = torch.zeros_like(mov_node_pos).detach() + wl_grad[mov_lhs:mov_rhs] = conn_node_grad[mov_lhs:mov_rhs] + _, _, density_grad = density_map_layer.merged_density_loss_grad( mov_node_pos, mov_node_size, init_density_map ) - wl_grad, density_grad = calc_grad( - optimizer, mov_node_pos, wl_loss, density_loss - ) + if ps.zero_macro_grad: + wl_grad[mov_lhs:mov_rhs].masked_fill_(data.is_mov_macro[mov_lhs:mov_rhs].unsqueeze(1), 0) + density_grad[mov_lhs:mov_rhs].masked_fill_(data.is_mov_macro[mov_lhs:mov_rhs].unsqueeze(1), 0) + (mov_node_pos * 0.0).sum().backward() + optimizer.zero_grad(set_to_none=False) + if not ps.rerun_route or route_fn is None: init_density_weight = (wl_grad.norm(p=1) / density_grad.norm(p=1)).detach() # init_density_weight = (wl_grad.norm(p=1) / grad_mat.norm(p=1)).detach() - ps.set_init_param(init_density_weight, data, density_loss) + ps.set_init_param(init_density_weight, data) else: _, filler_lhs = data.movable_connected_index filler_rhs = mov_node_pos.shape[0] diff --git a/src/param_scheduler.py b/src/param_scheduler.py index 79f49bc..6f19510 100644 --- a/src/param_scheduler.py +++ b/src/param_scheduler.py @@ -70,6 +70,7 @@ class MetricRecorder: class ParamScheduler: def __init__(self, data: PlaceData, args, logger) -> None: self.__logger__ = logger + self.__args__ = args self.data = data self.iter = 0 self.init_iter = 0 @@ -97,7 +98,7 @@ class ParamScheduler: self.best_sol_rollback: torch.Tensor = None self.best_metric_rollback = {"overflow": float("inf"), "hpwl": float("inf")} - # params + # global place params self.precond_coef = 1.0 self.precond_weight = None self.density_weight_start = args.density_weight @@ -113,13 +114,15 @@ class ParamScheduler: self.life = self.max_life self.stop_overflow = args.stop_overflow self.skip_update = False if args.enable_skip_update else None - self.enable_fence = data.enable_fence self.min_enlarge_density_interval = 1000 self.last_enlarge_density_iter = -self.min_enlarge_density_interval # skip density force - self.enable_sample_force = True + self.enable_sample_force = args.enable_sample_force self.force_ratio = 0.0 + self.enable_fence = data.enable_fence + + # routability parameter self.enable_route = args.use_route_force or args.use_cell_inflate self.use_cell_inflate = args.use_cell_inflate self.use_route_force = args.use_route_force @@ -140,14 +143,21 @@ class ParamScheduler: self.max_route_opt = 5 self.gr_sol_recorder = [] - def set_init_param(self, init_density_weight, data: PlaceData, init_density_loss): + # mixed size parameter + self.enable_mixed_size = args.mixed_size + self.include_macros = args.include_macros + self.zero_macro_grad = False + + def set_init_param(self, init_density_weight, data: PlaceData): # init_density_weight self.init_iter = self.iter self.all_init_iters.append(self.init_iter) self.precond_coef = 1.0 + self.mu = 1.0 self.density_weight = copy.deepcopy(self.density_weight_start) * init_density_weight self.wa_coeff = copy.deepcopy(self.wa_coeff_start) self.update_precond_weight(data) + self.set_mixsize_init_param() def set_route_init_param( self, init_density_weight, init_route_weight, init_congest_weight, data: PlaceData, args @@ -157,12 +167,38 @@ class ParamScheduler: self.init_iter = self.iter self.all_init_iters.append(self.init_iter) self.precond_coef = 1.0 + self.mu = 1.0 self.base_route_weight = init_route_weight * args.route_weight self.base_congest_weight = init_congest_weight * args.congest_weight self.route_weight = copy.deepcopy(self.density_weight) * self.base_route_weight self.congest_weight = copy.deepcopy(self.density_weight) * self.base_congest_weight self.pseudo_weight = args.pseudo_weight # same scale as wirelength weight self.update_precond_weight(data) + self.set_mixsize_init_param() + + def set_mixsize_init_param(self): + args = self.__args__ + if self.include_macros: + self.skip_update = None + self.enable_sample_force = False + if self.enable_mixed_size: + if not self.zero_macro_grad: + # simultaneously place macro and std cells + self.include_macros = True + self.stop_overflow = args.stop_overflow * 2.0 + self.enable_sample_force = False + self.skip_update = None + self.enable_route = False + self.use_cell_inflate = False + self.use_route_force = False + else: + self.include_macros = False + self.stop_overflow = args.stop_overflow + self.enable_sample_force = args.enable_sample_force + self.skip_update = False if args.enable_skip_update else None + self.enable_route = args.use_route_force or args.use_cell_inflate + self.use_cell_inflate = args.use_cell_inflate + self.use_route_force = args.use_route_force def reset_best_sol(self): # best solution @@ -384,6 +420,7 @@ class ParamScheduler: if ( self.recorder.overflow[ptr] < self.stop_overflow * 5 and self.recorder.overflow[ptr] >= self.stop_overflow + and not self.include_macros ): if self.check_plateau(self.recorder.overflow, window=50, threshold=0.05): # kill the program since it has converged diff --git a/src/run_placement_nesterov.py b/src/run_placement_nesterov.py index fdf6a5e..85041e3 100644 --- a/src/run_placement_nesterov.py +++ b/src/run_placement_nesterov.py @@ -20,9 +20,6 @@ def run_placement_main_nesterov(args, logger): assert args.use_eplace_nesterov logger.info("Start place %s/%s" % (args.dataset , args.design_name)) logger.info("Use Nesterov optimizer!") - if args.scale_design: - logger.warning("Eplace's nesterov optimizer cannot support normalized die. Disable scale_design.") - args.scale_design = False data = data.to(device) data = data.preprocess() logger.info(data) @@ -140,6 +137,7 @@ def run_placement_main_nesterov(args, logger): # exit(0) terminate_signal = False route_early_terminate_signal = False + log_info = False for iteration in range(args.inner_iter): # optimizer.zero_grad() # zero grad inside obj_and_grad_fn obj = optimizer.step(obj_and_grad_fn) @@ -148,6 +146,68 @@ def run_placement_main_nesterov(args, logger): ps.step(hpwl, overflow, mov_node_pos, data) if ps.need_to_early_stop(): terminate_signal = True + log_info = True + + if ps.enable_mixed_size and not ps.zero_macro_grad and terminate_signal: + ps.zero_macro_grad = True + # Find best gp node_pos (including macros and std cells) + best_res = ps.get_best_solution() + if best_res[0] is not None: + best_sol, hpwl, overflow = best_res + # fillers are unused from now, we don't copy there data + mov_node_pos[mov_lhs:mov_rhs].data.copy_(best_sol[mov_lhs:mov_rhs]) + node_pos = mov_node_pos[mov_lhs:mov_rhs] + node_pos = torch.cat([node_pos, data.node_pos[mov_rhs:]], dim=0) + # Evaluate the mixed placement solution + hpwl, overflow = evaluate_placement(node_pos, init_density_map, ps, data, args) + hpwl, overflow = hpwl.item(), overflow.item() + if args.draw_placement: + info = ("%d_mixed_gp" % (iteration + 1), hpwl, data.design_name) + draw_fig_with_cairo_cpp(node_pos, data.node_size, data, info, args) + logger.info("After Mixed-GP, best solution eval, exact HPWL: %.4E exact Overflow: %.4f" % (hpwl, overflow)) + # Run macro legalization to change node_pos inplace + macro_legalization_main(node_pos, data, args, logger) + if args.draw_placement: + info = ("%d_mixed_gp_ml" % (iteration + 1), hpwl, data.design_name) + draw_fig_with_cairo_cpp(node_pos, data.node_size, data, info, args) + # Write node_pos into database to provide an initial solution for std cell placement + data.node_pos[mov_lhs:mov_rhs].data.copy_(node_pos[mov_lhs:mov_rhs]) + # Prepare for std cell placement + init_density_map = get_init_density_map(rawdb, gpdb, data, args, logger, ps=ps) + data.__total_mov_area_without_filler__ = torch.sum(data.node_area[mov_lhs:mov_rhs][torch.logical_not(data.is_mov_macro[mov_lhs:mov_rhs])]).item() + mov_node_pos, mov_node_size, expand_ratio = data.get_mov_node_info(init_method="randn_center") + mov_macros_idx = data.is_mov_macro[mov_lhs:mov_rhs] + mov_node_pos[mov_lhs:mov_rhs][mov_macros_idx] = data.node_pos[mov_lhs:mov_rhs][mov_macros_idx] + mov_node_pos = mov_node_pos.requires_grad_(True) + trunc_node_pos_fn = get_trunc_node_pos_fn(mov_node_size, data) + density_map_layer.expand_ratio = expand_ratio + density_map_layer.sorted_maps = data.sorted_maps + # ignore the density and grad computation of macros by node_weight + density_map_layer.cache_node_weight[mov_lhs:mov_rhs][mov_macros_idx] = -1.0 + # update partial function correspondingly + obj_and_grad_fn.keywords["constraint_fn"] = trunc_node_pos_fn + obj_and_grad_fn.keywords["mov_node_size"] = mov_node_size + obj_and_grad_fn.keywords["expand_ratio"] = expand_ratio + obj_and_grad_fn.keywords["init_density_map"] = init_density_map + evaluator_fn.keywords["constraint_fn"] = trunc_node_pos_fn + evaluator_fn.keywords["mov_node_size"] = mov_node_size + evaluator_fn.keywords["init_density_map"] = init_density_map + # reset nesterov optimizer + logger.info("Reset optimizer...") + optimizer = NesterovOptimizer([mov_node_pos], lr=0) + # initialization + init_params( + mov_node_pos, trunc_node_pos_fn, mov_lhs, mov_rhs, conn_fix_node_pos, + density_map_layer, mov_node_size, expand_ratio, init_density_map, optimizer, + ps, data, args, route_fn=calc_route_force + ) + # init learnig rate + cur_lr = estimate_initial_learning_rate(obj_and_grad_fn, trunc_node_pos_fn, mov_node_pos, args.lr) + for param_group in optimizer.param_groups: + param_group["lr"] = cur_lr.item() + ps.reset_best_sol() + terminate_signal = False # reset signal + logger.info("Re-run std cell placement with fixed macros.") if ps.use_cell_inflate and ps.curr_optimizer_cnt < ps.max_route_opt and terminate_signal: terminate_signal = False # reset signal @@ -185,6 +245,7 @@ def run_placement_main_nesterov(args, logger): ps.rerun_route = False if ps.rerun_route: + log_info = True new_mov_node_size, new_expand_ratio = None, None if ps.use_cell_inflate: output = route_inflation( @@ -245,7 +306,8 @@ def run_placement_main_nesterov(args, logger): ) ps.reset_best_sol() - if iteration % args.log_freq == 0 or iteration == args.inner_iter - 1 or ps.rerun_route or terminate_signal: + if iteration % args.log_freq == 0 or iteration == args.inner_iter - 1 or log_info: + log_info = False log_str = ( "iter: %d | masked_hpwl: %.2E overflow: %.4f obj: %.4E " "density_weight: %.4E wa_coeff: %.4E" @@ -290,6 +352,10 @@ def run_placement_main_nesterov(args, logger): best_sol, hpwl, overflow = best_res # fillers are unused from now, we don't copy there data mov_node_pos[mov_lhs:mov_rhs].data.copy_(best_sol[mov_lhs:mov_rhs]) + if ps.enable_mixed_size and ps.zero_macro_grad: + # rollback macro_pos to previous legalized results since trunc_node_pos_fn may change them + mov_macros_idx = data.is_mov_macro[mov_lhs:mov_rhs] + mov_node_pos.data[mov_lhs:mov_rhs][mov_macros_idx] = data.node_pos[mov_lhs:mov_rhs][mov_macros_idx] if ps.enable_route: route_inflation_roll_back(args, logger, data, mov_node_size) if not route_early_terminate_signal: @@ -314,9 +380,7 @@ def run_placement_main_nesterov(args, logger): ) # Eval - hpwl, overflow = evaluate_placement( - node_pos, density_map_layer, init_density_map, data, args - ) + hpwl, overflow = evaluate_placement(node_pos, init_density_map, ps, data, args) hpwl, overflow = hpwl.item(), overflow.item() info = ("%d_gp" % (iteration + 1), hpwl, data.design_name) if args.draw_placement: diff --git a/utils/get_design_params.py b/utils/get_design_params.py index fdd3f92..de661a4 100644 --- a/utils/get_design_params.py +++ b/utils/get_design_params.py @@ -4,6 +4,8 @@ import os def find_benchmark(dataset_root, benchmark): bm_to_root = { "ispd2005": os.path.join(dataset_root, "ispd2005"), + "ispd2006": os.path.join(dataset_root, "ispd2006"), + "mms": os.path.join(dataset_root, "mms"), "dac2012": os.path.join(dataset_root, "iccad2012dac2012"), "ispd2015": os.path.join(dataset_root, "ispd2015"), "ispd2015_fix": os.path.join(dataset_root, "ispd2015_fix"), @@ -11,6 +13,7 @@ def find_benchmark(dataset_root, benchmark): "ispd2019_no_fence": os.path.join(dataset_root, "ispd2019_no_fence"), "iccad2019": os.path.join(dataset_root, "iccad2019"), "ispd2018": os.path.join(dataset_root, "ispd2018"), + "iccad2015": os.path.join(dataset_root, "iccad2015"), } root = bm_to_root[benchmark] all_designs = [i for i in os.listdir(root) if os.path.isdir(os.path.join(root, i))] @@ -18,8 +21,8 @@ def find_benchmark(dataset_root, benchmark): def get_single_design_params(dataset_root, benchmark, design_name, placement=None): - if benchmark == "ispd2005": - return single_ispd2005(dataset_root, design_name, placement) + if benchmark in ["ispd2005", "ispd2006", "mms"]: + return single_ispd2005(dataset_root, design_name, benchmark, placement) elif benchmark == "dac2012": return single_dac2012(dataset_root, design_name, placement) elif benchmark == "ispd2015": @@ -34,6 +37,8 @@ def get_single_design_params(dataset_root, benchmark, design_name, placement=Non return single_iccad2019(dataset_root, design_name, placement) elif benchmark == "ispd2018": return single_ispd2018(dataset_root, design_name, placement) + elif benchmark.startswith("iccad2015"): + return single_iccad2015(dataset_root, benchmark, design_name, placement) else: raise NotImplementedError("benchmark %s is not found" % benchmark) @@ -47,8 +52,8 @@ def get_multiple_design_params(dataset_root, benchmark): return params_mul -def single_ispd2005(dataset_root, design_name, placement=None): - benchmark = "ispd2005" +def single_ispd2005(dataset_root, design_name, benchmark, placement=None): + benchmark = benchmark root, all_designs = find_benchmark(dataset_root, benchmark) if design_name not in all_designs: raise ValueError("Design Name %s should in %s" % (design_name, all_designs)) @@ -56,9 +61,10 @@ def single_ispd2005(dataset_root, design_name, placement=None): "benchmark": benchmark, "bookshelf_variety": "ispd2005", "aux": "%s/%s/%s.aux" % (root, design_name, design_name), - "pl": "%s/%s/%s.pl" % (root, design_name, design_name) if placement is None else placement, "design_name": design_name, } + if placement is not None: + params["pl"] = placement return params @@ -71,9 +77,10 @@ def single_dac2012(dataset_root, design_name, placement=None): "benchmark": benchmark, "bookshelf_variety": "dac2012", "aux": "%s/%s/%s.aux" % (root, design_name, design_name), - "pl": "%s/%s/%s.pl" % (root, design_name, design_name) if placement is None else placement, "design_name": design_name, } + if placement is not None: + params["pl"] = placement return params @@ -170,6 +177,23 @@ def single_ispd2018(dataset_root, design_name, placement=None): return params +def single_iccad2015(dataset_root, benchmark, design_name, placement=None): + # configuration + # benchmark = "iccad2015" + root, all_designs = find_benchmark(dataset_root, benchmark) + if design_name not in all_designs: + raise ValueError("Design Name %s should in %s" % (design_name, all_designs)) + params = { + "benchmark": benchmark, + "tech_lef": "%s/tech.lef" % (root), + "cell_lef": "%s/%s/%s.lef" % (root, design_name, design_name), + "def": "%s/%s/%s.def" % (root, design_name, design_name) if placement is None else placement, + "verilog": "%s/%s/%s.v" % (root, design_name, design_name), + "design_name": design_name, + } + return params + + def get_custom_design_params(args): params = dict( [ diff --git a/utils/io_parser.py b/utils/io_parser.py index 6812ee6..f8770d1 100644 --- a/utils/io_parser.py +++ b/utils/io_parser.py @@ -71,19 +71,16 @@ class IOParser(object): print("def %s not exists." % params["def"]) return False if "aux" in params.keys(): - if "pl" not in params.keys(): - print("pl is not found!") + if not os.path.exists(params["aux"]): + print("aux %s not exists." % params["aux"]) + return False + if "pl" in params.keys() and not os.path.exists(params["pl"]): + print("pl %s not exists." % params["pl"]) return False if "output" in params.keys(): if "pl" != params["output"].split(".")[-1]: print("output format should be .pl") return False - if not os.path.exists(params["aux"]): - print("aux %s not exists." % params["aux"]) - return False - if not os.path.exists(params["pl"]): - print("pl %s not exists." % params["pl"]) - return False self.params = params if verbose_log: diff --git a/utils/logger.py b/utils/logger.py index 795aaf9..dfc6477 100644 --- a/utils/logger.py +++ b/utils/logger.py @@ -16,7 +16,7 @@ class CustomFormatter(logging.Formatter): # FIXME: An elapsed time gap exists between the C++ Timer and Python Timer, # but I don't know how to resolve it... time_format = "[%(relativeCreatedSecond)4d.%(relativeCreatedMSecond)03d] " - debug_msg = " (%(module)s.py Line%(lineno)d) %(msg)s" + debug_msg = " (%(module)s.py L%(lineno)d) %(msg)s" FORMATS = { logging.DEBUG: time_format + blue + "DEBUG" + reset + debug_msg, logging.INFO: time_format + "%(msg)s", @@ -47,7 +47,9 @@ def setup_logger(args, sys_argv) -> logging.Logger: screen_handler = logging.StreamHandler(stream=sys.stdout) screen_handler.setFormatter(formatter) logger = logging.getLogger() - logger.setLevel(logging.DEBUG) + logger.setLevel(logging.INFO) + if args.cpp_log_level == 0 or args.verbose_cpp_log: + logger.setLevel(logging.DEBUG) logger.addHandler(file_handler) logger.addHandler(screen_handler) diff --git a/utils/setup_dataset.py b/utils/setup_dataset.py index b4a0aa7..ed37355 100644 --- a/utils/setup_dataset.py +++ b/utils/setup_dataset.py @@ -44,6 +44,30 @@ def setup_design_args(args): elif args.design_name in ["bigblue3", "bigblue4"]: args.num_bin_x = args.num_bin_y = 2048 args.target_density = 1.0 + elif args.design_name in ["adaptec5"]: + args.target_density = 0.5 + args.num_bin_x = args.num_bin_y = 1024 + elif args.design_name in ["newblue1"]: + args.target_density = 0.8 + args.num_bin_x = args.num_bin_y = 512 + elif args.design_name in ["newblue2"]: + args.target_density = 0.9 + args.num_bin_x = args.num_bin_y = 1024 + elif args.design_name in ["newblue3"]: + args.target_density = 0.8 + args.num_bin_x = args.num_bin_y = 2048 + elif args.design_name in ["newblue4"]: + args.target_density = 0.5 + args.num_bin_x = args.num_bin_y = 1024 + elif args.design_name in ["newblue5"]: + args.target_density = 0.5 + args.num_bin_x = args.num_bin_y = 1024 + elif args.design_name in ["newblue6"]: + args.target_density = 0.8 + args.num_bin_x = args.num_bin_y = 2048 + elif args.design_name in ["newblue7"]: + args.target_density = 0.8 + args.num_bin_x = args.num_bin_y = 2048 elif args.design_name in ["mgc_des_perf_1"]: args.num_bin_x = args.num_bin_y = 512 args.target_density = 0.91