(1) support mixed size (2) support ispd2006/mms (3) remove legacy code
This commit is contained in:
parent
a9bde59465
commit
8e7c817f46
@ -20,7 +20,7 @@ Database::~Database() {
|
||||
void Database::load() {
|
||||
// ----- design related options -----
|
||||
|
||||
if (setting.BookshelfAux != "" && setting.BookshelfPl != "") {
|
||||
if (setting.BookshelfAux != "") {
|
||||
setting.Format = "bookshelf";
|
||||
readBSAux(setting.BookshelfAux, setting.BookshelfPl);
|
||||
}
|
||||
|
||||
@ -342,6 +342,12 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
readBSLine(fs, tokens);
|
||||
fs.close();
|
||||
|
||||
bool includePl = true;
|
||||
if (plFile == "") {
|
||||
logger.warning("No pl file specified. Try to find pl in aux file.");
|
||||
includePl = false;
|
||||
}
|
||||
|
||||
std::string fileNodes;
|
||||
std::string fileNets;
|
||||
std::string fileScl;
|
||||
@ -376,6 +382,10 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
logger.error("unrecognized file extension: %s", ext.c_str());
|
||||
}
|
||||
}
|
||||
if (includePl) {
|
||||
logger.info("pl file %s is given.", plFile.c_str());
|
||||
filePl = plFile;
|
||||
}
|
||||
// step 1: read floorplan, rows from:
|
||||
// scl - rows
|
||||
// step 2: read cell types from:
|
||||
@ -399,16 +409,12 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
|
||||
readBSRoute(fileRoute);
|
||||
readBSShapes(fileShapes);
|
||||
readBSWts(fileWts);
|
||||
// readBSPl ( filePl );
|
||||
readBSPl(plFile);
|
||||
readBSPl(filePl);
|
||||
readBSScl(fileScl);
|
||||
} else if (bsData.format == "ispd2005") {
|
||||
readBSNets(fileNets);
|
||||
// readBSRoute ( fileRoute );
|
||||
// readBSShapes( fileShapes);
|
||||
readBSWts(fileWts);
|
||||
// readBSPl ( filePl );
|
||||
readBSPl(plFile);
|
||||
readBSPl(filePl);
|
||||
readBSScl(fileScl);
|
||||
}
|
||||
|
||||
|
||||
@ -1477,10 +1477,10 @@ int readDefComponent(defrCallbackType_e c, defiComponent* co, defiUserData ud) {
|
||||
cell->unplace();
|
||||
} else if (co->isPlaced()) {
|
||||
cell->place(co->placementX(), co->placementY(), co->placementOrient());
|
||||
if (celltype->cls == "CORE") {
|
||||
if (celltype->cls == "CORE" || celltype->cls == "BLOCK") {
|
||||
cell->fixed(false);
|
||||
} else {
|
||||
// Set all non-CORE cells as fixed cells
|
||||
// Set all non-CORE yet non-BLOCK cells as fixed cells
|
||||
cell->fixed(true);
|
||||
}
|
||||
if (co->placementOrient() % 2 == 1) {
|
||||
|
||||
@ -20,6 +20,7 @@ PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
torch::Tensor,
|
||||
float,
|
||||
float,
|
||||
float,
|
||||
@ -34,6 +35,7 @@ PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
|
||||
.def("commit", &dp::DPTorchRawDB::commit)
|
||||
.def("rollback", &dp::DPTorchRawDB::rollback)
|
||||
.def("commit_from", &dp::DPTorchRawDB::commit_from)
|
||||
.def("commit_from_partial", &dp::DPTorchRawDB::commit_from_partial)
|
||||
.def("get_curr_cposx", &dp::DPTorchRawDB::get_curr_cposx, py::return_value_policy::move)
|
||||
.def("get_curr_cposy", &dp::DPTorchRawDB::get_curr_cposy, py::return_value_policy::move)
|
||||
.def("get_curr_lposx", &dp::DPTorchRawDB::get_curr_lposx, py::return_value_policy::move)
|
||||
@ -43,6 +45,7 @@ PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
|
||||
[](torch::Tensor node_lpos_init_,
|
||||
torch::Tensor node_size_,
|
||||
torch::Tensor node_weight_,
|
||||
torch::Tensor is_macro_,
|
||||
torch::Tensor pin_rel_lpos_,
|
||||
torch::Tensor pin_id2node_id_,
|
||||
torch::Tensor pin_id2net_id_,
|
||||
@ -66,6 +69,7 @@ PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
|
||||
return std::make_shared<dp::DPTorchRawDB>(node_lpos_init_,
|
||||
node_size_,
|
||||
node_weight_,
|
||||
is_macro_,
|
||||
pin_rel_lpos_,
|
||||
pin_id2node_id_,
|
||||
pin_id2net_id_,
|
||||
|
||||
@ -8,6 +8,7 @@ namespace dp {
|
||||
DPTorchRawDB::DPTorchRawDB(torch::Tensor node_lpos_init_,
|
||||
torch::Tensor node_size_,
|
||||
torch::Tensor node_weight_,
|
||||
torch::Tensor is_macro_,
|
||||
torch::Tensor pin_rel_lpos_,
|
||||
torch::Tensor pin_id2node_id_,
|
||||
torch::Tensor pin_id2net_id_,
|
||||
@ -74,6 +75,7 @@ DPTorchRawDB::DPTorchRawDB(torch::Tensor node_lpos_init_,
|
||||
|
||||
net_mask = net_mask_;
|
||||
node_weight = node_weight_;
|
||||
is_macro = is_macro_;
|
||||
|
||||
site_width = site_width_;
|
||||
row_height = row_height_;
|
||||
@ -180,6 +182,16 @@ void DPTorchRawDB::commit_from(torch::Tensor x_, torch::Tensor y_) {
|
||||
.copy_(y_.index({torch::indexing::Slice(0, num_movable_nodes)}));
|
||||
}
|
||||
|
||||
void DPTorchRawDB::commit_from_partial(torch::Tensor x_, torch::Tensor y_) {
|
||||
// commit external pos to new pos
|
||||
x.index({torch::indexing::Slice(0, num_movable_nodes)})
|
||||
.data()
|
||||
.copy_(x_.index({torch::indexing::Slice(0, num_movable_nodes)}));
|
||||
y.index({torch::indexing::Slice(0, num_movable_nodes)})
|
||||
.data()
|
||||
.copy_(y_.index({torch::indexing::Slice(0, num_movable_nodes)}));
|
||||
}
|
||||
|
||||
torch::Tensor DPTorchRawDB::get_curr_cposx() { return x + node_size_x / 2; }
|
||||
torch::Tensor DPTorchRawDB::get_curr_cposy() { return y + node_size_y / 2; }
|
||||
torch::Tensor DPTorchRawDB::get_curr_lposx() { return x; }
|
||||
|
||||
@ -10,6 +10,7 @@ public:
|
||||
DPTorchRawDB(torch::Tensor node_lpos_init_,
|
||||
torch::Tensor node_size_,
|
||||
torch::Tensor node_weight_,
|
||||
torch::Tensor is_macro_,
|
||||
torch::Tensor pin_rel_lpos_,
|
||||
torch::Tensor pin_id2node_id_,
|
||||
torch::Tensor pin_id2net_id_,
|
||||
@ -35,6 +36,7 @@ public:
|
||||
void commit();
|
||||
void rollback();
|
||||
void commit_from(torch::Tensor x_, torch::Tensor y_);
|
||||
void commit_from_partial(torch::Tensor x_, torch::Tensor y_);
|
||||
torch::Tensor get_curr_cposx();
|
||||
torch::Tensor get_curr_cposy();
|
||||
torch::Tensor get_curr_lposx();
|
||||
@ -48,6 +50,7 @@ public:
|
||||
torch::Tensor pin_rel_lpos;
|
||||
|
||||
torch::Tensor node_weight;
|
||||
torch::Tensor is_macro;
|
||||
|
||||
torch::Tensor init_x; // original pos (keep it const except committing)
|
||||
torch::Tensor init_y; // original pos (keep it const except committing)
|
||||
|
||||
@ -55,7 +55,7 @@ void distributeCells2Bins(const LegalizationData& db,
|
||||
num_legalized_nodes = num_movable_nodes;
|
||||
}
|
||||
for (int i = 0; i < num_legalized_nodes; i += 1) {
|
||||
if (!db.is_dummy_fixed(i)) {
|
||||
if (!db.is_mov_macro(i)) {
|
||||
int bin_id_x = (x[i] + node_size_x[i] / 2 - xl) / bin_size_x;
|
||||
int bin_id_y = (y[i] + node_size_y[i] / 2 - yl) / bin_size_y;
|
||||
|
||||
@ -87,7 +87,7 @@ void distributeFixedCells2Bins(const LegalizationData& db,
|
||||
std::vector<std::vector<int>>& bin_cells) {
|
||||
// one cell can be assigned to multiple bins
|
||||
for (int i = 0; i < num_nodes; i += 1) {
|
||||
if (db.is_dummy_fixed(i) || i >= num_movable_nodes) {
|
||||
if (db.is_mov_macro(i) || i >= num_movable_nodes) {
|
||||
int node_id = i;
|
||||
int bin_id_xl = std::max((int)floorDiv(x[node_id] - xl, bin_size_x, 0), 0);
|
||||
int bin_id_xh = std::min((int)ceilDiv((x[node_id] + node_size_x[node_id] - xl), bin_size_x, 0), num_bins_x);
|
||||
|
||||
@ -71,6 +71,7 @@ public:
|
||||
node2fence_region_map(at_db.node2fence_region_map.data_ptr<int>()),
|
||||
net_mask(at_db.net_mask.data_ptr<bool>()),
|
||||
node_weight(at_db.node_weight.data_ptr<float>()),
|
||||
is_macro(at_db.is_macro.data_ptr<bool>()),
|
||||
xl(at_db.xl),
|
||||
xh(at_db.xh),
|
||||
yl(at_db.yl),
|
||||
@ -95,6 +96,9 @@ public:
|
||||
const float* node_size_x;
|
||||
const float* node_size_y;
|
||||
|
||||
const float* node_weight;
|
||||
const bool* is_macro;
|
||||
|
||||
const float* pin_offset_x;
|
||||
const float* pin_offset_y;
|
||||
|
||||
@ -111,7 +115,6 @@ public:
|
||||
const int* node2fence_region_map;
|
||||
|
||||
const bool* net_mask;
|
||||
const float* node_weight;
|
||||
|
||||
/* chip info */
|
||||
float xl;
|
||||
@ -146,9 +149,8 @@ public:
|
||||
bin_size_x = (xh - xl) / num_bins_x_;
|
||||
bin_size_y = (yh - yl) / num_bins_y_;
|
||||
}
|
||||
inline bool is_dummy_fixed(int node_id) const {
|
||||
// DUMMY_FIXED_NUM_ROWS == 2
|
||||
return (node_id < num_movable_nodes && node_size_y[node_id] > (row_height * 2));
|
||||
inline bool is_mov_macro(int node_id) const {
|
||||
return (node_id < num_movable_nodes && is_macro[node_id]);
|
||||
}
|
||||
|
||||
inline float align2row(float y, float height) const {
|
||||
|
||||
@ -24,7 +24,7 @@ bool check_macro_legality(LegalizationData& db, const std::vector<int>& macros,
|
||||
float yh1 = yl1 + height1;
|
||||
float xh2 = xl2 + width2;
|
||||
float yh2 = yl2 + height2;
|
||||
if (std::min(xh1, xh2) > std::max(xl1, xl2) && std::min(yh1, yh2) > std::max(yl1, yl2)) {
|
||||
if (std::min(xh1, xh2) - std::max(xl1, xl2) > 1e-3 && std::min(yh1, yh2) - std::max(yl1, yl2) > 1e-3) {
|
||||
logger.error(
|
||||
"macro %d (%g, %g, %g, %g) var %d overlaps with macro %d "
|
||||
"(%g, %g, %g, %g) var %d, fixed: %d",
|
||||
@ -273,7 +273,7 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
|
||||
// collect macros
|
||||
std::vector<int> macros;
|
||||
for (int i = 0; i < db.num_movable_nodes; ++i) {
|
||||
if (db.is_dummy_fixed(i)) {
|
||||
if (db.is_mov_macro(i)) {
|
||||
// in some extreme case, some macros with 0 area should be ignored
|
||||
float area = db.node_size_x[i] * db.node_size_y[i];
|
||||
if (area > 0) {
|
||||
@ -281,7 +281,7 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
|
||||
}
|
||||
}
|
||||
}
|
||||
logger.info("Macro legalization: regard %lu cells as dummy fixed (movable macros)", macros.size());
|
||||
logger.info("Macro legalization: regard %lu cells as movable macros", macros.size());
|
||||
|
||||
// in case there is no movable macros
|
||||
if (macros.empty()) {
|
||||
@ -320,7 +320,19 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
|
||||
}
|
||||
};
|
||||
|
||||
// first round rough legalization with Hannan grid for clusters
|
||||
// 1) LP legalization, check displacement and legality
|
||||
auto displace = compute_displace(db, macros);
|
||||
logger.info("Macro displacement total %g, max %g, weighted total %g, max %g",
|
||||
displace.total_displace,
|
||||
displace.max_displace,
|
||||
displace.total_weighted_displace,
|
||||
displace.max_weighted_displace);
|
||||
bool legal = check_macro_legality(db, macros, true);
|
||||
update_best(legal, displace);
|
||||
|
||||
// 2) rough legalization with Hannan grid for clusters
|
||||
if (!legal) {
|
||||
logger.warning("LP not legal, try roughLegalize.");
|
||||
bool small_clusters_flag = true;
|
||||
bool blocked_macros_flag = false;
|
||||
roughLegalize(db, macros, fixed_macros, small_clusters_flag, blocked_macros_flag);
|
||||
@ -330,10 +342,13 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
|
||||
displace.max_displace,
|
||||
displace.total_weighted_displace,
|
||||
displace.max_weighted_displace);
|
||||
bool legal = check_macro_legality(db, macros, true);
|
||||
legal = check_macro_legality(db, macros, true);
|
||||
update_best(legal, displace);
|
||||
}
|
||||
|
||||
// try Hannan grid legalization if still not legal
|
||||
// 3) try Hannan grid legalization if still not legal
|
||||
if (!legal) {
|
||||
logger.warning("Not legal, try hannanLegalize.");
|
||||
legal = hannanLegalize(db, macros, fixed_macros, 10);
|
||||
auto displace = compute_displace(db, macros);
|
||||
logger.info("Macro displacement total %g, max %g, weighted total %g, max %g",
|
||||
@ -361,6 +376,7 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
|
||||
}
|
||||
}
|
||||
|
||||
if (legal) {
|
||||
logger.info("Align macros to site and rows");
|
||||
// align the lower left corner to row and site
|
||||
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
|
||||
@ -370,6 +386,12 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
|
||||
}
|
||||
|
||||
legal = check_macro_legality(db, macros, false);
|
||||
if (!legal) {
|
||||
logger.error("Macro legalization failed after aligning to site and row");
|
||||
}
|
||||
} else {
|
||||
logger.error("Macro legalization failed");
|
||||
}
|
||||
|
||||
return legal;
|
||||
}
|
||||
|
||||
@ -8,27 +8,6 @@ torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos,
|
||||
torch::Tensor net_mask,
|
||||
torch::Tensor hpwl_scale);
|
||||
|
||||
std::vector<torch::Tensor> wa_wirelength_cuda(torch::Tensor node_pos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
torch::Tensor pin_rel_cpos,
|
||||
torch::Tensor node2pin_list,
|
||||
torch::Tensor node2pin_list_end,
|
||||
torch::Tensor hyperedge_list,
|
||||
torch::Tensor hyperedge_list_end,
|
||||
torch::Tensor net_mask,
|
||||
float gamma,
|
||||
bool deterministic);
|
||||
|
||||
std::vector<torch::Tensor> wa_wirelength_hpwl_cuda(torch::Tensor node_pos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
torch::Tensor pin_rel_cpos,
|
||||
torch::Tensor node2pin_list,
|
||||
torch::Tensor node2pin_list_end,
|
||||
torch::Tensor hyperedge_list,
|
||||
torch::Tensor hyperedge_list_end,
|
||||
torch::Tensor net_mask,
|
||||
float gamma,
|
||||
bool deterministic);
|
||||
|
||||
std::vector<torch::Tensor> wa_wirelength_masked_scale_hpwl_cuda(torch::Tensor node_pos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
@ -66,68 +45,6 @@ torch::Tensor masked_scale_hpwl_sum(torch::Tensor node_pos,
|
||||
node_pos, pin_id2node_id, pin_rel_cpos, hyperedge_list, hyperedge_list_end, net_mask, hpwl_scale);
|
||||
}
|
||||
|
||||
std::vector<torch::Tensor> wa_wirelength(torch::Tensor node_pos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
torch::Tensor pin_rel_cpos,
|
||||
torch::Tensor node2pin_list,
|
||||
torch::Tensor node2pin_list_end,
|
||||
torch::Tensor hyperedge_list,
|
||||
torch::Tensor hyperedge_list_end,
|
||||
torch::Tensor net_mask,
|
||||
float gamma,
|
||||
bool deterministic) {
|
||||
CHECK_INPUT(node_pos);
|
||||
CHECK_INPUT(pin_id2node_id);
|
||||
CHECK_INPUT(pin_rel_cpos);
|
||||
CHECK_INPUT(node2pin_list);
|
||||
CHECK_INPUT(node2pin_list_end);
|
||||
CHECK_INPUT(hyperedge_list);
|
||||
CHECK_INPUT(hyperedge_list_end);
|
||||
CHECK_INPUT(net_mask);
|
||||
|
||||
return wa_wirelength_cuda(node_pos,
|
||||
pin_id2node_id,
|
||||
pin_rel_cpos,
|
||||
node2pin_list,
|
||||
node2pin_list_end,
|
||||
hyperedge_list,
|
||||
hyperedge_list_end,
|
||||
net_mask,
|
||||
gamma,
|
||||
deterministic);
|
||||
}
|
||||
|
||||
std::vector<torch::Tensor> wa_wirelength_hpwl(torch::Tensor node_pos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
torch::Tensor pin_rel_cpos,
|
||||
torch::Tensor node2pin_list,
|
||||
torch::Tensor node2pin_list_end,
|
||||
torch::Tensor hyperedge_list,
|
||||
torch::Tensor hyperedge_list_end,
|
||||
torch::Tensor net_mask,
|
||||
float gamma,
|
||||
bool deterministic) {
|
||||
CHECK_INPUT(node_pos);
|
||||
CHECK_INPUT(pin_id2node_id);
|
||||
CHECK_INPUT(pin_rel_cpos);
|
||||
CHECK_INPUT(node2pin_list);
|
||||
CHECK_INPUT(node2pin_list_end);
|
||||
CHECK_INPUT(hyperedge_list);
|
||||
CHECK_INPUT(hyperedge_list_end);
|
||||
CHECK_INPUT(net_mask);
|
||||
|
||||
return wa_wirelength_hpwl_cuda(node_pos,
|
||||
pin_id2node_id,
|
||||
pin_rel_cpos,
|
||||
node2pin_list,
|
||||
node2pin_list_end,
|
||||
hyperedge_list,
|
||||
hyperedge_list_end,
|
||||
net_mask,
|
||||
gamma,
|
||||
deterministic);
|
||||
}
|
||||
|
||||
std::vector<torch::Tensor> wa_wirelength_masked_scale_hpwl(torch::Tensor node_pos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
torch::Tensor pin_rel_cpos,
|
||||
@ -164,8 +81,6 @@ std::vector<torch::Tensor> wa_wirelength_masked_scale_hpwl(torch::Tensor node_po
|
||||
|
||||
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
|
||||
m.def("masked_scale_hpwl_sum", &masked_scale_hpwl_sum, "calculate the sum of scaled HPWL");
|
||||
m.def("merged_forward_backward", &wa_wirelength, "calculate WA wirelength and pin grad");
|
||||
m.def("merged_forward_backward_with_hpwl", &wa_wirelength_hpwl, "calculate WA wirelength, pin grad and hpwl");
|
||||
m.def("merged_forward_backward_with_masked_scale_hpwl",
|
||||
&wa_wirelength_masked_scale_hpwl,
|
||||
"calculate WA wirelength, pin grad and the scaled hpwl");
|
||||
|
||||
@ -74,134 +74,6 @@ __global__ void masked_scale_hpwl_cuda_kernel(
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void wa_wirelength_kernel(
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_pos,
|
||||
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list,
|
||||
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list_end,
|
||||
const torch::PackedTensorAccessor32<bool, 1, torch::RestrictPtrTraits> net_mask,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> partial_wa_wl,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_grad,
|
||||
int num_nets,
|
||||
float inv_gamma) {
|
||||
const int index = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
const int i = index >> 1; // net index
|
||||
if (i < num_nets && net_mask[i]) {
|
||||
const int c = index & 1; // channel index
|
||||
int64_t start_idx = 0;
|
||||
if (i != 0) {
|
||||
start_idx = hyperedge_list_end[i - 1];
|
||||
}
|
||||
int64_t end_idx = hyperedge_list_end[i];
|
||||
if (end_idx != start_idx) {
|
||||
int64_t pin_id = hyperedge_list[start_idx];
|
||||
float x_min = pin_pos[pin_id][c];
|
||||
float x_max = pin_pos[pin_id][c];
|
||||
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
|
||||
float xx = pin_pos[hyperedge_list[idx]][c];
|
||||
x_min = min(xx, x_min);
|
||||
x_max = max(xx, x_max);
|
||||
}
|
||||
|
||||
float xexp_x_sum = 0;
|
||||
float xexp_nx_sum = 0;
|
||||
float exp_x_sum = 0;
|
||||
float exp_nx_sum = 0;
|
||||
|
||||
for (int64_t idx = start_idx; idx < end_idx; idx++) {
|
||||
float xx = pin_pos[hyperedge_list[idx]][c];
|
||||
float exp_x = exp((xx - x_max) * inv_gamma);
|
||||
float exp_nx = exp((x_min - xx) * inv_gamma);
|
||||
|
||||
xexp_x_sum += xx * exp_x;
|
||||
xexp_nx_sum += xx * exp_nx;
|
||||
exp_x_sum += exp_x;
|
||||
exp_nx_sum += exp_nx;
|
||||
}
|
||||
|
||||
float wl = xexp_x_sum / exp_x_sum - xexp_nx_sum / exp_nx_sum;
|
||||
partial_wa_wl[i][c] = wl;
|
||||
|
||||
float b_x = inv_gamma / (exp_x_sum);
|
||||
float a_x = (1.0 - b_x * xexp_x_sum) / exp_x_sum;
|
||||
float b_nx = -inv_gamma / (exp_nx_sum);
|
||||
float a_nx = (1.0 - b_nx * xexp_nx_sum) / exp_nx_sum;
|
||||
|
||||
for (int64_t idx = start_idx; idx < end_idx; idx++) {
|
||||
float xx = pin_pos[hyperedge_list[idx]][c];
|
||||
float exp_x = exp((xx - x_max) * inv_gamma);
|
||||
float exp_nx = exp((x_min - xx) * inv_gamma);
|
||||
|
||||
pin_grad[hyperedge_list[idx]][c] = (a_x + b_x * xx) * exp_x - (a_nx + b_nx * xx) * exp_nx;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void wa_wirelength_hpwl_kernel(
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_pos,
|
||||
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list,
|
||||
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list_end,
|
||||
const torch::PackedTensorAccessor32<bool, 1, torch::RestrictPtrTraits> net_mask,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> partial_wa_wl,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> partial_hpwl,
|
||||
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_grad,
|
||||
int num_nets,
|
||||
float inv_gamma) {
|
||||
const int index = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
const int i = index >> 1; // net index
|
||||
if (i < num_nets && net_mask[i]) {
|
||||
const int c = index & 1; // channel index
|
||||
int64_t start_idx = 0;
|
||||
if (i != 0) {
|
||||
start_idx = hyperedge_list_end[i - 1];
|
||||
}
|
||||
int64_t end_idx = hyperedge_list_end[i];
|
||||
if (end_idx != start_idx) {
|
||||
int64_t pin_id = hyperedge_list[start_idx];
|
||||
float x_min = pin_pos[pin_id][c];
|
||||
float x_max = pin_pos[pin_id][c];
|
||||
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
|
||||
float xx = pin_pos[hyperedge_list[idx]][c];
|
||||
x_min = min(xx, x_min);
|
||||
x_max = max(xx, x_max);
|
||||
}
|
||||
partial_hpwl[i][c] = abs(x_max - x_min);
|
||||
|
||||
float xexp_x_sum = 0;
|
||||
float xexp_nx_sum = 0;
|
||||
float exp_x_sum = 0;
|
||||
float exp_nx_sum = 0;
|
||||
|
||||
for (int64_t idx = start_idx; idx < end_idx; idx++) {
|
||||
float xx = pin_pos[hyperedge_list[idx]][c];
|
||||
float exp_x = exp((xx - x_max) * inv_gamma);
|
||||
float exp_nx = exp((x_min - xx) * inv_gamma);
|
||||
|
||||
xexp_x_sum += xx * exp_x;
|
||||
xexp_nx_sum += xx * exp_nx;
|
||||
exp_x_sum += exp_x;
|
||||
exp_nx_sum += exp_nx;
|
||||
}
|
||||
|
||||
float wl = xexp_x_sum / exp_x_sum - xexp_nx_sum / exp_nx_sum;
|
||||
partial_wa_wl[i][c] = wl;
|
||||
|
||||
float b_x = inv_gamma / (exp_x_sum);
|
||||
float a_x = (1.0 - b_x * xexp_x_sum) / exp_x_sum;
|
||||
float b_nx = -inv_gamma / (exp_nx_sum);
|
||||
float a_nx = (1.0 - b_nx * xexp_nx_sum) / exp_nx_sum;
|
||||
|
||||
for (int64_t idx = start_idx; idx < end_idx; idx++) {
|
||||
float xx = pin_pos[hyperedge_list[idx]][c];
|
||||
float exp_x = exp((xx - x_max) * inv_gamma);
|
||||
float exp_nx = exp((x_min - xx) * inv_gamma);
|
||||
|
||||
pin_grad[hyperedge_list[idx]][c] = (a_x + b_x * xx) * exp_x - (a_nx + b_nx * xx) * exp_nx;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void wa_wirelength_masked_scale_hpwl_kernel(
|
||||
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_pos,
|
||||
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list,
|
||||
@ -296,57 +168,6 @@ void calc_node_grad_cuda(torch::Tensor node_grad,
|
||||
}
|
||||
}
|
||||
|
||||
std::vector<torch::Tensor> wa_wirelength_cuda(torch::Tensor node_pos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
torch::Tensor pin_rel_cpos,
|
||||
torch::Tensor node2pin_list,
|
||||
torch::Tensor node2pin_list_end,
|
||||
torch::Tensor hyperedge_list,
|
||||
torch::Tensor hyperedge_list_end,
|
||||
torch::Tensor net_mask,
|
||||
float gamma,
|
||||
bool deterministic) {
|
||||
cudaSetDevice(node_pos.get_device());
|
||||
auto stream = at::cuda::getCurrentCUDAStream();
|
||||
|
||||
const auto num_nodes = node_pos.size(0);
|
||||
const auto num_pins = pin_id2node_id.size(0);
|
||||
const auto num_nets = hyperedge_list_end.size(0);
|
||||
const auto num_channels = 2; // x, y
|
||||
|
||||
auto pin_pos = pin_rel_cpos.clone(); // pin
|
||||
auto partial_wa_wl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
|
||||
auto pin_grad = torch::zeros({num_pins, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
|
||||
|
||||
const int threads = 128;
|
||||
const int blocks = (num_pins * 2 + threads - 1) / threads;
|
||||
|
||||
node_pos_to_pin_pos_cuda_kernel<<<blocks, threads, 0, stream>>>(
|
||||
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
pin_id2node_id.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
num_pins);
|
||||
|
||||
const int threads2 = 128;
|
||||
const int blocks2 = (num_nets * 2 + threads2 - 1) / threads2;
|
||||
|
||||
float inv_gamma = 1 / gamma;
|
||||
wa_wirelength_kernel<<<blocks2, threads2, 0, stream>>>(
|
||||
pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
hyperedge_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
hyperedge_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
net_mask.packed_accessor32<bool, 1, torch::RestrictPtrTraits>(),
|
||||
partial_wa_wl.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
pin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
num_nets,
|
||||
inv_gamma);
|
||||
|
||||
auto node_grad = torch::zeros({num_nodes, num_channels}, torch::dtype(pin_grad.dtype()).device(pin_grad.device()));
|
||||
calc_node_grad_cuda(
|
||||
node_grad, pin_id2node_id, pin_grad, node2pin_list, node2pin_list_end, num_nodes, deterministic);
|
||||
|
||||
return {partial_wa_wl, node_grad};
|
||||
}
|
||||
|
||||
torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
@ -391,60 +212,6 @@ torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos,
|
||||
return total_hpwl;
|
||||
}
|
||||
|
||||
std::vector<torch::Tensor> wa_wirelength_hpwl_cuda(torch::Tensor node_pos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
torch::Tensor pin_rel_cpos,
|
||||
torch::Tensor node2pin_list,
|
||||
torch::Tensor node2pin_list_end,
|
||||
torch::Tensor hyperedge_list,
|
||||
torch::Tensor hyperedge_list_end,
|
||||
torch::Tensor net_mask,
|
||||
float gamma,
|
||||
bool deterministic) {
|
||||
cudaSetDevice(node_pos.get_device());
|
||||
auto stream = at::cuda::getCurrentCUDAStream();
|
||||
|
||||
const auto num_nodes = node_pos.size(0);
|
||||
const auto num_pins = pin_id2node_id.size(0);
|
||||
const auto num_nets = hyperedge_list_end.size(0);
|
||||
const auto num_channels = 2; // x, y
|
||||
|
||||
auto pin_pos = pin_rel_cpos.clone(); // pin
|
||||
auto partial_wa_wl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
|
||||
auto partial_hpwl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
|
||||
auto pin_grad = torch::zeros({num_pins, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
|
||||
|
||||
const int threads = 128;
|
||||
const int blocks = (num_pins * 2 + threads - 1) / threads;
|
||||
|
||||
node_pos_to_pin_pos_cuda_kernel<<<blocks, threads, 0, stream>>>(
|
||||
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
pin_id2node_id.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
num_pins);
|
||||
|
||||
const int threads2 = 128;
|
||||
const int blocks2 = (num_nets * 2 + threads2 - 1) / threads2;
|
||||
|
||||
float inv_gamma = 1 / gamma;
|
||||
wa_wirelength_hpwl_kernel<<<blocks2, threads2, 0, stream>>>(
|
||||
pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
hyperedge_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
hyperedge_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||
net_mask.packed_accessor32<bool, 1, torch::RestrictPtrTraits>(),
|
||||
partial_wa_wl.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
partial_hpwl.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
pin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||
num_nets,
|
||||
inv_gamma);
|
||||
|
||||
auto node_grad = torch::zeros({num_nodes, num_channels}, torch::dtype(pin_grad.dtype()).device(pin_grad.device()));
|
||||
calc_node_grad_cuda(
|
||||
node_grad, pin_id2node_id, pin_grad, node2pin_list, node2pin_list_end, num_nodes, deterministic);
|
||||
|
||||
return {partial_wa_wl, node_grad, partial_hpwl};
|
||||
}
|
||||
|
||||
std::vector<torch::Tensor> wa_wirelength_masked_scale_hpwl_cuda(torch::Tensor node_pos,
|
||||
torch::Tensor pin_id2node_id,
|
||||
torch::Tensor pin_rel_cpos,
|
||||
|
||||
@ -7,6 +7,18 @@ tar xvzf ispd2005.tar.gz
|
||||
rm -rf ispd2005.tar.gz
|
||||
mv ispd2005/ raw/
|
||||
|
||||
echo "=== Downloading ispd2006"
|
||||
wget --no-check-certificate "https://mycuhk-my.sharepoint.com/:u:/g/personal/1155136644_link_cuhk_edu_hk/EYBauANXekFAn1nlKRtec8YBecrfXNmocajWhqNfKWhRvA?e=wL1Z5z&download=1" -O ispd2006.tar.gz
|
||||
tar xvzf ispd2006.tar.gz
|
||||
rm -rf ispd2006.tar.gz
|
||||
mv ispd2006/ raw/
|
||||
|
||||
echo "=== Downloading mms"
|
||||
wget --no-check-certificate "https://mycuhk-my.sharepoint.com/:u:/g/personal/1155136644_link_cuhk_edu_hk/EQpwvzotaWBGlIm9zpOfIL4B_DRFb5jNOQ5mKHLd7yrByw?e=ovrb3e&download=1" -O mms.tar.gz
|
||||
tar xvzf mms.tar.gz
|
||||
rm -rf mms.tar.gz
|
||||
mv mms/ raw/
|
||||
|
||||
echo "=== Downloading ispd2015 ==="
|
||||
wget --no-check-certificate "https://mycuhk-my.sharepoint.com/:u:/g/personal/1155136644_link_cuhk_edu_hk/Ea4YjKNvi-9CnekS41Pw-GgBEhIRNnp6AhMDU9_xElLjNA?e=YSUMhQ&download=1" -O ispd2015.tar.gz
|
||||
tar xvzf ispd2015.tar.gz
|
||||
|
||||
6
main.py
6
main.py
@ -22,21 +22,19 @@ def get_option():
|
||||
parser.add_argument('--wa_coeff', type=float, default=4.0, help='wa coeff')
|
||||
parser.add_argument('--num_bin_x', type=int, default=512, help='#binX for density function')
|
||||
parser.add_argument('--num_bin_y', type=int, default=512, help='#binY for density function')
|
||||
parser.add_argument('--threshold', type=float, default=4.0, help='normalized node area threshold for using naive mode')
|
||||
parser.add_argument('--density_weight', type=float, default=8e-5, help='the weight of density loss')
|
||||
parser.add_argument('--density_weight_coef', type=float, default=1.05, help='the ratio of density_weight')
|
||||
parser.add_argument('--use_init_density_weight', type=str2bool, default=True, help='enable dynamic initialization of density_weight')
|
||||
parser.add_argument('--target_density', type=float, default=1.0, help='placement target density')
|
||||
parser.add_argument('--use_filler', type=str2bool, default=True, help='placement filler')
|
||||
parser.add_argument('--noise_ratio', type=float, default=0.025, help='noise ratio for initialization')
|
||||
parser.add_argument('--ignore_net_degree', type=int, default=100, help='threshold of net degree to ignore in wirelength calculation')
|
||||
parser.add_argument('--scale_design', type=str2bool, default=False, help='normalize die area')
|
||||
parser.add_argument('--use_eplace_nesterov', type=str2bool, default=True, help='enable eplace nesterov optimizer')
|
||||
parser.add_argument('--clamp_node', type=str2bool, default=True, help='enable eplace node clamp trick')
|
||||
parser.add_argument('--use_precond', type=str2bool, default=True, help='apply precond')
|
||||
parser.add_argument('--stop_overflow', type=float, default=0.07, help='stop overflow in scheduler')
|
||||
parser.add_argument('--enable_skip_update', type=str2bool, default=True, help='enable skip update')
|
||||
parser.add_argument("--loss_type", type=str, default="direct", help="loss type")
|
||||
parser.add_argument('--enable_sample_force', type=str2bool, default=True, help='enable sample force')
|
||||
parser.add_argument("--mixed_size", type=str2bool, default=False, help="enable mixed size placement")
|
||||
|
||||
# global routing params
|
||||
parser.add_argument('--use_cell_inflate', type=str2bool, default=False, help='use cell inflation')
|
||||
|
||||
@ -5,6 +5,6 @@ from .evaluator import *
|
||||
from .initializer import *
|
||||
from .nesterov_optimizer import NesterovOptimizer
|
||||
from .param_scheduler import ParamScheduler
|
||||
from .detail_placement import detail_placement_main
|
||||
from .detail_placement import detail_placement_main, macro_legalization_main
|
||||
from .run_placement_nesterov import run_placement_main_nesterov
|
||||
from .run_placement import run_placement_main
|
||||
@ -1,16 +1,6 @@
|
||||
import torch
|
||||
from .param_scheduler import ParamScheduler
|
||||
from .core import merged_wl_loss_grad, WAWirelengthLoss, WAWirelengthLossAndHPWL
|
||||
|
||||
|
||||
def calc_loss(wl_loss, density_loss, ps, args):
|
||||
if args.loss_type == "weighted_sum":
|
||||
loss = (wl_loss + ps.density_weight * density_loss) / (1 + ps.density_weight)
|
||||
elif args.loss_type == "direct":
|
||||
loss = wl_loss + ps.density_weight * density_loss
|
||||
else:
|
||||
raise NotImplementedError("Loss type not defined")
|
||||
return loss
|
||||
from .core import merged_wl_loss_grad
|
||||
|
||||
|
||||
def apply_precond(mov_node_pos: torch.Tensor, ps: ParamScheduler, args):
|
||||
@ -38,6 +28,8 @@ def calc_obj_and_grad(
|
||||
mov_node_pos = constraint_fn(mov_node_pos)
|
||||
conn_node_pos = mov_node_pos[mov_lhs:mov_rhs, ...]
|
||||
conn_node_pos = torch.cat([conn_node_pos, conn_fix_node_pos], dim=0)
|
||||
|
||||
assert merged_forward_backward
|
||||
if merged_forward_backward:
|
||||
if mov_node_pos.grad is not None:
|
||||
mov_node_pos.grad.zero_()
|
||||
@ -81,62 +73,11 @@ def calc_obj_and_grad(
|
||||
)
|
||||
mov_node_pos.grad += node_grad_by_density * ps.density_weight
|
||||
|
||||
if ps.zero_macro_grad:
|
||||
mov_node_pos.grad[mov_lhs:mov_rhs].masked_fill_(data.is_mov_macro[mov_lhs:mov_rhs].unsqueeze(1), 0)
|
||||
|
||||
grad = apply_precond(mov_node_pos, ps, args)
|
||||
loss = wl_loss + ps.density_weight * density_loss
|
||||
else:
|
||||
if mov_node_pos.grad is not None:
|
||||
mov_node_pos.grad.zero_()
|
||||
else:
|
||||
mov_node_pos.grad = torch.zeros_like(mov_node_pos).detach()
|
||||
wl_loss = WAWirelengthLoss.apply(
|
||||
conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos,
|
||||
data.node2pin_list, data.node2pin_list_end,
|
||||
data.hyperedge_list, data.hyperedge_list_end, data.net_mask,
|
||||
ps.wa_coeff, args.deterministic
|
||||
)
|
||||
density_loss, _ = density_map_layer(
|
||||
mov_node_pos, mov_node_size, init_density_map, calc_overflow=False
|
||||
)
|
||||
loss = calc_loss(wl_loss, density_loss, ps, args)
|
||||
loss.backward()
|
||||
grad = apply_precond(mov_node_pos, ps, args)
|
||||
|
||||
return loss, grad
|
||||
|
||||
|
||||
def calc_grad(
|
||||
optimizer: torch.optim.Optimizer, mov_node_pos: torch.Tensor, wl_loss, density_loss
|
||||
):
|
||||
optimizer.zero_grad(set_to_none=False)
|
||||
wl_loss.backward(retain_graph=True)
|
||||
wl_grad = mov_node_pos.grad.detach().clone()
|
||||
optimizer.zero_grad(set_to_none=False)
|
||||
density_loss.backward(retain_graph=True)
|
||||
density_grad = mov_node_pos.grad.detach().clone()
|
||||
optimizer.zero_grad(set_to_none=False)
|
||||
return wl_grad, density_grad
|
||||
|
||||
|
||||
def fast_optimization(
|
||||
mov_node_pos, trunc_node_pos_fn, mov_lhs, mov_rhs, conn_fix_node_pos,
|
||||
density_map_layer, mov_node_size, init_density_map, ps, data, args
|
||||
):
|
||||
mov_node_pos = trunc_node_pos_fn(mov_node_pos)
|
||||
conn_node_pos = mov_node_pos[mov_lhs:mov_rhs, ...]
|
||||
conn_node_pos = torch.cat(
|
||||
[conn_node_pos, conn_fix_node_pos], dim=0
|
||||
)
|
||||
wl_loss, hpwl = WAWirelengthLossAndHPWL.apply(
|
||||
conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos,
|
||||
data.node2pin_list, data.node2pin_list_end,
|
||||
data.hyperedge_list, data.hyperedge_list_end, data.net_mask,
|
||||
ps.wa_coeff, data.hpwl_scale, args.deterministic
|
||||
)
|
||||
density_loss, overflow = density_map_layer(
|
||||
mov_node_pos, mov_node_size, init_density_map
|
||||
)
|
||||
loss = calc_loss(wl_loss, density_loss, ps, args)
|
||||
loss.backward()
|
||||
apply_precond(mov_node_pos, ps, args)
|
||||
# calculate objective (hpwl, overflow)
|
||||
return hpwl.detach(), overflow.detach(), mov_node_pos
|
||||
@ -1,4 +1,4 @@
|
||||
from .flute import Flute, get_flute_wl
|
||||
from .electronic_density_layer import ElectronicDensityLayer
|
||||
from .wa_wirelength_hpwl import WAWirelengthLossAndHPWL, WAWirelengthLoss, masked_scale_hpwl, merged_wl_loss_grad
|
||||
from .wa_wirelength_hpwl import masked_scale_hpwl, merged_wl_loss_grad
|
||||
from .route_force import get_route_force, run_gr_and_fft, run_gr_and_fft_main, route_inflation, route_inflation_roll_back
|
||||
@ -262,35 +262,6 @@ class ElectronicDensityLayer(torch.nn.Module):
|
||||
|
||||
return node_weight
|
||||
|
||||
def get_density_map_naive(
|
||||
self,
|
||||
node_pos,
|
||||
node_size,
|
||||
init_density_map=None,
|
||||
):
|
||||
node_weight = node_size.new_ones(node_pos.shape[0])
|
||||
if init_density_map is None:
|
||||
node_pos.new_zeros(self.num_bin_x, self.num_bin_y)
|
||||
aux_mat = init_density_map.clone()
|
||||
num_nodes = node_pos.shape[0]
|
||||
density_map = density_map_cuda.forward_naive(
|
||||
node_pos,
|
||||
node_size,
|
||||
node_weight,
|
||||
self.unit_len,
|
||||
aux_mat,
|
||||
self.num_bin_x,
|
||||
self.num_bin_y,
|
||||
num_nodes,
|
||||
-1.0,
|
||||
-1.0,
|
||||
1e-4,
|
||||
False,
|
||||
self.deterministic,
|
||||
)
|
||||
|
||||
return density_map
|
||||
|
||||
def direct_calc_overflow(
|
||||
self,
|
||||
node_pos,
|
||||
|
||||
947
src/core/macro_legalization.py
Normal file
947
src/core/macro_legalization.py
Normal file
@ -0,0 +1,947 @@
|
||||
# We follow the work [1] and [2] to implement the macro legalizer.
|
||||
# [1] Cong, Jason, and Min Xie. "A robust mixed-size legalization and detailed placement algorithm." IEEE TCAD 2008.
|
||||
# [2] Moffitt, M. D., Ng, A. N., Markov, I. L., & Pollack, M. E. "Constraint-driven floorplan repair." ACM TODAES 2008.
|
||||
|
||||
# NOTE: Known Issue: Adding graph edge constraint in LP is very slow when num_macros is large.
|
||||
# Because pulp lib use Python OrderDict to store all constraints, when #Constraints
|
||||
# is large, the performance is bad.
|
||||
# Re-write the LP in C++ may achieve some speed up.
|
||||
# TODO: 1) macro spreading for routability optimization
|
||||
# 2) graph pruning for speed up
|
||||
|
||||
import numpy as np
|
||||
import numba as nb
|
||||
import logging
|
||||
import pulp as pl
|
||||
import igraph as ig
|
||||
|
||||
|
||||
pulp_logger = logging.getLogger('pulp')
|
||||
pulp_logger.setLevel(logging.INFO)
|
||||
use_numba_parallel = False
|
||||
|
||||
@nb.jit(cache=True, parallel=use_numba_parallel)
|
||||
def check_macro_legality(macro_pos, macro_size, macro_fixed, die_info, check_all=True):
|
||||
num_macros = macro_pos.shape[0]
|
||||
# overlap = np.zeros((num_macros, num_macros), dtype=np.bool8)
|
||||
legal = True
|
||||
for i in nb.prange(num_macros):
|
||||
lx_i = macro_pos[i][0] - macro_size[i][0] / 2
|
||||
ly_i = macro_pos[i][1] - macro_size[i][1] / 2
|
||||
hx_i = macro_pos[i][0] + macro_size[i][0] / 2
|
||||
hy_i = macro_pos[i][1] + macro_size[i][1] / 2
|
||||
for j in nb.prange(num_macros):
|
||||
if i >= j:
|
||||
continue
|
||||
lx_j = macro_pos[j][0] - macro_size[j][0] / 2
|
||||
ly_j = macro_pos[j][1] - macro_size[j][1] / 2
|
||||
hx_j = macro_pos[j][0] + macro_size[j][0] / 2
|
||||
hy_j = macro_pos[j][1] + macro_size[j][1] / 2
|
||||
if min(hx_i, hx_j) - max(lx_i, lx_j) > 1e-3 and min(hy_i, hy_j) - max(ly_i, ly_j) > 1e-3:
|
||||
# overlap[i][j] = True
|
||||
# overlap[j][i] = True
|
||||
legal = False
|
||||
if not check_all and not legal:
|
||||
return legal
|
||||
print("Macro", i, "and Macro", j, "Overlap.")
|
||||
|
||||
return legal
|
||||
|
||||
|
||||
|
||||
@nb.jit(nopython=True, nogil=True, cache=True, parallel=use_numba_parallel)
|
||||
def constraint_graph_construction(
|
||||
macro_pos, macro_size, macro_fixed, die_info, prune=True
|
||||
):
|
||||
num_macros = macro_pos.shape[0]
|
||||
macro_lpos = macro_pos - macro_size / 2
|
||||
edge_type = np.zeros((num_macros, num_macros), dtype=np.int8) # 0: x, 1: y, -1: None
|
||||
edge_dist_x = np.zeros((num_macros, num_macros), dtype=np.float32)
|
||||
edge_dist_y = np.zeros((num_macros, num_macros), dtype=np.float32)
|
||||
edge_type[:, :] = -1
|
||||
for i in nb.prange(num_macros):
|
||||
for j in nb.prange(num_macros):
|
||||
if i >= j:
|
||||
continue
|
||||
# Detect x/y order
|
||||
if macro_pos[i][0] <= macro_pos[j][0]:
|
||||
# x_i -> x_j
|
||||
x_order = 0
|
||||
else:
|
||||
# x_j -> x_i
|
||||
x_order = 1
|
||||
if macro_pos[i][1] <= macro_pos[j][1]:
|
||||
# y_i -> y_j
|
||||
y_order = 0
|
||||
else:
|
||||
# y_j -> y_i
|
||||
y_order = 1
|
||||
|
||||
# Calculate displacement
|
||||
lx_i = macro_pos[i][0] - macro_size[i][0] / 2
|
||||
lx_j = macro_pos[j][0] - macro_size[j][0] / 2
|
||||
hx_i = macro_pos[i][0] + macro_size[i][0] / 2
|
||||
hx_j = macro_pos[j][0] + macro_size[j][0] / 2
|
||||
if x_order == 0:
|
||||
dist_x = lx_j - hx_i
|
||||
else:
|
||||
dist_x = lx_i - hx_j
|
||||
|
||||
ly_i = macro_pos[i][1] - macro_size[i][1] / 2
|
||||
ly_j = macro_pos[j][1] - macro_size[j][1] / 2
|
||||
hy_i = macro_pos[i][1] + macro_size[i][1] / 2
|
||||
hy_j = macro_pos[j][1] + macro_size[j][1] / 2
|
||||
if y_order == 0:
|
||||
dist_y = ly_j - hy_i
|
||||
else:
|
||||
dist_y = ly_i - hy_j
|
||||
|
||||
# dist martix is undirected and symmetric
|
||||
edge_dist_x[i][j] = dist_x
|
||||
edge_dist_x[j][i] = dist_x
|
||||
edge_dist_y[i][j] = dist_y
|
||||
edge_dist_y[j][i] = dist_y
|
||||
|
||||
# Determine the edge type (horizontal or vertical)
|
||||
if dist_x >= 0 and dist_y >= 0:
|
||||
# non-overlap
|
||||
edge_type[i][j] = 0 if dist_x >= dist_y else 1
|
||||
elif dist_x >= 0 and dist_y < 0:
|
||||
# y projection overlap
|
||||
edge_type[i][j] = 0
|
||||
elif dist_x < 0 and dist_y >= 0:
|
||||
# x projection overlap
|
||||
edge_type[i][j] = 1
|
||||
elif dist_x < 0 and dist_y < 0:
|
||||
# overlap
|
||||
edge_type[i][j] = 0 if dist_x >= dist_y else 1
|
||||
|
||||
# Prune edges between objects without x/y projection overlap
|
||||
if prune:
|
||||
if edge_type[i][j] == 0 and not (ly_i <= hy_j and ly_j <= hy_i):
|
||||
edge_type[i][j] = -1
|
||||
if edge_type[i][j] == 1 and not (lx_i <= hx_j and lx_j <= hx_i):
|
||||
edge_type[i][j] = -1
|
||||
|
||||
# Make sure edge orders
|
||||
if edge_type[i][j] == 0 and x_order == 1:
|
||||
edge_type[i][j] = -1
|
||||
edge_type[j][i] = 0
|
||||
elif edge_type[i][j] == 1 and y_order == 1:
|
||||
edge_type[i][j] = -1
|
||||
edge_type[j][i] = 1
|
||||
|
||||
return edge_type, edge_dist_x, edge_dist_y
|
||||
|
||||
|
||||
@nb.jit(nopython=True, nogil=True, cache=True, parallel=use_numba_parallel)
|
||||
def initialize_xy_adj_weight(
|
||||
edge_type, adj_matrix, weight_matrix, macro_size, num_macros, num_nodes, s_id, t_id
|
||||
):
|
||||
adj_matrix[s_id, :num_macros, :] = 1
|
||||
adj_matrix[:num_macros, t_id, :] = 1
|
||||
|
||||
for i in nb.prange(num_nodes):
|
||||
for j in nb.prange(num_nodes):
|
||||
if i == j:
|
||||
continue
|
||||
if i < num_macros and j < num_macros:
|
||||
if edge_type[i][j] == 0:
|
||||
# horizontal
|
||||
adj_matrix[i][j][0] = 1
|
||||
elif edge_type[i][j] == 1:
|
||||
adj_matrix[i][j][1] = 1
|
||||
if i >= num_macros:
|
||||
weight_matrix[i][j][0] = np.divide(macro_size[j][0], 2)
|
||||
weight_matrix[i][j][1] = np.divide(macro_size[j][1], 2)
|
||||
elif j >= num_macros:
|
||||
weight_matrix[i][j][0] = np.divide(macro_size[i][0], 2)
|
||||
weight_matrix[i][j][1] = np.divide(macro_size[i][1], 2)
|
||||
else:
|
||||
weight_matrix[i][j][0] = np.divide(np.add(macro_size[i][0], macro_size[j][0]), 2)
|
||||
weight_matrix[i][j][1] = np.divide(np.add(macro_size[i][1], macro_size[j][1]), 2)
|
||||
|
||||
|
||||
@nb.jit(nopython=True, nogil=True, cache=True)
|
||||
def compute_L_value(
|
||||
macro_pos, macro_fixed, axis, topo_order_out, L, affected_L, adj_matrix, weight_matrix, die_ll, s_id, num_macros
|
||||
):
|
||||
for i in topo_order_out:
|
||||
if affected_L[i, axis] == 0:
|
||||
continue
|
||||
if i < num_macros:
|
||||
if macro_fixed[i]:
|
||||
L[i, axis] = macro_pos[i, axis]
|
||||
continue
|
||||
if i == s_id:
|
||||
L[i, axis] = die_ll[axis]
|
||||
else:
|
||||
is_preds = adj_matrix[:, i, axis]
|
||||
for j, is_pred in enumerate(is_preds):
|
||||
if is_pred == 0 or i == j:
|
||||
continue
|
||||
# j -> i
|
||||
L[i, axis] = max(L[j, axis] + weight_matrix[j][i][axis], L[i, axis])
|
||||
|
||||
|
||||
@nb.jit(nopython=True, nogil=True, cache=True)
|
||||
def compute_R_value(
|
||||
macro_pos, macro_fixed, axis, topo_order_in, R, affected_R, adj_matrix, weight_matrix, die_ur, t_id, num_macros
|
||||
):
|
||||
for i in topo_order_in:
|
||||
if affected_R[i, axis] == 0:
|
||||
continue
|
||||
if i < num_macros:
|
||||
if macro_fixed[i]:
|
||||
R[i, axis] = macro_pos[i, axis]
|
||||
continue
|
||||
if i == t_id:
|
||||
R[i, axis] = die_ur[axis]
|
||||
else:
|
||||
is_succs = adj_matrix[i, :, axis]
|
||||
for j, is_succ in enumerate(is_succs):
|
||||
if is_succ == 0 or i == j:
|
||||
continue
|
||||
# i -> j
|
||||
R[i, axis] = min(R[j, axis] - weight_matrix[i][j][axis], R[i, axis])
|
||||
|
||||
|
||||
def propagate_L_R(
|
||||
g, macro_pos, macro_fixed, affected_L, affected_R, adj_matrix, weight_matrix,
|
||||
L, R, die_ll, die_ur, s_id, t_id, num_macros, edges_pair=None
|
||||
):
|
||||
topo_order_out_X = nb.typed.List(g[0].topological_sorting(mode="out"))
|
||||
topo_order_in_X = nb.typed.List(g[0].topological_sorting(mode="in"))
|
||||
topo_order_out_Y = nb.typed.List(g[1].topological_sorting(mode="out"))
|
||||
topo_order_in_Y = nb.typed.List(g[1].topological_sorting(mode="in"))
|
||||
|
||||
if edges_pair:
|
||||
axis_del, (u_del, v_del), axis_add, (u_add, v_add) = edges_pair
|
||||
affected_L[:, :] = False
|
||||
affected_R[:, :] = False
|
||||
# In the old graph, u may not be topologically <= v
|
||||
bfs_order, _, _ = g[axis_del].bfs(u_del, mode='out')
|
||||
affected_L[np.array(bfs_order), axis_del] = True
|
||||
bfs_order, _, _ = g[axis_del].bfs(v_del, mode='out')
|
||||
affected_L[np.array(bfs_order), axis_del] = True
|
||||
bfs_order, _, _ = g[axis_del].bfs(u_del, mode='in')
|
||||
affected_R[np.array(bfs_order), axis_del] = True
|
||||
bfs_order, _, _ = g[axis_del].bfs(v_del, mode='in')
|
||||
affected_R[np.array(bfs_order), axis_del] = True
|
||||
# In the new graph, u -> v, so u should be topologically <= v
|
||||
bfs_order, _, _ = g[axis_add].bfs(u_add, mode='out')
|
||||
affected_L[np.array(bfs_order), axis_add] = True
|
||||
bfs_order, _, _ = g[axis_add].bfs(v_add, mode='in')
|
||||
affected_R[np.array(bfs_order), axis_add] = True
|
||||
|
||||
L[affected_L] = -np.inf
|
||||
R[affected_R] = np.inf
|
||||
compute_L_value(macro_pos, macro_fixed, 0, topo_order_out_X, L, affected_L,
|
||||
adj_matrix, weight_matrix, die_ll, s_id, num_macros)
|
||||
compute_R_value(macro_pos, macro_fixed, 0, topo_order_in_X, R, affected_R,
|
||||
adj_matrix, weight_matrix, die_ur, t_id, num_macros)
|
||||
compute_L_value(macro_pos, macro_fixed, 1, topo_order_out_Y, L, affected_L,
|
||||
adj_matrix, weight_matrix, die_ll, s_id, num_macros)
|
||||
compute_R_value(macro_pos, macro_fixed, 1, topo_order_in_Y, R, affected_R,
|
||||
adj_matrix, weight_matrix, die_ur, t_id, num_macros)
|
||||
|
||||
|
||||
@nb.jit(nopython=True, nogil=True, cache=True, parallel=use_numba_parallel)
|
||||
def compute_edge_slack(
|
||||
adj_matrix, weight_matrix, L, R, num_nodes, slack_matrix_e,
|
||||
):
|
||||
for i in nb.prange(num_nodes):
|
||||
for j in nb.prange(num_nodes):
|
||||
if adj_matrix[i][j][0] == 1:
|
||||
slack_matrix_e[i][j][0] = R[j][0] - L[i][0] - weight_matrix[i][j][0]
|
||||
if adj_matrix[i][j][1] == 1:
|
||||
slack_matrix_e[i][j][1] = R[j][1] - L[i][1] - weight_matrix[i][j][1]
|
||||
|
||||
|
||||
def slack_info(
|
||||
adj_matrix, weight_matrix, L, R, num_macros, num_nodes, slack_v, slack_matrix_e,
|
||||
update_edge_slack=True,
|
||||
):
|
||||
# Node Slack
|
||||
slack_v[:num_macros, :] = R[:num_macros, :] - L[:num_macros, :]
|
||||
x_nslack = np.minimum(slack_v[:, 0], 0)
|
||||
y_nslack = np.minimum(slack_v[:, 1], 0)
|
||||
x_tns, x_wns, nonzero_x = np.sum(x_nslack), np.min(x_nslack), np.sum(x_nslack < 0)
|
||||
y_tns, y_wns, nonzero_y = np.sum(y_nslack), np.min(y_nslack), np.sum(y_nslack < 0)
|
||||
info_v = (x_tns, x_wns, nonzero_x, y_tns, y_wns, nonzero_y)
|
||||
if update_edge_slack:
|
||||
slack_matrix_e[:,:,:] = 0
|
||||
compute_edge_slack(adj_matrix, weight_matrix, L, R, num_nodes, slack_matrix_e)
|
||||
x_nslack = np.minimum(slack_matrix_e[:, :, 0], 0)
|
||||
y_nslack = np.minimum(slack_matrix_e[:, :, 1], 0)
|
||||
x_tns, x_wns, nonzero_x = np.sum(x_nslack), np.min(x_nslack), np.sum(x_nslack < 0)
|
||||
y_tns, y_wns, nonzero_y = np.sum(y_nslack), np.min(y_nslack), np.sum(y_nslack < 0)
|
||||
info_e = (x_tns, x_wns, nonzero_x, y_tns, y_wns, nonzero_y)
|
||||
return info_v, info_e
|
||||
return info_v, None
|
||||
|
||||
|
||||
@nb.jit(nopython=True, nogil=True, cache=True)
|
||||
def mark_edge_to_move(adj_matrix, weight_matrix, slack_v, i, L, R, s_id, t_id, macro_pos):
|
||||
edges_pair = []
|
||||
x_slack = slack_v[i, 0]
|
||||
y_slack = slack_v[i, 1]
|
||||
if (x_slack >= 0 and y_slack >= 0) or (x_slack < 0 and y_slack < 0):
|
||||
return edges_pair
|
||||
if x_slack >= 0 and y_slack < 0:
|
||||
# need to handle in g[1]
|
||||
axis = 1
|
||||
else:
|
||||
# need to handle in g[0]
|
||||
axis = 0
|
||||
o_axis = 0 if axis == 1 else 1
|
||||
is_preds = adj_matrix[:, i, axis]
|
||||
for j, is_pred in enumerate(is_preds):
|
||||
# j -> i
|
||||
if is_pred == 0 or i == j:
|
||||
continue
|
||||
if j == s_id or j == t_id:
|
||||
continue
|
||||
if L[i][axis] == L[j][axis] + weight_matrix[j][i][axis]:
|
||||
edge_ij = False
|
||||
edge_ji = False
|
||||
if L[i][o_axis] + weight_matrix[i][j][o_axis] <= R[j][o_axis]:
|
||||
edge_ij = True
|
||||
is_succs_j = adj_matrix[j, :, o_axis]
|
||||
for k, is_succ in enumerate(is_succs_j):
|
||||
# i -> j -> k
|
||||
if is_succ == 0 or k == j:
|
||||
continue
|
||||
if L[i][o_axis] + weight_matrix[i][j][o_axis] + weight_matrix[j][k][o_axis] > R[k][o_axis]:
|
||||
edge_ij = False
|
||||
break
|
||||
if L[j][o_axis] + weight_matrix[j][k][o_axis] > R[k][o_axis]:
|
||||
edge_ij = False
|
||||
break
|
||||
if L[j][o_axis] + weight_matrix[j][i][o_axis] <= R[i][o_axis]:
|
||||
edge_ji = True
|
||||
is_succs_i = adj_matrix[i, :, o_axis]
|
||||
for k, is_succ in enumerate(is_succs_i):
|
||||
# j -> i -> k
|
||||
if is_succ == 0 or k == i:
|
||||
continue
|
||||
if L[j][o_axis] + weight_matrix[j][i][o_axis] + weight_matrix[i][k][o_axis] > R[k][o_axis]:
|
||||
edge_ij = False
|
||||
break
|
||||
if L[i][o_axis] + weight_matrix[i][k][o_axis] > R[k][o_axis]:
|
||||
edge_ij = False
|
||||
break
|
||||
if edge_ij and edge_ji:
|
||||
if macro_pos[i][o_axis] <= macro_pos[j][o_axis]:
|
||||
# i -> j
|
||||
edges_pair.append((axis, (j, i), o_axis, (i, j)))
|
||||
else:
|
||||
# j -> i
|
||||
edges_pair.append((axis, (j, i), o_axis, (j, i)))
|
||||
elif edge_ij:
|
||||
edges_pair.append((axis, (j, i), o_axis, (i, j)))
|
||||
elif edge_ji:
|
||||
edges_pair.append((axis, (j, i), o_axis, (j, i)))
|
||||
else:
|
||||
edges_pair.clear()
|
||||
|
||||
return edges_pair
|
||||
|
||||
|
||||
def longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur, logger, naive=False, prune=False):
|
||||
edge_type, _, _ = constraint_graph_construction(macro_pos, macro_size, macro_fixed, die_info, prune=prune)
|
||||
if naive:
|
||||
return edge_type, None, None
|
||||
logger.debug("Finish Graph construction.")
|
||||
die_lx, die_hx, die_ly, die_hy = die_info
|
||||
num_macros = macro_pos.shape[0]
|
||||
logger.debug("Longest Path Refinement #Macros: %d #FixedMacros: %d" % (num_macros, macro_fixed.sum()))
|
||||
num_nodes = num_macros + 2 # including source and target
|
||||
s_id = num_macros
|
||||
t_id = num_macros + 1
|
||||
dtype = macro_size.dtype
|
||||
|
||||
adj_matrix = np.zeros((num_nodes, num_nodes, 2), dtype=np.int8)
|
||||
weight_matrix = np.full((num_nodes, num_nodes, 2), -1, dtype=dtype)
|
||||
|
||||
initialize_xy_adj_weight(edge_type, adj_matrix, weight_matrix, macro_size,
|
||||
num_macros, num_nodes, s_id, t_id)
|
||||
|
||||
edge_type[:, :] = -1 # not used anymore
|
||||
|
||||
g_x = ig.Graph.Adjacency(adj_matrix[:,:,0], mode= "directed")
|
||||
g_y = ig.Graph.Adjacency(adj_matrix[:,:,1], mode= "directed")
|
||||
g = [g_x, g_y]
|
||||
# Use negative weight to find longest path
|
||||
if not g[0].is_dag():
|
||||
logger.warning("g_x is not a DAG.")
|
||||
if not g[1].is_dag():
|
||||
logger.warning("g_y is not a DAG.")
|
||||
|
||||
# Calcuate x_L, x_R, y_L and y_R
|
||||
L = np.full((num_nodes, 2), -np.inf, dtype=dtype)
|
||||
R = np.full((num_nodes, 2), np.inf, dtype=dtype)
|
||||
affected_L = np.ones((num_nodes, 2), dtype=np.bool8)
|
||||
affected_R = np.ones((num_nodes, 2), dtype=np.bool8)
|
||||
# Propagate all nodes' L and R
|
||||
propagate_L_R(g, macro_pos, macro_fixed, affected_L, affected_R, adj_matrix, weight_matrix,
|
||||
L, R, die_ll, die_ur, s_id, t_id, num_macros)
|
||||
# Calculate Node and Edge Slacks
|
||||
slack_v = np.zeros((num_macros, 2), dtype=dtype)
|
||||
slack_matrix_e = np.zeros((num_nodes, num_nodes, 2), dtype=dtype)
|
||||
info_v, info_e = slack_info(
|
||||
adj_matrix, weight_matrix, L, R, num_macros, num_nodes, slack_v, slack_matrix_e)
|
||||
logger.debug("Before longest path refinement:")
|
||||
logger.debug(" Node X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Node Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_v)
|
||||
logger.debug(" Edge X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Edge Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_e)
|
||||
|
||||
# plot_negative_slack_macro(macro_pos, macro_size, macro_fixed, die_info, slack_v)
|
||||
|
||||
macro_area = np.prod(macro_size, axis=1)
|
||||
num_trials = 0
|
||||
num_movement = 0
|
||||
while np.minimum(slack_v, 0).sum() < 0:
|
||||
if num_trials == 5:
|
||||
logger.error("Cannot fix longest path after %d trials." % num_trials)
|
||||
break
|
||||
macro_order = list(range(num_macros))
|
||||
sum_slack = slack_v.sum(axis=1)
|
||||
macro_order.sort(key=lambda x: (macro_area[x], -sum_slack[x], macro_pos[x][0], macro_pos[x][1], x))
|
||||
logger.debug("--- Trial %d ---" % num_trials)
|
||||
num_trials += 1
|
||||
for i in macro_order:
|
||||
# Mark edges to move
|
||||
edges_pair = mark_edge_to_move(adj_matrix, weight_matrix, slack_v, i, L, R, s_id, t_id, macro_pos)
|
||||
# Move selected edges
|
||||
for axis_del, (u_del, v_del), axis_add, (u_add, v_add) in edges_pair:
|
||||
g[axis_del].delete_edges([(u_del, v_del)])
|
||||
g[axis_add].add_edges([(u_add, v_add)])
|
||||
assert adj_matrix[u_del, v_del, axis_del] == 1
|
||||
assert adj_matrix[u_add, v_add, axis_add] == 0
|
||||
adj_matrix[u_del, v_del, axis_del] = 0
|
||||
adj_matrix[u_add, v_add, axis_add] = 1
|
||||
logger.debug("Move %d: G_%d (%d, %d) -> G_%d (%d, %d)." % (
|
||||
num_movement, axis_del, u_del, v_del, axis_add, u_add, v_add))
|
||||
|
||||
edges_pair = (axis_del, (u_del, v_del), axis_add, (u_add, v_add))
|
||||
# edges_pair = None # Debug only
|
||||
propagate_L_R(g, macro_pos, macro_fixed, affected_L, affected_R, adj_matrix, weight_matrix,
|
||||
L, R, die_ll, die_ur, s_id, t_id, num_macros, edges_pair=edges_pair)
|
||||
|
||||
info_v, _ = slack_info(adj_matrix, weight_matrix, L, R, num_macros, num_nodes, slack_v, None,
|
||||
update_edge_slack=False)
|
||||
logger.debug(" Updated Node X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Node Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_v)
|
||||
num_movement += 1
|
||||
# plot_negative_slack_macro(macro_pos, macro_size, macro_fixed, die_info, slack_v)
|
||||
|
||||
info_v, info_e = slack_info(
|
||||
adj_matrix, weight_matrix, L, R, num_macros, num_nodes, slack_v, slack_matrix_e)
|
||||
logger.debug("Finish longest path refinement:")
|
||||
logger.debug(" Node X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Node Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_v)
|
||||
logger.debug(" Edge X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Edge Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_e)
|
||||
|
||||
assert (adj_matrix[:num_macros,:num_macros] == 1).all(axis=2).sum() == 0
|
||||
edge_type[:, :] = -1
|
||||
edge_type[adj_matrix[:num_macros,:num_macros,0] == 1] = 0
|
||||
edge_type[adj_matrix[:num_macros,:num_macros,1] == 1] = 1
|
||||
|
||||
return edge_type, g_x, g_y
|
||||
|
||||
|
||||
def basic_variable(macro_pos, macro_size, macro_fixed, die_info):
|
||||
num_macros = macro_pos.shape[0]
|
||||
die_lx, die_hx, die_ly, die_hy = die_info
|
||||
x_set, y_set, dx_set, dy_set = [], [], [], []
|
||||
for i in range(num_macros):
|
||||
if not macro_fixed[i]:
|
||||
x_set.append(pl.LpVariable(
|
||||
"x_%d" % i, macro_size[i][0] / 2, die_hx - macro_size[i][0] / 2
|
||||
))
|
||||
y_set.append(pl.LpVariable(
|
||||
"y_%d" % i, macro_size[i][1] / 2, die_hy - macro_size[i][1] / 2
|
||||
))
|
||||
dx_set.append(pl.LpVariable("d_x_%d" % i, 0, die_hx))
|
||||
dy_set.append(pl.LpVariable("d_y_%d" % i, 0, die_hy))
|
||||
else:
|
||||
x_value = macro_pos[i][0]
|
||||
y_value = macro_pos[i][1]
|
||||
x_set.append(pl.LpVariable("x_%d" % i, x_value, x_value))
|
||||
y_set.append(pl.LpVariable("y_%d" % i, y_value, y_value))
|
||||
dx_set.append(pl.LpVariable("d_x_%d" % i, 0, 0))
|
||||
dy_set.append(pl.LpVariable("d_y_%d" % i, 0, 0))
|
||||
|
||||
for i in range(num_macros):
|
||||
x_value = macro_pos[i][0]
|
||||
y_value = macro_pos[i][1]
|
||||
x_set[i].setInitialValue(x_value)
|
||||
y_set[i].setInitialValue(y_value)
|
||||
dx_set[i].setInitialValue(0)
|
||||
dy_set[i].setInitialValue(0)
|
||||
return x_set, y_set, dx_set, dy_set
|
||||
|
||||
|
||||
def macro_legalization_xy(args, logger, macro_pos, macro_size, macro_fixed, macro_weights, die_info, die_ll, die_ur,
|
||||
num_items=None, lpbackend=None, naive=False, prune=False, edge_type=None):
|
||||
logger.info("Start macro_legalization_xy...")
|
||||
if num_items is not None:
|
||||
macro_pos_cache = np.copy(macro_pos)
|
||||
macro_pos = macro_pos[:num_items]
|
||||
macro_size = macro_size[:num_items]
|
||||
macro_fixed = macro_fixed[:num_items]
|
||||
|
||||
num_macros = macro_pos.shape[0]
|
||||
if edge_type is None:
|
||||
edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur,
|
||||
logger, naive=naive, prune=prune)
|
||||
logger.debug("Finish Graph X construction.")
|
||||
|
||||
prob_x = pl.LpProblem("MacroLegalizationX", pl.LpMinimize)
|
||||
prob_y = pl.LpProblem("MacroLegalizationY", pl.LpMinimize)
|
||||
x_set, y_set, dx_set, dy_set = basic_variable(macro_pos, macro_size, macro_fixed, die_info)
|
||||
prob_x += (
|
||||
pl.lpSum([macro_weights[i, 0] * dx_set[i] for i in range(num_macros)]),
|
||||
"Sum_of_Total_displacement",
|
||||
)
|
||||
prob_y += (
|
||||
pl.lpSum([macro_weights[i, 1] * dy_set[i] for i in range(num_macros)]),
|
||||
"Sum_of_Total_displacement",
|
||||
)
|
||||
for i in range(num_macros):
|
||||
ori_x = macro_pos[i][0]
|
||||
ori_y = macro_pos[i][1]
|
||||
prob_x += (
|
||||
x_set[i] - ori_x <= dx_set[i],
|
||||
"Displacement_x_%d" % i,
|
||||
)
|
||||
prob_x += (
|
||||
x_set[i] - ori_x >= -dx_set[i],
|
||||
"NegDisplacement_x_%d" % i,
|
||||
)
|
||||
prob_y += (
|
||||
y_set[i] - ori_y <= dy_set[i],
|
||||
"Displacement_y_%d" % i,
|
||||
)
|
||||
prob_y += (
|
||||
y_set[i] - ori_y >= -dy_set[i],
|
||||
"NegDisplacement_y_%d" % i,
|
||||
)
|
||||
|
||||
# 2) Graph Version X:
|
||||
dist_x = (macro_size[:,0] + macro_size[:,0].reshape(-1,1)) / 2
|
||||
for i in range(num_macros):
|
||||
for j in range(num_macros):
|
||||
if edge_type[i][j] == -1 or i == j:
|
||||
continue
|
||||
if macro_fixed[i] and macro_fixed[j]:
|
||||
continue
|
||||
if edge_type[i][j] == 0:
|
||||
prob_x += (
|
||||
x_set[i] + dist_x[i][j] <= x_set[j],
|
||||
"Horizontal_%d_%d" % (i, j),
|
||||
)
|
||||
|
||||
# Write LP for debugging
|
||||
# prob_x.writeLP("MacroLegalization.lp")
|
||||
|
||||
# Solve by pl
|
||||
logger.debug("Start solving...")
|
||||
prob_x.solve(lpbackend)
|
||||
|
||||
pl_status_x = pl.LpStatus[prob_x.status]
|
||||
solve_success = pl_status_x == "Optimal"
|
||||
displacement_x = pl.value(prob_x.objective)
|
||||
|
||||
# Commit Solver Results
|
||||
macro_pos_new = np.copy(macro_pos)
|
||||
for v in prob_x.variables():
|
||||
if str(v.name).startswith("x_"):
|
||||
macro_pos_new[int(str(v.name).split("_")[1])][0] = float(v.varValue)
|
||||
macro_pos = macro_pos_new
|
||||
|
||||
# 3) Graph Version Y: Need to update graph edges since placement is changed
|
||||
edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur,
|
||||
logger, naive=naive, prune=prune)
|
||||
logger.debug("Finish Graph Y construction.")
|
||||
dist_y = (macro_size[:,1] + macro_size[:,1].reshape(-1,1)) / 2
|
||||
for i in range(num_macros):
|
||||
for j in range(num_macros):
|
||||
if edge_type[i][j] == -1 or i == j:
|
||||
continue
|
||||
if macro_fixed[i] and macro_fixed[j]:
|
||||
continue
|
||||
if edge_type[i][j] == 1:
|
||||
prob_y += (
|
||||
y_set[i] + dist_y[i][j] <= y_set[j],
|
||||
"Vertical_%d_%d" % (i, j),
|
||||
)
|
||||
|
||||
# Write LP for debugging
|
||||
# prob_y.writeLP("MacroLegalization.lp")
|
||||
|
||||
# Solve by pl
|
||||
logger.debug("Start solving...")
|
||||
prob_y.solve(lpbackend)
|
||||
|
||||
pl_status_y = pl.LpStatus[prob_y.status]
|
||||
solve_success = (pl_status_y == "Optimal") and solve_success
|
||||
displacement_y = pl.value(prob_y.objective)
|
||||
|
||||
# Commit Solver Results
|
||||
macro_pos_new = np.copy(macro_pos)
|
||||
for v in prob_y.variables():
|
||||
if str(v.name).startswith("y_"):
|
||||
macro_pos_new[int(str(v.name).split("_")[1])][1] = float(v.varValue)
|
||||
|
||||
logger.info(
|
||||
"X Status: %s, DisplaceX = %.2f | Y Status: %s, DisplaceY = %.2f | Total Displacement = %.2f" % (
|
||||
pl_status_x, displacement_x, pl_status_y, displacement_y, displacement_x + displacement_y
|
||||
))
|
||||
|
||||
# 4) Iterative Legalization:
|
||||
if num_items is not None:
|
||||
macro_pos_cache[:num_items] = macro_pos_new[:num_items]
|
||||
logger.info("#Macros: %d, #Macros in step: %d" % (macro_pos_cache.shape[0], macro_pos_new.shape[0]))
|
||||
macro_pos_new = macro_pos_cache
|
||||
|
||||
return macro_pos_new, solve_success, displacement_x + displacement_y
|
||||
|
||||
|
||||
def macro_legalization_mix(args, logger, macro_pos, macro_size, macro_fixed, macro_weights, die_info, die_ll, die_ur,
|
||||
num_items=None, lpbackend=None, naive=False, prune=False, edge_type=None):
|
||||
logger.info("Start macro_legalization_mix...")
|
||||
if num_items is not None:
|
||||
macro_pos_cache = np.copy(macro_pos)
|
||||
macro_pos = macro_pos[:num_items]
|
||||
macro_size = macro_size[:num_items]
|
||||
macro_fixed = macro_fixed[:num_items]
|
||||
|
||||
num_macros = macro_pos.shape[0]
|
||||
if edge_type is None:
|
||||
edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info,
|
||||
die_ll, die_ur, logger, naive=naive, prune=prune)
|
||||
logger.debug("Finish Graph construction.")
|
||||
|
||||
prob = pl.LpProblem("MacroLegalization", pl.LpMinimize)
|
||||
logger.debug("Setup lp variables")
|
||||
x_set, y_set, dx_set, dy_set = basic_variable(macro_pos, macro_size, macro_fixed, die_info)
|
||||
logger.debug("Setup lp objectives")
|
||||
prob += (
|
||||
pl.lpSum([macro_weights[i, 0] * dx_set[i] + macro_weights[i, 1] * dy_set[i] for i in range(num_macros)]),
|
||||
"Sum_of_Total_displacement",
|
||||
)
|
||||
logger.debug("Setup lp displacement constrains")
|
||||
for i in range(num_macros):
|
||||
ori_x = macro_pos[i][0]
|
||||
ori_y = macro_pos[i][1]
|
||||
prob += (
|
||||
x_set[i] - ori_x <= dx_set[i],
|
||||
"Displacement_x_%d" % i,
|
||||
)
|
||||
prob += (
|
||||
x_set[i] - ori_x >= -dx_set[i],
|
||||
"NegDisplacement_x_%d" % i,
|
||||
)
|
||||
prob += (
|
||||
y_set[i] - ori_y <= dy_set[i],
|
||||
"Displacement_y_%d" % i,
|
||||
)
|
||||
prob += (
|
||||
y_set[i] - ori_y >= -dy_set[i],
|
||||
"NegDisplacement_y_%d" % i,
|
||||
)
|
||||
logger.debug("Setup lp edge constraints")
|
||||
dist_x = (macro_size[:,0] + macro_size[:,0].reshape(-1,1)) / 2
|
||||
dist_y = (macro_size[:,1] + macro_size[:,1].reshape(-1,1)) / 2
|
||||
for i in range(num_macros):
|
||||
for j in range(num_macros):
|
||||
if edge_type[i][j] == -1 or i == j:
|
||||
continue
|
||||
if macro_fixed[i] and macro_fixed[j]:
|
||||
continue
|
||||
if edge_type[i][j] == 0:
|
||||
prob += (
|
||||
x_set[i] + dist_x[i][j] <= x_set[j],
|
||||
"Horizontal_%d_%d" % (i, j),
|
||||
)
|
||||
elif edge_type[i][j] == 1:
|
||||
prob += (
|
||||
y_set[i] + dist_y[i][j] <= y_set[j],
|
||||
"Vertical_%d_%d" % (i, j),
|
||||
)
|
||||
# Write LP for debugging
|
||||
# prob.writeLP("MacroLegalization.lp")
|
||||
|
||||
# Solve by pl
|
||||
logger.debug("Start solving...")
|
||||
prob.solve(lpbackend)
|
||||
|
||||
solve_success = pl.LpStatus[prob.status] == "Optimal"
|
||||
displacement = pl.value(prob.objective)
|
||||
logger.info("Status: %s, Total Displacement of MacroLegalization = %.2f" % (pl.LpStatus[prob.status], displacement))
|
||||
|
||||
# Commit Solver Results
|
||||
macro_pos_new = np.copy(macro_pos)
|
||||
for v in prob.variables():
|
||||
if str(v.name).startswith("x_"):
|
||||
macro_pos_new[int(str(v.name).split("_")[1])][0] = float(v.varValue)
|
||||
if str(v.name).startswith("y_"):
|
||||
macro_pos_new[int(str(v.name).split("_")[1])][1] = float(v.varValue)
|
||||
|
||||
if num_items is not None:
|
||||
macro_pos_cache[:num_items] = macro_pos_new[:num_items]
|
||||
logger.info("#Macros: %d, #Macros in step: %d" % (macro_pos_cache.shape[0], macro_pos_new.shape[0]))
|
||||
macro_pos_new = macro_pos_cache
|
||||
|
||||
return macro_pos_new, solve_success, displacement
|
||||
|
||||
|
||||
def macro_legalization_ilp(args, logger, macro_pos, macro_size, macro_fixed, macro_weights, die_info, die_ll, die_ur,
|
||||
num_items=None, lpbackend=None, edge_type=None):
|
||||
logger.info("Start macro_legalization_ilp...")
|
||||
if num_items is not None:
|
||||
macro_pos_cache = np.copy(macro_pos)
|
||||
macro_pos = macro_pos[:num_items]
|
||||
macro_size = macro_size[:num_items]
|
||||
macro_fixed = macro_fixed[:num_items]
|
||||
num_macros = macro_pos.shape[0]
|
||||
|
||||
prob = pl.LpProblem("MacroLegalization", pl.LpMinimize)
|
||||
die_lx, die_hx, die_ly, die_hy = die_info
|
||||
x_set, y_set, dx_set, dy_set = basic_variable(macro_pos, macro_size, macro_fixed, die_info)
|
||||
prob += (
|
||||
pl.lpSum([macro_weights[i, 0] * dx_set[i] + macro_weights[i, 1] * dy_set[i] for i in range(num_macros)]),
|
||||
"Sum_of_Total_displacement",
|
||||
)
|
||||
for i in range(num_macros):
|
||||
ori_x = macro_pos[i][0]
|
||||
ori_y = macro_pos[i][1]
|
||||
prob += (
|
||||
x_set[i] - ori_x <= dx_set[i],
|
||||
"Displacement_x_%d" % i,
|
||||
)
|
||||
prob += (
|
||||
x_set[i] - ori_x >= -dx_set[i],
|
||||
"NegDisplacement_x_%d" % i,
|
||||
)
|
||||
prob += (
|
||||
y_set[i] - ori_y <= dy_set[i],
|
||||
"Displacement_y_%d" % i,
|
||||
)
|
||||
prob += (
|
||||
y_set[i] - ori_y >= -dy_set[i],
|
||||
"NegDisplacement_y_%d" % i,
|
||||
)
|
||||
|
||||
# Naive Binary Variables Version:
|
||||
dist_x = (macro_size[:,0] + macro_size[:,0].reshape(-1,1)) / 2
|
||||
dist_y = (macro_size[:,1] + macro_size[:,1].reshape(-1,1)) / 2
|
||||
choices = pl.LpVariable.dicts("Choice", (range(num_macros), range(num_macros), range(2)), 0, 1, cat=pl.const.LpInteger)
|
||||
for i in range(num_macros):
|
||||
for j in range(num_macros):
|
||||
if i >= j:
|
||||
continue
|
||||
if macro_fixed[i] and macro_fixed[j]:
|
||||
continue
|
||||
prob += (
|
||||
x_set[i] + dist_x[i][j] <= x_set[j] + die_hx * (choices[i][j][0] + choices[i][j][1]),
|
||||
"XLhs_%d_%d" % (i, j),
|
||||
)
|
||||
prob += (
|
||||
x_set[i] - dist_x[i][j] >= x_set[j] - die_hx * (1 + choices[i][j][0] - choices[i][j][1]),
|
||||
"XRhs_%d_%d" % (i, j),
|
||||
)
|
||||
prob += (
|
||||
y_set[i] + dist_y[i][j] <= y_set[j] + die_hy * (1 - choices[i][j][0] + choices[i][j][1]),
|
||||
"YLhs_%d_%d" % (i, j),
|
||||
)
|
||||
prob += (
|
||||
y_set[i] - dist_y[i][j] >= y_set[j] - die_hy * (2 - choices[i][j][0] - choices[i][j][1]),
|
||||
"YRhs_%d_%d" % (i, j),
|
||||
)
|
||||
|
||||
# Write LP for debugging
|
||||
# prob.writeLP("MacroLegalization.lp")
|
||||
|
||||
# Solve by pl
|
||||
logger.debug("Start solving...")
|
||||
prob.solve(lpbackend)
|
||||
|
||||
solve_success = pl.LpStatus[prob.status] == "Optimal"
|
||||
displacement = pl.value(prob.objective)
|
||||
logger.info("Status: %s, Total Displacement of MacroLegalization = %.2f" % (pl.LpStatus[prob.status], displacement))
|
||||
|
||||
# Commit Solver Results
|
||||
macro_pos_new = np.copy(macro_pos)
|
||||
for v in prob.variables():
|
||||
if str(v.name).startswith("x_"):
|
||||
macro_pos_new[int(str(v.name).split("_")[1])][0] = float(v.varValue)
|
||||
if str(v.name).startswith("y_"):
|
||||
macro_pos_new[int(str(v.name).split("_")[1])][1] = float(v.varValue)
|
||||
|
||||
if num_items is not None:
|
||||
macro_pos_cache[:num_items] = macro_pos_new[:num_items]
|
||||
logger.info("#Macros: %d, #Macros in step: %d" % (macro_pos_cache.shape[0], macro_pos_new.shape[0]))
|
||||
macro_pos_new = macro_pos_cache
|
||||
|
||||
return macro_pos_new, solve_success, displacement
|
||||
|
||||
|
||||
def macro_rough_align(macro_pos, macro_size, macro_fixed, die_ll, die_ur, inv_scalar):
|
||||
is_mov = np.logical_not(macro_fixed)
|
||||
|
||||
macro_pos_lb = macro_size / 2 + die_ll + 1e-4
|
||||
macro_pos_ub = die_ur - macro_size / 2 + die_ll - 1e-4
|
||||
mov_macro_pos_lb = macro_pos_lb[is_mov]
|
||||
mov_macro_pos_ub = macro_pos_ub[is_mov]
|
||||
macro_pos[is_mov] = macro_pos[is_mov].clip(min=mov_macro_pos_lb, max=mov_macro_pos_ub)
|
||||
|
||||
macro_lpos = macro_pos - macro_size / 2
|
||||
macro_lpos = np.multiply(macro_lpos, inv_scalar, out=macro_lpos)
|
||||
macro_lpos = np.round_(macro_lpos, out=macro_lpos)
|
||||
macro_lpos = np.divide(macro_lpos, inv_scalar, out=macro_lpos)
|
||||
|
||||
macro_pos_ = macro_lpos + macro_size / 2
|
||||
macro_pos[is_mov] = macro_pos_[is_mov]
|
||||
|
||||
return macro_pos
|
||||
|
||||
|
||||
def plot_macros(macro_pos, macro_size, macro_fixed, die_info, given_colors=None, img_path=None):
|
||||
import matplotlib as mpl
|
||||
import matplotlib.pyplot as plt
|
||||
from matplotlib.collections import PatchCollection
|
||||
from matplotlib.patches import Rectangle
|
||||
base_x = 8
|
||||
die_lx, die_hx, die_ly, die_hy = die_info
|
||||
base_y = base_x / die_hx * die_hy
|
||||
fig, ax = plt.subplots(1, figsize=(base_x, base_y))
|
||||
die_rects = [Rectangle((die_lx, die_ly), die_hx - die_lx, die_hy - die_ly)]
|
||||
pc = PatchCollection(die_rects, facecolor='w', alpha=0.5, edgecolor='black')
|
||||
ax.add_collection(pc)
|
||||
all_color = plt.get_cmap('Set2').colors
|
||||
for i in range(macro_pos.shape[0]):
|
||||
x, y = macro_pos[i]
|
||||
w, h = macro_size[i]
|
||||
color_idx = given_colors[i] if given_colors is not None else 0
|
||||
color = all_color[color_idx]
|
||||
rect = Rectangle((x - w / 2, y - h / 2), w, h, facecolor=color, alpha=0.5, edgecolor='black')
|
||||
ax.add_patch(rect)
|
||||
ax.text(x, y, "%d" % i, fontsize=14)
|
||||
|
||||
plt.xlim(-die_hx * 0.02, die_hx * 1.02)
|
||||
plt.ylim(-die_hy * 0.02, die_hy * 1.02)
|
||||
ax.axis('off')
|
||||
ax.get_xaxis().set_visible(False)
|
||||
ax.get_yaxis().set_visible(False)
|
||||
plt.tight_layout()
|
||||
img_path = img_path if img_path is not None else "test.png"
|
||||
plt.savefig(img_path, bbox_inches='tight')
|
||||
plt.close()
|
||||
|
||||
|
||||
def plot_negative_slack_macro(macro_pos, macro_size, macro_fixed, die_info, slack_v, img_path=None):
|
||||
num_macros = slack_v.shape[0]
|
||||
x_nslack = np.minimum(slack_v[:, 0], 0)
|
||||
y_nslack = np.minimum(slack_v[:, 1], 0)
|
||||
print(x_nslack)
|
||||
print(y_nslack)
|
||||
unique_values = np.sort(np.unique(x_nslack + y_nslack))[::-1]
|
||||
given_colors = np.zeros(num_macros, dtype=np.int8)
|
||||
for color_idx, value in enumerate(unique_values):
|
||||
given_colors[(x_nslack + y_nslack) == value] = color_idx
|
||||
plot_macros(macro_pos, macro_size, macro_fixed, die_info, given_colors=given_colors, img_path=img_path)
|
||||
|
||||
|
||||
def macro_legalization_multi(macro_info, args, logger):
|
||||
macro_pos, macro_size, macro_fixed, macro_weights, die_ll, die_ur, die_info, inv_scalar = macro_info
|
||||
macro_pos = macro_pos.cpu().numpy()
|
||||
macro_size = macro_size.cpu().numpy()
|
||||
macro_fixed = macro_fixed.cpu().numpy()
|
||||
macro_weights = macro_weights.cpu().numpy()
|
||||
die_ll = die_ll.cpu().numpy()
|
||||
die_ur = die_ur.cpu().numpy()
|
||||
die_info = die_info.cpu().numpy()
|
||||
inv_scalar = inv_scalar.cpu().numpy()
|
||||
|
||||
# plot_macros(macro_pos, macro_size, macro_fixed, die_info, img_path="legalized_before.png")
|
||||
nb.set_num_threads(args.num_threads)
|
||||
|
||||
if check_macro_legality(macro_pos, macro_size, macro_fixed, die_info, check_all=False):
|
||||
# Macros are legal, skip legalization
|
||||
return macro_pos, True
|
||||
|
||||
macro_pos = macro_rough_align(macro_pos, macro_size, macro_fixed, die_ll, die_ur, inv_scalar)
|
||||
|
||||
def macro_lg_handler(
|
||||
ml_func, method_name, solver, best_result, *func_args, max_times=1, timeLimit=None, **ml_func_kwargs
|
||||
):
|
||||
total_displacement = 0
|
||||
for i in range(max_times):
|
||||
solver.timeLimit = (i + 1) * 20 if timeLimit is None else timeLimit
|
||||
logger.info("Use cbc to solve LP. TimeLimit = %ds." % solver.timeLimit)
|
||||
macro_pos_tmp, solve_success, displacement = ml_func(*func_args, lpbackend=solver, **ml_func_kwargs)
|
||||
if not check_macro_legality(macro_pos_tmp, macro_size, macro_fixed, die_info):
|
||||
# update macro_pos to macro_pos_tmp
|
||||
func_args = (*func_args[:2], macro_pos_tmp, *func_args[3:])
|
||||
ml_func_kwargs["edge_type"] = None
|
||||
solve_success = False
|
||||
total_displacement += displacement
|
||||
if solve_success:
|
||||
break
|
||||
if not solve_success and not best_result[1] and method_name != "ilp":
|
||||
best_result = (macro_pos_tmp, False, total_displacement, method_name)
|
||||
if solve_success and (not best_result[1] or total_displacement < best_result[2]):
|
||||
best_result = (macro_pos_tmp, True, total_displacement, method_name)
|
||||
return best_result
|
||||
|
||||
# best_result: (macro_pos, solve_success, displacement, method_name)
|
||||
best_result = (None, False, float('inf'), None)
|
||||
func_args = (args, logger, macro_pos, macro_size, macro_fixed, macro_weights, die_info, die_ll, die_ur)
|
||||
solver = pl.PULP_CBC_CMD(msg=0, timeLimit=20, threads=args.num_threads)
|
||||
if macro_pos.shape[0] > 500:
|
||||
logger.info("Too many macros, try pruned graph version first.")
|
||||
edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur,
|
||||
logger, naive=False, prune=True)
|
||||
best_result = macro_lg_handler(macro_legalization_mix, "mix", solver, best_result, *func_args, max_times=1, edge_type=edge_type, prune=True)
|
||||
best_result = macro_lg_handler(macro_legalization_xy, "xy", solver, best_result, *func_args, max_times=3, edge_type=edge_type, prune=True)
|
||||
|
||||
if not best_result[1]:
|
||||
edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur,
|
||||
logger, naive=False, prune=False)
|
||||
best_result = macro_lg_handler(macro_legalization_mix, "mix", solver, best_result, *func_args, max_times=1, edge_type=edge_type)
|
||||
best_result = macro_lg_handler(macro_legalization_xy, "xy", solver, best_result, *func_args, max_times=3, edge_type=edge_type)
|
||||
|
||||
if not best_result[1]:
|
||||
logger.warning("Both LPs are infeasible. Try ILP version.")
|
||||
macro_lg_handler(macro_legalization_ilp, "ilp", solver, best_result, *func_args, timeLimit=120)
|
||||
if not best_result[1]:
|
||||
logger.error("ILP is infeasible.")
|
||||
|
||||
# commit the result no matter it is legal or not
|
||||
macro_pos, solve_success, displacement, method_name = best_result
|
||||
# plot_macros(macro_pos, macro_size, macro_fixed, die_info, img_path="legalized_after.png")
|
||||
if solve_success:
|
||||
logger.info("Macro Legalization Success. Select macro legalization [%s]. Displacement = %.2f" % (method_name, displacement))
|
||||
|
||||
return macro_pos, solve_success
|
||||
|
||||
|
||||
# For debug only
|
||||
# numba_logger = logging.getLogger('numba')
|
||||
# numba_logger.setLevel(logging.WARNING)
|
||||
# import random
|
||||
# import torch
|
||||
# import time
|
||||
# random.seed(0)
|
||||
# class Args:
|
||||
# def __init__(self) -> None:
|
||||
# self.num_threads = 20
|
||||
# args = Args()
|
||||
# logger = logging.getLogger(__name__)
|
||||
# logging.basicConfig(encoding='utf-8', level=logging.DEBUG, format='%(asctime)s %(message)s')
|
||||
# if __name__ == "__main__":
|
||||
# macro_info = torch.load("macro_info.pt")
|
||||
# logger.info("Start...")
|
||||
# start_time = time.time()
|
||||
# macro_pos, solve_success = macro_legalization_multi(macro_info, args, logger)
|
||||
# logger.info("Macro legalization time: %.2f" % (time.time() - start_time))
|
||||
@ -7,84 +7,6 @@ class HPWLCache:
|
||||
|
||||
hpwl_cache = HPWLCache()
|
||||
|
||||
class WAWirelengthLossAndHPWL(torch.autograd.Function):
|
||||
@staticmethod
|
||||
def forward(
|
||||
ctx,
|
||||
node_pos,
|
||||
pin_id2node_id,
|
||||
pin_rel_cpos,
|
||||
node2pin_list,
|
||||
node2pin_list_end,
|
||||
hyperedge_list,
|
||||
hyperedge_list_end,
|
||||
net_mask,
|
||||
gamma,
|
||||
hpwl_scale,
|
||||
deterministic,
|
||||
):
|
||||
|
||||
(
|
||||
partial_wa_wl,
|
||||
node_grad,
|
||||
partial_hpwl,
|
||||
) = wa_wirelength_hpwl_cuda.merged_forward_backward_with_hpwl(
|
||||
node_pos,
|
||||
pin_id2node_id,
|
||||
pin_rel_cpos,
|
||||
node2pin_list,
|
||||
node2pin_list_end,
|
||||
hyperedge_list,
|
||||
hyperedge_list_end,
|
||||
net_mask,
|
||||
gamma,
|
||||
deterministic,
|
||||
)
|
||||
sum_hpwl = torch.round(partial_hpwl * hpwl_scale).sum()
|
||||
ctx.save_for_backward(node_grad)
|
||||
return torch.sum(partial_wa_wl), sum_hpwl
|
||||
|
||||
@staticmethod
|
||||
def backward(ctx, wa_grad_out, hpwl_grad_out):
|
||||
node_grad = ctx.saved_tensors[0]
|
||||
return (node_grad * wa_grad_out,) + (None,) * 10
|
||||
|
||||
|
||||
class WAWirelengthLoss(torch.autograd.Function):
|
||||
@staticmethod
|
||||
def forward(
|
||||
ctx,
|
||||
node_pos,
|
||||
pin_id2node_id,
|
||||
pin_rel_cpos,
|
||||
node2pin_list,
|
||||
node2pin_list_end,
|
||||
hyperedge_list,
|
||||
hyperedge_list_end,
|
||||
net_mask,
|
||||
gamma,
|
||||
deterministic,
|
||||
):
|
||||
partial_wa_wl, node_grad = wa_wirelength_hpwl_cuda.merged_forward_backward(
|
||||
node_pos,
|
||||
pin_id2node_id,
|
||||
pin_rel_cpos,
|
||||
node2pin_list,
|
||||
node2pin_list_end,
|
||||
hyperedge_list,
|
||||
hyperedge_list_end,
|
||||
net_mask,
|
||||
gamma,
|
||||
deterministic,
|
||||
)
|
||||
ctx.save_for_backward(node_grad)
|
||||
return torch.sum(partial_wa_wl)
|
||||
|
||||
@staticmethod
|
||||
def backward(ctx, wa_grad_out):
|
||||
node_grad = ctx.saved_tensors[0]
|
||||
return (node_grad * wa_grad_out,) + (None,) * 9
|
||||
|
||||
|
||||
def merged_wl_loss_grad(
|
||||
node_pos,
|
||||
|
||||
139
src/database.py
139
src/database.py
@ -317,6 +317,21 @@ class PlaceData(object):
|
||||
if hasattr(self, "__num_fillers__"):
|
||||
return self.__num_fillers__
|
||||
|
||||
@property
|
||||
def num_macros(self):
|
||||
if hasattr(self, "__num_macros__"):
|
||||
return self.__num_macros__
|
||||
|
||||
@property
|
||||
def num_movable_macros(self):
|
||||
if hasattr(self, "__num_movable_macros__"):
|
||||
return self.__num_movable_macros__
|
||||
|
||||
@property
|
||||
def num_fixed_macros(self):
|
||||
if hasattr(self, "__num_fixed_macros__"):
|
||||
return self.__num_fixed_macros__
|
||||
|
||||
@property
|
||||
def num_bin_x(self):
|
||||
if hasattr(self, "__num_bin_x__"):
|
||||
@ -540,7 +555,7 @@ class PlaceData(object):
|
||||
self.pin_rel_cpos /= scalar_at
|
||||
self.pin_rel_lpos /= scalar_at
|
||||
self.pin_size /= scalar_at
|
||||
self.__die_scale__ *= self.site_width
|
||||
self.__die_scale__ *= scalar_at
|
||||
return self
|
||||
|
||||
def prescale(self):
|
||||
@ -591,10 +606,28 @@ class PlaceData(object):
|
||||
self.net_mask = torch.logical_and(
|
||||
self.net_to_num_pins <= args.ignore_net_degree, self.net_to_num_pins >= 2
|
||||
) # 0: ignore, 1: consider in wirelength calculation
|
||||
# obj related
|
||||
# macros -> all mov nodes has ultra-large areas with >= 3 row height and all fixed nodes
|
||||
# But nodes with zero width or zero height are not considered as macros
|
||||
mov_lhs, mov_rhs = self.movable_index
|
||||
mov_cell_area = torch.prod(self.node_size[mov_lhs:mov_rhs, ...], 1)
|
||||
self.__total_mov_area_without_filler__ = torch.sum(mov_cell_area).item()
|
||||
num_movable_nodes = mov_rhs - mov_lhs
|
||||
self.is_macro: torch.Tensor = self.node_size[:,1] * self.die_scale[1] / self.row_height > 2.01
|
||||
mov_node_area = torch.prod(self.node_size[mov_lhs:mov_rhs, ...], 1)
|
||||
mov_node_area_order = torch.argsort(mov_node_area)
|
||||
macro_area_threshold = 10 * torch.mean(mov_node_area[
|
||||
mov_node_area_order[:int(num_movable_nodes * 0.999)]
|
||||
])
|
||||
self.is_macro.logical_and_((self.node_area > macro_area_threshold).squeeze(1))
|
||||
self.is_macro[mov_rhs:] = True
|
||||
self.is_macro.logical_and_((self.node_size * self.die_scale > 1e-4).all(dim=1))
|
||||
|
||||
self.is_mov_macro = self.is_macro.clone()
|
||||
self.is_mov_macro[mov_rhs:] = False
|
||||
|
||||
self.__num_macros__ = torch.sum(self.is_macro).item()
|
||||
self.__num_movable_macros__ = torch.sum(self.is_mov_macro).item()
|
||||
self.__num_fixed_macros__ = self.__num_macros__ - self.__num_movable_macros__
|
||||
# obj related
|
||||
self.__total_mov_area_without_filler__ = torch.sum(mov_node_area).item()
|
||||
self.__bin_area__ = torch.prod(self.unit_len).item()
|
||||
return self
|
||||
|
||||
@ -630,12 +663,25 @@ class PlaceData(object):
|
||||
# init_density_map are all normalized to (0.0, 1.0)
|
||||
fixed_node_area = ori_dmap * self.bin_area
|
||||
placeable_area = die_area - fixed_node_area
|
||||
if True:
|
||||
mov_cell_area = torch.prod(mov_node_size, 1)
|
||||
num_movable_nodes = mov_rhs - mov_lhs
|
||||
mov_node_xsize_order = torch.argsort(mov_node_size[:, 0])
|
||||
mov_node_area = torch.prod(mov_node_size, 1).sum()
|
||||
mov_macro_area = self.node_area[self.is_mov_macro].sum()
|
||||
mov_stdcell_area = mov_node_area - mov_macro_area
|
||||
stdcell_placeable_area = placeable_area - mov_macro_area
|
||||
stdcell_util = mov_stdcell_area / stdcell_placeable_area
|
||||
if stdcell_util.item() > args.target_density:
|
||||
logger.warning("Stdcell util %.2f is larger than target density %.2f. Increase target density to %.2f." % (
|
||||
stdcell_util, args.target_density, stdcell_util))
|
||||
args.target_density = stdcell_util
|
||||
if self.is_mov_macro.sum().item() <= 1:
|
||||
mov_node_xsize = mov_node_size[:, 0]
|
||||
else:
|
||||
mov_node_xsize = mov_node_size[:, 0][
|
||||
torch.logical_not(self.is_mov_macro[mov_lhs:mov_rhs])
|
||||
].contiguous()
|
||||
num_movable_nodes = mov_node_xsize.shape[0]
|
||||
mov_node_xsize_order = torch.argsort(mov_node_xsize)
|
||||
filler_size_x = torch.mean(
|
||||
mov_node_size[:, 0][
|
||||
mov_node_xsize[
|
||||
mov_node_xsize_order[
|
||||
int(num_movable_nodes * 0.05) : int(
|
||||
num_movable_nodes * 0.95
|
||||
@ -645,7 +691,7 @@ class PlaceData(object):
|
||||
)
|
||||
filler_size_y = self.site_height / self.die_scale[1]
|
||||
total_filler_area = max(
|
||||
args.target_density * placeable_area - torch.sum(mov_cell_area),
|
||||
args.target_density * stdcell_placeable_area - mov_stdcell_area,
|
||||
0.0,
|
||||
)
|
||||
single_filler_size = torch.tensor(
|
||||
@ -656,16 +702,7 @@ class PlaceData(object):
|
||||
self.__num_fillers__ = int(
|
||||
torch.round(total_filler_area / (filler_size_x * filler_size_y))
|
||||
)
|
||||
else:
|
||||
mov_cell_area = torch.prod(mov_node_size, 1)
|
||||
total_filler_area = max(
|
||||
args.target_density * placeable_area
|
||||
- torch.sum(mov_cell_area).item(),
|
||||
0.0,
|
||||
)
|
||||
single_filler_area = torch.mean(mov_cell_area)
|
||||
single_filler_size = single_filler_area.sqrt().repeat(2)
|
||||
self.__num_fillers__ = int(total_filler_area / single_filler_area)
|
||||
|
||||
if self.num_fillers > 0:
|
||||
self.filler_size = single_filler_size.repeat(self.num_fillers, 1)
|
||||
logger.info(
|
||||
@ -689,18 +726,24 @@ class PlaceData(object):
|
||||
args.use_filler = False
|
||||
|
||||
die_area, placeable_area = die_area.item(), placeable_area.item()
|
||||
fixed_node_area, mov_cell_area = fixed_node_area.item(), torch.sum(mov_cell_area).item()
|
||||
fixed_node_area, mov_node_area = fixed_node_area.item(), mov_node_area.item()
|
||||
mov_macro_area, mov_stdcell_area = mov_macro_area.item(), mov_stdcell_area.item()
|
||||
total_filler_area = float(total_filler_area)
|
||||
logger.info(
|
||||
"DieArea: %.3E FixArea: %.3E (%.1f%%) PlaceableArea: %.3E (%.1f%%) MovArea: %.3E (%.1f%%) FillerArea: %.3E (%.1f%%) "
|
||||
"MovMacroArea: %.3E (%.1f%%) MovStdCellArea: %.3E (%.1f%%)"
|
||||
% (
|
||||
die_area,
|
||||
fixed_node_area, fixed_node_area / die_area * 100,
|
||||
placeable_area, placeable_area / die_area * 100,
|
||||
mov_cell_area, mov_cell_area / die_area * 100,
|
||||
mov_node_area, mov_node_area / die_area * 100,
|
||||
total_filler_area, total_filler_area / die_area * 100,
|
||||
mov_macro_area, mov_macro_area / die_area * 100,
|
||||
mov_stdcell_area, mov_stdcell_area / die_area * 100,
|
||||
)
|
||||
)
|
||||
if mov_node_area > placeable_area:
|
||||
logger.warning("MovArea > PlaceableArea. Not enough olaceblae area to place all movable nodes.")
|
||||
|
||||
return self
|
||||
|
||||
@ -769,6 +812,11 @@ class PlaceData(object):
|
||||
num_fltiopin,
|
||||
)
|
||||
)
|
||||
content += (
|
||||
"#Macros = %d, #MovMacros = %d, #FixMacros = %d\n" % (
|
||||
self.num_macros, self.num_movable_macros, self.num_fixed_macros
|
||||
)
|
||||
)
|
||||
content += "Core Info " + str([i for i in self.die_info.cpu().numpy()]) + "\n"
|
||||
content += "Site Width = %d, Row Height = %d\n" % (
|
||||
self.site_width,
|
||||
@ -783,6 +831,12 @@ class PlaceData(object):
|
||||
content += "target density = %.2f\n" % (args.target_density)
|
||||
content += "==================="
|
||||
logger.info(content)
|
||||
args.include_macros = True if self.num_movable_macros > 10 else False
|
||||
if args.include_macros and not args.mixed_size:
|
||||
logger.warning("Detect many macros. Suggest to turn on mixed_size in cmd args.")
|
||||
if self.num_movable_macros == 0 and args.mixed_size:
|
||||
logger.warning("#MovMacros is 0. Turn off mixed size mode.")
|
||||
args.mixed_size = False
|
||||
return self
|
||||
|
||||
def preprocess(self):
|
||||
@ -790,8 +844,8 @@ class PlaceData(object):
|
||||
self.backup_ori_var()
|
||||
self.preshift()
|
||||
self.prescale_by_site_width()
|
||||
if args.scale_design:
|
||||
self.prescale()
|
||||
# if args.scale_design:
|
||||
# self.prescale()
|
||||
self.pre_compute_var()
|
||||
self.init_fence_region()
|
||||
self.logging_statistics()
|
||||
@ -803,6 +857,21 @@ class PlaceData(object):
|
||||
self.compute_sorted_node_map()
|
||||
return self
|
||||
|
||||
def get_filler_pos(self):
|
||||
if self.num_fillers > 0:
|
||||
if self.enable_fence:
|
||||
raise NotImplementedError("We haven't yet supported fence region.")
|
||||
else:
|
||||
filler_pos = torch.rand(
|
||||
(self.num_fillers, 2),
|
||||
dtype=self.node_size.dtype,
|
||||
device=self.node_size.device,
|
||||
)
|
||||
scale = self.die_ur - self.die_ll
|
||||
shift = self.die_ll
|
||||
filler_pos = filler_pos * scale + shift
|
||||
return filler_pos
|
||||
|
||||
def get_mov_node_info(self, init_method="randn_center"):
|
||||
args = self.__args__
|
||||
mov_lhs, mov_rhs = self.movable_index
|
||||
@ -813,19 +882,15 @@ class PlaceData(object):
|
||||
scale = (self.die_ur - self.die_ll) * 0.001
|
||||
loc = (self.die_ur + self.die_ll) * 0.5
|
||||
mov_node_pos = torch.randn_like(mov_node_pos) * scale + loc
|
||||
elif init_method == "randn_center_lxly":
|
||||
# TODO: Mixed-size placement is very sensitive to the initial location.
|
||||
# An elegant yet effective initialization may be needed.
|
||||
scale = (self.die_ur - self.die_ll) * 0.001
|
||||
loc = (self.die_ur + self.die_ll) * 0.5
|
||||
mov_node_pos = torch.randn_like(mov_node_pos) * scale + loc + mov_node_size / 2
|
||||
|
||||
if self.num_fillers > 0:
|
||||
if self.enable_fence:
|
||||
raise NotImplementedError("We haven't yet supported fence region.")
|
||||
else:
|
||||
filler_pos = torch.rand(
|
||||
(self.num_fillers, 2),
|
||||
dtype=mov_node_size.dtype,
|
||||
device=mov_node_size.device,
|
||||
)
|
||||
scale = self.die_ur - self.die_ll
|
||||
shift = self.die_ll
|
||||
filler_pos = filler_pos * scale + shift
|
||||
filler_pos = self.get_filler_pos()
|
||||
mov_node_pos = torch.cat([mov_node_pos, filler_pos], dim=0)
|
||||
mov_node_size = torch.cat([mov_node_size, self.filler_size], dim=0)
|
||||
|
||||
@ -844,6 +909,10 @@ class PlaceData(object):
|
||||
expand_ratio = mov_node_area / clamp_mov_node_area
|
||||
mov_node_size = clamp_mov_node_size
|
||||
|
||||
if args.target_density < 1.0:
|
||||
expand_ratio[mov_lhs:mov_rhs].masked_fill_(
|
||||
self.is_mov_macro[mov_lhs:mov_rhs], args.target_density)
|
||||
|
||||
return mov_node_pos, mov_node_size, expand_ratio
|
||||
|
||||
def write_pl(self, node_pos, gp_prefix):
|
||||
|
||||
@ -1,10 +1,10 @@
|
||||
import torch
|
||||
from .database import PlaceData
|
||||
from .evaluator import get_obj_hpwl
|
||||
from .core.macro_legalization import macro_legalization_multi
|
||||
from utils.visualization import draw_fig_with_cairo_cpp
|
||||
from cpp_to_py import gpudp, routedp
|
||||
import numba as nb
|
||||
import numpy as np
|
||||
import os
|
||||
import time
|
||||
|
||||
@ -13,6 +13,7 @@ class PreprocessDatabaseCache:
|
||||
def __init__(self) -> None:
|
||||
self.node_size = None
|
||||
self.node_weight = None
|
||||
self.is_macro = None
|
||||
self.pin_id2node_id = None
|
||||
self.node2pin_list = None
|
||||
self.node2pin_list_end = None
|
||||
@ -20,6 +21,7 @@ class PreprocessDatabaseCache:
|
||||
def reset(self):
|
||||
self.node_size = None
|
||||
self.node_weight = None
|
||||
self.is_macro = None
|
||||
self.pin_id2node_id = None
|
||||
self.node2pin_list = None
|
||||
self.node2pin_list_end = None
|
||||
@ -76,10 +78,11 @@ def rearrange_dpdb_node_info(node_pos: torch.Tensor, data: PlaceData):
|
||||
if preprocess_db_cache.node_size is not None:
|
||||
node_size = preprocess_db_cache.node_size.clone()
|
||||
node_weight = preprocess_db_cache.node_weight
|
||||
is_macro = preprocess_db_cache.is_macro
|
||||
pin_id2node_id = preprocess_db_cache.pin_id2node_id
|
||||
node2pin_list = preprocess_db_cache.node2pin_list
|
||||
node2pin_list_end = preprocess_db_cache.node2pin_list_end
|
||||
return node_lpos, node_size, node_weight, pin_id2node_id, node2pin_list, node2pin_list_end
|
||||
return node_lpos, node_size, node_weight, is_macro, pin_id2node_id, node2pin_list, node2pin_list_end
|
||||
|
||||
pin_id2node_id: torch.Tensor = data.pin_id2node_id.clone().int().cpu().numpy()
|
||||
|
||||
@ -100,6 +103,14 @@ def rearrange_dpdb_node_info(node_pos: torch.Tensor, data: PlaceData):
|
||||
node_weight_ori[blkg_rhs:floatiopin_rhs]
|
||||
), dim=0)
|
||||
|
||||
is_macro = torch.cat((
|
||||
data.is_macro[:fix_rhs],
|
||||
data.is_macro[iopin_rhs:blkg_rhs],
|
||||
data.is_macro[floatiopin_rhs:floatfix_rhs],
|
||||
data.is_macro[fix_rhs:iopin_rhs],
|
||||
data.is_macro[blkg_rhs:floatiopin_rhs]
|
||||
), dim=0)
|
||||
|
||||
old_node2pin_list_end: torch.Tensor = data.node2pin_list_end.int()
|
||||
old_node2pin_list: torch.Tensor = data.node2pin_list.int()
|
||||
|
||||
@ -137,18 +148,19 @@ def rearrange_dpdb_node_info(node_pos: torch.Tensor, data: PlaceData):
|
||||
if preprocess_db_cache.node_size is None:
|
||||
preprocess_db_cache.node_size = node_size
|
||||
preprocess_db_cache.node_weight = node_weight
|
||||
preprocess_db_cache.is_macro = is_macro
|
||||
preprocess_db_cache.pin_id2node_id = pin_id2node_id
|
||||
preprocess_db_cache.node2pin_list = node2pin_list
|
||||
preprocess_db_cache.node2pin_list_end = node2pin_list_end
|
||||
|
||||
return node_lpos, node_size, node_weight, pin_id2node_id, node2pin_list, node2pin_list_end
|
||||
return node_lpos, node_size, node_weight, is_macro, pin_id2node_id, node2pin_list, node2pin_list_end
|
||||
|
||||
|
||||
def setup_detailed_rawdb(
|
||||
node_pos: torch.Tensor, use_cpu_db_: bool, data: PlaceData, args, logger, after_lg=True
|
||||
):
|
||||
curr_site_width = 1.0 # prescale_by_site_width
|
||||
node_lpos, node_size, node_weight, pin_id2node_id, node2pin_list, node2pin_list_end = rearrange_dpdb_node_info(
|
||||
node_lpos, node_size, node_weight, is_macro, pin_id2node_id, node2pin_list, node2pin_list_end = rearrange_dpdb_node_info(
|
||||
node_pos, data
|
||||
)
|
||||
if after_lg:
|
||||
@ -164,19 +176,6 @@ def setup_detailed_rawdb(
|
||||
|
||||
mov_lhs, mov_rhs = data.movable_index
|
||||
conn_mov_lhs, conn_mov_rhs = data.movable_connected_index
|
||||
if args.scale_design:
|
||||
# scale back
|
||||
die_scale = data.die_scale / data.site_width # assume site width == 1 in dp
|
||||
node_lpos = node_lpos * die_scale
|
||||
node_size = node_size * die_scale
|
||||
pin_rel_lpos = data.pin_rel_lpos * die_scale
|
||||
die_info = (data.die_info.reshape(2, 2).t() * die_scale).t().reshape(-1)
|
||||
region_boxes = (
|
||||
(data.region_boxes.reshape(-1, 2, 2).permute(0, 2, 1) * die_scale)
|
||||
.permute(0, 2, 1)
|
||||
.reshape(-1, 4)
|
||||
) # [:, 0] -> lx, [:, 1] -> hx, [:, 2] -> ly, [:, 3] -> hy
|
||||
else:
|
||||
pin_rel_lpos = data.pin_rel_lpos
|
||||
die_info = data.die_info
|
||||
region_boxes = data.region_boxes
|
||||
@ -206,6 +205,7 @@ def setup_detailed_rawdb(
|
||||
node_lpos.cpu(),
|
||||
node_size.cpu(),
|
||||
node_weight.cpu(),
|
||||
is_macro.cpu(),
|
||||
pin_rel_lpos.cpu(),
|
||||
pin_id2node_id.cpu(),
|
||||
data.pin_id2net_id.int().cpu(),
|
||||
@ -232,6 +232,7 @@ def setup_detailed_rawdb(
|
||||
node_lpos,
|
||||
node_size,
|
||||
node_weight,
|
||||
is_macro.cpu(),
|
||||
pin_rel_lpos,
|
||||
pin_id2node_id,
|
||||
data.pin_id2net_id.int(),
|
||||
@ -304,17 +305,73 @@ def commit_to_node_pos(node_pos: torch.Tensor, data:PlaceData, dp_rawdb):
|
||||
return node_pos
|
||||
|
||||
|
||||
def run_macro_legalization(node_pos, data: PlaceData, lg_rawdb, args, logger):
|
||||
if data.num_movable_macros == 0 or (data.num_movable_macros == 1 and data.num_fixed_macros == 0):
|
||||
return True
|
||||
if data.num_fixed_macros != 0:
|
||||
logger.warning("Including fixed macros. LP formula may be infeasible.")
|
||||
# Pre-process: get macro info
|
||||
mov_lhs, mov_rhs = data.movable_index
|
||||
macro_size = data.node_size[data.is_macro].contiguous()
|
||||
macro_pos = node_pos[data.is_macro].detach().contiguous()
|
||||
macro_fixed = (torch.cat([
|
||||
torch.zeros(mov_rhs - mov_lhs, dtype=torch.bool, device=node_pos.device),
|
||||
torch.ones(node_pos.shape[0] - mov_rhs, dtype=torch.bool, device=node_pos.device)
|
||||
])[data.is_macro]).contiguous()
|
||||
macro_weights = torch.ones_like(macro_pos)
|
||||
# macro_id = torch.zeros_like(data.is_macro, dtype=torch.long)
|
||||
# macro_id[data.is_macro] = torch.cumsum(data.is_macro, 0)[data.is_macro]
|
||||
inv_scalar = torch.tensor(
|
||||
[round(1.0 / get_ori_scale_factor(data))], dtype=torch.float32, device=node_pos.device
|
||||
)
|
||||
macro_info = (
|
||||
macro_pos, macro_size, macro_fixed, macro_weights,
|
||||
data.die_ll, data.die_ur, data.die_info, inv_scalar
|
||||
)
|
||||
# Macro legalization
|
||||
macro_pos, solve_success = macro_legalization_multi(macro_info, args, logger)
|
||||
node_pos_cache = node_pos.clone()
|
||||
node_pos_cache[data.is_macro] = torch.from_numpy(macro_pos).to(node_pos.device)
|
||||
node_pos[mov_lhs:mov_rhs] = node_pos_cache[mov_lhs:mov_rhs]
|
||||
# Post-process: round to integer and commit to lg_rawdb
|
||||
node_lpos, _, _, _, _, _, _ = rearrange_dpdb_node_info(node_pos, data)
|
||||
_, floatmov_rhs, _ = data.node_type_indices[1]
|
||||
node_lpos[:floatmov_rhs][data.is_macro[:floatmov_rhs]] = node_lpos[:floatmov_rhs][
|
||||
data.is_macro[:floatmov_rhs]].contiguous().mul_(inv_scalar).round_().div_(inv_scalar)
|
||||
|
||||
lg_rawdb.commit_from_partial(node_lpos[:, 0], node_lpos[:, 1])
|
||||
|
||||
return solve_success
|
||||
|
||||
|
||||
def macro_legalization_main(node_pos: torch.Tensor, data: PlaceData, args, logger, lg_rawdb=None):
|
||||
if lg_rawdb is None:
|
||||
lg_rawdb = setup_detailed_rawdb(node_pos, True, data, args, logger, after_lg=False)
|
||||
logger.info("Start running Macro Legalization... #Macros: %d, #MovMacros: %d." % (
|
||||
data.num_macros, data.num_movable_macros
|
||||
))
|
||||
ml_time = time.time()
|
||||
run_macro_legalization(node_pos, data, lg_rawdb, args, logger)
|
||||
# align to site/row and check
|
||||
if gpudp.macroLegalization(lg_rawdb, data.num_bin_x, data.num_bin_y):
|
||||
lg_rawdb.commit()
|
||||
logger.info("Check Pass in Macro Legalization")
|
||||
else:
|
||||
logger.error("Check failed in Macro Legalization.")
|
||||
# Commit result
|
||||
commit_to_node_pos(node_pos, data, lg_rawdb)
|
||||
torch.cuda.synchronize(node_pos.device)
|
||||
logger.info("***** Finish Macro Legalization, HPWL: %.4E Time: %.4f *****" % (
|
||||
get_obj_hpwl(node_pos, data, args).item(), time.time() - ml_time
|
||||
))
|
||||
|
||||
|
||||
def run_lg(node_pos: torch.Tensor, data: PlaceData, args, logger):
|
||||
# CPU legalization
|
||||
lg_rawdb = setup_detailed_rawdb(node_pos, True, data, args, logger, after_lg=False)
|
||||
|
||||
# run LG
|
||||
logger.info("Start running Macro Legalization...")
|
||||
ml_time = time.time()
|
||||
num_bins_x, num_bins_y = data.num_bin_x, data.num_bin_y
|
||||
if gpudp.macroLegalization(lg_rawdb, num_bins_x, num_bins_y):
|
||||
lg_rawdb.commit()
|
||||
logger.info("Finish Macro Legalization. Time: %.4f" % (time.time() - ml_time))
|
||||
macro_legalization_main(node_pos, data, args, logger, lg_rawdb)
|
||||
|
||||
total_cell_area = torch.sum(torch.prod(data.node_size, 1)).item()
|
||||
die_area = torch.prod(data.die_ur - data.die_ll).item()
|
||||
@ -346,8 +403,6 @@ def run_lg(node_pos: torch.Tensor, data: PlaceData, args, logger):
|
||||
# # Commit result
|
||||
# commit_to_node_pos(node_pos, data, lg_rawdb)
|
||||
# torch.cuda.synchronize(node_pos.device)
|
||||
# if args.scale_design:
|
||||
# node_pos /= data.die_scale
|
||||
# info = (-1, 0, data.design_name)
|
||||
# draw_fig_with_cairo_cpp(node_pos, data.node_size, data, info, args, base_size=4096)
|
||||
|
||||
@ -368,9 +423,6 @@ def run_lg(node_pos: torch.Tensor, data: PlaceData, args, logger):
|
||||
get_obj_hpwl(node_pos, data, args).item(), time.time() - gl_time
|
||||
))
|
||||
|
||||
if args.scale_design:
|
||||
node_pos /= data.die_scale
|
||||
|
||||
del lg_rawdb
|
||||
|
||||
return node_pos
|
||||
@ -449,9 +501,6 @@ def run_dp(node_pos: torch.Tensor, data: PlaceData, args, logger):
|
||||
dp_handler(gpudp.globalSwap, "Global Swap", num_bins_x // 2, num_bins_y // 2, gs_bs, gs_iter)
|
||||
dp_handler(gpudp.kReorder, "K-Reorder 2", num_bins_x, num_bins_y, kr_K, kr_iter)
|
||||
|
||||
if args.scale_design:
|
||||
node_pos /= data.die_scale
|
||||
|
||||
del dp_rawdb
|
||||
|
||||
return node_pos
|
||||
@ -479,12 +528,6 @@ def run_dp_route_opt(node_pos: torch.Tensor, gpdb, rawdb, ps, data: PlaceData, a
|
||||
node_lpos[:floatmov_rhs].mul_(inv_scalar).round_().div_(inv_scalar)
|
||||
|
||||
node_size = data.node_size.cpu()
|
||||
if args.scale_design:
|
||||
# scale back
|
||||
die_scale = data.die_scale / data.site_width # assume site width == 1 in dp
|
||||
node_lpos = node_lpos * die_scale
|
||||
node_size = node_size * die_scale
|
||||
die_info = (data.die_info.reshape(2, 2).t() * die_scale).t().reshape(-1)
|
||||
site_width = 1.0
|
||||
row_height = data.row_height / data.site_width
|
||||
die_info = data.die_info.cpu()
|
||||
|
||||
@ -1,7 +1,7 @@
|
||||
import torch
|
||||
from .database import PlaceData
|
||||
from .core import masked_scale_hpwl
|
||||
from cpp_to_py import hpwl_cuda
|
||||
from cpp_to_py import hpwl_cuda, density_map_cuda
|
||||
|
||||
def get_hpwl(data, pos):
|
||||
# CUDA only
|
||||
@ -23,23 +23,39 @@ def get_obj_hpwl(node_pos, data: PlaceData, args):
|
||||
hpwl = torch.sum(get_hpwl(data, pin_pos.detach()))
|
||||
return hpwl
|
||||
|
||||
def get_obj_overflow(node_pos, density_map_layer, init_density_map, data: PlaceData, args):
|
||||
def get_obj_overflow(node_pos, init_density_map, ps, data: PlaceData, args):
|
||||
mov_lhs, mov_rhs = data.movable_index
|
||||
density_map = density_map_layer.get_density_map_naive(
|
||||
node_pos[mov_lhs:mov_rhs], data.node_size[mov_lhs:mov_rhs], init_density_map
|
||||
node_pos = node_pos[mov_lhs:mov_rhs]
|
||||
node_size = data.node_size[mov_lhs:mov_rhs]
|
||||
if ps.zero_macro_grad:
|
||||
node_pos = node_pos[
|
||||
torch.logical_not(data.is_mov_macro[mov_lhs:mov_rhs])
|
||||
].contiguous()
|
||||
node_size = node_size[
|
||||
torch.logical_not(data.is_mov_macro[mov_lhs:mov_rhs])
|
||||
].contiguous()
|
||||
node_weight = node_size.new_ones(node_pos.shape[0])
|
||||
|
||||
if init_density_map is None:
|
||||
init_density_map = node_pos.new_zeros(data.num_bin_x, data.num_bin_y)
|
||||
aux_mat = init_density_map.clone()
|
||||
density_map = density_map_cuda.forward_naive(
|
||||
node_pos, node_size, node_weight, data.unit_len, aux_mat, data.num_bin_x,
|
||||
data.num_bin_y, node_pos.shape[0], -1.0, -1.0, 1e-4, False, args.deterministic,
|
||||
)
|
||||
|
||||
with torch.no_grad():
|
||||
overflow_sum = ((density_map - args.target_density) * data.bin_area).clamp_(min=0.0).sum()
|
||||
overflow = overflow_sum / data.total_mov_area_without_filler
|
||||
return overflow
|
||||
|
||||
def evaluate_placement(node_pos, density_map_layer, init_density_map, data: PlaceData, args):
|
||||
def evaluate_placement(node_pos, init_density_map, ps, data: PlaceData, args):
|
||||
# NOTE: since some nets are masked in global placement, hpwl may
|
||||
# underestimate, this function return the exact value of hpwl
|
||||
# Original overflow calculation uses the clamp node size (expand ratio),
|
||||
# this function uses the exact node size to evaluate the overflow
|
||||
hpwl = get_obj_hpwl(node_pos, data, args)
|
||||
overflow = get_obj_overflow(node_pos, density_map_layer, init_density_map, data, args)
|
||||
overflow = get_obj_overflow(node_pos, init_density_map, ps, data, args)
|
||||
return hpwl, overflow
|
||||
|
||||
def fast_evaluator(
|
||||
|
||||
@ -1,24 +1,28 @@
|
||||
import torch
|
||||
from .database import PlaceData
|
||||
from cpp_to_py import density_map_cuda
|
||||
from .core import WAWirelengthLossAndHPWL
|
||||
from .calculator import calc_grad
|
||||
from .core import merged_wl_loss_grad
|
||||
|
||||
|
||||
def get_init_density_map(rawdb, gpdb, data: PlaceData, args, logger):
|
||||
def get_init_density_map(rawdb, gpdb, data: PlaceData, args, logger, ps=None):
|
||||
lhs, rhs = data.fixed_index
|
||||
device = data.node_size.get_device()
|
||||
dtype = data.node_size.dtype
|
||||
zeros_density_map = torch.zeros(
|
||||
(data.num_bin_x, data.num_bin_y), device=device, dtype=dtype,
|
||||
)
|
||||
if lhs == rhs:
|
||||
if lhs == rhs and (ps is None or not ps.zero_macro_grad):
|
||||
data.init_density_map = zeros_density_map
|
||||
return zeros_density_map
|
||||
# get fix nodes which are located inside die
|
||||
node_pos = data.node_pos[lhs:rhs]
|
||||
node_size = data.node_size[lhs:rhs]
|
||||
node_weight = node_size.new_ones(node_size.shape[0])
|
||||
if ps is not None and ps.zero_macro_grad:
|
||||
# compute the mov + fixed macro density map
|
||||
node_pos = torch.cat([data.node_pos[data.is_mov_macro].contiguous(), node_pos])
|
||||
node_size = torch.cat([data.node_size[data.is_mov_macro].contiguous(), node_size])
|
||||
node_weight = node_size.new_ones(node_size.shape[0])
|
||||
init_density_map = density_map_cuda.forward_naive(
|
||||
node_pos, node_size, node_weight, data.unit_len, zeros_density_map,
|
||||
data.num_bin_x, data.num_bin_y, node_pos.shape[0], -1.0, -1.0, 1e-4, False,
|
||||
@ -105,22 +109,27 @@ def init_params(
|
||||
conn_node_pos = torch.cat(
|
||||
[conn_node_pos, conn_fix_node_pos], dim=0
|
||||
)
|
||||
wl_loss, hpwl = WAWirelengthLossAndHPWL.apply(
|
||||
_, conn_node_grad = merged_wl_loss_grad(
|
||||
conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos,
|
||||
data.node2pin_list, data.node2pin_list_end,
|
||||
data.hyperedge_list, data.hyperedge_list_end, data.net_mask,
|
||||
ps.wa_coeff, data.hpwl_scale, args.deterministic
|
||||
data.hpwl_scale, ps.wa_coeff, args.deterministic
|
||||
)
|
||||
density_loss, overflow = density_map_layer(
|
||||
wl_grad = torch.zeros_like(mov_node_pos).detach()
|
||||
wl_grad[mov_lhs:mov_rhs] = conn_node_grad[mov_lhs:mov_rhs]
|
||||
_, _, density_grad = density_map_layer.merged_density_loss_grad(
|
||||
mov_node_pos, mov_node_size, init_density_map
|
||||
)
|
||||
wl_grad, density_grad = calc_grad(
|
||||
optimizer, mov_node_pos, wl_loss, density_loss
|
||||
)
|
||||
if ps.zero_macro_grad:
|
||||
wl_grad[mov_lhs:mov_rhs].masked_fill_(data.is_mov_macro[mov_lhs:mov_rhs].unsqueeze(1), 0)
|
||||
density_grad[mov_lhs:mov_rhs].masked_fill_(data.is_mov_macro[mov_lhs:mov_rhs].unsqueeze(1), 0)
|
||||
(mov_node_pos * 0.0).sum().backward()
|
||||
optimizer.zero_grad(set_to_none=False)
|
||||
|
||||
if not ps.rerun_route or route_fn is None:
|
||||
init_density_weight = (wl_grad.norm(p=1) / density_grad.norm(p=1)).detach()
|
||||
# init_density_weight = (wl_grad.norm(p=1) / grad_mat.norm(p=1)).detach()
|
||||
ps.set_init_param(init_density_weight, data, density_loss)
|
||||
ps.set_init_param(init_density_weight, data)
|
||||
else:
|
||||
_, filler_lhs = data.movable_connected_index
|
||||
filler_rhs = mov_node_pos.shape[0]
|
||||
|
||||
@ -70,6 +70,7 @@ class MetricRecorder:
|
||||
class ParamScheduler:
|
||||
def __init__(self, data: PlaceData, args, logger) -> None:
|
||||
self.__logger__ = logger
|
||||
self.__args__ = args
|
||||
self.data = data
|
||||
self.iter = 0
|
||||
self.init_iter = 0
|
||||
@ -97,7 +98,7 @@ class ParamScheduler:
|
||||
self.best_sol_rollback: torch.Tensor = None
|
||||
self.best_metric_rollback = {"overflow": float("inf"), "hpwl": float("inf")}
|
||||
|
||||
# params
|
||||
# global place params
|
||||
self.precond_coef = 1.0
|
||||
self.precond_weight = None
|
||||
self.density_weight_start = args.density_weight
|
||||
@ -113,13 +114,15 @@ class ParamScheduler:
|
||||
self.life = self.max_life
|
||||
self.stop_overflow = args.stop_overflow
|
||||
self.skip_update = False if args.enable_skip_update else None
|
||||
self.enable_fence = data.enable_fence
|
||||
self.min_enlarge_density_interval = 1000
|
||||
self.last_enlarge_density_iter = -self.min_enlarge_density_interval
|
||||
# skip density force
|
||||
self.enable_sample_force = True
|
||||
self.enable_sample_force = args.enable_sample_force
|
||||
self.force_ratio = 0.0
|
||||
|
||||
self.enable_fence = data.enable_fence
|
||||
|
||||
# routability parameter
|
||||
self.enable_route = args.use_route_force or args.use_cell_inflate
|
||||
self.use_cell_inflate = args.use_cell_inflate
|
||||
self.use_route_force = args.use_route_force
|
||||
@ -140,14 +143,21 @@ class ParamScheduler:
|
||||
self.max_route_opt = 5
|
||||
self.gr_sol_recorder = []
|
||||
|
||||
def set_init_param(self, init_density_weight, data: PlaceData, init_density_loss):
|
||||
# mixed size parameter
|
||||
self.enable_mixed_size = args.mixed_size
|
||||
self.include_macros = args.include_macros
|
||||
self.zero_macro_grad = False
|
||||
|
||||
def set_init_param(self, init_density_weight, data: PlaceData):
|
||||
# init_density_weight
|
||||
self.init_iter = self.iter
|
||||
self.all_init_iters.append(self.init_iter)
|
||||
self.precond_coef = 1.0
|
||||
self.mu = 1.0
|
||||
self.density_weight = copy.deepcopy(self.density_weight_start) * init_density_weight
|
||||
self.wa_coeff = copy.deepcopy(self.wa_coeff_start)
|
||||
self.update_precond_weight(data)
|
||||
self.set_mixsize_init_param()
|
||||
|
||||
def set_route_init_param(
|
||||
self, init_density_weight, init_route_weight, init_congest_weight, data: PlaceData, args
|
||||
@ -157,12 +167,38 @@ class ParamScheduler:
|
||||
self.init_iter = self.iter
|
||||
self.all_init_iters.append(self.init_iter)
|
||||
self.precond_coef = 1.0
|
||||
self.mu = 1.0
|
||||
self.base_route_weight = init_route_weight * args.route_weight
|
||||
self.base_congest_weight = init_congest_weight * args.congest_weight
|
||||
self.route_weight = copy.deepcopy(self.density_weight) * self.base_route_weight
|
||||
self.congest_weight = copy.deepcopy(self.density_weight) * self.base_congest_weight
|
||||
self.pseudo_weight = args.pseudo_weight # same scale as wirelength weight
|
||||
self.update_precond_weight(data)
|
||||
self.set_mixsize_init_param()
|
||||
|
||||
def set_mixsize_init_param(self):
|
||||
args = self.__args__
|
||||
if self.include_macros:
|
||||
self.skip_update = None
|
||||
self.enable_sample_force = False
|
||||
if self.enable_mixed_size:
|
||||
if not self.zero_macro_grad:
|
||||
# simultaneously place macro and std cells
|
||||
self.include_macros = True
|
||||
self.stop_overflow = args.stop_overflow * 2.0
|
||||
self.enable_sample_force = False
|
||||
self.skip_update = None
|
||||
self.enable_route = False
|
||||
self.use_cell_inflate = False
|
||||
self.use_route_force = False
|
||||
else:
|
||||
self.include_macros = False
|
||||
self.stop_overflow = args.stop_overflow
|
||||
self.enable_sample_force = args.enable_sample_force
|
||||
self.skip_update = False if args.enable_skip_update else None
|
||||
self.enable_route = args.use_route_force or args.use_cell_inflate
|
||||
self.use_cell_inflate = args.use_cell_inflate
|
||||
self.use_route_force = args.use_route_force
|
||||
|
||||
def reset_best_sol(self):
|
||||
# best solution
|
||||
@ -384,6 +420,7 @@ class ParamScheduler:
|
||||
if (
|
||||
self.recorder.overflow[ptr] < self.stop_overflow * 5
|
||||
and self.recorder.overflow[ptr] >= self.stop_overflow
|
||||
and not self.include_macros
|
||||
):
|
||||
if self.check_plateau(self.recorder.overflow, window=50, threshold=0.05):
|
||||
# kill the program since it has converged
|
||||
|
||||
@ -20,9 +20,6 @@ def run_placement_main_nesterov(args, logger):
|
||||
assert args.use_eplace_nesterov
|
||||
logger.info("Start place %s/%s" % (args.dataset , args.design_name))
|
||||
logger.info("Use Nesterov optimizer!")
|
||||
if args.scale_design:
|
||||
logger.warning("Eplace's nesterov optimizer cannot support normalized die. Disable scale_design.")
|
||||
args.scale_design = False
|
||||
data = data.to(device)
|
||||
data = data.preprocess()
|
||||
logger.info(data)
|
||||
@ -140,6 +137,7 @@ def run_placement_main_nesterov(args, logger):
|
||||
# exit(0)
|
||||
terminate_signal = False
|
||||
route_early_terminate_signal = False
|
||||
log_info = False
|
||||
for iteration in range(args.inner_iter):
|
||||
# optimizer.zero_grad() # zero grad inside obj_and_grad_fn
|
||||
obj = optimizer.step(obj_and_grad_fn)
|
||||
@ -148,6 +146,68 @@ def run_placement_main_nesterov(args, logger):
|
||||
ps.step(hpwl, overflow, mov_node_pos, data)
|
||||
if ps.need_to_early_stop():
|
||||
terminate_signal = True
|
||||
log_info = True
|
||||
|
||||
if ps.enable_mixed_size and not ps.zero_macro_grad and terminate_signal:
|
||||
ps.zero_macro_grad = True
|
||||
# Find best gp node_pos (including macros and std cells)
|
||||
best_res = ps.get_best_solution()
|
||||
if best_res[0] is not None:
|
||||
best_sol, hpwl, overflow = best_res
|
||||
# fillers are unused from now, we don't copy there data
|
||||
mov_node_pos[mov_lhs:mov_rhs].data.copy_(best_sol[mov_lhs:mov_rhs])
|
||||
node_pos = mov_node_pos[mov_lhs:mov_rhs]
|
||||
node_pos = torch.cat([node_pos, data.node_pos[mov_rhs:]], dim=0)
|
||||
# Evaluate the mixed placement solution
|
||||
hpwl, overflow = evaluate_placement(node_pos, init_density_map, ps, data, args)
|
||||
hpwl, overflow = hpwl.item(), overflow.item()
|
||||
if args.draw_placement:
|
||||
info = ("%d_mixed_gp" % (iteration + 1), hpwl, data.design_name)
|
||||
draw_fig_with_cairo_cpp(node_pos, data.node_size, data, info, args)
|
||||
logger.info("After Mixed-GP, best solution eval, exact HPWL: %.4E exact Overflow: %.4f" % (hpwl, overflow))
|
||||
# Run macro legalization to change node_pos inplace
|
||||
macro_legalization_main(node_pos, data, args, logger)
|
||||
if args.draw_placement:
|
||||
info = ("%d_mixed_gp_ml" % (iteration + 1), hpwl, data.design_name)
|
||||
draw_fig_with_cairo_cpp(node_pos, data.node_size, data, info, args)
|
||||
# Write node_pos into database to provide an initial solution for std cell placement
|
||||
data.node_pos[mov_lhs:mov_rhs].data.copy_(node_pos[mov_lhs:mov_rhs])
|
||||
# Prepare for std cell placement
|
||||
init_density_map = get_init_density_map(rawdb, gpdb, data, args, logger, ps=ps)
|
||||
data.__total_mov_area_without_filler__ = torch.sum(data.node_area[mov_lhs:mov_rhs][torch.logical_not(data.is_mov_macro[mov_lhs:mov_rhs])]).item()
|
||||
mov_node_pos, mov_node_size, expand_ratio = data.get_mov_node_info(init_method="randn_center")
|
||||
mov_macros_idx = data.is_mov_macro[mov_lhs:mov_rhs]
|
||||
mov_node_pos[mov_lhs:mov_rhs][mov_macros_idx] = data.node_pos[mov_lhs:mov_rhs][mov_macros_idx]
|
||||
mov_node_pos = mov_node_pos.requires_grad_(True)
|
||||
trunc_node_pos_fn = get_trunc_node_pos_fn(mov_node_size, data)
|
||||
density_map_layer.expand_ratio = expand_ratio
|
||||
density_map_layer.sorted_maps = data.sorted_maps
|
||||
# ignore the density and grad computation of macros by node_weight
|
||||
density_map_layer.cache_node_weight[mov_lhs:mov_rhs][mov_macros_idx] = -1.0
|
||||
# update partial function correspondingly
|
||||
obj_and_grad_fn.keywords["constraint_fn"] = trunc_node_pos_fn
|
||||
obj_and_grad_fn.keywords["mov_node_size"] = mov_node_size
|
||||
obj_and_grad_fn.keywords["expand_ratio"] = expand_ratio
|
||||
obj_and_grad_fn.keywords["init_density_map"] = init_density_map
|
||||
evaluator_fn.keywords["constraint_fn"] = trunc_node_pos_fn
|
||||
evaluator_fn.keywords["mov_node_size"] = mov_node_size
|
||||
evaluator_fn.keywords["init_density_map"] = init_density_map
|
||||
# reset nesterov optimizer
|
||||
logger.info("Reset optimizer...")
|
||||
optimizer = NesterovOptimizer([mov_node_pos], lr=0)
|
||||
# initialization
|
||||
init_params(
|
||||
mov_node_pos, trunc_node_pos_fn, mov_lhs, mov_rhs, conn_fix_node_pos,
|
||||
density_map_layer, mov_node_size, expand_ratio, init_density_map, optimizer,
|
||||
ps, data, args, route_fn=calc_route_force
|
||||
)
|
||||
# init learnig rate
|
||||
cur_lr = estimate_initial_learning_rate(obj_and_grad_fn, trunc_node_pos_fn, mov_node_pos, args.lr)
|
||||
for param_group in optimizer.param_groups:
|
||||
param_group["lr"] = cur_lr.item()
|
||||
ps.reset_best_sol()
|
||||
terminate_signal = False # reset signal
|
||||
logger.info("Re-run std cell placement with fixed macros.")
|
||||
|
||||
if ps.use_cell_inflate and ps.curr_optimizer_cnt < ps.max_route_opt and terminate_signal:
|
||||
terminate_signal = False # reset signal
|
||||
@ -185,6 +245,7 @@ def run_placement_main_nesterov(args, logger):
|
||||
ps.rerun_route = False
|
||||
|
||||
if ps.rerun_route:
|
||||
log_info = True
|
||||
new_mov_node_size, new_expand_ratio = None, None
|
||||
if ps.use_cell_inflate:
|
||||
output = route_inflation(
|
||||
@ -245,7 +306,8 @@ def run_placement_main_nesterov(args, logger):
|
||||
)
|
||||
ps.reset_best_sol()
|
||||
|
||||
if iteration % args.log_freq == 0 or iteration == args.inner_iter - 1 or ps.rerun_route or terminate_signal:
|
||||
if iteration % args.log_freq == 0 or iteration == args.inner_iter - 1 or log_info:
|
||||
log_info = False
|
||||
log_str = (
|
||||
"iter: %d | masked_hpwl: %.2E overflow: %.4f obj: %.4E "
|
||||
"density_weight: %.4E wa_coeff: %.4E"
|
||||
@ -290,6 +352,10 @@ def run_placement_main_nesterov(args, logger):
|
||||
best_sol, hpwl, overflow = best_res
|
||||
# fillers are unused from now, we don't copy there data
|
||||
mov_node_pos[mov_lhs:mov_rhs].data.copy_(best_sol[mov_lhs:mov_rhs])
|
||||
if ps.enable_mixed_size and ps.zero_macro_grad:
|
||||
# rollback macro_pos to previous legalized results since trunc_node_pos_fn may change them
|
||||
mov_macros_idx = data.is_mov_macro[mov_lhs:mov_rhs]
|
||||
mov_node_pos.data[mov_lhs:mov_rhs][mov_macros_idx] = data.node_pos[mov_lhs:mov_rhs][mov_macros_idx]
|
||||
if ps.enable_route:
|
||||
route_inflation_roll_back(args, logger, data, mov_node_size)
|
||||
if not route_early_terminate_signal:
|
||||
@ -314,9 +380,7 @@ def run_placement_main_nesterov(args, logger):
|
||||
)
|
||||
|
||||
# Eval
|
||||
hpwl, overflow = evaluate_placement(
|
||||
node_pos, density_map_layer, init_density_map, data, args
|
||||
)
|
||||
hpwl, overflow = evaluate_placement(node_pos, init_density_map, ps, data, args)
|
||||
hpwl, overflow = hpwl.item(), overflow.item()
|
||||
info = ("%d_gp" % (iteration + 1), hpwl, data.design_name)
|
||||
if args.draw_placement:
|
||||
|
||||
@ -4,6 +4,8 @@ import os
|
||||
def find_benchmark(dataset_root, benchmark):
|
||||
bm_to_root = {
|
||||
"ispd2005": os.path.join(dataset_root, "ispd2005"),
|
||||
"ispd2006": os.path.join(dataset_root, "ispd2006"),
|
||||
"mms": os.path.join(dataset_root, "mms"),
|
||||
"dac2012": os.path.join(dataset_root, "iccad2012dac2012"),
|
||||
"ispd2015": os.path.join(dataset_root, "ispd2015"),
|
||||
"ispd2015_fix": os.path.join(dataset_root, "ispd2015_fix"),
|
||||
@ -11,6 +13,7 @@ def find_benchmark(dataset_root, benchmark):
|
||||
"ispd2019_no_fence": os.path.join(dataset_root, "ispd2019_no_fence"),
|
||||
"iccad2019": os.path.join(dataset_root, "iccad2019"),
|
||||
"ispd2018": os.path.join(dataset_root, "ispd2018"),
|
||||
"iccad2015": os.path.join(dataset_root, "iccad2015"),
|
||||
}
|
||||
root = bm_to_root[benchmark]
|
||||
all_designs = [i for i in os.listdir(root) if os.path.isdir(os.path.join(root, i))]
|
||||
@ -18,8 +21,8 @@ def find_benchmark(dataset_root, benchmark):
|
||||
|
||||
|
||||
def get_single_design_params(dataset_root, benchmark, design_name, placement=None):
|
||||
if benchmark == "ispd2005":
|
||||
return single_ispd2005(dataset_root, design_name, placement)
|
||||
if benchmark in ["ispd2005", "ispd2006", "mms"]:
|
||||
return single_ispd2005(dataset_root, design_name, benchmark, placement)
|
||||
elif benchmark == "dac2012":
|
||||
return single_dac2012(dataset_root, design_name, placement)
|
||||
elif benchmark == "ispd2015":
|
||||
@ -34,6 +37,8 @@ def get_single_design_params(dataset_root, benchmark, design_name, placement=Non
|
||||
return single_iccad2019(dataset_root, design_name, placement)
|
||||
elif benchmark == "ispd2018":
|
||||
return single_ispd2018(dataset_root, design_name, placement)
|
||||
elif benchmark.startswith("iccad2015"):
|
||||
return single_iccad2015(dataset_root, benchmark, design_name, placement)
|
||||
else:
|
||||
raise NotImplementedError("benchmark %s is not found" % benchmark)
|
||||
|
||||
@ -47,8 +52,8 @@ def get_multiple_design_params(dataset_root, benchmark):
|
||||
return params_mul
|
||||
|
||||
|
||||
def single_ispd2005(dataset_root, design_name, placement=None):
|
||||
benchmark = "ispd2005"
|
||||
def single_ispd2005(dataset_root, design_name, benchmark, placement=None):
|
||||
benchmark = benchmark
|
||||
root, all_designs = find_benchmark(dataset_root, benchmark)
|
||||
if design_name not in all_designs:
|
||||
raise ValueError("Design Name %s should in %s" % (design_name, all_designs))
|
||||
@ -56,9 +61,10 @@ def single_ispd2005(dataset_root, design_name, placement=None):
|
||||
"benchmark": benchmark,
|
||||
"bookshelf_variety": "ispd2005",
|
||||
"aux": "%s/%s/%s.aux" % (root, design_name, design_name),
|
||||
"pl": "%s/%s/%s.pl" % (root, design_name, design_name) if placement is None else placement,
|
||||
"design_name": design_name,
|
||||
}
|
||||
if placement is not None:
|
||||
params["pl"] = placement
|
||||
return params
|
||||
|
||||
|
||||
@ -71,9 +77,10 @@ def single_dac2012(dataset_root, design_name, placement=None):
|
||||
"benchmark": benchmark,
|
||||
"bookshelf_variety": "dac2012",
|
||||
"aux": "%s/%s/%s.aux" % (root, design_name, design_name),
|
||||
"pl": "%s/%s/%s.pl" % (root, design_name, design_name) if placement is None else placement,
|
||||
"design_name": design_name,
|
||||
}
|
||||
if placement is not None:
|
||||
params["pl"] = placement
|
||||
return params
|
||||
|
||||
|
||||
@ -170,6 +177,23 @@ def single_ispd2018(dataset_root, design_name, placement=None):
|
||||
return params
|
||||
|
||||
|
||||
def single_iccad2015(dataset_root, benchmark, design_name, placement=None):
|
||||
# configuration
|
||||
# benchmark = "iccad2015"
|
||||
root, all_designs = find_benchmark(dataset_root, benchmark)
|
||||
if design_name not in all_designs:
|
||||
raise ValueError("Design Name %s should in %s" % (design_name, all_designs))
|
||||
params = {
|
||||
"benchmark": benchmark,
|
||||
"tech_lef": "%s/tech.lef" % (root),
|
||||
"cell_lef": "%s/%s/%s.lef" % (root, design_name, design_name),
|
||||
"def": "%s/%s/%s.def" % (root, design_name, design_name) if placement is None else placement,
|
||||
"verilog": "%s/%s/%s.v" % (root, design_name, design_name),
|
||||
"design_name": design_name,
|
||||
}
|
||||
return params
|
||||
|
||||
|
||||
def get_custom_design_params(args):
|
||||
params = dict(
|
||||
[
|
||||
|
||||
@ -71,19 +71,16 @@ class IOParser(object):
|
||||
print("def %s not exists." % params["def"])
|
||||
return False
|
||||
if "aux" in params.keys():
|
||||
if "pl" not in params.keys():
|
||||
print("pl is not found!")
|
||||
if not os.path.exists(params["aux"]):
|
||||
print("aux %s not exists." % params["aux"])
|
||||
return False
|
||||
if "pl" in params.keys() and not os.path.exists(params["pl"]):
|
||||
print("pl %s not exists." % params["pl"])
|
||||
return False
|
||||
if "output" in params.keys():
|
||||
if "pl" != params["output"].split(".")[-1]:
|
||||
print("output format should be .pl")
|
||||
return False
|
||||
if not os.path.exists(params["aux"]):
|
||||
print("aux %s not exists." % params["aux"])
|
||||
return False
|
||||
if not os.path.exists(params["pl"]):
|
||||
print("pl %s not exists." % params["pl"])
|
||||
return False
|
||||
self.params = params
|
||||
|
||||
if verbose_log:
|
||||
|
||||
@ -16,7 +16,7 @@ class CustomFormatter(logging.Formatter):
|
||||
# FIXME: An elapsed time gap exists between the C++ Timer and Python Timer,
|
||||
# but I don't know how to resolve it...
|
||||
time_format = "[%(relativeCreatedSecond)4d.%(relativeCreatedMSecond)03d] "
|
||||
debug_msg = " (%(module)s.py Line%(lineno)d) %(msg)s"
|
||||
debug_msg = " (%(module)s.py L%(lineno)d) %(msg)s"
|
||||
FORMATS = {
|
||||
logging.DEBUG: time_format + blue + "DEBUG" + reset + debug_msg,
|
||||
logging.INFO: time_format + "%(msg)s",
|
||||
@ -47,6 +47,8 @@ def setup_logger(args, sys_argv) -> logging.Logger:
|
||||
screen_handler = logging.StreamHandler(stream=sys.stdout)
|
||||
screen_handler.setFormatter(formatter)
|
||||
logger = logging.getLogger()
|
||||
logger.setLevel(logging.INFO)
|
||||
if args.cpp_log_level == 0 or args.verbose_cpp_log:
|
||||
logger.setLevel(logging.DEBUG)
|
||||
logger.addHandler(file_handler)
|
||||
logger.addHandler(screen_handler)
|
||||
|
||||
@ -44,6 +44,30 @@ def setup_design_args(args):
|
||||
elif args.design_name in ["bigblue3", "bigblue4"]:
|
||||
args.num_bin_x = args.num_bin_y = 2048
|
||||
args.target_density = 1.0
|
||||
elif args.design_name in ["adaptec5"]:
|
||||
args.target_density = 0.5
|
||||
args.num_bin_x = args.num_bin_y = 1024
|
||||
elif args.design_name in ["newblue1"]:
|
||||
args.target_density = 0.8
|
||||
args.num_bin_x = args.num_bin_y = 512
|
||||
elif args.design_name in ["newblue2"]:
|
||||
args.target_density = 0.9
|
||||
args.num_bin_x = args.num_bin_y = 1024
|
||||
elif args.design_name in ["newblue3"]:
|
||||
args.target_density = 0.8
|
||||
args.num_bin_x = args.num_bin_y = 2048
|
||||
elif args.design_name in ["newblue4"]:
|
||||
args.target_density = 0.5
|
||||
args.num_bin_x = args.num_bin_y = 1024
|
||||
elif args.design_name in ["newblue5"]:
|
||||
args.target_density = 0.5
|
||||
args.num_bin_x = args.num_bin_y = 1024
|
||||
elif args.design_name in ["newblue6"]:
|
||||
args.target_density = 0.8
|
||||
args.num_bin_x = args.num_bin_y = 2048
|
||||
elif args.design_name in ["newblue7"]:
|
||||
args.target_density = 0.8
|
||||
args.num_bin_x = args.num_bin_y = 2048
|
||||
elif args.design_name in ["mgc_des_perf_1"]:
|
||||
args.num_bin_x = args.num_bin_y = 512
|
||||
args.target_density = 0.91
|
||||
|
||||
Loading…
Reference in New Issue
Block a user