(1) support mixed size (2) support ispd2006/mms (3) remove legacy code

This commit is contained in:
liulixinkerry 2024-06-05 11:12:55 +08:00
parent a9bde59465
commit 8e7c817f46
29 changed files with 1473 additions and 666 deletions

View File

@ -20,7 +20,7 @@ Database::~Database() {
void Database::load() {
// ----- design related options -----
if (setting.BookshelfAux != "" && setting.BookshelfPl != "") {
if (setting.BookshelfAux != "") {
setting.Format = "bookshelf";
readBSAux(setting.BookshelfAux, setting.BookshelfPl);
}

View File

@ -342,6 +342,12 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
readBSLine(fs, tokens);
fs.close();
bool includePl = true;
if (plFile == "") {
logger.warning("No pl file specified. Try to find pl in aux file.");
includePl = false;
}
std::string fileNodes;
std::string fileNets;
std::string fileScl;
@ -376,6 +382,10 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
logger.error("unrecognized file extension: %s", ext.c_str());
}
}
if (includePl) {
logger.info("pl file %s is given.", plFile.c_str());
filePl = plFile;
}
// step 1: read floorplan, rows from:
// scl - rows
// step 2: read cell types from:
@ -399,16 +409,12 @@ bool Database::readBSAux(const std::string& auxFile, const std::string& plFile)
readBSRoute(fileRoute);
readBSShapes(fileShapes);
readBSWts(fileWts);
// readBSPl ( filePl );
readBSPl(plFile);
readBSPl(filePl);
readBSScl(fileScl);
} else if (bsData.format == "ispd2005") {
readBSNets(fileNets);
// readBSRoute ( fileRoute );
// readBSShapes( fileShapes);
readBSWts(fileWts);
// readBSPl ( filePl );
readBSPl(plFile);
readBSPl(filePl);
readBSScl(fileScl);
}

View File

@ -1477,10 +1477,10 @@ int readDefComponent(defrCallbackType_e c, defiComponent* co, defiUserData ud) {
cell->unplace();
} else if (co->isPlaced()) {
cell->place(co->placementX(), co->placementY(), co->placementOrient());
if (celltype->cls == "CORE") {
if (celltype->cls == "CORE" || celltype->cls == "BLOCK") {
cell->fixed(false);
} else {
// Set all non-CORE cells as fixed cells
// Set all non-CORE yet non-BLOCK cells as fixed cells
cell->fixed(true);
}
if (co->placementOrient() % 2 == 1) {

View File

@ -20,6 +20,7 @@ PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
torch::Tensor,
torch::Tensor,
torch::Tensor,
torch::Tensor,
float,
float,
float,
@ -34,6 +35,7 @@ PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
.def("commit", &dp::DPTorchRawDB::commit)
.def("rollback", &dp::DPTorchRawDB::rollback)
.def("commit_from", &dp::DPTorchRawDB::commit_from)
.def("commit_from_partial", &dp::DPTorchRawDB::commit_from_partial)
.def("get_curr_cposx", &dp::DPTorchRawDB::get_curr_cposx, py::return_value_policy::move)
.def("get_curr_cposy", &dp::DPTorchRawDB::get_curr_cposy, py::return_value_policy::move)
.def("get_curr_lposx", &dp::DPTorchRawDB::get_curr_lposx, py::return_value_policy::move)
@ -43,6 +45,7 @@ PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
[](torch::Tensor node_lpos_init_,
torch::Tensor node_size_,
torch::Tensor node_weight_,
torch::Tensor is_macro_,
torch::Tensor pin_rel_lpos_,
torch::Tensor pin_id2node_id_,
torch::Tensor pin_id2net_id_,
@ -66,6 +69,7 @@ PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
return std::make_shared<dp::DPTorchRawDB>(node_lpos_init_,
node_size_,
node_weight_,
is_macro_,
pin_rel_lpos_,
pin_id2node_id_,
pin_id2net_id_,

View File

@ -8,6 +8,7 @@ namespace dp {
DPTorchRawDB::DPTorchRawDB(torch::Tensor node_lpos_init_,
torch::Tensor node_size_,
torch::Tensor node_weight_,
torch::Tensor is_macro_,
torch::Tensor pin_rel_lpos_,
torch::Tensor pin_id2node_id_,
torch::Tensor pin_id2net_id_,
@ -74,6 +75,7 @@ DPTorchRawDB::DPTorchRawDB(torch::Tensor node_lpos_init_,
net_mask = net_mask_;
node_weight = node_weight_;
is_macro = is_macro_;
site_width = site_width_;
row_height = row_height_;
@ -180,6 +182,16 @@ void DPTorchRawDB::commit_from(torch::Tensor x_, torch::Tensor y_) {
.copy_(y_.index({torch::indexing::Slice(0, num_movable_nodes)}));
}
void DPTorchRawDB::commit_from_partial(torch::Tensor x_, torch::Tensor y_) {
// commit external pos to new pos
x.index({torch::indexing::Slice(0, num_movable_nodes)})
.data()
.copy_(x_.index({torch::indexing::Slice(0, num_movable_nodes)}));
y.index({torch::indexing::Slice(0, num_movable_nodes)})
.data()
.copy_(y_.index({torch::indexing::Slice(0, num_movable_nodes)}));
}
torch::Tensor DPTorchRawDB::get_curr_cposx() { return x + node_size_x / 2; }
torch::Tensor DPTorchRawDB::get_curr_cposy() { return y + node_size_y / 2; }
torch::Tensor DPTorchRawDB::get_curr_lposx() { return x; }

View File

@ -10,6 +10,7 @@ public:
DPTorchRawDB(torch::Tensor node_lpos_init_,
torch::Tensor node_size_,
torch::Tensor node_weight_,
torch::Tensor is_macro_,
torch::Tensor pin_rel_lpos_,
torch::Tensor pin_id2node_id_,
torch::Tensor pin_id2net_id_,
@ -35,6 +36,7 @@ public:
void commit();
void rollback();
void commit_from(torch::Tensor x_, torch::Tensor y_);
void commit_from_partial(torch::Tensor x_, torch::Tensor y_);
torch::Tensor get_curr_cposx();
torch::Tensor get_curr_cposy();
torch::Tensor get_curr_lposx();
@ -48,6 +50,7 @@ public:
torch::Tensor pin_rel_lpos;
torch::Tensor node_weight;
torch::Tensor is_macro;
torch::Tensor init_x; // original pos (keep it const except committing)
torch::Tensor init_y; // original pos (keep it const except committing)

View File

@ -55,7 +55,7 @@ void distributeCells2Bins(const LegalizationData& db,
num_legalized_nodes = num_movable_nodes;
}
for (int i = 0; i < num_legalized_nodes; i += 1) {
if (!db.is_dummy_fixed(i)) {
if (!db.is_mov_macro(i)) {
int bin_id_x = (x[i] + node_size_x[i] / 2 - xl) / bin_size_x;
int bin_id_y = (y[i] + node_size_y[i] / 2 - yl) / bin_size_y;
@ -87,7 +87,7 @@ void distributeFixedCells2Bins(const LegalizationData& db,
std::vector<std::vector<int>>& bin_cells) {
// one cell can be assigned to multiple bins
for (int i = 0; i < num_nodes; i += 1) {
if (db.is_dummy_fixed(i) || i >= num_movable_nodes) {
if (db.is_mov_macro(i) || i >= num_movable_nodes) {
int node_id = i;
int bin_id_xl = std::max((int)floorDiv(x[node_id] - xl, bin_size_x, 0), 0);
int bin_id_xh = std::min((int)ceilDiv((x[node_id] + node_size_x[node_id] - xl), bin_size_x, 0), num_bins_x);

View File

@ -71,6 +71,7 @@ public:
node2fence_region_map(at_db.node2fence_region_map.data_ptr<int>()),
net_mask(at_db.net_mask.data_ptr<bool>()),
node_weight(at_db.node_weight.data_ptr<float>()),
is_macro(at_db.is_macro.data_ptr<bool>()),
xl(at_db.xl),
xh(at_db.xh),
yl(at_db.yl),
@ -95,6 +96,9 @@ public:
const float* node_size_x;
const float* node_size_y;
const float* node_weight;
const bool* is_macro;
const float* pin_offset_x;
const float* pin_offset_y;
@ -111,7 +115,6 @@ public:
const int* node2fence_region_map;
const bool* net_mask;
const float* node_weight;
/* chip info */
float xl;
@ -146,9 +149,8 @@ public:
bin_size_x = (xh - xl) / num_bins_x_;
bin_size_y = (yh - yl) / num_bins_y_;
}
inline bool is_dummy_fixed(int node_id) const {
// DUMMY_FIXED_NUM_ROWS == 2
return (node_id < num_movable_nodes && node_size_y[node_id] > (row_height * 2));
inline bool is_mov_macro(int node_id) const {
return (node_id < num_movable_nodes && is_macro[node_id]);
}
inline float align2row(float y, float height) const {

View File

@ -24,7 +24,7 @@ bool check_macro_legality(LegalizationData& db, const std::vector<int>& macros,
float yh1 = yl1 + height1;
float xh2 = xl2 + width2;
float yh2 = yl2 + height2;
if (std::min(xh1, xh2) > std::max(xl1, xl2) && std::min(yh1, yh2) > std::max(yl1, yl2)) {
if (std::min(xh1, xh2) - std::max(xl1, xl2) > 1e-3 && std::min(yh1, yh2) - std::max(yl1, yl2) > 1e-3) {
logger.error(
"macro %d (%g, %g, %g, %g) var %d overlaps with macro %d "
"(%g, %g, %g, %g) var %d, fixed: %d",
@ -273,7 +273,7 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
// collect macros
std::vector<int> macros;
for (int i = 0; i < db.num_movable_nodes; ++i) {
if (db.is_dummy_fixed(i)) {
if (db.is_mov_macro(i)) {
// in some extreme case, some macros with 0 area should be ignored
float area = db.node_size_x[i] * db.node_size_y[i];
if (area > 0) {
@ -281,7 +281,7 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
}
}
}
logger.info("Macro legalization: regard %lu cells as dummy fixed (movable macros)", macros.size());
logger.info("Macro legalization: regard %lu cells as movable macros", macros.size());
// in case there is no movable macros
if (macros.empty()) {
@ -320,7 +320,19 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
}
};
// first round rough legalization with Hannan grid for clusters
// 1) LP legalization, check displacement and legality
auto displace = compute_displace(db, macros);
logger.info("Macro displacement total %g, max %g, weighted total %g, max %g",
displace.total_displace,
displace.max_displace,
displace.total_weighted_displace,
displace.max_weighted_displace);
bool legal = check_macro_legality(db, macros, true);
update_best(legal, displace);
// 2) rough legalization with Hannan grid for clusters
if (!legal) {
logger.warning("LP not legal, try roughLegalize.");
bool small_clusters_flag = true;
bool blocked_macros_flag = false;
roughLegalize(db, macros, fixed_macros, small_clusters_flag, blocked_macros_flag);
@ -330,10 +342,13 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
displace.max_displace,
displace.total_weighted_displace,
displace.max_weighted_displace);
bool legal = check_macro_legality(db, macros, true);
legal = check_macro_legality(db, macros, true);
update_best(legal, displace);
}
// try Hannan grid legalization if still not legal
// 3) try Hannan grid legalization if still not legal
if (!legal) {
logger.warning("Not legal, try hannanLegalize.");
legal = hannanLegalize(db, macros, fixed_macros, 10);
auto displace = compute_displace(db, macros);
logger.info("Macro displacement total %g, max %g, weighted total %g, max %g",
@ -361,6 +376,7 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
}
}
if (legal) {
logger.info("Align macros to site and rows");
// align the lower left corner to row and site
for (unsigned int i = 0, ie = macros.size(); i < ie; ++i) {
@ -370,6 +386,12 @@ bool macroLegalization(DPTorchRawDB& at_db, int num_bins_x, int num_bins_y) {
}
legal = check_macro_legality(db, macros, false);
if (!legal) {
logger.error("Macro legalization failed after aligning to site and row");
}
} else {
logger.error("Macro legalization failed");
}
return legal;
}

View File

@ -8,27 +8,6 @@ torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos,
torch::Tensor net_mask,
torch::Tensor hpwl_scale);
std::vector<torch::Tensor> wa_wirelength_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
float gamma,
bool deterministic);
std::vector<torch::Tensor> wa_wirelength_hpwl_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
float gamma,
bool deterministic);
std::vector<torch::Tensor> wa_wirelength_masked_scale_hpwl_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
@ -66,68 +45,6 @@ torch::Tensor masked_scale_hpwl_sum(torch::Tensor node_pos,
node_pos, pin_id2node_id, pin_rel_cpos, hyperedge_list, hyperedge_list_end, net_mask, hpwl_scale);
}
std::vector<torch::Tensor> wa_wirelength(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
float gamma,
bool deterministic) {
CHECK_INPUT(node_pos);
CHECK_INPUT(pin_id2node_id);
CHECK_INPUT(pin_rel_cpos);
CHECK_INPUT(node2pin_list);
CHECK_INPUT(node2pin_list_end);
CHECK_INPUT(hyperedge_list);
CHECK_INPUT(hyperedge_list_end);
CHECK_INPUT(net_mask);
return wa_wirelength_cuda(node_pos,
pin_id2node_id,
pin_rel_cpos,
node2pin_list,
node2pin_list_end,
hyperedge_list,
hyperedge_list_end,
net_mask,
gamma,
deterministic);
}
std::vector<torch::Tensor> wa_wirelength_hpwl(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
float gamma,
bool deterministic) {
CHECK_INPUT(node_pos);
CHECK_INPUT(pin_id2node_id);
CHECK_INPUT(pin_rel_cpos);
CHECK_INPUT(node2pin_list);
CHECK_INPUT(node2pin_list_end);
CHECK_INPUT(hyperedge_list);
CHECK_INPUT(hyperedge_list_end);
CHECK_INPUT(net_mask);
return wa_wirelength_hpwl_cuda(node_pos,
pin_id2node_id,
pin_rel_cpos,
node2pin_list,
node2pin_list_end,
hyperedge_list,
hyperedge_list_end,
net_mask,
gamma,
deterministic);
}
std::vector<torch::Tensor> wa_wirelength_masked_scale_hpwl(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
@ -164,8 +81,6 @@ std::vector<torch::Tensor> wa_wirelength_masked_scale_hpwl(torch::Tensor node_po
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
m.def("masked_scale_hpwl_sum", &masked_scale_hpwl_sum, "calculate the sum of scaled HPWL");
m.def("merged_forward_backward", &wa_wirelength, "calculate WA wirelength and pin grad");
m.def("merged_forward_backward_with_hpwl", &wa_wirelength_hpwl, "calculate WA wirelength, pin grad and hpwl");
m.def("merged_forward_backward_with_masked_scale_hpwl",
&wa_wirelength_masked_scale_hpwl,
"calculate WA wirelength, pin grad and the scaled hpwl");

View File

@ -74,134 +74,6 @@ __global__ void masked_scale_hpwl_cuda_kernel(
}
}
__global__ void wa_wirelength_kernel(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_pos,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list_end,
const torch::PackedTensorAccessor32<bool, 1, torch::RestrictPtrTraits> net_mask,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> partial_wa_wl,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_grad,
int num_nets,
float inv_gamma) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int i = index >> 1; // net index
if (i < num_nets && net_mask[i]) {
const int c = index & 1; // channel index
int64_t start_idx = 0;
if (i != 0) {
start_idx = hyperedge_list_end[i - 1];
}
int64_t end_idx = hyperedge_list_end[i];
if (end_idx != start_idx) {
int64_t pin_id = hyperedge_list[start_idx];
float x_min = pin_pos[pin_id][c];
float x_max = pin_pos[pin_id][c];
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
float xx = pin_pos[hyperedge_list[idx]][c];
x_min = min(xx, x_min);
x_max = max(xx, x_max);
}
float xexp_x_sum = 0;
float xexp_nx_sum = 0;
float exp_x_sum = 0;
float exp_nx_sum = 0;
for (int64_t idx = start_idx; idx < end_idx; idx++) {
float xx = pin_pos[hyperedge_list[idx]][c];
float exp_x = exp((xx - x_max) * inv_gamma);
float exp_nx = exp((x_min - xx) * inv_gamma);
xexp_x_sum += xx * exp_x;
xexp_nx_sum += xx * exp_nx;
exp_x_sum += exp_x;
exp_nx_sum += exp_nx;
}
float wl = xexp_x_sum / exp_x_sum - xexp_nx_sum / exp_nx_sum;
partial_wa_wl[i][c] = wl;
float b_x = inv_gamma / (exp_x_sum);
float a_x = (1.0 - b_x * xexp_x_sum) / exp_x_sum;
float b_nx = -inv_gamma / (exp_nx_sum);
float a_nx = (1.0 - b_nx * xexp_nx_sum) / exp_nx_sum;
for (int64_t idx = start_idx; idx < end_idx; idx++) {
float xx = pin_pos[hyperedge_list[idx]][c];
float exp_x = exp((xx - x_max) * inv_gamma);
float exp_nx = exp((x_min - xx) * inv_gamma);
pin_grad[hyperedge_list[idx]][c] = (a_x + b_x * xx) * exp_x - (a_nx + b_nx * xx) * exp_nx;
}
}
}
}
__global__ void wa_wirelength_hpwl_kernel(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_pos,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list_end,
const torch::PackedTensorAccessor32<bool, 1, torch::RestrictPtrTraits> net_mask,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> partial_wa_wl,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> partial_hpwl,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_grad,
int num_nets,
float inv_gamma) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int i = index >> 1; // net index
if (i < num_nets && net_mask[i]) {
const int c = index & 1; // channel index
int64_t start_idx = 0;
if (i != 0) {
start_idx = hyperedge_list_end[i - 1];
}
int64_t end_idx = hyperedge_list_end[i];
if (end_idx != start_idx) {
int64_t pin_id = hyperedge_list[start_idx];
float x_min = pin_pos[pin_id][c];
float x_max = pin_pos[pin_id][c];
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
float xx = pin_pos[hyperedge_list[idx]][c];
x_min = min(xx, x_min);
x_max = max(xx, x_max);
}
partial_hpwl[i][c] = abs(x_max - x_min);
float xexp_x_sum = 0;
float xexp_nx_sum = 0;
float exp_x_sum = 0;
float exp_nx_sum = 0;
for (int64_t idx = start_idx; idx < end_idx; idx++) {
float xx = pin_pos[hyperedge_list[idx]][c];
float exp_x = exp((xx - x_max) * inv_gamma);
float exp_nx = exp((x_min - xx) * inv_gamma);
xexp_x_sum += xx * exp_x;
xexp_nx_sum += xx * exp_nx;
exp_x_sum += exp_x;
exp_nx_sum += exp_nx;
}
float wl = xexp_x_sum / exp_x_sum - xexp_nx_sum / exp_nx_sum;
partial_wa_wl[i][c] = wl;
float b_x = inv_gamma / (exp_x_sum);
float a_x = (1.0 - b_x * xexp_x_sum) / exp_x_sum;
float b_nx = -inv_gamma / (exp_nx_sum);
float a_nx = (1.0 - b_nx * xexp_nx_sum) / exp_nx_sum;
for (int64_t idx = start_idx; idx < end_idx; idx++) {
float xx = pin_pos[hyperedge_list[idx]][c];
float exp_x = exp((xx - x_max) * inv_gamma);
float exp_nx = exp((x_min - xx) * inv_gamma);
pin_grad[hyperedge_list[idx]][c] = (a_x + b_x * xx) * exp_x - (a_nx + b_nx * xx) * exp_nx;
}
}
}
}
__global__ void wa_wirelength_masked_scale_hpwl_kernel(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_pos,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list,
@ -296,57 +168,6 @@ void calc_node_grad_cuda(torch::Tensor node_grad,
}
}
std::vector<torch::Tensor> wa_wirelength_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
float gamma,
bool deterministic) {
cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
const auto num_nodes = node_pos.size(0);
const auto num_pins = pin_id2node_id.size(0);
const auto num_nets = hyperedge_list_end.size(0);
const auto num_channels = 2; // x, y
auto pin_pos = pin_rel_cpos.clone(); // pin
auto partial_wa_wl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
auto pin_grad = torch::zeros({num_pins, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
const int threads = 128;
const int blocks = (num_pins * 2 + threads - 1) / threads;
node_pos_to_pin_pos_cuda_kernel<<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
pin_id2node_id.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
num_pins);
const int threads2 = 128;
const int blocks2 = (num_nets * 2 + threads2 - 1) / threads2;
float inv_gamma = 1 / gamma;
wa_wirelength_kernel<<<blocks2, threads2, 0, stream>>>(
pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
hyperedge_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
hyperedge_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
net_mask.packed_accessor32<bool, 1, torch::RestrictPtrTraits>(),
partial_wa_wl.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
pin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
num_nets,
inv_gamma);
auto node_grad = torch::zeros({num_nodes, num_channels}, torch::dtype(pin_grad.dtype()).device(pin_grad.device()));
calc_node_grad_cuda(
node_grad, pin_id2node_id, pin_grad, node2pin_list, node2pin_list_end, num_nodes, deterministic);
return {partial_wa_wl, node_grad};
}
torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
@ -391,60 +212,6 @@ torch::Tensor masked_scale_hpwl_sum_cuda(torch::Tensor node_pos,
return total_hpwl;
}
std::vector<torch::Tensor> wa_wirelength_hpwl_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
float gamma,
bool deterministic) {
cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
const auto num_nodes = node_pos.size(0);
const auto num_pins = pin_id2node_id.size(0);
const auto num_nets = hyperedge_list_end.size(0);
const auto num_channels = 2; // x, y
auto pin_pos = pin_rel_cpos.clone(); // pin
auto partial_wa_wl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
auto partial_hpwl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
auto pin_grad = torch::zeros({num_pins, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
const int threads = 128;
const int blocks = (num_pins * 2 + threads - 1) / threads;
node_pos_to_pin_pos_cuda_kernel<<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
pin_id2node_id.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
num_pins);
const int threads2 = 128;
const int blocks2 = (num_nets * 2 + threads2 - 1) / threads2;
float inv_gamma = 1 / gamma;
wa_wirelength_hpwl_kernel<<<blocks2, threads2, 0, stream>>>(
pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
hyperedge_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
hyperedge_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
net_mask.packed_accessor32<bool, 1, torch::RestrictPtrTraits>(),
partial_wa_wl.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
partial_hpwl.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
pin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
num_nets,
inv_gamma);
auto node_grad = torch::zeros({num_nodes, num_channels}, torch::dtype(pin_grad.dtype()).device(pin_grad.device()));
calc_node_grad_cuda(
node_grad, pin_id2node_id, pin_grad, node2pin_list, node2pin_list_end, num_nodes, deterministic);
return {partial_wa_wl, node_grad, partial_hpwl};
}
std::vector<torch::Tensor> wa_wirelength_masked_scale_hpwl_cuda(torch::Tensor node_pos,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,

View File

@ -7,6 +7,18 @@ tar xvzf ispd2005.tar.gz
rm -rf ispd2005.tar.gz
mv ispd2005/ raw/
echo "=== Downloading ispd2006"
wget --no-check-certificate "https://mycuhk-my.sharepoint.com/:u:/g/personal/1155136644_link_cuhk_edu_hk/EYBauANXekFAn1nlKRtec8YBecrfXNmocajWhqNfKWhRvA?e=wL1Z5z&download=1" -O ispd2006.tar.gz
tar xvzf ispd2006.tar.gz
rm -rf ispd2006.tar.gz
mv ispd2006/ raw/
echo "=== Downloading mms"
wget --no-check-certificate "https://mycuhk-my.sharepoint.com/:u:/g/personal/1155136644_link_cuhk_edu_hk/EQpwvzotaWBGlIm9zpOfIL4B_DRFb5jNOQ5mKHLd7yrByw?e=ovrb3e&download=1" -O mms.tar.gz
tar xvzf mms.tar.gz
rm -rf mms.tar.gz
mv mms/ raw/
echo "=== Downloading ispd2015 ==="
wget --no-check-certificate "https://mycuhk-my.sharepoint.com/:u:/g/personal/1155136644_link_cuhk_edu_hk/Ea4YjKNvi-9CnekS41Pw-GgBEhIRNnp6AhMDU9_xElLjNA?e=YSUMhQ&download=1" -O ispd2015.tar.gz
tar xvzf ispd2015.tar.gz

View File

@ -22,21 +22,19 @@ def get_option():
parser.add_argument('--wa_coeff', type=float, default=4.0, help='wa coeff')
parser.add_argument('--num_bin_x', type=int, default=512, help='#binX for density function')
parser.add_argument('--num_bin_y', type=int, default=512, help='#binY for density function')
parser.add_argument('--threshold', type=float, default=4.0, help='normalized node area threshold for using naive mode')
parser.add_argument('--density_weight', type=float, default=8e-5, help='the weight of density loss')
parser.add_argument('--density_weight_coef', type=float, default=1.05, help='the ratio of density_weight')
parser.add_argument('--use_init_density_weight', type=str2bool, default=True, help='enable dynamic initialization of density_weight')
parser.add_argument('--target_density', type=float, default=1.0, help='placement target density')
parser.add_argument('--use_filler', type=str2bool, default=True, help='placement filler')
parser.add_argument('--noise_ratio', type=float, default=0.025, help='noise ratio for initialization')
parser.add_argument('--ignore_net_degree', type=int, default=100, help='threshold of net degree to ignore in wirelength calculation')
parser.add_argument('--scale_design', type=str2bool, default=False, help='normalize die area')
parser.add_argument('--use_eplace_nesterov', type=str2bool, default=True, help='enable eplace nesterov optimizer')
parser.add_argument('--clamp_node', type=str2bool, default=True, help='enable eplace node clamp trick')
parser.add_argument('--use_precond', type=str2bool, default=True, help='apply precond')
parser.add_argument('--stop_overflow', type=float, default=0.07, help='stop overflow in scheduler')
parser.add_argument('--enable_skip_update', type=str2bool, default=True, help='enable skip update')
parser.add_argument("--loss_type", type=str, default="direct", help="loss type")
parser.add_argument('--enable_sample_force', type=str2bool, default=True, help='enable sample force')
parser.add_argument("--mixed_size", type=str2bool, default=False, help="enable mixed size placement")
# global routing params
parser.add_argument('--use_cell_inflate', type=str2bool, default=False, help='use cell inflation')

View File

@ -5,6 +5,6 @@ from .evaluator import *
from .initializer import *
from .nesterov_optimizer import NesterovOptimizer
from .param_scheduler import ParamScheduler
from .detail_placement import detail_placement_main
from .detail_placement import detail_placement_main, macro_legalization_main
from .run_placement_nesterov import run_placement_main_nesterov
from .run_placement import run_placement_main

View File

@ -1,16 +1,6 @@
import torch
from .param_scheduler import ParamScheduler
from .core import merged_wl_loss_grad, WAWirelengthLoss, WAWirelengthLossAndHPWL
def calc_loss(wl_loss, density_loss, ps, args):
if args.loss_type == "weighted_sum":
loss = (wl_loss + ps.density_weight * density_loss) / (1 + ps.density_weight)
elif args.loss_type == "direct":
loss = wl_loss + ps.density_weight * density_loss
else:
raise NotImplementedError("Loss type not defined")
return loss
from .core import merged_wl_loss_grad
def apply_precond(mov_node_pos: torch.Tensor, ps: ParamScheduler, args):
@ -38,6 +28,8 @@ def calc_obj_and_grad(
mov_node_pos = constraint_fn(mov_node_pos)
conn_node_pos = mov_node_pos[mov_lhs:mov_rhs, ...]
conn_node_pos = torch.cat([conn_node_pos, conn_fix_node_pos], dim=0)
assert merged_forward_backward
if merged_forward_backward:
if mov_node_pos.grad is not None:
mov_node_pos.grad.zero_()
@ -81,62 +73,11 @@ def calc_obj_and_grad(
)
mov_node_pos.grad += node_grad_by_density * ps.density_weight
if ps.zero_macro_grad:
mov_node_pos.grad[mov_lhs:mov_rhs].masked_fill_(data.is_mov_macro[mov_lhs:mov_rhs].unsqueeze(1), 0)
grad = apply_precond(mov_node_pos, ps, args)
loss = wl_loss + ps.density_weight * density_loss
else:
if mov_node_pos.grad is not None:
mov_node_pos.grad.zero_()
else:
mov_node_pos.grad = torch.zeros_like(mov_node_pos).detach()
wl_loss = WAWirelengthLoss.apply(
conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos,
data.node2pin_list, data.node2pin_list_end,
data.hyperedge_list, data.hyperedge_list_end, data.net_mask,
ps.wa_coeff, args.deterministic
)
density_loss, _ = density_map_layer(
mov_node_pos, mov_node_size, init_density_map, calc_overflow=False
)
loss = calc_loss(wl_loss, density_loss, ps, args)
loss.backward()
grad = apply_precond(mov_node_pos, ps, args)
return loss, grad
def calc_grad(
optimizer: torch.optim.Optimizer, mov_node_pos: torch.Tensor, wl_loss, density_loss
):
optimizer.zero_grad(set_to_none=False)
wl_loss.backward(retain_graph=True)
wl_grad = mov_node_pos.grad.detach().clone()
optimizer.zero_grad(set_to_none=False)
density_loss.backward(retain_graph=True)
density_grad = mov_node_pos.grad.detach().clone()
optimizer.zero_grad(set_to_none=False)
return wl_grad, density_grad
def fast_optimization(
mov_node_pos, trunc_node_pos_fn, mov_lhs, mov_rhs, conn_fix_node_pos,
density_map_layer, mov_node_size, init_density_map, ps, data, args
):
mov_node_pos = trunc_node_pos_fn(mov_node_pos)
conn_node_pos = mov_node_pos[mov_lhs:mov_rhs, ...]
conn_node_pos = torch.cat(
[conn_node_pos, conn_fix_node_pos], dim=0
)
wl_loss, hpwl = WAWirelengthLossAndHPWL.apply(
conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos,
data.node2pin_list, data.node2pin_list_end,
data.hyperedge_list, data.hyperedge_list_end, data.net_mask,
ps.wa_coeff, data.hpwl_scale, args.deterministic
)
density_loss, overflow = density_map_layer(
mov_node_pos, mov_node_size, init_density_map
)
loss = calc_loss(wl_loss, density_loss, ps, args)
loss.backward()
apply_precond(mov_node_pos, ps, args)
# calculate objective (hpwl, overflow)
return hpwl.detach(), overflow.detach(), mov_node_pos

View File

@ -1,4 +1,4 @@
from .flute import Flute, get_flute_wl
from .electronic_density_layer import ElectronicDensityLayer
from .wa_wirelength_hpwl import WAWirelengthLossAndHPWL, WAWirelengthLoss, masked_scale_hpwl, merged_wl_loss_grad
from .wa_wirelength_hpwl import masked_scale_hpwl, merged_wl_loss_grad
from .route_force import get_route_force, run_gr_and_fft, run_gr_and_fft_main, route_inflation, route_inflation_roll_back

View File

@ -262,35 +262,6 @@ class ElectronicDensityLayer(torch.nn.Module):
return node_weight
def get_density_map_naive(
self,
node_pos,
node_size,
init_density_map=None,
):
node_weight = node_size.new_ones(node_pos.shape[0])
if init_density_map is None:
node_pos.new_zeros(self.num_bin_x, self.num_bin_y)
aux_mat = init_density_map.clone()
num_nodes = node_pos.shape[0]
density_map = density_map_cuda.forward_naive(
node_pos,
node_size,
node_weight,
self.unit_len,
aux_mat,
self.num_bin_x,
self.num_bin_y,
num_nodes,
-1.0,
-1.0,
1e-4,
False,
self.deterministic,
)
return density_map
def direct_calc_overflow(
self,
node_pos,

View File

@ -0,0 +1,947 @@
# We follow the work [1] and [2] to implement the macro legalizer.
# [1] Cong, Jason, and Min Xie. "A robust mixed-size legalization and detailed placement algorithm." IEEE TCAD 2008.
# [2] Moffitt, M. D., Ng, A. N., Markov, I. L., & Pollack, M. E. "Constraint-driven floorplan repair." ACM TODAES 2008.
# NOTE: Known Issue: Adding graph edge constraint in LP is very slow when num_macros is large.
# Because pulp lib use Python OrderDict to store all constraints, when #Constraints
# is large, the performance is bad.
# Re-write the LP in C++ may achieve some speed up.
# TODO: 1) macro spreading for routability optimization
# 2) graph pruning for speed up
import numpy as np
import numba as nb
import logging
import pulp as pl
import igraph as ig
pulp_logger = logging.getLogger('pulp')
pulp_logger.setLevel(logging.INFO)
use_numba_parallel = False
@nb.jit(cache=True, parallel=use_numba_parallel)
def check_macro_legality(macro_pos, macro_size, macro_fixed, die_info, check_all=True):
num_macros = macro_pos.shape[0]
# overlap = np.zeros((num_macros, num_macros), dtype=np.bool8)
legal = True
for i in nb.prange(num_macros):
lx_i = macro_pos[i][0] - macro_size[i][0] / 2
ly_i = macro_pos[i][1] - macro_size[i][1] / 2
hx_i = macro_pos[i][0] + macro_size[i][0] / 2
hy_i = macro_pos[i][1] + macro_size[i][1] / 2
for j in nb.prange(num_macros):
if i >= j:
continue
lx_j = macro_pos[j][0] - macro_size[j][0] / 2
ly_j = macro_pos[j][1] - macro_size[j][1] / 2
hx_j = macro_pos[j][0] + macro_size[j][0] / 2
hy_j = macro_pos[j][1] + macro_size[j][1] / 2
if min(hx_i, hx_j) - max(lx_i, lx_j) > 1e-3 and min(hy_i, hy_j) - max(ly_i, ly_j) > 1e-3:
# overlap[i][j] = True
# overlap[j][i] = True
legal = False
if not check_all and not legal:
return legal
print("Macro", i, "and Macro", j, "Overlap.")
return legal
@nb.jit(nopython=True, nogil=True, cache=True, parallel=use_numba_parallel)
def constraint_graph_construction(
macro_pos, macro_size, macro_fixed, die_info, prune=True
):
num_macros = macro_pos.shape[0]
macro_lpos = macro_pos - macro_size / 2
edge_type = np.zeros((num_macros, num_macros), dtype=np.int8) # 0: x, 1: y, -1: None
edge_dist_x = np.zeros((num_macros, num_macros), dtype=np.float32)
edge_dist_y = np.zeros((num_macros, num_macros), dtype=np.float32)
edge_type[:, :] = -1
for i in nb.prange(num_macros):
for j in nb.prange(num_macros):
if i >= j:
continue
# Detect x/y order
if macro_pos[i][0] <= macro_pos[j][0]:
# x_i -> x_j
x_order = 0
else:
# x_j -> x_i
x_order = 1
if macro_pos[i][1] <= macro_pos[j][1]:
# y_i -> y_j
y_order = 0
else:
# y_j -> y_i
y_order = 1
# Calculate displacement
lx_i = macro_pos[i][0] - macro_size[i][0] / 2
lx_j = macro_pos[j][0] - macro_size[j][0] / 2
hx_i = macro_pos[i][0] + macro_size[i][0] / 2
hx_j = macro_pos[j][0] + macro_size[j][0] / 2
if x_order == 0:
dist_x = lx_j - hx_i
else:
dist_x = lx_i - hx_j
ly_i = macro_pos[i][1] - macro_size[i][1] / 2
ly_j = macro_pos[j][1] - macro_size[j][1] / 2
hy_i = macro_pos[i][1] + macro_size[i][1] / 2
hy_j = macro_pos[j][1] + macro_size[j][1] / 2
if y_order == 0:
dist_y = ly_j - hy_i
else:
dist_y = ly_i - hy_j
# dist martix is undirected and symmetric
edge_dist_x[i][j] = dist_x
edge_dist_x[j][i] = dist_x
edge_dist_y[i][j] = dist_y
edge_dist_y[j][i] = dist_y
# Determine the edge type (horizontal or vertical)
if dist_x >= 0 and dist_y >= 0:
# non-overlap
edge_type[i][j] = 0 if dist_x >= dist_y else 1
elif dist_x >= 0 and dist_y < 0:
# y projection overlap
edge_type[i][j] = 0
elif dist_x < 0 and dist_y >= 0:
# x projection overlap
edge_type[i][j] = 1
elif dist_x < 0 and dist_y < 0:
# overlap
edge_type[i][j] = 0 if dist_x >= dist_y else 1
# Prune edges between objects without x/y projection overlap
if prune:
if edge_type[i][j] == 0 and not (ly_i <= hy_j and ly_j <= hy_i):
edge_type[i][j] = -1
if edge_type[i][j] == 1 and not (lx_i <= hx_j and lx_j <= hx_i):
edge_type[i][j] = -1
# Make sure edge orders
if edge_type[i][j] == 0 and x_order == 1:
edge_type[i][j] = -1
edge_type[j][i] = 0
elif edge_type[i][j] == 1 and y_order == 1:
edge_type[i][j] = -1
edge_type[j][i] = 1
return edge_type, edge_dist_x, edge_dist_y
@nb.jit(nopython=True, nogil=True, cache=True, parallel=use_numba_parallel)
def initialize_xy_adj_weight(
edge_type, adj_matrix, weight_matrix, macro_size, num_macros, num_nodes, s_id, t_id
):
adj_matrix[s_id, :num_macros, :] = 1
adj_matrix[:num_macros, t_id, :] = 1
for i in nb.prange(num_nodes):
for j in nb.prange(num_nodes):
if i == j:
continue
if i < num_macros and j < num_macros:
if edge_type[i][j] == 0:
# horizontal
adj_matrix[i][j][0] = 1
elif edge_type[i][j] == 1:
adj_matrix[i][j][1] = 1
if i >= num_macros:
weight_matrix[i][j][0] = np.divide(macro_size[j][0], 2)
weight_matrix[i][j][1] = np.divide(macro_size[j][1], 2)
elif j >= num_macros:
weight_matrix[i][j][0] = np.divide(macro_size[i][0], 2)
weight_matrix[i][j][1] = np.divide(macro_size[i][1], 2)
else:
weight_matrix[i][j][0] = np.divide(np.add(macro_size[i][0], macro_size[j][0]), 2)
weight_matrix[i][j][1] = np.divide(np.add(macro_size[i][1], macro_size[j][1]), 2)
@nb.jit(nopython=True, nogil=True, cache=True)
def compute_L_value(
macro_pos, macro_fixed, axis, topo_order_out, L, affected_L, adj_matrix, weight_matrix, die_ll, s_id, num_macros
):
for i in topo_order_out:
if affected_L[i, axis] == 0:
continue
if i < num_macros:
if macro_fixed[i]:
L[i, axis] = macro_pos[i, axis]
continue
if i == s_id:
L[i, axis] = die_ll[axis]
else:
is_preds = adj_matrix[:, i, axis]
for j, is_pred in enumerate(is_preds):
if is_pred == 0 or i == j:
continue
# j -> i
L[i, axis] = max(L[j, axis] + weight_matrix[j][i][axis], L[i, axis])
@nb.jit(nopython=True, nogil=True, cache=True)
def compute_R_value(
macro_pos, macro_fixed, axis, topo_order_in, R, affected_R, adj_matrix, weight_matrix, die_ur, t_id, num_macros
):
for i in topo_order_in:
if affected_R[i, axis] == 0:
continue
if i < num_macros:
if macro_fixed[i]:
R[i, axis] = macro_pos[i, axis]
continue
if i == t_id:
R[i, axis] = die_ur[axis]
else:
is_succs = adj_matrix[i, :, axis]
for j, is_succ in enumerate(is_succs):
if is_succ == 0 or i == j:
continue
# i -> j
R[i, axis] = min(R[j, axis] - weight_matrix[i][j][axis], R[i, axis])
def propagate_L_R(
g, macro_pos, macro_fixed, affected_L, affected_R, adj_matrix, weight_matrix,
L, R, die_ll, die_ur, s_id, t_id, num_macros, edges_pair=None
):
topo_order_out_X = nb.typed.List(g[0].topological_sorting(mode="out"))
topo_order_in_X = nb.typed.List(g[0].topological_sorting(mode="in"))
topo_order_out_Y = nb.typed.List(g[1].topological_sorting(mode="out"))
topo_order_in_Y = nb.typed.List(g[1].topological_sorting(mode="in"))
if edges_pair:
axis_del, (u_del, v_del), axis_add, (u_add, v_add) = edges_pair
affected_L[:, :] = False
affected_R[:, :] = False
# In the old graph, u may not be topologically <= v
bfs_order, _, _ = g[axis_del].bfs(u_del, mode='out')
affected_L[np.array(bfs_order), axis_del] = True
bfs_order, _, _ = g[axis_del].bfs(v_del, mode='out')
affected_L[np.array(bfs_order), axis_del] = True
bfs_order, _, _ = g[axis_del].bfs(u_del, mode='in')
affected_R[np.array(bfs_order), axis_del] = True
bfs_order, _, _ = g[axis_del].bfs(v_del, mode='in')
affected_R[np.array(bfs_order), axis_del] = True
# In the new graph, u -> v, so u should be topologically <= v
bfs_order, _, _ = g[axis_add].bfs(u_add, mode='out')
affected_L[np.array(bfs_order), axis_add] = True
bfs_order, _, _ = g[axis_add].bfs(v_add, mode='in')
affected_R[np.array(bfs_order), axis_add] = True
L[affected_L] = -np.inf
R[affected_R] = np.inf
compute_L_value(macro_pos, macro_fixed, 0, topo_order_out_X, L, affected_L,
adj_matrix, weight_matrix, die_ll, s_id, num_macros)
compute_R_value(macro_pos, macro_fixed, 0, topo_order_in_X, R, affected_R,
adj_matrix, weight_matrix, die_ur, t_id, num_macros)
compute_L_value(macro_pos, macro_fixed, 1, topo_order_out_Y, L, affected_L,
adj_matrix, weight_matrix, die_ll, s_id, num_macros)
compute_R_value(macro_pos, macro_fixed, 1, topo_order_in_Y, R, affected_R,
adj_matrix, weight_matrix, die_ur, t_id, num_macros)
@nb.jit(nopython=True, nogil=True, cache=True, parallel=use_numba_parallel)
def compute_edge_slack(
adj_matrix, weight_matrix, L, R, num_nodes, slack_matrix_e,
):
for i in nb.prange(num_nodes):
for j in nb.prange(num_nodes):
if adj_matrix[i][j][0] == 1:
slack_matrix_e[i][j][0] = R[j][0] - L[i][0] - weight_matrix[i][j][0]
if adj_matrix[i][j][1] == 1:
slack_matrix_e[i][j][1] = R[j][1] - L[i][1] - weight_matrix[i][j][1]
def slack_info(
adj_matrix, weight_matrix, L, R, num_macros, num_nodes, slack_v, slack_matrix_e,
update_edge_slack=True,
):
# Node Slack
slack_v[:num_macros, :] = R[:num_macros, :] - L[:num_macros, :]
x_nslack = np.minimum(slack_v[:, 0], 0)
y_nslack = np.minimum(slack_v[:, 1], 0)
x_tns, x_wns, nonzero_x = np.sum(x_nslack), np.min(x_nslack), np.sum(x_nslack < 0)
y_tns, y_wns, nonzero_y = np.sum(y_nslack), np.min(y_nslack), np.sum(y_nslack < 0)
info_v = (x_tns, x_wns, nonzero_x, y_tns, y_wns, nonzero_y)
if update_edge_slack:
slack_matrix_e[:,:,:] = 0
compute_edge_slack(adj_matrix, weight_matrix, L, R, num_nodes, slack_matrix_e)
x_nslack = np.minimum(slack_matrix_e[:, :, 0], 0)
y_nslack = np.minimum(slack_matrix_e[:, :, 1], 0)
x_tns, x_wns, nonzero_x = np.sum(x_nslack), np.min(x_nslack), np.sum(x_nslack < 0)
y_tns, y_wns, nonzero_y = np.sum(y_nslack), np.min(y_nslack), np.sum(y_nslack < 0)
info_e = (x_tns, x_wns, nonzero_x, y_tns, y_wns, nonzero_y)
return info_v, info_e
return info_v, None
@nb.jit(nopython=True, nogil=True, cache=True)
def mark_edge_to_move(adj_matrix, weight_matrix, slack_v, i, L, R, s_id, t_id, macro_pos):
edges_pair = []
x_slack = slack_v[i, 0]
y_slack = slack_v[i, 1]
if (x_slack >= 0 and y_slack >= 0) or (x_slack < 0 and y_slack < 0):
return edges_pair
if x_slack >= 0 and y_slack < 0:
# need to handle in g[1]
axis = 1
else:
# need to handle in g[0]
axis = 0
o_axis = 0 if axis == 1 else 1
is_preds = adj_matrix[:, i, axis]
for j, is_pred in enumerate(is_preds):
# j -> i
if is_pred == 0 or i == j:
continue
if j == s_id or j == t_id:
continue
if L[i][axis] == L[j][axis] + weight_matrix[j][i][axis]:
edge_ij = False
edge_ji = False
if L[i][o_axis] + weight_matrix[i][j][o_axis] <= R[j][o_axis]:
edge_ij = True
is_succs_j = adj_matrix[j, :, o_axis]
for k, is_succ in enumerate(is_succs_j):
# i -> j -> k
if is_succ == 0 or k == j:
continue
if L[i][o_axis] + weight_matrix[i][j][o_axis] + weight_matrix[j][k][o_axis] > R[k][o_axis]:
edge_ij = False
break
if L[j][o_axis] + weight_matrix[j][k][o_axis] > R[k][o_axis]:
edge_ij = False
break
if L[j][o_axis] + weight_matrix[j][i][o_axis] <= R[i][o_axis]:
edge_ji = True
is_succs_i = adj_matrix[i, :, o_axis]
for k, is_succ in enumerate(is_succs_i):
# j -> i -> k
if is_succ == 0 or k == i:
continue
if L[j][o_axis] + weight_matrix[j][i][o_axis] + weight_matrix[i][k][o_axis] > R[k][o_axis]:
edge_ij = False
break
if L[i][o_axis] + weight_matrix[i][k][o_axis] > R[k][o_axis]:
edge_ij = False
break
if edge_ij and edge_ji:
if macro_pos[i][o_axis] <= macro_pos[j][o_axis]:
# i -> j
edges_pair.append((axis, (j, i), o_axis, (i, j)))
else:
# j -> i
edges_pair.append((axis, (j, i), o_axis, (j, i)))
elif edge_ij:
edges_pair.append((axis, (j, i), o_axis, (i, j)))
elif edge_ji:
edges_pair.append((axis, (j, i), o_axis, (j, i)))
else:
edges_pair.clear()
return edges_pair
def longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur, logger, naive=False, prune=False):
edge_type, _, _ = constraint_graph_construction(macro_pos, macro_size, macro_fixed, die_info, prune=prune)
if naive:
return edge_type, None, None
logger.debug("Finish Graph construction.")
die_lx, die_hx, die_ly, die_hy = die_info
num_macros = macro_pos.shape[0]
logger.debug("Longest Path Refinement #Macros: %d #FixedMacros: %d" % (num_macros, macro_fixed.sum()))
num_nodes = num_macros + 2 # including source and target
s_id = num_macros
t_id = num_macros + 1
dtype = macro_size.dtype
adj_matrix = np.zeros((num_nodes, num_nodes, 2), dtype=np.int8)
weight_matrix = np.full((num_nodes, num_nodes, 2), -1, dtype=dtype)
initialize_xy_adj_weight(edge_type, adj_matrix, weight_matrix, macro_size,
num_macros, num_nodes, s_id, t_id)
edge_type[:, :] = -1 # not used anymore
g_x = ig.Graph.Adjacency(adj_matrix[:,:,0], mode= "directed")
g_y = ig.Graph.Adjacency(adj_matrix[:,:,1], mode= "directed")
g = [g_x, g_y]
# Use negative weight to find longest path
if not g[0].is_dag():
logger.warning("g_x is not a DAG.")
if not g[1].is_dag():
logger.warning("g_y is not a DAG.")
# Calcuate x_L, x_R, y_L and y_R
L = np.full((num_nodes, 2), -np.inf, dtype=dtype)
R = np.full((num_nodes, 2), np.inf, dtype=dtype)
affected_L = np.ones((num_nodes, 2), dtype=np.bool8)
affected_R = np.ones((num_nodes, 2), dtype=np.bool8)
# Propagate all nodes' L and R
propagate_L_R(g, macro_pos, macro_fixed, affected_L, affected_R, adj_matrix, weight_matrix,
L, R, die_ll, die_ur, s_id, t_id, num_macros)
# Calculate Node and Edge Slacks
slack_v = np.zeros((num_macros, 2), dtype=dtype)
slack_matrix_e = np.zeros((num_nodes, num_nodes, 2), dtype=dtype)
info_v, info_e = slack_info(
adj_matrix, weight_matrix, L, R, num_macros, num_nodes, slack_v, slack_matrix_e)
logger.debug("Before longest path refinement:")
logger.debug(" Node X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Node Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_v)
logger.debug(" Edge X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Edge Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_e)
# plot_negative_slack_macro(macro_pos, macro_size, macro_fixed, die_info, slack_v)
macro_area = np.prod(macro_size, axis=1)
num_trials = 0
num_movement = 0
while np.minimum(slack_v, 0).sum() < 0:
if num_trials == 5:
logger.error("Cannot fix longest path after %d trials." % num_trials)
break
macro_order = list(range(num_macros))
sum_slack = slack_v.sum(axis=1)
macro_order.sort(key=lambda x: (macro_area[x], -sum_slack[x], macro_pos[x][0], macro_pos[x][1], x))
logger.debug("--- Trial %d ---" % num_trials)
num_trials += 1
for i in macro_order:
# Mark edges to move
edges_pair = mark_edge_to_move(adj_matrix, weight_matrix, slack_v, i, L, R, s_id, t_id, macro_pos)
# Move selected edges
for axis_del, (u_del, v_del), axis_add, (u_add, v_add) in edges_pair:
g[axis_del].delete_edges([(u_del, v_del)])
g[axis_add].add_edges([(u_add, v_add)])
assert adj_matrix[u_del, v_del, axis_del] == 1
assert adj_matrix[u_add, v_add, axis_add] == 0
adj_matrix[u_del, v_del, axis_del] = 0
adj_matrix[u_add, v_add, axis_add] = 1
logger.debug("Move %d: G_%d (%d, %d) -> G_%d (%d, %d)." % (
num_movement, axis_del, u_del, v_del, axis_add, u_add, v_add))
edges_pair = (axis_del, (u_del, v_del), axis_add, (u_add, v_add))
# edges_pair = None # Debug only
propagate_L_R(g, macro_pos, macro_fixed, affected_L, affected_R, adj_matrix, weight_matrix,
L, R, die_ll, die_ur, s_id, t_id, num_macros, edges_pair=edges_pair)
info_v, _ = slack_info(adj_matrix, weight_matrix, L, R, num_macros, num_nodes, slack_v, None,
update_edge_slack=False)
logger.debug(" Updated Node X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Node Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_v)
num_movement += 1
# plot_negative_slack_macro(macro_pos, macro_size, macro_fixed, die_info, slack_v)
info_v, info_e = slack_info(
adj_matrix, weight_matrix, L, R, num_macros, num_nodes, slack_v, slack_matrix_e)
logger.debug("Finish longest path refinement:")
logger.debug(" Node X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Node Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_v)
logger.debug(" Edge X: TNS/WNS/#NegSlks %.2f/%.2f/%d | Edge Y: TNS/WNS/#NegSlks %.2f/%.2f/%d" % info_e)
assert (adj_matrix[:num_macros,:num_macros] == 1).all(axis=2).sum() == 0
edge_type[:, :] = -1
edge_type[adj_matrix[:num_macros,:num_macros,0] == 1] = 0
edge_type[adj_matrix[:num_macros,:num_macros,1] == 1] = 1
return edge_type, g_x, g_y
def basic_variable(macro_pos, macro_size, macro_fixed, die_info):
num_macros = macro_pos.shape[0]
die_lx, die_hx, die_ly, die_hy = die_info
x_set, y_set, dx_set, dy_set = [], [], [], []
for i in range(num_macros):
if not macro_fixed[i]:
x_set.append(pl.LpVariable(
"x_%d" % i, macro_size[i][0] / 2, die_hx - macro_size[i][0] / 2
))
y_set.append(pl.LpVariable(
"y_%d" % i, macro_size[i][1] / 2, die_hy - macro_size[i][1] / 2
))
dx_set.append(pl.LpVariable("d_x_%d" % i, 0, die_hx))
dy_set.append(pl.LpVariable("d_y_%d" % i, 0, die_hy))
else:
x_value = macro_pos[i][0]
y_value = macro_pos[i][1]
x_set.append(pl.LpVariable("x_%d" % i, x_value, x_value))
y_set.append(pl.LpVariable("y_%d" % i, y_value, y_value))
dx_set.append(pl.LpVariable("d_x_%d" % i, 0, 0))
dy_set.append(pl.LpVariable("d_y_%d" % i, 0, 0))
for i in range(num_macros):
x_value = macro_pos[i][0]
y_value = macro_pos[i][1]
x_set[i].setInitialValue(x_value)
y_set[i].setInitialValue(y_value)
dx_set[i].setInitialValue(0)
dy_set[i].setInitialValue(0)
return x_set, y_set, dx_set, dy_set
def macro_legalization_xy(args, logger, macro_pos, macro_size, macro_fixed, macro_weights, die_info, die_ll, die_ur,
num_items=None, lpbackend=None, naive=False, prune=False, edge_type=None):
logger.info("Start macro_legalization_xy...")
if num_items is not None:
macro_pos_cache = np.copy(macro_pos)
macro_pos = macro_pos[:num_items]
macro_size = macro_size[:num_items]
macro_fixed = macro_fixed[:num_items]
num_macros = macro_pos.shape[0]
if edge_type is None:
edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur,
logger, naive=naive, prune=prune)
logger.debug("Finish Graph X construction.")
prob_x = pl.LpProblem("MacroLegalizationX", pl.LpMinimize)
prob_y = pl.LpProblem("MacroLegalizationY", pl.LpMinimize)
x_set, y_set, dx_set, dy_set = basic_variable(macro_pos, macro_size, macro_fixed, die_info)
prob_x += (
pl.lpSum([macro_weights[i, 0] * dx_set[i] for i in range(num_macros)]),
"Sum_of_Total_displacement",
)
prob_y += (
pl.lpSum([macro_weights[i, 1] * dy_set[i] for i in range(num_macros)]),
"Sum_of_Total_displacement",
)
for i in range(num_macros):
ori_x = macro_pos[i][0]
ori_y = macro_pos[i][1]
prob_x += (
x_set[i] - ori_x <= dx_set[i],
"Displacement_x_%d" % i,
)
prob_x += (
x_set[i] - ori_x >= -dx_set[i],
"NegDisplacement_x_%d" % i,
)
prob_y += (
y_set[i] - ori_y <= dy_set[i],
"Displacement_y_%d" % i,
)
prob_y += (
y_set[i] - ori_y >= -dy_set[i],
"NegDisplacement_y_%d" % i,
)
# 2) Graph Version X:
dist_x = (macro_size[:,0] + macro_size[:,0].reshape(-1,1)) / 2
for i in range(num_macros):
for j in range(num_macros):
if edge_type[i][j] == -1 or i == j:
continue
if macro_fixed[i] and macro_fixed[j]:
continue
if edge_type[i][j] == 0:
prob_x += (
x_set[i] + dist_x[i][j] <= x_set[j],
"Horizontal_%d_%d" % (i, j),
)
# Write LP for debugging
# prob_x.writeLP("MacroLegalization.lp")
# Solve by pl
logger.debug("Start solving...")
prob_x.solve(lpbackend)
pl_status_x = pl.LpStatus[prob_x.status]
solve_success = pl_status_x == "Optimal"
displacement_x = pl.value(prob_x.objective)
# Commit Solver Results
macro_pos_new = np.copy(macro_pos)
for v in prob_x.variables():
if str(v.name).startswith("x_"):
macro_pos_new[int(str(v.name).split("_")[1])][0] = float(v.varValue)
macro_pos = macro_pos_new
# 3) Graph Version Y: Need to update graph edges since placement is changed
edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur,
logger, naive=naive, prune=prune)
logger.debug("Finish Graph Y construction.")
dist_y = (macro_size[:,1] + macro_size[:,1].reshape(-1,1)) / 2
for i in range(num_macros):
for j in range(num_macros):
if edge_type[i][j] == -1 or i == j:
continue
if macro_fixed[i] and macro_fixed[j]:
continue
if edge_type[i][j] == 1:
prob_y += (
y_set[i] + dist_y[i][j] <= y_set[j],
"Vertical_%d_%d" % (i, j),
)
# Write LP for debugging
# prob_y.writeLP("MacroLegalization.lp")
# Solve by pl
logger.debug("Start solving...")
prob_y.solve(lpbackend)
pl_status_y = pl.LpStatus[prob_y.status]
solve_success = (pl_status_y == "Optimal") and solve_success
displacement_y = pl.value(prob_y.objective)
# Commit Solver Results
macro_pos_new = np.copy(macro_pos)
for v in prob_y.variables():
if str(v.name).startswith("y_"):
macro_pos_new[int(str(v.name).split("_")[1])][1] = float(v.varValue)
logger.info(
"X Status: %s, DisplaceX = %.2f | Y Status: %s, DisplaceY = %.2f | Total Displacement = %.2f" % (
pl_status_x, displacement_x, pl_status_y, displacement_y, displacement_x + displacement_y
))
# 4) Iterative Legalization:
if num_items is not None:
macro_pos_cache[:num_items] = macro_pos_new[:num_items]
logger.info("#Macros: %d, #Macros in step: %d" % (macro_pos_cache.shape[0], macro_pos_new.shape[0]))
macro_pos_new = macro_pos_cache
return macro_pos_new, solve_success, displacement_x + displacement_y
def macro_legalization_mix(args, logger, macro_pos, macro_size, macro_fixed, macro_weights, die_info, die_ll, die_ur,
num_items=None, lpbackend=None, naive=False, prune=False, edge_type=None):
logger.info("Start macro_legalization_mix...")
if num_items is not None:
macro_pos_cache = np.copy(macro_pos)
macro_pos = macro_pos[:num_items]
macro_size = macro_size[:num_items]
macro_fixed = macro_fixed[:num_items]
num_macros = macro_pos.shape[0]
if edge_type is None:
edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info,
die_ll, die_ur, logger, naive=naive, prune=prune)
logger.debug("Finish Graph construction.")
prob = pl.LpProblem("MacroLegalization", pl.LpMinimize)
logger.debug("Setup lp variables")
x_set, y_set, dx_set, dy_set = basic_variable(macro_pos, macro_size, macro_fixed, die_info)
logger.debug("Setup lp objectives")
prob += (
pl.lpSum([macro_weights[i, 0] * dx_set[i] + macro_weights[i, 1] * dy_set[i] for i in range(num_macros)]),
"Sum_of_Total_displacement",
)
logger.debug("Setup lp displacement constrains")
for i in range(num_macros):
ori_x = macro_pos[i][0]
ori_y = macro_pos[i][1]
prob += (
x_set[i] - ori_x <= dx_set[i],
"Displacement_x_%d" % i,
)
prob += (
x_set[i] - ori_x >= -dx_set[i],
"NegDisplacement_x_%d" % i,
)
prob += (
y_set[i] - ori_y <= dy_set[i],
"Displacement_y_%d" % i,
)
prob += (
y_set[i] - ori_y >= -dy_set[i],
"NegDisplacement_y_%d" % i,
)
logger.debug("Setup lp edge constraints")
dist_x = (macro_size[:,0] + macro_size[:,0].reshape(-1,1)) / 2
dist_y = (macro_size[:,1] + macro_size[:,1].reshape(-1,1)) / 2
for i in range(num_macros):
for j in range(num_macros):
if edge_type[i][j] == -1 or i == j:
continue
if macro_fixed[i] and macro_fixed[j]:
continue
if edge_type[i][j] == 0:
prob += (
x_set[i] + dist_x[i][j] <= x_set[j],
"Horizontal_%d_%d" % (i, j),
)
elif edge_type[i][j] == 1:
prob += (
y_set[i] + dist_y[i][j] <= y_set[j],
"Vertical_%d_%d" % (i, j),
)
# Write LP for debugging
# prob.writeLP("MacroLegalization.lp")
# Solve by pl
logger.debug("Start solving...")
prob.solve(lpbackend)
solve_success = pl.LpStatus[prob.status] == "Optimal"
displacement = pl.value(prob.objective)
logger.info("Status: %s, Total Displacement of MacroLegalization = %.2f" % (pl.LpStatus[prob.status], displacement))
# Commit Solver Results
macro_pos_new = np.copy(macro_pos)
for v in prob.variables():
if str(v.name).startswith("x_"):
macro_pos_new[int(str(v.name).split("_")[1])][0] = float(v.varValue)
if str(v.name).startswith("y_"):
macro_pos_new[int(str(v.name).split("_")[1])][1] = float(v.varValue)
if num_items is not None:
macro_pos_cache[:num_items] = macro_pos_new[:num_items]
logger.info("#Macros: %d, #Macros in step: %d" % (macro_pos_cache.shape[0], macro_pos_new.shape[0]))
macro_pos_new = macro_pos_cache
return macro_pos_new, solve_success, displacement
def macro_legalization_ilp(args, logger, macro_pos, macro_size, macro_fixed, macro_weights, die_info, die_ll, die_ur,
num_items=None, lpbackend=None, edge_type=None):
logger.info("Start macro_legalization_ilp...")
if num_items is not None:
macro_pos_cache = np.copy(macro_pos)
macro_pos = macro_pos[:num_items]
macro_size = macro_size[:num_items]
macro_fixed = macro_fixed[:num_items]
num_macros = macro_pos.shape[0]
prob = pl.LpProblem("MacroLegalization", pl.LpMinimize)
die_lx, die_hx, die_ly, die_hy = die_info
x_set, y_set, dx_set, dy_set = basic_variable(macro_pos, macro_size, macro_fixed, die_info)
prob += (
pl.lpSum([macro_weights[i, 0] * dx_set[i] + macro_weights[i, 1] * dy_set[i] for i in range(num_macros)]),
"Sum_of_Total_displacement",
)
for i in range(num_macros):
ori_x = macro_pos[i][0]
ori_y = macro_pos[i][1]
prob += (
x_set[i] - ori_x <= dx_set[i],
"Displacement_x_%d" % i,
)
prob += (
x_set[i] - ori_x >= -dx_set[i],
"NegDisplacement_x_%d" % i,
)
prob += (
y_set[i] - ori_y <= dy_set[i],
"Displacement_y_%d" % i,
)
prob += (
y_set[i] - ori_y >= -dy_set[i],
"NegDisplacement_y_%d" % i,
)
# Naive Binary Variables Version:
dist_x = (macro_size[:,0] + macro_size[:,0].reshape(-1,1)) / 2
dist_y = (macro_size[:,1] + macro_size[:,1].reshape(-1,1)) / 2
choices = pl.LpVariable.dicts("Choice", (range(num_macros), range(num_macros), range(2)), 0, 1, cat=pl.const.LpInteger)
for i in range(num_macros):
for j in range(num_macros):
if i >= j:
continue
if macro_fixed[i] and macro_fixed[j]:
continue
prob += (
x_set[i] + dist_x[i][j] <= x_set[j] + die_hx * (choices[i][j][0] + choices[i][j][1]),
"XLhs_%d_%d" % (i, j),
)
prob += (
x_set[i] - dist_x[i][j] >= x_set[j] - die_hx * (1 + choices[i][j][0] - choices[i][j][1]),
"XRhs_%d_%d" % (i, j),
)
prob += (
y_set[i] + dist_y[i][j] <= y_set[j] + die_hy * (1 - choices[i][j][0] + choices[i][j][1]),
"YLhs_%d_%d" % (i, j),
)
prob += (
y_set[i] - dist_y[i][j] >= y_set[j] - die_hy * (2 - choices[i][j][0] - choices[i][j][1]),
"YRhs_%d_%d" % (i, j),
)
# Write LP for debugging
# prob.writeLP("MacroLegalization.lp")
# Solve by pl
logger.debug("Start solving...")
prob.solve(lpbackend)
solve_success = pl.LpStatus[prob.status] == "Optimal"
displacement = pl.value(prob.objective)
logger.info("Status: %s, Total Displacement of MacroLegalization = %.2f" % (pl.LpStatus[prob.status], displacement))
# Commit Solver Results
macro_pos_new = np.copy(macro_pos)
for v in prob.variables():
if str(v.name).startswith("x_"):
macro_pos_new[int(str(v.name).split("_")[1])][0] = float(v.varValue)
if str(v.name).startswith("y_"):
macro_pos_new[int(str(v.name).split("_")[1])][1] = float(v.varValue)
if num_items is not None:
macro_pos_cache[:num_items] = macro_pos_new[:num_items]
logger.info("#Macros: %d, #Macros in step: %d" % (macro_pos_cache.shape[0], macro_pos_new.shape[0]))
macro_pos_new = macro_pos_cache
return macro_pos_new, solve_success, displacement
def macro_rough_align(macro_pos, macro_size, macro_fixed, die_ll, die_ur, inv_scalar):
is_mov = np.logical_not(macro_fixed)
macro_pos_lb = macro_size / 2 + die_ll + 1e-4
macro_pos_ub = die_ur - macro_size / 2 + die_ll - 1e-4
mov_macro_pos_lb = macro_pos_lb[is_mov]
mov_macro_pos_ub = macro_pos_ub[is_mov]
macro_pos[is_mov] = macro_pos[is_mov].clip(min=mov_macro_pos_lb, max=mov_macro_pos_ub)
macro_lpos = macro_pos - macro_size / 2
macro_lpos = np.multiply(macro_lpos, inv_scalar, out=macro_lpos)
macro_lpos = np.round_(macro_lpos, out=macro_lpos)
macro_lpos = np.divide(macro_lpos, inv_scalar, out=macro_lpos)
macro_pos_ = macro_lpos + macro_size / 2
macro_pos[is_mov] = macro_pos_[is_mov]
return macro_pos
def plot_macros(macro_pos, macro_size, macro_fixed, die_info, given_colors=None, img_path=None):
import matplotlib as mpl
import matplotlib.pyplot as plt
from matplotlib.collections import PatchCollection
from matplotlib.patches import Rectangle
base_x = 8
die_lx, die_hx, die_ly, die_hy = die_info
base_y = base_x / die_hx * die_hy
fig, ax = plt.subplots(1, figsize=(base_x, base_y))
die_rects = [Rectangle((die_lx, die_ly), die_hx - die_lx, die_hy - die_ly)]
pc = PatchCollection(die_rects, facecolor='w', alpha=0.5, edgecolor='black')
ax.add_collection(pc)
all_color = plt.get_cmap('Set2').colors
for i in range(macro_pos.shape[0]):
x, y = macro_pos[i]
w, h = macro_size[i]
color_idx = given_colors[i] if given_colors is not None else 0
color = all_color[color_idx]
rect = Rectangle((x - w / 2, y - h / 2), w, h, facecolor=color, alpha=0.5, edgecolor='black')
ax.add_patch(rect)
ax.text(x, y, "%d" % i, fontsize=14)
plt.xlim(-die_hx * 0.02, die_hx * 1.02)
plt.ylim(-die_hy * 0.02, die_hy * 1.02)
ax.axis('off')
ax.get_xaxis().set_visible(False)
ax.get_yaxis().set_visible(False)
plt.tight_layout()
img_path = img_path if img_path is not None else "test.png"
plt.savefig(img_path, bbox_inches='tight')
plt.close()
def plot_negative_slack_macro(macro_pos, macro_size, macro_fixed, die_info, slack_v, img_path=None):
num_macros = slack_v.shape[0]
x_nslack = np.minimum(slack_v[:, 0], 0)
y_nslack = np.minimum(slack_v[:, 1], 0)
print(x_nslack)
print(y_nslack)
unique_values = np.sort(np.unique(x_nslack + y_nslack))[::-1]
given_colors = np.zeros(num_macros, dtype=np.int8)
for color_idx, value in enumerate(unique_values):
given_colors[(x_nslack + y_nslack) == value] = color_idx
plot_macros(macro_pos, macro_size, macro_fixed, die_info, given_colors=given_colors, img_path=img_path)
def macro_legalization_multi(macro_info, args, logger):
macro_pos, macro_size, macro_fixed, macro_weights, die_ll, die_ur, die_info, inv_scalar = macro_info
macro_pos = macro_pos.cpu().numpy()
macro_size = macro_size.cpu().numpy()
macro_fixed = macro_fixed.cpu().numpy()
macro_weights = macro_weights.cpu().numpy()
die_ll = die_ll.cpu().numpy()
die_ur = die_ur.cpu().numpy()
die_info = die_info.cpu().numpy()
inv_scalar = inv_scalar.cpu().numpy()
# plot_macros(macro_pos, macro_size, macro_fixed, die_info, img_path="legalized_before.png")
nb.set_num_threads(args.num_threads)
if check_macro_legality(macro_pos, macro_size, macro_fixed, die_info, check_all=False):
# Macros are legal, skip legalization
return macro_pos, True
macro_pos = macro_rough_align(macro_pos, macro_size, macro_fixed, die_ll, die_ur, inv_scalar)
def macro_lg_handler(
ml_func, method_name, solver, best_result, *func_args, max_times=1, timeLimit=None, **ml_func_kwargs
):
total_displacement = 0
for i in range(max_times):
solver.timeLimit = (i + 1) * 20 if timeLimit is None else timeLimit
logger.info("Use cbc to solve LP. TimeLimit = %ds." % solver.timeLimit)
macro_pos_tmp, solve_success, displacement = ml_func(*func_args, lpbackend=solver, **ml_func_kwargs)
if not check_macro_legality(macro_pos_tmp, macro_size, macro_fixed, die_info):
# update macro_pos to macro_pos_tmp
func_args = (*func_args[:2], macro_pos_tmp, *func_args[3:])
ml_func_kwargs["edge_type"] = None
solve_success = False
total_displacement += displacement
if solve_success:
break
if not solve_success and not best_result[1] and method_name != "ilp":
best_result = (macro_pos_tmp, False, total_displacement, method_name)
if solve_success and (not best_result[1] or total_displacement < best_result[2]):
best_result = (macro_pos_tmp, True, total_displacement, method_name)
return best_result
# best_result: (macro_pos, solve_success, displacement, method_name)
best_result = (None, False, float('inf'), None)
func_args = (args, logger, macro_pos, macro_size, macro_fixed, macro_weights, die_info, die_ll, die_ur)
solver = pl.PULP_CBC_CMD(msg=0, timeLimit=20, threads=args.num_threads)
if macro_pos.shape[0] > 500:
logger.info("Too many macros, try pruned graph version first.")
edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur,
logger, naive=False, prune=True)
best_result = macro_lg_handler(macro_legalization_mix, "mix", solver, best_result, *func_args, max_times=1, edge_type=edge_type, prune=True)
best_result = macro_lg_handler(macro_legalization_xy, "xy", solver, best_result, *func_args, max_times=3, edge_type=edge_type, prune=True)
if not best_result[1]:
edge_type, _, _ = longest_path_refinement(macro_pos, macro_size, macro_fixed, die_info, die_ll, die_ur,
logger, naive=False, prune=False)
best_result = macro_lg_handler(macro_legalization_mix, "mix", solver, best_result, *func_args, max_times=1, edge_type=edge_type)
best_result = macro_lg_handler(macro_legalization_xy, "xy", solver, best_result, *func_args, max_times=3, edge_type=edge_type)
if not best_result[1]:
logger.warning("Both LPs are infeasible. Try ILP version.")
macro_lg_handler(macro_legalization_ilp, "ilp", solver, best_result, *func_args, timeLimit=120)
if not best_result[1]:
logger.error("ILP is infeasible.")
# commit the result no matter it is legal or not
macro_pos, solve_success, displacement, method_name = best_result
# plot_macros(macro_pos, macro_size, macro_fixed, die_info, img_path="legalized_after.png")
if solve_success:
logger.info("Macro Legalization Success. Select macro legalization [%s]. Displacement = %.2f" % (method_name, displacement))
return macro_pos, solve_success
# For debug only
# numba_logger = logging.getLogger('numba')
# numba_logger.setLevel(logging.WARNING)
# import random
# import torch
# import time
# random.seed(0)
# class Args:
# def __init__(self) -> None:
# self.num_threads = 20
# args = Args()
# logger = logging.getLogger(__name__)
# logging.basicConfig(encoding='utf-8', level=logging.DEBUG, format='%(asctime)s %(message)s')
# if __name__ == "__main__":
# macro_info = torch.load("macro_info.pt")
# logger.info("Start...")
# start_time = time.time()
# macro_pos, solve_success = macro_legalization_multi(macro_info, args, logger)
# logger.info("Macro legalization time: %.2f" % (time.time() - start_time))

View File

@ -7,84 +7,6 @@ class HPWLCache:
hpwl_cache = HPWLCache()
class WAWirelengthLossAndHPWL(torch.autograd.Function):
@staticmethod
def forward(
ctx,
node_pos,
pin_id2node_id,
pin_rel_cpos,
node2pin_list,
node2pin_list_end,
hyperedge_list,
hyperedge_list_end,
net_mask,
gamma,
hpwl_scale,
deterministic,
):
(
partial_wa_wl,
node_grad,
partial_hpwl,
) = wa_wirelength_hpwl_cuda.merged_forward_backward_with_hpwl(
node_pos,
pin_id2node_id,
pin_rel_cpos,
node2pin_list,
node2pin_list_end,
hyperedge_list,
hyperedge_list_end,
net_mask,
gamma,
deterministic,
)
sum_hpwl = torch.round(partial_hpwl * hpwl_scale).sum()
ctx.save_for_backward(node_grad)
return torch.sum(partial_wa_wl), sum_hpwl
@staticmethod
def backward(ctx, wa_grad_out, hpwl_grad_out):
node_grad = ctx.saved_tensors[0]
return (node_grad * wa_grad_out,) + (None,) * 10
class WAWirelengthLoss(torch.autograd.Function):
@staticmethod
def forward(
ctx,
node_pos,
pin_id2node_id,
pin_rel_cpos,
node2pin_list,
node2pin_list_end,
hyperedge_list,
hyperedge_list_end,
net_mask,
gamma,
deterministic,
):
partial_wa_wl, node_grad = wa_wirelength_hpwl_cuda.merged_forward_backward(
node_pos,
pin_id2node_id,
pin_rel_cpos,
node2pin_list,
node2pin_list_end,
hyperedge_list,
hyperedge_list_end,
net_mask,
gamma,
deterministic,
)
ctx.save_for_backward(node_grad)
return torch.sum(partial_wa_wl)
@staticmethod
def backward(ctx, wa_grad_out):
node_grad = ctx.saved_tensors[0]
return (node_grad * wa_grad_out,) + (None,) * 9
def merged_wl_loss_grad(
node_pos,

View File

@ -317,6 +317,21 @@ class PlaceData(object):
if hasattr(self, "__num_fillers__"):
return self.__num_fillers__
@property
def num_macros(self):
if hasattr(self, "__num_macros__"):
return self.__num_macros__
@property
def num_movable_macros(self):
if hasattr(self, "__num_movable_macros__"):
return self.__num_movable_macros__
@property
def num_fixed_macros(self):
if hasattr(self, "__num_fixed_macros__"):
return self.__num_fixed_macros__
@property
def num_bin_x(self):
if hasattr(self, "__num_bin_x__"):
@ -540,7 +555,7 @@ class PlaceData(object):
self.pin_rel_cpos /= scalar_at
self.pin_rel_lpos /= scalar_at
self.pin_size /= scalar_at
self.__die_scale__ *= self.site_width
self.__die_scale__ *= scalar_at
return self
def prescale(self):
@ -591,10 +606,28 @@ class PlaceData(object):
self.net_mask = torch.logical_and(
self.net_to_num_pins <= args.ignore_net_degree, self.net_to_num_pins >= 2
) # 0: ignore, 1: consider in wirelength calculation
# obj related
# macros -> all mov nodes has ultra-large areas with >= 3 row height and all fixed nodes
# But nodes with zero width or zero height are not considered as macros
mov_lhs, mov_rhs = self.movable_index
mov_cell_area = torch.prod(self.node_size[mov_lhs:mov_rhs, ...], 1)
self.__total_mov_area_without_filler__ = torch.sum(mov_cell_area).item()
num_movable_nodes = mov_rhs - mov_lhs
self.is_macro: torch.Tensor = self.node_size[:,1] * self.die_scale[1] / self.row_height > 2.01
mov_node_area = torch.prod(self.node_size[mov_lhs:mov_rhs, ...], 1)
mov_node_area_order = torch.argsort(mov_node_area)
macro_area_threshold = 10 * torch.mean(mov_node_area[
mov_node_area_order[:int(num_movable_nodes * 0.999)]
])
self.is_macro.logical_and_((self.node_area > macro_area_threshold).squeeze(1))
self.is_macro[mov_rhs:] = True
self.is_macro.logical_and_((self.node_size * self.die_scale > 1e-4).all(dim=1))
self.is_mov_macro = self.is_macro.clone()
self.is_mov_macro[mov_rhs:] = False
self.__num_macros__ = torch.sum(self.is_macro).item()
self.__num_movable_macros__ = torch.sum(self.is_mov_macro).item()
self.__num_fixed_macros__ = self.__num_macros__ - self.__num_movable_macros__
# obj related
self.__total_mov_area_without_filler__ = torch.sum(mov_node_area).item()
self.__bin_area__ = torch.prod(self.unit_len).item()
return self
@ -630,12 +663,25 @@ class PlaceData(object):
# init_density_map are all normalized to (0.0, 1.0)
fixed_node_area = ori_dmap * self.bin_area
placeable_area = die_area - fixed_node_area
if True:
mov_cell_area = torch.prod(mov_node_size, 1)
num_movable_nodes = mov_rhs - mov_lhs
mov_node_xsize_order = torch.argsort(mov_node_size[:, 0])
mov_node_area = torch.prod(mov_node_size, 1).sum()
mov_macro_area = self.node_area[self.is_mov_macro].sum()
mov_stdcell_area = mov_node_area - mov_macro_area
stdcell_placeable_area = placeable_area - mov_macro_area
stdcell_util = mov_stdcell_area / stdcell_placeable_area
if stdcell_util.item() > args.target_density:
logger.warning("Stdcell util %.2f is larger than target density %.2f. Increase target density to %.2f." % (
stdcell_util, args.target_density, stdcell_util))
args.target_density = stdcell_util
if self.is_mov_macro.sum().item() <= 1:
mov_node_xsize = mov_node_size[:, 0]
else:
mov_node_xsize = mov_node_size[:, 0][
torch.logical_not(self.is_mov_macro[mov_lhs:mov_rhs])
].contiguous()
num_movable_nodes = mov_node_xsize.shape[0]
mov_node_xsize_order = torch.argsort(mov_node_xsize)
filler_size_x = torch.mean(
mov_node_size[:, 0][
mov_node_xsize[
mov_node_xsize_order[
int(num_movable_nodes * 0.05) : int(
num_movable_nodes * 0.95
@ -645,7 +691,7 @@ class PlaceData(object):
)
filler_size_y = self.site_height / self.die_scale[1]
total_filler_area = max(
args.target_density * placeable_area - torch.sum(mov_cell_area),
args.target_density * stdcell_placeable_area - mov_stdcell_area,
0.0,
)
single_filler_size = torch.tensor(
@ -656,16 +702,7 @@ class PlaceData(object):
self.__num_fillers__ = int(
torch.round(total_filler_area / (filler_size_x * filler_size_y))
)
else:
mov_cell_area = torch.prod(mov_node_size, 1)
total_filler_area = max(
args.target_density * placeable_area
- torch.sum(mov_cell_area).item(),
0.0,
)
single_filler_area = torch.mean(mov_cell_area)
single_filler_size = single_filler_area.sqrt().repeat(2)
self.__num_fillers__ = int(total_filler_area / single_filler_area)
if self.num_fillers > 0:
self.filler_size = single_filler_size.repeat(self.num_fillers, 1)
logger.info(
@ -689,18 +726,24 @@ class PlaceData(object):
args.use_filler = False
die_area, placeable_area = die_area.item(), placeable_area.item()
fixed_node_area, mov_cell_area = fixed_node_area.item(), torch.sum(mov_cell_area).item()
fixed_node_area, mov_node_area = fixed_node_area.item(), mov_node_area.item()
mov_macro_area, mov_stdcell_area = mov_macro_area.item(), mov_stdcell_area.item()
total_filler_area = float(total_filler_area)
logger.info(
"DieArea: %.3E FixArea: %.3E (%.1f%%) PlaceableArea: %.3E (%.1f%%) MovArea: %.3E (%.1f%%) FillerArea: %.3E (%.1f%%)"
"DieArea: %.3E FixArea: %.3E (%.1f%%) PlaceableArea: %.3E (%.1f%%) MovArea: %.3E (%.1f%%) FillerArea: %.3E (%.1f%%) "
"MovMacroArea: %.3E (%.1f%%) MovStdCellArea: %.3E (%.1f%%)"
% (
die_area,
fixed_node_area, fixed_node_area / die_area * 100,
placeable_area, placeable_area / die_area * 100,
mov_cell_area, mov_cell_area / die_area * 100,
mov_node_area, mov_node_area / die_area * 100,
total_filler_area, total_filler_area / die_area * 100,
mov_macro_area, mov_macro_area / die_area * 100,
mov_stdcell_area, mov_stdcell_area / die_area * 100,
)
)
if mov_node_area > placeable_area:
logger.warning("MovArea > PlaceableArea. Not enough olaceblae area to place all movable nodes.")
return self
@ -769,6 +812,11 @@ class PlaceData(object):
num_fltiopin,
)
)
content += (
"#Macros = %d, #MovMacros = %d, #FixMacros = %d\n" % (
self.num_macros, self.num_movable_macros, self.num_fixed_macros
)
)
content += "Core Info " + str([i for i in self.die_info.cpu().numpy()]) + "\n"
content += "Site Width = %d, Row Height = %d\n" % (
self.site_width,
@ -783,6 +831,12 @@ class PlaceData(object):
content += "target density = %.2f\n" % (args.target_density)
content += "==================="
logger.info(content)
args.include_macros = True if self.num_movable_macros > 10 else False
if args.include_macros and not args.mixed_size:
logger.warning("Detect many macros. Suggest to turn on mixed_size in cmd args.")
if self.num_movable_macros == 0 and args.mixed_size:
logger.warning("#MovMacros is 0. Turn off mixed size mode.")
args.mixed_size = False
return self
def preprocess(self):
@ -790,8 +844,8 @@ class PlaceData(object):
self.backup_ori_var()
self.preshift()
self.prescale_by_site_width()
if args.scale_design:
self.prescale()
# if args.scale_design:
# self.prescale()
self.pre_compute_var()
self.init_fence_region()
self.logging_statistics()
@ -803,6 +857,21 @@ class PlaceData(object):
self.compute_sorted_node_map()
return self
def get_filler_pos(self):
if self.num_fillers > 0:
if self.enable_fence:
raise NotImplementedError("We haven't yet supported fence region.")
else:
filler_pos = torch.rand(
(self.num_fillers, 2),
dtype=self.node_size.dtype,
device=self.node_size.device,
)
scale = self.die_ur - self.die_ll
shift = self.die_ll
filler_pos = filler_pos * scale + shift
return filler_pos
def get_mov_node_info(self, init_method="randn_center"):
args = self.__args__
mov_lhs, mov_rhs = self.movable_index
@ -813,19 +882,15 @@ class PlaceData(object):
scale = (self.die_ur - self.die_ll) * 0.001
loc = (self.die_ur + self.die_ll) * 0.5
mov_node_pos = torch.randn_like(mov_node_pos) * scale + loc
elif init_method == "randn_center_lxly":
# TODO: Mixed-size placement is very sensitive to the initial location.
# An elegant yet effective initialization may be needed.
scale = (self.die_ur - self.die_ll) * 0.001
loc = (self.die_ur + self.die_ll) * 0.5
mov_node_pos = torch.randn_like(mov_node_pos) * scale + loc + mov_node_size / 2
if self.num_fillers > 0:
if self.enable_fence:
raise NotImplementedError("We haven't yet supported fence region.")
else:
filler_pos = torch.rand(
(self.num_fillers, 2),
dtype=mov_node_size.dtype,
device=mov_node_size.device,
)
scale = self.die_ur - self.die_ll
shift = self.die_ll
filler_pos = filler_pos * scale + shift
filler_pos = self.get_filler_pos()
mov_node_pos = torch.cat([mov_node_pos, filler_pos], dim=0)
mov_node_size = torch.cat([mov_node_size, self.filler_size], dim=0)
@ -844,6 +909,10 @@ class PlaceData(object):
expand_ratio = mov_node_area / clamp_mov_node_area
mov_node_size = clamp_mov_node_size
if args.target_density < 1.0:
expand_ratio[mov_lhs:mov_rhs].masked_fill_(
self.is_mov_macro[mov_lhs:mov_rhs], args.target_density)
return mov_node_pos, mov_node_size, expand_ratio
def write_pl(self, node_pos, gp_prefix):

View File

@ -1,10 +1,10 @@
import torch
from .database import PlaceData
from .evaluator import get_obj_hpwl
from .core.macro_legalization import macro_legalization_multi
from utils.visualization import draw_fig_with_cairo_cpp
from cpp_to_py import gpudp, routedp
import numba as nb
import numpy as np
import os
import time
@ -13,6 +13,7 @@ class PreprocessDatabaseCache:
def __init__(self) -> None:
self.node_size = None
self.node_weight = None
self.is_macro = None
self.pin_id2node_id = None
self.node2pin_list = None
self.node2pin_list_end = None
@ -20,6 +21,7 @@ class PreprocessDatabaseCache:
def reset(self):
self.node_size = None
self.node_weight = None
self.is_macro = None
self.pin_id2node_id = None
self.node2pin_list = None
self.node2pin_list_end = None
@ -76,10 +78,11 @@ def rearrange_dpdb_node_info(node_pos: torch.Tensor, data: PlaceData):
if preprocess_db_cache.node_size is not None:
node_size = preprocess_db_cache.node_size.clone()
node_weight = preprocess_db_cache.node_weight
is_macro = preprocess_db_cache.is_macro
pin_id2node_id = preprocess_db_cache.pin_id2node_id
node2pin_list = preprocess_db_cache.node2pin_list
node2pin_list_end = preprocess_db_cache.node2pin_list_end
return node_lpos, node_size, node_weight, pin_id2node_id, node2pin_list, node2pin_list_end
return node_lpos, node_size, node_weight, is_macro, pin_id2node_id, node2pin_list, node2pin_list_end
pin_id2node_id: torch.Tensor = data.pin_id2node_id.clone().int().cpu().numpy()
@ -100,6 +103,14 @@ def rearrange_dpdb_node_info(node_pos: torch.Tensor, data: PlaceData):
node_weight_ori[blkg_rhs:floatiopin_rhs]
), dim=0)
is_macro = torch.cat((
data.is_macro[:fix_rhs],
data.is_macro[iopin_rhs:blkg_rhs],
data.is_macro[floatiopin_rhs:floatfix_rhs],
data.is_macro[fix_rhs:iopin_rhs],
data.is_macro[blkg_rhs:floatiopin_rhs]
), dim=0)
old_node2pin_list_end: torch.Tensor = data.node2pin_list_end.int()
old_node2pin_list: torch.Tensor = data.node2pin_list.int()
@ -137,18 +148,19 @@ def rearrange_dpdb_node_info(node_pos: torch.Tensor, data: PlaceData):
if preprocess_db_cache.node_size is None:
preprocess_db_cache.node_size = node_size
preprocess_db_cache.node_weight = node_weight
preprocess_db_cache.is_macro = is_macro
preprocess_db_cache.pin_id2node_id = pin_id2node_id
preprocess_db_cache.node2pin_list = node2pin_list
preprocess_db_cache.node2pin_list_end = node2pin_list_end
return node_lpos, node_size, node_weight, pin_id2node_id, node2pin_list, node2pin_list_end
return node_lpos, node_size, node_weight, is_macro, pin_id2node_id, node2pin_list, node2pin_list_end
def setup_detailed_rawdb(
node_pos: torch.Tensor, use_cpu_db_: bool, data: PlaceData, args, logger, after_lg=True
):
curr_site_width = 1.0 # prescale_by_site_width
node_lpos, node_size, node_weight, pin_id2node_id, node2pin_list, node2pin_list_end = rearrange_dpdb_node_info(
node_lpos, node_size, node_weight, is_macro, pin_id2node_id, node2pin_list, node2pin_list_end = rearrange_dpdb_node_info(
node_pos, data
)
if after_lg:
@ -164,19 +176,6 @@ def setup_detailed_rawdb(
mov_lhs, mov_rhs = data.movable_index
conn_mov_lhs, conn_mov_rhs = data.movable_connected_index
if args.scale_design:
# scale back
die_scale = data.die_scale / data.site_width # assume site width == 1 in dp
node_lpos = node_lpos * die_scale
node_size = node_size * die_scale
pin_rel_lpos = data.pin_rel_lpos * die_scale
die_info = (data.die_info.reshape(2, 2).t() * die_scale).t().reshape(-1)
region_boxes = (
(data.region_boxes.reshape(-1, 2, 2).permute(0, 2, 1) * die_scale)
.permute(0, 2, 1)
.reshape(-1, 4)
) # [:, 0] -> lx, [:, 1] -> hx, [:, 2] -> ly, [:, 3] -> hy
else:
pin_rel_lpos = data.pin_rel_lpos
die_info = data.die_info
region_boxes = data.region_boxes
@ -206,6 +205,7 @@ def setup_detailed_rawdb(
node_lpos.cpu(),
node_size.cpu(),
node_weight.cpu(),
is_macro.cpu(),
pin_rel_lpos.cpu(),
pin_id2node_id.cpu(),
data.pin_id2net_id.int().cpu(),
@ -232,6 +232,7 @@ def setup_detailed_rawdb(
node_lpos,
node_size,
node_weight,
is_macro.cpu(),
pin_rel_lpos,
pin_id2node_id,
data.pin_id2net_id.int(),
@ -304,17 +305,73 @@ def commit_to_node_pos(node_pos: torch.Tensor, data:PlaceData, dp_rawdb):
return node_pos
def run_macro_legalization(node_pos, data: PlaceData, lg_rawdb, args, logger):
if data.num_movable_macros == 0 or (data.num_movable_macros == 1 and data.num_fixed_macros == 0):
return True
if data.num_fixed_macros != 0:
logger.warning("Including fixed macros. LP formula may be infeasible.")
# Pre-process: get macro info
mov_lhs, mov_rhs = data.movable_index
macro_size = data.node_size[data.is_macro].contiguous()
macro_pos = node_pos[data.is_macro].detach().contiguous()
macro_fixed = (torch.cat([
torch.zeros(mov_rhs - mov_lhs, dtype=torch.bool, device=node_pos.device),
torch.ones(node_pos.shape[0] - mov_rhs, dtype=torch.bool, device=node_pos.device)
])[data.is_macro]).contiguous()
macro_weights = torch.ones_like(macro_pos)
# macro_id = torch.zeros_like(data.is_macro, dtype=torch.long)
# macro_id[data.is_macro] = torch.cumsum(data.is_macro, 0)[data.is_macro]
inv_scalar = torch.tensor(
[round(1.0 / get_ori_scale_factor(data))], dtype=torch.float32, device=node_pos.device
)
macro_info = (
macro_pos, macro_size, macro_fixed, macro_weights,
data.die_ll, data.die_ur, data.die_info, inv_scalar
)
# Macro legalization
macro_pos, solve_success = macro_legalization_multi(macro_info, args, logger)
node_pos_cache = node_pos.clone()
node_pos_cache[data.is_macro] = torch.from_numpy(macro_pos).to(node_pos.device)
node_pos[mov_lhs:mov_rhs] = node_pos_cache[mov_lhs:mov_rhs]
# Post-process: round to integer and commit to lg_rawdb
node_lpos, _, _, _, _, _, _ = rearrange_dpdb_node_info(node_pos, data)
_, floatmov_rhs, _ = data.node_type_indices[1]
node_lpos[:floatmov_rhs][data.is_macro[:floatmov_rhs]] = node_lpos[:floatmov_rhs][
data.is_macro[:floatmov_rhs]].contiguous().mul_(inv_scalar).round_().div_(inv_scalar)
lg_rawdb.commit_from_partial(node_lpos[:, 0], node_lpos[:, 1])
return solve_success
def macro_legalization_main(node_pos: torch.Tensor, data: PlaceData, args, logger, lg_rawdb=None):
if lg_rawdb is None:
lg_rawdb = setup_detailed_rawdb(node_pos, True, data, args, logger, after_lg=False)
logger.info("Start running Macro Legalization... #Macros: %d, #MovMacros: %d." % (
data.num_macros, data.num_movable_macros
))
ml_time = time.time()
run_macro_legalization(node_pos, data, lg_rawdb, args, logger)
# align to site/row and check
if gpudp.macroLegalization(lg_rawdb, data.num_bin_x, data.num_bin_y):
lg_rawdb.commit()
logger.info("Check Pass in Macro Legalization")
else:
logger.error("Check failed in Macro Legalization.")
# Commit result
commit_to_node_pos(node_pos, data, lg_rawdb)
torch.cuda.synchronize(node_pos.device)
logger.info("***** Finish Macro Legalization, HPWL: %.4E Time: %.4f *****" % (
get_obj_hpwl(node_pos, data, args).item(), time.time() - ml_time
))
def run_lg(node_pos: torch.Tensor, data: PlaceData, args, logger):
# CPU legalization
lg_rawdb = setup_detailed_rawdb(node_pos, True, data, args, logger, after_lg=False)
# run LG
logger.info("Start running Macro Legalization...")
ml_time = time.time()
num_bins_x, num_bins_y = data.num_bin_x, data.num_bin_y
if gpudp.macroLegalization(lg_rawdb, num_bins_x, num_bins_y):
lg_rawdb.commit()
logger.info("Finish Macro Legalization. Time: %.4f" % (time.time() - ml_time))
macro_legalization_main(node_pos, data, args, logger, lg_rawdb)
total_cell_area = torch.sum(torch.prod(data.node_size, 1)).item()
die_area = torch.prod(data.die_ur - data.die_ll).item()
@ -346,8 +403,6 @@ def run_lg(node_pos: torch.Tensor, data: PlaceData, args, logger):
# # Commit result
# commit_to_node_pos(node_pos, data, lg_rawdb)
# torch.cuda.synchronize(node_pos.device)
# if args.scale_design:
# node_pos /= data.die_scale
# info = (-1, 0, data.design_name)
# draw_fig_with_cairo_cpp(node_pos, data.node_size, data, info, args, base_size=4096)
@ -368,9 +423,6 @@ def run_lg(node_pos: torch.Tensor, data: PlaceData, args, logger):
get_obj_hpwl(node_pos, data, args).item(), time.time() - gl_time
))
if args.scale_design:
node_pos /= data.die_scale
del lg_rawdb
return node_pos
@ -449,9 +501,6 @@ def run_dp(node_pos: torch.Tensor, data: PlaceData, args, logger):
dp_handler(gpudp.globalSwap, "Global Swap", num_bins_x // 2, num_bins_y // 2, gs_bs, gs_iter)
dp_handler(gpudp.kReorder, "K-Reorder 2", num_bins_x, num_bins_y, kr_K, kr_iter)
if args.scale_design:
node_pos /= data.die_scale
del dp_rawdb
return node_pos
@ -479,12 +528,6 @@ def run_dp_route_opt(node_pos: torch.Tensor, gpdb, rawdb, ps, data: PlaceData, a
node_lpos[:floatmov_rhs].mul_(inv_scalar).round_().div_(inv_scalar)
node_size = data.node_size.cpu()
if args.scale_design:
# scale back
die_scale = data.die_scale / data.site_width # assume site width == 1 in dp
node_lpos = node_lpos * die_scale
node_size = node_size * die_scale
die_info = (data.die_info.reshape(2, 2).t() * die_scale).t().reshape(-1)
site_width = 1.0
row_height = data.row_height / data.site_width
die_info = data.die_info.cpu()

View File

@ -1,7 +1,7 @@
import torch
from .database import PlaceData
from .core import masked_scale_hpwl
from cpp_to_py import hpwl_cuda
from cpp_to_py import hpwl_cuda, density_map_cuda
def get_hpwl(data, pos):
# CUDA only
@ -23,23 +23,39 @@ def get_obj_hpwl(node_pos, data: PlaceData, args):
hpwl = torch.sum(get_hpwl(data, pin_pos.detach()))
return hpwl
def get_obj_overflow(node_pos, density_map_layer, init_density_map, data: PlaceData, args):
def get_obj_overflow(node_pos, init_density_map, ps, data: PlaceData, args):
mov_lhs, mov_rhs = data.movable_index
density_map = density_map_layer.get_density_map_naive(
node_pos[mov_lhs:mov_rhs], data.node_size[mov_lhs:mov_rhs], init_density_map
node_pos = node_pos[mov_lhs:mov_rhs]
node_size = data.node_size[mov_lhs:mov_rhs]
if ps.zero_macro_grad:
node_pos = node_pos[
torch.logical_not(data.is_mov_macro[mov_lhs:mov_rhs])
].contiguous()
node_size = node_size[
torch.logical_not(data.is_mov_macro[mov_lhs:mov_rhs])
].contiguous()
node_weight = node_size.new_ones(node_pos.shape[0])
if init_density_map is None:
init_density_map = node_pos.new_zeros(data.num_bin_x, data.num_bin_y)
aux_mat = init_density_map.clone()
density_map = density_map_cuda.forward_naive(
node_pos, node_size, node_weight, data.unit_len, aux_mat, data.num_bin_x,
data.num_bin_y, node_pos.shape[0], -1.0, -1.0, 1e-4, False, args.deterministic,
)
with torch.no_grad():
overflow_sum = ((density_map - args.target_density) * data.bin_area).clamp_(min=0.0).sum()
overflow = overflow_sum / data.total_mov_area_without_filler
return overflow
def evaluate_placement(node_pos, density_map_layer, init_density_map, data: PlaceData, args):
def evaluate_placement(node_pos, init_density_map, ps, data: PlaceData, args):
# NOTE: since some nets are masked in global placement, hpwl may
# underestimate, this function return the exact value of hpwl
# Original overflow calculation uses the clamp node size (expand ratio),
# this function uses the exact node size to evaluate the overflow
hpwl = get_obj_hpwl(node_pos, data, args)
overflow = get_obj_overflow(node_pos, density_map_layer, init_density_map, data, args)
overflow = get_obj_overflow(node_pos, init_density_map, ps, data, args)
return hpwl, overflow
def fast_evaluator(

View File

@ -1,24 +1,28 @@
import torch
from .database import PlaceData
from cpp_to_py import density_map_cuda
from .core import WAWirelengthLossAndHPWL
from .calculator import calc_grad
from .core import merged_wl_loss_grad
def get_init_density_map(rawdb, gpdb, data: PlaceData, args, logger):
def get_init_density_map(rawdb, gpdb, data: PlaceData, args, logger, ps=None):
lhs, rhs = data.fixed_index
device = data.node_size.get_device()
dtype = data.node_size.dtype
zeros_density_map = torch.zeros(
(data.num_bin_x, data.num_bin_y), device=device, dtype=dtype,
)
if lhs == rhs:
if lhs == rhs and (ps is None or not ps.zero_macro_grad):
data.init_density_map = zeros_density_map
return zeros_density_map
# get fix nodes which are located inside die
node_pos = data.node_pos[lhs:rhs]
node_size = data.node_size[lhs:rhs]
node_weight = node_size.new_ones(node_size.shape[0])
if ps is not None and ps.zero_macro_grad:
# compute the mov + fixed macro density map
node_pos = torch.cat([data.node_pos[data.is_mov_macro].contiguous(), node_pos])
node_size = torch.cat([data.node_size[data.is_mov_macro].contiguous(), node_size])
node_weight = node_size.new_ones(node_size.shape[0])
init_density_map = density_map_cuda.forward_naive(
node_pos, node_size, node_weight, data.unit_len, zeros_density_map,
data.num_bin_x, data.num_bin_y, node_pos.shape[0], -1.0, -1.0, 1e-4, False,
@ -105,22 +109,27 @@ def init_params(
conn_node_pos = torch.cat(
[conn_node_pos, conn_fix_node_pos], dim=0
)
wl_loss, hpwl = WAWirelengthLossAndHPWL.apply(
_, conn_node_grad = merged_wl_loss_grad(
conn_node_pos, data.pin_id2node_id, data.pin_rel_cpos,
data.node2pin_list, data.node2pin_list_end,
data.hyperedge_list, data.hyperedge_list_end, data.net_mask,
ps.wa_coeff, data.hpwl_scale, args.deterministic
data.hpwl_scale, ps.wa_coeff, args.deterministic
)
density_loss, overflow = density_map_layer(
wl_grad = torch.zeros_like(mov_node_pos).detach()
wl_grad[mov_lhs:mov_rhs] = conn_node_grad[mov_lhs:mov_rhs]
_, _, density_grad = density_map_layer.merged_density_loss_grad(
mov_node_pos, mov_node_size, init_density_map
)
wl_grad, density_grad = calc_grad(
optimizer, mov_node_pos, wl_loss, density_loss
)
if ps.zero_macro_grad:
wl_grad[mov_lhs:mov_rhs].masked_fill_(data.is_mov_macro[mov_lhs:mov_rhs].unsqueeze(1), 0)
density_grad[mov_lhs:mov_rhs].masked_fill_(data.is_mov_macro[mov_lhs:mov_rhs].unsqueeze(1), 0)
(mov_node_pos * 0.0).sum().backward()
optimizer.zero_grad(set_to_none=False)
if not ps.rerun_route or route_fn is None:
init_density_weight = (wl_grad.norm(p=1) / density_grad.norm(p=1)).detach()
# init_density_weight = (wl_grad.norm(p=1) / grad_mat.norm(p=1)).detach()
ps.set_init_param(init_density_weight, data, density_loss)
ps.set_init_param(init_density_weight, data)
else:
_, filler_lhs = data.movable_connected_index
filler_rhs = mov_node_pos.shape[0]

View File

@ -70,6 +70,7 @@ class MetricRecorder:
class ParamScheduler:
def __init__(self, data: PlaceData, args, logger) -> None:
self.__logger__ = logger
self.__args__ = args
self.data = data
self.iter = 0
self.init_iter = 0
@ -97,7 +98,7 @@ class ParamScheduler:
self.best_sol_rollback: torch.Tensor = None
self.best_metric_rollback = {"overflow": float("inf"), "hpwl": float("inf")}
# params
# global place params
self.precond_coef = 1.0
self.precond_weight = None
self.density_weight_start = args.density_weight
@ -113,13 +114,15 @@ class ParamScheduler:
self.life = self.max_life
self.stop_overflow = args.stop_overflow
self.skip_update = False if args.enable_skip_update else None
self.enable_fence = data.enable_fence
self.min_enlarge_density_interval = 1000
self.last_enlarge_density_iter = -self.min_enlarge_density_interval
# skip density force
self.enable_sample_force = True
self.enable_sample_force = args.enable_sample_force
self.force_ratio = 0.0
self.enable_fence = data.enable_fence
# routability parameter
self.enable_route = args.use_route_force or args.use_cell_inflate
self.use_cell_inflate = args.use_cell_inflate
self.use_route_force = args.use_route_force
@ -140,14 +143,21 @@ class ParamScheduler:
self.max_route_opt = 5
self.gr_sol_recorder = []
def set_init_param(self, init_density_weight, data: PlaceData, init_density_loss):
# mixed size parameter
self.enable_mixed_size = args.mixed_size
self.include_macros = args.include_macros
self.zero_macro_grad = False
def set_init_param(self, init_density_weight, data: PlaceData):
# init_density_weight
self.init_iter = self.iter
self.all_init_iters.append(self.init_iter)
self.precond_coef = 1.0
self.mu = 1.0
self.density_weight = copy.deepcopy(self.density_weight_start) * init_density_weight
self.wa_coeff = copy.deepcopy(self.wa_coeff_start)
self.update_precond_weight(data)
self.set_mixsize_init_param()
def set_route_init_param(
self, init_density_weight, init_route_weight, init_congest_weight, data: PlaceData, args
@ -157,12 +167,38 @@ class ParamScheduler:
self.init_iter = self.iter
self.all_init_iters.append(self.init_iter)
self.precond_coef = 1.0
self.mu = 1.0
self.base_route_weight = init_route_weight * args.route_weight
self.base_congest_weight = init_congest_weight * args.congest_weight
self.route_weight = copy.deepcopy(self.density_weight) * self.base_route_weight
self.congest_weight = copy.deepcopy(self.density_weight) * self.base_congest_weight
self.pseudo_weight = args.pseudo_weight # same scale as wirelength weight
self.update_precond_weight(data)
self.set_mixsize_init_param()
def set_mixsize_init_param(self):
args = self.__args__
if self.include_macros:
self.skip_update = None
self.enable_sample_force = False
if self.enable_mixed_size:
if not self.zero_macro_grad:
# simultaneously place macro and std cells
self.include_macros = True
self.stop_overflow = args.stop_overflow * 2.0
self.enable_sample_force = False
self.skip_update = None
self.enable_route = False
self.use_cell_inflate = False
self.use_route_force = False
else:
self.include_macros = False
self.stop_overflow = args.stop_overflow
self.enable_sample_force = args.enable_sample_force
self.skip_update = False if args.enable_skip_update else None
self.enable_route = args.use_route_force or args.use_cell_inflate
self.use_cell_inflate = args.use_cell_inflate
self.use_route_force = args.use_route_force
def reset_best_sol(self):
# best solution
@ -384,6 +420,7 @@ class ParamScheduler:
if (
self.recorder.overflow[ptr] < self.stop_overflow * 5
and self.recorder.overflow[ptr] >= self.stop_overflow
and not self.include_macros
):
if self.check_plateau(self.recorder.overflow, window=50, threshold=0.05):
# kill the program since it has converged

View File

@ -20,9 +20,6 @@ def run_placement_main_nesterov(args, logger):
assert args.use_eplace_nesterov
logger.info("Start place %s/%s" % (args.dataset , args.design_name))
logger.info("Use Nesterov optimizer!")
if args.scale_design:
logger.warning("Eplace's nesterov optimizer cannot support normalized die. Disable scale_design.")
args.scale_design = False
data = data.to(device)
data = data.preprocess()
logger.info(data)
@ -140,6 +137,7 @@ def run_placement_main_nesterov(args, logger):
# exit(0)
terminate_signal = False
route_early_terminate_signal = False
log_info = False
for iteration in range(args.inner_iter):
# optimizer.zero_grad() # zero grad inside obj_and_grad_fn
obj = optimizer.step(obj_and_grad_fn)
@ -148,6 +146,68 @@ def run_placement_main_nesterov(args, logger):
ps.step(hpwl, overflow, mov_node_pos, data)
if ps.need_to_early_stop():
terminate_signal = True
log_info = True
if ps.enable_mixed_size and not ps.zero_macro_grad and terminate_signal:
ps.zero_macro_grad = True
# Find best gp node_pos (including macros and std cells)
best_res = ps.get_best_solution()
if best_res[0] is not None:
best_sol, hpwl, overflow = best_res
# fillers are unused from now, we don't copy there data
mov_node_pos[mov_lhs:mov_rhs].data.copy_(best_sol[mov_lhs:mov_rhs])
node_pos = mov_node_pos[mov_lhs:mov_rhs]
node_pos = torch.cat([node_pos, data.node_pos[mov_rhs:]], dim=0)
# Evaluate the mixed placement solution
hpwl, overflow = evaluate_placement(node_pos, init_density_map, ps, data, args)
hpwl, overflow = hpwl.item(), overflow.item()
if args.draw_placement:
info = ("%d_mixed_gp" % (iteration + 1), hpwl, data.design_name)
draw_fig_with_cairo_cpp(node_pos, data.node_size, data, info, args)
logger.info("After Mixed-GP, best solution eval, exact HPWL: %.4E exact Overflow: %.4f" % (hpwl, overflow))
# Run macro legalization to change node_pos inplace
macro_legalization_main(node_pos, data, args, logger)
if args.draw_placement:
info = ("%d_mixed_gp_ml" % (iteration + 1), hpwl, data.design_name)
draw_fig_with_cairo_cpp(node_pos, data.node_size, data, info, args)
# Write node_pos into database to provide an initial solution for std cell placement
data.node_pos[mov_lhs:mov_rhs].data.copy_(node_pos[mov_lhs:mov_rhs])
# Prepare for std cell placement
init_density_map = get_init_density_map(rawdb, gpdb, data, args, logger, ps=ps)
data.__total_mov_area_without_filler__ = torch.sum(data.node_area[mov_lhs:mov_rhs][torch.logical_not(data.is_mov_macro[mov_lhs:mov_rhs])]).item()
mov_node_pos, mov_node_size, expand_ratio = data.get_mov_node_info(init_method="randn_center")
mov_macros_idx = data.is_mov_macro[mov_lhs:mov_rhs]
mov_node_pos[mov_lhs:mov_rhs][mov_macros_idx] = data.node_pos[mov_lhs:mov_rhs][mov_macros_idx]
mov_node_pos = mov_node_pos.requires_grad_(True)
trunc_node_pos_fn = get_trunc_node_pos_fn(mov_node_size, data)
density_map_layer.expand_ratio = expand_ratio
density_map_layer.sorted_maps = data.sorted_maps
# ignore the density and grad computation of macros by node_weight
density_map_layer.cache_node_weight[mov_lhs:mov_rhs][mov_macros_idx] = -1.0
# update partial function correspondingly
obj_and_grad_fn.keywords["constraint_fn"] = trunc_node_pos_fn
obj_and_grad_fn.keywords["mov_node_size"] = mov_node_size
obj_and_grad_fn.keywords["expand_ratio"] = expand_ratio
obj_and_grad_fn.keywords["init_density_map"] = init_density_map
evaluator_fn.keywords["constraint_fn"] = trunc_node_pos_fn
evaluator_fn.keywords["mov_node_size"] = mov_node_size
evaluator_fn.keywords["init_density_map"] = init_density_map
# reset nesterov optimizer
logger.info("Reset optimizer...")
optimizer = NesterovOptimizer([mov_node_pos], lr=0)
# initialization
init_params(
mov_node_pos, trunc_node_pos_fn, mov_lhs, mov_rhs, conn_fix_node_pos,
density_map_layer, mov_node_size, expand_ratio, init_density_map, optimizer,
ps, data, args, route_fn=calc_route_force
)
# init learnig rate
cur_lr = estimate_initial_learning_rate(obj_and_grad_fn, trunc_node_pos_fn, mov_node_pos, args.lr)
for param_group in optimizer.param_groups:
param_group["lr"] = cur_lr.item()
ps.reset_best_sol()
terminate_signal = False # reset signal
logger.info("Re-run std cell placement with fixed macros.")
if ps.use_cell_inflate and ps.curr_optimizer_cnt < ps.max_route_opt and terminate_signal:
terminate_signal = False # reset signal
@ -185,6 +245,7 @@ def run_placement_main_nesterov(args, logger):
ps.rerun_route = False
if ps.rerun_route:
log_info = True
new_mov_node_size, new_expand_ratio = None, None
if ps.use_cell_inflate:
output = route_inflation(
@ -245,7 +306,8 @@ def run_placement_main_nesterov(args, logger):
)
ps.reset_best_sol()
if iteration % args.log_freq == 0 or iteration == args.inner_iter - 1 or ps.rerun_route or terminate_signal:
if iteration % args.log_freq == 0 or iteration == args.inner_iter - 1 or log_info:
log_info = False
log_str = (
"iter: %d | masked_hpwl: %.2E overflow: %.4f obj: %.4E "
"density_weight: %.4E wa_coeff: %.4E"
@ -290,6 +352,10 @@ def run_placement_main_nesterov(args, logger):
best_sol, hpwl, overflow = best_res
# fillers are unused from now, we don't copy there data
mov_node_pos[mov_lhs:mov_rhs].data.copy_(best_sol[mov_lhs:mov_rhs])
if ps.enable_mixed_size and ps.zero_macro_grad:
# rollback macro_pos to previous legalized results since trunc_node_pos_fn may change them
mov_macros_idx = data.is_mov_macro[mov_lhs:mov_rhs]
mov_node_pos.data[mov_lhs:mov_rhs][mov_macros_idx] = data.node_pos[mov_lhs:mov_rhs][mov_macros_idx]
if ps.enable_route:
route_inflation_roll_back(args, logger, data, mov_node_size)
if not route_early_terminate_signal:
@ -314,9 +380,7 @@ def run_placement_main_nesterov(args, logger):
)
# Eval
hpwl, overflow = evaluate_placement(
node_pos, density_map_layer, init_density_map, data, args
)
hpwl, overflow = evaluate_placement(node_pos, init_density_map, ps, data, args)
hpwl, overflow = hpwl.item(), overflow.item()
info = ("%d_gp" % (iteration + 1), hpwl, data.design_name)
if args.draw_placement:

View File

@ -4,6 +4,8 @@ import os
def find_benchmark(dataset_root, benchmark):
bm_to_root = {
"ispd2005": os.path.join(dataset_root, "ispd2005"),
"ispd2006": os.path.join(dataset_root, "ispd2006"),
"mms": os.path.join(dataset_root, "mms"),
"dac2012": os.path.join(dataset_root, "iccad2012dac2012"),
"ispd2015": os.path.join(dataset_root, "ispd2015"),
"ispd2015_fix": os.path.join(dataset_root, "ispd2015_fix"),
@ -11,6 +13,7 @@ def find_benchmark(dataset_root, benchmark):
"ispd2019_no_fence": os.path.join(dataset_root, "ispd2019_no_fence"),
"iccad2019": os.path.join(dataset_root, "iccad2019"),
"ispd2018": os.path.join(dataset_root, "ispd2018"),
"iccad2015": os.path.join(dataset_root, "iccad2015"),
}
root = bm_to_root[benchmark]
all_designs = [i for i in os.listdir(root) if os.path.isdir(os.path.join(root, i))]
@ -18,8 +21,8 @@ def find_benchmark(dataset_root, benchmark):
def get_single_design_params(dataset_root, benchmark, design_name, placement=None):
if benchmark == "ispd2005":
return single_ispd2005(dataset_root, design_name, placement)
if benchmark in ["ispd2005", "ispd2006", "mms"]:
return single_ispd2005(dataset_root, design_name, benchmark, placement)
elif benchmark == "dac2012":
return single_dac2012(dataset_root, design_name, placement)
elif benchmark == "ispd2015":
@ -34,6 +37,8 @@ def get_single_design_params(dataset_root, benchmark, design_name, placement=Non
return single_iccad2019(dataset_root, design_name, placement)
elif benchmark == "ispd2018":
return single_ispd2018(dataset_root, design_name, placement)
elif benchmark.startswith("iccad2015"):
return single_iccad2015(dataset_root, benchmark, design_name, placement)
else:
raise NotImplementedError("benchmark %s is not found" % benchmark)
@ -47,8 +52,8 @@ def get_multiple_design_params(dataset_root, benchmark):
return params_mul
def single_ispd2005(dataset_root, design_name, placement=None):
benchmark = "ispd2005"
def single_ispd2005(dataset_root, design_name, benchmark, placement=None):
benchmark = benchmark
root, all_designs = find_benchmark(dataset_root, benchmark)
if design_name not in all_designs:
raise ValueError("Design Name %s should in %s" % (design_name, all_designs))
@ -56,9 +61,10 @@ def single_ispd2005(dataset_root, design_name, placement=None):
"benchmark": benchmark,
"bookshelf_variety": "ispd2005",
"aux": "%s/%s/%s.aux" % (root, design_name, design_name),
"pl": "%s/%s/%s.pl" % (root, design_name, design_name) if placement is None else placement,
"design_name": design_name,
}
if placement is not None:
params["pl"] = placement
return params
@ -71,9 +77,10 @@ def single_dac2012(dataset_root, design_name, placement=None):
"benchmark": benchmark,
"bookshelf_variety": "dac2012",
"aux": "%s/%s/%s.aux" % (root, design_name, design_name),
"pl": "%s/%s/%s.pl" % (root, design_name, design_name) if placement is None else placement,
"design_name": design_name,
}
if placement is not None:
params["pl"] = placement
return params
@ -170,6 +177,23 @@ def single_ispd2018(dataset_root, design_name, placement=None):
return params
def single_iccad2015(dataset_root, benchmark, design_name, placement=None):
# configuration
# benchmark = "iccad2015"
root, all_designs = find_benchmark(dataset_root, benchmark)
if design_name not in all_designs:
raise ValueError("Design Name %s should in %s" % (design_name, all_designs))
params = {
"benchmark": benchmark,
"tech_lef": "%s/tech.lef" % (root),
"cell_lef": "%s/%s/%s.lef" % (root, design_name, design_name),
"def": "%s/%s/%s.def" % (root, design_name, design_name) if placement is None else placement,
"verilog": "%s/%s/%s.v" % (root, design_name, design_name),
"design_name": design_name,
}
return params
def get_custom_design_params(args):
params = dict(
[

View File

@ -71,19 +71,16 @@ class IOParser(object):
print("def %s not exists." % params["def"])
return False
if "aux" in params.keys():
if "pl" not in params.keys():
print("pl is not found!")
if not os.path.exists(params["aux"]):
print("aux %s not exists." % params["aux"])
return False
if "pl" in params.keys() and not os.path.exists(params["pl"]):
print("pl %s not exists." % params["pl"])
return False
if "output" in params.keys():
if "pl" != params["output"].split(".")[-1]:
print("output format should be .pl")
return False
if not os.path.exists(params["aux"]):
print("aux %s not exists." % params["aux"])
return False
if not os.path.exists(params["pl"]):
print("pl %s not exists." % params["pl"])
return False
self.params = params
if verbose_log:

View File

@ -16,7 +16,7 @@ class CustomFormatter(logging.Formatter):
# FIXME: An elapsed time gap exists between the C++ Timer and Python Timer,
# but I don't know how to resolve it...
time_format = "[%(relativeCreatedSecond)4d.%(relativeCreatedMSecond)03d] "
debug_msg = " (%(module)s.py Line%(lineno)d) %(msg)s"
debug_msg = " (%(module)s.py L%(lineno)d) %(msg)s"
FORMATS = {
logging.DEBUG: time_format + blue + "DEBUG" + reset + debug_msg,
logging.INFO: time_format + "%(msg)s",
@ -47,6 +47,8 @@ def setup_logger(args, sys_argv) -> logging.Logger:
screen_handler = logging.StreamHandler(stream=sys.stdout)
screen_handler.setFormatter(formatter)
logger = logging.getLogger()
logger.setLevel(logging.INFO)
if args.cpp_log_level == 0 or args.verbose_cpp_log:
logger.setLevel(logging.DEBUG)
logger.addHandler(file_handler)
logger.addHandler(screen_handler)

View File

@ -44,6 +44,30 @@ def setup_design_args(args):
elif args.design_name in ["bigblue3", "bigblue4"]:
args.num_bin_x = args.num_bin_y = 2048
args.target_density = 1.0
elif args.design_name in ["adaptec5"]:
args.target_density = 0.5
args.num_bin_x = args.num_bin_y = 1024
elif args.design_name in ["newblue1"]:
args.target_density = 0.8
args.num_bin_x = args.num_bin_y = 512
elif args.design_name in ["newblue2"]:
args.target_density = 0.9
args.num_bin_x = args.num_bin_y = 1024
elif args.design_name in ["newblue3"]:
args.target_density = 0.8
args.num_bin_x = args.num_bin_y = 2048
elif args.design_name in ["newblue4"]:
args.target_density = 0.5
args.num_bin_x = args.num_bin_y = 1024
elif args.design_name in ["newblue5"]:
args.target_density = 0.5
args.num_bin_x = args.num_bin_y = 1024
elif args.design_name in ["newblue6"]:
args.target_density = 0.8
args.num_bin_x = args.num_bin_y = 2048
elif args.design_name in ["newblue7"]:
args.target_density = 0.8
args.num_bin_x = args.num_bin_y = 2048
elif args.design_name in ["mgc_des_perf_1"]:
args.num_bin_x = args.num_bin_y = 512
args.target_density = 0.91