add timing-driven wirelength model; add flute multi thread; setup iccad2015 dataset; refactor flow

This commit is contained in:
bunchgrape 2025-05-02 15:43:00 +08:00
parent 6aa1b0e253
commit 649f320384
21 changed files with 1007586 additions and 17 deletions

View File

@ -9,3 +9,5 @@ add_subdirectory(hpwl_cuda)
add_subdirectory(io_parser)
add_subdirectory(routedp)
add_subdirectory(wa_wirelength_hpwl_cuda)
add_subdirectory(gputimer)
add_subdirectory(wirelength_timing_cuda)

View File

@ -10,6 +10,8 @@ __all__ = [
"gpugr",
"gpudp",
"routedp",
"gputimer",
"wirelength_timing_cuda"
]
from .cpybin import (
dct_cuda,
@ -22,5 +24,7 @@ from .cpybin import (
gpugr,
gpudp,
routedp,
gputimer,
wirelength_timing_cuda
)

View File

@ -0,0 +1,9 @@
set(TARGET_NAME wirelength_timing_cuda)
add_pytorch_extension(${TARGET_NAME}
${TARGET_NAME}.cpp
${TARGET_NAME}_kernel.cu)
install(TARGETS
${TARGET_NAME}
DESTINATION ${XPLACE_LIB_DIR})

View File

@ -0,0 +1,52 @@
#include <torch/extension.h>
std::vector<torch::Tensor> wa_wirelength_timing_weight_cuda(torch::Tensor node_pos,
torch::Tensor timing_pin_grad,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
torch::Tensor net_weight,
torch::Tensor hpwl_scale,
float gamma,
bool deterministic);
#define CHECK_CUDA(x) TORCH_CHECK(x.device().is_cuda(), #x " must be a CUDA tensor")
#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
#define CHECK_INPUT(x) \
CHECK_CUDA(x); \
CHECK_CONTIGUOUS(x)
std::vector<torch::Tensor> wa_wirelength_timing_weight(torch::Tensor node_pos,
torch::Tensor timing_pin_grad,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
torch::Tensor net_weight,
torch::Tensor hpwl_scale,
float gamma,
bool deterministic) {
CHECK_INPUT(node_pos);
CHECK_INPUT(timing_pin_grad);
CHECK_INPUT(pin_id2node_id);
CHECK_INPUT(pin_rel_cpos);
CHECK_INPUT(node2pin_list);
CHECK_INPUT(node2pin_list_end);
CHECK_INPUT(hyperedge_list);
CHECK_INPUT(hyperedge_list_end);
CHECK_INPUT(net_mask);
CHECK_INPUT(net_weight);
CHECK_INPUT(hpwl_scale);
return wa_wirelength_timing_weight_cuda(
node_pos, timing_pin_grad, pin_id2node_id, pin_rel_cpos, node2pin_list, node2pin_list_end, hyperedge_list, hyperedge_list_end, net_mask, net_weight, hpwl_scale, gamma, deterministic);
}
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) { m.def("merged_wl_loss_grad_timing", &wa_wirelength_timing_weight, "calculate timing-driven WA wirelength pin grad"); }

View File

@ -0,0 +1,210 @@
#include <ATen/cuda/CUDAContext.h>
#include <cuda.h>
#include <cuda_runtime.h>
#include <torch/extension.h>
#include <vector>
__global__ void node_pos_to_pin_pos_cuda_kernel(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> pin_id2node_id,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_pos,
int num_pins) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int i = index >> 1; // pin index
if (i < num_pins) {
const int c = index & 1; // channel index
int64_t node_id = pin_id2node_id[i];
pin_pos[i][c] += node_pos[node_id][c];
}
}
__global__ void calc_node_grad_deterministic_cuda_kernel(
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_grad,
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_grad,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> node2pin_list,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> node2pin_list_end,
int num_nodes) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int i = index >> 1; // node index
if (i < num_nodes) {
const int c = index & 1; // channel index
int64_t start_idx = 0;
if (i != 0) {
start_idx = node2pin_list_end[i - 1];
}
int64_t end_idx = node2pin_list_end[i];
if (end_idx != start_idx) {
node_grad[i][c] += pin_grad[node2pin_list[start_idx]][c];
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
node_grad[i][c] += pin_grad[node2pin_list[idx]][c];
}
}
}
}
__global__ void wa_wirelength_pin_root_timing_kernel(
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_pos,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> timing_pin_weight,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list,
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list_end,
const torch::PackedTensorAccessor32<bool, 1, torch::RestrictPtrTraits> net_mask,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> net_weight,
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> hpwl_scale,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> partial_wa_wl,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> partial_hpwl,
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_grad,
int num_nets,
float inv_gamma) {
const int index = blockIdx.x * blockDim.x + threadIdx.x;
const int i = index >> 1; // net index
if (i < num_nets && net_mask[i]) {
const int c = index & 1; // channel index
int64_t start_idx = 0;
if (i != 0) {
start_idx = hyperedge_list_end[i - 1];
}
int64_t end_idx = hyperedge_list_end[i];
if (end_idx != start_idx) {
int64_t root_id = hyperedge_list[start_idx];
float x_min = pin_pos[root_id][c];
float x_max = pin_pos[root_id][c];
float root_x = pin_pos[root_id][c];
float recenter_exp_r = exp((root_x - x_max) * inv_gamma);
float recenter_exp_nr = exp((x_min - root_x) * inv_gamma);
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
float cur_x = pin_pos[hyperedge_list[idx]][c];
x_min = min(cur_x, x_min);
x_max = max(cur_x, x_max);
}
partial_hpwl[i][c] = round((x_max - x_min) * hpwl_scale[c]);
float sum_x_exp_x = root_x * recenter_exp_r;
float sum_x_exp_nx = root_x * recenter_exp_nr;
float sum_exp_x = recenter_exp_r;
float sum_exp_nx = recenter_exp_nr;
float wl_sum_x_exp_x = sum_x_exp_x;
float wl_sum_x_exp_nx = sum_x_exp_nx;
float wl_sum_exp_x = sum_exp_x;
float wl_sum_exp_nx = sum_exp_nx;
// pin-root gradient
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
int64_t pin_id = hyperedge_list[idx];
float cur_x = pin_pos[pin_id][c];
float recenter_exp_x = exp((cur_x - x_max) * inv_gamma);
float recenter_exp_nx = exp((x_min - cur_x) * inv_gamma);
float sum_x_exp_x = cur_x * recenter_exp_x + root_x * recenter_exp_r;
float sum_x_exp_nx = cur_x * recenter_exp_nx + root_x * recenter_exp_nr;
float sum_exp_x = recenter_exp_x + recenter_exp_r;
float sum_exp_nx = recenter_exp_nx + recenter_exp_nr;
wl_sum_x_exp_x += cur_x * recenter_exp_x;
wl_sum_x_exp_nx += cur_x * recenter_exp_nx;
wl_sum_exp_x += recenter_exp_x;
wl_sum_exp_nx += recenter_exp_nx;
float inv_sum_exp_x = 1 / sum_exp_x;
float inv_sum_exp_nx = 1 / sum_exp_nx;
float s_x = sum_x_exp_x * inv_sum_exp_x;
float ns_nx = sum_x_exp_nx * inv_sum_exp_nx;
partial_wa_wl[i][c] += s_x - ns_nx;
float x_coeff = inv_gamma * inv_sum_exp_x;
float nx_coeff = -inv_gamma * inv_sum_exp_nx;
float grad_const = (1 - inv_gamma * s_x) * inv_sum_exp_x;
float grad_nconst = (1 + inv_gamma * ns_nx) * inv_sum_exp_nx;
float x_grad = (grad_const + x_coeff * cur_x) * recenter_exp_x -
(grad_nconst + nx_coeff * cur_x) * recenter_exp_nx;
float root_grad = (grad_const + x_coeff * root_x) * recenter_exp_r -
(grad_nconst + nx_coeff * root_x) * recenter_exp_nr;
float delta_x = timing_pin_weight[pin_id];
pin_grad[pin_id][c] = x_grad * delta_x;
pin_grad[root_id][c] += root_grad * delta_x;
}
}
}
}
void calc_node_grad_cuda(torch::Tensor node_grad,
torch::Tensor pin_id2node_id,
torch::Tensor pin_grad,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
int num_nodes,
bool deterministic) {
if (deterministic) {
auto stream = at::cuda::getCurrentCUDAStream();
const int threads = 128;
const int blocks = (num_nodes * 2 + threads - 1) / threads;
calc_node_grad_deterministic_cuda_kernel<<<blocks, threads, 0, stream>>>(
node_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
pin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
node2pin_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
node2pin_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
num_nodes);
} else {
const auto pin_id2node_id_view = pin_id2node_id.unsqueeze(1).expand({-1, 2});
node_grad.scatter_add_(0, pin_id2node_id_view, pin_grad);
}
}
std::vector<torch::Tensor> wa_wirelength_timing_weight_cuda(torch::Tensor node_pos,
torch::Tensor timing_pin_weight,
torch::Tensor pin_id2node_id,
torch::Tensor pin_rel_cpos,
torch::Tensor node2pin_list,
torch::Tensor node2pin_list_end,
torch::Tensor hyperedge_list,
torch::Tensor hyperedge_list_end,
torch::Tensor net_mask,
torch::Tensor net_weight,
torch::Tensor hpwl_scale,
float gamma,
bool deterministic) {
cudaSetDevice(node_pos.get_device());
auto stream = at::cuda::getCurrentCUDAStream();
const auto num_nodes = node_pos.size(0);
const auto num_pins = pin_id2node_id.size(0);
const auto num_nets = hyperedge_list_end.size(0);
const auto num_channels = 2; // x, y
auto pin_pos = pin_rel_cpos.clone(); // pin
auto partial_wa_wl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
auto partial_hpwl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
auto pin_grad = torch::zeros({num_pins, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
const int threads = 128;
const int blocks = (num_pins * 2 + threads - 1) / threads;
node_pos_to_pin_pos_cuda_kernel<<<blocks, threads, 0, stream>>>(
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
pin_id2node_id.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
num_pins);
const int threads2 = 128;
const int blocks2 = (num_nets * 2 + threads2 - 1) / threads2;
float inv_gamma = 1 / gamma;
wa_wirelength_pin_root_timing_kernel<<<blocks2, threads2, 0, stream>>>(
pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
timing_pin_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
hyperedge_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
hyperedge_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
net_mask.packed_accessor32<bool, 1, torch::RestrictPtrTraits>(),
net_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
hpwl_scale.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
partial_wa_wl.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
partial_hpwl.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
pin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
num_nets,
inv_gamma);
auto node_grad = torch::zeros({num_nodes, num_channels}, torch::dtype(pin_grad.dtype()).device(pin_grad.device()));
calc_node_grad_cuda(
node_grad, pin_id2node_id, pin_grad, node2pin_list, node2pin_list_end, num_nodes, deterministic);
return {partial_wa_wl, node_grad, partial_hpwl};
}

13
main.py
View File

@ -49,6 +49,18 @@ def get_option():
parser.add_argument('--pseudo_weight', type=float, default=0, help='the weight of pseudo net')
parser.add_argument('--visualize_cgmap', type=str2bool, default=False, help='visualize congestion map')
# timing opt params
parser.add_argument('--timing_opt', type=str2bool, default=True, help='perform timing optimization')
parser.add_argument('--timing_freq', type=int, default=1, help='timing freq')
parser.add_argument('--calibration', type=str2bool, default=True, help='perform timer calibration')
parser.add_argument('--calibration_step', type=float, default=0.1, help='timing calibration step')
parser.add_argument('--timing_start_iter', type=int, default=100, help='start iteration of timing optimization')
parser.add_argument('--timing_init_weight', type=float, default=0.05, help='initial timing wirelength weight')
parser.add_argument('--decay_factor', type=float, default=0.3, help='decay factor of timing weight')
parser.add_argument('--decay_boost', type=float, default=3, help='dynamic decay boost factor')
parser.add_argument('--wire_resistance_per_micron', type=float, default=2.535, help='unit wire resistance, normalized across all layers')
parser.add_argument('--wire_capacitance_per_micron', type=float, default=0.16e-15, help='unit wire capacitance, normalized across all layers')
# detailed placement and evaluation
parser.add_argument('--legalization', type=str2bool, default=True, help='perform lg')
parser.add_argument('--detail_placement', type=str2bool, default=True, help='perform dp')
@ -77,6 +89,7 @@ def get_option():
args = parser.parse_args()
args.exp_id = datetime.datetime.now().strftime('%Y-%m-%d-%H:%M:%S') + args.exp_id
args.exp_id = "{}_{}".format(args.exp_id, args.design_name)
if args.dataset == "ispd2015":
print("We haven't yet support fence region in ispd2015, use ispd2015_fix instead")

View File

@ -1,7 +1,6 @@
import torch
from .param_scheduler import ParamScheduler
from .core import merged_wl_loss_grad
from .core import merged_wl_loss_grad, merged_wl_loss_grad_timing
def apply_precond(mov_node_pos: torch.Tensor, ps: ParamScheduler, args):
if not args.use_precond:
@ -51,6 +50,17 @@ def calc_obj_and_grad(
data.hpwl_scale, ps.wa_coeff, args.deterministic
)
mov_node_pos.grad[mov_lhs:mov_rhs] += conn_node_grad_by_wl[mov_lhs:mov_rhs]
if ps.enable_timing:
wl_loss_timing, conn_node_grad_by_timing = merged_wl_loss_grad_timing(
conn_node_pos, data.gputimer.timing_pin_weight,
data.pin_id2node_id, data.pin_rel_cpos,
data.node2pin_list, data.node2pin_list_end, data.hyperedge_list, data.hyperedge_list_end,
data.net_mask, data.net_weight, data.hpwl_scale, ps.wa_coeff, args.deterministic
)
mov_node_pos.grad[mov_lhs:mov_rhs] += conn_node_grad_by_timing[mov_lhs:mov_rhs]
wl_loss += wl_loss_timing
if ps.enable_sample_force:
if ps.iter > 3 and ps.iter % 20 == 0:
# ps.iter > 3 for warmup

View File

@ -2,3 +2,4 @@ from .flute import Flute, get_flute_wl
from .electronic_density_layer import ElectronicDensityLayer
from .wa_wirelength_hpwl import masked_scale_hpwl, merged_wl_loss_grad
from .route_force import get_route_force, run_gr_and_fft, run_gr_and_fft_main, route_inflation, route_inflation_roll_back
from .timing_opt import GPUTimer, merged_wl_loss_grad_timing

271
src/core/timing_opt.py Normal file
View File

@ -0,0 +1,271 @@
import torch
from cpp_to_py import gputimer, wirelength_timing_cuda
from src.param_scheduler import MetricRecorder
from utils import *
class GPUTimer():
def __init__(self, data, rawdb, gpdb, params, args):
self.metrics = [
"wns",
"tns",
]
self.recorder = MetricRecorder(**{m: [] for m in self.metrics})
self.data = data
self.net_names = data.net_names
self.pin_names = data.pin_names
self.microns = data.microns
self.wire_resistance_per_micron = args.wire_resistance_per_micron
self.wire_capacitance_per_micron = args.wire_capacitance_per_micron
self.node_size = data.node_size.detach().clone()
node_lpos = data.node_pos.detach() - self.node_size / 2
self.pin_rel_lpos = data.pin_rel_lpos.detach() + data.pin_size / 2
die_info = data.die_info
xl, xh, yl, yh = die_info.cpu().numpy()
self.mov_lhs, self.mov_rhs = data.movable_index
fix_lhs, fix_rhs = data.fixed_connected_index
num_movable_nodes = self.mov_rhs - self.mov_lhs
self.fix_conn_node_lpos = node_lpos[fix_lhs:fix_rhs]
self.conn_node_lpos = torch.cat([
node_lpos[self.mov_lhs:self.mov_rhs], self.fix_conn_node_lpos
], dim=0)
scale_factor = 1.0 / data.site_width
self.node_weight = data.node_special_type == 2
self.timing_raw_db = gputimer.create_timing_rawdb(
self.conn_node_lpos,
data.node_size,
self.pin_rel_lpos,
data.pin_id2node_id,
data.pin_id2net_id.int(),
data.node2pin_list,
data.node2pin_list_end,
data.hyperedge_list.int(),
data.hyperedge_list_end.int(),
data.net_mask,
num_movable_nodes,
scale_factor,
self.microns,
self.wire_resistance_per_micron,
self.wire_capacitance_per_micron
)
self.timer = gputimer.create_gputimer(params, rawdb, gpdb, self.timing_raw_db)
## Timing optimization
self.timer.init()
self.timer.levelize()
self.pin_slack = torch.zeros(data.num_pins, dtype=torch.float32, device=data.device)
self.timing_pin_weight = torch.ones(data.num_pins, dtype=torch.float32, device=data.device)
self.history_x = None
self.tns_record = []
self.wns_record = []
self.wns_max = []
self.delay_K_max = []
self.delay_1_max = []
self.w_2 = None
self.w_1 = None
self.a_1 = None
self.decay = args.decay_factor
self.decay_boost = args.decay_boost
self.init_alpha = 1.05
self.beta = 0
self.alpha = torch.ones(data.num_pins, dtype=torch.float32, device=data.device) * self.init_alpha
self.global_weight = 1
self.target_wns = 0
def update_timing(self, node_pos):
node_lpos = (node_pos.detach() - self.node_size / 2).to(self.data.device)
self.conn_node_lpos = torch.cat([
node_lpos[self.mov_lhs:self.mov_rhs], self.fix_conn_node_lpos
], dim=0)
self.timer.update_states()
self.timer.update_rc(node_lpos, False, False, False)
self.timer.update_timing()
def update_timing_eval(self, node_pos):
node_lpos = (node_pos.detach() - self.node_size / 2).to(self.data.device)
self.conn_node_lpos = torch.cat([
node_lpos[self.mov_lhs:self.mov_rhs], self.fix_conn_node_lpos
], dim=0)
self.timer.update_states()
self.timer.update_rc_flute(node_lpos, False)
self.timer.update_timing()
def update_timing_calibrated(self, node_pos, record=False):
node_lpos = (node_pos.detach() - self.node_size / 2).to(self.data.device)
self.conn_node_lpos = torch.cat([
node_lpos[self.mov_lhs:self.mov_rhs], self.fix_conn_node_lpos
], dim=0)
if record:
self.timer.update_states()
self.timer.update_rc_flute(node_lpos, True)
self.timer.update_states()
self.timer.update_rc(node_lpos, True, True, True)
else:
self.timer.update_states()
self.timer.update_rc(node_lpos, False, True, True)
self.timer.update_timing()
def report_timing_slack(self):
time_unit = self.timer.time_unit()
self.timer.update_endpoints()
wns_early, tns_early, wns_late, tns_late = self.timer.report_wns_and_tns()
wns_early = (wns_early.item() * (time_unit * 1e9))
wns_late = (wns_late.item() * (time_unit * 1e9))
tns_early = (tns_early.item() * (time_unit * 1e9))
tns_late = (tns_late.item() * (time_unit * 1e9))
self.push_metric(-wns_late, -tns_late)
return wns_early, tns_early, wns_late, tns_late
def report_pin_slack(self):
self.pin_slack = self.timer.report_pin_slack()
return self.pin_slack
def report_path(self, ep_name=None, el = -1, verbose=False):
if ep_name is not None:
ep_idx = self.pin_names.index(ep_name)
path, at, delay = self.timer.report_path(ep_idx, el, verbose)
else:
path, at, delay = self.timer.report_path(-1, el, verbose)
return path, at, delay
def report_arrival(self, pin_name):
pin_idx = self.pin_names.index(pin_name)
return self.timer.report_pin_at()[pin_idx]
def report_slew(self, pin_name):
pin_idx = self.pin_names.index(pin_name)
return self.timer.report_pin_slew()[pin_idx]
def report_load(self, pin_name):
pin_idx = self.pin_names.index(pin_name)
return self.timer.report_pin_load()[pin_idx]
def report_required(self, pin_name):
pin_idx = self.pin_names.index(pin_name)
return self.timer.report_pin_rat()[pin_idx]
def report_slack(self, pin_name):
pin_idx = self.pin_names.index(pin_name)
return self.timer.report_pin_slack()[pin_idx]
def get_node_critocality(self):
pin_slacks, _ = torch.min((torch.nan_to_num(self.report_pin_slack()) * (1e-9 / self.timer.time_unit())).clamp(max=0), 1)
endpoints_index = self.timer.endpoints_index().long()
endpoints_index = torch.unique(endpoints_index)
ep_id2node_id = self.data.pin_id2node_id[endpoints_index]
ep_slacks = pin_slacks[endpoints_index]
node_slacks = torch.zeros(self.node_weight.size(0), dtype=torch.float32, device=self.data.device)
node_slacks.scatter_add_(0, ep_id2node_id, ep_slacks)
node_critocality = torch.abs(node_slacks) / (torch.abs(node_slacks)).max()
return node_critocality
def step(self, ps, node_pos, data):
slacks, _ = torch.min(torch.nan_to_num(self.report_pin_slack()).clamp(max=0), 1)
delay_k, _ = self.timer.report_criticality_threshold(0.75, False, True)
delay_1, pin_visited = self.timer.report_criticality_threshold(0.99, False, True)
self.tns_record.append(self.recorder.tns[-1])
self.wns_record.append(self.recorder.wns[-1])
self.wns_max.append(slacks.min().item())
self.delay_K_max.append(delay_k.max().item())
self.delay_1_max.append(delay_1.max().item())
window = min(10, len(self.wns_max))
x_wns = torch.tensor(self.wns_max[-window:], dtype=torch.float32)
x_delay_k = torch.tensor(self.delay_K_max[-window:], dtype=torch.float32)
x_delay_1 = torch.tensor(self.delay_1_max[-window:], dtype=torch.float32)
wns_mean = x_wns[-1]
delay_k_mean = x_delay_k[-1]
delay_1_mean = x_delay_1[-1]
pin_weight = slacks.abs() / (np.abs(wns_mean)) * self.beta
pin_weight += (delay_k / delay_k_mean.clamp(min=1)) * self.beta * 2
pin_weight += torch.pow(2, (delay_1 / delay_1_mean.clamp(min=1))) * pin_visited.clamp(max=1)
w_0 = pin_weight
delta_w_0 = None
delta_w_1 = None
if self.w_1 is not None:
delta_w_0 = w_0 - self.w_1
if self.w_2 is not None:
delta_w_1 = self.w_1 - self.w_2
if delta_w_0 is not None and delta_w_1 is not None:
decay = (self.decay * torch.pow(5, delta_w_0.clamp(min=0)) / self.decay_boost).clamp(max=0.5)
else:
decay = 1
self.w_2 = self.w_1
self.w_1 = pin_weight.clone()
if self.history_x is None:
self.history_x = self.timing_pin_weight.clone()
self.timing_pin_weight = pin_weight.clamp(min=ps.timing_wl_weight)
self.timing_pin_weight = (decay * self.timing_pin_weight + (1 - decay) * self.history_x) * self.global_weight
self.timing_pin_weight = self.timing_pin_weight.contiguous().to(node_pos.device)
self.history_x = self.timing_pin_weight.clone()
def push_metric(self, wns, tns):
metrics_dict = {
"wns": wns,
"tns": tns,
}
self.recorder.push(**metrics_dict)
def visualize(self, args):
file_prefix = "%s_" % args.design_name
res_root = os.path.join(args.result_dir, args.exp_id)
prefix = os.path.join(res_root, args.eval_dir, file_prefix)
if not os.path.exists(os.path.dirname(prefix)):
os.makedirs(os.path.dirname(prefix))
self.recorder.visualize(prefix, True)
def merged_wl_loss_grad_timing(
node_pos,
timing_pin_grads,
pin_id2node_id,
pin_rel_cpos,
node2pin_list,
node2pin_list_end,
hyperedge_list,
hyperedge_list_end,
net_mask,
net_weight,
hpwl_scale,
gamma,
deterministic,
cache_hpwl=True,
):
(
partial_wa_wl,
node_grad,
partial_hpwl,
) = wirelength_timing_cuda.merged_wl_loss_grad_timing(
node_pos,
timing_pin_grads,
pin_id2node_id,
pin_rel_cpos,
node2pin_list,
node2pin_list_end,
hyperedge_list,
hyperedge_list_end,
net_mask,
net_weight,
hpwl_scale,
gamma,
deterministic,
)
return torch.sum(partial_wa_wl), node_grad

View File

@ -110,11 +110,12 @@ class PlaceData(object):
# TODO: more cases, hardcode?
self.node_special_type = torch.zeros(len(node_id2celltype_name), dtype=torch.int32)
## too slow...
# for node_id, celltype_name in enumerate(node_id2celltype_name):
# if celltype_name.startswith("CORE/BUF"):
# self.node_special_type[node_id] = 1
# if celltype_name.startswith("CORE/DFF"):
# self.node_special_type[node_id] = 2
for node_id, celltype_name in enumerate(node_id2celltype_name):
if celltype_name.startswith("CORE/BUF"):
self.node_special_type[node_id] = 1
if celltype_name.startswith("CORE/DFF"):
self.node_special_type[node_id] = 2
dataset_format = ""
if "aux" in dataset_path.keys():
@ -506,10 +507,16 @@ class PlaceData(object):
has_dict = any([isinstance(item, dict) for _, item in self])
if not has_dict:
info = [size_repr(key, item) for key, item in self]
info = [
size_repr(key, item) for key, item in self if type(item) != dict
]
return "{}({}, {})".format(cls, self.design_name, ", ".join(info))
else:
info = [size_repr(key, item, indent=2) for key, item in self]
info = [
size_repr(key, item, indent=2)
for key, item in self
if type(item) != dict
]
return "{}({}, \n{}\n)".format(cls, self.design_name, ",\n".join(info))
def backup_ori_var(self):
@ -606,6 +613,7 @@ class PlaceData(object):
self.net_mask = torch.logical_and(
self.net_to_num_pins <= args.ignore_net_degree, self.net_to_num_pins >= 2
) # 0: ignore, 1: consider in wirelength calculation
self.net_weight = torch.ones(self.num_nets, device=device, dtype=dtype)
# macros -> all mov nodes has ultra-large areas with >= 3 row height and all fixed nodes
# But nodes with zero width or zero height are not considered as macros
mov_lhs, mov_rhs = self.movable_index
@ -882,6 +890,7 @@ class PlaceData(object):
scale = (self.die_ur - self.die_ll) * 0.001
loc = (self.die_ur + self.die_ll) * 0.5
mov_node_pos = torch.randn_like(mov_node_pos) * scale + loc
# mov_node_pos = torch.randn(mov_node_pos.shape).to(mov_node_pos.device) * scale + loc
elif init_method == "randn_center_lxly":
# TODO: Mixed-size placement is very sensitive to the initial location.
# An elegant yet effective initialization may be needed.

View File

@ -1,7 +1,7 @@
import torch
from .database import PlaceData
from cpp_to_py import density_map_cuda
from .core import merged_wl_loss_grad
from .core import merged_wl_loss_grad, merged_wl_loss_grad_timing
def get_init_density_map(rawdb, gpdb, data: PlaceData, args, logger, ps=None):
@ -23,6 +23,9 @@ def get_init_density_map(rawdb, gpdb, data: PlaceData, args, logger, ps=None):
node_pos = torch.cat([data.node_pos[data.is_mov_macro].contiguous(), node_pos])
node_size = torch.cat([data.node_size[data.is_mov_macro].contiguous(), node_size])
node_weight = node_size.new_ones(node_size.shape[0])
if args.timing_opt: # TODO: sideline
node_size = node_size.clone()
node_size[:, 0] *= 1.05
init_density_map = density_map_cuda.forward_naive(
node_pos, node_size, node_weight, data.unit_len, zeros_density_map,
data.num_bin_x, data.num_bin_y, node_pos.shape[0], -1.0, -1.0, 1e-4, False,
@ -126,11 +129,11 @@ def init_params(
(mov_node_pos * 0.0).sum().backward()
optimizer.zero_grad(set_to_none=False)
if not ps.rerun_route or route_fn is None:
if (not ps.rerun_route or route_fn is None) and not args.timing_opt:
init_density_weight = (wl_grad.norm(p=1) / density_grad.norm(p=1)).detach()
# init_density_weight = (wl_grad.norm(p=1) / grad_mat.norm(p=1)).detach()
ps.set_init_param(init_density_weight, data)
else:
elif ps.rerun_route and route_fn is not None:
_, filler_lhs = data.movable_connected_index
filler_rhs = mov_node_pos.shape[0]
@ -147,7 +150,22 @@ def init_params(
ps.set_route_init_param(
init_density_weight, init_route_weight, init_congest_weight, data, args
)
elif args.timing_opt:
init_pin_weight = torch.ones(data.num_pins, dtype=torch.float32, device=data.device)
_, wl_grad_timing = merged_wl_loss_grad_timing(
conn_node_pos, init_pin_weight,
data.pin_id2node_id, data.pin_rel_cpos,
data.node2pin_list, data.node2pin_list_end, data.hyperedge_list, data.hyperedge_list_end,
data.net_mask, data.net_weight, data.hpwl_scale, ps.wa_coeff, args.deterministic
)
init_density_weight = ((wl_grad.norm(p=1) + wl_grad_timing.norm(p=1)) / density_grad.norm(p=1)).detach()
ps.set_init_param(init_density_weight, data)
ps.timing_wl_weight = args.timing_init_weight
data.gputimer.timing_pin_weight *= ps.timing_wl_weight
data.gputimer.beta = 5 * ps.timing_wl_weight
# timing mode fluctuation, larger divergence life
ps.max_life = 50
ps.life = ps.max_life
# Nesterove learning rate initialization
def estimate_initial_learning_rate(obj_and_grad_fn, constraint_fn, x_k, lr):

View File

@ -29,10 +29,12 @@ class MetricRecorder:
)
self[key].append(item)
def visualize(self, prefix):
def visualize(self, prefix, log = False):
for key, value in self:
x = list(range(len(value)))
plt.plot(x, value, label=key)
if log:
plt.yscale("log")
plt.legend()
plt.savefig(prefix + "%s.png" % key)
plt.close()
@ -148,6 +150,13 @@ class ParamScheduler:
self.include_macros = args.include_macros
self.zero_macro_grad = False
# timing parameter
self.timing_sol_recorder = []
self.enable_timing = False
self.timing_wl_weight = 0
self.best_sol_timing: torch.Tensor = None
self.best_metric_timing = {"wns": float("inf"), "tns": float("inf"), "iter": 0}
def set_init_param(self, init_density_weight, data: PlaceData):
# init_density_weight
self.init_iter = self.iter
@ -231,6 +240,19 @@ class ParamScheduler:
gr_metrics, hpwl, overflow, mov_node_pos.detach().clone()
))
def push_timing_sol(self, timing_metrics, hpwl, overflow, mov_node_pos: torch.Tensor):
if overflow > self.stop_overflow: # skip overflow solutions
return
wns_early, tns_early, wns_late, tns_late = timing_metrics
wns = -min(wns_early, wns_late, 0)
tns = -min(tns_early, tns_late, 0)
if wns <= self.best_metric_timing["wns"]:
self.best_metric_timing["wns"] = wns
self.best_metric_timing["iter"] = self.iter
self.best_sol_timing = mov_node_pos.detach().clone()
if tns < self.best_metric_timing["tns"]:
self.best_metric_timing["tns"] = tns
def step(self, hpwl, overflow, node_pos, data):
self.update_precond_weight(data)
self.push_metric(hpwl, overflow)
@ -583,6 +605,20 @@ class ParamScheduler:
)
return best_sol
def get_best_timing_sol(self):
logger = self.__logger__
if self.best_sol_timing is not None:
wns = self.best_metric_timing["wns"]
tns = self.best_metric_timing["tns"]
iterationn = self.best_metric_timing["iter"]
logger.info(
"Find best timing solution: WNS %.4f TNS %.4f at Iter %d" % (wns, tns, iterationn)
)
else:
logger.info("Cannot find best timing solution. Use the last solution.")
return None
return self.best_sol_timing.data
def visualize(self, args, logger):
file_prefix = "%s_" % args.design_name

View File

@ -10,7 +10,7 @@ def get_trunc_node_pos_fn(mov_node_size, data):
return x
return trunc_node_pos_fn
def global_placement_main(gpdb, rawdb, ps: ParamScheduler, data: PlaceData, args, logger):
def global_placement_main(gpdb, rawdb, ps: ParamScheduler, data: PlaceData, args, logger, params, gputimer=None):
init_density_map = data.init_density_map
if not args.global_placement:
logger.info("Global placement is switched off. Please make sure the input "
@ -135,12 +135,36 @@ def global_placement_main(gpdb, rawdb, ps: ParamScheduler, data: PlaceData, args
terminate_signal = False
route_early_terminate_signal = False
log_info = False
timing_cali_thrs_overflow = 0.5
timing_calibration = False
for iteration in range(args.inner_iter):
# optimizer.zero_grad() # zero grad inside obj_and_grad_fn
obj = optimizer.step(obj_and_grad_fn)
hpwl, overflow = evaluator_fn(mov_node_pos)
# update parameters
ps.step(hpwl, overflow, mov_node_pos, data)
# Perform timing-opt.
if args.timing_opt and iteration > args.timing_start_iter and iteration % 1 == 0:
ps.enable_timing = True
node_pos = torch.cat([mov_node_pos[mov_lhs:mov_rhs].clone(), data.node_pos[mov_rhs:]], dim=0)
if args.calibration and ps.recorder.overflow[-1] < timing_cali_thrs_overflow:
gputimer.update_timing_calibrated(node_pos, record=True)
timing_cali_thrs_overflow -= args.calibration_step
timing_calibration = True
elif timing_calibration:
gputimer.update_timing_calibrated(node_pos)
else:
gputimer.update_timing(node_pos)
timing_metrics = gputimer.report_timing_slack()
wns_early, tns_early, wns_late, tns_late = timing_metrics
ps.push_timing_sol(timing_metrics, hpwl, overflow, mov_node_pos)
if iteration % args.timing_freq == 0:
gputimer.step(ps, node_pos, data)
if ps.need_to_early_stop():
terminate_signal = True
log_info = True
@ -317,6 +341,10 @@ def global_placement_main(gpdb, rawdb, ps: ParamScheduler, data: PlaceData, args
ps.wa_coeff,
)
)
if ps.enable_timing:
log_str += " | early WNS/TNS: %.4f %.4f (ns) | late WNS/TNS: %.4f %.4f (ns)" % (
wns_early, tns_early, wns_late, tns_late
)
logger.info(log_str)
if args.draw_placement:
info = (iteration, hpwl, data.design_name)
@ -365,6 +393,10 @@ def global_placement_main(gpdb, rawdb, ps: ParamScheduler, data: PlaceData, args
ps.push_gr_sol(gr_metrics, hpwl, overflow, mov_node_pos)
best_sol_gr = ps.get_best_gr_sol()
mov_node_pos[mov_lhs:mov_rhs].data.copy_(best_sol_gr[mov_lhs:mov_rhs])
if ps.enable_timing and not ps.enable_route:
best_sol_timing = ps.get_best_timing_sol()
if best_sol_timing is not None:
mov_node_pos[mov_lhs:mov_rhs].data.copy_(best_sol_timing[mov_lhs:mov_rhs])
node_pos = mov_node_pos[mov_lhs:mov_rhs]
node_pos = torch.cat([node_pos, data.node_pos[mov_rhs:]], dim=0)
@ -423,8 +455,11 @@ def run_placement_main_nesterov(args, logger):
# global placement
node_pos, iteration, gp_hpwl, overflow, gp_time, gp_per_iter = global_placement_main(
gpdb, rawdb, ps, data, args, logger
gpdb, rawdb, ps, data, args, logger, params, gputimer
)
if args.timing_opt:
wns_early_gp, tns_early_gp, wns_late_gp, tns_late_gp = timing_eval_func(node_pos)
# detail placement
node_pos, dp_hpwl, top5overflow, lg_time, dp_time = detail_placement_main(
node_pos, gpdb, rawdb, ps, data, args, logger

43
thirdparty/flute_mp/CMakeLists.txt vendored Normal file
View File

@ -0,0 +1,43 @@
# cmake_minimum_required(VERSION 2.8.12)
# set(CMAKE_CXX_STANDARD 17)
# set(FLUTE_INCLUDE_DIR "${CMAKE_CURRENT_LIST_DIR}" CACHE INTERNAL "Directory where flute.h is located")
# set(CMAKE_CXX_STANDARD_REQUIRED ON)
# project(flute)
# if(NOT CMAKE_BUILD_TYPE)
# set(CMAKE_BUILD_TYPE "Release" CACHE STRING
# "Choose the type of build, options are: Debug Release."
# FORCE)
# endif(NOT CMAKE_BUILD_TYPE)
# set(SOURCES ${CMAKE_CURRENT_SOURCE_DIR}/flute.cpp)
# include_directories("${CMAKE_CURRENT_SOURCE_DIR}")
# set(CMAKE_POSITION_INDEPENDENT_CODE ON)
# add_library(${PROJECT_NAME} STATIC ${SOURCES})
# install(TARGETS ${PROJECT_NAME} DESTINATION thirdparty/${PROJECT_NAME})
# install(DIRECTORY lut.ICCAD2015 DESTINATION thirdparty/${PROJECT_NAME})
CMAKE_MINIMUM_REQUIRED (VERSION 3.1)
PROJECT(flute_mp)
SET(CMAKE_CXX_STANDARD 17)
set(FLUTE_MP_INCLUDE_DIR "${CMAKE_CURRENT_LIST_DIR}" CACHE INTERNAL "Directory where flute.h is located")
FILE(GLOB_RECURSE SRC_FILES_FLUTE_MP ${CMAKE_CURRENT_SOURCE_DIR}/*.cpp)
ADD_LIBRARY(flute_mp
SHARED
${SRC_FILES_FLUTE_MP}
)
TARGET_INCLUDE_DIRECTORIES(flute_mp PUBLIC ${FLUTE_MP_INCLUDE_DIR})
TARGET_COMPILE_OPTIONS(flute_mp PRIVATE -fPIC)
INSTALL(TARGETS flute_mp DESTINATION ${XPLACE_LIB_DIR})

50
thirdparty/flute_mp/LICENSE vendored Normal file
View File

@ -0,0 +1,50 @@
READ THIS LICENSE AGREEMENT CAREFULLY BEFORE USING THIS PRODUCT. BY USING
THIS PRODUCT YOU INDICATE YOUR ACCEPTANCE OF THE TERMS OF THE FOLLOWING
AGREEMENT. THESE TERMS APPLY TO YOU AND ANY SUBSEQUENT LICENSEE OF THIS
PRODUCT.
License Agreement for FLUTE
Copyright (c) 2004 by Dr. Chris C. N. Chu
All rights reserved
ATTRIBUTION ASSURANCE LICENSE (adapted from the original BSD license)
Redistribution and use in source and binary forms, with or without
modification, are permitted provided that the conditions below are
met. These conditions require a modest attribution to Dr. Chris C. N. Chu
(the "Author").
1. Redistributions of the source code, with or without modification (the
"Code"), must be accompanied by any documentation and, each time
the resulting executable program or a program dependent thereon is
launched, a prominent display (e.g., splash screen or banner text) of
the Author's attribution information, which includes:
(a) Dr. Chris C. N. Chu ("AUTHOR"),
(b) Iowa State University ("PROFESSIONAL IDENTIFICATION"), and
(c) http://class.ee.iastate.edu/cnchu/ ("URL").
2. Users who intend to use the Code for commercial purposes will notify
Author prior to such commercial use.
3. Neither the name nor any trademark of the Author may be used to
endorse or promote products derived from this software without
specific prior written permission.
4. Users are entirely responsible, to the exclusion of the Author and any
other persons, for compliance with (1) regulations set by owners or
administrators of employed equipment, (2) licensing terms of any other
software, and (3) local, national, and international regulations
regarding use, including those regarding import, export, and use of
encryption software.
THIS FREE SOFTWARE IS PROVIDED BY THE AUTHOR "AS IS" AND ANY EXPRESS OR
IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES
OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED.
IN NO EVENT SHALL THE AUTHOR OR ANY CONTRIBUTOR BE LIABLE FOR ANY DIRECT,
INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
(INCLUDING, BUT NOT LIMITED TO, EFFECTS OF UNAUTHORIZED OR MALICIOUS
NETWORK ACCESS; PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF
THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.

1378
thirdparty/flute_mp/flute.cpp vendored Normal file

File diff suppressed because it is too large Load Diff

73
thirdparty/flute_mp/flute.hpp vendored Normal file
View File

@ -0,0 +1,73 @@
#ifndef FLUTE_HPP_
#define FLUTE_HPP_
#include <algorithm>
#include <climits>
#include <cmath>
#include <cstdio>
#include <cstdlib>
namespace Flute {
#define DEGREE 9 // LUT will be used when d <= DEGREE, DEGREE <= 9
#define FLUTEROUTING 1 // 1 to construct routing, 0 to estimate WL only
#define REMOVE_DUPLICATE_PIN 0 // Remove dup. pin for flute_wl() & flute()
#define ACCURACY 8 // Default accuracy is 3
#define FLUTE_ACCURACY 8
// #define MAXD 2008840 // max. degree of a net that can be handled
#define MAXD 1000 // max. degree of a net that can be handled
using DType = int;
using DTYPE = int;
// TODO:
// 1. replace macros with constexpr variables
// 2. construct a param struct
// 3. replace all malloc-free contents
// 4. clear inconsistent comments
struct Branch {
DType x, y; // starting point of the branch
int n; // index of neighbor
};
struct Tree {
int deg; // degree
DType length; // total wirelength
Branch* branch; // array of tree branches
};
// Major functions
void readLUT(const char* powv, const char* post);
DType flute_wl(int d, DType* x, DType* y, int acc);
Tree flute(int d, DType* x, DType* y, int acc);
DType wirelength(Tree t);
void printtree(Tree t);
// Other useful functions
DType flutes_wl_LD(int d, DType* xs, DType* ys, int* s);
DType flutes_wl_MD(int d, DType* xs, DType* ys, int* s, int acc);
DType flutes_wl_RDP(int d, DType* xs, DType* ys, int* s, int acc);
Tree flutes_LD(int d, DType* xs, DType* ys, int* s);
Tree flutes_MD(int d, DType* xs, DType* ys, int* s, int acc);
Tree flutes_RDP(int d, DType* xs, DType* ys, int* s, int acc);
#if REMOVE_DUPLICATE_PIN == 1
#define flutes_wl(d, xs, ys, s, acc) flutes_wl_RDP(d, xs, ys, s, acc)
#define flutes(d, xs, ys, s, acc) flutes_RDP(d, xs, ys, s, acc)
#else
#define flutes_wl(d, xs, ys, s, acc) flutes_wl_ALLD(d, xs, ys, s, acc)
#define flutes(d, xs, ys, s, acc) flutes_ALLD(d, xs, ys, s, acc)
#endif
#define flutes_wl_ALLD(d, xs, ys, s, acc) flutes_wl_LMD(d, xs, ys, s, acc)
#define flutes_ALLD(d, xs, ys, s, acc) flutes_LMD(d, xs, ys, s, acc)
#define flutes_wl_LMD(d, xs, ys, s, acc) \
(d <= DEGREE ? flutes_wl_LD(d, xs, ys, s) : flutes_wl_MD(d, xs, ys, s, acc))
#define flutes_LMD(d, xs, ys, s, acc) \
(d <= DEGREE ? flutes_LD(d, xs, ys, s) : flutes_MD(d, xs, ys, s, acc))
} // namespace flute
#endif // FLUTE_HPP_

489580
thirdparty/flute_mp/lut.ICCAD2015/POST9.dat vendored Normal file

File diff suppressed because it is too large Load Diff

515760
thirdparty/flute_mp/lut.ICCAD2015/POWV9.dat vendored Normal file

File diff suppressed because it is too large Load Diff

View File

@ -150,6 +150,10 @@ class IOParser(object):
die_info = torch.tensor([dieLX, dieHX, dieLY, dieHY]).float()
# die_shift = torch.tensor([dieLX, dieLY])
# die_scale = torch.tensor([dieHX - dieLX, dieHY - dieLY])
net_names = gpdb.net_names()
pin_names = gpdb.pin_names()
node_names = gpdb.node_names()
microns = gpdb.microns()
siteWidth = gpdb.siteWidth()
siteHeight = gpdb.siteHeight()
@ -202,6 +206,10 @@ class IOParser(object):
design_info = {
"benchmark": self.params["benchmark"],
"dataset_path": self.params,
"node_names": node_names,
"net_names": net_names,
"pin_names": pin_names,
"microns": microns,
"node_type_indices": node_type_indices,
"node_id2node_name": node_id2node_name,
"node_id2celltype_name": node_id2celltype_name,

View File

@ -61,6 +61,13 @@ def setup_design_args(args):
elif args.design_name in ["bigblue3", "bigblue4"]:
args.num_bin_x = args.num_bin_y = 2048
args.target_density = 1.0
elif args.design_name in ["superblue1", "superblue3", "superblue4", "superblue5", "superblue16", "superblue18"]:
args.num_bin_x = args.num_bin_y = 512
args.target_density = 1.0
elif args.design_name in ["superblue7", "superblue10"]:
args.num_bin_x = args.num_bin_y = 1024
args.target_density = 1.0
args.start_iter = 150
elif args.design_name in ["adaptec5"]:
args.target_density = 0.5
args.num_bin_x = args.num_bin_y = 1024