add timing-driven wirelength model; add flute multi thread; setup iccad2015 dataset; refactor flow
This commit is contained in:
parent
6aa1b0e253
commit
649f320384
@ -9,3 +9,5 @@ add_subdirectory(hpwl_cuda)
|
|||||||
add_subdirectory(io_parser)
|
add_subdirectory(io_parser)
|
||||||
add_subdirectory(routedp)
|
add_subdirectory(routedp)
|
||||||
add_subdirectory(wa_wirelength_hpwl_cuda)
|
add_subdirectory(wa_wirelength_hpwl_cuda)
|
||||||
|
add_subdirectory(gputimer)
|
||||||
|
add_subdirectory(wirelength_timing_cuda)
|
||||||
@ -10,6 +10,8 @@ __all__ = [
|
|||||||
"gpugr",
|
"gpugr",
|
||||||
"gpudp",
|
"gpudp",
|
||||||
"routedp",
|
"routedp",
|
||||||
|
"gputimer",
|
||||||
|
"wirelength_timing_cuda"
|
||||||
]
|
]
|
||||||
from .cpybin import (
|
from .cpybin import (
|
||||||
dct_cuda,
|
dct_cuda,
|
||||||
@ -22,5 +24,7 @@ from .cpybin import (
|
|||||||
gpugr,
|
gpugr,
|
||||||
gpudp,
|
gpudp,
|
||||||
routedp,
|
routedp,
|
||||||
|
gputimer,
|
||||||
|
wirelength_timing_cuda
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
9
cpp_to_py/wirelength_timing_cuda/CMakeLists.txt
Normal file
9
cpp_to_py/wirelength_timing_cuda/CMakeLists.txt
Normal file
@ -0,0 +1,9 @@
|
|||||||
|
set(TARGET_NAME wirelength_timing_cuda)
|
||||||
|
|
||||||
|
add_pytorch_extension(${TARGET_NAME}
|
||||||
|
${TARGET_NAME}.cpp
|
||||||
|
${TARGET_NAME}_kernel.cu)
|
||||||
|
|
||||||
|
install(TARGETS
|
||||||
|
${TARGET_NAME}
|
||||||
|
DESTINATION ${XPLACE_LIB_DIR})
|
||||||
52
cpp_to_py/wirelength_timing_cuda/wirelength_timing_cuda.cpp
Normal file
52
cpp_to_py/wirelength_timing_cuda/wirelength_timing_cuda.cpp
Normal file
@ -0,0 +1,52 @@
|
|||||||
|
#include <torch/extension.h>
|
||||||
|
|
||||||
|
std::vector<torch::Tensor> wa_wirelength_timing_weight_cuda(torch::Tensor node_pos,
|
||||||
|
torch::Tensor timing_pin_grad,
|
||||||
|
torch::Tensor pin_id2node_id,
|
||||||
|
torch::Tensor pin_rel_cpos,
|
||||||
|
torch::Tensor node2pin_list,
|
||||||
|
torch::Tensor node2pin_list_end,
|
||||||
|
torch::Tensor hyperedge_list,
|
||||||
|
torch::Tensor hyperedge_list_end,
|
||||||
|
torch::Tensor net_mask,
|
||||||
|
torch::Tensor net_weight,
|
||||||
|
torch::Tensor hpwl_scale,
|
||||||
|
float gamma,
|
||||||
|
bool deterministic);
|
||||||
|
|
||||||
|
#define CHECK_CUDA(x) TORCH_CHECK(x.device().is_cuda(), #x " must be a CUDA tensor")
|
||||||
|
#define CHECK_CONTIGUOUS(x) TORCH_CHECK(x.is_contiguous(), #x " must be contiguous")
|
||||||
|
#define CHECK_INPUT(x) \
|
||||||
|
CHECK_CUDA(x); \
|
||||||
|
CHECK_CONTIGUOUS(x)
|
||||||
|
|
||||||
|
std::vector<torch::Tensor> wa_wirelength_timing_weight(torch::Tensor node_pos,
|
||||||
|
torch::Tensor timing_pin_grad,
|
||||||
|
torch::Tensor pin_id2node_id,
|
||||||
|
torch::Tensor pin_rel_cpos,
|
||||||
|
torch::Tensor node2pin_list,
|
||||||
|
torch::Tensor node2pin_list_end,
|
||||||
|
torch::Tensor hyperedge_list,
|
||||||
|
torch::Tensor hyperedge_list_end,
|
||||||
|
torch::Tensor net_mask,
|
||||||
|
torch::Tensor net_weight,
|
||||||
|
torch::Tensor hpwl_scale,
|
||||||
|
float gamma,
|
||||||
|
bool deterministic) {
|
||||||
|
CHECK_INPUT(node_pos);
|
||||||
|
CHECK_INPUT(timing_pin_grad);
|
||||||
|
CHECK_INPUT(pin_id2node_id);
|
||||||
|
CHECK_INPUT(pin_rel_cpos);
|
||||||
|
CHECK_INPUT(node2pin_list);
|
||||||
|
CHECK_INPUT(node2pin_list_end);
|
||||||
|
CHECK_INPUT(hyperedge_list);
|
||||||
|
CHECK_INPUT(hyperedge_list_end);
|
||||||
|
CHECK_INPUT(net_mask);
|
||||||
|
CHECK_INPUT(net_weight);
|
||||||
|
CHECK_INPUT(hpwl_scale);
|
||||||
|
|
||||||
|
return wa_wirelength_timing_weight_cuda(
|
||||||
|
node_pos, timing_pin_grad, pin_id2node_id, pin_rel_cpos, node2pin_list, node2pin_list_end, hyperedge_list, hyperedge_list_end, net_mask, net_weight, hpwl_scale, gamma, deterministic);
|
||||||
|
}
|
||||||
|
|
||||||
|
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) { m.def("merged_wl_loss_grad_timing", &wa_wirelength_timing_weight, "calculate timing-driven WA wirelength pin grad"); }
|
||||||
@ -0,0 +1,210 @@
|
|||||||
|
#include <ATen/cuda/CUDAContext.h>
|
||||||
|
#include <cuda.h>
|
||||||
|
#include <cuda_runtime.h>
|
||||||
|
#include <torch/extension.h>
|
||||||
|
|
||||||
|
#include <vector>
|
||||||
|
|
||||||
|
__global__ void node_pos_to_pin_pos_cuda_kernel(
|
||||||
|
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_pos,
|
||||||
|
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> pin_id2node_id,
|
||||||
|
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_pos,
|
||||||
|
int num_pins) {
|
||||||
|
const int index = blockIdx.x * blockDim.x + threadIdx.x;
|
||||||
|
const int i = index >> 1; // pin index
|
||||||
|
if (i < num_pins) {
|
||||||
|
const int c = index & 1; // channel index
|
||||||
|
int64_t node_id = pin_id2node_id[i];
|
||||||
|
pin_pos[i][c] += node_pos[node_id][c];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
__global__ void calc_node_grad_deterministic_cuda_kernel(
|
||||||
|
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> node_grad,
|
||||||
|
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_grad,
|
||||||
|
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> node2pin_list,
|
||||||
|
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> node2pin_list_end,
|
||||||
|
int num_nodes) {
|
||||||
|
const int index = blockIdx.x * blockDim.x + threadIdx.x;
|
||||||
|
const int i = index >> 1; // node index
|
||||||
|
if (i < num_nodes) {
|
||||||
|
const int c = index & 1; // channel index
|
||||||
|
int64_t start_idx = 0;
|
||||||
|
if (i != 0) {
|
||||||
|
start_idx = node2pin_list_end[i - 1];
|
||||||
|
}
|
||||||
|
int64_t end_idx = node2pin_list_end[i];
|
||||||
|
if (end_idx != start_idx) {
|
||||||
|
node_grad[i][c] += pin_grad[node2pin_list[start_idx]][c];
|
||||||
|
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
|
||||||
|
node_grad[i][c] += pin_grad[node2pin_list[idx]][c];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
__global__ void wa_wirelength_pin_root_timing_kernel(
|
||||||
|
const torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_pos,
|
||||||
|
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> timing_pin_weight,
|
||||||
|
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list,
|
||||||
|
const torch::PackedTensorAccessor32<int64_t, 1, torch::RestrictPtrTraits> hyperedge_list_end,
|
||||||
|
const torch::PackedTensorAccessor32<bool, 1, torch::RestrictPtrTraits> net_mask,
|
||||||
|
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> net_weight,
|
||||||
|
const torch::PackedTensorAccessor32<float, 1, torch::RestrictPtrTraits> hpwl_scale,
|
||||||
|
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> partial_wa_wl,
|
||||||
|
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> partial_hpwl,
|
||||||
|
torch::PackedTensorAccessor32<float, 2, torch::RestrictPtrTraits> pin_grad,
|
||||||
|
int num_nets,
|
||||||
|
float inv_gamma) {
|
||||||
|
const int index = blockIdx.x * blockDim.x + threadIdx.x;
|
||||||
|
const int i = index >> 1; // net index
|
||||||
|
if (i < num_nets && net_mask[i]) {
|
||||||
|
const int c = index & 1; // channel index
|
||||||
|
int64_t start_idx = 0;
|
||||||
|
if (i != 0) {
|
||||||
|
start_idx = hyperedge_list_end[i - 1];
|
||||||
|
}
|
||||||
|
int64_t end_idx = hyperedge_list_end[i];
|
||||||
|
if (end_idx != start_idx) {
|
||||||
|
int64_t root_id = hyperedge_list[start_idx];
|
||||||
|
float x_min = pin_pos[root_id][c];
|
||||||
|
float x_max = pin_pos[root_id][c];
|
||||||
|
float root_x = pin_pos[root_id][c];
|
||||||
|
float recenter_exp_r = exp((root_x - x_max) * inv_gamma);
|
||||||
|
float recenter_exp_nr = exp((x_min - root_x) * inv_gamma);
|
||||||
|
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
|
||||||
|
float cur_x = pin_pos[hyperedge_list[idx]][c];
|
||||||
|
x_min = min(cur_x, x_min);
|
||||||
|
x_max = max(cur_x, x_max);
|
||||||
|
}
|
||||||
|
partial_hpwl[i][c] = round((x_max - x_min) * hpwl_scale[c]);
|
||||||
|
|
||||||
|
float sum_x_exp_x = root_x * recenter_exp_r;
|
||||||
|
float sum_x_exp_nx = root_x * recenter_exp_nr;
|
||||||
|
float sum_exp_x = recenter_exp_r;
|
||||||
|
float sum_exp_nx = recenter_exp_nr;
|
||||||
|
float wl_sum_x_exp_x = sum_x_exp_x;
|
||||||
|
float wl_sum_x_exp_nx = sum_x_exp_nx;
|
||||||
|
float wl_sum_exp_x = sum_exp_x;
|
||||||
|
float wl_sum_exp_nx = sum_exp_nx;
|
||||||
|
// pin-root gradient
|
||||||
|
for (int64_t idx = start_idx + 1; idx < end_idx; idx++) {
|
||||||
|
int64_t pin_id = hyperedge_list[idx];
|
||||||
|
float cur_x = pin_pos[pin_id][c];
|
||||||
|
float recenter_exp_x = exp((cur_x - x_max) * inv_gamma);
|
||||||
|
float recenter_exp_nx = exp((x_min - cur_x) * inv_gamma);
|
||||||
|
|
||||||
|
float sum_x_exp_x = cur_x * recenter_exp_x + root_x * recenter_exp_r;
|
||||||
|
float sum_x_exp_nx = cur_x * recenter_exp_nx + root_x * recenter_exp_nr;
|
||||||
|
float sum_exp_x = recenter_exp_x + recenter_exp_r;
|
||||||
|
float sum_exp_nx = recenter_exp_nx + recenter_exp_nr;
|
||||||
|
wl_sum_x_exp_x += cur_x * recenter_exp_x;
|
||||||
|
wl_sum_x_exp_nx += cur_x * recenter_exp_nx;
|
||||||
|
wl_sum_exp_x += recenter_exp_x;
|
||||||
|
wl_sum_exp_nx += recenter_exp_nx;
|
||||||
|
|
||||||
|
float inv_sum_exp_x = 1 / sum_exp_x;
|
||||||
|
float inv_sum_exp_nx = 1 / sum_exp_nx;
|
||||||
|
float s_x = sum_x_exp_x * inv_sum_exp_x;
|
||||||
|
float ns_nx = sum_x_exp_nx * inv_sum_exp_nx;
|
||||||
|
partial_wa_wl[i][c] += s_x - ns_nx;
|
||||||
|
float x_coeff = inv_gamma * inv_sum_exp_x;
|
||||||
|
float nx_coeff = -inv_gamma * inv_sum_exp_nx;
|
||||||
|
float grad_const = (1 - inv_gamma * s_x) * inv_sum_exp_x;
|
||||||
|
float grad_nconst = (1 + inv_gamma * ns_nx) * inv_sum_exp_nx;
|
||||||
|
|
||||||
|
float x_grad = (grad_const + x_coeff * cur_x) * recenter_exp_x -
|
||||||
|
(grad_nconst + nx_coeff * cur_x) * recenter_exp_nx;
|
||||||
|
float root_grad = (grad_const + x_coeff * root_x) * recenter_exp_r -
|
||||||
|
(grad_nconst + nx_coeff * root_x) * recenter_exp_nr;
|
||||||
|
|
||||||
|
float delta_x = timing_pin_weight[pin_id];
|
||||||
|
pin_grad[pin_id][c] = x_grad * delta_x;
|
||||||
|
pin_grad[root_id][c] += root_grad * delta_x;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void calc_node_grad_cuda(torch::Tensor node_grad,
|
||||||
|
torch::Tensor pin_id2node_id,
|
||||||
|
torch::Tensor pin_grad,
|
||||||
|
torch::Tensor node2pin_list,
|
||||||
|
torch::Tensor node2pin_list_end,
|
||||||
|
int num_nodes,
|
||||||
|
bool deterministic) {
|
||||||
|
if (deterministic) {
|
||||||
|
auto stream = at::cuda::getCurrentCUDAStream();
|
||||||
|
const int threads = 128;
|
||||||
|
const int blocks = (num_nodes * 2 + threads - 1) / threads;
|
||||||
|
calc_node_grad_deterministic_cuda_kernel<<<blocks, threads, 0, stream>>>(
|
||||||
|
node_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||||
|
pin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||||
|
node2pin_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||||
|
node2pin_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||||
|
num_nodes);
|
||||||
|
} else {
|
||||||
|
const auto pin_id2node_id_view = pin_id2node_id.unsqueeze(1).expand({-1, 2});
|
||||||
|
node_grad.scatter_add_(0, pin_id2node_id_view, pin_grad);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
std::vector<torch::Tensor> wa_wirelength_timing_weight_cuda(torch::Tensor node_pos,
|
||||||
|
torch::Tensor timing_pin_weight,
|
||||||
|
torch::Tensor pin_id2node_id,
|
||||||
|
torch::Tensor pin_rel_cpos,
|
||||||
|
torch::Tensor node2pin_list,
|
||||||
|
torch::Tensor node2pin_list_end,
|
||||||
|
torch::Tensor hyperedge_list,
|
||||||
|
torch::Tensor hyperedge_list_end,
|
||||||
|
torch::Tensor net_mask,
|
||||||
|
torch::Tensor net_weight,
|
||||||
|
torch::Tensor hpwl_scale,
|
||||||
|
float gamma,
|
||||||
|
bool deterministic) {
|
||||||
|
cudaSetDevice(node_pos.get_device());
|
||||||
|
auto stream = at::cuda::getCurrentCUDAStream();
|
||||||
|
|
||||||
|
const auto num_nodes = node_pos.size(0);
|
||||||
|
const auto num_pins = pin_id2node_id.size(0);
|
||||||
|
const auto num_nets = hyperedge_list_end.size(0);
|
||||||
|
const auto num_channels = 2; // x, y
|
||||||
|
|
||||||
|
auto pin_pos = pin_rel_cpos.clone(); // pin
|
||||||
|
auto partial_wa_wl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
|
||||||
|
auto partial_hpwl = torch::zeros({num_nets, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
|
||||||
|
auto pin_grad = torch::zeros({num_pins, num_channels}, torch::dtype(pin_pos.dtype()).device(pin_pos.device()));
|
||||||
|
|
||||||
|
const int threads = 128;
|
||||||
|
const int blocks = (num_pins * 2 + threads - 1) / threads;
|
||||||
|
|
||||||
|
node_pos_to_pin_pos_cuda_kernel<<<blocks, threads, 0, stream>>>(
|
||||||
|
node_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||||
|
pin_id2node_id.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||||
|
pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||||
|
num_pins);
|
||||||
|
|
||||||
|
const int threads2 = 128;
|
||||||
|
const int blocks2 = (num_nets * 2 + threads2 - 1) / threads2;
|
||||||
|
|
||||||
|
float inv_gamma = 1 / gamma;
|
||||||
|
wa_wirelength_pin_root_timing_kernel<<<blocks2, threads2, 0, stream>>>(
|
||||||
|
pin_pos.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||||
|
timing_pin_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||||
|
hyperedge_list.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||||
|
hyperedge_list_end.packed_accessor32<int64_t, 1, torch::RestrictPtrTraits>(),
|
||||||
|
net_mask.packed_accessor32<bool, 1, torch::RestrictPtrTraits>(),
|
||||||
|
net_weight.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||||
|
hpwl_scale.packed_accessor32<float, 1, torch::RestrictPtrTraits>(),
|
||||||
|
partial_wa_wl.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||||
|
partial_hpwl.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||||
|
pin_grad.packed_accessor32<float, 2, torch::RestrictPtrTraits>(),
|
||||||
|
num_nets,
|
||||||
|
inv_gamma);
|
||||||
|
|
||||||
|
auto node_grad = torch::zeros({num_nodes, num_channels}, torch::dtype(pin_grad.dtype()).device(pin_grad.device()));
|
||||||
|
calc_node_grad_cuda(
|
||||||
|
node_grad, pin_id2node_id, pin_grad, node2pin_list, node2pin_list_end, num_nodes, deterministic);
|
||||||
|
|
||||||
|
return {partial_wa_wl, node_grad, partial_hpwl};
|
||||||
|
}
|
||||||
13
main.py
13
main.py
@ -49,6 +49,18 @@ def get_option():
|
|||||||
parser.add_argument('--pseudo_weight', type=float, default=0, help='the weight of pseudo net')
|
parser.add_argument('--pseudo_weight', type=float, default=0, help='the weight of pseudo net')
|
||||||
parser.add_argument('--visualize_cgmap', type=str2bool, default=False, help='visualize congestion map')
|
parser.add_argument('--visualize_cgmap', type=str2bool, default=False, help='visualize congestion map')
|
||||||
|
|
||||||
|
# timing opt params
|
||||||
|
parser.add_argument('--timing_opt', type=str2bool, default=True, help='perform timing optimization')
|
||||||
|
parser.add_argument('--timing_freq', type=int, default=1, help='timing freq')
|
||||||
|
parser.add_argument('--calibration', type=str2bool, default=True, help='perform timer calibration')
|
||||||
|
parser.add_argument('--calibration_step', type=float, default=0.1, help='timing calibration step')
|
||||||
|
parser.add_argument('--timing_start_iter', type=int, default=100, help='start iteration of timing optimization')
|
||||||
|
parser.add_argument('--timing_init_weight', type=float, default=0.05, help='initial timing wirelength weight')
|
||||||
|
parser.add_argument('--decay_factor', type=float, default=0.3, help='decay factor of timing weight')
|
||||||
|
parser.add_argument('--decay_boost', type=float, default=3, help='dynamic decay boost factor')
|
||||||
|
parser.add_argument('--wire_resistance_per_micron', type=float, default=2.535, help='unit wire resistance, normalized across all layers')
|
||||||
|
parser.add_argument('--wire_capacitance_per_micron', type=float, default=0.16e-15, help='unit wire capacitance, normalized across all layers')
|
||||||
|
|
||||||
# detailed placement and evaluation
|
# detailed placement and evaluation
|
||||||
parser.add_argument('--legalization', type=str2bool, default=True, help='perform lg')
|
parser.add_argument('--legalization', type=str2bool, default=True, help='perform lg')
|
||||||
parser.add_argument('--detail_placement', type=str2bool, default=True, help='perform dp')
|
parser.add_argument('--detail_placement', type=str2bool, default=True, help='perform dp')
|
||||||
@ -77,6 +89,7 @@ def get_option():
|
|||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
args.exp_id = datetime.datetime.now().strftime('%Y-%m-%d-%H:%M:%S') + args.exp_id
|
args.exp_id = datetime.datetime.now().strftime('%Y-%m-%d-%H:%M:%S') + args.exp_id
|
||||||
|
args.exp_id = "{}_{}".format(args.exp_id, args.design_name)
|
||||||
|
|
||||||
if args.dataset == "ispd2015":
|
if args.dataset == "ispd2015":
|
||||||
print("We haven't yet support fence region in ispd2015, use ispd2015_fix instead")
|
print("We haven't yet support fence region in ispd2015, use ispd2015_fix instead")
|
||||||
|
|||||||
@ -1,7 +1,6 @@
|
|||||||
import torch
|
import torch
|
||||||
from .param_scheduler import ParamScheduler
|
from .param_scheduler import ParamScheduler
|
||||||
from .core import merged_wl_loss_grad
|
from .core import merged_wl_loss_grad, merged_wl_loss_grad_timing
|
||||||
|
|
||||||
|
|
||||||
def apply_precond(mov_node_pos: torch.Tensor, ps: ParamScheduler, args):
|
def apply_precond(mov_node_pos: torch.Tensor, ps: ParamScheduler, args):
|
||||||
if not args.use_precond:
|
if not args.use_precond:
|
||||||
@ -51,6 +50,17 @@ def calc_obj_and_grad(
|
|||||||
data.hpwl_scale, ps.wa_coeff, args.deterministic
|
data.hpwl_scale, ps.wa_coeff, args.deterministic
|
||||||
)
|
)
|
||||||
mov_node_pos.grad[mov_lhs:mov_rhs] += conn_node_grad_by_wl[mov_lhs:mov_rhs]
|
mov_node_pos.grad[mov_lhs:mov_rhs] += conn_node_grad_by_wl[mov_lhs:mov_rhs]
|
||||||
|
|
||||||
|
if ps.enable_timing:
|
||||||
|
wl_loss_timing, conn_node_grad_by_timing = merged_wl_loss_grad_timing(
|
||||||
|
conn_node_pos, data.gputimer.timing_pin_weight,
|
||||||
|
data.pin_id2node_id, data.pin_rel_cpos,
|
||||||
|
data.node2pin_list, data.node2pin_list_end, data.hyperedge_list, data.hyperedge_list_end,
|
||||||
|
data.net_mask, data.net_weight, data.hpwl_scale, ps.wa_coeff, args.deterministic
|
||||||
|
)
|
||||||
|
mov_node_pos.grad[mov_lhs:mov_rhs] += conn_node_grad_by_timing[mov_lhs:mov_rhs]
|
||||||
|
wl_loss += wl_loss_timing
|
||||||
|
|
||||||
if ps.enable_sample_force:
|
if ps.enable_sample_force:
|
||||||
if ps.iter > 3 and ps.iter % 20 == 0:
|
if ps.iter > 3 and ps.iter % 20 == 0:
|
||||||
# ps.iter > 3 for warmup
|
# ps.iter > 3 for warmup
|
||||||
|
|||||||
@ -1,4 +1,5 @@
|
|||||||
from .flute import Flute, get_flute_wl
|
from .flute import Flute, get_flute_wl
|
||||||
from .electronic_density_layer import ElectronicDensityLayer
|
from .electronic_density_layer import ElectronicDensityLayer
|
||||||
from .wa_wirelength_hpwl import masked_scale_hpwl, merged_wl_loss_grad
|
from .wa_wirelength_hpwl import masked_scale_hpwl, merged_wl_loss_grad
|
||||||
from .route_force import get_route_force, run_gr_and_fft, run_gr_and_fft_main, route_inflation, route_inflation_roll_back
|
from .route_force import get_route_force, run_gr_and_fft, run_gr_and_fft_main, route_inflation, route_inflation_roll_back
|
||||||
|
from .timing_opt import GPUTimer, merged_wl_loss_grad_timing
|
||||||
271
src/core/timing_opt.py
Normal file
271
src/core/timing_opt.py
Normal file
@ -0,0 +1,271 @@
|
|||||||
|
import torch
|
||||||
|
from cpp_to_py import gputimer, wirelength_timing_cuda
|
||||||
|
from src.param_scheduler import MetricRecorder
|
||||||
|
from utils import *
|
||||||
|
|
||||||
|
class GPUTimer():
|
||||||
|
def __init__(self, data, rawdb, gpdb, params, args):
|
||||||
|
self.metrics = [
|
||||||
|
"wns",
|
||||||
|
"tns",
|
||||||
|
]
|
||||||
|
self.recorder = MetricRecorder(**{m: [] for m in self.metrics})
|
||||||
|
|
||||||
|
self.data = data
|
||||||
|
self.net_names = data.net_names
|
||||||
|
self.pin_names = data.pin_names
|
||||||
|
|
||||||
|
self.microns = data.microns
|
||||||
|
self.wire_resistance_per_micron = args.wire_resistance_per_micron
|
||||||
|
self.wire_capacitance_per_micron = args.wire_capacitance_per_micron
|
||||||
|
|
||||||
|
self.node_size = data.node_size.detach().clone()
|
||||||
|
node_lpos = data.node_pos.detach() - self.node_size / 2
|
||||||
|
self.pin_rel_lpos = data.pin_rel_lpos.detach() + data.pin_size / 2
|
||||||
|
|
||||||
|
die_info = data.die_info
|
||||||
|
xl, xh, yl, yh = die_info.cpu().numpy()
|
||||||
|
|
||||||
|
self.mov_lhs, self.mov_rhs = data.movable_index
|
||||||
|
fix_lhs, fix_rhs = data.fixed_connected_index
|
||||||
|
num_movable_nodes = self.mov_rhs - self.mov_lhs
|
||||||
|
self.fix_conn_node_lpos = node_lpos[fix_lhs:fix_rhs]
|
||||||
|
self.conn_node_lpos = torch.cat([
|
||||||
|
node_lpos[self.mov_lhs:self.mov_rhs], self.fix_conn_node_lpos
|
||||||
|
], dim=0)
|
||||||
|
|
||||||
|
scale_factor = 1.0 / data.site_width
|
||||||
|
self.node_weight = data.node_special_type == 2
|
||||||
|
|
||||||
|
self.timing_raw_db = gputimer.create_timing_rawdb(
|
||||||
|
self.conn_node_lpos,
|
||||||
|
data.node_size,
|
||||||
|
self.pin_rel_lpos,
|
||||||
|
data.pin_id2node_id,
|
||||||
|
data.pin_id2net_id.int(),
|
||||||
|
data.node2pin_list,
|
||||||
|
data.node2pin_list_end,
|
||||||
|
data.hyperedge_list.int(),
|
||||||
|
data.hyperedge_list_end.int(),
|
||||||
|
data.net_mask,
|
||||||
|
num_movable_nodes,
|
||||||
|
scale_factor,
|
||||||
|
self.microns,
|
||||||
|
self.wire_resistance_per_micron,
|
||||||
|
self.wire_capacitance_per_micron
|
||||||
|
)
|
||||||
|
|
||||||
|
self.timer = gputimer.create_gputimer(params, rawdb, gpdb, self.timing_raw_db)
|
||||||
|
|
||||||
|
## Timing optimization
|
||||||
|
self.timer.init()
|
||||||
|
self.timer.levelize()
|
||||||
|
self.pin_slack = torch.zeros(data.num_pins, dtype=torch.float32, device=data.device)
|
||||||
|
self.timing_pin_weight = torch.ones(data.num_pins, dtype=torch.float32, device=data.device)
|
||||||
|
self.history_x = None
|
||||||
|
|
||||||
|
self.tns_record = []
|
||||||
|
self.wns_record = []
|
||||||
|
self.wns_max = []
|
||||||
|
self.delay_K_max = []
|
||||||
|
self.delay_1_max = []
|
||||||
|
self.w_2 = None
|
||||||
|
self.w_1 = None
|
||||||
|
self.a_1 = None
|
||||||
|
self.decay = args.decay_factor
|
||||||
|
self.decay_boost = args.decay_boost
|
||||||
|
self.init_alpha = 1.05
|
||||||
|
self.beta = 0
|
||||||
|
self.alpha = torch.ones(data.num_pins, dtype=torch.float32, device=data.device) * self.init_alpha
|
||||||
|
self.global_weight = 1
|
||||||
|
|
||||||
|
self.target_wns = 0
|
||||||
|
|
||||||
|
def update_timing(self, node_pos):
|
||||||
|
node_lpos = (node_pos.detach() - self.node_size / 2).to(self.data.device)
|
||||||
|
self.conn_node_lpos = torch.cat([
|
||||||
|
node_lpos[self.mov_lhs:self.mov_rhs], self.fix_conn_node_lpos
|
||||||
|
], dim=0)
|
||||||
|
|
||||||
|
self.timer.update_states()
|
||||||
|
self.timer.update_rc(node_lpos, False, False, False)
|
||||||
|
self.timer.update_timing()
|
||||||
|
|
||||||
|
def update_timing_eval(self, node_pos):
|
||||||
|
node_lpos = (node_pos.detach() - self.node_size / 2).to(self.data.device)
|
||||||
|
self.conn_node_lpos = torch.cat([
|
||||||
|
node_lpos[self.mov_lhs:self.mov_rhs], self.fix_conn_node_lpos
|
||||||
|
], dim=0)
|
||||||
|
|
||||||
|
self.timer.update_states()
|
||||||
|
self.timer.update_rc_flute(node_lpos, False)
|
||||||
|
self.timer.update_timing()
|
||||||
|
|
||||||
|
def update_timing_calibrated(self, node_pos, record=False):
|
||||||
|
node_lpos = (node_pos.detach() - self.node_size / 2).to(self.data.device)
|
||||||
|
self.conn_node_lpos = torch.cat([
|
||||||
|
node_lpos[self.mov_lhs:self.mov_rhs], self.fix_conn_node_lpos
|
||||||
|
], dim=0)
|
||||||
|
|
||||||
|
if record:
|
||||||
|
self.timer.update_states()
|
||||||
|
self.timer.update_rc_flute(node_lpos, True)
|
||||||
|
self.timer.update_states()
|
||||||
|
self.timer.update_rc(node_lpos, True, True, True)
|
||||||
|
else:
|
||||||
|
self.timer.update_states()
|
||||||
|
self.timer.update_rc(node_lpos, False, True, True)
|
||||||
|
self.timer.update_timing()
|
||||||
|
|
||||||
|
|
||||||
|
def report_timing_slack(self):
|
||||||
|
time_unit = self.timer.time_unit()
|
||||||
|
self.timer.update_endpoints()
|
||||||
|
wns_early, tns_early, wns_late, tns_late = self.timer.report_wns_and_tns()
|
||||||
|
wns_early = (wns_early.item() * (time_unit * 1e9))
|
||||||
|
wns_late = (wns_late.item() * (time_unit * 1e9))
|
||||||
|
tns_early = (tns_early.item() * (time_unit * 1e9))
|
||||||
|
tns_late = (tns_late.item() * (time_unit * 1e9))
|
||||||
|
self.push_metric(-wns_late, -tns_late)
|
||||||
|
return wns_early, tns_early, wns_late, tns_late
|
||||||
|
|
||||||
|
def report_pin_slack(self):
|
||||||
|
self.pin_slack = self.timer.report_pin_slack()
|
||||||
|
return self.pin_slack
|
||||||
|
|
||||||
|
def report_path(self, ep_name=None, el = -1, verbose=False):
|
||||||
|
if ep_name is not None:
|
||||||
|
ep_idx = self.pin_names.index(ep_name)
|
||||||
|
path, at, delay = self.timer.report_path(ep_idx, el, verbose)
|
||||||
|
else:
|
||||||
|
path, at, delay = self.timer.report_path(-1, el, verbose)
|
||||||
|
return path, at, delay
|
||||||
|
|
||||||
|
def report_arrival(self, pin_name):
|
||||||
|
pin_idx = self.pin_names.index(pin_name)
|
||||||
|
return self.timer.report_pin_at()[pin_idx]
|
||||||
|
|
||||||
|
def report_slew(self, pin_name):
|
||||||
|
pin_idx = self.pin_names.index(pin_name)
|
||||||
|
return self.timer.report_pin_slew()[pin_idx]
|
||||||
|
|
||||||
|
def report_load(self, pin_name):
|
||||||
|
pin_idx = self.pin_names.index(pin_name)
|
||||||
|
return self.timer.report_pin_load()[pin_idx]
|
||||||
|
|
||||||
|
def report_required(self, pin_name):
|
||||||
|
pin_idx = self.pin_names.index(pin_name)
|
||||||
|
return self.timer.report_pin_rat()[pin_idx]
|
||||||
|
|
||||||
|
def report_slack(self, pin_name):
|
||||||
|
pin_idx = self.pin_names.index(pin_name)
|
||||||
|
return self.timer.report_pin_slack()[pin_idx]
|
||||||
|
|
||||||
|
def get_node_critocality(self):
|
||||||
|
pin_slacks, _ = torch.min((torch.nan_to_num(self.report_pin_slack()) * (1e-9 / self.timer.time_unit())).clamp(max=0), 1)
|
||||||
|
endpoints_index = self.timer.endpoints_index().long()
|
||||||
|
endpoints_index = torch.unique(endpoints_index)
|
||||||
|
ep_id2node_id = self.data.pin_id2node_id[endpoints_index]
|
||||||
|
ep_slacks = pin_slacks[endpoints_index]
|
||||||
|
node_slacks = torch.zeros(self.node_weight.size(0), dtype=torch.float32, device=self.data.device)
|
||||||
|
node_slacks.scatter_add_(0, ep_id2node_id, ep_slacks)
|
||||||
|
node_critocality = torch.abs(node_slacks) / (torch.abs(node_slacks)).max()
|
||||||
|
return node_critocality
|
||||||
|
|
||||||
|
def step(self, ps, node_pos, data):
|
||||||
|
slacks, _ = torch.min(torch.nan_to_num(self.report_pin_slack()).clamp(max=0), 1)
|
||||||
|
delay_k, _ = self.timer.report_criticality_threshold(0.75, False, True)
|
||||||
|
delay_1, pin_visited = self.timer.report_criticality_threshold(0.99, False, True)
|
||||||
|
|
||||||
|
self.tns_record.append(self.recorder.tns[-1])
|
||||||
|
self.wns_record.append(self.recorder.wns[-1])
|
||||||
|
self.wns_max.append(slacks.min().item())
|
||||||
|
self.delay_K_max.append(delay_k.max().item())
|
||||||
|
self.delay_1_max.append(delay_1.max().item())
|
||||||
|
|
||||||
|
window = min(10, len(self.wns_max))
|
||||||
|
x_wns = torch.tensor(self.wns_max[-window:], dtype=torch.float32)
|
||||||
|
x_delay_k = torch.tensor(self.delay_K_max[-window:], dtype=torch.float32)
|
||||||
|
x_delay_1 = torch.tensor(self.delay_1_max[-window:], dtype=torch.float32)
|
||||||
|
|
||||||
|
wns_mean = x_wns[-1]
|
||||||
|
delay_k_mean = x_delay_k[-1]
|
||||||
|
delay_1_mean = x_delay_1[-1]
|
||||||
|
|
||||||
|
pin_weight = slacks.abs() / (np.abs(wns_mean)) * self.beta
|
||||||
|
pin_weight += (delay_k / delay_k_mean.clamp(min=1)) * self.beta * 2
|
||||||
|
pin_weight += torch.pow(2, (delay_1 / delay_1_mean.clamp(min=1))) * pin_visited.clamp(max=1)
|
||||||
|
|
||||||
|
w_0 = pin_weight
|
||||||
|
delta_w_0 = None
|
||||||
|
delta_w_1 = None
|
||||||
|
if self.w_1 is not None:
|
||||||
|
delta_w_0 = w_0 - self.w_1
|
||||||
|
if self.w_2 is not None:
|
||||||
|
delta_w_1 = self.w_1 - self.w_2
|
||||||
|
|
||||||
|
if delta_w_0 is not None and delta_w_1 is not None:
|
||||||
|
decay = (self.decay * torch.pow(5, delta_w_0.clamp(min=0)) / self.decay_boost).clamp(max=0.5)
|
||||||
|
else:
|
||||||
|
decay = 1
|
||||||
|
|
||||||
|
self.w_2 = self.w_1
|
||||||
|
self.w_1 = pin_weight.clone()
|
||||||
|
if self.history_x is None:
|
||||||
|
self.history_x = self.timing_pin_weight.clone()
|
||||||
|
self.timing_pin_weight = pin_weight.clamp(min=ps.timing_wl_weight)
|
||||||
|
self.timing_pin_weight = (decay * self.timing_pin_weight + (1 - decay) * self.history_x) * self.global_weight
|
||||||
|
self.timing_pin_weight = self.timing_pin_weight.contiguous().to(node_pos.device)
|
||||||
|
self.history_x = self.timing_pin_weight.clone()
|
||||||
|
|
||||||
|
def push_metric(self, wns, tns):
|
||||||
|
metrics_dict = {
|
||||||
|
"wns": wns,
|
||||||
|
"tns": tns,
|
||||||
|
}
|
||||||
|
self.recorder.push(**metrics_dict)
|
||||||
|
|
||||||
|
def visualize(self, args):
|
||||||
|
file_prefix = "%s_" % args.design_name
|
||||||
|
res_root = os.path.join(args.result_dir, args.exp_id)
|
||||||
|
prefix = os.path.join(res_root, args.eval_dir, file_prefix)
|
||||||
|
if not os.path.exists(os.path.dirname(prefix)):
|
||||||
|
os.makedirs(os.path.dirname(prefix))
|
||||||
|
self.recorder.visualize(prefix, True)
|
||||||
|
|
||||||
|
def merged_wl_loss_grad_timing(
|
||||||
|
node_pos,
|
||||||
|
timing_pin_grads,
|
||||||
|
pin_id2node_id,
|
||||||
|
pin_rel_cpos,
|
||||||
|
node2pin_list,
|
||||||
|
node2pin_list_end,
|
||||||
|
hyperedge_list,
|
||||||
|
hyperedge_list_end,
|
||||||
|
net_mask,
|
||||||
|
net_weight,
|
||||||
|
hpwl_scale,
|
||||||
|
gamma,
|
||||||
|
deterministic,
|
||||||
|
cache_hpwl=True,
|
||||||
|
):
|
||||||
|
(
|
||||||
|
partial_wa_wl,
|
||||||
|
node_grad,
|
||||||
|
partial_hpwl,
|
||||||
|
) = wirelength_timing_cuda.merged_wl_loss_grad_timing(
|
||||||
|
node_pos,
|
||||||
|
timing_pin_grads,
|
||||||
|
pin_id2node_id,
|
||||||
|
pin_rel_cpos,
|
||||||
|
node2pin_list,
|
||||||
|
node2pin_list_end,
|
||||||
|
hyperedge_list,
|
||||||
|
hyperedge_list_end,
|
||||||
|
net_mask,
|
||||||
|
net_weight,
|
||||||
|
hpwl_scale,
|
||||||
|
gamma,
|
||||||
|
deterministic,
|
||||||
|
)
|
||||||
|
return torch.sum(partial_wa_wl), node_grad
|
||||||
@ -110,11 +110,12 @@ class PlaceData(object):
|
|||||||
# TODO: more cases, hardcode?
|
# TODO: more cases, hardcode?
|
||||||
self.node_special_type = torch.zeros(len(node_id2celltype_name), dtype=torch.int32)
|
self.node_special_type = torch.zeros(len(node_id2celltype_name), dtype=torch.int32)
|
||||||
## too slow...
|
## too slow...
|
||||||
# for node_id, celltype_name in enumerate(node_id2celltype_name):
|
for node_id, celltype_name in enumerate(node_id2celltype_name):
|
||||||
# if celltype_name.startswith("CORE/BUF"):
|
if celltype_name.startswith("CORE/BUF"):
|
||||||
# self.node_special_type[node_id] = 1
|
self.node_special_type[node_id] = 1
|
||||||
# if celltype_name.startswith("CORE/DFF"):
|
if celltype_name.startswith("CORE/DFF"):
|
||||||
# self.node_special_type[node_id] = 2
|
self.node_special_type[node_id] = 2
|
||||||
|
|
||||||
|
|
||||||
dataset_format = ""
|
dataset_format = ""
|
||||||
if "aux" in dataset_path.keys():
|
if "aux" in dataset_path.keys():
|
||||||
@ -506,10 +507,16 @@ class PlaceData(object):
|
|||||||
has_dict = any([isinstance(item, dict) for _, item in self])
|
has_dict = any([isinstance(item, dict) for _, item in self])
|
||||||
|
|
||||||
if not has_dict:
|
if not has_dict:
|
||||||
info = [size_repr(key, item) for key, item in self]
|
info = [
|
||||||
|
size_repr(key, item) for key, item in self if type(item) != dict
|
||||||
|
]
|
||||||
return "{}({}, {})".format(cls, self.design_name, ", ".join(info))
|
return "{}({}, {})".format(cls, self.design_name, ", ".join(info))
|
||||||
else:
|
else:
|
||||||
info = [size_repr(key, item, indent=2) for key, item in self]
|
info = [
|
||||||
|
size_repr(key, item, indent=2)
|
||||||
|
for key, item in self
|
||||||
|
if type(item) != dict
|
||||||
|
]
|
||||||
return "{}({}, \n{}\n)".format(cls, self.design_name, ",\n".join(info))
|
return "{}({}, \n{}\n)".format(cls, self.design_name, ",\n".join(info))
|
||||||
|
|
||||||
def backup_ori_var(self):
|
def backup_ori_var(self):
|
||||||
@ -606,6 +613,7 @@ class PlaceData(object):
|
|||||||
self.net_mask = torch.logical_and(
|
self.net_mask = torch.logical_and(
|
||||||
self.net_to_num_pins <= args.ignore_net_degree, self.net_to_num_pins >= 2
|
self.net_to_num_pins <= args.ignore_net_degree, self.net_to_num_pins >= 2
|
||||||
) # 0: ignore, 1: consider in wirelength calculation
|
) # 0: ignore, 1: consider in wirelength calculation
|
||||||
|
self.net_weight = torch.ones(self.num_nets, device=device, dtype=dtype)
|
||||||
# macros -> all mov nodes has ultra-large areas with >= 3 row height and all fixed nodes
|
# macros -> all mov nodes has ultra-large areas with >= 3 row height and all fixed nodes
|
||||||
# But nodes with zero width or zero height are not considered as macros
|
# But nodes with zero width or zero height are not considered as macros
|
||||||
mov_lhs, mov_rhs = self.movable_index
|
mov_lhs, mov_rhs = self.movable_index
|
||||||
@ -882,6 +890,7 @@ class PlaceData(object):
|
|||||||
scale = (self.die_ur - self.die_ll) * 0.001
|
scale = (self.die_ur - self.die_ll) * 0.001
|
||||||
loc = (self.die_ur + self.die_ll) * 0.5
|
loc = (self.die_ur + self.die_ll) * 0.5
|
||||||
mov_node_pos = torch.randn_like(mov_node_pos) * scale + loc
|
mov_node_pos = torch.randn_like(mov_node_pos) * scale + loc
|
||||||
|
# mov_node_pos = torch.randn(mov_node_pos.shape).to(mov_node_pos.device) * scale + loc
|
||||||
elif init_method == "randn_center_lxly":
|
elif init_method == "randn_center_lxly":
|
||||||
# TODO: Mixed-size placement is very sensitive to the initial location.
|
# TODO: Mixed-size placement is very sensitive to the initial location.
|
||||||
# An elegant yet effective initialization may be needed.
|
# An elegant yet effective initialization may be needed.
|
||||||
|
|||||||
@ -1,7 +1,7 @@
|
|||||||
import torch
|
import torch
|
||||||
from .database import PlaceData
|
from .database import PlaceData
|
||||||
from cpp_to_py import density_map_cuda
|
from cpp_to_py import density_map_cuda
|
||||||
from .core import merged_wl_loss_grad
|
from .core import merged_wl_loss_grad, merged_wl_loss_grad_timing
|
||||||
|
|
||||||
|
|
||||||
def get_init_density_map(rawdb, gpdb, data: PlaceData, args, logger, ps=None):
|
def get_init_density_map(rawdb, gpdb, data: PlaceData, args, logger, ps=None):
|
||||||
@ -23,6 +23,9 @@ def get_init_density_map(rawdb, gpdb, data: PlaceData, args, logger, ps=None):
|
|||||||
node_pos = torch.cat([data.node_pos[data.is_mov_macro].contiguous(), node_pos])
|
node_pos = torch.cat([data.node_pos[data.is_mov_macro].contiguous(), node_pos])
|
||||||
node_size = torch.cat([data.node_size[data.is_mov_macro].contiguous(), node_size])
|
node_size = torch.cat([data.node_size[data.is_mov_macro].contiguous(), node_size])
|
||||||
node_weight = node_size.new_ones(node_size.shape[0])
|
node_weight = node_size.new_ones(node_size.shape[0])
|
||||||
|
if args.timing_opt: # TODO: sideline
|
||||||
|
node_size = node_size.clone()
|
||||||
|
node_size[:, 0] *= 1.05
|
||||||
init_density_map = density_map_cuda.forward_naive(
|
init_density_map = density_map_cuda.forward_naive(
|
||||||
node_pos, node_size, node_weight, data.unit_len, zeros_density_map,
|
node_pos, node_size, node_weight, data.unit_len, zeros_density_map,
|
||||||
data.num_bin_x, data.num_bin_y, node_pos.shape[0], -1.0, -1.0, 1e-4, False,
|
data.num_bin_x, data.num_bin_y, node_pos.shape[0], -1.0, -1.0, 1e-4, False,
|
||||||
@ -126,11 +129,11 @@ def init_params(
|
|||||||
(mov_node_pos * 0.0).sum().backward()
|
(mov_node_pos * 0.0).sum().backward()
|
||||||
optimizer.zero_grad(set_to_none=False)
|
optimizer.zero_grad(set_to_none=False)
|
||||||
|
|
||||||
if not ps.rerun_route or route_fn is None:
|
if (not ps.rerun_route or route_fn is None) and not args.timing_opt:
|
||||||
init_density_weight = (wl_grad.norm(p=1) / density_grad.norm(p=1)).detach()
|
init_density_weight = (wl_grad.norm(p=1) / density_grad.norm(p=1)).detach()
|
||||||
# init_density_weight = (wl_grad.norm(p=1) / grad_mat.norm(p=1)).detach()
|
# init_density_weight = (wl_grad.norm(p=1) / grad_mat.norm(p=1)).detach()
|
||||||
ps.set_init_param(init_density_weight, data)
|
ps.set_init_param(init_density_weight, data)
|
||||||
else:
|
elif ps.rerun_route and route_fn is not None:
|
||||||
_, filler_lhs = data.movable_connected_index
|
_, filler_lhs = data.movable_connected_index
|
||||||
filler_rhs = mov_node_pos.shape[0]
|
filler_rhs = mov_node_pos.shape[0]
|
||||||
|
|
||||||
@ -147,7 +150,22 @@ def init_params(
|
|||||||
ps.set_route_init_param(
|
ps.set_route_init_param(
|
||||||
init_density_weight, init_route_weight, init_congest_weight, data, args
|
init_density_weight, init_route_weight, init_congest_weight, data, args
|
||||||
)
|
)
|
||||||
|
elif args.timing_opt:
|
||||||
|
init_pin_weight = torch.ones(data.num_pins, dtype=torch.float32, device=data.device)
|
||||||
|
_, wl_grad_timing = merged_wl_loss_grad_timing(
|
||||||
|
conn_node_pos, init_pin_weight,
|
||||||
|
data.pin_id2node_id, data.pin_rel_cpos,
|
||||||
|
data.node2pin_list, data.node2pin_list_end, data.hyperedge_list, data.hyperedge_list_end,
|
||||||
|
data.net_mask, data.net_weight, data.hpwl_scale, ps.wa_coeff, args.deterministic
|
||||||
|
)
|
||||||
|
init_density_weight = ((wl_grad.norm(p=1) + wl_grad_timing.norm(p=1)) / density_grad.norm(p=1)).detach()
|
||||||
|
ps.set_init_param(init_density_weight, data)
|
||||||
|
ps.timing_wl_weight = args.timing_init_weight
|
||||||
|
data.gputimer.timing_pin_weight *= ps.timing_wl_weight
|
||||||
|
data.gputimer.beta = 5 * ps.timing_wl_weight
|
||||||
|
# timing mode fluctuation, larger divergence life
|
||||||
|
ps.max_life = 50
|
||||||
|
ps.life = ps.max_life
|
||||||
|
|
||||||
# Nesterove learning rate initialization
|
# Nesterove learning rate initialization
|
||||||
def estimate_initial_learning_rate(obj_and_grad_fn, constraint_fn, x_k, lr):
|
def estimate_initial_learning_rate(obj_and_grad_fn, constraint_fn, x_k, lr):
|
||||||
|
|||||||
@ -29,10 +29,12 @@ class MetricRecorder:
|
|||||||
)
|
)
|
||||||
self[key].append(item)
|
self[key].append(item)
|
||||||
|
|
||||||
def visualize(self, prefix):
|
def visualize(self, prefix, log = False):
|
||||||
for key, value in self:
|
for key, value in self:
|
||||||
x = list(range(len(value)))
|
x = list(range(len(value)))
|
||||||
plt.plot(x, value, label=key)
|
plt.plot(x, value, label=key)
|
||||||
|
if log:
|
||||||
|
plt.yscale("log")
|
||||||
plt.legend()
|
plt.legend()
|
||||||
plt.savefig(prefix + "%s.png" % key)
|
plt.savefig(prefix + "%s.png" % key)
|
||||||
plt.close()
|
plt.close()
|
||||||
@ -148,6 +150,13 @@ class ParamScheduler:
|
|||||||
self.include_macros = args.include_macros
|
self.include_macros = args.include_macros
|
||||||
self.zero_macro_grad = False
|
self.zero_macro_grad = False
|
||||||
|
|
||||||
|
# timing parameter
|
||||||
|
self.timing_sol_recorder = []
|
||||||
|
self.enable_timing = False
|
||||||
|
self.timing_wl_weight = 0
|
||||||
|
self.best_sol_timing: torch.Tensor = None
|
||||||
|
self.best_metric_timing = {"wns": float("inf"), "tns": float("inf"), "iter": 0}
|
||||||
|
|
||||||
def set_init_param(self, init_density_weight, data: PlaceData):
|
def set_init_param(self, init_density_weight, data: PlaceData):
|
||||||
# init_density_weight
|
# init_density_weight
|
||||||
self.init_iter = self.iter
|
self.init_iter = self.iter
|
||||||
@ -230,6 +239,19 @@ class ParamScheduler:
|
|||||||
self.gr_sol_recorder.append((
|
self.gr_sol_recorder.append((
|
||||||
gr_metrics, hpwl, overflow, mov_node_pos.detach().clone()
|
gr_metrics, hpwl, overflow, mov_node_pos.detach().clone()
|
||||||
))
|
))
|
||||||
|
|
||||||
|
def push_timing_sol(self, timing_metrics, hpwl, overflow, mov_node_pos: torch.Tensor):
|
||||||
|
if overflow > self.stop_overflow: # skip overflow solutions
|
||||||
|
return
|
||||||
|
wns_early, tns_early, wns_late, tns_late = timing_metrics
|
||||||
|
wns = -min(wns_early, wns_late, 0)
|
||||||
|
tns = -min(tns_early, tns_late, 0)
|
||||||
|
if wns <= self.best_metric_timing["wns"]:
|
||||||
|
self.best_metric_timing["wns"] = wns
|
||||||
|
self.best_metric_timing["iter"] = self.iter
|
||||||
|
self.best_sol_timing = mov_node_pos.detach().clone()
|
||||||
|
if tns < self.best_metric_timing["tns"]:
|
||||||
|
self.best_metric_timing["tns"] = tns
|
||||||
|
|
||||||
def step(self, hpwl, overflow, node_pos, data):
|
def step(self, hpwl, overflow, node_pos, data):
|
||||||
self.update_precond_weight(data)
|
self.update_precond_weight(data)
|
||||||
@ -582,6 +604,20 @@ class ParamScheduler:
|
|||||||
(best_idx, numOvflNets, gr_wirelength, gr_numVias, gr_numShorts, rc_hor_mean, rc_ver_mean)
|
(best_idx, numOvflNets, gr_wirelength, gr_numVias, gr_numShorts, rc_hor_mean, rc_ver_mean)
|
||||||
)
|
)
|
||||||
return best_sol
|
return best_sol
|
||||||
|
|
||||||
|
def get_best_timing_sol(self):
|
||||||
|
logger = self.__logger__
|
||||||
|
if self.best_sol_timing is not None:
|
||||||
|
wns = self.best_metric_timing["wns"]
|
||||||
|
tns = self.best_metric_timing["tns"]
|
||||||
|
iterationn = self.best_metric_timing["iter"]
|
||||||
|
logger.info(
|
||||||
|
"Find best timing solution: WNS %.4f TNS %.4f at Iter %d" % (wns, tns, iterationn)
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
logger.info("Cannot find best timing solution. Use the last solution.")
|
||||||
|
return None
|
||||||
|
return self.best_sol_timing.data
|
||||||
|
|
||||||
|
|
||||||
def visualize(self, args, logger):
|
def visualize(self, args, logger):
|
||||||
|
|||||||
@ -10,7 +10,7 @@ def get_trunc_node_pos_fn(mov_node_size, data):
|
|||||||
return x
|
return x
|
||||||
return trunc_node_pos_fn
|
return trunc_node_pos_fn
|
||||||
|
|
||||||
def global_placement_main(gpdb, rawdb, ps: ParamScheduler, data: PlaceData, args, logger):
|
def global_placement_main(gpdb, rawdb, ps: ParamScheduler, data: PlaceData, args, logger, params, gputimer=None):
|
||||||
init_density_map = data.init_density_map
|
init_density_map = data.init_density_map
|
||||||
if not args.global_placement:
|
if not args.global_placement:
|
||||||
logger.info("Global placement is switched off. Please make sure the input "
|
logger.info("Global placement is switched off. Please make sure the input "
|
||||||
@ -135,12 +135,36 @@ def global_placement_main(gpdb, rawdb, ps: ParamScheduler, data: PlaceData, args
|
|||||||
terminate_signal = False
|
terminate_signal = False
|
||||||
route_early_terminate_signal = False
|
route_early_terminate_signal = False
|
||||||
log_info = False
|
log_info = False
|
||||||
|
timing_cali_thrs_overflow = 0.5
|
||||||
|
timing_calibration = False
|
||||||
for iteration in range(args.inner_iter):
|
for iteration in range(args.inner_iter):
|
||||||
# optimizer.zero_grad() # zero grad inside obj_and_grad_fn
|
# optimizer.zero_grad() # zero grad inside obj_and_grad_fn
|
||||||
obj = optimizer.step(obj_and_grad_fn)
|
obj = optimizer.step(obj_and_grad_fn)
|
||||||
hpwl, overflow = evaluator_fn(mov_node_pos)
|
hpwl, overflow = evaluator_fn(mov_node_pos)
|
||||||
# update parameters
|
# update parameters
|
||||||
ps.step(hpwl, overflow, mov_node_pos, data)
|
ps.step(hpwl, overflow, mov_node_pos, data)
|
||||||
|
|
||||||
|
# Perform timing-opt.
|
||||||
|
if args.timing_opt and iteration > args.timing_start_iter and iteration % 1 == 0:
|
||||||
|
ps.enable_timing = True
|
||||||
|
node_pos = torch.cat([mov_node_pos[mov_lhs:mov_rhs].clone(), data.node_pos[mov_rhs:]], dim=0)
|
||||||
|
|
||||||
|
if args.calibration and ps.recorder.overflow[-1] < timing_cali_thrs_overflow:
|
||||||
|
gputimer.update_timing_calibrated(node_pos, record=True)
|
||||||
|
timing_cali_thrs_overflow -= args.calibration_step
|
||||||
|
timing_calibration = True
|
||||||
|
elif timing_calibration:
|
||||||
|
gputimer.update_timing_calibrated(node_pos)
|
||||||
|
else:
|
||||||
|
gputimer.update_timing(node_pos)
|
||||||
|
|
||||||
|
timing_metrics = gputimer.report_timing_slack()
|
||||||
|
wns_early, tns_early, wns_late, tns_late = timing_metrics
|
||||||
|
ps.push_timing_sol(timing_metrics, hpwl, overflow, mov_node_pos)
|
||||||
|
|
||||||
|
if iteration % args.timing_freq == 0:
|
||||||
|
gputimer.step(ps, node_pos, data)
|
||||||
|
|
||||||
if ps.need_to_early_stop():
|
if ps.need_to_early_stop():
|
||||||
terminate_signal = True
|
terminate_signal = True
|
||||||
log_info = True
|
log_info = True
|
||||||
@ -317,6 +341,10 @@ def global_placement_main(gpdb, rawdb, ps: ParamScheduler, data: PlaceData, args
|
|||||||
ps.wa_coeff,
|
ps.wa_coeff,
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
if ps.enable_timing:
|
||||||
|
log_str += " | early WNS/TNS: %.4f %.4f (ns) | late WNS/TNS: %.4f %.4f (ns)" % (
|
||||||
|
wns_early, tns_early, wns_late, tns_late
|
||||||
|
)
|
||||||
logger.info(log_str)
|
logger.info(log_str)
|
||||||
if args.draw_placement:
|
if args.draw_placement:
|
||||||
info = (iteration, hpwl, data.design_name)
|
info = (iteration, hpwl, data.design_name)
|
||||||
@ -365,6 +393,10 @@ def global_placement_main(gpdb, rawdb, ps: ParamScheduler, data: PlaceData, args
|
|||||||
ps.push_gr_sol(gr_metrics, hpwl, overflow, mov_node_pos)
|
ps.push_gr_sol(gr_metrics, hpwl, overflow, mov_node_pos)
|
||||||
best_sol_gr = ps.get_best_gr_sol()
|
best_sol_gr = ps.get_best_gr_sol()
|
||||||
mov_node_pos[mov_lhs:mov_rhs].data.copy_(best_sol_gr[mov_lhs:mov_rhs])
|
mov_node_pos[mov_lhs:mov_rhs].data.copy_(best_sol_gr[mov_lhs:mov_rhs])
|
||||||
|
if ps.enable_timing and not ps.enable_route:
|
||||||
|
best_sol_timing = ps.get_best_timing_sol()
|
||||||
|
if best_sol_timing is not None:
|
||||||
|
mov_node_pos[mov_lhs:mov_rhs].data.copy_(best_sol_timing[mov_lhs:mov_rhs])
|
||||||
|
|
||||||
node_pos = mov_node_pos[mov_lhs:mov_rhs]
|
node_pos = mov_node_pos[mov_lhs:mov_rhs]
|
||||||
node_pos = torch.cat([node_pos, data.node_pos[mov_rhs:]], dim=0)
|
node_pos = torch.cat([node_pos, data.node_pos[mov_rhs:]], dim=0)
|
||||||
@ -423,8 +455,11 @@ def run_placement_main_nesterov(args, logger):
|
|||||||
|
|
||||||
# global placement
|
# global placement
|
||||||
node_pos, iteration, gp_hpwl, overflow, gp_time, gp_per_iter = global_placement_main(
|
node_pos, iteration, gp_hpwl, overflow, gp_time, gp_per_iter = global_placement_main(
|
||||||
gpdb, rawdb, ps, data, args, logger
|
gpdb, rawdb, ps, data, args, logger, params, gputimer
|
||||||
)
|
)
|
||||||
|
if args.timing_opt:
|
||||||
|
wns_early_gp, tns_early_gp, wns_late_gp, tns_late_gp = timing_eval_func(node_pos)
|
||||||
|
|
||||||
# detail placement
|
# detail placement
|
||||||
node_pos, dp_hpwl, top5overflow, lg_time, dp_time = detail_placement_main(
|
node_pos, dp_hpwl, top5overflow, lg_time, dp_time = detail_placement_main(
|
||||||
node_pos, gpdb, rawdb, ps, data, args, logger
|
node_pos, gpdb, rawdb, ps, data, args, logger
|
||||||
|
|||||||
43
thirdparty/flute_mp/CMakeLists.txt
vendored
Normal file
43
thirdparty/flute_mp/CMakeLists.txt
vendored
Normal file
@ -0,0 +1,43 @@
|
|||||||
|
# cmake_minimum_required(VERSION 2.8.12)
|
||||||
|
|
||||||
|
# set(CMAKE_CXX_STANDARD 17)
|
||||||
|
|
||||||
|
# set(FLUTE_INCLUDE_DIR "${CMAKE_CURRENT_LIST_DIR}" CACHE INTERNAL "Directory where flute.h is located")
|
||||||
|
|
||||||
|
# set(CMAKE_CXX_STANDARD_REQUIRED ON)
|
||||||
|
|
||||||
|
# project(flute)
|
||||||
|
# if(NOT CMAKE_BUILD_TYPE)
|
||||||
|
# set(CMAKE_BUILD_TYPE "Release" CACHE STRING
|
||||||
|
# "Choose the type of build, options are: Debug Release."
|
||||||
|
# FORCE)
|
||||||
|
# endif(NOT CMAKE_BUILD_TYPE)
|
||||||
|
|
||||||
|
# set(SOURCES ${CMAKE_CURRENT_SOURCE_DIR}/flute.cpp)
|
||||||
|
|
||||||
|
# include_directories("${CMAKE_CURRENT_SOURCE_DIR}")
|
||||||
|
# set(CMAKE_POSITION_INDEPENDENT_CODE ON)
|
||||||
|
# add_library(${PROJECT_NAME} STATIC ${SOURCES})
|
||||||
|
|
||||||
|
# install(TARGETS ${PROJECT_NAME} DESTINATION thirdparty/${PROJECT_NAME})
|
||||||
|
# install(DIRECTORY lut.ICCAD2015 DESTINATION thirdparty/${PROJECT_NAME})
|
||||||
|
|
||||||
|
|
||||||
|
CMAKE_MINIMUM_REQUIRED (VERSION 3.1)
|
||||||
|
PROJECT(flute_mp)
|
||||||
|
|
||||||
|
SET(CMAKE_CXX_STANDARD 17)
|
||||||
|
|
||||||
|
set(FLUTE_MP_INCLUDE_DIR "${CMAKE_CURRENT_LIST_DIR}" CACHE INTERNAL "Directory where flute.h is located")
|
||||||
|
|
||||||
|
FILE(GLOB_RECURSE SRC_FILES_FLUTE_MP ${CMAKE_CURRENT_SOURCE_DIR}/*.cpp)
|
||||||
|
|
||||||
|
ADD_LIBRARY(flute_mp
|
||||||
|
SHARED
|
||||||
|
${SRC_FILES_FLUTE_MP}
|
||||||
|
)
|
||||||
|
|
||||||
|
TARGET_INCLUDE_DIRECTORIES(flute_mp PUBLIC ${FLUTE_MP_INCLUDE_DIR})
|
||||||
|
TARGET_COMPILE_OPTIONS(flute_mp PRIVATE -fPIC)
|
||||||
|
|
||||||
|
INSTALL(TARGETS flute_mp DESTINATION ${XPLACE_LIB_DIR})
|
||||||
50
thirdparty/flute_mp/LICENSE
vendored
Normal file
50
thirdparty/flute_mp/LICENSE
vendored
Normal file
@ -0,0 +1,50 @@
|
|||||||
|
READ THIS LICENSE AGREEMENT CAREFULLY BEFORE USING THIS PRODUCT. BY USING
|
||||||
|
THIS PRODUCT YOU INDICATE YOUR ACCEPTANCE OF THE TERMS OF THE FOLLOWING
|
||||||
|
AGREEMENT. THESE TERMS APPLY TO YOU AND ANY SUBSEQUENT LICENSEE OF THIS
|
||||||
|
PRODUCT.
|
||||||
|
|
||||||
|
License Agreement for FLUTE
|
||||||
|
|
||||||
|
Copyright (c) 2004 by Dr. Chris C. N. Chu
|
||||||
|
All rights reserved
|
||||||
|
|
||||||
|
ATTRIBUTION ASSURANCE LICENSE (adapted from the original BSD license)
|
||||||
|
Redistribution and use in source and binary forms, with or without
|
||||||
|
modification, are permitted provided that the conditions below are
|
||||||
|
met. These conditions require a modest attribution to Dr. Chris C. N. Chu
|
||||||
|
(the "Author").
|
||||||
|
|
||||||
|
1. Redistributions of the source code, with or without modification (the
|
||||||
|
"Code"), must be accompanied by any documentation and, each time
|
||||||
|
the resulting executable program or a program dependent thereon is
|
||||||
|
launched, a prominent display (e.g., splash screen or banner text) of
|
||||||
|
the Author's attribution information, which includes:
|
||||||
|
(a) Dr. Chris C. N. Chu ("AUTHOR"),
|
||||||
|
(b) Iowa State University ("PROFESSIONAL IDENTIFICATION"), and
|
||||||
|
(c) http://class.ee.iastate.edu/cnchu/ ("URL").
|
||||||
|
|
||||||
|
2. Users who intend to use the Code for commercial purposes will notify
|
||||||
|
Author prior to such commercial use.
|
||||||
|
|
||||||
|
3. Neither the name nor any trademark of the Author may be used to
|
||||||
|
endorse or promote products derived from this software without
|
||||||
|
specific prior written permission.
|
||||||
|
|
||||||
|
4. Users are entirely responsible, to the exclusion of the Author and any
|
||||||
|
other persons, for compliance with (1) regulations set by owners or
|
||||||
|
administrators of employed equipment, (2) licensing terms of any other
|
||||||
|
software, and (3) local, national, and international regulations
|
||||||
|
regarding use, including those regarding import, export, and use of
|
||||||
|
encryption software.
|
||||||
|
|
||||||
|
THIS FREE SOFTWARE IS PROVIDED BY THE AUTHOR "AS IS" AND ANY EXPRESS OR
|
||||||
|
IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES
|
||||||
|
OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED.
|
||||||
|
IN NO EVENT SHALL THE AUTHOR OR ANY CONTRIBUTOR BE LIABLE FOR ANY DIRECT,
|
||||||
|
INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
|
||||||
|
(INCLUDING, BUT NOT LIMITED TO, EFFECTS OF UNAUTHORIZED OR MALICIOUS
|
||||||
|
NETWORK ACCESS; PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
|
||||||
|
DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
|
||||||
|
THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||||
|
(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF
|
||||||
|
THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||||
1378
thirdparty/flute_mp/flute.cpp
vendored
Normal file
1378
thirdparty/flute_mp/flute.cpp
vendored
Normal file
File diff suppressed because it is too large
Load Diff
73
thirdparty/flute_mp/flute.hpp
vendored
Normal file
73
thirdparty/flute_mp/flute.hpp
vendored
Normal file
@ -0,0 +1,73 @@
|
|||||||
|
#ifndef FLUTE_HPP_
|
||||||
|
#define FLUTE_HPP_
|
||||||
|
|
||||||
|
#include <algorithm>
|
||||||
|
#include <climits>
|
||||||
|
#include <cmath>
|
||||||
|
#include <cstdio>
|
||||||
|
#include <cstdlib>
|
||||||
|
|
||||||
|
namespace Flute {
|
||||||
|
|
||||||
|
#define DEGREE 9 // LUT will be used when d <= DEGREE, DEGREE <= 9
|
||||||
|
#define FLUTEROUTING 1 // 1 to construct routing, 0 to estimate WL only
|
||||||
|
#define REMOVE_DUPLICATE_PIN 0 // Remove dup. pin for flute_wl() & flute()
|
||||||
|
#define ACCURACY 8 // Default accuracy is 3
|
||||||
|
#define FLUTE_ACCURACY 8
|
||||||
|
// #define MAXD 2008840 // max. degree of a net that can be handled
|
||||||
|
#define MAXD 1000 // max. degree of a net that can be handled
|
||||||
|
|
||||||
|
using DType = int;
|
||||||
|
using DTYPE = int;
|
||||||
|
|
||||||
|
// TODO:
|
||||||
|
// 1. replace macros with constexpr variables
|
||||||
|
// 2. construct a param struct
|
||||||
|
// 3. replace all malloc-free contents
|
||||||
|
// 4. clear inconsistent comments
|
||||||
|
|
||||||
|
struct Branch {
|
||||||
|
DType x, y; // starting point of the branch
|
||||||
|
int n; // index of neighbor
|
||||||
|
};
|
||||||
|
|
||||||
|
struct Tree {
|
||||||
|
int deg; // degree
|
||||||
|
DType length; // total wirelength
|
||||||
|
Branch* branch; // array of tree branches
|
||||||
|
};
|
||||||
|
|
||||||
|
// Major functions
|
||||||
|
void readLUT(const char* powv, const char* post);
|
||||||
|
DType flute_wl(int d, DType* x, DType* y, int acc);
|
||||||
|
Tree flute(int d, DType* x, DType* y, int acc);
|
||||||
|
DType wirelength(Tree t);
|
||||||
|
void printtree(Tree t);
|
||||||
|
|
||||||
|
// Other useful functions
|
||||||
|
DType flutes_wl_LD(int d, DType* xs, DType* ys, int* s);
|
||||||
|
DType flutes_wl_MD(int d, DType* xs, DType* ys, int* s, int acc);
|
||||||
|
DType flutes_wl_RDP(int d, DType* xs, DType* ys, int* s, int acc);
|
||||||
|
Tree flutes_LD(int d, DType* xs, DType* ys, int* s);
|
||||||
|
Tree flutes_MD(int d, DType* xs, DType* ys, int* s, int acc);
|
||||||
|
Tree flutes_RDP(int d, DType* xs, DType* ys, int* s, int acc);
|
||||||
|
|
||||||
|
#if REMOVE_DUPLICATE_PIN == 1
|
||||||
|
#define flutes_wl(d, xs, ys, s, acc) flutes_wl_RDP(d, xs, ys, s, acc)
|
||||||
|
#define flutes(d, xs, ys, s, acc) flutes_RDP(d, xs, ys, s, acc)
|
||||||
|
#else
|
||||||
|
#define flutes_wl(d, xs, ys, s, acc) flutes_wl_ALLD(d, xs, ys, s, acc)
|
||||||
|
#define flutes(d, xs, ys, s, acc) flutes_ALLD(d, xs, ys, s, acc)
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#define flutes_wl_ALLD(d, xs, ys, s, acc) flutes_wl_LMD(d, xs, ys, s, acc)
|
||||||
|
#define flutes_ALLD(d, xs, ys, s, acc) flutes_LMD(d, xs, ys, s, acc)
|
||||||
|
|
||||||
|
#define flutes_wl_LMD(d, xs, ys, s, acc) \
|
||||||
|
(d <= DEGREE ? flutes_wl_LD(d, xs, ys, s) : flutes_wl_MD(d, xs, ys, s, acc))
|
||||||
|
#define flutes_LMD(d, xs, ys, s, acc) \
|
||||||
|
(d <= DEGREE ? flutes_LD(d, xs, ys, s) : flutes_MD(d, xs, ys, s, acc))
|
||||||
|
|
||||||
|
} // namespace flute
|
||||||
|
|
||||||
|
#endif // FLUTE_HPP_
|
||||||
489580
thirdparty/flute_mp/lut.ICCAD2015/POST9.dat
vendored
Normal file
489580
thirdparty/flute_mp/lut.ICCAD2015/POST9.dat
vendored
Normal file
File diff suppressed because it is too large
Load Diff
515760
thirdparty/flute_mp/lut.ICCAD2015/POWV9.dat
vendored
Normal file
515760
thirdparty/flute_mp/lut.ICCAD2015/POWV9.dat
vendored
Normal file
File diff suppressed because it is too large
Load Diff
@ -150,6 +150,10 @@ class IOParser(object):
|
|||||||
die_info = torch.tensor([dieLX, dieHX, dieLY, dieHY]).float()
|
die_info = torch.tensor([dieLX, dieHX, dieLY, dieHY]).float()
|
||||||
# die_shift = torch.tensor([dieLX, dieLY])
|
# die_shift = torch.tensor([dieLX, dieLY])
|
||||||
# die_scale = torch.tensor([dieHX - dieLX, dieHY - dieLY])
|
# die_scale = torch.tensor([dieHX - dieLX, dieHY - dieLY])
|
||||||
|
net_names = gpdb.net_names()
|
||||||
|
pin_names = gpdb.pin_names()
|
||||||
|
node_names = gpdb.node_names()
|
||||||
|
microns = gpdb.microns()
|
||||||
|
|
||||||
siteWidth = gpdb.siteWidth()
|
siteWidth = gpdb.siteWidth()
|
||||||
siteHeight = gpdb.siteHeight()
|
siteHeight = gpdb.siteHeight()
|
||||||
@ -202,6 +206,10 @@ class IOParser(object):
|
|||||||
design_info = {
|
design_info = {
|
||||||
"benchmark": self.params["benchmark"],
|
"benchmark": self.params["benchmark"],
|
||||||
"dataset_path": self.params,
|
"dataset_path": self.params,
|
||||||
|
"node_names": node_names,
|
||||||
|
"net_names": net_names,
|
||||||
|
"pin_names": pin_names,
|
||||||
|
"microns": microns,
|
||||||
"node_type_indices": node_type_indices,
|
"node_type_indices": node_type_indices,
|
||||||
"node_id2node_name": node_id2node_name,
|
"node_id2node_name": node_id2node_name,
|
||||||
"node_id2celltype_name": node_id2celltype_name,
|
"node_id2celltype_name": node_id2celltype_name,
|
||||||
|
|||||||
@ -61,6 +61,13 @@ def setup_design_args(args):
|
|||||||
elif args.design_name in ["bigblue3", "bigblue4"]:
|
elif args.design_name in ["bigblue3", "bigblue4"]:
|
||||||
args.num_bin_x = args.num_bin_y = 2048
|
args.num_bin_x = args.num_bin_y = 2048
|
||||||
args.target_density = 1.0
|
args.target_density = 1.0
|
||||||
|
elif args.design_name in ["superblue1", "superblue3", "superblue4", "superblue5", "superblue16", "superblue18"]:
|
||||||
|
args.num_bin_x = args.num_bin_y = 512
|
||||||
|
args.target_density = 1.0
|
||||||
|
elif args.design_name in ["superblue7", "superblue10"]:
|
||||||
|
args.num_bin_x = args.num_bin_y = 1024
|
||||||
|
args.target_density = 1.0
|
||||||
|
args.start_iter = 150
|
||||||
elif args.design_name in ["adaptec5"]:
|
elif args.design_name in ["adaptec5"]:
|
||||||
args.target_density = 0.5
|
args.target_density = 0.5
|
||||||
args.num_bin_x = args.num_bin_y = 1024
|
args.num_bin_x = args.num_bin_y = 1024
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user