diff --git a/common/place/placer_static.cc b/common/place/placer_static.cc index f8c01d82..dee5dc70 100644 --- a/common/place/placer_static.cc +++ b/common/place/placer_static.cc @@ -505,13 +505,33 @@ class StaticPlacer } } - RealPair limit_to_reg(Region *reg, RealPair val) + RealPair get_region_gradient(Region *reg, RealPair pos) { + RealPair val(0, 0); if (reg == nullptr) return val; const auto &b = constraint_region_bounds_place[reg->name]; - return RealPair(std::max(std::min(val.x, b.x1), b.x0), - std::max(std::min(val.y, b.y1), b.y0)); + + float cx = (b.x0 + b.x1) / 2.f, w = (b.x1 - b.x0); + float cy = (b.y0 + b.y1) / 2.f, h = (b.y1 - b.y0); + const float thresh = 0.9f / 2.f; + + if (pos.x < (cx - w * thresh)) { + val.x = ((cx - w * thresh) - pos.x); + } else if (pos.x > (cx + w * thresh)) { + val.x = ((cx + w * thresh) - pos.x); + } + if (pos.y < (cy - h * thresh)) { + val.y = ((cy - h * thresh) - pos.y); + } else if (pos.y > (cy + h * thresh)) { + val.y = ((cy + h * thresh) - pos.y); + } + // if ((iter % 20) == 0 && !b.contains(int(pos.x + 0.5f), int(pos.y + 0.5f))) { + // log(" %s (%.1f, %.1f) (%d, %d, %d, %d) gx=%.2f gy=%.2f %s\n", + // ctx->nameOf(reg), pos.x, pos.y, b.x0, b.y0, b.x1, b.y1, + // val.x, val.y, b.contains(int(pos.x + 0.5f), int(pos.y + 0.5f)) ? "" : "***"); + // } + return val; } void init_cells() @@ -1045,7 +1065,7 @@ class StaticPlacer } // Third loop: compute total gradient, and precondition // TODO: ALM as well as simple penalty - for (auto &cell : mcells) { + for (int idx = 0; idx < int(mcells.size()); idx++) { #if 0 if (!cell.is_spacer) { printf("%d (%f, %f) wirelen_grad: (%f,%f) density_grad: (%f,%f)\n", iter, cell.ref_pos.x, @@ -1054,13 +1074,24 @@ class StaticPlacer } #endif // Preconditioner from replace for now - + auto &cell = mcells.at(idx); float precond = std::max(1.0f, float(cell.pin_count) + dens_penalty[cell.group] * cell.rect.area()); + + // Extra gradient to pull cells towards their region constraint bounds + RealPair region_constr_grad(0.f, 0.f); + if (idx < int(ccells.size()) && ccells.at(idx).base_cell->region != nullptr) { + region_constr_grad = + get_region_gradient(ccells.at(idx).base_cell->region, ref ? cell.ref_pos : cell.pos); + } + if (ref) { - cell.ref_total_grad = - ((cell.ref_wl_grad * -1) - cell.ref_dens_grad * dens_penalty[cell.group]) / precond; + cell.ref_total_grad = ((cell.ref_wl_grad * -1) - + cell.ref_dens_grad * dens_penalty[cell.group] - region_constr_grad) / + precond; } else { - cell.total_grad = ((cell.wl_grad * -1) - cell.dens_grad * dens_penalty[cell.group]) / precond; + cell.total_grad = + ((cell.wl_grad * -1) - cell.dens_grad * dens_penalty[cell.group] - region_constr_grad) / + precond; } } } @@ -1262,10 +1293,6 @@ class StaticPlacer cell.pos = clamp_loc(cell.ref_pos - cell.ref_total_grad * steplen); // compute reference position cell.ref_pos = clamp_loc(cell.pos + (cell.pos - cell.last_pos) * ((nesterov_a - 1) / a_next)); - if (idx < int(ccells.size()) && ccells.at(idx).base_cell->region != nullptr) { - cell.pos = limit_to_reg(ccells.at(idx).base_cell->region, cell.pos); - cell.ref_pos = limit_to_reg(ccells.at(idx).base_cell->region, cell.ref_pos); - } } nesterov_a = a_next; update_chains(); diff --git a/himbaechel/uarch/gatemate/CMakeLists.txt b/himbaechel/uarch/gatemate/CMakeLists.txt index 5a1d2e5b..fb4dd556 100644 --- a/himbaechel/uarch/gatemate/CMakeLists.txt +++ b/himbaechel/uarch/gatemate/CMakeLists.txt @@ -19,6 +19,7 @@ set(SOURCES pack_mult.cc pack_serdes.cc pack.h + partition.cc pll.cc route_clock.cc route_mult.cc diff --git a/himbaechel/uarch/gatemate/gatemate.cc b/himbaechel/uarch/gatemate/gatemate.cc index 6c08e76f..3918eb67 100644 --- a/himbaechel/uarch/gatemate/gatemate.cc +++ b/himbaechel/uarch/gatemate/gatemate.cc @@ -49,6 +49,7 @@ po::options_description GateMateImpl::getUArchOptions() specific.add_options()("clk-cp", "use CP lines for CLK and EN"); specific.add_options()("no-cpe-cp", "do not use CP lines pass through CPE"); specific.add_options()("no-bridges", "do not use CPE in bridge mode"); + specific.add_options()("auto-part", "use automatic hypergraph-based partitioning for multi-die designs"); return specific; } @@ -121,6 +122,7 @@ void GateMateImpl::init_database(Arch *arch) use_cp_for_clk = args.options.count("clk-cp") == 1; use_cp_for_cpe = args.options.count("no-cpe-cp") == 0; use_bridges = args.options.count("no-bridges") == 0; + auto_part = args.options.count("auto-part") == 1; } void GateMateImpl::init(Context *ctx) diff --git a/himbaechel/uarch/gatemate/gatemate.h b/himbaechel/uarch/gatemate/gatemate.h index 5fc003f0..585fcbe2 100644 --- a/himbaechel/uarch/gatemate/gatemate.h +++ b/himbaechel/uarch/gatemate/gatemate.h @@ -142,6 +142,8 @@ struct GateMateImpl : HimbaechelAPI void get_setuphold_from_tmg_db(IdString id_setup, IdString id_hold, DelayPair &setup, DelayPair &hold) const; void get_setuphold_from_tmg_db(IdString id_setuphold, DelayPair &setup, DelayPair &hold) const; + void partition_design(); + struct GateMateCellInfo { // slice info @@ -177,6 +179,7 @@ struct GateMateImpl : HimbaechelAPI bool use_cp_for_clk; bool use_cp_for_cpe; bool use_bridges; + bool auto_part; }; NEXTPNR_NAMESPACE_END diff --git a/himbaechel/uarch/gatemate/pack.cc b/himbaechel/uarch/gatemate/pack.cc index 11167ae2..3d9fd0a6 100644 --- a/himbaechel/uarch/gatemate/pack.cc +++ b/himbaechel/uarch/gatemate/pack.cc @@ -651,6 +651,11 @@ void GateMateImpl::pack() packer.copy_clocks(); packer.remove_constants(); packer.remove_double_constrained(); + + if (auto_part) { + partition_design(); + } + if (forced_die != IdString()) { for (auto &cell : ctx->cells) { if (cell.second->belStrength != PlaceStrength::STRENGTH_FIXED) diff --git a/himbaechel/uarch/gatemate/partition.cc b/himbaechel/uarch/gatemate/partition.cc new file mode 100644 index 00000000..2d1d29db --- /dev/null +++ b/himbaechel/uarch/gatemate/partition.cc @@ -0,0 +1,851 @@ +/* + * nextpnr -- Next Generation Place and Route + * + * Copyright (C) 2024 The Project Peppercorn Authors. + * + * Permission to use, copy, modify, and/or distribute this software for any + * purpose with or without fee is hereby granted, provided that the above + * copyright notice and this permission notice appear in all copies. + * + * THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES + * WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF + * MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR + * ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES + * WHATSOEVER RESULTING FROM LOSS OF USE, DATA OR PROFITS, WHETHER IN AN + * ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING OUT OF + * OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE. + * + */ + +#include +#include + +#include "gatemate.h" + +#define HIMBAECHEL_CONSTIDS "uarch/gatemate/constids.inc" +#include "himbaechel_constids.h" + +NEXTPNR_NAMESPACE_BEGIN + +namespace { + +struct HypergraphEdge +{ + std::vector nodes; + int weight = 1; +}; + +struct HypergraphNode +{ + std::vector edges; + int partition = -1; + float area = 1; + bool fixed = false; +}; + +struct Hypergraph +{ + std::vector edges; + std::vector nodes; + void dump(std::ostream &out) const; + void read(std::istream &in); +}; + +struct PartitionConstraint +{ + double min_nodes, max_nodes; +}; + +struct gain_store +{ + gain_store(int num_nodes) { node2element.resize(num_nodes); } + + struct gain_element + { + int node; + int fwd_ptr = -1; + int bwd_ptr = -1; + }; + struct gain_pointer + { + int gain; + int elem_ptr = -1; + }; + struct gain_bucket + { + int head_ptr = -1; + int tail_ptr = -1; + }; + std::map buckets; + + void remove_elem(int gain, int elem_ptr) + { + auto &bucket = buckets.at(gain); + auto &elem = elements.at(elem_ptr); + if (elem.fwd_ptr == -1) { + bucket.tail_ptr = elem.bwd_ptr; + } else { + elements.at(elem.fwd_ptr).bwd_ptr = elem.bwd_ptr; + } + if (elem.bwd_ptr == -1) { + bucket.head_ptr = elem.fwd_ptr; + } else { + elements.at(elem.bwd_ptr).fwd_ptr = elem.fwd_ptr; + } + if (bucket.tail_ptr == -1) { + NPNR_ASSERT(bucket.head_ptr == -1); + buckets.erase(gain); + } + elem.fwd_ptr = next_free_elem; + elem.bwd_ptr = -1; + next_free_elem = elem_ptr; + + node2element.at(elem.node).elem_ptr = -1; + elem.node = -1; + }; + + void add_elem(int gain, int node, bool at_start = true) + { + int elem_ptr = next_free_elem; + if (elem_ptr == int(elements.size())) { + elements.emplace_back(); + ++next_free_elem; + } else { + next_free_elem = elements.at(elem_ptr).fwd_ptr; + } + auto &elem = elements.at(elem_ptr); + auto &bucket = buckets[gain]; + elem.node = node; + if (at_start) { + elem.fwd_ptr = bucket.head_ptr; + elem.bwd_ptr = -1; + if (bucket.head_ptr != -1) + elements.at(bucket.head_ptr).bwd_ptr = elem_ptr; + if (bucket.tail_ptr == -1) + bucket.tail_ptr = elem_ptr; + bucket.head_ptr = elem_ptr; + } else { + elem.bwd_ptr = bucket.tail_ptr; + elem.fwd_ptr = -1; + if (bucket.tail_ptr != -1) + elements.at(bucket.tail_ptr).fwd_ptr = elem_ptr; + if (bucket.head_ptr == -1) + bucket.head_ptr = elem_ptr; + bucket.tail_ptr = elem_ptr; + } + node2element.at(node).gain = gain; + node2element.at(node).elem_ptr = elem_ptr; + }; + + std::pair pop_node(bool lifo = true) + { + // Pop the node with the highest gain + NPNR_ASSERT(!buckets.empty()); + auto highest = buckets.rbegin(); + int elem_ptr = lifo ? highest->second.head_ptr : highest->second.tail_ptr; + int node = elements.at(elem_ptr).node; + int gain = highest->first; + remove_elem(gain, elem_ptr); + return {node, gain}; + } + + bool has_moves() { return !buckets.empty(); } + + int node_gain(int node) const + { + auto &c = node2element.at(node); + NPNR_ASSERT(c.elem_ptr != -1); + return c.gain; + } + + void update_node(int node, int delta) + { + auto &c = node2element.at(node); + int new_gain = c.gain + delta; + NPNR_ASSERT(c.elem_ptr != -1); + remove_elem(c.gain, c.elem_ptr); + add_elem(new_gain, node); + } + + // The linked list storage (so we don't have to malloc all the time) + std::vector elements; + int next_free_elem = 0; + // node to element map (indexed by node index) + std::vector node2element; +}; + +struct FMPartitioner +{ + + FMPartitioner(Context *ctx, Hypergraph &g, const std::vector &partitions) + : ctx(ctx), g(g), partitions(partitions), gains(g.nodes.size()) + { + init(); + } + + Context *ctx; + Hypergraph &g; + std::vector partitions; + gain_store gains; + std::vector locked; + std::vector part_area; + + void init() + { + locked.resize(g.nodes.size(), false); + part_area.resize(partitions.size(), 0); + gain_store empty(g.nodes.size()); + std::swap(gains, empty); + for (int i = 0; i < int(g.nodes.size()); i++) { + auto &c = g.nodes.at(i); + if (c.fixed) { + locked.at(i) = true; + NPNR_ASSERT(c.partition != -1); + part_area.at(c.partition) += c.area; + } + } + } + + inline bool above_eq_target() + { + for (int i = 0; i < int(partitions.size()); i++) + if (part_area.at(i) < partitions.at(i).min_nodes) + return false; + return true; + } + + inline double total_area_slack(bool max_area) + { + double slack = 0; + for (int i = 0; i < int(partitions.size()); i++) { + slack += part_area_slack(i, max_area); + } + return slack; + } + + inline double part_area_slack(int part, bool max_area) + { + if (max_area) + return std::max(0, partitions.at(part).max_nodes - part_area.at(part)); + else + return std::max(0, partitions.at(part).min_nodes - part_area.at(part)); + } + + void random_part() + { + /* + Perform a random initial partitioning + Probabilities are weighted based on relative occupancy of paritions relative to occupancy constraints + */ + std::fill(part_area.begin(), part_area.end(), 0); + for (int i = 0; i < int(g.nodes.size()); i++) { + auto &c = g.nodes.at(i); + if (c.fixed) + part_area.at(c.partition) += c.area; + } + std::vector> sorted_by_area; + + for (int i = 0; i < int(g.nodes.size()); i++) { + auto &c = g.nodes.at(i); + if (c.fixed) + continue; + sorted_by_area.emplace_back(i, c.area); + } + std::sort(sorted_by_area.begin(), sorted_by_area.end(), + [&](std::pair a, std::pair b) { return a.second > b.second; }); + for (auto item : sorted_by_area) { + auto &c = g.nodes.at(item.first); + double r = ctx->rng() / double(0x3fffffff); + double p = 0; + c.partition = int(partitions.size()) - 1; + + bool use_max_area = above_eq_target(); + double total_slack = total_area_slack(use_max_area); + + for (int i = 0; i < int(partitions.size()) - 1; i++) { + p += double(part_area_slack(i, use_max_area)) / std::max(1, total_slack); + if (r <= p) { + c.partition = i; + break; + } + } + + part_area.at(c.partition) += c.area; + } + } + + int compute_cost() + { + int cost = 0; + std::vector seen_parts(partitions.size(), false); + for (int i = 0; i < int(g.edges.size()); i++) { + std::fill(seen_parts.begin(), seen_parts.end(), false); + auto &e = g.edges.at(i); + for (int n : e.nodes) + seen_parts.at(g.nodes.at(n).partition) = true; + int num_parts = 0; + for (bool b : seen_parts) + if (b) + ++num_parts; + if (num_parts == 0) + continue; + cost += e.weight * (num_parts - 1); + } + return cost; + } + + double total_area() + { + double area = 0; + for (int i = 0; i < int(g.nodes.size()); i++) { + area += g.nodes.at(i).area; + } + return area; + } + + void coarsen(Hypergraph &coarsened, dict> &new2orig) + { + // This is a very basic algorithm for coarsening that probably doesn't give very good results + // Need to find a better one that + dict orig2new; + + double area_sum = 0; + int area_count = 0; + for (int i = 0; i < int(g.nodes.size()); i++) { + auto &n = g.nodes.at(i); + if (n.fixed || n.edges.empty()) + continue; + ++area_count; + area_sum += n.area; + } + + double average_area = area_sum / area_count; + + // Merge nodes + std::vector node_indices(g.nodes.size()); + for (int i = 0; i < int(g.nodes.size()); i++) + node_indices.at(i) = i; + ctx->sorted_shuffle(node_indices); + for (int i : node_indices) { + auto &n = g.nodes.at(i); + if (n.fixed) { + // Locked nodes are never merged + coarsened.nodes.emplace_back(); + auto &n2 = coarsened.nodes.back(); + n2.fixed = n.fixed; + n2.partition = n.partition; + n2.area = n.area; + new2orig[int(coarsened.nodes.size()) - 1].push_back(i); + orig2new[i] = int(coarsened.nodes.size()) - 1; + continue; + } + const int max_count = 2; + const int area_ratio = 3; + if (!n.edges.empty()) { + dict neighbours; + for (int thresh = 20; thresh < 200; thresh *= 1.2) { + for (int merge_edge : n.edges) { + auto &e = g.edges.at(merge_edge); + if (int(e.nodes.size()) > thresh) + continue; + for (int neighbour : e.nodes) { + if (neighbour == i) + continue; + auto &merge_node_data = g.nodes.at(neighbour); + if (merge_node_data.fixed) + continue; + if (orig2new.count(neighbour)) { + if (orig2new.count(i)) + continue; // don't merge two clusters + // Already a cluster + int n2_idx = orig2new.at(neighbour); + if ((coarsened.nodes.at(n2_idx).area + n.area) > (area_ratio * average_area)) + continue; + if (int(new2orig.at(n2_idx).size()) >= (max_count - 1)) + continue; + // Alias to the first node in the cluster + neighbours[new2orig.at(n2_idx).front()] += (1.0f / sqrt(e.nodes.size())); + } else if (orig2new.count(i)) { + int n2_idx = orig2new.at(i); + if ((coarsened.nodes.at(n2_idx).area + merge_node_data.area) > + (area_ratio * average_area)) + continue; + if (int(new2orig.at(n2_idx).size()) >= (max_count - 1)) + continue; + neighbours[neighbour] += (1.0f / sqrt(e.nodes.size())); + } else { + neighbours[neighbour] += (1.0f / sqrt(e.nodes.size())); + } + } + } + if (!neighbours.empty()) + break; + } + if (int(neighbours.size()) > 0) { + auto best_neighbour = + std::max_element(neighbours.begin(), neighbours.end(), + [&](const std::pair &a, const std::pair &b) { + return (a.second < b.second); + }); + int merge_node = best_neighbour->first; + auto &merge_node_data = g.nodes.at(merge_node); + if (orig2new.count(i)) { + int n2_idx = orig2new.at(i); + coarsened.nodes.at(n2_idx).area += merge_node_data.area; + new2orig[n2_idx].push_back(merge_node); + orig2new[merge_node] = n2_idx; + goto merged; + } else if (orig2new.count(merge_node)) { + // Already a cluster + int n2_idx = orig2new.at(merge_node); + coarsened.nodes.at(n2_idx).area += n.area; + new2orig[n2_idx].push_back(i); + orig2new[i] = n2_idx; + goto merged; + } else { + // Create a cluster + coarsened.nodes.emplace_back(); + auto &n2 = coarsened.nodes.back(); + n2.fixed = false; + n2.partition = -1; + n2.area = n.area + merge_node_data.area; + + new2orig[int(coarsened.nodes.size()) - 1].push_back(i); + new2orig[int(coarsened.nodes.size()) - 1].push_back(merge_node); + orig2new[i] = int(coarsened.nodes.size()) - 1; + orig2new[merge_node] = int(coarsened.nodes.size()) - 1; + goto merged; + } + } + } + if (0) { + merged: + continue; + } + // Didn't find anything to merge with + if (!orig2new.count(i)) { + coarsened.nodes.emplace_back(); + auto &n2 = coarsened.nodes.back(); + n2.fixed = n.fixed; + n2.partition = n.partition; + n2.area = n.area; + new2orig[int(coarsened.nodes.size()) - 1].push_back(i); + orig2new[i] = int(coarsened.nodes.size()) - 1; + continue; + } + } + // Reconstruct edges + std::unordered_set seen_nodes; + for (int i = 0; i < int(g.edges.size()); i++) { + auto &e = g.edges.at(i); + if (e.nodes.size() <= 1) + continue; + // Don't create a new edge if it now only connects to one node + if (std::all_of(e.nodes.begin(), e.nodes.end(), [&](int n) { return n == e.nodes.at(0); })) + continue; + + int e2_idx = int(coarsened.edges.size()); + coarsened.edges.emplace_back(); + auto &e2 = coarsened.edges.back(); + e2.weight = e.weight; + + seen_nodes.clear(); + for (auto n : e.nodes) { + int n2 = orig2new.at(n); + if (seen_nodes.count(n2)) + continue; + seen_nodes.insert(n2); + coarsened.nodes.at(n2).edges.push_back(e2_idx); + e2.nodes.push_back(n2); + } + } + } + + void assert_area() + { + std::fill(part_area.begin(), part_area.end(), 0); + for (int i = 0; i < int(g.nodes.size()); i++) { + auto &c = g.nodes.at(i); + part_area.at(c.partition) += c.area; + } + for (int i = 0; i < int(partitions.size()); i++) + NPNR_ASSERT((part_area.at(i) >= partitions.at(i).min_nodes) && + (part_area.at(i) <= partitions.at(i).max_nodes)); + } + + void uncoarsen(const Hypergraph &coarsened, const dict> &new2orig) + { + for (const auto &item : new2orig) { + const auto &new_node = coarsened.nodes.at(item.first); + for (int old_node_idx : item.second) { + auto &old_node = g.nodes.at(old_node_idx); + if (old_node.fixed) + continue; + old_node.partition = new_node.partition; + } + } + std::fill(part_area.begin(), part_area.end(), 0); + for (int i = 0; i < int(g.nodes.size()); i++) { + auto &c = g.nodes.at(i); + part_area.at(c.partition) += c.area; + } + } + + void gain_update(int node, int src_part, int dst_part) + { + // Figure 13 - Pseudo-code for a faster gain update that takes advantage of special cases. + auto &n = g.nodes.at(node); + for (int e_idx : n.edges) { + auto &e = g.edges.at(e_idx); + if (e.nodes.size() == 2) { + for (int n2_idx : e.nodes) { + if (n2_idx == node) + continue; + if (locked.at(n2_idx)) + break; + auto &n2 = g.nodes.at(n2_idx); + if (n2.partition == src_part) + gains.update_node(n2_idx, 2 * e.weight); + else + gains.update_node(n2_idx, -2 * e.weight); + break; + } + continue; + } + if (e.nodes.size() == 1) + continue; + int src_tally = 0; + int dst_tally = 0; + for (int n2_idx : e.nodes) { + auto &n2 = g.nodes.at(n2_idx); + if (n2.partition == src_part) + ++src_tally; + if (n2.partition == dst_part) + ++dst_tally; + } + if (dst_tally == 0) { + // This move is the first node on the edge to enter the dst partition + for (int n2_idx : e.nodes) { + if (n2_idx == node || locked.at(n2_idx)) + continue; + gains.update_node(n2_idx, e.weight); + } + } else if (src_tally == 1) { + // This move is the last node on the edge to leave the src partition + for (int n2_idx : e.nodes) { + if (n2_idx == node || locked.at(n2_idx)) + continue; + gains.update_node(n2_idx, -e.weight); + } + } else { + // None of the special cases apply + for (int n2_idx : e.nodes) { + if (n2_idx == node || locked.at(n2_idx)) + continue; + auto &n2 = g.nodes.at(n2_idx); + if (n2.partition == src_part && src_tally == 2) { + // This other node is the last one left in the src partition + // other than the one being moved + gains.update_node(n2_idx, e.weight); + } + if (n2.partition == dst_part && dst_tally == 1) { + // This other node is the only other one in the dst partition + // other than the one being moved + gains.update_node(n2_idx, -e.weight); + } + } + } + } + } + + void setup_initial_gains() + { + // Setup the starting gains + for (int i = 0; i < int(g.nodes.size()); i++) { + auto &n = g.nodes.at(i); + if (n.fixed) + continue; + int src_part = n.partition; + int dst_part = 1 - n.partition; + int gain = 0; + for (int e_idx : n.edges) { + auto &e = g.edges.at(e_idx); + if (e.nodes.size() == 2) { + // Special-casing for two-element nodes + for (int n2_idx : e.nodes) { + if (n2_idx == i) + continue; + if (locked.at(n2_idx)) + break; + auto &n2 = g.nodes.at(n2_idx); + if (n2.partition == src_part) + gain -= e.weight; // now introducing a split + else + gain += e.weight; // now removing a split + break; + } + continue; + } + + if (e.nodes.size() == 1) + continue; + + int src_tally = 0; + int dst_tally = 0; + for (int n2_idx : e.nodes) { + auto &n2 = g.nodes.at(n2_idx); + if (n2.partition == src_part) + ++src_tally; + if (n2.partition == dst_part) + ++dst_tally; + } + + if (src_tally == 1) { + gain += e.weight; // now removing a split + continue; + } + + if (dst_tally == 0) { + gain -= e.weight; // now introducing a split + continue; + } + } + gains.add_elem(gain, i); + } + } + + void run() + { + std::vector> moves_made; + std::vector> reinsert; + + int score = 0; + int best_score = 0; + int best_score_idx = -1; + + setup_initial_gains(); + + int start_cost = compute_cost(); + + while (true) { + int move_node = -1; + int move_gain = 0; + reinsert.clear(); + // Find a legal move + while (gains.has_moves()) { + auto move = gains.pop_node(); + int n_idx = move.first; + auto &n = g.nodes.at(n_idx); + int src_part = n.partition; + int dst_part = 1 - n.partition; + if (/*(part_area.at(src_part) >= partitions.at(src_part).min_nodes) && */ ( + (part_area.at(src_part) - n.area) < partitions.at(src_part).min_nodes)) + goto fail; + if (/*(part_area.at(dst_part) <= partitions.at(dst_part).max_nodes) && */ ( + (part_area.at(dst_part) + n.area) > partitions.at(dst_part).max_nodes)) + goto fail; + move_node = n_idx; + move_gain = move.second; + break; + fail: + reinsert.push_back(move); + } + if (move_node == -1) + break; + // Re-add the illegal moves we popped + for (auto re : reinsert) + gains.add_elem(re.second, re.first); + + auto &n = g.nodes.at(move_node); + int src_part = n.partition; + int dst_part = 1 - n.partition; + + // Update gains + gain_update(move_node, src_part, dst_part); + + // Update areas + part_area.at(src_part) -= n.area; + part_area.at(dst_part) += n.area; + n.partition = dst_part; + + // Update score + score += move_gain; + if ((best_score_idx == -1) || (score > best_score)) { + best_score_idx = int(moves_made.size()); + best_score = score; + } + + // Add move to list + moves_made.emplace_back(move_node, src_part); + locked.at(move_node) = true; + } + + // Revert moves after the best score + for (int i = best_score_idx + 1; i < int(moves_made.size()); i++) { + auto &mm = moves_made.at(i); + auto &n = g.nodes.at(mm.first); + part_area.at(n.partition) -= n.area; + part_area.at(mm.second) += n.area; + n.partition = mm.second; + } + + if (ctx->verbose) + log_info(" start: %d end: %d incr_gain: %d moves_made: %d\n", start_cost, compute_cost(), best_score, + int(moves_made.size())); + } +}; + +int partition_recursive(Context *ctx, Hypergraph &g, const std::vector &partitions, int level) +{ + int non_fixed_nodes = 0; + for (auto &n : g.nodes) + if (!n.fixed && !n.edges.empty()) + ++non_fixed_nodes; + FMPartitioner fm(ctx, g, partitions); + fm.init(); + if (ctx->verbose) + log_info("enter level=%d, N=%d, A=%f\n", level, non_fixed_nodes, fm.total_area()); + if (non_fixed_nodes <= 20) { + // Final level in the hierarchy + // Initial random partioning as our seed + fm.random_part(); + } else { + // Coarse the hypergraph, partition the coarsened graph and use that result as our seed + Hypergraph coarsened; + dict> new2orig; + fm.coarsen(coarsened, new2orig); + partition_recursive(ctx, coarsened, partitions, level + 1); + fm.uncoarsen(coarsened, new2orig); + } + if (ctx->verbose) { + log_info("re-enter level=%d, N=%d, cost=%d", level, non_fixed_nodes, fm.compute_cost()); + for (int i = 0; i < int(partitions.size()); i++) { + log(", A%d=%f", i, fm.part_area.at(i)); + } + log("\n"); + } + + fm.assert_area(); + // The FM optimisation phase + fm.run(); + for (int i = 0; i < 5; i++) { + FMPartitioner fm(ctx, g, partitions); + fm.init(); + fm.assert_area(); + fm.run(); + } + fm.assert_area(); + + // Status print + + if (ctx->verbose) { + log_info("exit level=%d, N=%d, cost=%d", level, non_fixed_nodes, fm.compute_cost()); + for (int i = 0; i < int(partitions.size()); i++) { + log(", A%d=%f", i, fm.part_area.at(i)); + } + log("\n"); + } + + fm.assert_area(); + return fm.compute_cost(); +} + +float get_cell_area(Context *ctx, const CellInfo *ci) +{ + if (ci->type == id_CC_BRAM_20K) { + return 100.0f; + } else if (ci->type == id_CC_BRAM_40K) { + return 200.0f; + } + return 1.0f; +} + +template void recursive_visit_children(const CellInfo *ci, Tfunc func) { + for (auto child : ci->constr_children) { + func(child); + recursive_visit_children(child, func); + } +} + +} // namespace + +void GateMateImpl::partition_design() +{ + if (dies != 2) { + log_error("Partitioning is currently only supported for the CCGM1A2 device.\n"); + } + log_info("Partitioning design across dies...\n"); + // Build the hypergraph + dict cell2node; + Hypergraph g; + double total_area = 0; + for (auto &cell : ctx->cells) { + CellInfo *ci = cell.second.get(); + if (ci->cluster != ClusterId() && ci->name != ci->cluster) + continue; // not cluster root + + int node_idx = g.nodes.size(); + cell2node[ci->name] = node_idx; + g.nodes.emplace_back(); + auto &n = g.nodes.back(); + n.area = get_cell_area(ctx, ci); + + recursive_visit_children(ci, [&](CellInfo *child) { + n.area += get_cell_area(ctx, child); + cell2node[child->name] = node_idx; + }); + + total_area += n.area; + if (ci->bel != BelId()) { + // constrained bels + auto tile_data = tile_extra_data(ci->bel.tile); + n.fixed = true; + n.partition = tile_data->die; + } + } + // Import edges + for (auto &net : ctx->nets) { + NetInfo *ni = net.second.get(); + if (!ni->driver.cell || ni->driver.cell->type.in(id_CC_BUFG, id_CC_PLL, id_CC_PLL_ADV)) + continue; // undriven or global nets + int edge_idx = int(g.edges.size()); + g.edges.emplace_back(); + auto &e = g.edges.back(); + auto import_port = [&](const PortRef &pr) { + auto node_idx = cell2node.at(pr.cell->name); + e.nodes.push_back(node_idx); + g.nodes.at(node_idx).edges.push_back(edge_idx); + }; + import_port(ni->driver); + for (auto &usr : ni->users) + import_port(usr); + } + // Run the partitioner + std::vector partitions; + for (int i = 0; i < dies; i++) { + // TODO: better partition constraints (35000 for a target per-die utilisation of ~75% max) + partitions.emplace_back(); + partitions.back().min_nodes = 0; + partitions.back().max_nodes = std::max(35000, total_area * 0.6); + } + + int cost = partition_recursive(ctx, g, partitions, 0); + log_info("Hypergraph partitioning complete, final cost: %d\n", cost); + + std::vector partition_area(dies); + for (const auto &c2n : cell2node) { + CellInfo *ci = ctx->cells.at(c2n.first).get(); + if (ci->cluster != ClusterId() && ci->cluster != ci->name) + continue; // not cluster root + auto &n = g.nodes.at(c2n.second); + partition_area.at(n.partition) += n.area; + if (!n.fixed) { + ctx->constrainCellToRegion(c2n.first, index_to_die.at(n.partition)); + } + } + for (int i = 0; i < dies; i++) { + log_info(" die %s area: %f\n", index_to_die.at(i).c_str(ctx), partition_area.at(i)); + } +} + +NEXTPNR_NAMESPACE_END