From b42d7c2d9e51eaa50bc3b90423fdfa3b9e00bd91 Mon Sep 17 00:00:00 2001 From: Geza Lore Date: Mon, 3 Aug 2026 17:00:31 +0100 Subject: [PATCH] Internals: Move critical path propagators to MTaskGraph from Contraction Code movement only prep for further work. No functional change, output should be identical. --- src/V3OrderMTaskContraction.cpp | 198 +----------------------- src/V3OrderMTaskGraph.cpp | 5 +- src/V3OrderMTaskGraph.h | 257 ++++++++++++++++++++++++++++---- 3 files changed, 237 insertions(+), 223 deletions(-) diff --git a/src/V3OrderMTaskContraction.cpp b/src/V3OrderMTaskContraction.cpp index a2897a01b..6d4ea3c8a 100644 --- a/src/V3OrderMTaskContraction.cpp +++ b/src/V3OrderMTaskContraction.cpp @@ -37,7 +37,6 @@ #include #include #include -#include VL_DEFINE_DEBUG_FUNCTIONS; @@ -480,186 +479,6 @@ static void partCheckCriticalPaths(V3Graph& mTaskGraph) { } } -// ###################################################################### -// PropagateCp - -template -class PropagateCp final { - // Propagate increasing critical path (CP) costs through a graph. - // - // Usage: - // * Client increases the cost and/or CP at a node or small set of nodes - // (often a pair in practice, eg. edge contraction.) - // * Client calls PropagateCp::cpHasIncreased() one or more times. - // Each call indicates that the inclusive CP of some "seed" vertex - // has increased to a given value. - // * NOTE: PropagateCp will neither read nor modify the cost - // or CPs at the seed vertices, it only accesses and modifies - // vertices wayward from the seeds. - // * Client calls PropagateCp::go(). Internally, this iteratively - // propagates the new CPs wayward through the graph. - // - - // TYPES - - // We keep pending vertices in a heap during critical path propagation - struct PendingKey final { - LogicMTask* m_mtaskp; // The vertex in the heap - uint64_t m_score; // The score of this entry - void increase(uint64_t score) { - UDEBUGONLY(UASSERT(score >= m_score, "Must increase");); - m_score = score; - } - bool operator<(const PendingKey& other) const { - if (m_score != other.m_score) return m_score < other.m_score; - return *m_mtaskp < *other.m_mtaskp; - } - }; - - using PendingHeap = PairingHeap; - using PendingHeapNode = typename PendingHeap::Node; - - // MEMBERS - PendingHeap m_pendingHeap; // Heap of pending rescores - - // We allocate this many heap nodes at once - static constexpr size_t ALLOC_CHUNK_SIZE = 128; - PendingHeapNode* m_freep = nullptr; // List of free heap nodes - std::vector> m_allocated; // Allocated heap nodes - - const bool m_slowAsserts; // Enable nontrivial asserts - // Used only with slow asserts to check MTasks visited only once - std::unordered_set m_seen; - -public: - // CONSTRUCTORS - explicit PropagateCp(bool slowAsserts) - : m_slowAsserts{slowAsserts} {} - - // METHODS -private: - // Allocate a HeapNode for the given element - PendingHeapNode* allocNode() { - // If no free nodes available, then make some - if (!m_freep) { - // Allocate in chunks for efficiency - m_allocated.emplace_back(new PendingHeapNode[ALLOC_CHUNK_SIZE]); - // Set up free list pointer - m_freep = m_allocated.back().get(); - // Set up free list chain - for (size_t i = 1; i < ALLOC_CHUNK_SIZE; ++i) { - m_freep[i - 1].m_next.m_ptr = &m_freep[i]; - } - // Clear the next pointer of the last entry - m_freep[ALLOC_CHUNK_SIZE - 1].m_next.m_ptr = nullptr; - } - // Free nodes are available, pick up the first one - PendingHeapNode* const resultp = m_freep; - m_freep = resultp->m_next.m_ptr; - resultp->m_next.m_ptr = nullptr; - return resultp; - } - - // Release a heap node (make it available for future allocation) - void freeNode(PendingHeapNode* nodep) { - // Re-use the existing link pointers and simply prepend it to the free list - nodep->m_next.m_ptr = m_freep; - m_freep = nodep; - } - -public: - void cpHasIncreased(V3GraphVertex* vxp, uint64_t newInclusiveCp) { - constexpr GraphWay way{N_Way}; - constexpr GraphWay inv{way.invert()}; - - // For *vxp, whose CP-inclusive has just increased to - // newInclusiveCp, iterate to all wayward nodes, update the edges - // of each, and add each to m_pending if its overall CP has grown. - for (V3GraphEdge& graphEdge : vxp->edges()) { - MTaskEdge& edge = static_cast(graphEdge); - - LogicMTask* const relativep = edge.furtherMTaskp(); - EdgeHeap::Node& edgeHeapNode = edge.m_edgeHeapNode[inv]; - if (newInclusiveCp > edgeHeapNode.key().m_score) { - relativep->m_edgeHeap[inv].increaseKey(&edgeHeapNode, newInclusiveCp); - } - - const uint64_t critPathCost = relativep->critPathCost(way); - - if (critPathCost >= newInclusiveCp) continue; - - // relativep's critPathCost() is out of step with its longest !wayward edge. - // Schedule that to be resolved. - const uint64_t newVal = newInclusiveCp - critPathCost; - - void*& pendingNodepRef = relativep->m_propagateHeapNodep; - if (PendingHeapNode* const nodep = static_cast(pendingNodepRef)) { - // Already in heap. Increase score if needed. - if (newVal > nodep->key().m_score) m_pendingHeap.increaseKey(nodep, newVal); - continue; - } - - // Add to heap - PendingHeapNode* const nodep = allocNode(); - pendingNodepRef = nodep; - m_pendingHeap.insert(nodep, {relativep, newVal}); - } - } - - void go() { - constexpr GraphWay way{N_Way}; - constexpr GraphWay inv{way.invert()}; - - // m_pending maps each pending vertex to the amount that it wayward - // CP will grow. - // - // We can iterate over the pending set in reverse order, always - // choosing the nodes with the largest pending CP-growth. - // - // The intuition is: if the original seed node had its CP grow by - // 50, the most any wayward node can possibly grow is also 50. So - // for anything pending to grow by 50, we know we can process it - // once and we won't have to grow its CP again on the current pass. - // After we're done with all the grow-by-50s, nothing else will - // grow by 50 again on the current pass, and we can process the - // grow-by-49s and we know we'll only have to process each one - // once. And so on. - // - // This generalizes to multiple seed nodes also. - while (!m_pendingHeap.empty()) { - // Pop max element from heap - PendingHeapNode* const maxp = m_pendingHeap.max(); - m_pendingHeap.remove(maxp); - // Pick up values - LogicMTask* const mtaskp = maxp->key().m_mtaskp; - const uint64_t cpGrowBy = maxp->key().m_score; - // Free the heap node, we are done with it - freeNode(maxp); - mtaskp->m_propagateHeapNodep = nullptr; - // Update the critPathCost of mtaskp, that was out-of-date with respect to its edges - const uint64_t startCp = mtaskp->critPathCost(way); - const uint64_t newCp = startCp + cpGrowBy; - if (VL_UNLIKELY(m_slowAsserts)) { - // Check that CP matches that of the longest edge wayward of vxp. - const uint64_t edgeCp = mtaskp->m_edgeHeap[inv].max()->key().m_score; - UASSERT_OBJ(edgeCp == newCp, mtaskp, "CP doesn't match longest wayward edge"); - // Confirm that we only set each node's CP once. That's an - // important property of PropagateCp which allows it to be far - // faster than a recursive algorithm on some graphs. - const bool first = m_seen.insert(mtaskp).second; - UASSERT_OBJ(first, mtaskp, "Set CP on node twice"); - } - mtaskp->setCritPathCost(way, newCp); - cpHasIncreased(mtaskp, newCp + mtaskp->cost()); - } - - if (VL_UNLIKELY(m_slowAsserts)) m_seen.clear(); - } - -private: - VL_UNCOPYABLE(PropagateCp); -}; - //###################################################################### // Contraction @@ -686,9 +505,6 @@ class Contraction final { // fixed for that lifetime: merging only ever deletes vertices, never creates them. std::unique_ptr m_mtaskDatap; - PropagateCp m_forwardPropagator{m_slowAsserts}; // Forward propagator - PropagateCp m_reversePropagator{m_slowAsserts}; // Reverse propagator - // Singular source vertex of the OrderMTaskGraph LogicMTask* const m_entryMTaskp = m_mTaskGraph.entryp(); // Singular sink vertex of the dependency graph @@ -931,20 +747,22 @@ class Contraction final { recipientp->setCritPathCost(GraphWay::FORWARD, recipientNewCpFwd.cp); if (recipientNewCpFwd.propagate) { - m_forwardPropagator.cpHasIncreased(recipientp, recipientNewCpFwd.propagateCp); + m_mTaskGraph.forwardPropagator().cpHasIncreased(recipientp, + recipientNewCpFwd.propagateCp); } recipientp->setCritPathCost(GraphWay::REVERSE, recipientNewCpRev.cp); if (recipientNewCpRev.propagate) { - m_reversePropagator.cpHasIncreased(recipientp, recipientNewCpRev.propagateCp); + m_mTaskGraph.reversePropagator().cpHasIncreased(recipientp, + recipientNewCpRev.propagateCp); } if (donorNewCpFwd.propagate) { - m_forwardPropagator.cpHasIncreased(donorp, donorNewCpFwd.propagateCp); + m_mTaskGraph.forwardPropagator().cpHasIncreased(donorp, donorNewCpFwd.propagateCp); } if (donorNewCpRev.propagate) { - m_reversePropagator.cpHasIncreased(donorp, donorNewCpRev.propagateCp); + m_mTaskGraph.reversePropagator().cpHasIncreased(donorp, donorNewCpRev.propagateCp); } - m_forwardPropagator.go(); - m_reversePropagator.go(); + m_mTaskGraph.forwardPropagator().go(); + m_mTaskGraph.reversePropagator().go(); // Remove all other SiblingMCs that include recipientp or donorp. We remove all siblingMCs // of recipientp so we do not get huge numbers of SiblingMCs. We'll recreate them below, up diff --git a/src/V3OrderMTaskGraph.cpp b/src/V3OrderMTaskGraph.cpp index 5c32ad275..e47b61fb4 100644 --- a/src/V3OrderMTaskGraph.cpp +++ b/src/V3OrderMTaskGraph.cpp @@ -18,6 +18,7 @@ #include "V3OrderMTaskGraph.h" +#include "V3Global.h" #include "V3InstrCount.h" VL_DEFINE_DEBUG_FUNCTIONS; @@ -28,7 +29,9 @@ VL_DEFINE_DEBUG_FUNCTIONS; OrderMTaskGraph::OrderMTaskGraph(OrderMoveGraph& moveGraph) : m_moveGraph{moveGraph} , m_entryp{new LogicMTask{*this, nullptr}} - , m_exitp{new LogicMTask{*this, nullptr}} {} + , m_exitp{new LogicMTask{*this, nullptr}} + , m_forwardPropagator{v3Global.opt.debugPartition()} + , m_reversePropagator{v3Global.opt.debugPartition()} {} uint64_t OrderMTaskGraph::totalCost() const { uint64_t cost = 0; diff --git a/src/V3OrderMTaskGraph.h b/src/V3OrderMTaskGraph.h index 2db922fea..8643d0787 100644 --- a/src/V3OrderMTaskGraph.h +++ b/src/V3OrderMTaskGraph.h @@ -20,6 +20,10 @@ // candidate machinery: any auxiliary data the algorithms need is attached // externally via the vertex/edge user pointers. // +// PropagateCp propagates increasing critical path costs through the graph. +// OrderMTaskGraph owns one instance for each direction, which the algorithms +// operating on the graph use to keep the critical paths up to date. +// //************************************************************************* #ifndef VERILATOR_V3ORDERMTASKGRAPH_H_ @@ -37,43 +41,13 @@ #include #include #include +#include class LogicMTask; +class OrderMTaskGraph; template class PropagateCp; -//============================================================================= -// OrderMTaskGraph - -// The graph of LogicMTask vertices and MTaskEdge edges, used during multi-threaded scheduling. -class OrderMTaskGraph final : public V3Graph { - OrderMoveGraph& m_moveGraph; // The OrderMoveGraph this graph is built from - LogicMTask* const m_entryp; // The singular entry point vertex - LogicMTask* const m_exitp; // The singular exit point vertex - - // CONSTRUCTOR - explicit OrderMTaskGraph(OrderMoveGraph& moveGraph); // Used by build(), hence private - VL_UNCOPYABLE(OrderMTaskGraph); - VL_UNMOVABLE(OrderMTaskGraph); - -public: - // ACCESSORS - OrderMoveGraph& moveGraph() const { return m_moveGraph; } - LogicMTask* entryp() const { return m_entryp; } - LogicMTask* exitp() const { return m_exitp; } - - // METHODS - uint64_t totalCost() const; // O(V), called once - - // STATIC METHODS - // Build an MTask graph from 'moveGraph' - static std::unique_ptr build(OrderMoveGraph& moveGraph) VL_MT_DISABLED; - // Fix data hazards in the MTask graph - static void fixDataHazards(OrderMTaskGraph& mtaskGraph) VL_MT_DISABLED; - // Coarsen the MTask graph by merging MTasks until the given critical-path limit is reached - static void contract(OrderMTaskGraph& mtaskGraph, uint64_t scoreLimit) VL_MT_DISABLED; -}; - //============================================================================= // We keep MTaskEdge graph edges in a PairingHeap, sorted by score and id @@ -327,6 +301,225 @@ public: } }; +//============================================================================= +// PropagateCp + +template +class PropagateCp final { + // Propagate increasing critical path (CP) costs through a graph. + // + // Usage: + // * Client increases the cost and/or CP at a node or small set of nodes + // (often a pair in practice, eg. edge contraction.) + // * Client calls PropagateCp::cpHasIncreased() one or more times. + // Each call indicates that the inclusive CP of some "seed" vertex + // has increased to a given value. + // * NOTE: PropagateCp will neither read nor modify the cost + // or CPs at the seed vertices, it only accesses and modifies + // vertices wayward from the seeds. + // * Client calls PropagateCp::go(). Internally, this iteratively + // propagates the new CPs wayward through the graph. + // + + // TYPES + + // We keep pending vertices in a heap during critical path propagation + struct PendingKey final { + LogicMTask* m_mtaskp; // The vertex in the heap + uint64_t m_score; // The score of this entry + void increase(uint64_t score) { + UDEBUGONLY(UASSERT(score >= m_score, "Must increase");); + m_score = score; + } + bool operator<(const PendingKey& other) const { + if (m_score != other.m_score) return m_score < other.m_score; + return *m_mtaskp < *other.m_mtaskp; + } + }; + + using PendingHeap = PairingHeap; + using PendingHeapNode = typename PendingHeap::Node; + + // MEMBERS + PendingHeap m_pendingHeap; // Heap of pending rescores + + // We allocate this many heap nodes at once + static constexpr size_t ALLOC_CHUNK_SIZE = 128; + PendingHeapNode* m_freep = nullptr; // List of free heap nodes + std::vector> m_allocated; // Allocated heap nodes + + const bool m_slowAsserts; // Enable nontrivial asserts + // Used only with slow asserts to check MTasks visited only once + std::unordered_set m_seen; + +public: + // CONSTRUCTORS + explicit PropagateCp(bool slowAsserts) + : m_slowAsserts{slowAsserts} {} + + // METHODS +private: + // Allocate a HeapNode for the given element + PendingHeapNode* allocNode() { + // If no free nodes available, then make some + if (!m_freep) { + // Allocate in chunks for efficiency + m_allocated.emplace_back(new PendingHeapNode[ALLOC_CHUNK_SIZE]); + // Set up free list pointer + m_freep = m_allocated.back().get(); + // Set up free list chain + for (size_t i = 1; i < ALLOC_CHUNK_SIZE; ++i) { + m_freep[i - 1].m_next.m_ptr = &m_freep[i]; + } + // Clear the next pointer of the last entry + m_freep[ALLOC_CHUNK_SIZE - 1].m_next.m_ptr = nullptr; + } + // Free nodes are available, pick up the first one + PendingHeapNode* const resultp = m_freep; + m_freep = resultp->m_next.m_ptr; + resultp->m_next.m_ptr = nullptr; + return resultp; + } + + // Release a heap node (make it available for future allocation) + void freeNode(PendingHeapNode* nodep) { + // Re-use the existing link pointers and simply prepend it to the free list + nodep->m_next.m_ptr = m_freep; + m_freep = nodep; + } + +public: + void cpHasIncreased(LogicMTask* vxp, uint64_t newInclusiveCp) { + constexpr GraphWay way{N_Way}; + constexpr GraphWay inv{way.invert()}; + + // For *vxp, whose CP-inclusive has just increased to + // newInclusiveCp, iterate to all wayward nodes, update the edges + // of each, and add each to m_pending if its overall CP has grown. + for (V3GraphEdge& graphEdge : vxp->edges()) { + MTaskEdge& edge = static_cast(graphEdge); + + LogicMTask* const relativep = edge.furtherMTaskp(); + EdgeHeap::Node& edgeHeapNode = edge.m_edgeHeapNode[inv]; + if (newInclusiveCp > edgeHeapNode.key().m_score) { + relativep->m_edgeHeap[inv].increaseKey(&edgeHeapNode, newInclusiveCp); + } + + const uint64_t critPathCost = relativep->critPathCost(way); + + if (critPathCost >= newInclusiveCp) continue; + + // relativep's critPathCost() is out of step with its longest !wayward edge. + // Schedule that to be resolved. + const uint64_t newVal = newInclusiveCp - critPathCost; + + void*& pendingNodepRef = relativep->m_propagateHeapNodep; + if (PendingHeapNode* const nodep = static_cast(pendingNodepRef)) { + // Already in heap. Increase score if needed. + if (newVal > nodep->key().m_score) m_pendingHeap.increaseKey(nodep, newVal); + continue; + } + + // Add to heap + PendingHeapNode* const nodep = allocNode(); + pendingNodepRef = nodep; + m_pendingHeap.insert(nodep, {relativep, newVal}); + } + } + + void go() { + constexpr GraphWay way{N_Way}; + constexpr GraphWay inv{way.invert()}; + + // m_pending maps each pending vertex to the amount that it wayward + // CP will grow. + // + // We can iterate over the pending set in reverse order, always + // choosing the nodes with the largest pending CP-growth. + // + // The intuition is: if the original seed node had its CP grow by + // 50, the most any wayward node can possibly grow is also 50. So + // for anything pending to grow by 50, we know we can process it + // once and we won't have to grow its CP again on the current pass. + // After we're done with all the grow-by-50s, nothing else will + // grow by 50 again on the current pass, and we can process the + // grow-by-49s and we know we'll only have to process each one + // once. And so on. + // + // This generalizes to multiple seed nodes also. + while (!m_pendingHeap.empty()) { + // Pop max element from heap + PendingHeapNode* const maxp = m_pendingHeap.max(); + m_pendingHeap.remove(maxp); + // Pick up values + LogicMTask* const mtaskp = maxp->key().m_mtaskp; + const uint64_t cpGrowBy = maxp->key().m_score; + // Free the heap node, we are done with it + freeNode(maxp); + mtaskp->m_propagateHeapNodep = nullptr; + // Update the critPathCost of mtaskp, that was out-of-date with respect to its edges + const uint64_t startCp = mtaskp->critPathCost(way); + const uint64_t newCp = startCp + cpGrowBy; + if (VL_UNLIKELY(m_slowAsserts)) { + // Check that CP matches that of the longest edge wayward of vxp. + const uint64_t edgeCp = mtaskp->m_edgeHeap[inv].max()->key().m_score; + UASSERT_OBJ(edgeCp == newCp, mtaskp, "CP doesn't match longest wayward edge"); + // Confirm that we only set each node's CP once. That's an + // important property of PropagateCp which allows it to be far + // faster than a recursive algorithm on some graphs. + const bool first = m_seen.insert(mtaskp).second; + UASSERT_OBJ(first, mtaskp, "Set CP on node twice"); + } + mtaskp->setCritPathCost(way, newCp); + cpHasIncreased(mtaskp, newCp + mtaskp->cost()); + } + + if (VL_UNLIKELY(m_slowAsserts)) m_seen.clear(); + } + +private: + VL_UNCOPYABLE(PropagateCp); +}; + +//============================================================================= +// OrderMTaskGraph + +// The graph of LogicMTask vertices and MTaskEdge edges, used during multi-threaded scheduling. +class OrderMTaskGraph final : public V3Graph { + OrderMoveGraph& m_moveGraph; // The OrderMoveGraph this graph is built from + LogicMTask* const m_entryp; // The singular entry point vertex + LogicMTask* const m_exitp; // The singular exit point vertex + + // The critical path propagators, one for each direction. Owned here so the algorithms + // operating on this graph (contraction, hazard fixing) share them. + PropagateCp m_forwardPropagator; // Forward propagator + PropagateCp m_reversePropagator; // Reverse propagator + + // CONSTRUCTOR + explicit OrderMTaskGraph(OrderMoveGraph& moveGraph); // Used by build(), hence private + VL_UNCOPYABLE(OrderMTaskGraph); + VL_UNMOVABLE(OrderMTaskGraph); + +public: + // ACCESSORS + OrderMoveGraph& moveGraph() const { return m_moveGraph; } + LogicMTask* entryp() const { return m_entryp; } + LogicMTask* exitp() const { return m_exitp; } + PropagateCp& forwardPropagator() { return m_forwardPropagator; } + PropagateCp& reversePropagator() { return m_reversePropagator; } + + // METHODS + uint64_t totalCost() const; // O(V), called once + + // STATIC METHODS + // Build an MTask graph from 'moveGraph' + static std::unique_ptr build(OrderMoveGraph& moveGraph) VL_MT_DISABLED; + // Fix data hazards in the MTask graph + static void fixDataHazards(OrderMTaskGraph& mtaskGraph) VL_MT_DISABLED; + // Coarsen the MTask graph by merging MTasks until the given critical-path limit is reached + static void contract(OrderMTaskGraph& mtaskGraph, uint64_t scoreLimit) VL_MT_DISABLED; +}; + //============================================================================= // MTaskEdge method definitions (need the full definition of LogicMTask)