Fix unordered data hazards in multi-threaded scheduling (#8133)

The OrderGraph used during V3Order step deliberately omits some variable
accesses from the dependency graph. E.g.: a read of a variable that is
in the reading block's own hybrid sensitivity list emits no edge, nor
does a read ignored due to a force/release, nor an access to a variable
marked 'ignoreSchedWrite' and friends. For serial mode that is fine, the
logic runs one block at a time. In parallel mode two such blocks can run
concurrently, and if one writes what the other reads, that is a data
race at runtime.

These accesses cannot be recovered from the graph edges. They are now
collected from the AST while the OrderGraph is built, and held by the
OrderLogicVertex performing them.

FixDataHazards is reworked around these access lists stored in
OrderLogicVertex, so it is now aware of all variable accesses the logic
makes, including those not encoded by the dependency graph edges. The
previous heuristic of fixing data hazards by merging same-rank MTasks is
removed. Additional edges are inserted instead to prescribe a fixed
ordering of conflicting MTasks. To insert edges without unduly
increasing the critical path, or introducing cycles, new edges are
added such that they preserve topological ordering, and they are
inserted between vertices sorted by critical path length. See algorithm
details in the code.

Also add a data hazard checker under '--debug-partition', reporting every
unordered accessor pair left in the final MTask graph.

This fixes the race demonstrated by t_sched_hybrid_hazard (#7913),
which is no longer expected to fail.

Under ThreadSanitizer over the vltmt tests: 17 failing before, 3 after,
with no regressions. The 3 remaining are different defects.
This commit is contained in:
Geza Lore
2026-08-18 08:50:50 +02:00
committed by GitHub
parent 96ea587df0
commit d4a18d4dfb
10 changed files with 477 additions and 259 deletions
+38 -7
View File
@@ -76,20 +76,24 @@ public:
class OrderGraphBuilder final : public VNVisitor {
// TYPES
enum VarUsage : uint8_t { VU_CON = 0x1, VU_GEN = 0x2 };
enum VarAccess : uint8_t { VA_READ = 0x1, VA_WRITE = 0x2 };
using VarVertexType = OrderUser::VarVertexType;
// NODE STATE
// AstVarScope::user1 -> OrderUser instance for variable (via m_orderUser)
// AstVarScope::user2 -> VarUsage within logic blocks
// AstVarScope::user3 -> bool: Hybrid sensitivity
// AstVarScope::user4 -> VarAccess within logic blocks
const VNUser1InUse user1InUse;
const VNUser2InUse user2InUse;
const VNUser3InUse user3InUse;
const VNUser4InUse user4InUse;
AstUser1Allocator<AstVarScope, OrderUser> m_orderUser;
// STATE
OrderGraph* const m_graphp = new OrderGraph; // The ordering graph built by this visitor
OrderLogicVertex* m_logicVxp = nullptr; // Current logic block being analyzed
std::vector<AstVarScope*> m_accessedVscps; // Variables accessed by the current logic block
// Map from Trigger reference AstSenItem to the original AstSenTree
const V3Order::TrigToSenMap& m_trigToSen;
@@ -106,13 +110,15 @@ class OrderGraphBuilder final : public VNVisitor {
bool m_inPost = false; // Underneath AstAlwaysPost
std::function<bool(const AstVarScope*)> m_readTriggersCombLogic;
V3Sched::util::VarScopeSet m_forceReadEdgeIgnores;
const bool m_parallel; // Ordering for multi-threaded execution (record variable accesses)
// METHODS
void iterateLogic(AstNode* nodep) {
UASSERT_OBJ(!m_logicVxp, nodep, "Should not nest");
// Reset VarUsage
// Reset VarUsage and VarAccess
AstNode::user2ClearTree();
AstNode::user4ClearTree();
m_forceReadEdgeIgnores.clear();
if (!m_inClocked)
V3Sched::util::collectForceReadEdgeIgnores(nodep, m_forceReadEdgeIgnores);
@@ -120,6 +126,17 @@ class OrderGraphBuilder final : public VNVisitor {
m_logicVxp = new OrderLogicVertex{m_graphp, m_scopep, m_domainp, m_hybridp, nodep};
// Gather variable dependencies based on usage
iterateChildren(nodep);
if (m_parallel) {
// Emit one access record for each variable this logic block accessed
for (AstVarScope* const vscp : m_accessedVscps) {
const int recorded = vscp->user4();
const VAccess access = recorded == (VA_READ | VA_WRITE) ? VAccess::READWRITE
: recorded == VA_WRITE ? VAccess::WRITE
: VAccess::READ;
m_logicVxp->addVarAccess(vscp, access);
}
m_accessedVscps.clear();
}
// Finished with this logic
m_logicVxp = nullptr;
m_forceReadEdgeIgnores.clear();
@@ -184,6 +201,16 @@ class OrderGraphBuilder final : public VNVisitor {
// Variable reference in logic. Add data dependency.
// Record the raw access for the multi-threaded data hazard fixer
if (m_parallel) {
uint8_t recorded = 0;
if (nodep->access().isWriteOrRW()) recorded |= VA_WRITE;
if (nodep->access().isReadOrRW()) recorded |= VA_READ;
UASSERT_OBJ(recorded, nodep, "Unknown variable access type");
// Accumulate access type, record the variable on first access only
if (!varscp->user4Or(recorded)) m_accessedVscps.push_back(varscp);
}
// Check whether this variable was already generated/consumed in the same logic. We
// don't want to add extra edges if the logic has many usages of the same variable,
// so only proceed on first encounter.
@@ -357,8 +384,9 @@ class OrderGraphBuilder final : public VNVisitor {
// CONSTRUCTOR
OrderGraphBuilder(AstNetlist* /*nodep*/, const std::vector<V3Sched::LogicByScope*>& coll,
const V3Order::TrigToSenMap& trigToSen)
: m_trigToSen{trigToSen} {
const V3Order::TrigToSenMap& trigToSen, bool parallel)
: m_trigToSen{trigToSen}
, m_parallel{parallel} {
// Build the graph
for (const V3Sched::LogicByScope* const lbsp : coll) {
for (const auto& pair : *lbsp) {
@@ -375,14 +403,17 @@ public:
// this visitor does change the tree (removes some nodes related to DPI export trigger).
static std::unique_ptr<OrderGraph> apply(AstNetlist* nodep,
const std::vector<V3Sched::LogicByScope*>& coll,
const V3Order::TrigToSenMap& trigToSen) {
return std::unique_ptr<OrderGraph>{OrderGraphBuilder{nodep, coll, trigToSen}.m_graphp};
const V3Order::TrigToSenMap& trigToSen,
bool parallel) {
return std::unique_ptr<OrderGraph>{
OrderGraphBuilder{nodep, coll, trigToSen, parallel}.m_graphp};
}
};
std::unique_ptr<OrderGraph>
V3Order::buildOrderGraph(AstNetlist* netlistp, //
const std::vector<V3Sched::LogicByScope*>& coll, //
const V3Order::TrigToSenMap& trigToSen) {
return OrderGraphBuilder::apply(netlistp, coll, trigToSen);
const V3Order::TrigToSenMap& trigToSen, //
bool parallel) {
return OrderGraphBuilder::apply(netlistp, coll, trigToSen, parallel);
}