From 56331db7eedbed721bba475918b64b5e02f2a29c Mon Sep 17 00:00:00 2001 From: Geza Lore Date: Tue, 6 Oct 2026 07:30:50 +0100 Subject: [PATCH] Optimize concatenation of bitwise operations in DFG (#8625) Add DFG peephole patterns that replace 'Concat(a OP b, c OP d)' with 'Concat(a, c) OP Concat(b, d)' for And/Or/Xor, when one of the resulting Concats folds (or will fold after one more such push), including when the operands are adjacent terms of a nested Concat. Repeated application turns logic written bit-by-bit, into word-wide operations, which previously could produce very large, slow-to-compile C++. --- src/V3DfgPeephole.cpp | 108 ++++++++++++++++++++++++++++++++ src/V3DfgPeepholePatterns.h | 2 + src/V3DfgVertices.h | 4 ++ test_regress/t/t_dfg_peephole.v | 49 +++++++++++++++ 4 files changed, 163 insertions(+) diff --git a/src/V3DfgPeephole.cpp b/src/V3DfgPeephole.cpp index b9d28fd95..0584c622d 100644 --- a/src/V3DfgPeephole.cpp +++ b/src/V3DfgPeephole.cpp @@ -907,6 +907,91 @@ class V3DfgPeephole final : public DfgVisitor { return false; } + // Returns true if 'Concat(lhsp, rhsp)' is folded into a single vertex by another pattern + static bool isConcatFoldable(const DfgVertex* lhsp, const DfgVertex* rhsp) { + // Concatenation of the same vertex is a replicate + if (isSame(lhsp, rhsp)) return true; + if (const DfgRep* const lRepp = lhsp->cast()) { + if (isSame(lRepp->srcp(), rhsp)) return true; + } + if (const DfgRep* const rRepp = rhsp->cast()) { + if (isSame(lhsp, rRepp->srcp())) return true; + } + // Concatenation of adjoining selects is a select + if (const DfgSel* const lSelp = lhsp->cast()) { + if (const DfgSel* const rSelp = rhsp->cast()) { + return isSame(lSelp->fromp(), rSelp->fromp()) + && lSelp->lsb() == rSelp->lsb() + rSelp->width(); + } + } + // Concatenation of Nots is pushed through the Nots (PUSH_CONCAT_THROUGH_NOTS) + if (const DfgNot* const lNotp = lhsp->cast()) { + if (const DfgNot* const rNotp = rhsp->cast()) { + return !lNotp->hasMultipleSinks() // + && !rNotp->hasMultipleSinks() // + && isConcatFoldable(lNotp->srcp(), rNotp->srcp()); + } + } + return false; + } + + // Returns true if 'lhsp' and 'rhsp' are the same bitwise operation + static bool isSameBitwise(const DfgVertex* lhsp, const DfgVertex* rhsp) { + if (lhsp->type() != rhsp->type()) return false; + return lhsp->is() || lhsp->is() || lhsp->is(); + } + + // Returns true if 'Concat(lhsp, rhsp)' folds into a single vertex, either directly, or + // after pushing it through the same bitwise operation on both sides + static bool isConcatSimplifiable(const DfgVertex* lhsp, const DfgVertex* rhsp) { + if (isConcatFoldable(lhsp, rhsp)) return true; + if (!isSameBitwise(lhsp, rhsp)) return false; + if (lhsp->hasMultipleSinks() || rhsp->hasMultipleSinks()) return false; + const DfgVertexBinary* const lp = lhsp->as(); + const DfgVertexBinary* const rp = rhsp->as(); + if (isConcatFoldable(lp->lhsp(), rp->lhsp()) && isConcatFoldable(lp->rhsp(), rp->rhsp())) + return true; + if (isConcatFoldable(lp->lhsp(), rp->rhsp()) && isConcatFoldable(lp->rhsp(), rp->lhsp())) + return true; + return false; + } + + // Returns true if replacing 'Concat(a OP b, c OP d)' (the Concat of 'lhsp' and 'rhsp') with + // 'Concat(a, c) OP Concat(b, d)' removes vertices, that is if at least one of the resulting + // Concats simplifies. The other might be pushed further by a later application. Sets 'swap' + // if instead 'Concat(a, d) OP Concat(b, c)' should be used. + static bool isConcatPushableThroughBitwise(const DfgVertex* lhsp, const DfgVertex* rhsp, + bool& swap) { + if (!isSameBitwise(lhsp, rhsp)) return false; + if (lhsp->hasMultipleSinks() || rhsp->hasMultipleSinks()) return false; + const DfgVertexBinary* const lp = lhsp->as(); + const DfgVertexBinary* const rp = rhsp->as(); + swap = false; + if (isConcatSimplifiable(lp->lhsp(), rp->lhsp())) return true; + if (isConcatSimplifiable(lp->rhsp(), rp->rhsp())) return true; + swap = true; + if (isConcatSimplifiable(lp->lhsp(), rp->rhsp())) return true; + if (isConcatSimplifiable(lp->rhsp(), rp->lhsp())) return true; + return false; + } + + // Create 'Concat(a, c) OP Concat(b, d)' from 'lhsp' = 'a OP b' and 'rhsp' = 'c OP d', or + // 'Concat(a, d) OP Concat(b, c)' if 'swap' is set + DfgVertex* makeConcatPushedThroughBitwise(FileLine* flp, const DfgVertex* lhsp, + const DfgVertex* rhsp, bool swap) { + const DfgVertexBinary* const lp = lhsp->as(); + const DfgVertexBinary* const rp = rhsp->as(); + const DfgDataType& dtype = DfgDataType::packed(lp->width() + rp->width()); + DfgVertex* const cp = swap ? rp->rhsp() : rp->lhsp(); + DfgVertex* const dp = swap ? rp->lhsp() : rp->rhsp(); + DfgConcat* const newLhsp = make(flp, dtype, lp->lhsp(), cp); + DfgConcat* const newRhsp = make(flp, dtype, lp->rhsp(), dp); + if (lp->is()) return make(flp, dtype, newLhsp, newRhsp); + if (lp->is()) return make(flp, dtype, newLhsp, newRhsp); + UASSERT_OBJ(lp->is(), lp, "Should be a bitwise operation"); + return make(flp, dtype, newLhsp, newRhsp); + } + template VL_ATTR_WARN_UNUSED_RESULT bool tryReplaceBitwiseWithReduction(Bitwise* vtxp) { UASSERT_OBJ(vtxp->width() == 1, vtxp, "Width must be 1"); @@ -2017,6 +2102,29 @@ class V3DfgPeephole final : public DfgVisitor { } } + { + bool swap = false; + if (isConcatPushableThroughBitwise(lhsp, rhsp, swap)) { + APPLYING(PUSH_CONCAT_THROUGH_BITWISE) { + replace(makeConcatPushedThroughBitwise(flp, lhsp, rhsp, swap)); + return; + } + } + // The mirrored RHS match is not needed due to RIGHT_LEANING_ASSOC + if (DfgConcat* const rConcatp = rhsp->cast()) { + DfgVertex* const rlp = rConcatp->lhsp(); + if (!rConcatp->hasMultipleSinks() + && isConcatPushableThroughBitwise(lhsp, rlp, swap)) { + APPLYING(PUSH_NESTED_CONCAT_THROUGH_BITWISE_ON_LHS) { + DfgVertex* const newp + = makeConcatPushedThroughBitwise(flp, lhsp, rlp, swap); + replace(make(vtxp, newp, rConcatp->rhsp())); + return; + } + } + } + } + if (DfgSel* const lSelp = lhsp->cast()) { if (DfgSel* const rSelp = rhsp->cast()) { if (isSame(lSelp->fromp(), rSelp->fromp())) { diff --git a/src/V3DfgPeepholePatterns.h b/src/V3DfgPeepholePatterns.h index f9062b728..64c735960 100644 --- a/src/V3DfgPeepholePatterns.h +++ b/src/V3DfgPeepholePatterns.h @@ -57,9 +57,11 @@ _FOR_EACH_DFG_PEEPHOLE_OPTIMIZATION_APPLY(macro, PUSH_BITWISE_THROUGH_SEL) \ _FOR_EACH_DFG_PEEPHOLE_OPTIMIZATION_APPLY(macro, PUSH_COMMUTATIVE_BINARY_THROUGH_COND) \ _FOR_EACH_DFG_PEEPHOLE_OPTIMIZATION_APPLY(macro, PUSH_COMPARE_OP_THROUGH_CONCAT) \ + _FOR_EACH_DFG_PEEPHOLE_OPTIMIZATION_APPLY(macro, PUSH_CONCAT_THROUGH_BITWISE) \ _FOR_EACH_DFG_PEEPHOLE_OPTIMIZATION_APPLY(macro, PUSH_CONCAT_THROUGH_COND_LHS) \ _FOR_EACH_DFG_PEEPHOLE_OPTIMIZATION_APPLY(macro, PUSH_CONCAT_THROUGH_COND_RHS) \ _FOR_EACH_DFG_PEEPHOLE_OPTIMIZATION_APPLY(macro, PUSH_CONCAT_THROUGH_NOTS) \ + _FOR_EACH_DFG_PEEPHOLE_OPTIMIZATION_APPLY(macro, PUSH_NESTED_CONCAT_THROUGH_BITWISE_ON_LHS) \ _FOR_EACH_DFG_PEEPHOLE_OPTIMIZATION_APPLY(macro, PUSH_NOT_THROUGH_COND) \ _FOR_EACH_DFG_PEEPHOLE_OPTIMIZATION_APPLY(macro, PUSH_REDUCTION_THROUGH_BITWISE_OF_CONCAT) \ _FOR_EACH_DFG_PEEPHOLE_OPTIMIZATION_APPLY(macro, PUSH_REDUCTION_THROUGH_BITWISE_OF_SELS) \ diff --git a/src/V3DfgVertices.h b/src/V3DfgVertices.h index f260afd5e..10b799084 100644 --- a/src/V3DfgVertices.h +++ b/src/V3DfgVertices.h @@ -374,6 +374,10 @@ protected: public: ASTGEN_MEMBERS_DfgVertexBinary; + DfgVertex* lhsp() const { return inputp(0); } + void lhsp(DfgVertex* vtxp) { inputp(0, vtxp); } + DfgVertex* rhsp() const { return inputp(1); } + void rhsp(DfgVertex* vtxp) { inputp(1, vtxp); } }; class DfgMatchMasked final : public DfgVertexBinary { diff --git a/test_regress/t/t_dfg_peephole.v b/test_regress/t/t_dfg_peephole.v index 8a1e480aa..d1fe16b86 100644 --- a/test_regress/t/t_dfg_peephole.v +++ b/test_regress/t/t_dfg_peephole.v @@ -217,6 +217,55 @@ module t ( `signal(REPLACE_CONCAT_ZERO_AND_SEL_TOP_WITH_SHIFTR, {62'd0, rand_a[63:62]}); `signal(REPLACE_CONCAT_SEL_BOTTOM_AND_ZERO_WITH_SHIFTL, {rand_a[1:0], 62'd0}); `signal(PUSH_CONCAT_THROUGH_NOTS, {~(rand_a+64'd101), ~(rand_b+64'd101)} ); + // Operands of the Concats below are separate wires, as V3Const would otherwise merge them before Dfg. + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_XOR_L = rand_a[5] ^ rand_b[7]; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_XOR_R = rand_a[4] ^ rand_b[7]; + wire [1:0] tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_L_L = rand_a[10:9] & {2{rand_b[11]}}; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_L_R = rand_a[3] & rand_b[11]; + wire [1:0] tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_L_DIFF_L = {2{srand_b[41]}} & srand_a[42:41]; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_L_DIFF_R = srand_b[43] & srand_a[44]; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_R_L = rand_a[13] & rand_b[15]; + wire [1:0] tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_R_R = rand_a[12:11] & {2{rand_b[15]}}; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_R_SWAP_L = rand_b[16] & rand_a[13]; + wire [1:0] tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_R_SWAP_R = {2{rand_b[16]}} & rand_a[7:6]; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_L = ~rand_a[23] & rand_b[25]; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_R = ~rand_a[22] & rand_b[25]; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_DIFF_L = ~rand_a[30] & rand_b[33]; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_DIFF_R = ~rand_b[31] & rand_b[33]; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_SHARED_L_L = ~rand_a[19] & rand_b[21]; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_SHARED_L_R = ~rand_a[18] & rand_b[22]; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_SHARED_R_L = ~rand_a[39] & rand_b[21]; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_SHARED_R_R = ~rand_a[38] & rand_b[22]; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_L = (srand_b[50] & srand_a[46]) | (srand_b[51] & srand_a[48]); + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_R = (srand_b[50] & srand_a[45]) | (srand_b[51] & srand_a[47]); + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_SWAP_L = (rand_a[41] & rand_b[45]) | (rand_a[43] & rand_b[46]); + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_SWAP_R = (rand_a[40] & rand_b[45]) | (rand_a[42] & rand_b[46]); + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_PARTIAL_L = (rand_a[41] & rand_b[47]) | (rand_a[43] & rand_b[46]); + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_PARTIAL_R = (rand_a[40] & rand_b[48]) | (rand_a[42] & rand_b[46]); + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_PARTIAL_SWAP_L = (srand_a[61] & srand_b[60]) | (srand_a[63] & srand_b[62]); + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_PARTIAL_SWAP_R = (srand_b[57] & srand_a[60]) | (srand_a[62] & srand_b[62]); + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_SHARED_R_L = rand_a[61] & rand_b[3]; + wire tmp_PUSH_CONCAT_THROUGH_BITWISE_SHARED_R_R = rand_a[60] & rand_b[3]; + wire tmp_PUSH_NESTED_CONCAT_THROUGH_BITWISE_ON_LHS_L = rand_a[55] & rand_b[57]; + wire [4:0] tmp_PUSH_NESTED_CONCAT_THROUGH_BITWISE_ON_LHS_R = {rand_a[54] & rand_b[57], rand_b[63:60]}; + `signal(PUSH_CONCAT_THROUGH_BITWISE_XOR, {tmp_PUSH_CONCAT_THROUGH_BITWISE_XOR_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_XOR_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_REP_L, {tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_L_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_L_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_REP_L_DIFF, {tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_L_DIFF_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_L_DIFF_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_REP_R, {tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_R_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_R_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_REP_R_SWAP, {tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_R_SWAP_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_REP_R_SWAP_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_NOT, {tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_NOT_DIFF, {tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_DIFF_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_DIFF_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_NOT_SHARED_L, {tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_SHARED_L_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_SHARED_L_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_NOT_SHARED_L_SHARED, ~rand_a[19]); + `signal(PUSH_CONCAT_THROUGH_BITWISE_NOT_SHARED_R, {tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_SHARED_R_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_NOT_SHARED_R_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_NOT_SHARED_R_SHARED, ~rand_a[38]); + `signal(PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD, {tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_SWAP, {tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_SWAP_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_SWAP_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_PARTIAL, {tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_PARTIAL_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_PARTIAL_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_PARTIAL_SWAP, {tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_PARTIAL_SWAP_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_LOOKAHEAD_PARTIAL_SWAP_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_SHARED_R, {tmp_PUSH_CONCAT_THROUGH_BITWISE_SHARED_R_L, tmp_PUSH_CONCAT_THROUGH_BITWISE_SHARED_R_R}); + `signal(PUSH_CONCAT_THROUGH_BITWISE_SHARED_R_SHARED, rand_a[60] & rand_b[3]); + `signal(PUSH_NESTED_CONCAT_THROUGH_BITWISE_ON_LHS, {tmp_PUSH_NESTED_CONCAT_THROUGH_BITWISE_ON_LHS_L, tmp_PUSH_NESTED_CONCAT_THROUGH_BITWISE_ON_LHS_R}); `signal(REMOVE_CONCAT_OF_ADJOINING_SELS, {rand_a[10:3], rand_a[2:1]}); `signal(REPLACE_NESTED_CONCAT_OF_ADJOINING_SELS_ON_LHS_CAT, {rand_a[2:1], rand_b}); `signal(REPLACE_NESTED_CONCAT_OF_ADJOINING_SELS_ON_RHS_CAT, {rand_b, rand_a[10:3]});