From 8f8dd9f9812c8c97a57f26074236adeaf23b244d Mon Sep 17 00:00:00 2001 From: Christopher Haster Date: Sun, 14 Apr 2024 12:00:09 -0500 Subject: [PATCH] rbyd-rr: Eagerly flip, adopt branch before/after to disambiguate ysplits It's been annoying for a while how many flip operations we need in lfsr_rbyd_appendattr to implement diverging range removals correctly. Unfortuantely, we need all of these flips since we need to know the original alt ordering in order to know how to split yellow nodes. Keep in mind yellow splits depend on what alts exist in our history: b .-------'| .-'| | b | .----'| => .-----|-'| | | | p[0].jump = 0xa10 00000b20: altrle 0x402 w0 0xa20 <'-> jump = 0xa20 00000b30: altble 0x403 w0 0xa30 <--- branch_ = 0xb30 Anything jump out? That's right! only branch_ is > branch. This holds even after flips: branch = 0xb20 branch = 0xb20 flip2 branch = 0xb20 p[0].jump = 0xa10 flip p[0].jump = 0xa10 --.---> p[0].jump = 0xb30 jump = 0xa20 --.--> jump = 0xb30 --'-.-> jump = 0xa20 branch_ = 0xb30 --'--> branch_ = 0xa20 ----'-> branch_ = 0xa10 This is provable by noting that our alts can't even encode forward jumps. So... proof by lack of encoding? We can use this to determine which yellow split is needed even after flipping: - branch_ < branch && jump < branch => take yellow alt - branch_ < branch && jump > branch => take red alt - branch_ > branch => take black alt This lets us move/deduplicate the flipping logic before the diverging logic and operate in a sort of "flipped space", where branch_ is always the next branch we will take. Unfortunately we do need to flip red alts that don't get split back before descending down red nodes, which sort of matches our weird access pattern, but this extra flip is well worth the code savings elsewhere. --- This greatly simplifies the state space of lfsr_rbyd_appendattr, and it already shows in code size measurements: code stack before: 34528 2864 after: 34320 (-0.6%) 2864 (+0.0%) code frame stack appendattr before: 2378 216 568 appendattr after: 2280 (-4.1%) 216 (+0.0%) 568 (+0.0%) But this is really only after simplifying the diverging logic and yellow splits. I think there may be even more savings if we can figure out how to move all of the alt logic into the "flipped space"... --- lfs.c | 286 +++++++++++++++++++++++++++------------------------------- 1 file changed, 134 insertions(+), 152 deletions(-) diff --git a/lfs.c b/lfs.c index f2bc4b66..5ba837a3 100644 --- a/lfs.c +++ b/lfs.c @@ -2928,119 +2928,56 @@ trunk:; jump = branch - jump; lfs_size_t branch_ = branch + d; - // do bounds want to take different paths? begin diverging - if (!diverged - // diverging black? - && (((lfsr_tag_isblack(alt) - // give up if we find a yellow alt - || lfsr_tag_isred(p[0].alt)) - && lfsr_tag_diverging2( - alt, weight, - p[0].alt, p[0].weight, - lower_rid, upper_rid, - a_rid, a_tag, - b_rid, b_tag)) - // diverging red? - || (lfsr_tag_isred(p[0].alt) - && lfsr_tag_diverging( - p[0].alt, p[0].weight, - lower_rid, upper_rid, - a_rid, a_tag, - b_rid, b_tag)))) { - diverged = true; - - // diverging red? flip - if (lfsr_tag_isred(p[0].alt) - && lfsr_tag_diverging( - p[0].alt, p[0].weight, - lower_rid, upper_rid, - a_rid, a_tag, - b_rid, b_tag)) { - if (lfsr_tag_isparallel(alt, p[0].alt)) { - lfsr_tag_flip2(&alt, &weight, - p[0].alt, p[0].weight, - lower_rid, upper_rid); - lfs_swap32(&jump, &branch_); - } - - lfs_swap16(&p[0].alt, &alt); - lfs_swap32(&p[0].weight, &weight); - lfs_swap32(&p[0].jump, &jump); - p[0].alt |= LFSR_TAG_R; - alt &= ~LFSR_TAG_R; - - // both diverging? collapse - if (lfsr_tag_diverging( - p[0].alt, p[0].weight, - lower_rid, upper_rid, - a_rid, a_tag, - b_rid, b_tag)) { - LFS_ASSERT(!lfsr_tag_isparallel(alt, p[0].alt)); - lfsr_tag_flip2(&alt, &weight, - p[0].alt, p[0].weight, - lower_rid, upper_rid); - lfs_swap32(&jump, &branch_); - - weight += p[0].weight; - jump = p[0].jump; - lfsr_p_pop(p); - } - } - - // diverging upper? stitch together both trunks - if (a_rid > b_rid || a_tag > b_tag) { - if (lfsr_tag_isgt(alt)) { - lfsr_tag_flip2( - &alt, &weight, - p[0].alt, p[0].weight, - lower_rid, upper_rid); - lfs_swap32(&jump, &branch_); - } - lfsr_tag_trim2( - alt, weight, - p[0].alt, p[0].weight, - &lower_rid, &upper_rid, - &lower_tag, &upper_tag); - - // stitch together both trunks - err = lfsr_p_push(lfs, rbyd, p, - LFSR_TAG_ALT(LFSR_TAG_LE, LFSR_TAG_B, d_tag), - d_rid - (lower_rid - weight), - jump); - if (err) { - return err; - } - - // continue to next alt - branch = branch_; - continue; - } + // yellow alts should be parallel + LFS_ASSERT(!(lfsr_tag_isred(alt) && lfsr_tag_isred(p[0].alt)) + || lfsr_tag_isparallel(alt, p[0].alt)); // force diverged alts to be pruned - } else if (diverged + if (diverged && lfsr_tag_diverging2( alt, weight, p[0].alt, p[0].weight, lower_rid, upper_rid, a_rid, a_tag, b_rid, b_tag)) { - if (lfsr_tag_follow2( - alt, weight, - p[0].alt, p[0].weight, - lower_rid, upper_rid, - a_rid, a_tag)) { - lfsr_tag_flip2( - &alt, &weight, + // both diverged? collapse + if (lfsr_tag_isred(p[0].alt) + && lfsr_tag_diverging( p[0].alt, p[0].weight, - lower_rid, upper_rid); - lfs_swap32(&jump, &branch_); - } + lower_rid, upper_rid, + a_rid, a_tag, + b_rid, b_tag)) { + if (!lfsr_tag_isparallel(alt, p[0].alt)) { + lfsr_tag_flip2(&alt, &weight, + p[0].alt, p[0].weight, + lower_rid, upper_rid); + lfs_swap32(&jump, &branch_); + } - lfsr_tag_trim( - alt, weight, - &lower_rid, &upper_rid, - &lower_tag, &upper_tag); - weight = 0; + weight += p[0].weight; + jump = p[0].jump; + lfsr_p_pop(p); + + // one diverged? trim so alt is pruned + } else { + if (lfsr_tag_follow2( + alt, weight, + p[0].alt, p[0].weight, + lower_rid, upper_rid, + a_rid, a_tag)) { + lfsr_tag_flip2( + &alt, &weight, + p[0].alt, p[0].weight, + lower_rid, upper_rid); + lfs_swap32(&jump, &branch_); + } + + lfsr_tag_trim( + alt, weight, + &lower_rid, &upper_rid, + &lower_tag, &upper_tag); + weight = 0; + } } // prune? @@ -3093,10 +3030,89 @@ trunk:; } } - // two reds makes a yellow, split? - if (lfsr_tag_isred(alt) && lfsr_tag_isred(p[0].alt)) { - LFS_ASSERT(lfsr_tag_isparallel(alt, p[0].alt)); + // take black alt? needs a flip + // b + // .-'| => .-'| + // 1 2 1 2 1 + if (lfsr_tag_follow2( + alt, weight, + p[0].alt, p[0].weight, + lower_rid, upper_rid, + a_rid, a_tag)) { + lfsr_tag_flip2(&alt, &weight, + p[0].alt, p[0].weight, + lower_rid, upper_rid); + lfs_swap32(&jump, &branch_); + } + // should've taken red alt? needs a flip + // r + // .----'| .-'| + // | | >b + // | .-'| .--|-'| + // 1 2 3 1 2 3 1 + if (lfsr_tag_isred(p[0].alt) + && lfsr_tag_follow(p[0].alt, p[0].weight, + lower_rid, upper_rid, + a_rid, a_tag)) { + lfs_swap16(&p[0].alt, &alt); + lfs_swap32(&p[0].weight, &weight); + lfs_swap32(&p[0].jump, &jump); + alt = (alt & ~LFSR_TAG_R) | (p[0].alt & LFSR_TAG_R); + p[0].alt |= LFSR_TAG_R; + + lfsr_tag_flip2(&alt, &weight, + p[0].alt, p[0].weight, + lower_rid, upper_rid); + lfs_swap32(&jump, &branch_); + } + + // do bounds want to take different paths? begin diverging + if (!diverged + // eagerly diverge on lower trunk + && ((a_rid < b_rid || a_tag < b_tag) + // diverging black? + || lfsr_tag_isblack(alt) + // give up if we find a yellow alt + || lfsr_tag_isred(p[0].alt)) + // we only need to check the second alt, either + // lower or upper trunk will flip on follow + && lfsr_tag_diverging2( + alt, weight, + p[0].alt, p[0].weight, + lower_rid, upper_rid, + a_rid, a_tag, + b_rid, b_tag)) { + diverged = true; + + // diverging upper? stitch together both trunks + if (a_rid > b_rid || a_tag > b_tag) { + lfsr_tag_trim2( + alt, weight, + p[0].alt, p[0].weight, + &lower_rid, &upper_rid, + &lower_tag, &upper_tag); + + // stitch together both trunks + err = lfsr_p_push(lfs, rbyd, p, + LFSR_TAG_ALT(LFSR_TAG_LE, LFSR_TAG_B, d_tag), + d_rid - (lower_rid - weight), + jump); + if (err) { + return err; + } + + // continue to next alt + branch = branch_; + continue; + } + } + + // two reds makes a yellow, split? + // + // note we've lost the original yellow edge because of flips, but + // we know the red edge is the only branch_ > branch + if (lfsr_tag_isred(alt) && lfsr_tag_isred(p[0].alt)) { // if we take the red or yellow alt we can just point // to the black alt // b @@ -3106,19 +3122,12 @@ trunk:; // | | branch) { + lfs_swap16(&p[0].alt, &alt); + lfs_swap32(&p[0].weight, &weight); + lfs_swap32(&p[0].jump, &jump); + } alt &= ~LFSR_TAG_R; lfsr_tag_trim( @@ -3155,45 +3164,18 @@ trunk:; } } - // black alts terminate 2-3-4 nodes - if (lfsr_tag_isblack(alt)) { - // take black alt? needs a flip - // b - // .-'| => .-'| - // 1 2 1 2 1 - if (lfsr_tag_follow2( - alt, weight, - p[0].alt, p[0].weight, - lower_rid, upper_rid, - a_rid, a_tag)) { - lfsr_tag_flip2(&alt, &weight, - p[0].alt, p[0].weight, - lower_rid, upper_rid); - lfs_swap32(&jump, &branch_); - } - - // should've taken red alt? needs a flip - // r - // .----'| .-'| - // | | >b - // | .-'| .--|-'| - // 1 2 3 1 2 3 1 - if (lfsr_tag_isred(p[0].alt) - && lfsr_tag_follow(p[0].alt, p[0].weight, - lower_rid, upper_rid, - a_rid, a_tag)) { - lfs_swap16(&p[0].alt, &alt); - lfs_swap32(&p[0].weight, &weight); - lfs_swap32(&p[0].jump, &jump); - p[0].alt |= LFSR_TAG_R; - alt &= ~LFSR_TAG_R; - + // red alt? we need to read the rest of the 2-3-4 node + if (lfsr_tag_isred(alt)) { + // undo flip temporarily + if (branch_ < branch) { lfsr_tag_flip2(&alt, &weight, p[0].alt, p[0].weight, lower_rid, upper_rid); lfs_swap32(&jump, &branch_); } + // black alt? terminate 2-3-4 nodes + } else { // trim alts from our current bounds lfsr_tag_trim2( alt, weight,