diff --git a/scripts/dbgbtree.py b/scripts/dbgbtree.py index 4307501b..c0691a7c 100755 --- a/scripts/dbgbtree.py +++ b/scripts/dbgbtree.py @@ -262,25 +262,216 @@ def tagrepr(tag, weight=None, size=None, off=None): ' w%d' % weight if weight is not None else '', ' %d' % size if size is not None else '') +# tree branches are an abstract thing for tree rendering +class TreeBranch: + def __init__(self, a, b, depth=0, color='b'): + # a and b are context specific + self.a = a + self.b = b + self.depth = depth + self.color = color + + def __repr__(self): + return '%s(%s, %s, %s, %s)' % ( + self.__class__.__name__, + self.a, + self.b, + self.depth, + self.color) + + def __eq__(self, other): + return ((self.a, self.b, self.depth, self.color) + == (other.a, other.b, other.depth, other.color)) + + def __ne__(self, other): + return not self.__eq__(other) + + def __hash__(self): + return hash((self.a, self.b, self.depth)) + +def treerepr(tree, x, depth=None, color=False): + # find the max depth from the tree + if depth is None: + depth = max((t.depth+1 for t in tree), default=0) + if depth == 0: + return '' + + def branchrepr(tree, x, d, was): + for t in tree: + if t.depth == d and t.b == x: + if any(t.depth == d and t.a == x + for t in tree): + return '+-', t.color, t.color + elif any(t.depth == d + and x > min(t.a, t.b) + and x < max(t.a, t.b) + for t in tree): + return '|-', t.color, t.color + elif t.a < t.b: + return '\'-', t.color, t.color + else: + return '.-', t.color, t.color + for t in tree: + if t.depth == d and t.a == x: + return '+ ', t.color, None + for t in tree: + if (t.depth == d + and x > min(t.a, t.b) + and x < max(t.a, t.b)): + return '| ', t.color, was + if was: + return '--', was, was + return ' ', None, None + + trunk = [] + was = None + for d in range(depth): + t, c, was = branchrepr(tree, x, d, was) + + trunk.append('%s%s%s%s' % ( + '\x1b[33m' if color and c == 'y' + else '\x1b[31m' if color and c == 'r' + else '\x1b[90m' if color and c == 'b' + else '', + t, + ('>' if was else ' ') if d == depth-1 else '', + '\x1b[m' if color and c else '')) + + return '%s ' % ''.join(trunk) + +# compute the difference between two paths, returning everything +# in a after the paths diverge, as well as the relevant index +def pathdelta(a, b): + if not isinstance(a, list): + a = list(a) + i = 0 + for a_, b_ in zip(a, b): + if a_ == b_: + i += 1 + else: + break + + return [(i+j, a_) for j, a_ in enumerate(a[i:])] + + +# a simple wrapper over an open file with bd geometry +class Bd: + def __init__(self, f, block_size=None, block_count=None): + self.f = f + self.block_size = block_size + self.block_count = block_count + + def __repr__(self): + return '<%s %sx%s>' % ( + self.__class__.__name__, + self.block_size, + self.block_count) + + def read(self, size=-1): + return self.f.read(size) + + def seek(self, block, off, whence=0): + pos = self.f.seek(block*self.block_size + off, whence) + return pos // block_size, pos % block_size + + def readblock(self, block): + self.f.seek(block*self.block_size) + return self.f.read(self.block_size) + +# tagged data in an rbyd +class Rattr: + def __init__(self, tag, weight, block, toff, off, data): + self.tag = tag + self.weight = weight + self.block = block + self.toff = toff + self.off = off + self.data = data + + @property + def size(self): + return len(self.data) + + def __repr__(self): + return '<%s %s>' % (self.__class__.__name__, self.tagrepr()) + + def tagrepr(self): + return tagrepr(self.tag, self.weight, self.size) + + def __bool__(self): + return bool(self.data) + + def __len__(self): + return len(self.data) + + def __getitem__(self, key): + return self.data[key] + + def __iter__(self): + return iter(self.data) + + def __eq__(self, other): + return ((self.tag, self.weight, self.data) + == (other.tag, other.weight, other.data)) + + def __ne__(self, other): + return not self.__eq__(other) + + def __hash__(self): + return hash((self.tag, self.weight, self.data)) + +class Ralt: + def __init__(self, tag, weight, block, toff, off, jump, color=None): + self.tag = tag + self.weight = weight + self.block = block + self.toff = toff + self.off = off + self.jump = jump + + if color is not None: + self.color = color + else: + self.color = 'r' if tag & TAG_R else 'b' + + @property + def joff(self): + return self.toff - self.jump + + def __repr__(self): + return '<%s %s>' % (self.__class__.__name__, self.tagrepr()) + + def tagrepr(self): + return tagrepr(self.tag, self.weight, self.jump, self.toff) + + def __eq__(self, other): + return ((self.tag, self.weight, self.jump) + == (other.tag, other.weight, other.jump)) + + def __ne__(self, other): + return not self.__eq__(other) + + def __hash__(self): + return hash((self.tag, self.weight, self.jump)) -# this type is used for tree representations -TBranch = co.namedtuple('TBranch', 'a, b, d, c') # our core rbyd type class Rbyd: - def __init__(self, blocks, data, rev, eoff, trunk, weight, cksum, - gcksumdelta): + def __init__(self, data, blocks, trunk, weight, rev, eoff, cksum, *, + gcksumdelta=None, + corrupt=False): if isinstance(blocks, int): - blocks = (blocks,) + blocks = [blocks] - self.blocks = tuple(blocks) self.data = data - self.rev = rev - self.eoff = eoff + self.blocks = list(blocks) self.trunk = trunk self.weight = weight + self.rev = rev + self.eoff = eoff self.cksum = cksum self.gcksumdelta = gcksumdelta + self.corrupt = corrupt @property def block(self): @@ -294,15 +485,31 @@ class Rbyd: ','.join('%x' % block for block in self.blocks), self.trunk) + def __repr__(self): + return '<%s %s w%s>' % ( + self.__class__.__name__, + self.addr(), + self.weight) + + def __bool__(self): + return not self.corrupt + + def __eq__(self, other): + return ((frozenset(self.blocks), self.trunk) + == (frozenset(other.blocks), other.trunk)) + + def __ne__(self, other): + return not self.__eq__(other) + + def __hash__(self): + return hash((frozenset(self.blocks), self.trunk)) + @classmethod - def fetch(cls, f, block_size, block, trunk=None, cksum=None): - # multiple blocks? - if (not isinstance(block, int) - and not isinstance(block, Rbyd) - and len(block) > 1): + def fetch(cls, bd, blocks, trunk=None, cksum=None): + # multiple blocks? unfortunately this must be a list + if isinstance(blocks, list): # fetch all blocks - rbyds = [cls.fetch(f, block_size, block, trunk, cksum) - for block in block] + rbyds = [cls.fetch(bd, block, trunk, cksum) for block in blocks] # determine most recent revision i = 0 for i_, rbyd in enumerate(rbyds): @@ -320,26 +527,25 @@ class Rbyd: for j in range(len(rbyds)-1)) return rbyd - # block may be an rbyd, in which case we inherit the - # already-read data - # - # this helps avoid race conditions with cksums and shrubs - if isinstance(block, Rbyd): - # inherit the trunk too I guess? - if trunk is None: - trunk = block.trunk - block, data = block.block, block.data - else: - # block may encode a trunk - block = block[0] if not isinstance(block, int) else block - if isinstance(block, tuple): - if trunk is None: - trunk = block[1] - block = block[0] + block = blocks - # seek to the block - f.seek(block * block_size) - data = f.read(block_size) + # blocks may also encode trunks + block, trunk = ( + block[0] if isinstance(block, tuple) + else block, + trunk if trunk is not None + else block[1] if isinstance(block, tuple) + else None) + + # bd can be either a bd reference or preread data + # + # preread data can be useful for avoiding race conditions + # with cksums and shrubs + if isinstance(bd, Bd): + # seek/read the block + data = bd.readblock(block) + else: + data = bd # fetch the rbyd rev = fromle32(data[0:4]) @@ -376,7 +582,8 @@ class Rbyd: # found a gcksumdelta? if (tag & 0xff00) == TAG_GCKSUMDELTA: - gcksumdelta_ = (tag, w, j_-d, d, data[j_:j_+size]) + gcksumdelta_ = Rattr(tag, w, + block, j_-d, d, data[j_:j_+size]) # found a cksum? else: @@ -433,60 +640,76 @@ class Rbyd: # cksum mismatch? if cksum is not None and cksum_ != cksum: - return cls(block, data, rev, 0, 0, 0, cksum_, gcksumdelta) + return cls(data, block, 0, 0, rev, 0, cksum_, + corrupt=True) - return cls(block, data, rev, eoff, trunk_, weight, cksum_, gcksumdelta) + return cls(data, block, trunk_, weight, rev, eoff, cksum_, + gcksumdelta=gcksumdelta, + corrupt=not trunk_) - def lookup(self, rid, tag): + def lookupnext(self, rid=-1, tag=None, *, + path=False): if not self: - return True, 0, -1, 0, 0, 0, b'', [] + return None, None, None, None, *(([],) if path else ()) - tag = max(tag, 0x1) + tag = max(tag or 0, 0x1) lower = 0 upper = self.weight - path = [] + path_ = [] # descend down tree j = self.trunk while True: - _, alt, weight_, jump, d = fromtag(self.data[j:]) + _, alt, w, jump, d = fromtag(self.data[j:]) # found an alt? if alt & TAG_ALT: # follow? - if ((rid, tag & 0xfff) > (upper-weight_-1, alt & 0xfff) + if ((rid, tag & 0xfff) > (upper-w-1, alt & 0xfff) if alt & TAG_GT else ((rid, tag & 0xfff) - <= (lower+weight_-1, alt & 0xfff))): - lower += upper-lower-weight_ if alt & TAG_GT else 0 - upper -= upper-lower-weight_ if not alt & TAG_GT else 0 + <= (lower+w-1, alt & 0xfff))): + lower += upper-lower-w if alt & TAG_GT else 0 + upper -= upper-lower-w if not alt & TAG_GT else 0 j = j - jump - # figure out which color - if alt & TAG_R: - _, nalt, _, _, _ = fromtag(self.data[j+jump+d:]) - if nalt & TAG_R: - path.append((j+jump, j, True, 'y')) + if path: + # figure out which color + if alt & TAG_R: + _, nalt, _, _, _ = fromtag(self.data[j+jump+d:]) + if nalt & TAG_R: + color = 'y' + else: + color = 'r' else: - path.append((j+jump, j, True, 'r')) - else: - path.append((j+jump, j, True, 'b')) + color = 'b' + + path_.append(( + Ralt(alt, w, self.block, j+jump, j+jump+d, + jump, color), + True)) # stay on path else: - lower += weight_ if not alt & TAG_GT else 0 - upper -= weight_ if alt & TAG_GT else 0 + lower += w if not alt & TAG_GT else 0 + upper -= w if alt & TAG_GT else 0 j = j + d - # figure out which color - if alt & TAG_R: - _, nalt, _, _, _ = fromtag(self.data[j:]) - if nalt & TAG_R: - path.append((j-d, j, False, 'y')) + if path: + # figure out which color + if alt & TAG_R: + _, nalt, _, _, _ = fromtag(self.data[j:]) + if nalt & TAG_R: + color = 'y' + else: + color = 'r' else: - path.append((j-d, j, False, 'r')) - else: - path.append((j-d, j, False, 'b')) + color = 'b' + + path_.append(( + Ralt(alt, w, self.block, j-d, j, + jump, color), + False)) # found tag else: @@ -494,55 +717,146 @@ class Rbyd: tag_ = alt w_ = upper-lower - done = not tag_ or (rid_, tag_) < (rid, tag) + if not tag_ or (rid_, tag_) < (rid, tag): + return None, None, None, None, *(([],) if path else ()) - return (done, rid_, tag_, w_, j, d, - self.data[j+d:j+d+jump], - path) + return (rid_, + tag_, + w_, + Rattr(tag_, w_, self.block, j, j+d, + self.data[j+d:j+d+jump]), + *((path_,) if path else ())) - def __bool__(self): - return bool(self.trunk) + def lookup(self, rid, tag=None, mask=None, *, + path=False): + if tag is None: + tag, mask = 0, 0xffff - def __eq__(self, other): - return self.block == other.block and self.trunk == other.trunk + rid_, tag_, w_, rattr_, *path_ = self.lookupnext( + rid, tag & ~(mask or 0), + path=path) + if (rid_ is None + or rid_ != rid + or (tag_ & ~(mask or 0)) != (tag & ~(mask or 0))): + if mask is not None: + return None, None, *path_ + elif path: + return None, *path_ + else: + return None - def __ne__(self, other): - return not self.__eq__(other) + if mask is not None: + return tag_, rattr_, *path_ + elif path: + return rattr_, *path_ + else: + return rattr_ - def __iter__(self): - tag = 0 + def __getitem__(self, key): + if not isinstance(key, tuple): + key = (key,) + + return self.lookup(*key) + + def __contains__(self, key): + if not isinstance(key, tuple): + key = (key,) + + v = self.lookup(*key) + if isinstance(v, tuple): + return v[0] is not None + else: + return v is not None + + def rids(self, *, + path=False): rid = -1 - while True: - done, rid, tag, w, j, d, data, _ = self.lookup(rid, tag+0x1) - if done: + rid, tag, w, rattr, *path_ = self.lookupnext(rid, + path=path) + # found end of tree? + if rid is None: break - yield rid, tag, w, j, d, data + yield rid, tag, w, rattr, *path_ + rid += 1 + + def rattrs(self, rid=None, *, + path=False): + if rid is None: + rid, tag = -1, 0 + while True: + rid, tag, w, rattr, *path_ = self.lookupnext(rid, tag+0x1, + path=path) + # found end of tree? + if rid is None: + break + + yield rid, tag, w, rattr, *path_ + else: + tag = 0 + while True: + rid_, tag, w, rattr, *path_ = self.lookupnext(rid, tag+0x1, + path=path) + # found end of tree? + if rid_ is None or rid_ != rid: + break + + yield tag, rattr, *path_ + + def __iter__(self): + return self.rattrs() + + # lookup by name + def namelookup(self, did, name): + # binary search + best = (False, None, None, None, None) + lower = 0 + upper = self.weight + while lower < upper: + rid, tag, w, rattr = self.lookupnext( + lower + (upper-1-lower)//2) + if rid is None: + break + + # treat vestigial names as a catch-all + if ((tag == TAG_NAME and rid-(w-1) == 0) + or (tag & 0xff00) != TAG_NAME): + did_ = 0 + name_ = b'' + else: + did_, d = fromleb128(rattr[:]) + name_ = rattr[d:] + + # bisect search space + if (did_, name_) > (did, name): + upper = rid-(w-1) + elif (did_, name_) < (did, name): + lower = rid + 1 + # keep track of best match + best = (False, rid, tag, w, rattr) + else: + # found a match + return True, rid, tag, w, rattr + + return best # create tree representation for debugging - def tree(self, *, - rbyd=False): + def tree(self, **args): trunks = co.defaultdict(lambda: (-1, 0)) alts = co.defaultdict(lambda: {}) - rid, tag = -1, 0 - while True: - done, rid, tag, w, j, d, data, path = self.lookup(rid, tag+0x1) - # found end of tree? - if done: - break - + for rid, tag, w, rattr, path in self.rattrs(path=True): # keep track of trunks/alts - trunks[j] = (rid, tag) + trunks[rattr.toff] = (rid, tag) - for j_, j__, followed, c in path: + for ralt, followed in path: if followed: - alts[j_] |= {'f': j__, 'c': c} + alts[ralt.toff] |= {'f': ralt.joff, 'c': ralt.color} else: - alts[j_] |= {'nf': j__, 'c': c} + alts[ralt.toff] |= {'nf': ralt.off, 'c': ralt.color} - if rbyd: + if args.get('tree_rbyd'): # treat unreachable alts as converging paths for j_, alt in alts.items(): if 'f' not in alt: @@ -553,71 +867,469 @@ class Rbyd: else: # prune any alts with unreachable edges pruned = {} - for j_, alt in alts.items(): + for j, alt in alts.items(): if 'f' not in alt: - pruned[j_] = alt['nf'] + pruned[j] = alt['nf'] elif 'nf' not in alt: - pruned[j_] = alt['f'] - for j_ in pruned.keys(): - del alts[j_] + pruned[j] = alt['f'] + for j in pruned.keys(): + del alts[j] - for j_, alt in alts.items(): + for j, alt in alts.items(): while alt['f'] in pruned: alt['f'] = pruned[alt['f']] while alt['nf'] in pruned: alt['nf'] = pruned[alt['nf']] # find the trunk and depth of each alt - def rec_trunk(j_): - if j_ not in alts: - return trunks[j_] + def rec_trunk(j): + if j not in alts: + return trunks[j] else: - if 'nft' not in alts[j_]: - alts[j_]['nft'] = rec_trunk(alts[j_]['nf']) - return alts[j_]['nft'] + if 'nft' not in alts[j]: + alts[j]['nft'] = rec_trunk(alts[j]['nf']) + return alts[j]['nft'] - for j_ in alts.keys(): - rec_trunk(j_) - for j_, alt in alts.items(): + for j in alts.keys(): + rec_trunk(j) + for j, alt in alts.items(): if alt['f'] in alts: alt['ft'] = alts[alt['f']]['nft'] else: alt['ft'] = trunks[alt['f']] - def rec_height(j_): - if j_ not in alts: + def rec_height(j): + if j not in alts: return 0 else: - if 'h' not in alts[j_]: - alts[j_]['h'] = max( - rec_height(alts[j_]['f']), - rec_height(alts[j_]['nf'])) + 1 - return alts[j_]['h'] + if 'h' not in alts[j]: + alts[j]['h'] = max( + rec_height(alts[j]['f']), + rec_height(alts[j]['nf'])) + 1 + return alts[j]['h'] - for j_ in alts.keys(): - rec_height(j_) + for j in alts.keys(): + rec_height(j) t_depth = max((alt['h']+1 for alt in alts.values()), default=0) # convert to more general tree representation tree = set() for j, alt in alts.items(): - # note all non-trunk edges should be black - tree.add(TBranch( - a=alt['nft'], - b=alt['nft'], - d=t_depth-1 - alt['h'], - c=alt['c'], - )) + # note all non-trunk edges should be colored black + tree.add(TreeBranch( + alt['nft'], + alt['nft'], + t_depth-1 - alt['h'], + alt['c'])) if alt['ft'] != alt['nft']: - tree.add(TBranch( - a=alt['nft'], - b=alt['ft'], - d=t_depth-1 - alt['h'], - c='b', - )) + tree.add(TreeBranch( + alt['nft'], + alt['ft'], + t_depth-1 - alt['h'], + 'b')) + + return tree + + +# our rbyd btree type +class Btree: + def __init__(self, bd, rbyd, *, + corrupt=False): + self.bd = bd + self.rbyd = rbyd + self.corrupt = corrupt or rbyd.corrupt + + @property + def block(self): + return self.rbyd.block + + @property + def blocks(self): + return self.rbyd.blocks + + @property + def trunk(self): + return self.rbyd.trunk + + @property + def weight(self): + return self.rbyd.weight + + @property + def rev(self): + return self.rbyd.rev + + @property + def eoff(self): + return self.rbyd.eoff + + @property + def cksum(self): + return self.rbyd.cksum + + def addr(self): + return self.rbyd.addr() + + def __repr__(self): + return '<%s %s w%s>' % ( + self.__class__.__name__, + self.addr(), + self.weight) + + def __bool__(self): + return not self.corrupt + + def __eq__(self, other): + return self.rbyd == other.rbyd + + def __ne__(self, other): + return not self.__eq__(other) + + def __hash__(self): + return hash(self.rbyd) + + @classmethod + def fetch(cls, bd, blocks, trunk=None, cksum=None): + # we need a real bd reference here + assert isinstance(bd, Bd) + + rbyd = Rbyd.fetch(bd, blocks, trunk, cksum) + return cls(bd, rbyd) + + def lookupleaf(self, bid, *, + path=None, + depth=None): + rbyd = self.rbyd + rid = bid + depth_ = 1 + path_ = [] + + while True: + # first tag indicates the branch's weight + rid_, tag_, w_, rattr_ = rbyd.lookupnext(rid) + # if we hit this the rbyd is probably corrupt + if rid_ is None: + # still keep track of best guess in path + if path: + path_.append((bid, rbyd, rid, None, None, None)) + return (None, None, None, None, + *((path_,) if path else ())) + + # keep track of path + if path: + path_.append((bid + (rid_-rid), rbyd, rid_, tag_, w_, rattr_)) + + # find branch tag if there is one + _, branch_ = rbyd.lookup(rid_, TAG_BRANCH, 0x3) + + # descend down branch? + if branch_ is not None and ( + not depth or depth_ < depth): + block, trunk, cksum = frombranch(branch_[:]) + rbyd = Rbyd.fetch(self.bd, block, trunk, cksum) + # keep track of expected trunk/weight if corrupted + if not rbyd: + rbyd.trunk = trunk + rbyd.weight = w_ + + rid -= (rid_-(w_-1)) + depth_ += 1 + + else: + return (bid + (rid_-rid), rbyd, rid_, w_, + *((path_,) if path else ())) + + def lookup(self, bid, tag=None, mask=None, *, + path=False, + depth=None): + # lookup rbyd in btree + bid_, rbyd_, rid_, w_, *path_ = self.lookupleaf(bid, + path=path, + depth=depth) + if bid_ is None: + if mask is not None: + return None, None, None, None, *path_ + else: + return None, None, None, *path_ + + # lookup tag in rbyd + tag_, rattr_ = rbyd_.lookup(rid_, tag, mask or 0) + + # note rbyd/rid is still accessible with either path or lookupleaf + if mask is not None: + return bid_, w_, tag_, rattr_, *path_ + else: + return bid_, w_, rattr_, *path_ + + def __getitem__(self, key): + if not isinstance(key, tuple): + key = (key,) + + return self.lookup(*key) + + def __contains__(self, key): + if not isinstance(key, tuple): + key = (key,) + + return self.lookup(*key)[0] is not None + + # note leaves only iterates over leaf rbyds, whereas traverse + # traverses all rbyds + def leaves(self, *, + path=False, + depth=None): + # include our root rbyd even if the weight is zero + if self.weight == 0 and (depth is None or depth > 0): + yield -1, self.rbyd, *(([],) if path else()) + + bid = 0 + while True: + bid, rbyd, rid, w, *path_ = self.lookupleaf(bid, + path=path, + depth=depth) + if bid is None: + break + + yield (bid-rid + (rbyd.weight-1), rbyd, + *((path_[0][:-1],) if path else ())) + bid += (rbyd.weight - w) + 1 + + def traverse(self, *, + path=False, + depth=None): + ppath_ = [] + for bid, rbyd, path_ in self.leaves( + path=True, + depth=depth): + for d, (bid_, rbyd_, rid_, tag_, w_, rattr_) in pathdelta( + path_, ppath_): + yield (bid_-rid_ + (rbyd_.weight-1), rbyd_, + *((path_[:d],) if path else ())) + ppath_ = path_ + + yield bid, rbyd, *((path_,) if path else ()) + + def bids(self, *, + path=False, + depth=None): + for bid, rbyd, *path_ in self.leaves( + path=path, + depth=depth): + for rid, tag, w, rattr in rbyd.rids(): + yield (bid-(rbyd.weight-1) + rid, w, tag, rattr, + *((path_[0]+[ + (bid-(rbyd.weight-1) + rid, + rbyd, rid, tag, w, rattr)],) + if path else ())) + + def rattrs(self, bid=None, *, + path=False, + depth=None): + if bid is None: + for bid, rbyd, *path_ in self.leaves( + path=path, + depth=depth): + for rid, tag, w, rattr in rbyd.rids(): + for tag, rattr in rbyd.rattrs(rid): + yield (bid-(rbyd.weight-1) + rid, w, tag, rattr, + *((path_[0]+[ + (bid-(rbyd.weight-1) + rid, + rbyd, rid, tag, w, rattr)],) + if path else ())) + else: + bid, rbyd, rid, w, *path_ = self.lookupleaf(bid, + path=path, + depth=depth) + for tag, rattr in rbyd.rattrs(rid): + yield tag, rattr, *path_ + + def __iter__(self): + return self.rattrs() + + # lookup by name + def namelookup(self, did, name, *, + depth=None): + rbyd = self.rbyd + bid = 0 + depth_ = 1 + + while True: + found_, rid_, tag_, w_, rattr_ = rbyd.namelookup(did, name) + + # find branch tag if there is one + _, branch_ = rbyd.lookup(rid_, TAG_BRANCH, 0x3) + + # found another branch + if branch is not None and ( + not depth or depth_ < depth): + # update our bid + bid += rid_ - (w_-1) + + block, trunk, cksum = frombranch(branch[:]) + rbyd = Rbyd.fetch(self.bd, block, trunk, cksum) + # keep track of expected trunk/weight if corrupted + if not rbyd: + rbyd.trunk = trunk + rbyd.weight = w_ + + depth_ += 1 + + # found best match + else: + return found_, bid + rid_, tag_, w_, rattr_ + + # btree rbyd-tree generation for debugging + def _tree_rtree(self, *, + depth=None, + inner=False, + **args): + # precompute rbyd trees so we know the max depth at each layer + # to nicely align trees + rtrees = {} + rdepths = {} + for bid, rbyd, path in self.traverse(path=True, depth=depth): + if not rbyd: + continue + + rtrees[rbyd] = rbyd.tree(**args) + rdepths[len(path)] = max( + rdepths.get(len(path), 0), + max((t.depth+1 for t in rtrees[rbyd]), default=0)) + + # map rbyd branches into our btree space + tree = set() + for bid, rbyd, path in self.traverse(path=True, depth=depth): + if not rbyd: + continue + + # yes we can find new rbyds if disk is being mutated, just + # ignore these + if rbyd not in rtrees: + continue + + rtree = rtrees[rbyd] + rdepth = max((t.depth+1 for t in rtree), default=0) + d = sum(rdepths[d]+1 for d in range(len(path))) + + for t in rtree: + # note we adjust our bid to be left-leaning, this allows a + # global order and make tree rendering quite a bit easier + a_rid, a_tag = t.a + b_rid, b_tag = t.b + _, _, a_w, _ = rbyd.lookupnext(a_rid) + _, _, b_w, _ = rbyd.lookupnext(b_rid) + tree.add(TreeBranch( + (bid-(rbyd.weight-1)+a_rid-(a_w-1), len(path), a_tag), + (bid-(rbyd.weight-1)+b_rid-(b_w-1), len(path), b_tag), + d + rdepths[len(path)]-rdepth + t.depth, + t.color)) + + # connect rbyd branches to rbyd roots + if path: + l_bid, l_rbyd, l_rid, l_tag, l_w, l_rattr = path[-1] + l_tag, l_rattr = l_rbyd.lookup(l_rid, TAG_BRANCH, 0x3) + + if rtree: + r_rid, r_tag = min(rtree, key=lambda t: t.depth).a + _, _, r_w, _ = rbyd.lookupnext(r_rid) + else: + r_rid, r_tag, r_w, _ = rbyd.lookupnext() + + tree.add(TreeBranch( + (l_bid-(l_w-1), len(path)-1, l_tag), + (bid-(rbyd.weight-1)+r_rid-(r_w-1), len(path), r_tag), + d-1)) + + # remap branches to leaves if we aren't showing inner branches + if not inner: + # step through each btree layer backwards + b_depth = max((t.a[1]+1 for t in tree), default=0) + + for d in reversed(range(b_depth-1)): + # find bid ranges at this level + bids = set() + for t in tree: + if t.a[1] == d: + bids.add(t.a[0]) + bids = sorted(bids) + + # find the best root for each bid range + roots = {} + for i in range(len(bids)): + for t in tree: + if (t.b[1] > d + and t.b[0] >= bids[i] + and (i == len(bids)-1 or t.b[0] < bids[i+1]) + and (bids[i] not in roots + or t.depth < roots[bids[i]].depth)): + roots[bids[i]] = t + + # remap branches to leaf-roots + tree_ = set() + for t in tree: + if t.a[1] == d and t.a[0] in roots: + t = TreeBranch( + roots[t.a[0]].b, + t.b, + t.depth, + t.color) + if t.b[1] == d and t.b[0] in roots: + t = TreeBranch( + t.a, + roots[t.b[0]].b, + t.depth, + t.color) + tree_.add(t) + tree = tree_ + + return tree + + # btree btree generation for debugging + def _tree_btree(self, *, + depth=None, + inner=False, + **args): + # find all branches + tree = set() + root = None + branches = {} + for bid, w, tag, rattr, path in self.bids( + path=True, + depth=depth): + # create branch for each jump in path + # + # note we adjust our bid to be left-leaning, this allows a + # global order and make tree rendering quite a bit easier + # + a = root + for d, (bid_, rbyd_, rid_, tag_, w_, rattr_) in enumerate(path): + b = (bid_-(w_-1), d, tag_) + + # remap branches to leaves if we aren't showing inner + # branches + if not inner: + if b not in branches: + bid_, rbyd_, rid_, tag_, w_, rattr_ = path[-1] + branches[b] = (bid_-(w_-1), len(path)-1, tag_) + b = branches[b] + + # render the root path on first rid, this is arbitrary + if root is None: + root, a = b, b + + tree.add(TreeBranch(a, b, d)) + a = b + + return tree + + # create tree representation for debugging + def tree(self, **args): + if args.get('tree_btree'): + return self._tree_btree(**args) + else: + return self._tree_rtree(**args) - return tree, t_depth def main(disk, roots=None, *, @@ -652,342 +1364,53 @@ def main(disk, roots=None, *, f.seek(0, os.SEEK_END) block_size = f.tell() - # fetch the root - btree = Rbyd.fetch(f, block_size, roots, trunk) + # fetch the btree + bd = Bd(f, block_size, block_count) + btree = Btree.fetch(bd, roots, trunk) print('btree %s w%d, rev %08x, cksum %08x' % ( btree.addr(), btree.weight, btree.rev, btree.cksum)) - # look up a bid, while keeping track of the search path - def btree_lookup(bid, *, - depth=None): - rbyd = btree - rid = bid - depth_ = 1 - path = [] - - # corrupted? return a corrupted block once - if not rbyd: - return bid > 0, bid, 0, rbyd, -1, [], path - - while True: - # collect all tags, normally you don't need to do this - # but we are debugging here - name = None - tags = [] - branch = None - rid_ = rid - tag = 0 - w = 0 - for i in it.count(): - done, rid__, tag, w_, j, d, data, _ = rbyd.lookup( - rid_, tag+0x1) - if done or (i != 0 and rid__ != rid_): - break - - # first tag indicates the branch's weight - if i == 0: - rid_, w = rid__, w_ - - # catch any branches - if tag & 0xfff == TAG_BRANCH: - branch = (tag, j, d, data) - - tags.append((tag, j, d, data)) - - # keep track of path - path.append((bid + (rid_-rid), w, rbyd, rid_, tags)) - - # descend down branch? - if branch is not None and ( - not depth or depth_ < depth): - tag, j, d, data = branch - block, trunk, cksum = frombranch(data) - rbyd = Rbyd.fetch(f, block_size, block, trunk, cksum) - - # corrupted? bail here so we can keep traversing the tree - if not rbyd: - return False, bid + (rid_-rid), w, rbyd, -1, [], path - - rid -= (rid_-(w-1)) - depth_ += 1 - else: - return not tags, bid + (rid_-rid), w, rbyd, rid_, tags, path - - # precompute rbyd-trees if requested + # precompute trees if requested t_width = 0 - if args.get('tree') or args.get('rbyd'): - # find the max depth of each layer to nicely align trees - bdepths = {} - bid = -1 - while True: - done, bid, w, rbyd, rid, tags, path = btree_lookup( - bid+1, depth=args.get('depth')) - if done: - break + if (args.get('tree') + or args.get('tree_rbyd') + or args.get('tree_btree')): + tree = btree.tree(**args) - for d, (bid, w, rbyd, rid, tags) in enumerate(path): - _, rdepth = rbyd.tree(rbyd=args.get('rbyd')) - bdepths[d] = max(bdepths.get(d, 0), rdepth) - - # find all branches - tree = set() - root = None - branches = {} - bid = -1 - while True: - done, bid, w, rbyd, rid, tags, path = btree_lookup( - bid+1, depth=args.get('depth')) - if done: - break - - d_ = 0 - leaf = None - for d, (bid, w, rbyd, rid, tags) in enumerate(path): - if not tags: - continue - - # map rbyd tree into B-tree space - rtree, rdepth = rbyd.tree(rbyd=args.get('rbyd')) - - # note we adjust our bid/rids to be left-leaning, - # this allows a global order and make tree rendering quite - # a bit easier - rtree_ = set() - for branch in rtree: - a_rid, a_tag = branch.a - b_rid, b_tag = branch.b - _, _, _, a_w, _, _, _, _ = rbyd.lookup(a_rid, 0) - _, _, _, b_w, _, _, _, _ = rbyd.lookup(b_rid, 0) - rtree_.add(TBranch( - a=(a_rid-(a_w-1), a_tag), - b=(b_rid-(b_w-1), b_tag), - d=branch.d, - c=branch.c, - )) - rtree = rtree_ - - # connect our branch to the rbyd's root - if leaf is not None: - root = min(rtree, - key=lambda branch: branch.d, - default=None) - - if root is not None: - r_rid, r_tag = root.a - else: - r_rid, r_tag = rid-(w-1), tags[0][0] - tree.add(TBranch( - a=leaf, - b=(bid-rid+r_rid, d, r_rid, r_tag), - d=d_-1, - c='b', - )) - - for branch in rtree: - # map rbyd branches into our btree space - a_rid, a_tag = branch.a - b_rid, b_tag = branch.b - tree.add(TBranch( - a=(bid-rid+a_rid, d, a_rid, a_tag), - b=(bid-rid+b_rid, d, b_rid, b_tag), - d=branch.d + d_ + bdepths.get(d, 0)-rdepth, - c=branch.c, - )) - - d_ += max(bdepths.get(d, 0), 1) - leaf = (bid-(w-1), d, rid-(w-1), - next( - (tag for tag, _, _, _ in tags - if tag & 0xfff == TAG_BRANCH), - TAG_BRANCH)) - - # remap branches to leaves if we aren't showing inner branches - if not args.get('inner'): - # step through each layer backwards - b_depth = max((branch.b[1]+1 for branch in tree), default=0) - - # keep track of the original bids, unfortunately because we - # store the bids in the branches we overwrite these - tree = {(branch.b[0] - branch.b[2], branch) for branch in tree} - - for bd in reversed(range(b_depth-1)): - # find leaf-roots at this level - roots = {} - for bid, branch in tree: - # choose the highest node as the root - if (branch.b[1] == b_depth-1 - and (bid not in roots - or branch.d < roots[bid].d)): - roots[bid] = branch - - # remap branches to leaf-roots - tree_ = set() - for bid, branch in tree: - if branch.a[1] == bd and branch.a[0] in roots: - branch = TBranch( - a=roots[branch.a[0]].b, - b=branch.b, - d=branch.d, - c=branch.c, - ) - if branch.b[1] == bd and branch.b[0] in roots: - branch = TBranch( - a=branch.a, - b=roots[branch.b[0]].b, - d=branch.d, - c=branch.c, - ) - tree_.add((bid, branch)) - tree = tree_ - - # strip out bids - tree = {branch for _, branch in tree} - - # precompute B-trees if requested - elif args.get('btree'): - # find all branches - tree = set() - root = None - branches = {} - bid = -1 - while True: - done, bid, w, rbyd, rid, tags, path = btree_lookup( - bid+1, depth=args.get('depth')) - if done: - break - - # if we're not showing inner nodes, prefer names higher in - # the tree since this avoids showing vestigial names - name = None - if not args.get('inner'): - name = None - for bid_, w_, rbyd_, rid_, tags_ in reversed(path): - for tag_, j_, d_, data_ in tags_: - if tag_ & 0x7f00 == TAG_NAME: - name = (tag_, j_, d_, data_) - - if rid_-(w_-1) != 0: - break - - a = root - for d, (bid, w, rbyd, rid, tags) in enumerate(path): - if not tags: - continue - - b = (bid-(w-1), d, rid-(w-1), - (name if name else tags[0])[0]) - - # remap branches to leaves if we aren't showing - # inner branches - if not args.get('inner'): - if b not in branches: - bid, w, rbyd, rid, tags = path[-1] - if not tags: - continue - branches[b] = ( - bid-(w-1), len(path)-1, rid-(w-1), - (name if name else tags[0])[0]) - b = branches[b] - - # found entry point? - if root is None: - root = b - a = root - - tree.add(TBranch( - a=a, - b=b, - d=d, - c='b', - )) - a = b - - # common tree renderer - if args.get('tree') or args.get('rbyd') or args.get('btree'): # find the max depth from the tree - t_depth = max((branch.d+1 for branch in tree), default=0) + t_depth = max((t.depth+1 for t in tree), default=0) if t_depth > 0: t_width = 2*t_depth + 2 - def treerepr(bid, w, bd, rid, tag): - if t_depth == 0: - return '' - - def branchrepr(x, d, was): - for branch in tree: - if branch.d == d and branch.b == x: - if any(branch.d == d and branch.a == x - for branch in tree): - return '+-', branch.c, branch.c - elif any(branch.d == d - and x > min(branch.a, branch.b) - and x < max(branch.a, branch.b) - for branch in tree): - return '|-', branch.c, branch.c - elif branch.a < branch.b: - return '\'-', branch.c, branch.c - else: - return '.-', branch.c, branch.c - for branch in tree: - if branch.d == d and branch.a == x: - return '+ ', branch.c, None - for branch in tree: - if (branch.d == d - and x > min(branch.a, branch.b) - and x < max(branch.a, branch.b)): - return '| ', branch.c, was - if was: - return '--', was, was - return ' ', None, None - - trunk = [] - was = None - for d in range(t_depth): - t, c, was = branchrepr( - (bid-(w-1), bd, rid-(w-1), tag), d, was) - - trunk.append('%s%s%s%s' % ( - '\x1b[33m' if color and c == 'y' - else '\x1b[31m' if color and c == 'r' - else '\x1b[90m' if color and c == 'b' - else '', - t, - ('>' if was else ' ') if d == t_depth-1 else '', - '\x1b[m' if color and c else '')) - - return '%s ' % ''.join(trunk) - - # dynamically size the id field w_width = mt.ceil(mt.log10(max(1, btree.weight)+1)) # prbyd here means the last rendered rbyd, we update # in dbg_branch to always print interleaved addresses prbyd = None - def dbg_branch(bid, w, rbyd, rid, tags, bd): + def dbg_branch(d, bid, rbyd, rid, w): nonlocal prbyd # show human-readable representation - for i, (tag, j, d, data) in enumerate(tags): + for i, (tag, rattr) in enumerate(rbyd.rattrs(rid)): print('%10s %s%*s %-*s %s' % ( '%04x.%04x:' % (rbyd.block, rbyd.trunk) if prbyd is None or rbyd != prbyd else '', - treerepr(bid, w, bd, rid, tag) + treerepr(tree, (bid-(w-1), d, tag), t_depth, color) if args.get('tree') - or args.get('rbyd') - or args.get('btree') + or args.get('tree_rbyd') + or args.get('tree_btree') else '', 2*w_width+1, '' if i != 0 else '%d-%d' % (bid-(w-1), bid) if w > 1 else bid if w > 0 else '', - 21+w_width, tagrepr( - tag, w if i == 0 else 0, len(data), None), - next(xxd(data, 8), '') + 21+w_width, rattr.tagrepr(), + next(xxd(rattr[:], 8), '') if not args.get('raw') and not args.get('no_truncate') else '')) @@ -995,49 +1418,28 @@ def main(disk, roots=None, *, # show on-disk encoding of tags/data if args.get('raw'): - for o, line in enumerate(xxd(rbyd.data[j:j+d])): + for o, line in enumerate(xxd( + rbyd.data[rattr.toff:rattr.off])): print('%9s: %*s%*s %s' % ( '%04x' % (j + o*16), t_width, '', 2*w_width+1, '', line)) if args.get('raw') or args.get('no_truncate'): - for o, line in enumerate(xxd(data)): + for o, line in enumerate(xxd(rattr[:])): print('%9s: %*s%*s %s' % ( - '%04x' % (j+d + o*16), + '%04x' % (rattr.off + o*16), t_width, '', 2*w_width+1, '', line)) - # traverse and print entries - bid = -1 prbyd = None ppath = [] corrupted = False - while True: - done, bid, w, rbyd, rid, tags, path = btree_lookup( - bid+1, depth=args.get('depth')) - if done: - break - - # print inner btree entries if requested - if args.get('inner'): - changed = False - for (x, px) in it.zip_longest( - enumerate(path[:-1]), - enumerate(ppath[:-1])): - if x is None: - break - if not (changed or px is None or x != px): - continue - changed = True - - # show the inner entry - d, (bid_, w_, rbyd_, rid_, tags_) = x - dbg_branch(bid_, w_, rbyd_, rid_, tags_, d) - ppath = path - + for bid, rbyd, path in btree.leaves( + path=True, + depth=args.get('depth')): # corrupted? try to keep printing the tree if not rbyd: print('%04x.%04x: %*s%s%s%s' % ( @@ -1050,25 +1452,17 @@ def main(disk, roots=None, *, corrupted = True continue - # if we're not showing inner nodes, prefer names higher in the tree - # since this avoids showing vestigial names - if not args.get('inner'): - name = None - for bid_, w_, rbyd_, rid_, tags_ in reversed(path): - for tag_, j_, d_, data_ in tags_: - if tag_ & 0x7f00 == TAG_NAME: - name = (tag_, j_, d_, data_) + # print inner btree entries if requested + if args.get('inner'): + for d, (bid_, rbyd_, rid_, tag_, w_, rattr_) in pathdelta( + path, ppath): + dbg_branch(d, bid_, rbyd_, rid_, w_) + ppath = path - if rid_-(w_-1) != 0: - break - - if name is not None: - tags = [name] + [(tag, j, d, data) - for tag, j, d, data in tags - if tag & 0x7f00 != TAG_NAME] - - # show the branch - dbg_branch(bid, w, rbyd, rid, tags, len(path)-1) + for rid, tag, w, rattr in rbyd.rids(): + bid_ = bid-(rbyd.weight-1) + rid + # show the branch + dbg_branch(len(path), bid_, rbyd, rid, w) if args.get('error_on_corrupt') and corrupted: sys.exit(2) @@ -1118,11 +1512,11 @@ if __name__ == "__main__": action='store_true', help="Show the underlying rbyd trees.") parser.add_argument( - '-B', '--btree', + '-B', '--tree-btree', action='store_true', help="Show the B-tree.") parser.add_argument( - '-R', '--rbyd', + '-R', '--tree-rbyd', action='store_true', help="Show the full underlying rbyd trees.") parser.add_argument( diff --git a/scripts/dbgrbyd.py b/scripts/dbgrbyd.py index f441bab0..8b1e5614 100755 --- a/scripts/dbgrbyd.py +++ b/scripts/dbgrbyd.py @@ -265,6 +265,108 @@ def tagrepr(tag, weight=None, size=None, off=None): ' w%d' % weight if weight is not None else '', ' %d' % size if size is not None else '') +# tree branches are an abstract thing for tree rendering +class TreeBranch: + def __init__(self, a, b, depth=0, color='b'): + # a and b are context specific + self.a = a + self.b = b + self.depth = depth + self.color = color + + def __repr__(self): + return '%s(%s, %s, %s, %s)' % ( + self.__class__.__name__, + self.a, + self.b, + self.depth, + self.color) + + def __eq__(self, other): + return ((self.a, self.b, self.depth, self.color) + == (other.a, other.b, other.depth, other.color)) + + def __ne__(self, other): + return not self.__eq__(other) + + def __hash__(self): + return hash((self.a, self.b, self.depth)) + +def treerepr(tree, x, depth=None, color=False): + # find the max depth from the tree + if depth is None: + depth = max((t.depth+1 for t in tree), default=0) + if depth == 0: + return '' + + def branchrepr(tree, x, d, was): + for t in tree: + if t.depth == d and t.b == x: + if any(t.depth == d and t.a == x + for t in tree): + return '+-', t.color, t.color + elif any(t.depth == d + and x > min(t.a, t.b) + and x < max(t.a, t.b) + for t in tree): + return '|-', t.color, t.color + elif t.a < t.b: + return '\'-', t.color, t.color + else: + return '.-', t.color, t.color + for t in tree: + if t.depth == d and t.a == x: + return '+ ', t.color, None + for t in tree: + if (t.depth == d + and x > min(t.a, t.b) + and x < max(t.a, t.b)): + return '| ', t.color, was + if was: + return '--', was, was + return ' ', None, None + + trunk = [] + was = None + for d in range(depth): + t, c, was = branchrepr(tree, x, d, was) + + trunk.append('%s%s%s%s' % ( + '\x1b[33m' if color and c == 'y' + else '\x1b[31m' if color and c == 'r' + else '\x1b[90m' if color and c == 'b' + else '', + t, + ('>' if was else ' ') if d == depth-1 else '', + '\x1b[m' if color and c else '')) + + return '%s ' % ''.join(trunk) + + +# a simple wrapper over an open file with bd geometry +class Bd: + def __init__(self, f, block_size=None, block_count=None): + self.f = f + self.block_size = block_size + self.block_count = block_count + + def __repr__(self): + return '<%s %sx%s>' % ( + self.__class__.__name__, + self.block_size, + self.block_count) + + def read(self, size=-1): + return self.f.read(size) + + def seek(self, block, off, whence=0): + pos = self.f.seek(block*self.block_size + off, whence) + return pos // block_size, pos % block_size + + def readblock(self, block): + self.f.seek(block*self.block_size) + return self.f.read(self.block_size) + # tagged data in an rbyd class Rattr: def __init__(self, tag, weight, block, toff, off, data): @@ -297,6 +399,16 @@ class Rattr: def __iter__(self): return iter(self.data) + def __eq__(self, other): + return ((self.tag, self.weight, self.data) + == (other.tag, other.weight, other.data)) + + def __ne__(self, other): + return not self.__eq__(other) + + def __hash__(self): + return hash((self.tag, self.weight, self.data)) + class Ralt: def __init__(self, tag, weight, block, toff, off, jump, color=None): self.tag = tag @@ -321,22 +433,16 @@ class Ralt: def tagrepr(self): return tagrepr(self.tag, self.weight, self.jump, self.toff) -# tree branches are an abstract thing for tree rendering -class TreeBranch: - def __init__(self, a, b, depth, color): - # note a and b are context specific - self.a = a - self.b = b - self.depth = depth - self.color = color + def __eq__(self, other): + return ((self.tag, self.weight, self.jump) + == (other.tag, other.weight, other.jump)) + + def __ne__(self, other): + return not self.__eq__(other) + + def __hash__(self): + return hash((self.tag, self.weight, self.jump)) - def __repr__(self): - return '%s(%s, %s, %s, %s)' % ( - self.__class__.__name__, - self.a, - self.b, - self.depth, - self.color) # our core rbyd type class Rbyd: @@ -369,26 +475,30 @@ class Rbyd: self.trunk) def __repr__(self): - return '<%s %s>' % (self.__class__.__name__, self.addr()) + return '<%s %s w%s>' % ( + self.__class__.__name__, + self.addr(), + self.weight) def __bool__(self): return not self.corrupt def __eq__(self, other): - return (self.blocks, self.trunk) == (other.blocks, other.trunk) + return ((frozenset(self.blocks), self.trunk) + == (frozenset(other.blocks), other.trunk)) def __ne__(self, other): return not self.__eq__(other) def __hash__(self): - return hash((self.blocks, self.trunk)) + return hash((frozenset(self.blocks), self.trunk)) @classmethod - def fetch(cls, data, block, trunk=None, cksum=None): + def fetch(cls, bd, blocks, trunk=None, cksum=None): # multiple blocks? unfortunately this must be a list - if isinstance(block, list): + if isinstance(blocks, list): # fetch all blocks - rbyds = [cls.fetch(data, block, trunk, cksum) for block in block] + rbyds = [cls.fetch(bd, block, trunk, cksum) for block in blocks] # determine most recent revision i = 0 for i_, rbyd in enumerate(rbyds): @@ -406,7 +516,9 @@ class Rbyd: for j in range(len(rbyds)-1)) return rbyd - # block may encode a trunk + block = blocks + + # blocks may also encode trunks block, trunk = ( block[0] if isinstance(block, tuple) else block, @@ -414,15 +526,15 @@ class Rbyd: else block[1] if isinstance(block, tuple) else None) - # data can be either disk + block_size tuple or data + # bd can be either a bd reference or preread data # - # note preread data can be useful for avoiding race conditions + # preread data can be useful for avoiding race conditions # with cksums and shrubs - if isinstance(data, tuple): - f, block_size, *_ = data - # seek to the block - f.seek(block * block_size) - data = f.read(block_size) + if isinstance(bd, Bd): + # seek/read the block + data = bd.readblock(block) + else: + data = bd # fetch the rbyd rev = fromle32(data[0:4]) @@ -517,17 +629,17 @@ class Rbyd: # cksum mismatch? if cksum is not None and cksum_ != cksum: - return cls(data, block, trunk or 0, 0, rev, 0, cksum_, + return cls(data, block, 0, 0, rev, 0, cksum_, corrupt=True) return cls(data, block, trunk_, weight, rev, eoff, cksum_, gcksumdelta=gcksumdelta, corrupt=not trunk_) - def lookupnext(self, rid, tag=None, *, + def lookupnext(self, rid=-1, tag=None, *, path=False): if not self: - return None, None, None, *(([],) if path else ()) + return None, None, None, None, *(([],) if path else ()) tag = max(tag or 0, 0x1) lower = 0 @@ -595,9 +707,11 @@ class Rbyd: w_ = upper-lower if not tag_ or (rid_, tag_) < (rid, tag): - return None, None, None, *(([],) if path else ()) + return None, None, None, None, *(([],) if path else ()) - return (rid_, tag_, + return (rid_, + tag_, + w_, Rattr(tag_, w_, self.block, j, j+d, self.data[j+d:j+d+jump]), *((path_,) if path else ())) @@ -607,7 +721,8 @@ class Rbyd: if tag is None: tag, mask = 0, 0xffff - rid_, tag_, rattr_, *path_ = self.lookupnext(rid, tag & ~(mask or 0), + rid_, tag_, w_, rattr_, *path_ = self.lookupnext( + rid, tag & ~(mask or 0), path=path) if (rid_ is None or rid_ != rid @@ -642,29 +757,59 @@ class Rbyd: else: return v is not None - def __iter__(self): - rid, tag = -1, 0 + def rids(self, *, + path=False): + rid = -1 while True: - rid, tag, rattr = self.lookupnext(rid, tag+0x1) + rid, tag, w, rattr, *path_ = self.lookupnext(rid, + path=path) # found end of tree? if rid is None: break - yield rid, tag, rattr + yield rid, tag, w, rattr, *path_ + rid += 1 + + def rattrs(self, rid=None, *, + path=False): + if rid is None: + rid, tag = -1, 0 + while True: + rid, tag, w, rattr, *path_ = self.lookupnext(rid, tag+0x1, + path=path) + # found end of tree? + if rid is None: + break + + yield rid, tag, w, rattr, *path_ + else: + tag = 0 + while True: + rid_, tag, w, rattr, *path_ = self.lookupnext(rid, tag+0x1, + path=path) + # found end of tree? + if rid_ is None or rid_ != rid: + break + + yield tag, rattr, *path_ + + def __iter__(self): + return self.rattrs() # lookup by name def namelookup(self, did, name): # binary search - best = (False, None, None, None) + best = (False, None, None, None, None) lower = 0 upper = self.weight while lower < upper: - rid, tag, rattr = self.lookupnext(lower + (upper-1-lower)//2) + rid, tag, w, rattr = self.lookupnext( + lower + (upper-1-lower)//2) if rid is None: break # treat vestigial names as a catch-all - if ((tag == TAG_NAME and rid-(rattr.weight-1) == 0) + if ((tag == TAG_NAME and rid-(w-1) == 0) or (tag & 0xff00) != TAG_NAME): did_ = 0 name_ = b'' @@ -674,31 +819,23 @@ class Rbyd: # bisect search space if (did_, name_) > (did, name): - upper = rid-(rattr.weight-1) + upper = rid-(w-1) elif (did_, name_) < (did, name): lower = rid + 1 # keep track of best match - best = (False, rid, tag, rattr) + best = (False, rid, tag, w, rattr) else: # found a match - return True, rid, tag, rattr + return True, rid, tag, w, rattr return best # create tree representation for debugging - def tree(self, *, - rbyd=False): + def tree(self, **args): trunks = co.defaultdict(lambda: (-1, 0)) alts = co.defaultdict(lambda: {}) - rid, tag = -1, 0 - while True: - rid, tag, rattr, path = self.lookupnext(rid, tag+0x1, - path=True) - # found end of tree? - if rid is None: - break - + for rid, tag, w, rattr, path in self.rattrs(path=True): # keep track of trunks/alts trunks[rattr.toff] = (rid, tag) @@ -708,7 +845,7 @@ class Rbyd: else: alts[ralt.toff] |= {'nf': ralt.off, 'c': ralt.color} - if rbyd: + if args.get('tree_rbyd'): # treat unreachable alts as converging paths for j_, alt in alts.items(): if 'f' not in alt: @@ -772,14 +909,14 @@ class Rbyd: tree.add(TreeBranch( alt['nft'], alt['nft'], - depth=t_depth-1 - alt['h'], - color=alt['c'])) + t_depth-1 - alt['h'], + alt['c'])) if alt['ft'] != alt['nft']: tree.add(TreeBranch( alt['nft'], alt['ft'], - depth=t_depth-1 - alt['h'], - color='b')) + t_depth-1 - alt['h'], + 'b')) return tree @@ -1164,74 +1301,25 @@ def dbg_tree(rbyd, *, # precompute tree t_width = 0 if args.get('tree') or args.get('tree_rbyd'): - tree = rbyd.tree(rbyd=args.get('tree_rbyd')) + tree = rbyd.tree(**args) # find the max depth from the tree - t_depth = max((b.depth+1 for b in tree), default=0) + t_depth = max((t.depth+1 for t in tree), default=0) if t_depth > 0: t_width = 2*t_depth + 2 - def treerepr(rid, tag): - if t_depth == 0: - return '' - - def branchrepr(x, d, was): - for b in tree: - if b.depth == d and b.b == x: - if any(b.depth == d and b.a == x - for b in tree): - return '+-', b.color, b.color - elif any(b.depth == d - and x > min(b.a, b.b) - and x < max(b.a, b.b) - for b in tree): - return '|-', b.color, b.color - elif b.a < b.b: - return '\'-', b.color, b.color - else: - return '.-', b.color, b.color - for b in tree: - if b.depth == d and b.a == x: - return '+ ', b.color, None - for b in tree: - if (b.depth == d - and x > min(b.a, b.b) - and x < max(b.a, b.b)): - return '| ', b.color, was - if was: - return '--', was, was - return ' ', None, None - - trunk = [] - was = None - for d in range(t_depth): - t, c, was = branchrepr((rid, tag), d, was) - - trunk.append('%s%s%s%s' % ( - '\x1b[33m' if color and c == 'y' - else '\x1b[31m' if color and c == 'r' - else '\x1b[90m' if color and c == 'b' - else '', - t, - ('>' if was else ' ') if d == t_depth-1 else '', - '\x1b[m' if color and c else '')) - - return '%s ' % ''.join(trunk) - - # dynamically size the id field w_width = mt.ceil(mt.log10(max(1, rbyd.weight)+1)) - for i, (rid, tag, rattr) in enumerate(rbyd): + for i, (rid, tag, w, rattr) in enumerate(rbyd): # show human-readable tag representation print('%08x: %s%*s %-*s %s' % ( rattr.toff, - treerepr(rid, tag) + treerepr(tree, (rid, tag), t_depth, color) if args.get('tree') or args.get('tree_rbyd') else '', - 2*w_width+1, '%d-%d' % (rid-(rattr.weight-1), rid) - if rattr.weight > 1 - else rid if rattr.weight > 0 or i == 0 + 2*w_width+1, '%d-%d' % (rid-(w-1), rid) if w > 1 + else rid if w > 0 or i == 0 else '', 21+w_width, rattr.tagrepr(), next(xxd(rattr[:8], 8), '') @@ -1290,7 +1378,8 @@ def main(disk, blocks=None, *, block_size = f.tell() # fetch the rbyd - rbyd = Rbyd.fetch((f, block_size), blocks) + bd = Bd(f, block_size, block_count) + rbyd = Rbyd.fetch(bd, blocks) print('rbyd %s w%d, rev %08x, size %d, cksum %08x' % ( rbyd.addr(),