Adopted scaling of did hashes based on the number of metadata entries

Instead of truncating to exactly 28-bits for nice leb128 alignment, we
now truncate to ~the number of metadata entries, which must be >= ~2x
the number dids since each did needs a dir entry and dstart entry.

This has the downside of needing to actually keep track of an estimate
of the number of metadata entries, which is made a bit difficult due to
integer overflow issues (we can have more than 2^32 metadata entries),
but has the upside of allowing a full 2^32 number of dids worst case.
This is really unlikely, but it's nice to not need another configuration
option to control the did limit.

Another option would be to scale the hashes based on the number dids,
which would be a more direct solution. Unfortunately determining the
number of dids during mount requires a O(m*log(m)) scan of each rbyd
to find either dir entries or dstart entries. This solution can easily
end up with an overestimate, but only needs to weight of each rbyd which
can be (and already is) found in O(m).
This commit is contained in:
Christopher Haster
2023-07-19 01:40:02 -05:00
parent c5e84e874f
commit 389987ee4f
2 changed files with 29 additions and 17 deletions
+28 -17
View File
@@ -1541,6 +1541,7 @@ static inline lfs_size_t lfsr_gdelta_size(
static int lfsr_gdelta_xor(lfs_t *lfs,
uint8_t *gdelta, lfs_size_t size,
lfsr_data_t xor) {
(void)size;
// expect xor to fit
LFS_ASSERT(lfsr_data_size(xor) <= size);
@@ -5813,7 +5814,7 @@ static int lfsr_mdir_commit(lfs_t *lfs, lfsr_mdir_t *mdir, lfs_ssize_t *rid,
// handle possible mtree updates, this gets a bit messy
lfsr_mdir_t mroot_ = lfs->mroot;
lfsr_btree_t mtree_ = lfs->mtree;
lfsr_mdir_t msibling_;
lfsr_mdir_t msibling_ = {.mid=LFSR_MID_RM, .rbyd.weight = 0};
bool dirtymroot = false;
bool dirtymtree = false;
@@ -6181,6 +6182,9 @@ static int lfsr_mdir_commit(lfs_t *lfs, lfsr_mdir_t *mdir, lfs_ssize_t *rid,
// update our mroot and mtree
lfs->mroot = mroot_;
lfs->mtree = mtree_;
lfs->mlimit = lfs_max32(lfs->mlimit, lfs_max32(
mdir_.rbyd.weight,
msibling_.rbyd.weight));
return 0;
}
@@ -6345,12 +6349,6 @@ static int lfsr_mtree_pathlookup(lfs_t *lfs, const char *path,
if (d < 0) {
return d;
}
// TODO should we put this in lfsr_data_readleb128?
// TODO should readtag then call lfsr_data_readleb128?
if (did > 0x7fffffff) {
return LFS_ERR_CORRUPT;
}
}
// lookup up this dname in the mtree
@@ -6436,6 +6434,10 @@ static int lfsr_mtree_traversal_next(lfs_t *lfs,
return err;
}
// keep track of the largest mdir we've seen, reset here
// since this is the first mdir we see
lfs->mlimit = traversal->mdir.rbyd.weight;
if (mid_) {
*mid_ = -1;
}
@@ -6473,6 +6475,9 @@ static int lfsr_mtree_traversal_next(lfs_t *lfs,
return err;
}
// keep track of the largest mdir we've seen
lfs->mlimit = lfs_max32(lfs->mlimit, traversal->mdir.rbyd.weight);
if (mid_) {
*mid_ = -1;
}
@@ -6591,6 +6596,9 @@ static int lfsr_mtree_traversal_next(lfs_t *lfs,
return err;
}
// keep track of the largest mdir we've seen
lfs->mlimit = lfs_max32(lfs->mlimit, traversal->mdir.rbyd.weight);
if (mid_) {
*mid_ = mid;
}
@@ -7255,11 +7263,20 @@ int lfsr_mkdir(lfs_t *lfs, const char *path) {
// hopefully few collisions, we use a hash of the full path. Since
// we have a CRC handy, we can use that.
//
// We truncate to 28-bits to more optimally use our leb128 encoding.
// TODO should we have a configurable limit for this? dir_limit? or
// just loose ~3 bits from the configured file limit?
// We also truncate to make better use of our leb128 encoding. This is
// pretty arbitrary, but we don't want collisions, so we truncate to
// our best >= estimate of the number of metadata entries. Worst case
// this is >= ~2x the number of dids in the system, since each dir needs
// a dir entry and dstart entry.
//
lfs_size_t did = lfs_crc32c(0, path, strlen(path)) & 0xfffffff;
lfs_size_t did = lfs_crc32c(0, path, strlen(path))
// This complicated bit of logic determines the upper limit of
// metadata entries, limited to 2^32. This is done post-log to
// try to avoid issues with integer overflow.
& ((1 << lfs_min32(
lfs_nlog2(lfsr_mtree_weight(lfs))
+ lfs_nlog2(lfs->mlimit),
32)) - 1);
// Check if we have a collision. If we do, search for the next
// available did
@@ -7360,12 +7377,6 @@ int lfsr_dir_open(lfs_t *lfs, lfsr_dir_t *dir, const char *path) {
if (d < 0) {
return d;
}
// TODO should we put this in lfsr_data_readleb128?
// TODO should readtag then call lfsr_data_readleb128?
if (did > 0x7fffffff) {
return LFS_ERR_CORRUPT;
}
}
// now lookup our dstart in the mtree
+1
View File
@@ -492,6 +492,7 @@ typedef struct lfs {
// begin lfsr things
lfsr_mdir_t mroot;
lfsr_btree_t mtree;
lfs_size_t mlimit;
// TODO do we really need separate decoded/encoded grms?
lfsr_grm_t grm_;