Adopted scaling of did hashes based on the number of metadata entries
Instead of truncating to exactly 28-bits for nice leb128 alignment, we now truncate to ~the number of metadata entries, which must be >= ~2x the number dids since each did needs a dir entry and dstart entry. This has the downside of needing to actually keep track of an estimate of the number of metadata entries, which is made a bit difficult due to integer overflow issues (we can have more than 2^32 metadata entries), but has the upside of allowing a full 2^32 number of dids worst case. This is really unlikely, but it's nice to not need another configuration option to control the did limit. Another option would be to scale the hashes based on the number dids, which would be a more direct solution. Unfortunately determining the number of dids during mount requires a O(m*log(m)) scan of each rbyd to find either dir entries or dstart entries. This solution can easily end up with an overestimate, but only needs to weight of each rbyd which can be (and already is) found in O(m).
This commit is contained in:
@@ -1541,6 +1541,7 @@ static inline lfs_size_t lfsr_gdelta_size(
|
|||||||
static int lfsr_gdelta_xor(lfs_t *lfs,
|
static int lfsr_gdelta_xor(lfs_t *lfs,
|
||||||
uint8_t *gdelta, lfs_size_t size,
|
uint8_t *gdelta, lfs_size_t size,
|
||||||
lfsr_data_t xor) {
|
lfsr_data_t xor) {
|
||||||
|
(void)size;
|
||||||
// expect xor to fit
|
// expect xor to fit
|
||||||
LFS_ASSERT(lfsr_data_size(xor) <= size);
|
LFS_ASSERT(lfsr_data_size(xor) <= size);
|
||||||
|
|
||||||
@@ -5813,7 +5814,7 @@ static int lfsr_mdir_commit(lfs_t *lfs, lfsr_mdir_t *mdir, lfs_ssize_t *rid,
|
|||||||
// handle possible mtree updates, this gets a bit messy
|
// handle possible mtree updates, this gets a bit messy
|
||||||
lfsr_mdir_t mroot_ = lfs->mroot;
|
lfsr_mdir_t mroot_ = lfs->mroot;
|
||||||
lfsr_btree_t mtree_ = lfs->mtree;
|
lfsr_btree_t mtree_ = lfs->mtree;
|
||||||
lfsr_mdir_t msibling_;
|
lfsr_mdir_t msibling_ = {.mid=LFSR_MID_RM, .rbyd.weight = 0};
|
||||||
bool dirtymroot = false;
|
bool dirtymroot = false;
|
||||||
bool dirtymtree = false;
|
bool dirtymtree = false;
|
||||||
|
|
||||||
@@ -6181,6 +6182,9 @@ static int lfsr_mdir_commit(lfs_t *lfs, lfsr_mdir_t *mdir, lfs_ssize_t *rid,
|
|||||||
// update our mroot and mtree
|
// update our mroot and mtree
|
||||||
lfs->mroot = mroot_;
|
lfs->mroot = mroot_;
|
||||||
lfs->mtree = mtree_;
|
lfs->mtree = mtree_;
|
||||||
|
lfs->mlimit = lfs_max32(lfs->mlimit, lfs_max32(
|
||||||
|
mdir_.rbyd.weight,
|
||||||
|
msibling_.rbyd.weight));
|
||||||
|
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
@@ -6345,12 +6349,6 @@ static int lfsr_mtree_pathlookup(lfs_t *lfs, const char *path,
|
|||||||
if (d < 0) {
|
if (d < 0) {
|
||||||
return d;
|
return d;
|
||||||
}
|
}
|
||||||
|
|
||||||
// TODO should we put this in lfsr_data_readleb128?
|
|
||||||
// TODO should readtag then call lfsr_data_readleb128?
|
|
||||||
if (did > 0x7fffffff) {
|
|
||||||
return LFS_ERR_CORRUPT;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// lookup up this dname in the mtree
|
// lookup up this dname in the mtree
|
||||||
@@ -6436,6 +6434,10 @@ static int lfsr_mtree_traversal_next(lfs_t *lfs,
|
|||||||
return err;
|
return err;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// keep track of the largest mdir we've seen, reset here
|
||||||
|
// since this is the first mdir we see
|
||||||
|
lfs->mlimit = traversal->mdir.rbyd.weight;
|
||||||
|
|
||||||
if (mid_) {
|
if (mid_) {
|
||||||
*mid_ = -1;
|
*mid_ = -1;
|
||||||
}
|
}
|
||||||
@@ -6473,6 +6475,9 @@ static int lfsr_mtree_traversal_next(lfs_t *lfs,
|
|||||||
return err;
|
return err;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// keep track of the largest mdir we've seen
|
||||||
|
lfs->mlimit = lfs_max32(lfs->mlimit, traversal->mdir.rbyd.weight);
|
||||||
|
|
||||||
if (mid_) {
|
if (mid_) {
|
||||||
*mid_ = -1;
|
*mid_ = -1;
|
||||||
}
|
}
|
||||||
@@ -6591,6 +6596,9 @@ static int lfsr_mtree_traversal_next(lfs_t *lfs,
|
|||||||
return err;
|
return err;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// keep track of the largest mdir we've seen
|
||||||
|
lfs->mlimit = lfs_max32(lfs->mlimit, traversal->mdir.rbyd.weight);
|
||||||
|
|
||||||
if (mid_) {
|
if (mid_) {
|
||||||
*mid_ = mid;
|
*mid_ = mid;
|
||||||
}
|
}
|
||||||
@@ -7255,11 +7263,20 @@ int lfsr_mkdir(lfs_t *lfs, const char *path) {
|
|||||||
// hopefully few collisions, we use a hash of the full path. Since
|
// hopefully few collisions, we use a hash of the full path. Since
|
||||||
// we have a CRC handy, we can use that.
|
// we have a CRC handy, we can use that.
|
||||||
//
|
//
|
||||||
// We truncate to 28-bits to more optimally use our leb128 encoding.
|
// We also truncate to make better use of our leb128 encoding. This is
|
||||||
// TODO should we have a configurable limit for this? dir_limit? or
|
// pretty arbitrary, but we don't want collisions, so we truncate to
|
||||||
// just loose ~3 bits from the configured file limit?
|
// our best >= estimate of the number of metadata entries. Worst case
|
||||||
|
// this is >= ~2x the number of dids in the system, since each dir needs
|
||||||
|
// a dir entry and dstart entry.
|
||||||
//
|
//
|
||||||
lfs_size_t did = lfs_crc32c(0, path, strlen(path)) & 0xfffffff;
|
lfs_size_t did = lfs_crc32c(0, path, strlen(path))
|
||||||
|
// This complicated bit of logic determines the upper limit of
|
||||||
|
// metadata entries, limited to 2^32. This is done post-log to
|
||||||
|
// try to avoid issues with integer overflow.
|
||||||
|
& ((1 << lfs_min32(
|
||||||
|
lfs_nlog2(lfsr_mtree_weight(lfs))
|
||||||
|
+ lfs_nlog2(lfs->mlimit),
|
||||||
|
32)) - 1);
|
||||||
|
|
||||||
// Check if we have a collision. If we do, search for the next
|
// Check if we have a collision. If we do, search for the next
|
||||||
// available did
|
// available did
|
||||||
@@ -7360,12 +7377,6 @@ int lfsr_dir_open(lfs_t *lfs, lfsr_dir_t *dir, const char *path) {
|
|||||||
if (d < 0) {
|
if (d < 0) {
|
||||||
return d;
|
return d;
|
||||||
}
|
}
|
||||||
|
|
||||||
// TODO should we put this in lfsr_data_readleb128?
|
|
||||||
// TODO should readtag then call lfsr_data_readleb128?
|
|
||||||
if (did > 0x7fffffff) {
|
|
||||||
return LFS_ERR_CORRUPT;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// now lookup our dstart in the mtree
|
// now lookup our dstart in the mtree
|
||||||
|
|||||||
@@ -492,6 +492,7 @@ typedef struct lfs {
|
|||||||
// begin lfsr things
|
// begin lfsr things
|
||||||
lfsr_mdir_t mroot;
|
lfsr_mdir_t mroot;
|
||||||
lfsr_btree_t mtree;
|
lfsr_btree_t mtree;
|
||||||
|
lfs_size_t mlimit;
|
||||||
|
|
||||||
// TODO do we really need separate decoded/encoded grms?
|
// TODO do we really need separate decoded/encoded grms?
|
||||||
lfsr_grm_t grm_;
|
lfsr_grm_t grm_;
|
||||||
|
|||||||
Reference in New Issue
Block a user