Reworked bshrubs a tiny bit, allow forced recalculations
Now setting estimate=-1 will force a recalculation on the next bshrub
commit. This centralizing the annoying bshrub estimate calculation, so
code that allocs/fetches bshrubs can be simplifies to just struct
assignment.
As a plus, you don't pay the estimate calculation cost when only reading
a file.
Also dropped lfsr_bshrub_alloc/fetch, these weren't really useful
functions and their naming could be misleading.
I wonder if bshrubs will ever stop feeling like a big hack.
Also tweaked bshrub estimates a bit so they only include
bsprouts/bshrubs. I think we can get away with this by including
max(bptr, trunk, btree) + shrub_size when we get around to calculating
file limits.
We are now including an extra tag when estimating bsprout size, but
that's not the end of the world. At least all calls to
lfsr_bsprout_estimate__ are consistent now.
These changes were more for code organization/runtime improvements, the
benefit to code cost is minimal:
code stack
before: 33364 3072
after: 33336 (-0.1%) 3072 (+0.0%)
This commit is contained in:
@@ -5397,22 +5397,14 @@ static lfs_ssize_t lfsr_mdir_estimate__(lfs_t *lfs, const lfsr_mdir_t *mdir,
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
// include the cost of this tag
|
|
||||||
dsize_ += LFSR_ATTR_ESTIMATE;
|
|
||||||
|
|
||||||
// special handling for sprouts, just to avoid duplicate cost
|
// special handling for sprouts, just to avoid duplicate cost
|
||||||
if (tag == LFSR_TAG_DATA) {
|
if (tag == LFSR_TAG_DATA) {
|
||||||
// TODO don't include tag in attr estimate?
|
|
||||||
// we already included the size of the tag in our attr
|
|
||||||
// estimate, undo that for now
|
|
||||||
dsize_ -= LFSR_TAG_DSIZE;
|
|
||||||
|
|
||||||
lfs_ssize_t dsize__ = lfsr_bsprout_estimate__(lfs,
|
lfs_ssize_t dsize__ = lfsr_bsprout_estimate__(lfs,
|
||||||
(const lfsr_bsprout_t*)&data);
|
(const lfsr_bsprout_t*)&data);
|
||||||
if (dsize__ < 0) {
|
if (dsize__ < 0) {
|
||||||
return dsize__;
|
return dsize__;
|
||||||
}
|
}
|
||||||
dsize_ += dsize__;
|
dsize_ += LFSR_ATTR_ESTIMATE + dsize__;
|
||||||
|
|
||||||
// special handling for shrub trunks, we need to include the
|
// special handling for shrub trunks, we need to include the
|
||||||
// compacted cost of the shrub in our estimate
|
// compacted cost of the shrub in our estimate
|
||||||
@@ -5436,11 +5428,11 @@ static lfs_ssize_t lfsr_mdir_estimate__(lfs_t *lfs, const lfsr_mdir_t *mdir,
|
|||||||
if (dsize__ < 0) {
|
if (dsize__ < 0) {
|
||||||
return dsize__;
|
return dsize__;
|
||||||
}
|
}
|
||||||
dsize_ += dsize__;
|
dsize_ += LFSR_ATTR_ESTIMATE + dsize__;
|
||||||
|
|
||||||
} else {
|
} else {
|
||||||
// include the cost of this data
|
// include the cost of this tag
|
||||||
dsize_ += lfsr_data_size(&data);
|
dsize_ += LFSR_ATTR_ESTIMATE + lfsr_data_size(&data);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -6974,40 +6966,6 @@ static int lfsr_bshrub_compact__(lfs_t *lfs, lfsr_rbyd_t *rbyd_,
|
|||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
// lfsr_bshruballoc is a bit of a misnomer, this doesn't alloc, just
|
|
||||||
// prepares a new bshrub in the given mdir
|
|
||||||
static int lfsr_bshrub_alloc(lfs_t *lfs,
|
|
||||||
const lfsr_mdir_t *mdir, lfsr_bshrub_t *bshrub,
|
|
||||||
lfs_size_t estimate) {
|
|
||||||
(void)lfs;
|
|
||||||
bshrub->rbyd.blocks[0] = mdir->rbyd.blocks[0];
|
|
||||||
bshrub->rbyd.trunk = 0;
|
|
||||||
bshrub->rbyd.weight = 0;
|
|
||||||
bshrub->estimate = estimate;
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
// bshrubs don't really need to be fetched since the mdir must be
|
|
||||||
// fetched, but we do need to find the bshrubs estimate
|
|
||||||
static int lfsr_bshrub_fetch(lfs_t *lfs,
|
|
||||||
const lfsr_mdir_t *mdir, lfsr_bshrub_t *bshrub,
|
|
||||||
lfs_size_t trunk, lfsr_rid_t weight) {
|
|
||||||
bshrub->rbyd.blocks[0] = mdir->rbyd.blocks[0];
|
|
||||||
bshrub->rbyd.trunk = trunk;
|
|
||||||
bshrub->rbyd.weight = weight;
|
|
||||||
|
|
||||||
// find an estimate of the current shrub size, we need this
|
|
||||||
// to prevent our shrub from overflowing the mdir
|
|
||||||
lfs_ssize_t estimate = lfsr_rbyd_estimate(lfs,
|
|
||||||
&bshrub->rbyd, -1, -1, NULL);
|
|
||||||
if (estimate < 0) {
|
|
||||||
return estimate;
|
|
||||||
}
|
|
||||||
bshrub->estimate = estimate;
|
|
||||||
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
static int lfsr_bshrub_lookupnext_(lfs_t *lfs,
|
static int lfsr_bshrub_lookupnext_(lfs_t *lfs,
|
||||||
const lfsr_mdir_t *mdir, const lfsr_bshrub_t *bshrub,
|
const lfsr_mdir_t *mdir, const lfsr_bshrub_t *bshrub,
|
||||||
lfsr_bid_t bid,
|
lfsr_bid_t bid,
|
||||||
@@ -7037,72 +6995,18 @@ static int lfsr_bshrub_lookup(lfs_t *lfs,
|
|||||||
tag_, weight_, data_);
|
tag_, weight_, data_);
|
||||||
}
|
}
|
||||||
|
|
||||||
static int lfsr_bshrub_commit(lfs_t *lfs,
|
// find a tight upper bound on the _full_ bshrub size, this includes
|
||||||
lfsr_mdir_t *mdir, lfsr_bshrub_t *bshrub,
|
// any on-disk bshrubs, and all pending bshrubs
|
||||||
const lfsr_attr_t *attrs, lfs_size_t attr_count) {
|
static lfs_ssize_t lfsr_bshrub_estimate(lfs_t *lfs,
|
||||||
// we need some scratch space for tail-recursive attrs
|
lfsr_mdir_t *mdir, const lfsr_bshrub_t *bshrub) {
|
||||||
// TODO combined scratch pool?
|
(void)bshrub;
|
||||||
lfsr_attr_t scratch_attrs[4];
|
lfs_size_t estimate = 0;
|
||||||
uint8_t scratch_buf[2*LFSR_BRANCH_DSIZE];
|
|
||||||
|
|
||||||
// try to commit to the btree
|
|
||||||
int err = lfsr_btree_commit_(lfs, &bshrub->rbyd,
|
|
||||||
lfsr_bshrub_isbshrub(mdir, bshrub),
|
|
||||||
scratch_attrs, scratch_buf,
|
|
||||||
attrs, attr_count,
|
|
||||||
&attrs, &attr_count);
|
|
||||||
if (err) {
|
|
||||||
return err;
|
|
||||||
}
|
|
||||||
|
|
||||||
// when btree is shrubbed, lfsr_btree_commit_ stops at the root
|
|
||||||
// and returns with pending attrs
|
|
||||||
//
|
|
||||||
// note! lfsr_bshrub_isbshrub may have changed state due to collapsed
|
|
||||||
// parents, splits, etc
|
|
||||||
//
|
|
||||||
if (attr_count > 0) {
|
|
||||||
// new bshrub?
|
|
||||||
if (bshrub->rbyd.trunk == 0) {
|
|
||||||
err = lfsr_bshrub_alloc(lfs, mdir, bshrub,
|
|
||||||
LFSR_ATTR_ESTIMATE + LFSR_BTREE_DSIZE);
|
|
||||||
if (err) {
|
|
||||||
return err;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// we need to prevent our shrub from overflowing our mdir somehow
|
|
||||||
//
|
|
||||||
// maintaining an accurate estimate is tricky and error-prone,
|
|
||||||
// but recalculating an estimate every commit is expensive
|
|
||||||
//
|
|
||||||
// Instead, we keep track of an estimate of how many bytes have
|
|
||||||
// been progged to the shrub since the last estimate, and recalculate
|
|
||||||
// the estimate when this overflows our shrub_size. This mirrors how
|
|
||||||
// block_size and rbyds interact, and amortizes the estimate cost.
|
|
||||||
|
|
||||||
// figure out how much data this commit progs
|
|
||||||
lfs_size_t commit_estimate = 0;
|
|
||||||
for (lfs_size_t i = 0; i < attr_count; i++) {
|
|
||||||
// only include tag overhead if tag is not a grow tag
|
|
||||||
if (!lfsr_tag_isgrow(attrs[i].tag)) {
|
|
||||||
commit_estimate += LFSR_ATTR_ESTIMATE;
|
|
||||||
}
|
|
||||||
commit_estimate += lfsr_data_size(&attrs[i].data);
|
|
||||||
}
|
|
||||||
|
|
||||||
// does our estimate exceed our shrub_size? need to recalculate an
|
|
||||||
// accurate our estimate
|
|
||||||
lfs_size_t estimate = bshrub->estimate + commit_estimate;
|
|
||||||
if (estimate > lfs->cfg->shrub_size) {
|
|
||||||
// don't forget to include our pending commit
|
|
||||||
estimate = commit_estimate;
|
|
||||||
|
|
||||||
// include all unique sprouts/shrubs related to our file,
|
// include all unique sprouts/shrubs related to our file,
|
||||||
// including the on-disk sprout/shrub
|
// including the on-disk sprout/shrub
|
||||||
lfsr_tag_t tag;
|
lfsr_tag_t tag;
|
||||||
lfsr_data_t data;
|
lfsr_data_t data;
|
||||||
err = lfsr_mdir_lookupnext(lfs, mdir, mdir->mid, LFSR_TAG_DATA,
|
int err = lfsr_mdir_lookupnext(lfs, mdir, mdir->mid, LFSR_TAG_DATA,
|
||||||
&tag, &data);
|
&tag, &data);
|
||||||
if (err && err != LFS_ERR_NOENT) {
|
if (err && err != LFS_ERR_NOENT) {
|
||||||
return err;
|
return err;
|
||||||
@@ -7114,7 +7018,7 @@ static int lfsr_bshrub_commit(lfs_t *lfs,
|
|||||||
if (dsize < 0) {
|
if (dsize < 0) {
|
||||||
return dsize;
|
return dsize;
|
||||||
}
|
}
|
||||||
estimate += lfsr_data_size(&data);
|
estimate += dsize;
|
||||||
|
|
||||||
} else if (err != LFS_ERR_NOENT && tag == LFSR_TAG_BSHRUB) {
|
} else if (err != LFS_ERR_NOENT && tag == LFSR_TAG_BSHRUB) {
|
||||||
lfsr_rbyd_t shrub = mdir->rbyd;
|
lfsr_rbyd_t shrub = mdir->rbyd;
|
||||||
@@ -7158,9 +7062,83 @@ static int lfsr_bshrub_commit(lfs_t *lfs,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
return estimate;
|
||||||
|
}
|
||||||
|
|
||||||
|
static int lfsr_bshrub_commit(lfs_t *lfs,
|
||||||
|
lfsr_mdir_t *mdir, lfsr_bshrub_t *bshrub,
|
||||||
|
const lfsr_attr_t *attrs, lfs_size_t attr_count) {
|
||||||
|
// we need some scratch space for tail-recursive attrs
|
||||||
|
// TODO combined scratch pool?
|
||||||
|
lfsr_attr_t scratch_attrs[4];
|
||||||
|
uint8_t scratch_buf[2*LFSR_BRANCH_DSIZE];
|
||||||
|
|
||||||
|
// try to commit to the btree
|
||||||
|
int err = lfsr_btree_commit_(lfs, &bshrub->rbyd,
|
||||||
|
lfsr_bshrub_isbshrub(mdir, bshrub),
|
||||||
|
scratch_attrs, scratch_buf,
|
||||||
|
attrs, attr_count,
|
||||||
|
&attrs, &attr_count);
|
||||||
|
if (err) {
|
||||||
|
return err;
|
||||||
|
}
|
||||||
|
|
||||||
|
// when btree is shrubbed, lfsr_btree_commit_ stops at the root
|
||||||
|
// and returns with pending attrs
|
||||||
|
//
|
||||||
|
// note! lfsr_bshrub_isbshrub may have changed state due to collapsed
|
||||||
|
// parents, splits, etc
|
||||||
|
//
|
||||||
|
if (attr_count > 0) {
|
||||||
|
// new bshrub?
|
||||||
|
if (bshrub->rbyd.trunk == 0) {
|
||||||
|
bshrub->rbyd.blocks[0] = mdir->rbyd.blocks[0];
|
||||||
|
bshrub->rbyd.trunk = 0;
|
||||||
|
bshrub->rbyd.weight = 0;
|
||||||
|
// force estimate recalculation
|
||||||
|
bshrub->estimate = -1;
|
||||||
|
}
|
||||||
|
|
||||||
|
// we need to prevent our shrub from overflowing our mdir somehow
|
||||||
|
//
|
||||||
|
// maintaining an accurate estimate is tricky and error-prone,
|
||||||
|
// but recalculating an estimate every commit is expensive
|
||||||
|
//
|
||||||
|
// Instead, we keep track of an estimate of how many bytes have
|
||||||
|
// been progged to the shrub since the last estimate, and recalculate
|
||||||
|
// the estimate when this overflows our shrub_size. This mirrors how
|
||||||
|
// block_size and rbyds interact, and amortizes the estimate cost.
|
||||||
|
|
||||||
|
// figure out how much data this commit progs
|
||||||
|
lfs_size_t commit_estimate = 0;
|
||||||
|
for (lfs_size_t i = 0; i < attr_count; i++) {
|
||||||
|
// only include tag overhead if tag is not a grow tag
|
||||||
|
if (!lfsr_tag_isgrow(attrs[i].tag)) {
|
||||||
|
commit_estimate += LFSR_ATTR_ESTIMATE;
|
||||||
|
}
|
||||||
|
commit_estimate += lfsr_data_size(&attrs[i].data);
|
||||||
|
}
|
||||||
|
|
||||||
|
// avoid some overflow issues here
|
||||||
|
lfs_ssize_t estimate = bshrub->estimate;
|
||||||
|
if ((lfs_size_t)estimate <= lfs->cfg->shrub_size) {
|
||||||
|
estimate += commit_estimate;
|
||||||
|
}
|
||||||
|
|
||||||
|
// does our estimate exceed our shrub_size? need to recalculate an
|
||||||
|
// accurate our estimate
|
||||||
|
if ((lfs_size_t)estimate > lfs->cfg->shrub_size) {
|
||||||
|
estimate = lfsr_bshrub_estimate(lfs, mdir, bshrub);
|
||||||
|
if (estimate < 0) {
|
||||||
|
return estimate;
|
||||||
|
}
|
||||||
|
|
||||||
|
// don't forget to include our pending commit
|
||||||
|
estimate += commit_estimate;
|
||||||
|
|
||||||
// do we overflow shrub_size/2? the 1/2 here prevents runaway
|
// do we overflow shrub_size/2? the 1/2 here prevents runaway
|
||||||
// performance when the shrub is near full
|
// performance when the shrub is near full
|
||||||
if (estimate > lfs->cfg->shrub_size/2) {
|
if ((lfs_size_t)estimate > lfs->cfg->shrub_size/2) {
|
||||||
goto evict;
|
goto evict;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -7186,7 +7164,7 @@ static int lfsr_bshrub_commit(lfs_t *lfs,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
LFS_ASSERT(bshrub->estimate == estimate);
|
LFS_ASSERT(bshrub->estimate == (lfs_size_t)estimate);
|
||||||
}
|
}
|
||||||
|
|
||||||
LFS_ASSERT(bshrub->rbyd.trunk != 0);
|
LFS_ASSERT(bshrub->rbyd.trunk != 0);
|
||||||
@@ -9382,12 +9360,12 @@ int lfsr_file_opencfg(lfs_t *lfs, lfsr_file_t *file,
|
|||||||
return err;
|
return err;
|
||||||
}
|
}
|
||||||
|
|
||||||
int err = lfsr_bshrub_fetch(lfs,
|
file->ftree.u.bshrub.rbyd.blocks[0]
|
||||||
&file->ftree.mdir, &file->ftree.u.bshrub,
|
= file->ftree.mdir.rbyd.blocks[0];
|
||||||
trunk, weight);
|
file->ftree.u.bshrub.rbyd.trunk = trunk;
|
||||||
if (err) {
|
file->ftree.u.bshrub.rbyd.weight = weight;
|
||||||
return err;
|
// force estimate recalculation if we write to this shrub
|
||||||
}
|
file->ftree.u.bshrub.estimate = -1;
|
||||||
|
|
||||||
// or a btree
|
// or a btree
|
||||||
} else if (err != LFS_ERR_NOENT && tag == LFSR_TAG_BTREE) {
|
} else if (err != LFS_ERR_NOENT && tag == LFSR_TAG_BTREE) {
|
||||||
@@ -9671,21 +9649,18 @@ static int lfsr_ftree_carve(lfs_t *lfs, lfsr_ftree_t *ftree,
|
|||||||
lfs_size_t attr_count_ = 0;
|
lfs_size_t attr_count_ = 0;
|
||||||
uint8_t buf[LFSR_BPTR_DSIZE+LFSR_ECKSUM_DSIZE];
|
uint8_t buf[LFSR_BPTR_DSIZE+LFSR_ECKSUM_DSIZE];
|
||||||
lfs_size_t buf_size = 0;
|
lfs_size_t buf_size = 0;
|
||||||
lfs_size_t estimate = 0;
|
|
||||||
|
|
||||||
// these also check if ftree is non-zero
|
// these also check if ftree is non-zero
|
||||||
if (lfsr_ftree_isbsprout(ftree)) {
|
if (lfsr_ftree_isbsprout(ftree)) {
|
||||||
attrs_[attr_count_++] = LFSR_ATTR(0,
|
attrs_[attr_count_++] = LFSR_ATTR(0,
|
||||||
DATA, +lfsr_ftree_size(ftree),
|
DATA, +lfsr_ftree_size(ftree),
|
||||||
DATA(ftree->u.bsprout.data));
|
DATA(ftree->u.bsprout.data));
|
||||||
estimate += LFSR_ATTR_ESTIMATE + lfsr_ftree_size(ftree);
|
|
||||||
|
|
||||||
} else if (lfsr_ftree_isbleaf(ftree)) {
|
} else if (lfsr_ftree_isbleaf(ftree)) {
|
||||||
attrs_[attr_count_++] = LFSR_ATTR(0,
|
attrs_[attr_count_++] = LFSR_ATTR(0,
|
||||||
BLOCK, +lfsr_ftree_size(ftree),
|
BLOCK, +lfsr_ftree_size(ftree),
|
||||||
FROMBPTR(&ftree->u.bptr, &buf[buf_size]));
|
FROMBPTR(&ftree->u.bptr, &buf[buf_size]));
|
||||||
buf_size += LFSR_BPTR_DSIZE;
|
buf_size += LFSR_BPTR_DSIZE;
|
||||||
estimate += LFSR_ATTR_ESTIMATE + LFSR_BPTR_DSIZE;
|
|
||||||
|
|
||||||
// append becksum?
|
// append becksum?
|
||||||
if (ftree->u.bleaf.becksum.size != -1) {
|
if (ftree->u.bleaf.becksum.size != -1) {
|
||||||
@@ -9693,20 +9668,19 @@ static int lfsr_ftree_carve(lfs_t *lfs, lfsr_ftree_t *ftree,
|
|||||||
BECKSUM, 0,
|
BECKSUM, 0,
|
||||||
FROMECKSUM(&ftree->u.bleaf.becksum, &buf[buf_size]));
|
FROMECKSUM(&ftree->u.bleaf.becksum, &buf[buf_size]));
|
||||||
buf_size += LFSR_ECKSUM_DSIZE;
|
buf_size += LFSR_ECKSUM_DSIZE;
|
||||||
estimate += LFSR_ATTR_ESTIMATE + LFSR_ECKSUM_DSIZE;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
int err = lfsr_bshrub_alloc(lfs, &ftree->mdir, &ftree->u.bshrub,
|
ftree->u.bshrub.rbyd.blocks[0] = ftree->mdir.rbyd.blocks[0];
|
||||||
estimate);
|
ftree->u.bshrub.rbyd.trunk = 0;
|
||||||
if (err) {
|
ftree->u.bshrub.rbyd.weight = 0;
|
||||||
return err;
|
// force estimate recalculation
|
||||||
}
|
ftree->u.bshrub.estimate = -1;
|
||||||
|
|
||||||
if (attr_count_ > 0) {
|
if (attr_count_ > 0) {
|
||||||
LFS_ASSERT(attr_count_ <= sizeof(attrs_)/sizeof(lfsr_attr_t));
|
LFS_ASSERT(attr_count_ <= sizeof(attrs_)/sizeof(lfsr_attr_t));
|
||||||
LFS_ASSERT(buf_size <= sizeof(buf));
|
LFS_ASSERT(buf_size <= sizeof(buf));
|
||||||
err = lfsr_bshrub_commit(lfs, &ftree->mdir, &ftree->u.bshrub,
|
int err = lfsr_bshrub_commit(lfs, &ftree->mdir, &ftree->u.bshrub,
|
||||||
attrs_, attr_count_);
|
attrs_, attr_count_);
|
||||||
if (err) {
|
if (err) {
|
||||||
return err;
|
return err;
|
||||||
|
|||||||
Reference in New Issue
Block a user