From 2a72dd1700b1172a8eafe6c6ac2720a374182356 Mon Sep 17 00:00:00 2001 From: Christopher Haster Date: Sat, 31 Jan 2026 14:35:03 -0600 Subject: [PATCH] runners: Treat erase timing as strictly per-byte Initial results with the new timing calculations looked weird. Turns out different block sizes perform surprisingly when they all cost the same! Fortunately, erases are the one operation where per-byte vs per-op timing doesn't really matter, so reverting to only per-byte timing solves this problem. Now, erasing 2 4KiB blocks should take the same time as 1 8KiB block, instead of twice as long. --- Arguably, erase timing shouldn't be _strictly_ linear w.r.t. block size. There's a reason denser storage usually ends up with larger block sizes after all. But preventing the block size from messing with per-byte timings is much more interesting from a filesystem design perspective. It also matches the behavior of artificially increasing block size to reduce block allocator pressure. Unfortunately, this also raises concerns with read/prog timing when varying geometry is involved... Should we stick to the per-byte timing in such cases? Is there a better timing model out there without too much additional complexity? --- runners/bench_defines.h | 26 ++++++++++++++++++-------- 1 file changed, 18 insertions(+), 8 deletions(-) diff --git a/runners/bench_defines.h b/runners/bench_defines.h index 45d1ada3..96ba348c 100644 --- a/runners/bench_defines.h +++ b/runners/bench_defines.h @@ -35,16 +35,21 @@ // FR=104 MHz, quad prog (9.6 ns * 8/4) // => +~19 ns for bus (not read!) // + // simple: // readed=40ns/B fR=50 MHz, quad read (20 ns * 8/4) // progged=1582ns/B tPP=0.4 ms, page=256 (0.4 ms / 256 + bus) // erased=10986ns/B tSE=45 ms, sector=4096 (45 ms / 4096) // + // less-simple: // reads=0ns (no transaction cost) // progs=400000ns tPP=0.4 ms, page=256 - // erases=45000000ns tSE=45 ms, sector=4096 + // erases=0ns (no transaction cost) // readed=40ns/B fR=50 MHz, quad read (20 ns * 8/4) // progged=1484ns/B tPP=0.4 ms (((4096/256)*0.4ms - 0.4ms)/4096 + bus) - // erased=0ns/B (no per-byte cost) + // erased=10986ns/B tSE=45 ms, sector=4096 (45 ms / 4096) + // + // note we always treat erases as per-byte to simplify benchmarking + // across different block sizes // #ifdef BENCH_SIMPLE BENCH_DEFINE(READS_TIMING, 0 ) @@ -56,10 +61,10 @@ #else BENCH_DEFINE(READS_TIMING, 0 ) BENCH_DEFINE(PROGS_TIMING, 400000 ) - BENCH_DEFINE(ERASES_TIMING, 45000000 ) + BENCH_DEFINE(ERASES_TIMING, 0 ) BENCH_DEFINE(READED_TIMING, 40 ) BENCH_DEFINE(PROGGED_TIMING, 1484 ) - BENCH_DEFINE(ERASED_TIMING, 0 ) + BENCH_DEFINE(ERASED_TIMING, 10986 ) #endif #else // default timings for NAND flash, based on w25n01gv: @@ -68,16 +73,21 @@ // FR=104 MHz, quad read/prog (9.6 ns * 8/4) // => +~19 ns for bus // + // simple: // readed=31ns/B tRD1=25 us, p=2048, s=512 (25 us / 2048 + bus) // progged=141ns/B tPP=250 us, p=2048, s=512 (250 us / 2048 + bus) // erased=15ns/B tBE=2 ms, block=131072 (2 ms / 131072) // + // less-simple: // reads=25000ns tRD1=25 us, p=2048, s=512 // progs=250000ns tPP=250 us, p=2048, s=512 - // erases=2000000ns tBE=2 ms, block=131072 + // erases=0ns (no transaction cost) // readed=31ns/B tRD1=25 us (((131072/2048)*25us - 25us)/131072 + bus) // progged=139ns/B tPP=250 us (((131072/2048)*250us - 250us)/131072 + bus) - // erased=0ns/B (no per-byte cost) + // erased=15ns/B tBE=2 ms, block=131072 (2 ms / 131072) + // + // note we always treat erases as per-byte to simplify benchmarking + // across different block sizes // #ifdef BENCH_SIMPLE BENCH_DEFINE(READS_TIMING, 0 ) @@ -89,10 +99,10 @@ #else BENCH_DEFINE(READS_TIMING, 25000 ) BENCH_DEFINE(PROGS_TIMING, 250000 ) - BENCH_DEFINE(ERASES_TIMING, 2000000 ) + BENCH_DEFINE(ERASES_TIMING, 0 ) BENCH_DEFINE(READED_TIMING, 31 ) BENCH_DEFINE(PROGGED_TIMING, 139 ) - BENCH_DEFINE(ERASED_TIMING, 0 ) + BENCH_DEFINE(ERASED_TIMING, 15 ) #endif #endif #ifndef BENCH_KIWIBD