scripts: Added -w/--word-bits to bound dbgleb128/dbgle32 parsing

This is limited to dbgle32.py, dbgleb128.py, and dbgtag.py for now.

This more closely matches how littlefs behaves, in that we read a
bounded number of bytes before leb128 decoding. This minimizes bugs
related to leb128 overflow and avoids reading inherently undecodable
data.

The previous unbounded behavior is still available with -w0.

Note this gives dbgle32.py much more flexibility in that it can now
decode other integer widths. Uh, ignore the name for now. At least it's
self documenting that the default is 32-bits...

---

Also fixed a bug in fromleb128 where size was reported incorrectly on
offset + truncated leb128.
This commit is contained in:
Christopher Haster
2025-04-14 16:26:21 -05:00
parent 0cea8b96fb
commit bd70270e11
9 changed files with 91 additions and 28 deletions
+1 -1
View File
@@ -213,7 +213,7 @@ def fromleb128(data, j=0):
if not b & 0x80: if not b & 0x80:
return word, d+1 return word, d+1
d += 1 d += 1
return word, len(data) return word, d
def fromtag(data, j=0): def fromtag(data, j=0):
d = 0 d = 0
+1 -1
View File
@@ -243,7 +243,7 @@ def fromleb128(data, j=0):
if not b & 0x80: if not b & 0x80:
return word, d+1 return word, d+1
d += 1 d += 1
return word, len(data) return word, d
def fromtag(data, j=0): def fromtag(data, j=0):
d = 0 d = 0
+1 -1
View File
@@ -151,7 +151,7 @@ def fromleb128(data, j=0):
if not b & 0x80: if not b & 0x80:
return word, d+1 return word, d+1
d += 1 d += 1
return word, len(data) return word, d
def fromtag(data, j=0): def fromtag(data, j=0):
d = 0 d = 0
+28 -9
View File
@@ -5,6 +5,7 @@ if __name__ == "__main__":
__import__('sys').path.pop(0) __import__('sys').path.pop(0)
import io import io
import math as mt
import os import os
import struct import struct
import sys import sys
@@ -21,18 +22,27 @@ def openio(path, mode='r', buffering=-1):
else: else:
return open(path, mode, buffering) return open(path, mode, buffering)
def fromle32(data, j=0): def dbg_le32s(data, *,
return struct.unpack('<I', data[j:j+4].ljust(4, b'\0'))[0] word_bits=32):
# figure out le32 size in bytes
if word_bits != 0:
n = mt.ceil(word_bits / 8)
def dbg_le32s(data): # parse le32s, or le<whatevers>
lines = [] lines = []
j = 0 j = 0
while j < len(data): while j < len(data):
word = fromle32(data, j) word = 0
d = 0
while (j+d < len(data)
and (d < n if word_bits != 0 else True)):
word |= data[j+d] << d
d += 1
lines.append(( lines.append((
' '.join('%02x' % b for b in data[j:j+4]), ' '.join('%02x' % b for b in data[j:j+d]),
word)) word))
j += 4 j += d
# figure out widths # figure out widths
w = [0] w = [0]
@@ -47,18 +57,21 @@ def dbg_le32s(data):
def main(le32s, *, def main(le32s, *,
hex=False, hex=False,
input=None): input=None,
word_bits=32):
hex_ = hex; del hex hex_ = hex; del hex
# interpret as a sequence of hex bytes # interpret as a sequence of hex bytes
if hex_: if hex_:
bytes_ = [b for le32 in le32s for b in le32.split()] bytes_ = [b for le32 in le32s for b in le32.split()]
dbg_le32s(bytes(int(b, 16) for b in bytes_)) dbg_le32s(bytes(int(b, 16) for b in bytes_),
word_bits=word_bits)
# parse le32s in a file # parse le32s in a file
elif input: elif input:
with openio(input, 'rb') as f: with openio(input, 'rb') as f:
dbg_le32s(f.read()) dbg_le32s(f.read(),
word_bits=word_bits)
# we don't currently have a default interpretation # we don't currently have a default interpretation
else: else:
@@ -84,6 +97,12 @@ if __name__ == "__main__":
parser.add_argument( parser.add_argument(
'-i', '--input', '-i', '--input',
help="Read le32s from this file. Can use - for stdin.") help="Read le32s from this file. Can use - for stdin.")
parser.add_argument(
'-w', '--word-bits',
nargs='?',
type=lambda x: int(x, 0),
const=0,
help="Word size in bits. 0 is unbounded. Defaults to 32.")
sys.exit(main(**{k: v sys.exit(main(**{k: v
for k, v in vars(parser.parse_intermixed_args()).items() for k, v in vars(parser.parse_intermixed_args()).items()
if v is not None})) if v is not None}))
+27 -5
View File
@@ -5,6 +5,7 @@ if __name__ == "__main__":
__import__('sys').path.pop(0) __import__('sys').path.pop(0)
import io import io
import math as mt
import os import os
import struct import struct
import sys import sys
@@ -31,13 +32,25 @@ def fromleb128(data, j=0):
if not b & 0x80: if not b & 0x80:
return word, d+1 return word, d+1
d += 1 d += 1
return word, len(data) return word, d
def dbg_leb128s(data): def dbg_leb128s(data, *,
word_bits=32):
# figure out leb128 size in bytes
if word_bits != 0:
n = mt.ceil(word_bits / 7)
# parse leb128s
lines = [] lines = []
j = 0 j = 0
while j < len(data): while j < len(data):
# bounded leb128s?
if word_bits != 0:
word, d = fromleb128(data[j:j+n])
# unbounded?
else:
word, d = fromleb128(data, j) word, d = fromleb128(data, j)
lines.append(( lines.append((
' '.join('%02x' % b for b in data[j:j+d]), ' '.join('%02x' % b for b in data[j:j+d]),
word)) word))
@@ -56,18 +69,21 @@ def dbg_leb128s(data):
def main(leb128s, *, def main(leb128s, *,
hex=False, hex=False,
input=None): input=None,
word_bits=32):
hex_ = hex; del hex hex_ = hex; del hex
# interpret as a sequence of hex bytes # interpret as a sequence of hex bytes
if hex_: if hex_:
bytes_ = [b for leb128 in leb128s for b in leb128.split()] bytes_ = [b for leb128 in leb128s for b in leb128.split()]
dbg_leb128s(bytes(int(b, 16) for b in bytes_)) dbg_leb128s(bytes(int(b, 16) for b in bytes_),
word_bits=word_bits)
# parse leb128s in a file # parse leb128s in a file
elif input: elif input:
with openio(input, 'rb') as f: with openio(input, 'rb') as f:
dbg_leb128s(f.read()) dbg_leb128s(f.read(),
word_bits=word_bits)
# we don't currently have a default interpretation # we don't currently have a default interpretation
else: else:
@@ -93,6 +109,12 @@ if __name__ == "__main__":
parser.add_argument( parser.add_argument(
'-i', '--input', '-i', '--input',
help="Read leb128s from this file. Can use - for stdin.") help="Read leb128s from this file. Can use - for stdin.")
parser.add_argument(
'-w', '--word-bits',
nargs='?',
type=lambda x: int(x, 0),
const=0,
help="Word size in bits. 0 is unbounded. Defaults to 32.")
sys.exit(main(**{k: v sys.exit(main(**{k: v
for k, v in vars(parser.parse_intermixed_args()).items() for k, v in vars(parser.parse_intermixed_args()).items()
if v is not None})) if v is not None}))
+1 -1
View File
@@ -170,7 +170,7 @@ def fromleb128(data, j=0):
if not b & 0x80: if not b & 0x80:
return word, d+1 return word, d+1
d += 1 d += 1
return word, len(data) return word, d
def fromtag(data, j=0): def fromtag(data, j=0):
d = 0 d = 0
+1 -1
View File
@@ -151,7 +151,7 @@ def fromleb128(data, j=0):
if not b & 0x80: if not b & 0x80:
return word, d+1 return word, d+1
d += 1 d += 1
return word, len(data) return word, d
def fromtag(data, j=0): def fromtag(data, j=0):
d = 0 d = 0
+1 -1
View File
@@ -161,7 +161,7 @@ def fromleb128(data, j=0):
if not b & 0x80: if not b & 0x80:
return word, d+1 return word, d+1
d += 1 d += 1
return word, len(data) return word, d
def fromtag(data, j=0): def fromtag(data, j=0):
d = 0 d = 0
+28 -6
View File
@@ -5,6 +5,7 @@ if __name__ == "__main__":
__import__('sys').path.pop(0) __import__('sys').path.pop(0)
import io import io
import math as mt
import os import os
import struct import struct
import sys import sys
@@ -74,7 +75,7 @@ def fromleb128(data, j=0):
if not b & 0x80: if not b & 0x80:
return word, d+1 return word, d+1
d += 1 d += 1
return word, len(data) return word, d
def fromtag(data, j=0): def fromtag(data, j=0):
d = 0 d = 0
@@ -234,7 +235,12 @@ def list_tags():
w[0], 'LFSR_'+n, w[0], 'LFSR_'+n,
c)) c))
def dbg_tags(data): def dbg_tags(data, *,
word_bits=32):
# figure out tag size in bytes
if word_bits != 0:
n = 2 + 2*mt.ceil(word_bits / 7)
lines = [] lines = []
# interpret as ints? # interpret as ints?
if not isinstance(data, bytes): if not isinstance(data, bytes):
@@ -247,7 +253,13 @@ def dbg_tags(data):
else: else:
j = 0 j = 0
while j < len(data): while j < len(data):
# bounded tags?
if word_bits != 0:
v, tag, w, size, d = fromtag(data[j:j+n])
# unbounded?
else:
v, tag, w, size, d = fromtag(data, j) v, tag, w, size, d = fromtag(data, j)
lines.append(( lines.append((
' '.join('%02x' % b for b in data[j:j+d]), ' '.join('%02x' % b for b in data[j:j+d]),
tagrepr(tag, w, size))) tagrepr(tag, w, size)))
@@ -271,7 +283,8 @@ def dbg_tags(data):
def main(tags, *, def main(tags, *,
list=False, list=False,
hex=False, hex=False,
input=None): input=None,
word_bits=32):
list_ = list; del list list_ = list; del list
hex_ = hex; del hex hex_ = hex; del hex
@@ -282,16 +295,19 @@ def main(tags, *,
# interpret as a sequence of hex bytes # interpret as a sequence of hex bytes
elif hex_: elif hex_:
bytes_ = [b for tag in tags for b in tag.split()] bytes_ = [b for tag in tags for b in tag.split()]
dbg_tags(bytes(int(b, 16) for b in bytes_)) dbg_tags(bytes(int(b, 16) for b in bytes_),
word_bits=word_bits)
# parse tags in a file # parse tags in a file
elif input: elif input:
with openio(input, 'rb') as f: with openio(input, 'rb') as f:
dbg_tags(f.read()) dbg_tags(f.read(),
word_bits=word_bits)
# default to interpreting as ints # default to interpreting as ints
else: else:
dbg_tags(int(tag, 0) for tag in tags) dbg_tags((int(tag, 0) for tag in tags),
word_bits=word_bits)
if __name__ == "__main__": if __name__ == "__main__":
@@ -315,6 +331,12 @@ if __name__ == "__main__":
parser.add_argument( parser.add_argument(
'-i', '--input', '-i', '--input',
help="Read tags from this file. Can use - for stdin.") help="Read tags from this file. Can use - for stdin.")
parser.add_argument(
'-w', '--word-bits',
nargs='?',
type=lambda x: int(x, 0),
const=0,
help="Word size in bits. 0 is unbounded. Defaults to 32.")
sys.exit(main(**{k: v sys.exit(main(**{k: v
for k, v in vars(parser.parse_intermixed_args()).items() for k, v in vars(parser.parse_intermixed_args()).items()
if v is not None})) if v is not None}))