Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 6 additions & 1 deletion src/extract/lzo1x.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -149,7 +149,12 @@ bool lzo1x_decompress_safe(const uint8_t* in, size_t in_len, uint8_t* out, size_
if (t >= 16) goto match;
{
if (!in_avail(1)) return false;
size_t off = 1 + 0x0800 + (t >> 2) + (static_cast<size_t>(*ip++) << 2);
// A short match FOLLOWING a match (length 2) uses a base distance of 1.
// Only the short match after a *literal run* (after_literal_run, length
// 3) adds the 0x0800 (M2_MAX_OFFSET) base -- that offset must NOT be
// applied here, or every match-after-match back-reference is 2048 too
// far and copy_match fails (aborting the whole block).
size_t off = 1 + (t >> 2) + (static_cast<size_t>(*ip++) << 2);
if (!copy_match(off, 2)) return false;
}
goto match_done;
Expand Down
24 changes: 13 additions & 11 deletions src/extract/ubifs.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -152,22 +152,24 @@ std::map<uint32_t, std::vector<uint8_t>> deubi(const Reader& r, uint64_t base, b

for (auto& [vol_id, lebs] : map) {
if (lebs.empty()) continue;
uint32_t max_lnum = lebs.rbegin()->first;
uint64_t total = uint64_t(max_lnum + 1) * (leb_len_common ? leb_len_common : 0);
// A real volume cannot hold more data than the source image (its LEBs
// live in PEBs within it). A corrupt/mutated lnum otherwise balloons the
// reconstructed image — e.g. a 256 KB input with one stray lnum=8000
// would allocate tens of MB (up to MAX_VOLUME_BYTES) of mostly-zero
// padding: a memory/disk amplification bomb whose huge mmap later faults.
// Pack the volume's LEBs contiguously in logical (lnum ascending) order.
// The UBIFS node scanner locates nodes by magic + CRC, so their exact
// lnum*leb_len positions are not needed. Sizing/placing by lnum instead
// was both wasteful and, for a *sparse* volume (few LEBs at high lnums,
// as a large mostly-empty data volume produces), forced a clamp to the
// source size that silently dropped every LEB past it. Sizing by the
// actual LEB count keeps every LEB and is inherently bomb-proof: a stray
// lnum can no longer balloon the image (its cost is one LEB, not its
// logical offset).
uint64_t total = uint64_t(lebs.size()) * (leb_len_common ? leb_len_common : 0);
if (total == 0 || total > MAX_VOLUME_BYTES) { truncated = true; continue; }
if (total > r.size()) { total = r.size(); truncated = true; } // clamp the bomb
std::vector<uint8_t>& img = volumes[vol_id];
img.assign(total, 0);
img.reserve(total);
for (auto& [lnum, leb] : lebs) {
(void)lnum;
auto span = r.bytes(leb.off, leb.len);
if (!span) { truncated = true; continue; }
uint64_t dst = uint64_t(lnum) * leb.len;
if (dst + leb.len <= img.size()) std::memcpy(img.data() + dst, span->data(), leb.len);
img.insert(img.end(), span->begin(), span->end());
}
}
return volumes;
Expand Down
61 changes: 61 additions & 0 deletions tests/test_extract.py
Original file line number Diff line number Diff line change
Expand Up @@ -222,6 +222,66 @@ def test_ubifs(work, src, expected):
return failures


def test_ubi_sparse_volume(work):
"""A UBI volume whose LEBs sit at sparse logical numbers (as a large, mostly
empty data volume produces) must reconstruct with EVERY LEB, including the
high-lnum ones. Regression: the reconstructor sized/placed LEBs by lnum and
clamped the image to the source size, silently dropping every LEB past it.
Self-contained: a hand-built 3-PEB UBI (no mkfs tools). Returns failures."""
import struct
import zlib

PEB, VIDOFF, DATAOFF = 0x20000, 2048, 4096
LEB = PEB - DATAOFF
crc = lambda b: zlib.crc32(b) & 0xffffffff

def ec():
h = bytearray(64)
h[0:4] = b"UBI#"; h[4] = 1
struct.pack_into(">I", h, 16, VIDOFF)
struct.pack_into(">I", h, 20, DATAOFF)
struct.pack_into(">I", h, 24, 0x12345678)
struct.pack_into(">I", h, 60, crc(bytes(h[0:60])))
return h

def vid(vol_id, lnum, sqnum):
h = bytearray(64)
h[0:4] = b"UBI!"; h[4] = 1; h[5] = 1 # version, vol_type=dynamic
struct.pack_into(">I", h, 8, vol_id)
struct.pack_into(">I", h, 12, lnum)
struct.pack_into(">Q", h, 40, sqnum)
struct.pack_into(">I", h, 60, crc(bytes(h[0:60])))
return h

def peb(vol_id, lnum, sqnum, marker):
p = bytearray(b"\xff" * PEB)
p[0:64] = ec()
p[VIDOFF:VIDOFF + 64] = vid(vol_id, lnum, sqnum)
p[DATAOFF:DATAOFF + LEB] = (marker * LEB)[:LEB]
return bytes(p)

# one volume, LEBs at lnum 0, 1, and 250 (sparse). The last is far past the
# 3-PEB image size, so the old size-clamp dropped it.
img = peb(1, 0, 10, b"LEB0") + peb(1, 1, 11, b"LEB1") + peb(1, 250, 12, b"LEBX")
src = os.path.join(work, "sparse_ubi.bin")
with open(src, "wb") as f:
f.write(img)
outdir = os.path.join(work, "sparse_ubi.out")
subprocess.run([MORIA, "--extract", "-C", outdir, src], capture_output=True)
blob = b""
for dp, _, fs in os.walk(outdir):
for f in fs:
if f.endswith(".img"):
with open(os.path.join(dp, f), "rb") as fh:
blob += fh.read()
if b"LEB0" in blob and b"LEB1" in blob and b"LEBX" in blob:
print("PASS [ubi-sparse]: all LEBs recovered incl. the sparse high-lnum one")
return 0
print(f"FAIL [ubi-sparse]: LEB0={b'LEB0' in blob} LEB1={b'LEB1' in blob} "
f"LEBX(sparse)={b'LEBX' in blob}")
return 1


def test_tar(work, src, expected):
"""Pack the tree with GNU tar (gnu/pax/ustar), extract with moria, compare.
Needs `tar`. Exercises long names, prefix, and pax path records. Returns
Expand Down Expand Up @@ -1660,6 +1720,7 @@ def main():
failures += test_ext(work, src, expected)
failures += test_jffs2(work, src, expected)
failures += test_ubifs(work, src, expected)
failures += test_ubi_sparse_volume(work)
failures += test_tar(work, src, expected)
failures += test_romfs(work, src, expected)
failures += test_yaffs2(work, src, expected)
Expand Down
8 changes: 8 additions & 0 deletions tests/unit/unit_tests.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -257,6 +257,14 @@ static void test_lzo1x() {
{0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43}},
{{0x05,0x68,0x65,0x61,0x64,0x65,0x72,0x3a,0xff,0x32,0x00,0x00,0x0d,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0x74,0x72,0x61,0x69,0x6c,0x65,0x72,0x11,0x00,0x00},
{0x68,0x65,0x61,0x64,0x65,0x72,0x3a,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0x74,0x72,0x61,0x69,0x6c,0x65,0x72}},
// Regression: a length-2 match immediately FOLLOWING another match (the
// match_next path). The old decoder added the 0x0800 (M2_MAX_OFFSET) base
// that belongs only to the after-a-literal-run short match, over-shooting
// this back-reference by 2048 and failing the whole block -- which
// zero-filled every LZO-compressed filesystem (e.g. a UBIFS/LZO rootfs).
// liblzo2 (python-lzo, raw lzo1x) roundtrips these exact bytes.
{{0x17,0x4f,0x43,0x51,0x4f,0x49,0x63,0x94,0x00,0x93,0x00,0x69,0x6c,0x65,0x0c,0x01,0x11,0x00,0x00},
{0x4f,0x43,0x51,0x4f,0x49,0x63,0x4f,0x43,0x51,0x4f,0x49,0x4f,0x43,0x51,0x4f,0x49,0x69,0x6c,0x65,0x4f,0x43}},
};
for (const auto& v : vecs) CHECK(lzo_ok(v.comp, v.plain));

Expand Down
Loading