From b576a6cca35bb0896c5c3dd1bb2ce2119e1ded7a Mon Sep 17 00:00:00 2001 From: Matt Brown Date: Fri, 18 Sep 2026 18:44:45 -0400 Subject: [PATCH 1/2] fix(lzo1x): correct the match-after-match short-match distance base The internal LZO1X decompressor applied the 0x0800 (M2_MAX_OFFSET) distance base to the length-2 short match that follows another match (the match_next path). That base belongs only to the length-3 short match that follows a literal run. Adding it here over-shot every match-after-match back-reference by 2048 bytes, so copy_match rejected it and the whole block failed to decode. Because this is the shared lzo1x_decompress_safe, it silently corrupted any LZO-compressed extraction (UBIFS/squashfs/jffs2/f2fs/btrfs) as soon as a stream used a length-2 match after a match -- common in real data, but absent from the existing known-answer vectors, so it went unnoticed. A UBIFS/LZO rootfs came out ~98% zero-filled (status "partial"); with the fix it extracts cleanly. Add a known-answer vector that exercises the match_next path: it fails on the old decoder and round-trips through liblzo2. --- src/extract/lzo1x.cpp | 7 ++++++- tests/unit/unit_tests.cpp | 8 ++++++++ 2 files changed, 14 insertions(+), 1 deletion(-) diff --git a/src/extract/lzo1x.cpp b/src/extract/lzo1x.cpp index d2eddf7..0206023 100644 --- a/src/extract/lzo1x.cpp +++ b/src/extract/lzo1x.cpp @@ -149,7 +149,12 @@ bool lzo1x_decompress_safe(const uint8_t* in, size_t in_len, uint8_t* out, size_ if (t >= 16) goto match; { if (!in_avail(1)) return false; - size_t off = 1 + 0x0800 + (t >> 2) + (static_cast(*ip++) << 2); + // A short match FOLLOWING a match (length 2) uses a base distance of 1. + // Only the short match after a *literal run* (after_literal_run, length + // 3) adds the 0x0800 (M2_MAX_OFFSET) base -- that offset must NOT be + // applied here, or every match-after-match back-reference is 2048 too + // far and copy_match fails (aborting the whole block). + size_t off = 1 + (t >> 2) + (static_cast(*ip++) << 2); if (!copy_match(off, 2)) return false; } goto match_done; diff --git a/tests/unit/unit_tests.cpp b/tests/unit/unit_tests.cpp index 1a9e36a..35449cb 100644 --- a/tests/unit/unit_tests.cpp +++ b/tests/unit/unit_tests.cpp @@ -257,6 +257,14 @@ static void test_lzo1x() { {0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43}}, {{0x05,0x68,0x65,0x61,0x64,0x65,0x72,0x3a,0xff,0x32,0x00,0x00,0x0d,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0x74,0x72,0x61,0x69,0x6c,0x65,0x72,0x11,0x00,0x00}, {0x68,0x65,0x61,0x64,0x65,0x72,0x3a,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0x74,0x72,0x61,0x69,0x6c,0x65,0x72}}, + // Regression: a length-2 match immediately FOLLOWING another match (the + // match_next path). The old decoder added the 0x0800 (M2_MAX_OFFSET) base + // that belongs only to the after-a-literal-run short match, over-shooting + // this back-reference by 2048 and failing the whole block -- which + // zero-filled every LZO-compressed filesystem (e.g. a UBIFS/LZO rootfs). + // liblzo2 (python-lzo, raw lzo1x) roundtrips these exact bytes. + {{0x17,0x4f,0x43,0x51,0x4f,0x49,0x63,0x94,0x00,0x93,0x00,0x69,0x6c,0x65,0x0c,0x01,0x11,0x00,0x00}, + {0x4f,0x43,0x51,0x4f,0x49,0x63,0x4f,0x43,0x51,0x4f,0x49,0x4f,0x43,0x51,0x4f,0x49,0x69,0x6c,0x65,0x4f,0x43}}, }; for (const auto& v : vecs) CHECK(lzo_ok(v.comp, v.plain)); From 840c05b50da0aa05de5297950e9d086808f327d4 Mon Sep 17 00:00:00 2001 From: Matt Brown Date: Fri, 18 Sep 2026 18:55:39 -0400 Subject: [PATCH 2/2] fix(ubi): reconstruct sparse volumes without dropping high-lnum LEBs reconstruct_volumes sized each volume image by (max_lnum+1)*leb_len and placed each LEB at lnum*leb_len, then clamped the image to the source size to defend against a stray lnum ballooning the allocation. For a legitimately sparse volume -- a large, mostly-empty data volume whose few LEBs sit at high logical numbers -- that clamp silently discarded every LEB past the source size (e.g. a volume with LEBs up to lnum 745 in an ~18 MB image kept only ~20% of them), so most of the filesystem was lost. Pack a volume's LEBs contiguously in lnum order instead. The UBIFS node scanner locates nodes by magic + CRC, so exact lnum*leb_len offsets are unnecessary; sizing by the actual LEB count keeps every LEB and is inherently bomb-proof (a stray lnum now costs one LEB, not its logical offset). Non-sparse volumes are unaffected -- their LEBs are already contiguous. Add a self-contained regression test: a hand-built 3-PEB UBI with a volume whose LEBs are at lnum 0, 1, and 250 -- the last is dropped by the old clamp and recovered by the fix. --- src/extract/ubifs.cpp | 24 +++++++++-------- tests/test_extract.py | 61 +++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 74 insertions(+), 11 deletions(-) diff --git a/src/extract/ubifs.cpp b/src/extract/ubifs.cpp index d9c5384..a7cc796 100644 --- a/src/extract/ubifs.cpp +++ b/src/extract/ubifs.cpp @@ -152,22 +152,24 @@ std::map> deubi(const Reader& r, uint64_t base, b for (auto& [vol_id, lebs] : map) { if (lebs.empty()) continue; - uint32_t max_lnum = lebs.rbegin()->first; - uint64_t total = uint64_t(max_lnum + 1) * (leb_len_common ? leb_len_common : 0); - // A real volume cannot hold more data than the source image (its LEBs - // live in PEBs within it). A corrupt/mutated lnum otherwise balloons the - // reconstructed image — e.g. a 256 KB input with one stray lnum=8000 - // would allocate tens of MB (up to MAX_VOLUME_BYTES) of mostly-zero - // padding: a memory/disk amplification bomb whose huge mmap later faults. + // Pack the volume's LEBs contiguously in logical (lnum ascending) order. + // The UBIFS node scanner locates nodes by magic + CRC, so their exact + // lnum*leb_len positions are not needed. Sizing/placing by lnum instead + // was both wasteful and, for a *sparse* volume (few LEBs at high lnums, + // as a large mostly-empty data volume produces), forced a clamp to the + // source size that silently dropped every LEB past it. Sizing by the + // actual LEB count keeps every LEB and is inherently bomb-proof: a stray + // lnum can no longer balloon the image (its cost is one LEB, not its + // logical offset). + uint64_t total = uint64_t(lebs.size()) * (leb_len_common ? leb_len_common : 0); if (total == 0 || total > MAX_VOLUME_BYTES) { truncated = true; continue; } - if (total > r.size()) { total = r.size(); truncated = true; } // clamp the bomb std::vector& img = volumes[vol_id]; - img.assign(total, 0); + img.reserve(total); for (auto& [lnum, leb] : lebs) { + (void)lnum; auto span = r.bytes(leb.off, leb.len); if (!span) { truncated = true; continue; } - uint64_t dst = uint64_t(lnum) * leb.len; - if (dst + leb.len <= img.size()) std::memcpy(img.data() + dst, span->data(), leb.len); + img.insert(img.end(), span->begin(), span->end()); } } return volumes; diff --git a/tests/test_extract.py b/tests/test_extract.py index 0866292..61f2151 100644 --- a/tests/test_extract.py +++ b/tests/test_extract.py @@ -222,6 +222,66 @@ def test_ubifs(work, src, expected): return failures +def test_ubi_sparse_volume(work): + """A UBI volume whose LEBs sit at sparse logical numbers (as a large, mostly + empty data volume produces) must reconstruct with EVERY LEB, including the + high-lnum ones. Regression: the reconstructor sized/placed LEBs by lnum and + clamped the image to the source size, silently dropping every LEB past it. + Self-contained: a hand-built 3-PEB UBI (no mkfs tools). Returns failures.""" + import struct + import zlib + + PEB, VIDOFF, DATAOFF = 0x20000, 2048, 4096 + LEB = PEB - DATAOFF + crc = lambda b: zlib.crc32(b) & 0xffffffff + + def ec(): + h = bytearray(64) + h[0:4] = b"UBI#"; h[4] = 1 + struct.pack_into(">I", h, 16, VIDOFF) + struct.pack_into(">I", h, 20, DATAOFF) + struct.pack_into(">I", h, 24, 0x12345678) + struct.pack_into(">I", h, 60, crc(bytes(h[0:60]))) + return h + + def vid(vol_id, lnum, sqnum): + h = bytearray(64) + h[0:4] = b"UBI!"; h[4] = 1; h[5] = 1 # version, vol_type=dynamic + struct.pack_into(">I", h, 8, vol_id) + struct.pack_into(">I", h, 12, lnum) + struct.pack_into(">Q", h, 40, sqnum) + struct.pack_into(">I", h, 60, crc(bytes(h[0:60]))) + return h + + def peb(vol_id, lnum, sqnum, marker): + p = bytearray(b"\xff" * PEB) + p[0:64] = ec() + p[VIDOFF:VIDOFF + 64] = vid(vol_id, lnum, sqnum) + p[DATAOFF:DATAOFF + LEB] = (marker * LEB)[:LEB] + return bytes(p) + + # one volume, LEBs at lnum 0, 1, and 250 (sparse). The last is far past the + # 3-PEB image size, so the old size-clamp dropped it. + img = peb(1, 0, 10, b"LEB0") + peb(1, 1, 11, b"LEB1") + peb(1, 250, 12, b"LEBX") + src = os.path.join(work, "sparse_ubi.bin") + with open(src, "wb") as f: + f.write(img) + outdir = os.path.join(work, "sparse_ubi.out") + subprocess.run([MORIA, "--extract", "-C", outdir, src], capture_output=True) + blob = b"" + for dp, _, fs in os.walk(outdir): + for f in fs: + if f.endswith(".img"): + with open(os.path.join(dp, f), "rb") as fh: + blob += fh.read() + if b"LEB0" in blob and b"LEB1" in blob and b"LEBX" in blob: + print("PASS [ubi-sparse]: all LEBs recovered incl. the sparse high-lnum one") + return 0 + print(f"FAIL [ubi-sparse]: LEB0={b'LEB0' in blob} LEB1={b'LEB1' in blob} " + f"LEBX(sparse)={b'LEBX' in blob}") + return 1 + + def test_tar(work, src, expected): """Pack the tree with GNU tar (gnu/pax/ustar), extract with moria, compare. Needs `tar`. Exercises long names, prefix, and pax path records. Returns @@ -1660,6 +1720,7 @@ def main(): failures += test_ext(work, src, expected) failures += test_jffs2(work, src, expected) failures += test_ubifs(work, src, expected) + failures += test_ubi_sparse_volume(work) failures += test_tar(work, src, expected) failures += test_romfs(work, src, expected) failures += test_yaffs2(work, src, expected)