diff --git a/src/extract/lzo1x.cpp b/src/extract/lzo1x.cpp index d2eddf7..0206023 100644 --- a/src/extract/lzo1x.cpp +++ b/src/extract/lzo1x.cpp @@ -149,7 +149,12 @@ bool lzo1x_decompress_safe(const uint8_t* in, size_t in_len, uint8_t* out, size_ if (t >= 16) goto match; { if (!in_avail(1)) return false; - size_t off = 1 + 0x0800 + (t >> 2) + (static_cast(*ip++) << 2); + // A short match FOLLOWING a match (length 2) uses a base distance of 1. + // Only the short match after a *literal run* (after_literal_run, length + // 3) adds the 0x0800 (M2_MAX_OFFSET) base -- that offset must NOT be + // applied here, or every match-after-match back-reference is 2048 too + // far and copy_match fails (aborting the whole block). + size_t off = 1 + (t >> 2) + (static_cast(*ip++) << 2); if (!copy_match(off, 2)) return false; } goto match_done; diff --git a/src/extract/ubifs.cpp b/src/extract/ubifs.cpp index d9c5384..a7cc796 100644 --- a/src/extract/ubifs.cpp +++ b/src/extract/ubifs.cpp @@ -152,22 +152,24 @@ std::map> deubi(const Reader& r, uint64_t base, b for (auto& [vol_id, lebs] : map) { if (lebs.empty()) continue; - uint32_t max_lnum = lebs.rbegin()->first; - uint64_t total = uint64_t(max_lnum + 1) * (leb_len_common ? leb_len_common : 0); - // A real volume cannot hold more data than the source image (its LEBs - // live in PEBs within it). A corrupt/mutated lnum otherwise balloons the - // reconstructed image — e.g. a 256 KB input with one stray lnum=8000 - // would allocate tens of MB (up to MAX_VOLUME_BYTES) of mostly-zero - // padding: a memory/disk amplification bomb whose huge mmap later faults. + // Pack the volume's LEBs contiguously in logical (lnum ascending) order. + // The UBIFS node scanner locates nodes by magic + CRC, so their exact + // lnum*leb_len positions are not needed. Sizing/placing by lnum instead + // was both wasteful and, for a *sparse* volume (few LEBs at high lnums, + // as a large mostly-empty data volume produces), forced a clamp to the + // source size that silently dropped every LEB past it. Sizing by the + // actual LEB count keeps every LEB and is inherently bomb-proof: a stray + // lnum can no longer balloon the image (its cost is one LEB, not its + // logical offset). + uint64_t total = uint64_t(lebs.size()) * (leb_len_common ? leb_len_common : 0); if (total == 0 || total > MAX_VOLUME_BYTES) { truncated = true; continue; } - if (total > r.size()) { total = r.size(); truncated = true; } // clamp the bomb std::vector& img = volumes[vol_id]; - img.assign(total, 0); + img.reserve(total); for (auto& [lnum, leb] : lebs) { + (void)lnum; auto span = r.bytes(leb.off, leb.len); if (!span) { truncated = true; continue; } - uint64_t dst = uint64_t(lnum) * leb.len; - if (dst + leb.len <= img.size()) std::memcpy(img.data() + dst, span->data(), leb.len); + img.insert(img.end(), span->begin(), span->end()); } } return volumes; diff --git a/tests/test_extract.py b/tests/test_extract.py index 0866292..61f2151 100644 --- a/tests/test_extract.py +++ b/tests/test_extract.py @@ -222,6 +222,66 @@ def test_ubifs(work, src, expected): return failures +def test_ubi_sparse_volume(work): + """A UBI volume whose LEBs sit at sparse logical numbers (as a large, mostly + empty data volume produces) must reconstruct with EVERY LEB, including the + high-lnum ones. Regression: the reconstructor sized/placed LEBs by lnum and + clamped the image to the source size, silently dropping every LEB past it. + Self-contained: a hand-built 3-PEB UBI (no mkfs tools). Returns failures.""" + import struct + import zlib + + PEB, VIDOFF, DATAOFF = 0x20000, 2048, 4096 + LEB = PEB - DATAOFF + crc = lambda b: zlib.crc32(b) & 0xffffffff + + def ec(): + h = bytearray(64) + h[0:4] = b"UBI#"; h[4] = 1 + struct.pack_into(">I", h, 16, VIDOFF) + struct.pack_into(">I", h, 20, DATAOFF) + struct.pack_into(">I", h, 24, 0x12345678) + struct.pack_into(">I", h, 60, crc(bytes(h[0:60]))) + return h + + def vid(vol_id, lnum, sqnum): + h = bytearray(64) + h[0:4] = b"UBI!"; h[4] = 1; h[5] = 1 # version, vol_type=dynamic + struct.pack_into(">I", h, 8, vol_id) + struct.pack_into(">I", h, 12, lnum) + struct.pack_into(">Q", h, 40, sqnum) + struct.pack_into(">I", h, 60, crc(bytes(h[0:60]))) + return h + + def peb(vol_id, lnum, sqnum, marker): + p = bytearray(b"\xff" * PEB) + p[0:64] = ec() + p[VIDOFF:VIDOFF + 64] = vid(vol_id, lnum, sqnum) + p[DATAOFF:DATAOFF + LEB] = (marker * LEB)[:LEB] + return bytes(p) + + # one volume, LEBs at lnum 0, 1, and 250 (sparse). The last is far past the + # 3-PEB image size, so the old size-clamp dropped it. + img = peb(1, 0, 10, b"LEB0") + peb(1, 1, 11, b"LEB1") + peb(1, 250, 12, b"LEBX") + src = os.path.join(work, "sparse_ubi.bin") + with open(src, "wb") as f: + f.write(img) + outdir = os.path.join(work, "sparse_ubi.out") + subprocess.run([MORIA, "--extract", "-C", outdir, src], capture_output=True) + blob = b"" + for dp, _, fs in os.walk(outdir): + for f in fs: + if f.endswith(".img"): + with open(os.path.join(dp, f), "rb") as fh: + blob += fh.read() + if b"LEB0" in blob and b"LEB1" in blob and b"LEBX" in blob: + print("PASS [ubi-sparse]: all LEBs recovered incl. the sparse high-lnum one") + return 0 + print(f"FAIL [ubi-sparse]: LEB0={b'LEB0' in blob} LEB1={b'LEB1' in blob} " + f"LEBX(sparse)={b'LEBX' in blob}") + return 1 + + def test_tar(work, src, expected): """Pack the tree with GNU tar (gnu/pax/ustar), extract with moria, compare. Needs `tar`. Exercises long names, prefix, and pax path records. Returns @@ -1660,6 +1720,7 @@ def main(): failures += test_ext(work, src, expected) failures += test_jffs2(work, src, expected) failures += test_ubifs(work, src, expected) + failures += test_ubi_sparse_volume(work) failures += test_tar(work, src, expected) failures += test_romfs(work, src, expected) failures += test_yaffs2(work, src, expected) diff --git a/tests/unit/unit_tests.cpp b/tests/unit/unit_tests.cpp index 1a9e36a..35449cb 100644 --- a/tests/unit/unit_tests.cpp +++ b/tests/unit/unit_tests.cpp @@ -257,6 +257,14 @@ static void test_lzo1x() { {0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43,0x41,0x42,0x43}}, {{0x05,0x68,0x65,0x61,0x64,0x65,0x72,0x3a,0xff,0x32,0x00,0x00,0x0d,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0x74,0x72,0x61,0x69,0x6c,0x65,0x72,0x11,0x00,0x00}, {0x68,0x65,0x61,0x64,0x65,0x72,0x3a,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,0x74,0x72,0x61,0x69,0x6c,0x65,0x72}}, + // Regression: a length-2 match immediately FOLLOWING another match (the + // match_next path). The old decoder added the 0x0800 (M2_MAX_OFFSET) base + // that belongs only to the after-a-literal-run short match, over-shooting + // this back-reference by 2048 and failing the whole block -- which + // zero-filled every LZO-compressed filesystem (e.g. a UBIFS/LZO rootfs). + // liblzo2 (python-lzo, raw lzo1x) roundtrips these exact bytes. + {{0x17,0x4f,0x43,0x51,0x4f,0x49,0x63,0x94,0x00,0x93,0x00,0x69,0x6c,0x65,0x0c,0x01,0x11,0x00,0x00}, + {0x4f,0x43,0x51,0x4f,0x49,0x63,0x4f,0x43,0x51,0x4f,0x49,0x4f,0x43,0x51,0x4f,0x49,0x69,0x6c,0x65,0x4f,0x43}}, }; for (const auto& v : vecs) CHECK(lzo_ok(v.comp, v.plain));