diff --git a/CMakeLists.txt b/CMakeLists.txt index 4ba6bdf..aa896eb 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -132,6 +132,7 @@ add_executable(moria src/extract/spiffs.cpp src/extract/littlefs.cpp src/validators/luks.cpp + src/validators/lzma.cpp ) target_include_directories(moria PRIVATE src third_party) target_compile_definitions(moria PRIVATE MORIA_VERSION="${PROJECT_VERSION}") @@ -232,6 +233,7 @@ if(Python3_Interpreter_FOUND) test_esp32_nvs test_legacy_fs test_luks + test_lzma test_rae_rfp test_vbf test_uboot_env diff --git a/signatures/lzma.toml b/signatures/lzma.toml new file mode 100644 index 0000000..1fe38ae --- /dev/null +++ b/signatures/lzma.toml @@ -0,0 +1,24 @@ +name = "lzma" +category = "compression" + +# Legacy standalone LZMA1 (".lzma alone") has no magic number. Its 13-byte header +# is [props u8][dict_size u32 LE][uncompressed_size u64 LE], then a raw LZMA1 +# stream. Anchor on the near-universal firmware header prefix `5D 00 00`: props +# 0x5D (lc3/lp0/pb2, the toolchain default emitted by xz --format=lzma, LZMA SDK, +# and every vendor loader observed) followed by a dict size that is a multiple of +# 64 KiB (so its low 16 bits are 0). The validator then checks the header fields +# and trial-decodes to a clean end-of-stream marker, so a stray `5D 00 00` in +# unrelated data is rejected (no false positive). Identify + extract. +magic_offset = 0 +validator = "lzma" +confidence = "structural" + +[[magic]] +hex = "5d0000" +endian = "little" + +[doc] +description = "LZMA-compressed data (legacy .lzma alone container, LZMA1)." +references = [ + { title = "LZMA SDK / .lzma format", url = "https://tukaani.org/xz/xz-file-format.txt" }, +] diff --git a/src/extract/compressed.cpp b/src/extract/compressed.cpp index 6ef9687..b7acb33 100644 --- a/src/extract/compressed.cpp +++ b/src/extract/compressed.cpp @@ -19,6 +19,7 @@ constexpr uint64_t MAX_OUT = uint64_t(8) << 30; // decompressed-size cap std::optional codec_for(const std::string& type) { if (type == "gzip") return Compressor::Gzip; if (type == "xz") return Compressor::Xz; + if (type == "lzma") return Compressor::Lzma; if (type == "zstd") return Compressor::Zstd; if (type == "lz4") return Compressor::Lz4; if (type == "lz4_legacy") return Compressor::Lz4Legacy; diff --git a/src/extract/decompress.cpp b/src/extract/decompress.cpp index 2a6400d..00979cc 100644 --- a/src/extract/decompress.cpp +++ b/src/extract/decompress.cpp @@ -241,6 +241,43 @@ std::optional> upx_lzma_block_exact( #endif } +std::optional lzma_alone_probe([[maybe_unused]] std::span src, + [[maybe_unused]] size_t out_cap, + [[maybe_unused]] size_t* in_consumed) { + if (in_consumed) *in_consumed = 0; +#ifdef MORIA_HAVE_LZMA + if (src.size() < 13 || out_cap == 0) return std::nullopt; + lzma_stream s = LZMA_STREAM_INIT; + if (lzma_alone_decoder(&s, UINT64_MAX) != LZMA_OK) return std::nullopt; + s.next_in = src.data(); + s.avail_in = src.size(); + std::vector out; + lzma_ret rc = LZMA_OK; + while (true) { + const size_t old = out.size(); + if (old >= out_cap) { lzma_end(&s); return std::nullopt; } // no end marker within cap + const size_t grow = std::min(1u << 20, out_cap - old); + out.resize(old + grow); + s.next_out = out.data() + old; + s.avail_out = grow; + rc = lzma_code(&s, LZMA_FINISH); + out.resize(old + (grow - s.avail_out)); + if (rc == LZMA_STREAM_END) break; // clean end marker: a real stream + if (rc != LZMA_OK) { lzma_end(&s); return std::nullopt; } // range-coder / data error + if (s.avail_out != 0 && s.avail_in == 0) { // ran out of input, no end marker + lzma_end(&s); + return std::nullopt; + } + } + const uint64_t decoded = out.size(); + if (in_consumed) *in_consumed = src.size() - s.avail_in; + lzma_end(&s); + return decoded; +#else + return std::nullopt; +#endif +} + namespace { #ifdef MORIA_HAVE_ZLIB diff --git a/src/extract/decompress.hpp b/src/extract/decompress.hpp index 5a03bc0..40977bf 100644 --- a/src/extract/decompress.hpp +++ b/src/extract/decompress.hpp @@ -66,6 +66,19 @@ std::optional> upx_lzma_block_exact(std::span lzma_alone_probe(std::span src, size_t out_cap, + size_t* in_consumed = nullptr); + // Streaming decompression of a whole standalone stream/container whose decoded // size is not known in advance (a gzip/xz/zstd/lz4-frame firmware wrapper). Grows // the output buffer as it goes, stopping at `cap` bytes. Unlike decompress(), diff --git a/src/extract/manifest.cpp b/src/extract/manifest.cpp index c619024..91d38be 100644 --- a/src/extract/manifest.cpp +++ b/src/extract/manifest.cpp @@ -68,7 +68,8 @@ Extractor find_extractor(const std::string& type) { if (type == "vbmeta") return extract_vbmeta; if (type == "upx") return extract_upx; if (type == "fit") return extract_fit; - if (type == "gzip" || type == "xz" || type == "zstd" || type == "lz4" || type == "lz4_legacy") + if (type == "gzip" || type == "xz" || type == "zstd" || type == "lz4" || type == "lz4_legacy" || + type == "lzma") return extract_compressed; if (type == "yaffs2") return extract_yaffs2; if (type == "ubifs" || type == "ubi") return extract_ubifs; diff --git a/src/main.cpp b/src/main.cpp index d363bdb..1ee4088 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -275,7 +275,7 @@ void extract_findings(ft::Reader& reader, const std::vector& findin if (!ex) continue; if (f.type != "android_sparse" && f.offset < sparse_end) continue; const bool is_comp = f.type == "gzip" || f.type == "xz" || f.type == "zstd" || - f.type == "lz4"; + f.type == "lz4" || f.type == "lzma"; if (!is_comp && f.offset < comp_end) continue; // inside a compressed stream if (f.type == "jffs2") { if (f.offset >= jffs2_cov) continue; diff --git a/src/mime.hpp b/src/mime.hpp index 79cc502..296c46d 100644 --- a/src/mime.hpp +++ b/src/mime.hpp @@ -20,6 +20,7 @@ inline const char* mime_for_type(const std::string& t) { // --- registered IANA / de-facto standard types ------------------------ if (t == "gzip") return "application/gzip"; if (t == "xz") return "application/x-xz"; + if (t == "lzma") return "application/x-lzma"; if (t == "bzip2") return "application/x-bzip2"; if (t == "zstd") return "application/zstd"; if (t == "lz4" || t == "lz4_legacy") return "application/x-lz4"; diff --git a/src/validators/lzma.cpp b/src/validators/lzma.cpp new file mode 100644 index 0000000..8d1d132 --- /dev/null +++ b/src/validators/lzma.cpp @@ -0,0 +1,69 @@ +// lzma.cpp — legacy standalone LZMA1 (".lzma alone") identifier. See lzma.hpp. +#include "validators/lzma.hpp" + +#include +#include + +#include "extract/decompress.hpp" + +namespace ft { + +namespace { + +// The .lzma header: [props u8][dict_size u32 LE][uncompressed_size u64 LE]. +constexpr uint64_t kUnknownSize = UINT64_MAX; // all-0xFF = size not stored +// props = (pb * 5 + lp) * 9 + lc, with lc<=8, lp<=4, pb<=4 → max 224. +constexpr uint8_t kMaxProps = 224; +// Dictionary bounds. Real firmware uses 8/16/32/64 MiB; the format allows less. +// A multiple of 64 KiB (the `5D 00 00` anchor guarantees the low 16 bits are 0), +// required to be a power of two in [4 KiB, 768 MiB]. +constexpr uint32_t kMinDict = 1u << 12; +constexpr uint32_t kMaxDict = 768u << 20; +// Cap the identify-time trial decode. Larger than any plausible firmware LZMA +// payload, so a real stream reaches its end marker, while a decompression bomb +// cannot make identify decode unbounded output. +constexpr size_t kProbeCap = 512u << 20; + +} // namespace + +bool validate_lzma(ValidatorCtx& ctx) { + const Reader& r = ctx.reader; + const size_t off = ctx.offset; + + auto props = r.at(off, Endian::Little); + auto dict = r.at(off + 1, Endian::Little); + auto usize = r.at(off + 5, Endian::Little); + if (!props || !dict || !usize) return false; // header runs past EOF + + if (*props > kMaxProps) return false; + if (*dict < kMinDict || *dict > kMaxDict) return false; + if ((*dict & (*dict - 1)) != 0) return false; // not a power of two + + const bool known = (*usize != kUnknownSize); + if (known && (*usize == 0 || *usize > kProbeCap)) return false; // implausible declared size + + // FP-proof gate: trial-decode to a clean end-of-stream marker. A stray + // `5D 00 00 ...` cannot reach the marker, so it is rejected here. + auto src = r.bytes(off, r.size() - off); + if (!src) return false; + size_t consumed = 0; + auto decoded = lzma_alone_probe(*src, kProbeCap, &consumed); + if (!decoded) return false; + if (known && *decoded != *usize) return false; // declared size must match actual + + Finding& out = ctx.out; + out.type = "lzma"; + out.category = "compression"; + out.endian = Endian::Little; + out.offset = off; + out.size = consumed; // the exact compressed span, so the region is claimed + const uint32_t dict_mib = *dict >> 20; + std::string dict_str = dict_mib ? std::to_string(dict_mib) + " MiB" : std::to_string(*dict) + " B"; + out.label = std::to_string(*decoded) + " bytes, dict " + dict_str; + out.set_confidence(Confidence::Verified, + "LZMA1 alone: decoded " + std::to_string(*decoded) + + " bytes to end marker, dict " + dict_str); + return true; +} + +} // namespace ft diff --git a/src/validators/lzma.hpp b/src/validators/lzma.hpp new file mode 100644 index 0000000..15976a4 --- /dev/null +++ b/src/validators/lzma.hpp @@ -0,0 +1,12 @@ +// lzma.hpp — legacy standalone LZMA1 (".lzma alone") identifier. +#pragma once +#include "signature.hpp" +namespace ft { +// The legacy .lzma "alone" container has no magic number: a 13-byte header +// (props byte, 4-byte dict size, 8-byte uncompressed size) then a raw LZMA1 +// stream. Anchored on the near-universal firmware header prefix `5D 00 00` +// (props 0x5D = lc3/lp0/pb2, dict a multiple of 64 KiB), this validates the +// header fields and trial-decodes to a clean end-of-stream marker, so a stray +// `5D 00 00` in unrelated data cannot false-positive. Reports `verified`. +bool validate_lzma(ValidatorCtx& ctx); +} // namespace ft diff --git a/src/validators/registry.cpp b/src/validators/registry.cpp index e7daff3..bb94c74 100644 --- a/src/validators/registry.cpp +++ b/src/validators/registry.cpp @@ -13,6 +13,7 @@ #include "validators/legacy_fs.hpp" #include "validators/littlefs.hpp" #include "validators/luks.hpp" +#include "validators/lzma.hpp" #include "validators/partition.hpp" #include "validators/rae_rfp.hpp" #include "validators/spiffs.hpp" @@ -67,6 +68,7 @@ Validator find_validator(const std::string& name) { if (name == "logfs") return validate_logfs; if (name == "littlefs") return validate_littlefs; if (name == "spiffs") return validate_spiffs; + if (name == "lzma") return validate_lzma; return nullptr; } diff --git a/tests/gen_samples.py b/tests/gen_samples.py index ac65f31..a025f17 100755 --- a/tests/gen_samples.py +++ b/tests/gen_samples.py @@ -335,6 +335,15 @@ def zstd(): return bytes([0x28, 0xB5, 0x2F, 0xFD, 0x00]) + b"\0" * 16 # frame header descriptor, reserved bit 0 +def lzma_alone(): + # A real legacy .lzma "alone" stream (props 0x5D, 8 MiB dict, size-unknown + # marker) that decodes to a clean end marker. Unlike the magic-only gzip/xz + # fixtures, the lzma validator trial-decodes, so the fixture must be genuine. + import lzma + payload = b"MORIA lzma alone fixture: the quick brown fox. " * 400 + return lzma.compress(payload, format=lzma.FORMAT_ALONE) + + def cpio_newc(): fields = ["00000000"] * 13 fields[6] = "00000000" # c_filesize @@ -1122,6 +1131,7 @@ def spiffs_img(): ("blob.lz4", lz4, "lz4", STRUCTURAL), ("blob.lz4l", lz4_legacy, "lz4_legacy", STRUCTURAL), ("blob.zst", zstd, "zstd", STRUCTURAL), + ("blob.lzma", lzma_alone, "lzma", VERIFIED), ("initramfs.cpio", cpio_newc, "cpio", CONSISTENT), ("archive.tar", tar, "tar", STRUCTURAL), ("archive.7z", sevenzip, "7z", STRUCTURAL), diff --git a/tests/run.sh b/tests/run.sh index 64d8c88..20b7030 100755 --- a/tests/run.sh +++ b/tests/run.sh @@ -33,6 +33,8 @@ python3 tests/test_esp32_nvs.py python3 tests/test_legacy_fs.py # LUKS1/LUKS2 identify with header parameters (synthetic + real cryptsetup). python3 tests/test_luks.py +# Legacy standalone .lzma identify (verified via trial-decode) + byte-exact extract + FP guards. +python3 tests/test_lzma.py # RFP section extraction + LZARI decode round-trip (self-contained hard gate). python3 tests/test_rae_rfp.py # VBF block extraction + LZSS decode round-trip (self-contained hard gate). diff --git a/tests/test_lzma.py b/tests/test_lzma.py new file mode 100644 index 0000000..56d87db --- /dev/null +++ b/tests/test_lzma.py @@ -0,0 +1,132 @@ +#!/usr/bin/env python3 +"""Legacy standalone LZMA (".lzma alone") identify + extract regression (issue #43). + +moria must recognize a legacy LZMA1 "alone" stream (no magic number: a 13-byte +[props][dict_size][uncompressed_size] header, then a raw LZMA1 stream) and, under +-e, decompress it byte-exact. The format is anchored on the near-universal +firmware header prefix `5D 00 00` and confirmed by a trial-decode to a clean +end-of-stream marker, so the false-positive guards below are the point of the +whole exercise. + +Real streams here come from Python's stdlib lzma module (FORMAT_ALONE), so the +"real stream" cases need no external tool. + +Layers: + 1. Identify a genuine unknown-size .lzma stream -> verified. + 2. Extract byte-exact. + 3. Known-size header: correct declared size passes; wrong declared size rejected. + 4. FP guards: valid header + random payload rejected; a truncated stream (no end + marker) rejected; a sub-threshold dict rejected. + +Run: python3 tests/test_lzma.py +""" +import json +import lzma +import os +import struct +import subprocess +import sys +import tempfile + +HERE = os.path.dirname(os.path.abspath(__file__)) +MORIA = os.path.join(HERE, "..", "build", "moria") + + +def moria_json(data: bytes, extra=None): + with tempfile.NamedTemporaryFile(suffix=".bin") as f: + f.write(data) + f.flush() + cmd = [MORIA, "-j"] + (extra or []) + [f.name] + r = subprocess.run(cmd, capture_output=True, timeout=120) + return json.loads(r.stdout) + + +def lzma_findings(data: bytes): + return [x for x in moria_json(data).get("findings", []) if x["type"] == "lzma"] + + +def real_alone(payload: bytes) -> bytes: + """A genuine unknown-size (0xFF...) .lzma-alone stream: props 0x5D, 8 MiB dict.""" + return lzma.compress(payload, format=lzma.FORMAT_ALONE) + + +def with_known_size(stream: bytes, size: int) -> bytes: + """Patch the 8-byte uncompressed-size field (offset 5) to a definite value.""" + b = bytearray(stream) + struct.pack_into("