From a19ca9904f05d9d83e6a7597dfa8579ece593ef4 Mon Sep 17 00:00:00 2001 From: skmagiik Date: Tue, 29 Sep 2026 21:14:25 -0400 Subject: [PATCH 1/3] Add Windows CE and CAB extraction support --- CMakeLists.txt | 14 + README.md | 7 +- fuzz/build.sh | 2 +- fuzz/fuzz_lzx.cpp | 44 +++ moria.1 | 6 +- signatures/cab.toml | 51 +++ signatures/wince_hive.toml | 33 ++ signatures/wince_rom.toml | 37 +++ src/archives.cpp | 41 +++ src/cab_parse.cpp | 146 ++++++++ src/cab_parse.hpp | 72 ++++ src/extract/cab.cpp | 307 +++++++++++++++++ src/extract/cab.hpp | 30 ++ src/extract/ce_setup.cpp | 305 +++++++++++++++++ src/extract/ce_setup.hpp | 54 +++ src/extract/decompress.cpp | 32 ++ src/extract/decompress.hpp | 9 + src/extract/lzx.cpp | 606 ++++++++++++++++++++++++++++++++++ src/extract/lzx.hpp | 46 +++ src/extract/manifest.cpp | 6 + src/extract/wince_hive.cpp | 83 +++++ src/extract/wince_hive.hpp | 24 ++ src/extract/wince_rom.cpp | 384 +++++++++++++++++++++ src/extract/wince_rom.hpp | 30 ++ src/validators/cab.cpp | 57 ++++ src/validators/cab.hpp | 10 + src/validators/registry.cpp | 6 + src/validators/wince_hive.cpp | 34 ++ src/validators/wince_hive.hpp | 10 + src/validators/wince_rom.cpp | 66 ++++ src/validators/wince_rom.hpp | 11 + src/wince_hive_parse.cpp | 181 ++++++++++ src/wince_hive_parse.hpp | 41 +++ src/wince_rom_parse.cpp | 207 ++++++++++++ src/wince_rom_parse.hpp | 72 ++++ tests/gen_samples.py | 86 +++++ tests/lzxbuild.py | 79 +++++ tests/run.sh | 6 + tests/test_cab.py | 277 ++++++++++++++++ tests/test_wince_hive.py | 132 ++++++++ tests/test_wince_rom.py | 246 ++++++++++++++ 41 files changed, 3883 insertions(+), 7 deletions(-) create mode 100644 fuzz/fuzz_lzx.cpp create mode 100644 signatures/cab.toml create mode 100644 signatures/wince_hive.toml create mode 100644 signatures/wince_rom.toml create mode 100644 src/cab_parse.cpp create mode 100644 src/cab_parse.hpp create mode 100644 src/extract/cab.cpp create mode 100644 src/extract/cab.hpp create mode 100644 src/extract/ce_setup.cpp create mode 100644 src/extract/ce_setup.hpp create mode 100644 src/extract/lzx.cpp create mode 100644 src/extract/lzx.hpp create mode 100644 src/extract/wince_hive.cpp create mode 100644 src/extract/wince_hive.hpp create mode 100644 src/extract/wince_rom.cpp create mode 100644 src/extract/wince_rom.hpp create mode 100644 src/validators/cab.cpp create mode 100644 src/validators/cab.hpp create mode 100644 src/validators/wince_hive.cpp create mode 100644 src/validators/wince_hive.hpp create mode 100644 src/validators/wince_rom.cpp create mode 100644 src/validators/wince_rom.hpp create mode 100644 src/wince_hive_parse.cpp create mode 100644 src/wince_hive_parse.hpp create mode 100644 src/wince_rom_parse.cpp create mode 100644 src/wince_rom_parse.hpp create mode 100644 tests/lzxbuild.py create mode 100644 tests/test_cab.py create mode 100644 tests/test_wince_hive.py create mode 100644 tests/test_wince_rom.py diff --git a/CMakeLists.txt b/CMakeLists.txt index aa896eb..2e1857d 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -68,6 +68,14 @@ add_executable(moria src/extract/esp32_nvs.cpp src/esp32_nvs_parse.cpp src/extract/rae_rfp.cpp + src/extract/wince_rom.cpp + src/extract/cab.cpp + src/extract/wince_hive.cpp + src/wince_hive_parse.cpp + src/extract/ce_setup.cpp + src/cab_parse.cpp + src/extract/lzx.cpp + src/wince_rom_parse.cpp src/extract/lzari.cpp src/extract/vbf.cpp src/extract/vbmeta.cpp @@ -113,6 +121,9 @@ add_executable(moria src/validators/zip.cpp src/validators/jpeg.cpp src/validators/rae_rfp.cpp + src/validators/wince_rom.cpp + src/validators/cab.cpp + src/validators/wince_hive.cpp src/validators/vbf.cpp src/validators/verity.cpp src/validators/vbmeta.cpp @@ -235,6 +246,9 @@ if(Python3_Interpreter_FOUND) test_luks test_lzma test_rae_rfp + test_wince_rom + test_wince_hive + test_cab test_vbf test_uboot_env test_vbmeta diff --git a/README.md b/README.md index 44645b8..23753dd 100644 --- a/README.md +++ b/README.md @@ -36,7 +36,7 @@ moria -j # JSON, for tools and agents moria -e # extract to .extracted/ (-C DIR to choose the output dir) moria -c # carve raw byte ranges to .carved/ (no parsing) moria -E # entropy pass: flag unidentified / possibly-encrypted regions -moria --list # list tar/cpio/zip members without extracting +moria --list # list tar/cpio/zip/cab/CE-ROM members without extracting moria --broad # also load the ~2.5k general file-type signatures moria --help ``` @@ -48,9 +48,10 @@ moria --help Unpacked in-process, no external tools and no sudo: - **Filesystems:** SquashFS, ext2/3/4, F2FS, FAT12/16/32, exFAT, NTFS, HFS+/HFSX, XFS, btrfs, JFFS2, UBI/UBIFS, romfs, YAFFS2, cramfs, EROFS -- **Archives and images:** ZIP, tar, cpio, ISO 9660, Android sparse, Android boot -- **Kernels and wrappers:** U-Boot uImage, U-Boot FIT, standalone gzip / xz / zstd / lz4 streams +- **Archives and images:** ZIP, tar, cpio, MS-CAB (stored/MSZIP/LZX), ISO 9660, Android sparse, Android boot +- **Kernels and wrappers:** U-Boot uImage, U-Boot FIT, Windows CE XIP ROM (nk.bin / .cos), standalone gzip / xz / zstd / lz4 streams - **Firmware packages:** RAE Systems / Honeywell RFP (section table; LZARI-decompresses each section) +- **Configuration stores:** U-Boot environment, ESP32 NVS, Windows CE registry hives (`.hv`) ## Signatures diff --git a/fuzz/build.sh b/fuzz/build.sh index 6b5f6f4..84e8fcd 100755 --- a/fuzz/build.sh +++ b/fuzz/build.sh @@ -11,7 +11,7 @@ mapfile -t SRCS < <(find src -name '*.cpp' ! -name 'main.cpp' | sort) # Xcode clang lacks the libFuzzer runtime; override with a clang that has it # (e.g. CXX=/opt/homebrew/opt/llvm/bin/clang++ on macOS). CXX="${CXX:-clang++}" -for target in fuzz_scan fuzz_esp32_part fuzz_esp32_nvs; do +for target in fuzz_scan fuzz_esp32_part fuzz_esp32_nvs fuzz_lzx; do "$CXX" -std=c++20 -O1 -g -fsanitize=fuzzer,address -I src -I third_party \ -DFT_FUZZ_SIGDIR="\"$PWD/signatures\"" \ "${SRCS[@]}" "fuzz/${target}.cpp" -o "$target" diff --git a/fuzz/fuzz_lzx.cpp b/fuzz/fuzz_lzx.cpp new file mode 100644 index 0000000..84e85ef --- /dev/null +++ b/fuzz/fuzz_lzx.cpp @@ -0,0 +1,44 @@ +// fuzz_lzx.cpp — libFuzzer entry point for the LZX decompressors. +// +// The CE-ROM (CECompress) and MS-CAB framings both drive the same Huffman/match +// decoder from lengths and offsets taken straight out of the input, so they are +// the sharpest edge in the Windows CE / cabinet path. Both entry points are fed +// directly here — a declared output length is taken from the fuzzer bytes too, +// so the "header claims a huge decode" case is covered — and the same bytes go +// through the full scan+identify pipeline so the .cos / .cab validators see +// adversarial input as well. +// +// Build via fuzz/build.sh. Run: ./fuzz_lzx -max_total_time=60 +#include +#include +#include + +#include "extract/lzx.hpp" +#include "reader.hpp" +#include "scan.hpp" +#include "sigload.hpp" + +#ifndef FT_FUZZ_SIGDIR +#define FT_FUZZ_SIGDIR "signatures" +#endif + +extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) { + static const ft::LoadResult sigs = ft::load_signatures(FT_FUZZ_SIGDIR); + if (size < 4) return 0; + + // First three bytes pick the declared output length (capped so the fuzzer + // spends its time on the decoder, not on allocating) and the LZX window. + const size_t out_len = 1 + ((size_t(data[0]) << 8 | data[1]) & 0x3FFFF); + const unsigned window = 15 + (data[2] % 8); // includes the invalid 22 + std::span body(data + 3, size - 3); + + auto ce = ft::ce_decompress_rom(body, out_len); + (void)ce; + auto cab = ft::lzx_decompress_cab(body, out_len, window); + (void)cab; + + ft::Reader reader(std::span(data, size)); + auto findings = ft::scan(reader, sigs.signatures); + (void)findings; + return 0; +} diff --git a/moria.1 b/moria.1 index e8d1d28..548edf2 100644 --- a/moria.1 +++ b/moria.1 @@ -13,7 +13,7 @@ embedded formats (filesystems, containers, bootloaders, executables, compression, media) with byte offsets and confidence tiers. It is identification-first; with .B \-\-extract -it also unpacks SquashFS, ext2/3/4, F2FS, JFFS2, UBIFS, romfs, YAFFS2, cramfs, FAT, exFAT, NTFS, HFS+, XFS, btrfs, EROFS, and ISO9660 filesystems, ZIP/tar/cpio archives, U-Boot uImage/FIT containers, and RAE Systems/Honeywell RFP firmware packages (LZARI-decompressing their sections) +it also unpacks SquashFS, ext2/3/4, F2FS, JFFS2, UBIFS, romfs, YAFFS2, cramfs, FAT, exFAT, NTFS, HFS+, XFS, btrfs, EROFS, and ISO9660 filesystems, ZIP/tar/cpio/MS-CAB archives, U-Boot uImage/FIT containers, Windows CE XIP ROM images and registry hives, and RAE Systems/Honeywell RFP firmware packages (LZARI-decompressing their sections) internally (no sudo). Every other format is still identified and located by offset, ready for another tool such as binwalk. .PP Output is human-readable by default. A single file shows a findings tree @@ -51,11 +51,11 @@ is not reported as an unidentified high-entropy region, so the hint fires only o genuinely unclaimed high-entropy data. .TP .B \-\-list -List tar/cpio/zip member names and sizes. Does not extract. +List tar/cpio/zip/MS-CAB member names and sizes, or a Windows CE ROM's modules and files. Does not extract. .SS Extract .TP .BR \-e ", " \-\-extract -Unpack identified SquashFS, ext2/3/4, F2FS, JFFS2, UBIFS (incl. UBI), romfs, YAFFS2, cramfs, FAT, exFAT, NTFS, EROFS, and ISO9660 filesystems, ZIP/tar/cpio archives, U-Boot uImage/FIT containers, Android boot/sparse images, and RAE Systems/Honeywell RFP firmware packages into +Unpack identified SquashFS, ext2/3/4, F2FS, JFFS2, UBIFS (incl. UBI), romfs, YAFFS2, cramfs, FAT, exFAT, NTFS, EROFS, and ISO9660 filesystems, ZIP/tar/cpio/MS-CAB archives (rebuilding a Windows CE installer cabinet's real install tree from its _setup.xml), U-Boot uImage/FIT containers, Android boot/sparse images, Windows CE XIP ROM images (rebuilding each XIP module into a flat PE), Windows CE registry hives (dumping their recoverable values to text), and RAE Systems/Honeywell RFP firmware packages into .IR file .extracted/ (one subdir per region, named .IR 0x\- ). diff --git a/signatures/cab.toml b/signatures/cab.toml new file mode 100644 index 0000000..42ec149 --- /dev/null +++ b/signatures/cab.toml @@ -0,0 +1,51 @@ +# NOTE: all root-level keys must precede the first [[magic]] / [doc] table. +name = "cab" +category = "archive" + +# Microsoft Cabinet. "MSCF" then three reserved DWORDs that are zero in every +# cabinet the format defines, the total size, the offset of the file table, and +# the format version (1.3 for every cabinet in the wild). Folders hold the +# compressed streams; files are slices of a folder's decompressed output, so the +# validator walks both tables before believing the match. +struct = """ + bytes[4] signature; + u32 reserved1; + u32 cb_cabinet; + u32 reserved2; + u32 coff_files; + u32 reserved3; + u8 version_minor; + u8 version_major; + u16 folders; + u16 files; + u16 flags; +""" + +constraints = [ + "reserved1 == 0", + "reserved2 == 0", + "reserved3 == 0", + "version_major == 1", + "cb_cabinet > 36", + "coff_files >= 36", + "coff_files < cb_cabinet", + "folders > 0", + "files > 0", + "flags <= 7", +] + +size = "cb_cabinet" +size_clamp = true + +# Walks the folder/file tables, names the codec, flags a spanned set. +validator = "cab" + +[[magic]] +ascii = "MSCF" +endian = "little" + +[doc] +description = "Microsoft Cabinet archive (MSCF). Folder streams are stored, MSZIP (DEFLATE), Quantum, or LZX; Windows CE installer cabinets additionally carry a _setup.xml describing the real install paths and registry keys." +references = [ + { title = "[MS-CAB] Cabinet File Format", url = "https://learn.microsoft.com/en-us/openspecs/windows_protocols/ms-cab/" }, +] diff --git a/signatures/wince_hive.toml b/signatures/wince_hive.toml new file mode 100644 index 0000000..51b2a77 --- /dev/null +++ b/signatures/wince_hive.toml @@ -0,0 +1,33 @@ +# NOTE: all root-level keys must precede the first [[magic]] / [doc] table. +name = "wince_hive" +category = "container" + +# Windows CE registry hive (boot.hv / default.hv / user.hv). The file opens with +# the hive block size, a zero DWORD, and the "EKIM" signature at offset 8, then +# a header of GUIDs and hashes before the cell data. CE's cell layout is not NT's +# regf, so the key tree is not walked; the validator instead proves the file is a +# hive by recovering value records from the cell data, and the extractor writes +# what it recovered. +magic_offset = 8 + +struct = """ + u32 block_size; + u32 reserved; + bytes[4] signature; +""" + +constraints = [ + "reserved == 0", + "block_size >= 256", + "block_size <= 65536", + "block_size & (block_size - 1) == 0", +] + +validator = "wince_hive" + +[[magic]] +ascii = "EKIM" +endian = "little" + +[doc] +description = "Windows CE registry hive (EKIM). Holds the device's registry — service configuration, credentials, and certificates. Value records are recovered structurally; CE's cell layout means the key tree they belonged to is not reconstructed." diff --git a/signatures/wince_rom.toml b/signatures/wince_rom.toml new file mode 100644 index 0000000..feab491 --- /dev/null +++ b/signatures/wince_rom.toml @@ -0,0 +1,37 @@ +# NOTE: all root-level keys must precede the first [[magic]] / [doc] table. +name = "wince_rom" +category = "container" + +# Windows CE XIP ROM image (nk.bin, and the Crestron .cos control-processor +# images built from one). The image opens with the CPU's branch to the +# bootstrap; at 0x40 sits the ROM signature "ECEC" followed by pTOC, the VIRTUAL +# address of the ROMHDR, and the same value as an offset from the image start. +# The ROMHDR carries the physical span plus the module and file table counts. +# Nothing here is a length in the header itself, so the validator resolves pTOC +# and derives the span from physfirst/physlast. +magic_offset = 64 + +struct = """ + skip[64]; + bytes[4] signature; + u32 ptoc; + u32 toc_offset; +""" + +constraints = [ + "ptoc != 0", +] + +# Resolves pTOC -> ROMHDR, sizes the image, sets arch/compression. An image +# whose pTOC does not resolve stays `magic` tier and is not extracted. +validator = "wince_rom" + +[[magic]] +ascii = "ECEC" +endian = "little" + +[doc] +description = "Windows CE XIP ROM image (ROMHDR + TOC): kernel modules stored as split E32/O32 PE images plus ROM files, both optionally CECompress-compressed." +references = [ + { title = "Windows CE ROMHDR / TOCentry (romimage)", url = "https://learn.microsoft.com/en-us/previous-versions/windows/embedded/ms924510(v=msdn.10)" }, +] diff --git a/src/archives.cpp b/src/archives.cpp index 1e2254a..b91e135 100644 --- a/src/archives.cpp +++ b/src/archives.cpp @@ -4,6 +4,9 @@ #include #include +#include "cab_parse.hpp" +#include "wince_rom_parse.hpp" + namespace ft { namespace { @@ -121,12 +124,50 @@ void list_zip(const Reader& r, Finding& f) { } } +// A cabinet's members are slices of a folder's decompressed stream; listing +// needs only the two tables, never the streams. +void list_cab(const Reader& r, Finding& f) { + CabHeader h; + std::vector folders; + std::vector files; + if (!cab_header(r, f.offset, h) || !cab_folders(r, f.offset, h, folders) || + !cab_files(r, f.offset, h, files)) + return; + for (const CabFile& cf : files) { + if (f.members.size() >= MAX_MEMBERS) { f.members_truncated = true; break; } + const size_t fi = cf.ifolder < folders.size() ? cf.ifolder : 0; + f.members.push_back({cf.name, cf.size, cab_comp_name(folders[fi].comp()), {}}); + } +} + +// A CE ROM carries two member lists: XIP modules and plain ROM files. Both are +// named in the TOC, so listing is free. +void list_wince_rom(const Reader& r, Finding& f) { + CeRomHeader h; + if (!ce_rom_header(r, f.offset, h)) return; + std::vector modules; + std::vector files; + ce_rom_modules(r, f.offset, h, modules); + ce_rom_files(r, f.offset, h, files); + for (const CeModule& m : modules) { + if (f.members.size() >= MAX_MEMBERS) { f.members_truncated = true; return; } + f.members.push_back({m.name, m.size, "module", {}}); + } + for (const CeFile& cf : files) { + if (f.members.size() >= MAX_MEMBERS) { f.members_truncated = true; return; } + f.members.push_back( + {cf.name, cf.real, cf.comp != cf.real ? "file · cecompress" : "file", {}}); + } +} + } // namespace void list_members(const Reader& r, Finding& f) { if (f.type == "tar") list_tar(r, f); else if (f.type == "cpio") list_cpio(r, f); else if (f.type == "zip") list_zip(r, f); + else if (f.type == "cab") list_cab(r, f); + else if (f.type == "wince_rom") list_wince_rom(r, f); } } // namespace ft diff --git a/src/cab_parse.cpp b/src/cab_parse.cpp new file mode 100644 index 0000000..3415085 --- /dev/null +++ b/src/cab_parse.cpp @@ -0,0 +1,146 @@ +// cab_parse.cpp — MSCF header and table walking. See the header. +#include "cab_parse.hpp" + +namespace ft { + +namespace { + +constexpr size_t kHeaderFixed = 0x24; // CFHEADER up to (and excluding) the reserve fields +constexpr size_t kFolderFixed = 8; +constexpr size_t kFileFixed = 16; +constexpr size_t kMaxName = 512; + +// Read an ASCIIZ field, advancing `off`. False if it is unterminated or absurd. +bool read_asciiz(const Reader& r, size_t& off, std::string& out, size_t limit) { + out.clear(); + for (size_t i = 0; i < limit; ++i) { + auto c = r.at(off + i, Endian::Little); + if (!c) return false; + if (*c == 0) { + off += i + 1; + return true; + } + out += static_cast(*c); + } + return false; +} + +} // namespace + +bool cab_header(const Reader& r, size_t base, CabHeader& out) { + auto u16 = [&](size_t o) { return r.at(base + o, Endian::Little); }; + auto u32 = [&](size_t o) { return r.at(base + o, Endian::Little); }; + + auto cb = u32(0x08); + auto coff = u32(0x10); + auto nfold = u16(0x1a); + auto nfile = u16(0x1c); + auto flags = u16(0x1e); + auto setid = u16(0x20); + auto icab = u16(0x22); + auto vmin = r.at(base + 0x18, Endian::Little); + auto vmaj = r.at(base + 0x19, Endian::Little); + if (!cb || !coff || !nfold || !nfile || !flags || !setid || !icab || !vmin || !vmaj) + return false; + + out.cb_cabinet = *cb; + out.coff_files = *coff; + out.nfolders = *nfold; + out.nfiles = *nfile; + out.flags = *flags; + out.set_id = *setid; + out.icabinet = *icab; + out.ver_minor = *vmin; + out.ver_major = *vmaj; + + size_t off = base + kHeaderFixed; + if (out.flags & kCabReservePresent) { + auto rh = u16(0x24); + auto rf = r.at(base + 0x26, Endian::Little); + auto rd = r.at(base + 0x27, Endian::Little); + if (!rh || !rf || !rd) return false; + out.res_header = *rh; + out.res_folder = *rf; + out.res_data = *rd; + off += 4 + out.res_header; + } + // Spanning-set names sit between the reserve area and the folder table. + std::string tmp; + if (out.flags & kCabPrevCabinet) { + if (!read_asciiz(r, off, tmp, kMaxName)) return false; + if (!read_asciiz(r, off, tmp, kMaxName)) return false; + } + if (out.flags & kCabNextCabinet) { + if (!read_asciiz(r, off, tmp, kMaxName)) return false; + if (!read_asciiz(r, off, tmp, kMaxName)) return false; + } + out.folders_off = off; + + // Basic shape: version 1.3, both tables inside the file, the file table + // where the header says it is. + if (out.ver_major != 1) return false; + if (out.nfolders == 0 || out.nfiles == 0) return false; + const size_t stride = kFolderFixed + out.res_folder; + if (!r.bytes(out.folders_off, size_t(out.nfolders) * stride)) return false; + if (base + size_t(out.coff_files) < out.folders_off + size_t(out.nfolders) * stride) + return false; + if (!r.bytes(base + out.coff_files, size_t(out.nfiles) * kFileFixed)) return false; + return true; +} + +bool cab_folders(const Reader& r, size_t base, const CabHeader& h, std::vector& out) { + const size_t stride = kFolderFixed + h.res_folder; + if (!r.bytes(h.folders_off, size_t(h.nfolders) * stride)) return false; + out.reserve(h.nfolders); + for (uint16_t i = 0; i < h.nfolders; ++i) { + const size_t e = h.folders_off + size_t(i) * stride; + CabFolder f; + f.coff_data = r.at(e, Endian::Little).value_or(0); + f.ndata = r.at(e + 4, Endian::Little).value_or(0); + f.type = r.at(e + 6, Endian::Little).value_or(0); + // A folder's data must start inside the cabinet. + if (base + size_t(f.coff_data) >= r.size()) return false; + out.push_back(f); + } + return true; +} + +bool cab_files(const Reader& r, size_t base, const CabHeader& h, std::vector& out) { + size_t off = base + h.coff_files; + out.reserve(h.nfiles); + for (uint16_t i = 0; i < h.nfiles; ++i) { + auto u16 = [&](size_t o) { return r.at(off + o, Endian::Little); }; + auto u32 = [&](size_t o) { return r.at(off + o, Endian::Little); }; + auto size = u32(0); + auto fo = u32(4); + auto ifol = u16(8); + auto date = u16(10); + auto time = u16(12); + auto attr = u16(14); + if (!size || !fo || !ifol || !date || !time || !attr) return false; + CabFile f; + f.size = *size; + f.folder_off = *fo; + f.ifolder = *ifol; + f.date = *date; + f.time = *time; + f.attribs = *attr; + off += kFileFixed; + if (!read_asciiz(r, off, f.name, kMaxName)) return false; + if (f.name.empty()) continue; + out.push_back(std::move(f)); + } + return true; +} + +const char* cab_comp_name(CabComp c) { + switch (c) { + case CabComp::None: return "none"; + case CabComp::MsZip: return "mszip"; + case CabComp::Quantum: return "quantum"; + case CabComp::Lzx: return "lzx"; + } + return "unknown"; +} + +} // namespace ft diff --git a/src/cab_parse.hpp b/src/cab_parse.hpp new file mode 100644 index 0000000..4b3c206 --- /dev/null +++ b/src/cab_parse.hpp @@ -0,0 +1,72 @@ +// cab_parse.hpp — Microsoft Cabinet (MSCF) header and table walking. +// +// A cabinet is three tables and a blob: CFHEADER, then cFolders CFFOLDERs +// (each a compression type plus a pointer to its chain of CFDATA blocks), then +// cFiles CFFILEs (each a name plus an offset INTO ITS FOLDER'S decompressed +// stream), then the CFDATA blocks themselves. Files are not individually +// compressed — a folder is one stream and its files are slices of it. +// +// Any of the three tables may carry per-structure reserved bytes whose widths +// live in the header, so every stride here is computed, never assumed. +// Shared by the validator and the extractor. +#pragma once + +#include +#include +#include + +#include "reader.hpp" + +namespace ft { + +// CFHEADER.flags +constexpr uint16_t kCabPrevCabinet = 0x0001; +constexpr uint16_t kCabNextCabinet = 0x0002; +constexpr uint16_t kCabReservePresent = 0x0004; + +// CFFOLDER.typeCompress, low nibble. +enum class CabComp : uint8_t { None = 0, MsZip = 1, Quantum = 2, Lzx = 3 }; + +struct CabHeader { + uint32_t cb_cabinet = 0; // total cabinet size, from the header + uint32_t coff_files = 0; // file offset of the first CFFILE + uint16_t nfolders = 0; + uint16_t nfiles = 0; + uint16_t flags = 0; + uint16_t set_id = 0; + uint16_t icabinet = 0; + uint8_t ver_major = 0, ver_minor = 0; + uint16_t res_header = 0; // abReserve sizes, 0 unless kCabReservePresent + uint8_t res_folder = 0, res_data = 0; + size_t folders_off = 0; // file offset of the first CFFOLDER +}; + +struct CabFolder { + uint32_t coff_data = 0; // file offset of this folder's first CFDATA + uint16_t ndata = 0; // number of CFDATA blocks + uint16_t type = 0; // raw typeCompress (low nibble = codec, bits 8-12 = LZX window) + CabComp comp() const { return static_cast(type & 0x0f); } + unsigned lzx_window() const { return (type >> 8) & 0x1f; } +}; + +struct CabFile { + std::string name; // backslash-separated, as stored + uint32_t size = 0; // uncompressed size + uint32_t folder_off = 0; // offset into the folder's decompressed stream + uint16_t ifolder = 0; + uint16_t attribs = 0; + uint16_t date = 0, time = 0; +}; + +// Parse the CFHEADER at `base`, resolving the optional reserve fields and the +// prev/next cabinet name strings so folders_off lands on the first CFFOLDER. +bool cab_header(const Reader& r, size_t base, CabHeader& out); + +// Walk the folder / file tables. Both return false if the table does not fit +// inside the file; individual malformed entries are skipped. +bool cab_folders(const Reader& r, size_t base, const CabHeader& h, std::vector& out); +bool cab_files(const Reader& r, size_t base, const CabHeader& h, std::vector& out); + +const char* cab_comp_name(CabComp c); + +} // namespace ft diff --git a/src/extract/cab.cpp b/src/extract/cab.cpp new file mode 100644 index 0000000..f66de70 --- /dev/null +++ b/src/extract/cab.cpp @@ -0,0 +1,307 @@ +// cab.cpp — Microsoft Cabinet extraction. See cab.hpp. +// +// Files are not stored individually: a CFFOLDER is one compressed stream split +// across CFDATA blocks, and each CFFILE names a byte range inside that stream's +// decompressed output. So a folder is decoded once and then sliced, and folders +// with no files pointing at them are never decoded at all. +#include "extract/cab.hpp" + +#include +#include +#include +#include +#include +#include + +#include "cab_parse.hpp" +#include "extract/ce_setup.hpp" +#include "extract/decompress.hpp" +#include "extract/lzx.hpp" +#include "extract/safepath.hpp" + +namespace ft { + +namespace { + +constexpr uint64_t kMaxFolderOut = uint64_t(1) << 30; // 1 GiB of output per folder +constexpr size_t kMszipFrame = 32768; // MSZIP block size and window + +// CFFILE.iFolder values that mean "this file continues across cabinets". +constexpr uint16_t kFolderContinuedFrom = 0xFFFD; +constexpr uint16_t kFolderContinuedTo = 0xFFFE; +constexpr uint16_t kFolderContinuedBoth = 0xFFFF; + +std::string lower(std::string s) { + for (char& c : s) c = static_cast(std::tolower(static_cast(c))); + return s; +} + +bool ends_with(const std::string& s, const char* suffix) { + const size_t n = std::strlen(suffix); + return s.size() >= n && std::equal(s.end() - n, s.end(), suffix); +} + +// A CAB member name is a backslash-separated relative path. Keep the structure +// but strip anything that could escape the output root or upset a filesystem. +std::string safe_member_path(const std::string& name) { + std::string norm = name; + for (char& c : norm) + if (c == '\\') c = '/'; + std::string out; + size_t i = 0; + while (i < norm.size()) { + size_t j = norm.find('/', i); + if (j == std::string::npos) j = norm.size(); + std::string part = norm.substr(i, j - i); + i = j + 1; + if (part.empty() || part == "." || part == "..") continue; + for (char& c : part) { + const unsigned char u = static_cast(c); + if (u < 0x20 || c == ':' || c == '*' || c == '?' || c == '"' || c == '<' || c == '>' || + c == '|') + c = '_'; + } + if (!out.empty()) out += '/'; + out += part; + } + return out.empty() ? std::string("_unnamed") : out; +} + +// Decode one folder's CFDATA chain. `why` names the reason on failure so the +// caller can put it in the manifest rather than silently dropping the folder. +bool folder_bytes(const Reader& r, size_t base, const CabHeader& h, const CabFolder& fo, + std::vector& out, std::string& why) { + const size_t stride = 8 + h.res_data; + const CabComp codec = fo.comp(); + if (codec == CabComp::Quantum) { + why = "unsupported:quantum"; + return false; + } + + size_t off = base + fo.coff_data; + std::vector lzx_in; // LZX is one bitstream across the whole folder + uint64_t lzx_out = 0; + + for (uint16_t i = 0; i < fo.ndata; ++i) { + auto cb = r.at(off + 4, Endian::Little); + auto cu = r.at(off + 6, Endian::Little); + if (!cb || !cu) { + why = "truncated CFDATA header"; + return false; + } + auto data = r.bytes(off + stride, *cb); + if (!data) { + why = "CFDATA runs past the end of the file"; + return false; + } + off += stride + *cb; + + if (out.size() + *cu > kMaxFolderOut || lzx_out + *cu > kMaxFolderOut) { + why = "folder exceeds the output cap"; + return false; + } + + switch (codec) { + case CabComp::None: + if (*cu != *cb) { + why = "stored block size mismatch"; + return false; + } + out.insert(out.end(), data->begin(), data->end()); + break; + case CabComp::MsZip: { + if (data->size() < 2 || (*data)[0] != 'C' || (*data)[1] != 'K') { + why = "MSZIP block missing its CK marker"; + return false; + } + // A block may match back into the previous block's output. + const size_t hist = std::min(out.size(), kMszipFrame); + auto blk = mszip_block(data->subspan(2), *cu, + std::span(out).last(hist)); + if (!blk) { + why = "MSZIP block failed to inflate"; + return false; + } + out.insert(out.end(), blk->begin(), blk->end()); + break; + } + case CabComp::Lzx: + lzx_in.insert(lzx_in.end(), data->begin(), data->end()); + lzx_out += *cu; + break; + case CabComp::Quantum: return false; // handled above + } + } + + if (codec == CabComp::Lzx) { + const unsigned win = fo.lzx_window(); + auto dec = lzx_decompress_cab(lzx_in, static_cast(lzx_out), win); + if (!dec) { + why = "LZX folder failed to decode"; + return false; + } + out = std::move(*dec); + } + return true; +} + +// Resolve a CFFILE's folder index, including the cross-cabinet sentinels. +size_t folder_index(uint16_t ifolder, size_t nfolders) { + switch (ifolder) { + case kFolderContinuedFrom: + case kFolderContinuedBoth: return 0; + case kFolderContinuedTo: return nfolders - 1; + default: return ifolder < nfolders ? ifolder : SIZE_MAX; + } +} + +} // namespace + +bool extract_cab(const Reader& r, const Finding& f, SafeRoot& root, const std::string& subdir, + Extracted& out) { + out.offset = f.offset; + out.type = "cab"; + out.root = subdir; + + CabHeader h; + std::vector folders; + std::vector files; + if (!cab_header(r, f.offset, h) || !cab_folders(r, f.offset, h, folders) || + !cab_files(r, f.offset, h, files)) { + out.status = "error:header"; + return true; + } + if (!root.make_dir(subdir)) { + out.status = "error:mkdir"; + return true; + } + out.consumed = h.cb_cabinet; + + // Group files by the folder whose stream they slice. + std::vector> by_folder(folders.size()); + bool spanned = false; + for (size_t i = 0; i < files.size(); ++i) { + const size_t fi = folder_index(files[i].ifolder, folders.size()); + if (fi == SIZE_MAX) { + out.warnings.push_back(files[i].name + ": folder index out of range"); + continue; + } + if (files[i].ifolder >= kFolderContinuedFrom) spanned = true; + by_folder[fi].push_back(i); + } + if (spanned) + out.warnings.push_back( + "file(s) continue across cabinets; only the part stored here is recoverable"); + + // The CE installer manifest tells us what every mangled member really is, + // so read it before writing anything. + CeSetup setup; + bool have_setup = false; + std::unordered_map install_map; // lowercased source -> dest + for (size_t fi = 0; fi < folders.size() && !have_setup; ++fi) { + for (size_t idx : by_folder[fi]) { + if (lower(files[idx].name) != "_setup.xml") continue; + std::vector stream; + std::string why; + if (!folder_bytes(r, f.offset, h, folders[fi], stream, why)) break; + const CabFile& cf = files[idx]; + if (uint64_t(cf.folder_off) + cf.size > stream.size()) break; + if (ce_setup_parse(std::span(stream).subspan(cf.folder_off, cf.size), + setup)) { + have_setup = true; + for (const CeSetupEntry& e : setup.files) install_map[lower(e.source)] = e.dest; + } + break; + } + } + + size_t failures = 0, mapped = 0, unmapped = 0; + bool made_fs = false, made_unmapped = false; + + for (size_t fi = 0; fi < folders.size(); ++fi) { + if (by_folder[fi].empty()) continue; + std::vector stream; + std::string why; + if (!folder_bytes(r, f.offset, h, folders[fi], stream, why)) { + ++failures; + out.warnings.push_back("folder " + std::to_string(fi) + ": " + why); + if (why.rfind("unsupported:", 0) == 0 && out.status.empty()) out.status = why; + continue; + } + + for (size_t idx : by_folder[fi]) { + const CabFile& cf = files[idx]; + if (uint64_t(cf.folder_off) + cf.size > stream.size()) { + ++failures; + out.warnings.push_back(cf.name + ": extends past its folder's data"); + continue; + } + const uint8_t* p = stream.data() + cf.folder_off; + const std::vector data(p, p + cf.size); + + std::string rel; + if (!have_setup) { + rel = subdir + "/" + safe_member_path(cf.name); + } else { + const std::string lname = lower(cf.name); + auto it = install_map.find(lname); + if (lname == "_setup.xml") { + rel = subdir + "/_setup.xml"; + } else if (it != install_map.end()) { + if (!made_fs && root.make_dir(subdir + "/fs")) { + made_fs = true; + out.dirs++; + } + rel = subdir + "/fs/" + it->second; + ++mapped; + } else if (ends_with(lname, ".999")) { + // cabwiz stores the setup DLL (Install_Init / Install_Exit) + // as the .999 member and the install header as .000. + rel = subdir + "/setup.dll"; + } else if (ends_with(lname, ".000")) { + rel = subdir + "/install-header.000"; + } else { + if (!made_unmapped && root.make_dir(subdir + "/unmapped")) { + made_unmapped = true; + out.dirs++; + } + rel = subdir + "/unmapped/" + safe_member_path(cf.name); + ++unmapped; + } + } + + if (!root.write_file(rel, data, 0644)) { + out.status = "error:write"; + return true; + } + out.files++; + out.bytes += data.size(); + } + } + + if (have_setup && !setup.registry.empty()) { + const std::string reg = ce_setup_reg_file(setup.registry); + const std::vector bytes(reg.begin(), reg.end()); + if (root.write_file(subdir + "/registry.reg", bytes, 0644)) { + out.files++; + out.bytes += bytes.size(); + } + } + // Members _setup.xml never mentions are kept under unmapped/ rather than + // dropped, but say so: they are the part of the install the manifest did + // not explain. + if (unmapped) + out.warnings.push_back(std::to_string(unmapped) + + " member(s) not named by _setup.xml, kept under unmapped/"); + + if (out.files == 0) { + if (out.status.empty()) out.status = "error:empty"; + } else if (failures || !out.status.empty()) { + if (out.status.rfind("unsupported:", 0) != 0) out.status = "partial"; + } else { + out.status = "ok"; + } + return true; +} + +} // namespace ft diff --git a/src/extract/cab.hpp b/src/extract/cab.hpp new file mode 100644 index 0000000..5142f48 --- /dev/null +++ b/src/extract/cab.hpp @@ -0,0 +1,30 @@ +// cab.hpp — Microsoft Cabinet extractor, including Windows CE installer CABs. +// +// A plain cabinet unpacks to its member names under the output subdir. +// +// A Windows CE installer cabinet needs one more step to be readable. cabwiz +// stores every payload under a mangled 8.3 name (CPHPRO~1.004) and keeps the +// real destination path, the registry keys, and the application name in a +// _setup.xml wap-provisioningdoc alongside them. Extracting the members alone +// gives you three dozen files called things like 3-SERI~1.001. So when a +// _setup.xml is present this rebuilds the tree the installer would have +// produced — `fs/` holding each payload at its real path with CE's directory +// macros expanded, `registry.reg` holding the keys, and the two cabwiz +// housekeeping members (the install header and the setup DLL) named for what +// they are. +#pragma once + +#include + +#include "extract/manifest.hpp" +#include "finding.hpp" +#include "reader.hpp" + +namespace ft { + +class SafeRoot; + +bool extract_cab(const Reader& r, const Finding& f, SafeRoot& root, const std::string& subdir, + Extracted& out); + +} // namespace ft diff --git a/src/extract/ce_setup.cpp b/src/extract/ce_setup.cpp new file mode 100644 index 0000000..014d1fb --- /dev/null +++ b/src/extract/ce_setup.cpp @@ -0,0 +1,305 @@ +// ce_setup.cpp — _setup.xml parsing and install-path expansion. See the header. +// +// The document is machine-generated and shallow, so this walks it with a small +// tag scanner and a stack of the `type` attributes rather than pulling in an +// XML library: the structure being read is "which nested characteristic types +// enclose this Extract", which is exactly a stack of names. +#include "extract/ce_setup.hpp" + +#include +#include +#include + +namespace ft { + +namespace { + +// CE special-directory macros used in _setup.xml destination paths. +struct Macro { + const char* macro; + const char* path; +}; +constexpr Macro kCeDirs[] = { + {"%CE1%", "\\Program Files"}, + {"%CE2%", "\\Windows"}, + {"%CE3%", "\\Windows\\Desktop"}, + {"%CE4%", "\\Windows\\StartUp"}, + {"%CE5%", "\\My Documents"}, + {"%CE6%", "\\Program Files\\Accessories"}, + {"%CE7%", "\\Program Files\\Communications"}, + {"%CE8%", "\\Program Files\\Games"}, + {"%CE9%", "\\Program Files\\Pocket Outlook"}, + {"%CE10%", "\\Program Files\\Office"}, + {"%CE11%", "\\Windows\\Start Menu\\Programs"}, + {"%CE12%", "\\Windows\\Start Menu\\Accessories"}, + {"%CE13%", "\\Windows\\Start Menu\\Communications"}, + {"%CE14%", "\\Windows\\Start Menu\\Programs\\Games"}, + {"%CE15%", "\\Windows\\Fonts"}, + {"%CE16%", "\\Windows\\Recent"}, + {"%CE17%", "\\Windows\\Start Menu"}, +}; + +std::string replace_all(std::string s, const std::string& from, const std::string& to) { + if (from.empty()) return s; + for (size_t p = s.find(from); p != std::string::npos; p = s.find(from, p + to.size())) + s.replace(p, from.size(), to); + return s; +} + +// Expand CE macros and normalise to a relative POSIX path with no traversal. +std::string ce_path(std::string p) { + for (const Macro& m : kCeDirs) p = replace_all(std::move(p), m.macro, m.path); + p = replace_all(std::move(p), "%InstallDir%", "InstallDir"); + for (char& c : p) + if (c == '\\') c = '/'; + std::string out; + size_t i = 0; + while (i < p.size()) { + size_t j = p.find('/', i); + if (j == std::string::npos) j = p.size(); + std::string part = p.substr(i, j - i); + i = j + 1; + if (part.empty() || part == "." || part == "..") continue; + // Anything still carrying a macro (%CE99%, %AppName%) would make an odd + // directory name; keep it readable but harmless. + for (char& c : part) { + const unsigned char u = static_cast(c); + if (u < 0x20 || c == ':' || c == '*' || c == '?' || c == '"' || c == '<' || c == '>' || + c == '|') + c = '_'; + } + if (!out.empty()) out += '/'; + out += part; + } + return out; +} + +std::string decode_entities(const std::string& s) { + if (s.find('&') == std::string::npos) return s; + std::string o; + o.reserve(s.size()); + for (size_t i = 0; i < s.size();) { + if (s[i] != '&') { + o += s[i++]; + continue; + } + const size_t semi = s.find(';', i); + if (semi == std::string::npos || semi - i > 10) { + o += s[i++]; + continue; + } + const std::string ent = s.substr(i + 1, semi - i - 1); + if (ent == "amp") o += '&'; + else if (ent == "lt") o += '<'; + else if (ent == "gt") o += '>'; + else if (ent == "quot") o += '"'; + else if (ent == "apos") o += '\''; + else if (ent.size() > 1 && ent[0] == '#') { + const long v = std::strtol(ent.c_str() + (ent[1] == 'x' || ent[1] == 'X' ? 2 : 1), + nullptr, (ent[1] == 'x' || ent[1] == 'X') ? 16 : 10); + if (v > 0 && v < 0x80) o += static_cast(v); + // Non-ASCII character references are left out rather than guessed at. + } else { + o += s.substr(i, semi - i + 1); // unknown entity: keep it verbatim + } + i = semi + 1; + } + return o; +} + +struct Tag { + std::string name; + std::vector> attrs; + bool closing = false; + bool self_closing = false; +}; + +bool is_space(char c) { return c == ' ' || c == '\t' || c == '\r' || c == '\n'; } + +// Scan the next element starting at or after `i`. False at end of document. +bool next_tag(const std::string& s, size_t& i, Tag& t) { + for (;;) { + const size_t lt = s.find('<', i); + if (lt == std::string::npos) return false; + size_t p = lt + 1; + if (p < s.size() && (s[p] == '!' || s[p] == '?')) { // comment / PI / doctype + const size_t gt = s.find('>', p); + if (gt == std::string::npos) return false; + i = gt + 1; + continue; + } + t = Tag{}; + if (p < s.size() && s[p] == '/') { + t.closing = true; + ++p; + } + const size_t ns = p; + while (p < s.size() && !is_space(s[p]) && s[p] != '>' && s[p] != '/') ++p; + t.name = s.substr(ns, p - ns); + + while (p < s.size()) { + while (p < s.size() && is_space(s[p])) ++p; + if (p >= s.size()) return false; + if (s[p] == '/') { + t.self_closing = true; + ++p; + continue; + } + if (s[p] == '>') { + i = p + 1; + return !t.name.empty(); + } + const size_t as = p; + while (p < s.size() && !is_space(s[p]) && s[p] != '=' && s[p] != '>' && s[p] != '/') + ++p; + const std::string key = s.substr(as, p - as); + std::string val; + while (p < s.size() && is_space(s[p])) ++p; + if (p < s.size() && s[p] == '=') { + ++p; + while (p < s.size() && is_space(s[p])) ++p; + if (p < s.size() && (s[p] == '"' || s[p] == '\'')) { + const char q = s[p++]; + const size_t vs = p; + while (p < s.size() && s[p] != q) ++p; + val = decode_entities(s.substr(vs, p - vs)); + if (p < s.size()) ++p; + } + } + if (!key.empty()) t.attrs.emplace_back(key, std::move(val)); + } + return false; + } +} + +const std::string* attr(const Tag& t, const char* name) { + for (const auto& a : t.attrs) + if (a.first == name) return &a.second; + return nullptr; +} + +std::string join(const std::vector& parts, size_t from, size_t to, char sep) { + std::string o; + for (size_t i = from; i < to; ++i) { + if (!o.empty()) o += sep; + o += parts[i]; + } + return o; +} + +} // namespace + +bool ce_setup_parse(std::span xml, CeSetup& out) { + const std::string s(reinterpret_cast(xml.data()), xml.size()); + size_t i = 0; + Tag t; + + // Stack of enclosing values. Its first element + // names the section (Install / FileOperation / Registry); the rest spell out + // either an install path or a registry key. + std::vector stack; + bool saw_section = false; + + while (next_tag(s, i, t)) { + if (t.name == "characteristic") { + if (t.closing) { + if (!stack.empty()) stack.pop_back(); + continue; + } + const std::string* ty = attr(t, "type"); + stack.push_back(ty ? *ty : std::string()); + if (stack.size() == 1) saw_section = true; + + // A registry key element is the key itself; record it now so keys + // with no values still appear. + if (stack.size() >= 2 && stack[0] == "Registry") + out.registry.push_back({join(stack, 1, stack.size(), '\\'), {}}); + + if (t.self_closing && !stack.empty()) stack.pop_back(); + continue; + } + if (t.name != "parm" || t.closing || stack.empty()) continue; + + const std::string* nm = attr(t, "name"); + const std::string* val = attr(t, "value"); + if (!nm) continue; + + if (stack[0] == "Install") { + if (*nm == "AppName" && val) out.appname = *val; + } else if (stack[0] == "FileOperation") { + // inside an Extract: everything between the + // section and the Extract is the destination path. + if (*nm == "Source" && val && stack.back() == "Extract" && stack.size() >= 3) { + CeSetupEntry e; + e.source = *val; + e.dest = ce_path(join(stack, 1, stack.size() - 1, '/')); + if (!e.source.empty() && !e.dest.empty()) out.files.push_back(std::move(e)); + } + } else if (stack[0] == "Registry") { + if (!out.registry.empty()) + out.registry.back().values.push_back( + {*nm, val ? *val : std::string(), + attr(t, "datatype") ? *attr(t, "datatype") : std::string()}); + } + } + // Drop registry keys that carried no values (pure path elements). + out.registry.erase(std::remove_if(out.registry.begin(), out.registry.end(), + [](const CeRegKey& k) { return k.values.empty(); }), + out.registry.end()); + return saw_section; +} + +std::string ce_setup_reg_file(const std::vector& keys) { + struct Hive { + const char* abbrev; + const char* full; + }; + constexpr Hive kHives[] = { + {"HKLM", "HKEY_LOCAL_MACHINE"}, + {"HKCU", "HKEY_CURRENT_USER"}, + {"HKCR", "HKEY_CLASSES_ROOT"}, + {"HKU", "HKEY_USERS"}, + }; + + std::string o = "Windows Registry Editor Version 5.00\n\n"; + for (const CeRegKey& k : keys) { + const size_t sep = k.key.find('\\'); + const std::string head = sep == std::string::npos ? k.key : k.key.substr(0, sep); + const std::string rest = sep == std::string::npos ? std::string() : k.key.substr(sep + 1); + std::string full; + for (const Hive& h : kHives) + if (head == h.abbrev) { + full = h.full; + if (!rest.empty()) full += "\\" + rest; + break; + } + // CE setup docs normally carry their own hive prefix; anything else is + // a machine key by convention. + if (full.empty()) full = std::string("HKEY_LOCAL_MACHINE\\") + k.key; + + o += "[" + full + "]\n"; + for (const CeRegValue& v : k.values) { + if (v.name.empty()) continue; + if (v.datatype == "integer") { + char* end = nullptr; + const long n = std::strtol(v.value.c_str(), &end, 10); + if (end && *end == '\0' && !v.value.empty()) { + char buf[32]; + std::snprintf(buf, sizeof(buf), "%08lx", static_cast(n)); + o += "\"" + v.name + "\"=dword:" + buf + "\n"; + continue; + } + } + std::string esc; + for (char c : v.value) { + if (c == '\\' || c == '"') esc += '\\'; + esc += c; + } + o += "\"" + v.name + "\"=\"" + esc + "\"\n"; + } + o += "\n"; + } + return o; +} + +} // namespace ft diff --git a/src/extract/ce_setup.hpp b/src/extract/ce_setup.hpp new file mode 100644 index 0000000..cc09ad7 --- /dev/null +++ b/src/extract/ce_setup.hpp @@ -0,0 +1,54 @@ +// ce_setup.hpp — the _setup.xml inside a Windows CE installer cabinet. +// +// cabwiz emits a wap-provisioningdoc that is the only place a CE installer CAB +// records what its files actually are. Three sections matter: +// +// parms, including AppName +// a directory tree of nested +// elements ending in an +// , +// where X is the mangled 8.3 name the payload is stored under +// nested key elements whose s +// are the values to write +// +// Destination paths use CE's directory macros (%CE2% = \Windows, and so on), +// expanded here into ordinary relative paths. +#pragma once + +#include +#include +#include +#include + +namespace ft { + +struct CeRegValue { + std::string name; + std::string value; + std::string datatype; // "string" | "integer" | ... +}; + +struct CeRegKey { + std::string key; // backslash-separated, usually starting at a hive prefix + std::vector values; +}; + +struct CeSetupEntry { + std::string source; // mangled 8.3 name of the CAB member + std::string dest; // install path, CE macros already expanded, relative + safe +}; + +struct CeSetup { + std::string appname; + std::vector files; + std::vector registry; +}; + +// Parse a _setup.xml. Returns false only if the document has no recognizable +// provisioning structure at all. +bool ce_setup_parse(std::span xml, CeSetup& out); + +// Render the collected registry keys as a Windows .reg file. +std::string ce_setup_reg_file(const std::vector& keys); + +} // namespace ft diff --git a/src/extract/decompress.cpp b/src/extract/decompress.cpp index 00979cc..44d4b1f 100644 --- a/src/extract/decompress.cpp +++ b/src/extract/decompress.cpp @@ -475,6 +475,38 @@ std::optional> decompress_stream([[maybe_unused]] Compresso } } +std::optional> mszip_block([[maybe_unused]] std::span src, + [[maybe_unused]] size_t out_len, + [[maybe_unused]] std::span history) { + if (out_len == 0) return std::nullopt; +#ifdef MORIA_HAVE_ZLIB + std::vector out(out_len); + z_stream s{}; + if (inflateInit2(&s, -15) != Z_OK) return std::nullopt; + // For raw inflate the dictionary is set up front; MSZIP carries the tail of + // the previous block's output as the window a match may reach back into. + if (!history.empty() && + inflateSetDictionary(&s, history.data(), static_cast(history.size())) != Z_OK) { + inflateEnd(&s); + return std::nullopt; + } + s.next_in = const_cast(src.data()); + s.avail_in = static_cast(src.size()); + s.next_out = out.data(); + s.avail_out = static_cast(out_len); + const int rc = inflate(&s, Z_FINISH); + const size_t produced = out_len - s.avail_out; + inflateEnd(&s); + // Z_BUF_ERROR after a full block is normal: the deflate stream ends without + // a final block marker in some MSZIP encoders. + if (rc != Z_STREAM_END && rc != Z_OK && rc != Z_BUF_ERROR) return std::nullopt; + if (produced != out_len) return std::nullopt; + return out; +#else + return std::nullopt; +#endif +} + std::optional> decompress(Compressor c, std::span src, size_t max_out) { if (src.empty() || max_out == 0) return std::nullopt; diff --git a/src/extract/decompress.hpp b/src/extract/decompress.hpp index 40977bf..e5d66b6 100644 --- a/src/extract/decompress.hpp +++ b/src/extract/decompress.hpp @@ -66,6 +66,15 @@ std::optional> upx_lzma_block_exact(std::span> mszip_block(std::span src, size_t out_len, + std::span history); + // Identify-time probe for a legacy standalone LZMA1 (".lzma alone") stream: the // FP-proof gate for a magicless format. Decodes `src` with the .lzma-alone // decoder, growing the output up to `out_cap` bytes, and succeeds ONLY if the diff --git a/src/extract/lzx.cpp b/src/extract/lzx.cpp new file mode 100644 index 0000000..097e2f6 --- /dev/null +++ b/src/extract/lzx.cpp @@ -0,0 +1,606 @@ +// lzx.cpp — LZX bitstream decoder plus the CE-ROM and MS-CAB framings. +// See lzx.hpp for what each entry point expects. +// +// The bitstream layer follows the published LZX description: 16-bit +// little-endian refills into a 32-bit buffer, MSB-first consumption, a pretree +// coding the main/length tree code lengths as deltas mod 17, three repeated +// match offsets (R0/R1/R2), and an optional x86 CALL (0xE8) translation pass +// over the decoded bytes. Written against moria's own bounds-checked buffers. +#include "extract/lzx.hpp" + +#include +#include + +namespace ft { + +namespace { + +constexpr unsigned kNumChars = 256; +constexpr unsigned kNumPrimaryLengths = 7; +constexpr unsigned kMinMatch = 2; +constexpr unsigned kMaxMatch = 257; + +constexpr unsigned kPretreeElems = 20; +constexpr unsigned kSecondaryElems = 249; +constexpr unsigned kAlignedElems = 8; + +constexpr unsigned kPretreeMaxSymbols = kPretreeElems; +constexpr unsigned kPretreeTableBits = 6; +constexpr unsigned kMaintreeMaxSymbols = kNumChars + (51u << 3); +constexpr unsigned kMaintreeTableBits = 11; +constexpr unsigned kLentreeMaxSymbols = kSecondaryElems; +constexpr unsigned kLentreeTableBits = 10; +constexpr unsigned kAligntreeMaxSymbols = kAlignedElems; +constexpr unsigned kAligntreeTableBits = 7; + +// read_lengths() can run past `last` by one repeat (max 51 entries), so every +// length table carries this much slack past its symbol count. +constexpr unsigned kLenTableSafety = 64; + +constexpr unsigned kBlockVerbatim = 1; +constexpr unsigned kBlockAligned = 2; +constexpr unsigned kBlockUncompressed = 3; + +constexpr size_t kCabFrameSize = 32768; + +// Number of position slots per window size (15..21 bits). +constexpr unsigned kPositionSlots[7] = {30, 32, 34, 36, 38, 42, 50}; + +constexpr uint8_t kExtraBits[51] = { + 0, 0, 0, 0, 1, 1, 2, 2, 3, 3, 4, 4, 5, 5, 6, 6, 7, + 7, 8, 8, 9, 9, 10, 10, 11, 11, 12, 12, 13, 13, 14, 14, 15, 15, + 16, 16, 17, 17, 17, 17, 17, 17, 17, 17, 17, 17, 17, 17, 17, 17, 17}; + +constexpr uint32_t kPositionBase[51] = { + 0, 1, 2, 3, 4, 6, 8, 12, 16, 24, + 32, 48, 64, 96, 128, 192, 256, 384, 512, 768, + 1024, 1536, 2048, 3072, 4096, 6144, 8192, 12288, 16384, 24576, + 32768, 49152, 65536, 98304, 131072, 196608, 262144, 393216, 524288, 655360, + 786432, 917504, 1048576, 1179648, 1310720, 1441792, 1572864, 1703936, 1835008, 1966080, + 2097152}; + +// MSB-first bit reader over a byte span, refilled 16 bits at a time in +// little-endian word order. Reads past the end yield zero bits and latch +// `overrun`, so a truncated stream fails cleanly instead of reading OOB. +class BitReader { +public: + explicit BitReader(std::span in) : in_(in) {} + + // n must be <= 17: the buffer holds 32 bits and refills in 16-bit units, so + // a larger request could need three injections and shift negatively. The + // two wider fields in LZX (the 24-bit block length and the 32-bit file size) + // are read as two calls by the caller. + uint32_t read(unsigned n) { + if (n == 0) return 0; + ensure(n); + uint32_t v = peek(n); + remove(n); + return v; + } + + void ensure(unsigned n) { + while (left_ < static_cast(n)) { + uint32_t lo = byte_(); + uint32_t hi = byte_(); + buf_ |= ((hi << 8) | lo) << (32 - 16 - left_); + left_ += 16; + } + } + + uint32_t peek(unsigned n) const { return buf_ >> (32 - n); } + void remove(unsigned n) { + buf_ <<= n; + left_ -= static_cast(n); + } + uint32_t raw() const { return buf_; } + int left() const { return left_; } + + // Drop the partially-consumed word and realign to the byte stream. Used + // around an uncompressed block, whose payload is read as plain bytes. + void align() { + buf_ = 0; + left_ = 0; + } + + size_t pos() const { return pos_; } + void set_pos(size_t p) { pos_ = p; } + size_t size() const { return in_.size(); } + // The 16-bit lookahead legitimately reads a few bytes past the last symbol + // of a well-formed stream; anything beyond that is a truncated stream. + bool overrun_past(size_t slack) const { return pos_ > in_.size() + slack; } + + // Read `n` plain bytes at the current position (used for uncompressed + // blocks and the stored R0/R1/R2 triple). False if the stream is short. + bool take_bytes(uint8_t* dst, size_t n) { + if (n > in_.size() - std::min(pos_, in_.size())) { + overrun_ = true; + return false; + } + std::memcpy(dst, in_.data() + pos_, n); + pos_ += n; + return true; + } + +private: + uint32_t byte_() { + if (pos_ >= in_.size()) { + overrun_ = true; + ++pos_; + return 0; + } + return in_[pos_++]; + } + + std::span in_; + size_t pos_ = 0; + uint32_t buf_ = 0; + int left_ = 0; + bool overrun_ = false; +}; + +// Build the two-level Huffman decode table used by LZX: a direct-indexed table +// for codes up to `nbits` long, with longer codes walking a binary tree grafted +// past the direct region. Layout matches the reference decoder so the same +// length arrays drive it. +bool make_decode_table(unsigned nsyms, unsigned nbits, const uint8_t* length, + std::vector& table) { + const size_t tsize = (size_t(1) << nbits) + (size_t(nsyms) << 1); + if (table.size() != tsize) table.assign(tsize, 0); + else std::fill(table.begin(), table.end(), 0); + + uint32_t table_mask = 1u << nbits; + uint32_t bit_mask = table_mask >> 1; + uint32_t pos = 0; + uint32_t next_symbol = bit_mask; + + unsigned bit_num = 1; + for (; bit_num <= nbits; ++bit_num, bit_mask >>= 1) { + for (unsigned sym = 0; sym < nsyms; ++sym) { + if (length[sym] != bit_num) continue; + uint32_t leaf = pos; + pos += bit_mask; + if (pos > table_mask) return false; // codes overflow the table + for (uint32_t k = 0; k < bit_mask; ++k) table[leaf++] = static_cast(sym); + } + } + + if (pos != table_mask) { + for (uint32_t sym = pos; sym < table_mask; ++sym) table[sym] = 0; + + // Codes longer than nbits: extend into the tree area past the table. + uint64_t pos_hi = uint64_t(pos) << 16; + uint64_t mask_hi = uint64_t(table_mask) << 16; + bit_mask = 1u << 15; + + for (; bit_num <= 16; ++bit_num, bit_mask >>= 1) { + for (unsigned sym = 0; sym < nsyms; ++sym) { + if (length[sym] != bit_num) continue; + uint32_t leaf = static_cast(pos_hi >> 16); + for (unsigned fill = 0; fill < bit_num - nbits; ++fill) { + if (leaf >= table.size()) return false; + if (table[leaf] == 0) { + const size_t lo = size_t(next_symbol) << 1; + if (lo + 1 >= table.size()) return false; + table[lo] = 0; + table[lo + 1] = 0; + table[leaf] = static_cast(next_symbol++); + } + leaf = uint32_t(table[leaf]) << 1; + if ((pos_hi >> (15 - fill)) & 1) leaf++; + } + if (leaf >= table.size()) return false; + table[leaf] = static_cast(sym); + pos_hi += bit_mask; + if (pos_hi > mask_hi) return false; + } + } + if (pos_hi != mask_hi) { + // An incomplete code is legal only when no symbol is used at all. + for (unsigned sym = 0; sym < nsyms; ++sym) + if (length[sym] != 0) return false; + } + return true; + } + return true; +} + +// One LZX stream. `out` grows linearly; matches resolve against it directly. +class LzxDecoder { +public: + LzxDecoder(std::span in, unsigned window_bits, size_t cap) + : bits_(in), cap_(cap) { + window_bits_ = window_bits; + main_elements_ = kNumChars + (kPositionSlots[window_bits - 15] << 3); + out.reserve(std::min(cap, size_t(1) << 20)); + } + + static bool window_ok(unsigned window_bits) { return window_bits >= 15 && window_bits <= 21; } + + // Read the one-time stream header (the optional x86 translation size). + bool read_header() { + if (bits_.read(1) == 1) { + uint32_t hi = bits_.read(16); + uint32_t lo = bits_.read(16); + intel_filesize_ = static_cast((hi << 16) | lo); + } + return true; + } + + // Decode until `out` holds at least `target` bytes. + bool decode_until(size_t target) { + while (out.size() < target) { + if (block_remaining_ == 0 && !start_block()) return false; + const size_t want = target - out.size(); + const size_t run = std::min(block_remaining_, want); + const size_t before = out.size(); + if (block_type_ == kBlockUncompressed) { + if (!copy_stored(run)) return false; + } else { + if (!decode_block(run)) return false; + } + const size_t produced = out.size() - before; + // A final match may overrun the requested run, but never its block. + if (produced == 0 || produced > block_remaining_) return false; + block_remaining_ -= static_cast(produced); + if (out.size() > cap_) return false; + if (bits_.overrun_past(8)) return false; + } + return true; + } + + bool intel_started() const { return intel_started_; } + int32_t intel_filesize() const { return intel_filesize_; } + + std::vector out; + +private: + bool start_block() { + if (block_type_ == kBlockUncompressed) { + if (block_length_ & 1) bits_.set_pos(bits_.pos() + 1); // odd-length pad byte + block_type_ = 0; + bits_.align(); + } + block_type_ = bits_.read(3); + const uint32_t hi = bits_.read(16); + const uint32_t lo = bits_.read(8); + block_length_ = (hi << 8) | lo; + block_remaining_ = block_length_; + if (block_length_ == 0) return false; // would not make progress + + if (block_type_ == kBlockAligned) { + for (unsigned i = 0; i < kAlignedElems; ++i) + aligntree_len_[i] = static_cast(bits_.read(3)); + if (!make_decode_table(kAligntreeMaxSymbols, kAligntreeTableBits, aligntree_len_, + aligntree_tab_)) + return false; + } + + if (block_type_ == kBlockVerbatim || block_type_ == kBlockAligned) { + if (!read_lengths(maintree_len_, 0, kNumChars)) return false; + if (!read_lengths(maintree_len_, kNumChars, main_elements_)) return false; + if (!make_decode_table(kMaintreeMaxSymbols, kMaintreeTableBits, maintree_len_, + maintree_tab_)) + return false; + if (maintree_len_[0xE8] != 0) intel_started_ = true; + if (!read_lengths(lentree_len_, 0, kSecondaryElems)) return false; + if (!make_decode_table(kLentreeMaxSymbols, kLentreeTableBits, lentree_len_, + lentree_tab_)) + return false; + return true; + } + if (block_type_ == kBlockUncompressed) { + intel_started_ = true; + // Realign to a byte boundary, then read the stored R0/R1/R2. + bits_.ensure(16); + if (bits_.left() > 16) bits_.set_pos(bits_.pos() - 2); + bits_.align(); + uint8_t r[12]; + if (!bits_.take_bytes(r, sizeof(r))) return false; + R0_ = le32(r); + R1_ = le32(r + 4); + R2_ = le32(r + 8); + return true; + } + return false; // invalid block type + } + + static uint32_t le32(const uint8_t* p) { + return uint32_t(p[0]) | (uint32_t(p[1]) << 8) | (uint32_t(p[2]) << 16) | + (uint32_t(p[3]) << 24); + } + + bool copy_stored(size_t n) { + if (out.size() + n > cap_) return false; + const size_t at = out.size(); + out.resize(at + n); + return bits_.take_bytes(out.data() + at, n); + } + + // Read a Huffman symbol: index the direct table, then walk the grafted tree + // for codes longer than the table's bit width. + bool read_sym(const std::vector& tab, const uint8_t* lens, unsigned nsyms, + unsigned nbits, unsigned max_codeword, unsigned& sym_out) { + if (tab.size() < (size_t(1) << nbits)) return false; + bits_.ensure(max_codeword); + uint32_t i = tab[bits_.peek(nbits)]; + if (i >= nsyms) { + uint32_t j = 1u << (32 - nbits); + for (;;) { + j >>= 1; + if (j == 0) return false; + i <<= 1; + i |= (bits_.raw() & j) ? 1u : 0u; + if (i >= tab.size()) return false; + i = tab[i]; + if (i < nsyms) break; + } + } + if (lens[i] == 0) return false; // unused symbol: the stream is corrupt + bits_.remove(lens[i]); + sym_out = i; + return true; + } + + // Decode the pretree, then the run-length-coded length deltas it describes. + bool read_lengths(uint8_t* lens, unsigned first, unsigned last) { + for (unsigned x = 0; x < kPretreeElems; ++x) + pretree_len_[x] = static_cast(bits_.read(4)); + if (!make_decode_table(kPretreeMaxSymbols, kPretreeTableBits, pretree_len_, pretree_tab_)) + return false; + + unsigned x = first; + while (x < last) { + unsigned z = 0; + if (!read_sym(pretree_tab_, pretree_len_, kPretreeMaxSymbols, kPretreeTableBits, 16, z)) + return false; + if (z == 17) { + unsigned y = bits_.read(4) + 4; + while (y--) lens[x++] = 0; + } else if (z == 18) { + unsigned y = bits_.read(5) + 20; + while (y--) lens[x++] = 0; + } else if (z == 19) { + unsigned y = bits_.read(1) + 4; + unsigned z2 = 0; + if (!read_sym(pretree_tab_, pretree_len_, kPretreeMaxSymbols, kPretreeTableBits, 16, + z2)) + return false; + const uint8_t v = static_cast((lens[x] + 17 - z2) % 17); + while (y--) lens[x++] = v; + } else { + lens[x] = static_cast((lens[x] + 17 - z) % 17); + x++; + } + } + return true; + } + + // Decode at least `run` bytes of a verbatim/aligned block; a trailing match + // may produce a little more (the caller accounts for it). + bool decode_block(size_t run) { + const size_t stop = out.size() + run; + while (out.size() < stop) { + unsigned main_element = 0; + if (!read_sym(maintree_tab_, maintree_len_, kMaintreeMaxSymbols, kMaintreeTableBits, 16, + main_element)) + return false; + if (main_element < kNumChars) { + if (out.size() >= cap_) return false; + out.push_back(static_cast(main_element)); + continue; + } + + main_element -= kNumChars; + size_t match_length = main_element & kNumPrimaryLengths; + if (match_length == kNumPrimaryLengths) { + unsigned footer = 0; + if (!read_sym(lentree_tab_, lentree_len_, kLentreeMaxSymbols, kLentreeTableBits, 16, + footer)) + return false; + match_length += footer; + } + match_length += kMinMatch; + if (match_length > kMaxMatch) return false; + + uint32_t slot = main_element >> 3; + uint32_t match_offset; + if (slot > 2) { + if (slot >= 51) return false; + const unsigned extra = kExtraBits[slot]; + uint32_t verbatim_bits, aligned_bits = 0; + // The LZX description says an aligned symbol is read when the + // extra-bit count exceeds 3; it is actually read when it is 3 + // or more. + if (block_type_ == kBlockAligned && extra >= 3) { + verbatim_bits = bits_.read(extra - 3) << 3; + unsigned a = 0; + if (!read_sym(aligntree_tab_, aligntree_len_, kAligntreeMaxSymbols, + kAligntreeTableBits, 8, a)) + return false; + aligned_bits = a; + } else { + verbatim_bits = bits_.read(extra); + } + match_offset = kPositionBase[slot] + verbatim_bits + aligned_bits - 2; + R2_ = R1_; + R1_ = R0_; + R0_ = match_offset; + } else if (slot == 0) { + match_offset = R0_; + } else if (slot == 1) { + match_offset = R1_; + R1_ = R0_; + R0_ = match_offset; + } else { + match_offset = R2_; + R2_ = R0_; + R0_ = match_offset; + } + + if (match_offset == 0 || match_offset > (1u << window_bits_)) return false; + if (out.size() + match_length > cap_) return false; + if (!emit_match(match_offset, match_length)) return false; + } + return true; + } + + // Copy a match. A reference reaching before the start of the stream lands in + // the never-written part of the LZX window, which the format initializes to + // 0xDC; reproduce that rather than reading out of bounds. + bool emit_match(uint32_t offset, size_t len) { + size_t pos = out.size(); + if (offset > pos) { + const size_t pre = std::min(offset - pos, len); + out.insert(out.end(), pre, 0xDC); + len -= pre; + if (len == 0) return true; + pos = 0; // the wrapped copy resumes at the start of the stream + } else { + pos -= offset; + } + for (size_t k = 0; k < len; ++k) { + if (pos >= out.size()) return false; + out.push_back(out[pos++]); + } + return true; + } + + BitReader bits_; + size_t cap_; + unsigned window_bits_ = 15; + unsigned main_elements_ = kNumChars; + + uint32_t R0_ = 1, R1_ = 1, R2_ = 1; + unsigned block_type_ = 0; + uint32_t block_length_ = 0; + uint32_t block_remaining_ = 0; + bool intel_started_ = false; + int32_t intel_filesize_ = 0; + + uint8_t pretree_len_[kPretreeMaxSymbols + kLenTableSafety] = {}; + uint8_t maintree_len_[kMaintreeMaxSymbols + kLenTableSafety] = {}; + uint8_t lentree_len_[kLentreeMaxSymbols + kLenTableSafety] = {}; + uint8_t aligntree_len_[kAligntreeMaxSymbols + kLenTableSafety] = {}; + std::vector pretree_tab_, maintree_tab_, lentree_tab_, aligntree_tab_; +}; + +// Undo the encoder's x86 CALL translation over one frame of decoded bytes: +// absolute targets within [-curpos, filesize) go back to relative. +void undo_e8(std::vector& data, size_t start, size_t len, size_t abs_start, + int32_t filesize) { + if (len <= 10) return; + size_t i = 0; + int64_t curpos = static_cast(abs_start); + const size_t limit = len - 10; + while (i < limit) { + if (data[start + i] != 0xE8) { + ++i; + ++curpos; + continue; + } + uint8_t* p = data.data() + start + i + 1; + int32_t abs_off = static_cast(uint32_t(p[0]) | (uint32_t(p[1]) << 8) | + (uint32_t(p[2]) << 16) | (uint32_t(p[3]) << 24)); + if (abs_off >= -curpos && abs_off < filesize) { + const int32_t rel = (abs_off >= 0) ? static_cast(abs_off - curpos) + : static_cast(abs_off + filesize); + const uint32_t u = static_cast(rel); + p[0] = uint8_t(u); + p[1] = uint8_t(u >> 8); + p[2] = uint8_t(u >> 16); + p[3] = uint8_t(u >> 24); + } + i += 5; + curpos += 5; + } +} + +// One CE ROM block: u32 window bits, u32 decoded size, 8 reserved bytes, then +// the LZX stream. Appends the decoded bytes to `out`. +bool ce_block(std::span blob, std::vector& out, size_t cap) { + if (blob.size() < 16) return false; + auto rd = [&](size_t o) { + return uint32_t(blob[o]) | (uint32_t(blob[o + 1]) << 8) | (uint32_t(blob[o + 2]) << 16) | + (uint32_t(blob[o + 3]) << 24); + }; + const uint32_t window_bits = rd(0); + const uint32_t want = rd(4); + if (!LzxDecoder::window_ok(window_bits)) return false; + if (want == 0 || want > cap) return false; + + LzxDecoder d(blob.subspan(16), window_bits, want + kMaxMatch); + if (!d.read_header()) return false; + if (!d.decode_until(want)) return false; + d.out.resize(want); + if (d.intel_started()) undo_e8(d.out, 0, d.out.size(), 0, d.intel_filesize()); + out.insert(out.end(), d.out.begin(), d.out.end()); + return true; +} + +uint32_t u24(std::span s, size_t off) { + return uint32_t(s[off]) | (uint32_t(s[off + 1]) << 8) | (uint32_t(s[off + 2]) << 16); +} + +} // namespace + +std::optional> ce_decompress_rom(std::span src, + size_t out_len) { + // Header: a table of 3-byte little-endian values. [0] is the total decoded + // size; [1..n-1] are the end offsets of each compressed block. Block data + // starts right after the table. + constexpr unsigned kBlockBits = 12; // 4 KiB blocks + if (src.size() < 3 || out_len == 0) return std::nullopt; + + const uint32_t total = u24(src, 0); + const size_t num_blocks = total == 0 ? 2 : ((total - 1) >> kBlockBits) + 2; + if (num_blocks < 2 || num_blocks > (size_t(1) << 22)) return std::nullopt; + size_t table_end = num_blocks * 3; + if (table_end > src.size()) return std::nullopt; + + std::vector out; + out.reserve(std::min(out_len, size_t(1) << 20)); + + size_t block_start = table_end; + size_t entry = 3; + size_t remaining = out_len; + for (size_t b = 1; b < num_blocks && remaining > 0; ++b, entry += 3) { + const uint32_t block_end = u24(src, entry); + if (block_end <= block_start || block_end > src.size()) return std::nullopt; + const size_t before = out.size(); + if (!ce_block(src.subspan(block_start, block_end - block_start), out, remaining)) + return std::nullopt; + const size_t produced = out.size() - before; + if (produced > remaining) return std::nullopt; + remaining -= produced; + block_start = block_end; + } + + if (out.empty()) return std::nullopt; + // A blob may declare fewer bytes than the caller's field (a module .data + // section whose tail is zero-filled at load time); pad rather than fail. + out.resize(out_len, 0); + return out; +} + +std::optional> lzx_decompress_cab(std::span src, size_t out_len, + unsigned window_bits) { + if (!LzxDecoder::window_ok(window_bits) || out_len == 0) return std::nullopt; + + LzxDecoder d(src, window_bits, out_len + kMaxMatch); + if (!d.read_header()) return std::nullopt; + + // Output is produced in 32 KiB frames; the x86 translation is per frame, + // using the frame's absolute position, and stops after 32768 frames. + size_t done = 0; + for (size_t frame = 0; done < out_len; ++frame) { + const size_t want = std::min(kCabFrameSize, out_len - done); + if (!d.decode_until(done + want)) return std::nullopt; + if (d.intel_started() && d.intel_filesize() != 0 && frame < 32768) + undo_e8(d.out, done, want, done, d.intel_filesize()); + done += want; + } + d.out.resize(out_len); + return std::move(d.out); +} + +} // namespace ft diff --git a/src/extract/lzx.hpp b/src/extract/lzx.hpp new file mode 100644 index 0000000..49f7d67 --- /dev/null +++ b/src/extract/lzx.hpp @@ -0,0 +1,46 @@ +// lzx.hpp — LZX decompression, in the two framings firmware actually ships. +// +// LZX is one bitstream format with two different envelopes around it: +// +// * Windows CE XIP ROM ("CECompress"/CEDecompressROM): the payload is split +// into independent 4 KiB blocks, each a self-contained LZX stream prefixed +// by its own window size and decoded length, indexed by a table of 3-byte +// end offsets. Used for compressed ROM files and module sections in a .cos +// / nk.bin image. +// * MS-CAB (CFFOLDER typeCompress 3): one continuous bitstream for the whole +// folder, output in 32 KiB frames, with the x86 E8 translation applied per +// frame rather than per stream. +// +// Both share the block/Huffman/match decoder below. Decoding is linear (matches +// resolve against the bytes already produced) rather than through a circular +// window: the semantics are identical because an LZX match offset never exceeds +// the window size, and it keeps the output a plain buffer. +// +// Every entry point is bounded: output can never exceed the caller's cap, a +// malformed stream yields nullopt rather than a partial lie. +#pragma once + +#include +#include +#include +#include + +namespace ft { + +// Decode a Windows CE ROM CECompress blob (the 3-byte block-offset table plus +// per-block LZX streams) to exactly `out_len` bytes. +// +// A blob's own header declares the decoded length, which for a module .data +// section can be SHORTER than the section's virtual size (the remainder is zero +// at load time). That is a valid encoding, so a short decode is padded with +// zeroes to out_len rather than rejected. Returns nullopt if the block table is +// inconsistent, a block fails to decode, or the stream would exceed out_len. +std::optional> ce_decompress_rom(std::span src, size_t out_len); + +// Decode one MS-CAB folder's LZX bitstream (all CFDATA payloads concatenated) +// to exactly `out_len` bytes. `window_bits` is 15..21, taken from the high byte +// of CFFOLDER.typeCompress. Returns nullopt on a malformed stream. +std::optional> lzx_decompress_cab(std::span src, size_t out_len, + unsigned window_bits); + +} // namespace ft diff --git a/src/extract/manifest.cpp b/src/extract/manifest.cpp index 91d38be..66b28bb 100644 --- a/src/extract/manifest.cpp +++ b/src/extract/manifest.cpp @@ -7,6 +7,7 @@ #include "extract/android_sparse.hpp" #include "extract/compressed.hpp" #include "extract/descramble.hpp" +#include "extract/cab.hpp" #include "extract/cpio.hpp" #include "extract/cramfs.hpp" #include "extract/erofs.hpp" @@ -36,6 +37,8 @@ #include "extract/upx.hpp" #include "extract/vbf.hpp" #include "extract/vbmeta.hpp" +#include "extract/wince_hive.hpp" +#include "extract/wince_rom.hpp" #include "extract/fit.hpp" namespace ft { @@ -43,6 +46,7 @@ namespace ft { Extractor find_extractor(const std::string& type) { if (is_scheme(type)) return extract_descramble; // vendor-encrypted -> descramble + recurse if (type == "squashfs") return extract_squashfs; + if (type == "cab") return extract_cab; if (type == "cpio") return extract_cpio; if (type == "ext") return extract_ext; if (type == "jffs2") return extract_jffs2; @@ -64,6 +68,8 @@ Extractor find_extractor(const std::string& type) { if (type == "uboot_env") return extract_uboot_env; if (type == "esp32_nvs") return extract_esp32_nvs; if (type == "rae_rfp") return extract_rae_rfp; + if (type == "wince_rom") return extract_wince_rom; + if (type == "wince_hive") return extract_wince_hive; if (type == "vbf") return extract_vbf; if (type == "vbmeta") return extract_vbmeta; if (type == "upx") return extract_upx; diff --git a/src/extract/wince_hive.cpp b/src/extract/wince_hive.cpp new file mode 100644 index 0000000..943c2e5 --- /dev/null +++ b/src/extract/wince_hive.cpp @@ -0,0 +1,83 @@ +// wince_hive.cpp — Windows CE registry hive value dump. See the header. +#include "extract/wince_hive.hpp" + +#include +#include + +#include "extract/safepath.hpp" +#include "wince_hive_parse.hpp" + +namespace ft { + +namespace { +constexpr size_t kMaxValues = 500000; // a bound on adversarial input, not on real hives + +// Keep a value on one line: control characters would break the dump's shape. +std::string one_line(const std::string& s) { + std::string o; + o.reserve(s.size()); + for (char c : s) { + const unsigned char u = static_cast(c); + if (u == '\t') o += "\\t"; + else if (u == '\r') o += "\\r"; + else if (u == '\n') o += "\\n"; + else if (u < 0x20 || u == 0x7f) o += '.'; + else o += c; + } + return o; +} +} // namespace + +bool extract_wince_hive(const Reader& r, const Finding& f, SafeRoot& root, + const std::string& subdir, Extracted& out) { + out.offset = f.offset; + out.type = "wince_hive"; + out.root = subdir; + + std::vector values; + ce_hive_values(r, f.offset, values, kMaxValues); + if (values.empty()) { + out.status = "error:no-values"; + return true; + } + if (!root.make_dir(subdir)) { + out.status = "error:mkdir"; + return true; + } + + std::string text = + "# Windows CE registry values recovered from the hive at offset " + + std::to_string(f.offset) + ".\n" + "# Recovered record by record: CE's cell layout is not walked, so these\n" + "# values carry no key path. Columns: hive offset, name, type, value.\n"; + char off[24]; + for (const CeHiveValue& v : values) { + std::snprintf(off, sizeof(off), "0x%08zx", v.offset); + text += off; + text += " "; + text += v.name; + text.append(v.name.size() < 40 ? 40 - v.name.size() : 1, ' '); + const char* tn = ce_hive_type_name(v.type); + text += tn; + const size_t tl = std::char_traits::length(tn); + text.append(tl < 15 ? 15 - tl : 1, ' '); + text += one_line(v.rendered); + text += '\n'; + } + + const std::vector bytes(text.begin(), text.end()); + if (!root.write_file(subdir + "/registry-values.txt", bytes, 0644)) { + out.status = "error:write"; + return true; + } + out.files = 1; + out.bytes = bytes.size(); + out.consumed = r.size() - f.offset; + out.status = values.size() >= kMaxValues ? "partial" : "ok"; + if (values.size() >= kMaxValues) + out.warnings.push_back("value recovery stopped at the " + std::to_string(kMaxValues) + + "-record cap"); + return true; +} + +} // namespace ft diff --git a/src/extract/wince_hive.hpp b/src/extract/wince_hive.hpp new file mode 100644 index 0000000..eb9be39 --- /dev/null +++ b/src/extract/wince_hive.hpp @@ -0,0 +1,24 @@ +// wince_hive.hpp — Windows CE registry hive value dump. +// +// The hive itself is already a file; what extraction adds is readability. CE's +// registry is where a device keeps its service configuration, its account and +// key material, and its certificates, all encoded as UTF-16 cell records that +// no text search will find. This writes every recoverable value out as one line +// of text — offset, name, type, value — so the rest of the toolchain (and a +// person with grep) can read it. +#pragma once + +#include + +#include "extract/manifest.hpp" +#include "finding.hpp" +#include "reader.hpp" + +namespace ft { + +class SafeRoot; + +bool extract_wince_hive(const Reader& r, const Finding& f, SafeRoot& root, + const std::string& subdir, Extracted& out); + +} // namespace ft diff --git a/src/extract/wince_rom.cpp b/src/extract/wince_rom.cpp new file mode 100644 index 0000000..b1444f1 --- /dev/null +++ b/src/extract/wince_rom.cpp @@ -0,0 +1,384 @@ +// wince_rom.cpp — Windows CE XIP ROM extraction. See wince_rom.hpp. +// +// ROM files are simple: copy the stored bytes, running them through the +// CECompress decoder when the stored size differs from the real size. +// +// Modules are the interesting half. romimage does not keep a module as a PE +// file; it keeps an E32ROMHDR (the optional-header fields that matter), an +// O32ROMHDR array (one per section: virtual size, RVA, stored size, and a +// pointer to the bytes somewhere else in the image), and the section bytes, +// each optionally CECompress-coded. Rebuilding those into a flat PE — a DOS +// stub, a COFF header, an optional header carrying the original image base and +// entry point, a section table, and every section placed so its file offset +// equals its RVA — turns 250-odd kernel modules back into files a disassembler +// opens without a loader script. +#include "extract/wince_rom.hpp" + +#include +#include +#include +#include +#include + +#include "extract/lzx.hpp" +#include "extract/safepath.hpp" +#include "wince_rom_parse.hpp" + +namespace ft { + +namespace { + +constexpr uint32_t kScnCompressed = 0x00002000; // section bytes are CECompress-coded +constexpr uint32_t kScnCode = 0x00000020; // IMAGE_SCN_CNT_CODE +constexpr uint32_t kScnWrite = 0x80000000; // IMAGE_SCN_MEM_WRITE +constexpr uint32_t kAlign = 0x1000; +constexpr size_t kMaxSections = 96; +constexpr uint64_t kMaxModule = uint64_t(256) << 20; +constexpr uint64_t kMaxFile = uint64_t(512) << 20; + +// Data-directory slots carried in the E32 ROM header, in its own order. +enum { EXP = 0, IMP, RES, EXC, SEC }; + +struct Obj { + uint32_t vsize, rva, psize, dataptr, realaddr, flags; +}; + +void put16(std::vector& b, size_t off, uint16_t v) { + b[off] = uint8_t(v); + b[off + 1] = uint8_t(v >> 8); +} + +void put32(std::vector& b, size_t off, uint32_t v) { + b[off] = uint8_t(v); + b[off + 1] = uint8_t(v >> 8); + b[off + 2] = uint8_t(v >> 16); + b[off + 3] = uint8_t(v >> 24); +} + +// Flatten a ROM path to one safe filename ("\\windows\\nk.exe" -> "nk.exe"). +std::string flat_name(const std::string& name) { + std::string s = name; + for (char& c : s) + if (c == '\\') c = '/'; + const size_t slash = s.find_last_of('/'); + if (slash != std::string::npos) s = s.substr(slash + 1); + std::string o; + for (char c : s) { + const unsigned char u = static_cast(c); + o += (u >= 0x20 && u < 0x7f && c != '/' && c != '"' && c != '*' && c != ':') ? c : '_'; + } + if (o.empty() || o == "." || o == "..") o = "_unnamed"; + return o; +} + +// ROM sections carry flags but no names. Recover the conventional ones from the +// data directories and the section flags so the rebuilt PE reads naturally. +std::string section_name(const Obj& o, const uint32_t dir_rva[6], + const uint32_t dir_size[6], std::unordered_map& used) { + std::string n; + if (dir_size[RES] && o.rva == dir_rva[RES]) n = ".rsrc"; + else if (dir_size[EXC] && o.rva == dir_rva[EXC]) n = ".pdata"; + else if (o.flags & kScnCode) n = ".text"; + else if (o.psize == 0) n = ".bss"; + else if (o.flags & kScnWrite) n = ".data"; + else n = ".rdata"; + if (used.count(n)) { + const std::string stem = n.substr(0, 6); + for (int i = 2;; ++i) { + std::string cand = stem + std::to_string(i); + if (!used.count(cand)) { + n = cand; + break; + } + } + } + used[n] = 1; + return n; +} + +// Rebuild one TOC module into a flat PE. `failed` counts sections whose +// CECompress payload would not decode; those are stored raw so nothing is lost. +bool build_pe(const Reader& r, size_t base, const CeRomHeader& h, const CeModule& m, + std::vector& pe, size_t& failed) { + const size_t e = ce_rom_offset(r, base, h, m.e32, 0x54); + if (e == SIZE_MAX) return false; + auto u16 = [&](size_t o) { return r.at(e + o, Endian::Little).value_or(0); }; + auto u32 = [&](size_t o) { return r.at(e + o, Endian::Little).value_or(0); }; + + const uint16_t objcnt = u16(0x00); + const uint16_t imgflags = u16(0x02); + const uint32_t entryrva = u32(0x04); + const uint32_t vbase = u32(0x08); + const uint32_t stackmax = u32(0x10); + const uint32_t vsize = u32(0x14); + const uint32_t timestamp = u32(0x20); + if (objcnt == 0 || objcnt > kMaxSections) return false; + + uint32_t dir_rva[6], dir_size[6]; + for (int i = 0; i < 6; ++i) { + dir_rva[i] = u32(0x24 + 8 * i); + dir_size[i] = u32(0x28 + 8 * i); + } + + const size_t ob = ce_rom_offset(r, base, h, m.o32, size_t(objcnt) * 24); + if (ob == SIZE_MAX) return false; + std::vector objs(objcnt); + for (uint16_t i = 0; i < objcnt; ++i) { + const size_t p = ob + size_t(i) * 24; + auto v = [&](size_t o) { return r.at(p + o, Endian::Little).value_or(0); }; + objs[i] = {v(0), v(4), v(8), v(12), v(16), v(20)}; + } + + // Resolve each section's real bytes before laying the file out. + std::vector> blobs(objcnt); + for (uint16_t i = 0; i < objcnt; ++i) { + const Obj& o = objs[i]; + const size_t n = std::min(o.vsize, o.psize); + if (n == 0) continue; + const size_t src = ce_rom_offset(r, base, h, o.dataptr, n); + if (src == SIZE_MAX) { + ++failed; + continue; + } + auto raw = r.bytes(src, n); + if (!raw) { + ++failed; + continue; + } + if (o.flags & kScnCompressed) { + auto dec = ce_decompress_rom(*raw, o.vsize); + if (dec) { + blobs[i] = std::move(*dec); + continue; + } + ++failed; // keep the compressed bytes rather than dropping the section + } + blobs[i].assign(raw->begin(), raw->end()); + } + + uint64_t end = kAlign; // headers occupy the first page + for (uint16_t i = 0; i < objcnt; ++i) { + if (blobs[i].empty()) continue; + const uint64_t e_i = + uint64_t(objs[i].rva) + (uint64_t(blobs[i].size()) + kAlign - 1) / kAlign * kAlign; + end = std::max(end, e_i); + } + if (end > kMaxModule) return false; + pe.assign(static_cast(end), 0); + + // DOS header: just enough to be a PE (signature + e_lfanew). + pe[0] = 'M'; + pe[1] = 'Z'; + put16(pe, 2, 0x90); + constexpr size_t kNt = 0x80; + put32(pe, 0x3c, kNt); + + // COFF header. The machine id comes from the ROM's own CPU field so a MIPS + // or SH ROM does not get mislabelled as ARM. + std::memcpy(pe.data() + kNt, "PE\0\0", 4); + put16(pe, kNt + 4, h.cpu ? h.cpu : 0x01c2); + put16(pe, kNt + 6, objcnt); + put32(pe, kNt + 8, timestamp); + put32(pe, kNt + 12, 0); // PointerToSymbolTable + put32(pe, kNt + 16, 0); // NumberOfSymbols + put16(pe, kNt + 20, 0xe0); + put16(pe, kNt + 22, imgflags); + + uint32_t code = 0, data = 0, text_rva = objs[0].rva; + bool have_text = false; + for (const Obj& o : objs) { + if (o.flags & kScnCode) { + code += o.psize; + if (!have_text) { + text_rva = o.rva; + have_text = true; + } + } else { + data += o.psize; + } + } + + const size_t opt = kNt + 24; + put16(pe, opt + 0, 0x10b); // PE32 + pe[opt + 2] = 0; // linker version + pe[opt + 3] = 0; + put32(pe, opt + 4, code); + put32(pe, opt + 8, data); + put32(pe, opt + 12, 0); + put32(pe, opt + 16, entryrva); + put32(pe, opt + 20, text_rva); + put32(pe, opt + 24, 0); + put32(pe, opt + 28, vbase); + put32(pe, opt + 32, kAlign); // SectionAlignment + put32(pe, opt + 36, kAlign); // FileAlignment == SectionAlignment: offset == RVA + put16(pe, opt + 40, 4); // OS version + put16(pe, opt + 42, 0); + put16(pe, opt + 44, 0); // image version + put16(pe, opt + 46, 0); + put16(pe, opt + 48, 4); // subsystem version + put16(pe, opt + 50, 0); + put32(pe, opt + 52, 0); + put32(pe, opt + 56, vsize); // SizeOfImage + put32(pe, opt + 60, kAlign); // SizeOfHeaders + put32(pe, opt + 64, 0); // CheckSum + put16(pe, opt + 68, 9); // IMAGE_SUBSYSTEM_WINDOWS_CE_GUI + put16(pe, opt + 70, 0); + put32(pe, opt + 72, stackmax); + put32(pe, opt + 76, 0x1000); + put32(pe, opt + 80, 0x10000); + put32(pe, opt + 84, 0x1000); + put32(pe, opt + 88, 0); + put32(pe, opt + 92, 16); // NumberOfRvaAndSizes + + const size_t dd = opt + 96; + const int order[5] = {EXP, IMP, RES, EXC, SEC}; + for (int i = 0; i < 16; ++i) { + const uint32_t rva = i < 5 ? dir_rva[order[i]] : 0; + const uint32_t sz = i < 5 ? dir_size[order[i]] : 0; + put32(pe, dd + i * 8, rva); + put32(pe, dd + i * 8 + 4, sz); + } + + const size_t st = opt + 0xe0; + if (st + size_t(objcnt) * 40 > kAlign) return false; // headers must fit one page + std::unordered_map used; + for (uint16_t i = 0; i < objcnt; ++i) { + const Obj& o = objs[i]; + const std::string nm = section_name(o, dir_rva, dir_size, used); + const size_t se = st + size_t(i) * 40; + std::memcpy(pe.data() + se, nm.data(), std::min(nm.size(), 8)); + put32(pe, se + 8, o.vsize); + put32(pe, se + 12, o.rva); + const uint32_t raw = static_cast((blobs[i].size() + kAlign - 1) / kAlign * kAlign); + put32(pe, se + 16, raw); + put32(pe, se + 20, blobs[i].empty() ? 0 : o.rva); + put32(pe, se + 24, 0); + put32(pe, se + 28, 0); + put16(pe, se + 32, 0); + put16(pe, se + 34, 0); + put32(pe, se + 36, o.flags & ~kScnCompressed); + if (!blobs[i].empty()) { + if (uint64_t(o.rva) + blobs[i].size() > pe.size()) return false; + std::memcpy(pe.data() + o.rva, blobs[i].data(), blobs[i].size()); + } + } + return true; +} + +// Give a repeated ROM name a distinct filename rather than overwriting. +std::string unique_name(std::unordered_map& used, const std::string& name) { + int& n = used[name]; + if (n++ == 0) return name; + const size_t dot = name.find_last_of('.'); + const std::string stem = dot == std::string::npos ? name : name.substr(0, dot); + const std::string ext = dot == std::string::npos ? std::string() : name.substr(dot); + return stem + "_" + std::to_string(n) + ext; +} + +} // namespace + +bool extract_wince_rom(const Reader& r, const Finding& f, SafeRoot& root, const std::string& subdir, + Extracted& out) { + out.offset = f.offset; + out.type = "wince_rom"; + out.root = subdir; + + CeRomHeader h; + if (!ce_rom_header(r, f.offset, h)) { + out.status = "error:no-romhdr"; + return true; + } + if (!root.make_dir(subdir)) { + out.status = "error:mkdir"; + return true; + } + + std::vector modules; + std::vector files; + ce_rom_modules(r, f.offset, h, modules); + ce_rom_files(r, f.offset, h, files); + + size_t mod_fail = 0, sect_fail = 0, file_fail = 0; + + if (!modules.empty() && root.make_dir(subdir + "/modules")) out.dirs++; + if (!files.empty() && root.make_dir(subdir + "/files")) out.dirs++; + + std::unordered_map used_mod, used_file; + for (const CeModule& m : modules) { + std::vector pe; + size_t failed = 0; + if (!build_pe(r, f.offset, h, m, pe, failed)) { + ++mod_fail; + out.warnings.push_back(m.name + ": module headers unreadable, skipped"); + continue; + } + sect_fail += failed; + if (failed) + out.warnings.push_back(m.name + ": " + std::to_string(failed) + + " section(s) failed to decompress, stored raw"); + const std::string rel = subdir + "/modules/" + unique_name(used_mod, flat_name(m.name)); + if (!root.write_file(rel, pe, 0644)) { + out.status = "error:write"; + return true; + } + out.files++; + out.bytes += pe.size(); + } + + for (const CeFile& fe : files) { + if (fe.comp == 0 || fe.real > kMaxFile) { + ++file_fail; + continue; + } + const size_t src = ce_rom_offset(r, f.offset, h, fe.load, fe.comp); + if (src == SIZE_MAX) { + ++file_fail; + out.warnings.push_back(fe.name + ": data outside the image"); + continue; + } + auto raw = r.bytes(src, fe.comp); + if (!raw) { + ++file_fail; + continue; + } + + std::string name = flat_name(fe.name); + std::vector data; + if (fe.comp != fe.real) { + auto dec = ce_decompress_rom(*raw, fe.real); + if (dec) { + data = std::move(*dec); + } else { + // Keep the stored bytes under a marked name so nothing is lost. + data.assign(raw->begin(), raw->end()); + name += ".cecompressed"; + ++file_fail; + out.warnings.push_back(fe.name + ": cecompress decode failed, stored raw"); + } + } else { + data.assign(raw->begin(), raw->end()); + } + + const std::string rel = subdir + "/files/" + unique_name(used_file, name); + if (!root.write_file(rel, data, 0644)) { + out.status = "error:write"; + return true; + } + out.files++; + out.bytes += data.size(); + } + + const uint64_t span = h.span(); + const size_t avail = r.size() - f.offset; + out.consumed = span <= avail ? span : avail; + + if (out.files == 0) + out.status = "error:empty"; + else if (mod_fail || sect_fail || file_fail) + out.status = "partial"; + else + out.status = "ok"; + return true; +} + +} // namespace ft diff --git a/src/extract/wince_rom.hpp b/src/extract/wince_rom.hpp new file mode 100644 index 0000000..ccba4bc --- /dev/null +++ b/src/extract/wince_rom.hpp @@ -0,0 +1,30 @@ +// wince_rom.hpp — Windows CE XIP ROM extractor. +// +// Produces two trees under the output subdir: +// +// modules/ one file per TOCentry. XIP modules are not stored as PE files — +// romimage splits each into an E32 ROM header, an O32 section +// table, and the section bytes scattered through the image — so +// each is rebuilt into a flat PE (file offset == RVA) that a +// disassembler or a PE parser can open directly. +// files/ one file per FILESentry, CECompress-decoded where needed. +// +// Everything here is what makes a .cos worth opening: the registry hives, the +// .CAB installers, the certificates and the vendor DLLs all live in files/, +// and moria's normal recursion then descends into them. +#pragma once + +#include + +#include "extract/manifest.hpp" +#include "finding.hpp" +#include "reader.hpp" + +namespace ft { + +class SafeRoot; + +bool extract_wince_rom(const Reader& r, const Finding& f, SafeRoot& root, const std::string& subdir, + Extracted& out); + +} // namespace ft diff --git a/src/validators/cab.cpp b/src/validators/cab.cpp new file mode 100644 index 0000000..6d69df9 --- /dev/null +++ b/src/validators/cab.cpp @@ -0,0 +1,57 @@ +// cab.cpp — Microsoft Cabinet validator. See the header. +#include "validators/cab.hpp" + +#include +#include +#include +#include + +#include "cab_parse.hpp" + +namespace ft { + +bool validate_cab(ValidatorCtx& ctx) { + const Reader& r = ctx.reader; + const size_t base = ctx.offset; + + CabHeader h; + if (!cab_header(r, base, h)) return false; + + std::vector folders; + if (!cab_folders(r, base, h, folders) || folders.empty()) return false; + std::vector files; + if (!cab_files(r, base, h, files) || files.empty()) return false; + + const size_t avail = r.size() - base; + if (h.cb_cabinet == 0 || h.cb_cabinet > avail + 0x1000) return false; + ctx.out.size = h.cb_cabinet <= avail ? h.cb_cabinet : avail; + + // Name the codec(s) in use. A cabinet may mix them per folder, though in + // practice one tool writes the whole set the same way. + std::set codecs; + for (const CabFolder& f : folders) codecs.insert(cab_comp_name(f.comp())); + std::string comp; + for (const std::string& c : codecs) comp += (comp.empty() ? "" : "+") + c; + ctx.out.compression = comp; + + char ver[16]; + std::snprintf(ver, sizeof(ver), "%u.%u", h.ver_major, h.ver_minor); + ctx.out.version = ver; + + char label[80]; + std::snprintf(label, sizeof(label), "%u folders, %u files", h.nfolders, h.nfiles); + ctx.out.label = label; + + if (h.flags & (kCabPrevCabinet | kCabNextCabinet)) + ctx.out.diagnostics.push_back({"warning", "cab-spanned", + "part of a multi-cabinet set; files continued from or into " + "another cabinet cannot be completed from this file alone"}); + + if (h.cb_cabinet == avail) + ctx.out.set_confidence(Confidence::Verified, "folder + file tables parse, cbCabinet == EOF"); + else + ctx.out.set_confidence(Confidence::Consistent, "folder + file tables parse"); + return true; +} + +} // namespace ft diff --git a/src/validators/cab.hpp b/src/validators/cab.hpp new file mode 100644 index 0000000..2582797 --- /dev/null +++ b/src/validators/cab.hpp @@ -0,0 +1,10 @@ +#pragma once +#include "signature.hpp" + +namespace ft { +// Validate a Microsoft Cabinet: walk the folder and file tables (honouring the +// optional per-structure reserve widths), size the finding by cbCabinet, and +// report the folder compression codec. A cabinet whose declared size matches +// the bytes available is `verified`. +bool validate_cab(ValidatorCtx& ctx); +} // namespace ft diff --git a/src/validators/registry.cpp b/src/validators/registry.cpp index bb94c74..296071c 100644 --- a/src/validators/registry.cpp +++ b/src/validators/registry.cpp @@ -1,6 +1,7 @@ #include "validators/registry.hpp" #include "validators/android_boot.hpp" +#include "validators/cab.hpp" #include "validators/cpio.hpp" #include "validators/deobf.hpp" #include "validators/dtb.hpp" @@ -27,6 +28,8 @@ #include "validators/vbf.hpp" #include "validators/vbmeta.hpp" #include "validators/verity.hpp" +#include "validators/wince_hive.hpp" +#include "validators/wince_rom.hpp" #include "validators/yaffs2.hpp" #include "validators/zip.hpp" @@ -38,6 +41,7 @@ Validator find_validator(const std::string& name) { if (name == "squashfs") return validate_squashfs; if (name == "uimage") return validate_uimage; if (name == "elf") return validate_elf; + if (name == "cab") return validate_cab; if (name == "cpio") return validate_cpio; if (name == "dtb") return validate_dtb; if (name == "ihex") return validate_ihex; @@ -47,6 +51,8 @@ Validator find_validator(const std::string& name) { if (name == "zip") return validate_zip; if (name == "jpeg") return validate_jpeg; if (name == "rae_rfp") return validate_rae_rfp; + if (name == "wince_rom") return validate_wince_rom; + if (name == "wince_hive") return validate_wince_hive; if (name == "vbf") return validate_vbf; if (name == "upx") return validate_upx; if (name == "verity") return validate_verity; diff --git a/src/validators/wince_hive.cpp b/src/validators/wince_hive.cpp new file mode 100644 index 0000000..1dbdac5 --- /dev/null +++ b/src/validators/wince_hive.cpp @@ -0,0 +1,34 @@ +// wince_hive.cpp — Windows CE registry hive validator. See the header. +#include "validators/wince_hive.hpp" + +#include +#include + +#include "wince_hive_parse.hpp" + +namespace ft { + +namespace { +// Enough recovered records that a coincidental header cannot have produced +// them, and a bound on how many the validator bothers to count. +constexpr size_t kMinValues = 8; +constexpr size_t kCountCap = 4096; +} // namespace + +bool validate_wince_hive(ValidatorCtx& ctx) { + std::vector values; + const size_t n = ce_hive_values(ctx.reader, ctx.offset, values, kCountCap); + if (n < kMinValues) return false; + + ctx.out.size = ctx.reader.size() - ctx.offset; + + char buf[64]; + std::snprintf(buf, sizeof(buf), "%zu%s values", n, n >= kCountCap ? "+" : ""); + ctx.out.label = buf; + std::snprintf(buf, sizeof(buf), "%zu%s registry value record(s) recovered", n, + n >= kCountCap ? "+" : ""); + ctx.out.set_confidence(Confidence::Consistent, buf); + return true; +} + +} // namespace ft diff --git a/src/validators/wince_hive.hpp b/src/validators/wince_hive.hpp new file mode 100644 index 0000000..c2df715 --- /dev/null +++ b/src/validators/wince_hive.hpp @@ -0,0 +1,10 @@ +#pragma once +#include "signature.hpp" + +namespace ft { +// Validate a Windows CE registry hive: confirm the header shape around the +// "EKIM" signature and require that real value records are recoverable from the +// cell data. The record recovery is what separates a hive from four coincidental +// bytes, so it decides the tier. +bool validate_wince_hive(ValidatorCtx& ctx); +} // namespace ft diff --git a/src/validators/wince_rom.cpp b/src/validators/wince_rom.cpp new file mode 100644 index 0000000..a11593e --- /dev/null +++ b/src/validators/wince_rom.cpp @@ -0,0 +1,66 @@ +// wince_rom.cpp — Windows CE XIP ROM validator. See the header. +// +// "ECEC" at offset 0x40 is only four bytes and is copied verbatim into images +// that are not ROMs at all (the Crestron eboot carries it with a pTOC pointing +// tens of megabytes past its own end). The real evidence is that pTOC resolves +// to a ROMHDR whose first field is the image's own virtual base and whose +// module + file tables fit inside the file; that resolution is what separates a +// ROM from a stray signature, so it decides the tier here. +#include "validators/wince_rom.hpp" + +#include + +#include "wince_rom_parse.hpp" + +namespace ft { + +bool validate_wince_rom(ValidatorCtx& ctx) { + const Reader& r = ctx.reader; + const size_t base = ctx.offset; + + CeRomHeader h; + if (!ce_rom_header(r, base, h)) { + // No ROM behind the signature. At the start of a file that is worth + // saying (a bootloader built from the same sources carries the header + // verbatim); four bytes deep inside some other image it is just noise, + // so drop it rather than report a container that isn't there. + if (base != 0) return false; + ctx.out.size = 0; + ctx.out.set_confidence(Confidence::Magic, + "ECEC signature; pTOC does not resolve to a ROMHDR"); + return true; + } + + const size_t avail = r.size() - base; + const uint64_t span = h.span(); + ctx.out.size = static_cast(span <= avail ? span : avail); + + if (const char* a = ce_cpu_arch(h.cpu); *a) ctx.out.arch = a; + + // Any file stored shorter than its real size is CECompress-coded; the same + // codec covers compressed module sections. + std::vector files; + bool compressed = false; + if (ce_rom_files(r, base, h, files)) + for (const CeFile& f : files) + if (f.comp != f.real) { + compressed = true; + break; + } + ctx.out.compression = compressed ? "cecompress" : "none"; + + char ev[160]; + std::snprintf(ev, sizeof(ev), "ROMHDR at 0x%zx: base 0x%08x, %u modules, %u files", h.hdr_off, + h.physfirst, h.nummods, h.numfiles); + // The tables resolving inside the image is a decode-grade check: a stray + // signature cannot produce a self-consistent TOC. Short of the full span + // being present, keep it one tier lower. + ctx.out.set_confidence(span <= avail ? Confidence::Verified : Confidence::Consistent, ev); + + char label[64]; + std::snprintf(label, sizeof(label), "%u modules, %u files", h.nummods, h.numfiles); + ctx.out.label = label; + return true; +} + +} // namespace ft diff --git a/src/validators/wince_rom.hpp b/src/validators/wince_rom.hpp new file mode 100644 index 0000000..e2480ab --- /dev/null +++ b/src/validators/wince_rom.hpp @@ -0,0 +1,11 @@ +#pragma once +#include "signature.hpp" + +namespace ft { +// Validate a Windows CE XIP ROM image: resolve pTOC to a ROMHDR whose virtual +// base, physical span, and module/file table sizes agree with the file, then +// size the finding by the image span and label it with the module/file counts. +// A bootloader that carries a stale "ECEC" signature but no resolvable ROMHDR +// stays at `magic` tier so it is identified without driving an extraction. +bool validate_wince_rom(ValidatorCtx& ctx); +} // namespace ft diff --git a/src/wince_hive_parse.cpp b/src/wince_hive_parse.cpp new file mode 100644 index 0000000..2c4cc2b --- /dev/null +++ b/src/wince_hive_parse.cpp @@ -0,0 +1,181 @@ +// wince_hive_parse.cpp — CE hive value recovery. See the header. +#include "wince_hive_parse.hpp" + +#include + +namespace ft { + +namespace { + +constexpr size_t kScanStart = 0x40; // past the signature + header GUIDs +constexpr uint16_t kMaxNameChars = 64; +constexpr uint16_t kMaxDataLen = 0x4000; + +// The registry types CE stores in a hive. +constexpr uint16_t kRegSz = 1; +constexpr uint16_t kRegExpandSz = 2; +constexpr uint16_t kRegBinary = 3; +constexpr uint16_t kRegDword = 4; +constexpr uint16_t kRegDwordBe = 5; +constexpr uint16_t kRegMultiSz = 7; +constexpr uint16_t kRegQword = 11; + +bool known_type(uint16_t t) { + return t == kRegSz || t == kRegExpandSz || t == kRegBinary || t == kRegDword || + t == kRegDwordBe || t == kRegMultiSz || t == kRegQword; +} + +// The characters a real registry value name is made of. Anything outside this +// is how a coincidental length triple gets rejected. +bool name_char(char32_t c) { + if (c >= 'A' && c <= 'Z') return true; + if (c >= 'a' && c <= 'z') return true; + if (c >= '0' && c <= '9') return true; + switch (c) { + case '_': case '-': case '.': case ' ': case ':': case '\\': case '/': case '#': + case '@': case '(': case ')': case '[': case ']': case '{': case '}': case '$': + case '%': case '+': case '*': case '?': case '!': case ',': case '\'': + return true; + default: + return false; + } +} + +void utf8_append(std::string& s, char32_t c) { + if (c < 0x80) { + s += static_cast(c); + } else if (c < 0x800) { + s += static_cast(0xC0 | (c >> 6)); + s += static_cast(0x80 | (c & 0x3F)); + } else if (c < 0x10000) { + s += static_cast(0xE0 | (c >> 12)); + s += static_cast(0x80 | ((c >> 6) & 0x3F)); + s += static_cast(0x80 | (c & 0x3F)); + } else { + s += static_cast(0xF0 | (c >> 18)); + s += static_cast(0x80 | ((c >> 12) & 0x3F)); + s += static_cast(0x80 | ((c >> 6) & 0x3F)); + s += static_cast(0x80 | (c & 0x3F)); + } +} + +// Decode UTF-16LE. `nul` chooses what an embedded NUL becomes: value strings +// are rendered with "|" between the members of a MULTI_SZ, names reject it. +bool utf16_to_utf8(std::span in, std::string& out, char nul) { + for (size_t i = 0; i + 1 < in.size(); i += 2) { + char32_t c = static_cast(in[i] | (uint32_t(in[i + 1]) << 8)); + if (c == 0) { + if (!nul) return false; + out += nul; + continue; + } + if (c >= 0xD800 && c < 0xDC00 && i + 3 < in.size()) { // surrogate pair + const char32_t lo = static_cast(in[i + 2] | (uint32_t(in[i + 3]) << 8)); + if (lo >= 0xDC00 && lo < 0xE000) { + c = 0x10000 + ((c - 0xD800) << 10) + (lo - 0xDC00); + i += 2; + } + } + utf8_append(out, c); + } + return true; +} + +std::string hex_of(std::span b) { + static const char* d = "0123456789abcdef"; + std::string s; + s.reserve(b.size() * 2); + for (uint8_t c : b) { + s += d[c >> 4]; + s += d[c & 0xF]; + } + return s; +} + +// Render a value's bytes for the text dump: strings as text, the fixed-width +// integers as numbers, everything else as hex. +std::string render(uint16_t type, std::span data) { + char buf[32]; + switch (type) { + case kRegSz: + case kRegExpandSz: + case kRegMultiSz: { + std::string s; + utf16_to_utf8(data, s, '|'); + while (!s.empty() && s.back() == '|') s.pop_back(); // trailing terminators + return s; + } + case kRegDword: { + const uint32_t v = uint32_t(data[0]) | (uint32_t(data[1]) << 8) | + (uint32_t(data[2]) << 16) | (uint32_t(data[3]) << 24); + std::snprintf(buf, sizeof(buf), "%u", v); + return buf; + } + case kRegQword: { + uint64_t v = 0; + for (int i = 7; i >= 0; --i) v = (v << 8) | data[i]; + std::snprintf(buf, sizeof(buf), "%llu", static_cast(v)); + return buf; + } + default: + return hex_of(data); + } +} + +} // namespace + +size_t ce_hive_values(const Reader& r, size_t base, std::vector& out, size_t limit) { + const size_t n = r.size(); + if (base + kScanStart + 6 > n) return 0; + size_t found = 0; + + for (size_t i = base + kScanStart; i + 6 <= n && found < limit; i += 2) { + const uint16_t type = r.at(i, Endian::Little).value_or(0); + if (!known_type(type)) continue; + const uint16_t data_len = r.at(i + 2, Endian::Little).value_or(0); + const uint16_t name_len = r.at(i + 4, Endian::Little).value_or(0); + if (name_len == 0 || name_len > kMaxNameChars) continue; + if (data_len == 0 || data_len > kMaxDataLen) continue; + if (type == kRegDword && data_len != 4) continue; + if (type == kRegDwordBe && data_len != 4) continue; + if (type == kRegQword && data_len != 8) continue; + + auto name_raw = r.bytes(i + 6, size_t(name_len) * 2); + if (!name_raw) continue; + auto data = r.bytes(i + 6 + size_t(name_len) * 2, data_len); + if (!data) continue; + + std::string name; + if (!utf16_to_utf8(*name_raw, name, 0)) continue; // an embedded NUL is not a name + // A registry value name is ASCII in practice; anything else (including + // a byte that only became valid UTF-8 by accident) is a false positive. + bool ok = !name.empty(); + for (unsigned char c : name) + if (!name_char(c)) { + ok = false; + break; + } + if (!ok) continue; + + out.push_back({i - base, std::move(name), type, render(type, *data)}); + ++found; + // Deliberately no skip past the record: a false positive here must not + // shift the scan away from the real records that follow it. + } + return found; +} + +const char* ce_hive_type_name(uint16_t type) { + switch (type) { + case kRegSz: return "REG_SZ"; + case kRegExpandSz: return "REG_EXPAND_SZ"; + case kRegBinary: return "REG_BINARY"; + case kRegDword: return "REG_DWORD"; + case kRegDwordBe: return "REG_DWORD_BE"; + case kRegMultiSz: return "REG_MULTI_SZ"; + case kRegQword: return "REG_QWORD"; + default: return ""; + } +} + +} // namespace ft diff --git a/src/wince_hive_parse.hpp b/src/wince_hive_parse.hpp new file mode 100644 index 0000000..bd1c4ab --- /dev/null +++ b/src/wince_hive_parse.hpp @@ -0,0 +1,41 @@ +// wince_hive_parse.hpp — value recovery from a Windows CE registry hive (.hv). +// +// A CE hive opens with a 0x400-byte block size, a zero DWORD, and the "EKIM" +// signature at offset 8, then a header of GUIDs and hashes before the cell +// data. The key tree is a CE-private cell layout that differs from NT's regf, +// so what is done here is RECOVERY, not a tree walk: every value record in the +// file is recovered structurally, without the key path it lived under. +// +// A value record is: +// u16 type; u16 data_len; u16 name_len; char16 name[name_len]; u8 data[data_len] +// +// Recovery scans every 2-byte position and accepts a record only when the type +// is one of the seven CE uses, the declared lengths fit the file, the fixed +// widths for DWORD/QWORD agree, and the name decodes to a plausible registry +// value name. The scan never skips ahead past an accepted record, so one false +// positive cannot desynchronise the rest of the file. +#pragma once + +#include +#include +#include + +#include "reader.hpp" + +namespace ft { + +struct CeHiveValue { + size_t offset = 0; // file offset of the record + std::string name; // UTF-8, converted from the stored UTF-16LE + uint16_t type = 0; // REG_SZ, REG_DWORD, ... + std::string rendered; // value as text: the string, the number, or hex +}; + +// Recover value records from the hive at `base`, appending to `out`, stopping +// at `limit` records. Returns the number recovered. +size_t ce_hive_values(const Reader& r, size_t base, std::vector& out, size_t limit); + +// Registry type name ("REG_SZ", ...), or "" for a type CE does not use. +const char* ce_hive_type_name(uint16_t type); + +} // namespace ft diff --git a/src/wince_rom_parse.cpp b/src/wince_rom_parse.cpp new file mode 100644 index 0000000..3e135be --- /dev/null +++ b/src/wince_rom_parse.cpp @@ -0,0 +1,207 @@ +// wince_rom_parse.cpp — ROMHDR location and table walking. See the header. +#include "wince_rom_parse.hpp" + +#include + +namespace ft { + +namespace { + +constexpr size_t kSigOff = 0x40; // "ECEC" + pTOC live here +constexpr size_t kRomHdrSize = 0x4c; // ROMHDR, followed by the two tables +constexpr size_t kTocEntry = 32; // TOCentry +constexpr size_t kFilesEntry = 28; // FILESentry +constexpr uint32_t kMaxModules = 4096; +constexpr uint32_t kMaxFiles = 8192; +constexpr size_t kMaxName = 260; + +// Virtual bases romimage has used for CE 4/5/6 kernel regions. Tried when the +// image does not carry a usable TOC-offset hint. +constexpr uint32_t kKnownBases[] = {0x80000000, 0x80200000, 0x80100000, 0x84000000, + 0x88000000, 0x8c000000, 0x00000000}; + +// pTOC does not always point exactly at the ROMHDR: some images place it a few +// bytes short (CE 6 images observed here put the header at pTOC + 8), so a +// candidate offset is probed across a small aligned skew. +constexpr size_t kMaxSkew = 0x40; + +// Does a ROMHDR at file offset `off` describe an image based at `vbase` that +// fits the file? This is the whole acceptance test for a TOC candidate. +bool hdr_plausible(const Reader& r, size_t base, size_t off, uint32_t vbase) { + if (off < base || off - base > r.size() - base) return false; + if (r.size() - off < kRomHdrSize) return false; + auto pf = r.at(off, Endian::Little); + auto pl = r.at(off + 4, Endian::Little); + auto nm = r.at(off + 8, Endian::Little); + auto nf = r.at(off + 0x28, Endian::Little); + if (!pf || !pl || !nm || !nf) return false; + if (*pf != vbase || *pf >= *pl) return false; + if (*nm == 0 || *nm > kMaxModules || *nf > kMaxFiles) return false; + // The image span must roughly match the file; a ROM is sometimes stored with + // its tail trimmed, so allow a little slack rather than demanding equality. + const uint64_t span = uint64_t(*pl) - *pf; + if (span > uint64_t(r.size() - base) + 0x10000) return false; + // Both tables have to be inside the file. + const uint64_t tables = uint64_t(kRomHdrSize) + uint64_t(*nm) * kTocEntry + + uint64_t(*nf) * kFilesEntry; + if (tables > r.size() - off) return false; + return true; +} + +bool load_hdr(const Reader& r, size_t off, CeRomHeader& h) { + auto u32 = [&](size_t o) { return r.at(off + o, Endian::Little).value_or(0); }; + h.hdr_off = off; + h.physfirst = u32(0x00); + h.physlast = u32(0x04); + h.nummods = u32(0x08); + h.ramstart = u32(0x0c); + h.ramfree = u32(0x10); + h.ramend = u32(0x14); + h.numfiles = u32(0x28); + h.kflags = u32(0x2c); + h.fsram = u32(0x30); + h.cpu = r.at(off + 0x3c, Endian::Little).value_or(0); + h.misc = r.at(off + 0x3e, Endian::Little).value_or(0); + return true; +} + +std::string read_cstr(const Reader& r, size_t off) { + std::string s; + for (size_t i = 0; i < kMaxName; ++i) { + auto c = r.at(off + i, Endian::Little); + if (!c || *c == 0) break; + s += static_cast(*c); + } + return s; +} + +} // namespace + +bool ce_rom_header(const Reader& r, size_t base, CeRomHeader& out) { + if (base > r.size() || r.size() - base < kSigOff + 8) return false; + auto ptoc = r.at(base + kSigOff + 4, Endian::Little); + if (!ptoc || *ptoc == 0) return false; + out.ptoc = *ptoc; + + // Candidate 1: the TOC offset romimage stores next to pTOC. When present it + // gives the virtual base directly (physfirst == pTOC - toc_offset). + auto toc_off = r.at(base + kSigOff + 8, Endian::Little); + if (toc_off && *toc_off != 0 && *toc_off <= *ptoc) { + const uint32_t vbase = *ptoc - *toc_off; + for (size_t skew = 0; skew < kMaxSkew; skew += 4) { + const size_t off = base + size_t(*toc_off) + skew; + if (hdr_plausible(r, base, off, vbase)) return load_hdr(r, off, out); + } + } + + // Candidate 2: the virtual bases romimage conventionally uses. + for (uint32_t vbase : kKnownBases) { + if (*ptoc < vbase) continue; + const uint64_t rel = uint64_t(*ptoc) - vbase; + if (rel > r.size() - base) continue; + for (size_t skew = 0; skew < kMaxSkew; skew += 4) { + const size_t off = base + size_t(rel) + skew; + if (hdr_plausible(r, base, off, vbase)) return load_hdr(r, off, out); + } + } + + // Candidate 3: an unconventional base. The ROMHDR's first field IS the + // virtual base, and its file offset is pTOC - base + skew, so a header at + // offset `o` must satisfy u32[o] == pTOC - (o - base) + skew. One aligned + // pass over the image finds it without guessing the base. + // + // Only for a ROM that is the whole file. "ECEC" is four bytes, so a large + // image can carry many stray matches, and running a full pass for each + // would be quadratic; every image romimage actually produces is resolved by + // one of the two cheap candidates above. + if (base != 0) return false; + const size_t end = r.size() >= kRomHdrSize ? r.size() - kRomHdrSize : 0; + for (size_t o = (base + 3) & ~size_t(3); o + 4 <= end; o += 4) { + auto v = r.at(o, Endian::Little); + if (!v) break; + const uint64_t rel = o - base; + // skew = v + rel - pTOC, valid only in [0, kMaxSkew) and 4-aligned. + const uint64_t skew = uint64_t(*v) + rel - *ptoc; + if (skew >= kMaxSkew || (skew & 3) != 0) continue; + if (hdr_plausible(r, base, o, *v)) return load_hdr(r, o, out); + } + return false; +} + +size_t ce_rom_offset(const Reader& r, size_t base, const CeRomHeader& h, uint32_t addr, + size_t need) { + if (addr < h.physfirst) return SIZE_MAX; + const uint64_t rel = uint64_t(addr) - h.physfirst; + const uint64_t off = uint64_t(base) + rel; + if (off > r.size() || need > r.size() - off) return SIZE_MAX; + return static_cast(off); +} + +bool ce_rom_modules(const Reader& r, size_t base, const CeRomHeader& h, + std::vector& out) { + const size_t table = h.hdr_off + kRomHdrSize; + if (!r.bytes(table, size_t(h.nummods) * kTocEntry)) return false; + out.reserve(h.nummods); + for (uint32_t i = 0; i < h.nummods; ++i) { + const size_t e = table + size_t(i) * kTocEntry; + auto u32 = [&](size_t o) { return r.at(e + o, Endian::Little).value_or(0); }; + CeModule m; + m.attr = u32(0x00); + m.size = u32(0x0c); + const uint32_t name_ptr = u32(0x10); + m.e32 = u32(0x14); + m.o32 = u32(0x18); + m.load = u32(0x1c); + const size_t no = ce_rom_offset(r, base, h, name_ptr, 1); + if (no == SIZE_MAX) continue; + m.name = read_cstr(r, no); + if (m.name.empty()) continue; + out.push_back(std::move(m)); + } + return true; +} + +bool ce_rom_files(const Reader& r, size_t base, const CeRomHeader& h, std::vector& out) { + const size_t table = h.hdr_off + kRomHdrSize + size_t(h.nummods) * kTocEntry; + if (!r.bytes(table, size_t(h.numfiles) * kFilesEntry)) return false; + out.reserve(h.numfiles); + for (uint32_t i = 0; i < h.numfiles; ++i) { + const size_t e = table + size_t(i) * kFilesEntry; + auto u32 = [&](size_t o) { return r.at(e + o, Endian::Little).value_or(0); }; + CeFile f; + f.attr = u32(0x00); + f.real = u32(0x0c); + f.comp = u32(0x10); + const uint32_t name_ptr = u32(0x14); + f.load = u32(0x18); + const size_t no = ce_rom_offset(r, base, h, name_ptr, 1); + if (no == SIZE_MAX) continue; + f.name = read_cstr(r, no); + if (f.name.empty()) continue; + out.push_back(std::move(f)); + } + return true; +} + +const char* ce_cpu_arch(uint16_t cpu) { + switch (cpu) { + case 0x014c: return "x86"; + case 0x0162: + case 0x0166: + case 0x0168: + case 0x0169: return "mips"; + case 0x01a2: + case 0x01a3: + case 0x01a4: + case 0x01a6: + case 0x01a8: return "sh"; + case 0x01c0: + case 0x01c2: + case 0x01c4: return "arm"; + case 0x0200: return "ia64"; + case 0x8664: return "x86_64"; + default: return ""; + } +} + +} // namespace ft diff --git a/src/wince_rom_parse.hpp b/src/wince_rom_parse.hpp new file mode 100644 index 0000000..437aac2 --- /dev/null +++ b/src/wince_rom_parse.hpp @@ -0,0 +1,72 @@ +// wince_rom_parse.hpp — Windows CE XIP ROM (ECEC / ROMHDR) table walking. +// +// A CE ROM image (nk.bin, or a Crestron .cos) begins with an ARM/MIPS branch to +// the bootstrap; at offset 0x40 sits the signature "ECEC" followed by pTOC, the +// VIRTUAL address of the ROMHDR. The ROMHDR gives the image's physical span, +// then two tables follow it: +// +// TOCentry[nummods] XIP modules — executables kept split into their E32/O32 +// ROM headers plus section data, not as whole PE files. +// FILESentry[numfiles] plain ROM files, optionally CECompress-compressed. +// +// Everything in those tables is a virtual address, so every lookup goes through +// offset_of() (addr - physfirst, relative to where the ROM starts in the file). +// Shared by the validator (which only needs the header) and the extractor. +#pragma once + +#include +#include +#include + +#include "reader.hpp" + +namespace ft { + +struct CeRomHeader { + size_t hdr_off = 0; // file offset of the ROMHDR itself + uint32_t ptoc = 0; // virtual address recorded at +0x44 + uint32_t physfirst = 0, physlast = 0; + uint32_t nummods = 0, numfiles = 0; + uint32_t ramstart = 0, ramfree = 0, ramend = 0; + uint32_t kflags = 0, fsram = 0; + uint16_t cpu = 0, misc = 0; + + uint32_t span() const { return physlast - physfirst; } +}; + +struct CeModule { + std::string name; + uint32_t attr = 0; + uint32_t size = 0; + uint32_t e32 = 0; // virtual address of the E32 ROM header + uint32_t o32 = 0; // virtual address of the O32 section table + uint32_t load = 0; +}; + +struct CeFile { + std::string name; + uint32_t attr = 0; + uint32_t real = 0; // decompressed size + uint32_t comp = 0; // stored size (== real when not compressed) + uint32_t load = 0; // virtual address of the stored bytes +}; + +// Locate and parse the ROMHDR for a ROM starting at `base`. Returns false when +// pTOC does not resolve to a self-consistent header (a bootloader that carries +// a stale ECEC signature but no ROM, say). +bool ce_rom_header(const Reader& r, size_t base, CeRomHeader& out); + +// Virtual address -> file offset, or SIZE_MAX when it falls outside the image. +size_t ce_rom_offset(const Reader& r, size_t base, const CeRomHeader& h, uint32_t addr, + size_t need); + +// Walk the module / file tables. Entries whose name or data lies outside the +// image are skipped; both return false only if the table itself is unreadable. +bool ce_rom_modules(const Reader& r, size_t base, const CeRomHeader& h, + std::vector& out); +bool ce_rom_files(const Reader& r, size_t base, const CeRomHeader& h, std::vector& out); + +// PE machine id -> short architecture name ("arm", "mips", ...); "" if unknown. +const char* ce_cpu_arch(uint16_t cpu); + +} // namespace ft diff --git a/tests/gen_samples.py b/tests/gen_samples.py index a025f17..6d9c134 100755 --- a/tests/gen_samples.py +++ b/tests/gen_samples.py @@ -684,6 +684,89 @@ def section(name, flags, data): return hdr + section(b"IniFile", 0, ini) + section(b"SIGN", 0, bytes(96)) +def wince_rom(): + # Windows CE XIP ROM: an ARM branch, "ECEC" + pTOC + the TOC offset at 0x40, + # then a ROMHDR whose physfirst is the image's virtual base, followed by the + # module and file tables. One module (headers only, no sections) and one + # stored file is enough for the validator to resolve pTOC -> ROMHDR. + base = 0x80200000 + img = bytearray(0x40) + struct.pack_into("> i) & 1) + self.n += 1 + if self.n == 16: + self.out += struct.pack("> 16, 16) + w.put(intel_filesize & 0xFFFF, 16) + else: + w.put(0, 1) + w.put(3, 3) + w.put(len(data), 24) + w.align_word() + out = bytes(w.out) + struct.pack(" len(frames): # fold the tail into the last frame's block + parts[-2] += parts[-1] + parts.pop() + return list(zip(parts, frames)) + + +SETUP_XML = """ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +""".encode() + +AGENT = b"MZ" + b"\x90\x00" + bytes(200) + b"agent payload" * 50 +HOSTKEY = b"-----BEGIN RSA PRIVATE KEY-----\n" + b"A" * 600 + b"\n-----END RSA PRIVATE KEY-----\n" +SETUP_DLL = b"MZ" + bytes(120) + b"setup dll body" +INSTALL_HDR = bytes(range(256)) * 4 + + +def run(path, outdir=None): + cmd = [MORIA, "-e", "-j", path, "-C", outdir] if outdir else [MORIA, "-j", path] + r = subprocess.run(cmd, capture_output=True, timeout=180) + if r.returncode != 0: + raise SystemExit("moria exited %d: %s" % (r.returncode, r.stderr.decode("replace"))) + return json.loads(r.stdout) + + +def fail(msg): + print("FAIL:", msg) + return 1 + + +def extract(tmp, name, blob): + path = os.path.join(tmp, name) + with open(path, "wb") as f: + f.write(blob) + outdir = os.path.join(tmp, name + ".out") + j = run(path, outdir) + ent = [e for e in j["extraction"]["extracted"] if e["type"] == "cab"] + if not ent: + raise AssertionError("no cab extraction for %s: %r" % (name, j["extraction"])) + return j, ent[0], os.path.join(outdir, ent[0]["root"]) + + +def check_roundtrip(tmp, name, folders, files, payloads, reserve=None, want_comp=None): + blob = build_cab(folders, files, reserve) + j, ent, root = extract(tmp, name, blob) + finds = [x for x in j["findings"] if x["type"] == "cab"] + if len(finds) != 1 or finds[0]["confidence_tier"] != "verified": + return fail("%s: findings %r" % (name, [(x["type"], x["confidence_tier"]) + for x in j["findings"]])) + if want_comp and finds[0].get("compression") != want_comp: + return fail("%s: compression %r, want %r" % (name, finds[0].get("compression"), want_comp)) + if ent["status"] != "ok": + return fail("%s: status %r (%r)" % (name, ent["status"], ent.get("warnings"))) + for member, want in payloads.items(): + p = os.path.join(root, member) + if not os.path.isfile(p): + return fail("%s: missing member %s" % (name, member)) + got = open(p, "rb").read() + if got != want: + return fail("%s: member %s differs (%d vs %d bytes)" % + (name, member, len(got), len(want))) + return 0 + + +def main(): + with tempfile.TemporaryDirectory() as tmp: + # 1. Stored folders, CE-style reserve fields, a file spanning two blocks. + a = b"stored-member-A;" * 400 # 6400 B: crosses the 4096 block boundary + b = b"stored-member-B\n" * 100 + data = a + b + rc = check_roundtrip( + tmp, "stored.cab", + folders=[(COMP_NONE, stored_blocks(data))], + files=[("dir\\a.txt", 0, 0, len(a)), ("b.txt", 0, len(a), len(b))], + payloads={"dir/a.txt": a, "b.txt": b}, + reserve=(8, 52, 4), want_comp="none") + if rc: + return rc + + # 2. MSZIP, several blocks, each matching back into the previous one. + big = b"".join(struct.pack("> 8)) & 0xFF for i in range(100000)) + rc = check_roundtrip( + tmp, "lzx.cab", + folders=[(COMP_LZX | (21 << 8), lzx_blocks(lz))], + files=[("frames.bin", 0, 0, len(lz))], + payloads={"frames.bin": lz}, want_comp="lzx") + if rc: + return rc + + # 4. A Windows CE installer cabinet, rebuilt from its _setup.xml. + members = [("_setup.xml", SETUP_XML), ("AGENT~1.001", AGENT), + ("HOSTKE~1.002", HOSTKEY), ("SETUP.999", SETUP_DLL), + ("HEADER.000", INSTALL_HDR), ("EXTRA~1.500", b"unlisted member")] + stream = b"".join(p for _, p in members) + files, off = [], 0 + for nm, p in members: + files.append((nm, 0, off, len(p))) + off += len(p) + blob = build_cab([(COMP_NONE, stored_blocks(stream))], files, reserve=(0, 52, 0)) + j, ent, root = extract(tmp, "ceinstall.cab", blob) + if ent["status"] != "ok": + return fail("CE cab status %r (%r)" % (ent["status"], ent.get("warnings"))) + + want = { + "_setup.xml": SETUP_XML, + "fs/Windows/Agent.exe": AGENT, + "fs/Sys/SSH/host_key.pem": HOSTKEY, + "setup.dll": SETUP_DLL, + "install-header.000": INSTALL_HDR, + "unmapped/EXTRA~1.500": b"unlisted member", + } + for rel, blobw in want.items(): + p = os.path.join(root, rel) + if not os.path.isfile(p): + return fail("CE cab: missing %s (have %r)" % + (rel, sorted(os.listdir(root)))) + if open(p, "rb").read() != blobw: + return fail("CE cab: %s differs" % rel) + + reg = open(os.path.join(root, "registry.reg"), "r").read() + for line in ("[HKEY_LOCAL_MACHINE\\Crestron\\System]", + '"Version"="1.8001.0298"', + '"Build"=dword:0000012a'): + if line not in reg: + return fail("CE cab: registry.reg missing %r:\n%s" % (line, reg)) + + print("ok: cab identify + stored/MSZIP/LZX extraction + CE install-tree rebuild") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/test_wince_hive.py b/tests/test_wince_hive.py new file mode 100644 index 0000000..42d7a35 --- /dev/null +++ b/tests/test_wince_hive.py @@ -0,0 +1,132 @@ +#!/usr/bin/env python3 +"""Windows CE registry hive identify + value-recovery regression. + +Builds a synthetic hive (the "EKIM" header plus value records of every type CE +stores, padded with plausible-looking noise), runs `moria -e`, and asserts the +dump recovers each value with the right name, type, and rendering — and that a +header with no recoverable records is rejected rather than reported as a hive. + +Self-contained; no external tools. Exit nonzero on any failure. +""" +import json +import os +import struct +import subprocess +import sys +import tempfile + +HERE = os.path.dirname(os.path.abspath(__file__)) +MORIA = os.path.join(HERE, "..", "build", "moria") + +REG_SZ, REG_EXPAND_SZ, REG_BINARY, REG_DWORD, REG_MULTI_SZ, REG_QWORD = 1, 2, 3, 4, 7, 11 + +# (name, type, data, expected rendering in the dump) +VALUES = [ + ("Version", REG_SZ, "1.8001.0298\0".encode("utf-16-le"), "1.8001.0298"), + ("Path", REG_EXPAND_SZ, "%CE2%\\svc.exe\0".encode("utf-16-le"), "%CE2%\\svc.exe"), + ("Order", REG_MULTI_SZ, "a\0bb\0ccc\0\0".encode("utf-16-le"), "a|bb|ccc"), + ("Flags", REG_DWORD, struct.pack(" 3 else "") + + for name, _vtype, _data, want in VALUES: + key = name.split()[0] + if key not in rows: + return fail("value %r missing from the dump:\n%s" % (name, dump)) + got_type, got_val = rows[key] + if got_val != want: + return fail("value %r rendered %r, want %r" % (name, got_val, want)) + if not got_type.startswith("REG_"): + return fail("value %r type column %r" % (name, got_type)) + + # Every record we planted must be found, and nothing much beyond them: + # the noise tail is there to catch a scan that accepts anything. + body = [l for l in dump.splitlines() if l and not l.startswith("#")] + if len(body) < len(VALUES): + return fail("recovered %d records, planted %d" % (len(body), len(VALUES))) + if len(body) > len(VALUES) + 4: + return fail("recovered %d records from %d planted: the scan is too loose\n%s" % + (len(body), len(VALUES), dump)) + + # A header with no recoverable records is not a hive. + empty = os.path.join(tmp, "notahive.bin") + with open(empty, "wb") as f: + f.write(build_hive(with_values=False)) + j = run(empty) + if any(x["type"] == "wince_hive" for x in j["findings"]): + return fail("EKIM header with no values was reported as a hive") + + print("ok: wince_hive identify + value recovery") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/test_wince_rom.py b/tests/test_wince_rom.py new file mode 100644 index 0000000..2049727 --- /dev/null +++ b/tests/test_wince_rom.py @@ -0,0 +1,246 @@ +#!/usr/bin/env python3 +"""Windows CE XIP ROM identify + extract regression. + +Builds a synthetic CE ROM (ECEC signature, ROMHDR, one XIP module split into an +E32/O32 pair with a stored and a CECompress-coded section, plus a stored and a +compressed ROM file), runs `moria -e`, and asserts: + + * the image is identified as wince_rom at `verified` tier with the right + module/file counts and span, + * both ROM files come back byte-for-byte, the compressed one decoded, + * the module is rebuilt into a PE whose sections land at their RVAs with the + right bytes, including the section that was CECompress-coded, + * a file that carries the ECEC signature but no resolvable ROMHDR (what a CE + bootloader built from the same sources looks like) stays at `magic` tier and + is not extracted. + +Self-contained; no external tools. Exit nonzero on any failure. +""" +import json +import os +import struct +import subprocess +import sys +import tempfile + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, HERE) +MORIA = os.path.join(HERE, "..", "build", "moria") + +from lzxbuild import ce_compress_blob # noqa: E402 + +BASE = 0x80200000 # virtual address the image is linked at +CPU_ARM = 0x01C2 # IMAGE_FILE_MACHINE_THUMB +SCN_COMPRESSED = 0x00002000 +SCN_CODE = 0x00000020 +SCN_WRITE = 0x80000000 + +TEXT = bytes(range(256)) * 12 # 3072 B, stored verbatim +DATA = (b"CRESTRON-CE-SECTION-" * 300)[:5000] # CECompress-coded section +PLAIN = b"[boot]\r\nDevice=CP3\r\n" * 40 # stored ROM file +PACKED = (b"ce-rom-file-payload;" * 900)[:17000] # CECompress-coded ROM file + +TEXT_RVA = 0x1000 +DATA_RVA = 0x2000 +# The .data section's virtual size exceeds its stored blob: CE zero-fills the +# tail at load time, and the extractor has to do the same rather than treat the +# short decode as a failure. +DATA_VSIZE = 0x2000 +DATA_LOADED = DATA + bytes(DATA_VSIZE - len(DATA)) +IMAGE_VSIZE = DATA_RVA + DATA_VSIZE + + +def build_rom(): + """Assemble a ROM image; returns the bytes.""" + img = bytearray(0x40) + img += b"ECEC" + struct.pack(" Date: Thu, 1 Oct 2026 14:35:37 -0400 Subject: [PATCH 2/3] feat(cfbf): read MSI/compound-file streams, and fix LZX framing in CAB --- CMakeLists.txt | 4 + README.md | 19 +-- moria.1 | 11 +- signatures/cfbf.toml | 54 +++++++ src/archives.cpp | 15 ++ src/cfbf_parse.cpp | 295 ++++++++++++++++++++++++++++++++++++ src/cfbf_parse.hpp | 78 ++++++++++ src/extract/cfbf.cpp | 111 ++++++++++++++ src/extract/cfbf.hpp | 27 ++++ src/extract/lzx.cpp | 43 +++++- src/extract/manifest.cpp | 5 + src/extract/manifest.hpp | 1 + src/main.cpp | 66 +++++++- src/validators/cfbf.cpp | 34 +++++ src/validators/cfbf.hpp | 11 ++ src/validators/registry.cpp | 2 + tests/cfbfbuild.py | 162 ++++++++++++++++++++ tests/gen_samples.py | 12 ++ tests/lzxbuild.py | 280 +++++++++++++++++++++++++++++++++- tests/run.sh | 2 + tests/test_cab.py | 54 ++++++- tests/test_cfbf.py | 258 +++++++++++++++++++++++++++++++ tests/test_wince_rom.py | 2 +- 23 files changed, 1507 insertions(+), 39 deletions(-) create mode 100644 signatures/cfbf.toml create mode 100644 src/cfbf_parse.cpp create mode 100644 src/cfbf_parse.hpp create mode 100644 src/extract/cfbf.cpp create mode 100644 src/extract/cfbf.hpp create mode 100644 src/validators/cfbf.cpp create mode 100644 src/validators/cfbf.hpp create mode 100644 tests/cfbfbuild.py create mode 100644 tests/test_cfbf.py diff --git a/CMakeLists.txt b/CMakeLists.txt index 2e1857d..675f80d 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -70,10 +70,12 @@ add_executable(moria src/extract/rae_rfp.cpp src/extract/wince_rom.cpp src/extract/cab.cpp + src/extract/cfbf.cpp src/extract/wince_hive.cpp src/wince_hive_parse.cpp src/extract/ce_setup.cpp src/cab_parse.cpp + src/cfbf_parse.cpp src/extract/lzx.cpp src/wince_rom_parse.cpp src/extract/lzari.cpp @@ -123,6 +125,7 @@ add_executable(moria src/validators/rae_rfp.cpp src/validators/wince_rom.cpp src/validators/cab.cpp + src/validators/cfbf.cpp src/validators/wince_hive.cpp src/validators/vbf.cpp src/validators/verity.cpp @@ -249,6 +252,7 @@ if(Python3_Interpreter_FOUND) test_wince_rom test_wince_hive test_cab + test_cfbf test_vbf test_uboot_env test_vbmeta diff --git a/README.md b/README.md index 23753dd..cf66d78 100644 --- a/README.md +++ b/README.md @@ -30,14 +30,15 @@ The `moria` binary is **self-contained**: all signature sets are embedded at bui Output is human-readable by default. Pass `-j` for JSON. ``` -moria # identify: a findings tree with offsets, types, and confidence -moria # scan a tree: a type summary plus the notable files -moria -j # JSON, for tools and agents -moria -e # extract to .extracted/ (-C DIR to choose the output dir) -moria -c # carve raw byte ranges to .carved/ (no parsing) -moria -E # entropy pass: flag unidentified / possibly-encrypted regions -moria --list # list tar/cpio/zip/cab/CE-ROM members without extracting -moria --broad # also load the ~2.5k general file-type signatures +moria # identify: a findings tree with offsets, types, and confidence +moria # scan a tree: a type summary plus the notable files +moria -j # JSON, for tools and agents +moria -e # extract to .extracted/ (-C DIR to choose the output dir) +moria -c # carve raw byte ranges to .carved/ (no parsing) +moria -E # entropy pass: flag unidentified / possibly-encrypted regions +moria --list # list tar/cpio/zip/cab/CE-ROM members without extracting +moria --broad # also load the ~2.5k general file-type signatures +moria --unpack-executables # unpack executables inside other extracted output (cfbf/msi files) moria --help ``` @@ -48,7 +49,7 @@ moria --help Unpacked in-process, no external tools and no sudo: - **Filesystems:** SquashFS, ext2/3/4, F2FS, FAT12/16/32, exFAT, NTFS, HFS+/HFSX, XFS, btrfs, JFFS2, UBI/UBIFS, romfs, YAFFS2, cramfs, EROFS -- **Archives and images:** ZIP, tar, cpio, MS-CAB (stored/MSZIP/LZX), ISO 9660, Android sparse, Android boot +- **Archives and images:** ZIP, tar, cpio, MS-CAB (stored/MSZIP/LZX), CFBF compound files / Windows Installer `.msi`, ISO 9660, Android sparse, Android boot - **Kernels and wrappers:** U-Boot uImage, U-Boot FIT, Windows CE XIP ROM (nk.bin / .cos), standalone gzip / xz / zstd / lz4 streams - **Firmware packages:** RAE Systems / Honeywell RFP (section table; LZARI-decompresses each section) - **Configuration stores:** U-Boot environment, ESP32 NVS, Windows CE registry hives (`.hv`) diff --git a/moria.1 b/moria.1 index 548edf2..c3ec816 100644 --- a/moria.1 +++ b/moria.1 @@ -13,7 +13,7 @@ embedded formats (filesystems, containers, bootloaders, executables, compression, media) with byte offsets and confidence tiers. It is identification-first; with .B \-\-extract -it also unpacks SquashFS, ext2/3/4, F2FS, JFFS2, UBIFS, romfs, YAFFS2, cramfs, FAT, exFAT, NTFS, HFS+, XFS, btrfs, EROFS, and ISO9660 filesystems, ZIP/tar/cpio/MS-CAB archives, U-Boot uImage/FIT containers, Windows CE XIP ROM images and registry hives, and RAE Systems/Honeywell RFP firmware packages (LZARI-decompressing their sections) +it also unpacks SquashFS, ext2/3/4, F2FS, JFFS2, UBIFS, romfs, YAFFS2, cramfs, FAT, exFAT, NTFS, HFS+, XFS, btrfs, EROFS, and ISO9660 filesystems, ZIP/tar/cpio/MS-CAB archives, CFBF compound files (Windows Installer .msi), U-Boot uImage/FIT containers, Windows CE XIP ROM images and registry hives, and RAE Systems/Honeywell RFP firmware packages (LZARI-decompressing their sections) internally (no sudo). Every other format is still identified and located by offset, ready for another tool such as binwalk. .PP Output is human-readable by default. A single file shows a findings tree @@ -55,7 +55,7 @@ List tar/cpio/zip/MS-CAB member names and sizes, or a Windows CE ROM's modules a .SS Extract .TP .BR \-e ", " \-\-extract -Unpack identified SquashFS, ext2/3/4, F2FS, JFFS2, UBIFS (incl. UBI), romfs, YAFFS2, cramfs, FAT, exFAT, NTFS, EROFS, and ISO9660 filesystems, ZIP/tar/cpio/MS-CAB archives (rebuilding a Windows CE installer cabinet's real install tree from its _setup.xml), U-Boot uImage/FIT containers, Android boot/sparse images, Windows CE XIP ROM images (rebuilding each XIP module into a flat PE), Windows CE registry hives (dumping their recoverable values to text), and RAE Systems/Honeywell RFP firmware packages into +Unpack identified SquashFS, ext2/3/4, F2FS, JFFS2, UBIFS (incl. UBI), romfs, YAFFS2, cramfs, FAT, exFAT, NTFS, EROFS, and ISO9660 filesystems, ZIP/tar/cpio/MS-CAB archives, CFBF compound files such as Windows Installer .msi packages, U-Boot uImage/FIT containers, Android boot/sparse images, Windows CE XIP ROM images, Windows CE registry hives, and RAE Systems/Honeywell RFP firmware packages into .IR file .extracted/ (one subdir per region, named .IR 0x\- ). @@ -91,6 +91,13 @@ output bytes (default 4294967296 = 4 GiB); the manifest is marked .IR capped . A per-extraction decompression-ratio guard (output/input over 1000:1 above 10 MiB) also stops descent into a likely decompression bomb. +.TP +.B \-\-unpack\-executables +Also unpack installer/document containers (CFBF/MSI) that are found +inside other extracted output. By default these are unpacked only when they +are the root container because a single .msi expands to several hundred +database\-table streams and burying the rest of the findings under them is +rarely what you want. .SS Carve .TP .BR \-c ", " \-\-carve diff --git a/signatures/cfbf.toml b/signatures/cfbf.toml new file mode 100644 index 0000000..86865f2 --- /dev/null +++ b/signatures/cfbf.toml @@ -0,0 +1,54 @@ +# NOTE: all root-level keys must precede the first [[magic]] / [doc] table. +name = "cfbf" +category = "container" + +# Compound File Binary Format (MSI, .doc/.xls, .msg): a miniature FAT +# filesystem. The +# header names the sector geometry and where the FAT, mini-FAT and directory +# start; the validator builds the FAT and walks the directory before believing +# the match, and sizes the finding by the last sector the FAT accounts for +# (nothing in the header records the file length). +struct = """ + bytes[8] signature; + bytes[16] clsid; + u16 version_minor; + u16 version_major; + u16 byte_order; + u16 sector_shift; + u16 mini_sector_shift; + u16 reserved1; + u32 reserved2; + u32 num_dir_sectors; + u32 num_fat_sectors; + u32 first_dir_sector; + u32 txn_signature; + u32 mini_cutoff; + u32 first_minifat_sector; + u32 num_minifat_sectors; + u32 first_difat_sector; + u32 num_difat_sectors; +""" + +constraints = [ + "byte_order == 0xfffe", + "version_major >= 3 && version_major <= 4", + "sector_shift >= 9 && sector_shift <= 12", + "mini_sector_shift == 6", + "mini_cutoff == 4096", + "num_fat_sectors > 0", + "reserved1 == 0", + "reserved2 == 0", +] + +# Builds the FAT, walks the directory, sizes the span, names MSI. +validator = "cfbf" + +[[magic]] +hex = "d0cf11e0a1b11ae1" +endian = "little" + +[doc] +description = "Compound File Binary Format (CFBF), the container behind Windows Installer .msi packages and legacy Office documents; the same on-disk layout was previously called OLE2 / OLE structured storage. Storages and streams are addressed through a FAT, so a stream's bytes are not contiguous in the file. Only the container is parsed, not the COM object semantics layered on it." +references = [ + { title = "[MS-CFB] Compound File Binary File Format", url = "https://learn.microsoft.com/en-us/openspecs/windows_protocols/ms-cfb/" }, +] diff --git a/src/archives.cpp b/src/archives.cpp index b91e135..e435369 100644 --- a/src/archives.cpp +++ b/src/archives.cpp @@ -5,6 +5,7 @@ #include #include "cab_parse.hpp" +#include "cfbf_parse.hpp" #include "wince_rom_parse.hpp" namespace ft { @@ -140,6 +141,19 @@ void list_cab(const Reader& r, Finding& f) { } } +// A compound file's directory already names every stream and its length, so +// listing costs only the header + FAT walk. Storages are skipped: they are +// directories, and only streams carry bytes. +void list_cfbf(const Reader& r, Finding& f) { + Cfbf c; + if (!cfbf_parse(r, f.offset, c)) return; + for (const CfbfEntry& e : c.entries) { + if (e.type != kCfbfStream || e.size == 0) continue; + if (f.members.size() >= MAX_MEMBERS) { f.members_truncated = true; break; } + f.members.push_back({e.name, static_cast(e.size), "stored", {}}); + } +} + // A CE ROM carries two member lists: XIP modules and plain ROM files. Both are // named in the TOC, so listing is free. void list_wince_rom(const Reader& r, Finding& f) { @@ -167,6 +181,7 @@ void list_members(const Reader& r, Finding& f) { else if (f.type == "cpio") list_cpio(r, f); else if (f.type == "zip") list_zip(r, f); else if (f.type == "cab") list_cab(r, f); + else if (f.type == "cfbf") list_cfbf(r, f); else if (f.type == "wince_rom") list_wince_rom(r, f); } diff --git a/src/cfbf_parse.cpp b/src/cfbf_parse.cpp new file mode 100644 index 0000000..43e9713 --- /dev/null +++ b/src/cfbf_parse.cpp @@ -0,0 +1,295 @@ +// cfbf_parse.cpp — compound file (CFBF) walking. See the header. +#include "cfbf_parse.hpp" + +#include +#include + +namespace ft { + +namespace { + +// A compound file is never this large in firmware; the cap keeps a corrupt +// sector count from making us allocate a FAT the file cannot back. +constexpr size_t kMaxSectors = 1u << 24; // 16M sectors +constexpr size_t kMaxDirEntries = 1u << 18; +constexpr uint64_t kMaxStream = uint64_t(1) << 32; + +// MSI encodes its stream names in the 0x3800..0x483F code-unit range, packing +// one or two characters of a 64-symbol alphabet into each UTF-16 unit. Without +// this the tables come out as CJK mojibake and Data1.cab is unrecognizable. +char mime_char(uint32_t id) { + if (id < 10) return static_cast('0' + id); + if (id < 36) return static_cast('A' + id - 10); + if (id < 62) return static_cast('a' + id - 36); + return id == 62 ? '.' : '_'; +} + +void append_utf8(std::string& s, uint32_t cp) { + if (cp < 0x80) { + s += static_cast(cp); + } else if (cp < 0x800) { + s += static_cast(0xC0 | (cp >> 6)); + s += static_cast(0x80 | (cp & 0x3F)); + } else { + s += static_cast(0xE0 | (cp >> 12)); + s += static_cast(0x80 | ((cp >> 6) & 0x3F)); + s += static_cast(0x80 | (cp & 0x3F)); + } +} + +// Decode a directory entry name. Returns the display form and whether any unit +// was MSI-encoded. +// +// U+4840 is not a payload unit: MSI puts it in front of a name to mark the +// stream as one of the database tables. It is dropped, so `_Validation` reads +// as itself rather than carrying a stray CJK character. +std::string decode_name(const std::vector& units, bool& mangled) { + mangled = false; + for (uint16_t u : units) + if (u >= 0x3800 && u <= 0x4840) mangled = true; + + std::string out; + for (uint16_t u : units) { + if (mangled && u == 0x4840) { + continue; // MSI table marker, not a character + } else if (mangled && u >= 0x3800 && u < 0x4800) { + const uint32_t v = uint32_t(u) - 0x3800; + out += mime_char(v & 0x3F); + out += mime_char((v >> 6) & 0x3F); + } else if (mangled && u >= 0x4800 && u < 0x4840) { + out += mime_char(uint32_t(u) - 0x4800); + } else if (u < 0x20) { + // \x05SummaryInformation and friends: keep the name readable. + out += '_'; + } else { + append_utf8(out, u); + } + } + return out; +} + +} // namespace + +bool cfbf_parse(const Reader& r, size_t base, Cfbf& out) { + auto u16 = [&](size_t o) { return r.at(base + o, Endian::Little); }; + auto u32 = [&](size_t o) { return r.at(base + o, Endian::Little); }; + + auto sig = r.bytes(base, 8); + static const uint8_t kMagic[8] = {0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1}; + if (!sig || !std::equal(sig->begin(), sig->end(), kMagic)) return false; + + auto minor = u16(0x18), major = u16(0x1A), order = u16(0x1C); + auto ssh = u16(0x1E), mssh = u16(0x20); + auto nfat = u32(0x2C), dir_start = u32(0x30), cutoff = u32(0x38); + auto mfat_start = u32(0x3C), nmfat = u32(0x40); + auto difat_start = u32(0x44), ndifat = u32(0x48); + if (!minor || !major || !order || !ssh || !mssh || !nfat || !dir_start || !cutoff || + !mfat_start || !nmfat || !difat_start || !ndifat) + return false; + + if (*order != 0xFFFE) return false; // little-endian only, per spec + if (*major != 3 && *major != 4) return false; + if (*ssh < 9 || *ssh > 12 || *mssh != 6) return false; + out.ver_major = *major; + out.ver_minor = *minor; + out.sector_size = 1u << *ssh; + out.mini_sector_size = 1u << *mssh; + out.mini_cutoff = *cutoff; + + const size_t ss = out.sector_size; + const size_t per_sector = ss / 4; + // Sector n starts one sector in: the header occupies the first sector (512 + // bytes of it, zero-padded to `ss` for the 4 KiB v4 layout). + auto sector_off = [&](uint32_t n) { return base + size_t(n + 1) * ss; }; + auto sector = [&](uint32_t n) { return r.bytes(sector_off(n), ss); }; + + // DIFAT: 109 entries inline, the rest in a chain of DIFAT sectors. + std::vector difat; + difat.reserve(109); + for (unsigned i = 0; i < 109; ++i) { + auto v = u32(0x4C + size_t(i) * 4); + if (!v) return false; + if (*v <= kCfbfMaxSect) difat.push_back(*v); + } + { + uint32_t s = *difat_start; + std::set seen; + for (uint32_t i = 0; i < *ndifat && s <= kCfbfMaxSect; ++i) { + if (!seen.insert(s).second) return false; // DIFAT loop + auto d = sector(s); + if (!d) return false; + for (size_t k = 0; k + 1 < per_sector; ++k) { + uint32_t v; + std::memcpy(&v, d->data() + k * 4, 4); + if (v <= kCfbfMaxSect) difat.push_back(v); + } + std::memcpy(&s, d->data() + (per_sector - 1) * 4, 4); + } + } + if (difat.empty() || difat.size() > kMaxSectors / per_sector + 1) return false; + + // FAT: the concatenation of every FAT sector the DIFAT names. + out.fat.clear(); + out.fat.reserve(std::min(difat.size(), *nfat) * per_sector); + for (size_t i = 0; i < difat.size() && i < *nfat; ++i) { + auto d = sector(difat[i]); + if (!d) return false; + for (size_t k = 0; k < per_sector; ++k) { + uint32_t v; + std::memcpy(&v, d->data() + k * 4, 4); + out.fat.push_back(v); + } + } + if (out.fat.empty()) return false; + + // The file ends after the last sector the FAT accounts for. Trailing FREE + // entries are padding inside the final FAT sector, not real sectors. + size_t last = SIZE_MAX; + for (size_t i = out.fat.size(); i-- > 0;) { + if (out.fat[i] != kCfbfFree) { + last = i; + break; + } + } + if (last == SIZE_MAX) return false; + out.span = uint64_t(last + 2) * ss; + const uint64_t avail = r.size() - base; + if (out.span > avail) out.span = avail; + + // Follow a chain through the regular FAT, collecting sector numbers. + auto walk = [&](uint32_t start, std::vector& chain) { + std::set seen; + uint32_t n = start; + while (n <= kCfbfMaxSect) { + if (n >= out.fat.size()) return false; + if (!seen.insert(n).second) return false; // chain loop + if (chain.size() > kMaxSectors) return false; + chain.push_back(n); + n = out.fat[n]; + } + return n == kCfbfEndChain; + }; + + // Mini-FAT, for streams below the cutoff. + out.minifat.clear(); + if (*mfat_start <= kCfbfMaxSect && *nmfat) { + std::vector chain; + if (walk(*mfat_start, chain)) { + for (uint32_t s : chain) { + auto d = sector(s); + if (!d) break; + for (size_t k = 0; k < per_sector; ++k) { + uint32_t v; + std::memcpy(&v, d->data() + k * 4, 4); + out.minifat.push_back(v); + } + } + } + } + + // Directory: 128-byte entries across the chain from first_dir_sector. + std::vector dir_chain; + if (!walk(*dir_start, dir_chain)) return false; + std::vector dir; + dir.reserve(dir_chain.size() * ss); + for (uint32_t s : dir_chain) { + auto d = sector(s); + if (!d) return false; + dir.insert(dir.end(), d->begin(), d->end()); + } + + const size_t count = std::min(dir.size() / 128, kMaxDirEntries); + out.entries.clear(); + out.entries.reserve(count); + for (size_t i = 0; i < count; ++i) { + const uint8_t* e = dir.data() + i * 128; + uint16_t nlen; + std::memcpy(&nlen, e + 0x40, 2); + CfbfEntry ce; + ce.type = e[0x42]; + std::memcpy(&ce.left, e + 0x44, 4); + std::memcpy(&ce.right, e + 0x48, 4); + std::memcpy(&ce.child, e + 0x4C, 4); + std::memcpy(&ce.start, e + 0x74, 4); + uint32_t lo, hi; + std::memcpy(&lo, e + 0x78, 4); + std::memcpy(&hi, e + 0x7C, 4); + ce.size = uint64_t(lo) | (uint64_t(hi) << 32); + // v3 files often leave the high dword as garbage; the spec says ignore it. + if (out.ver_major == 3) ce.size = lo; + if (ce.type != kCfbfStorage && ce.type != kCfbfStream && ce.type != kCfbfRoot) { + out.entries.push_back(CfbfEntry{}); // unallocated: keep indices aligned + continue; + } + if (nlen >= 2 && nlen <= 64) { + std::vector units; + for (size_t k = 0; k + 1 < size_t(nlen); k += 2) { + uint16_t u; + std::memcpy(&u, e + k, 2); + if (!u) break; // the length includes the NUL terminator + units.push_back(u); + } + ce.name = decode_name(units, ce.mangled); + } + if (ce.mangled) out.msi = true; + if (ce.type == kCfbfStream) out.streams++; + if (ce.type == kCfbfStorage) out.storages++; + if (ce.type == kCfbfRoot) { + out.mini_start = ce.start; + out.mini_size = ce.size; + } + out.entries.push_back(std::move(ce)); + } + if (out.entries.empty()) return false; + if (out.entries[0].type != kCfbfRoot) return false; + return true; +} + +bool cfbf_stream(const Reader& r, size_t base, const Cfbf& c, const CfbfEntry& e, + std::vector& out) { + out.clear(); + if (e.size == 0) return true; + if (e.size > kMaxStream || e.size > r.size()) return false; + + const size_t ss = c.sector_size; + auto read_chain = [&](const std::vector& fat, uint32_t start, size_t unit, + uint64_t want, auto&& fetch) { + std::set seen; + uint32_t n = start; + out.reserve(static_cast(want)); + while (out.size() < want) { + if (n > kCfbfMaxSect || n >= fat.size()) return false; + if (!seen.insert(n).second) return false; // loop + auto d = fetch(n); + if (!d) return false; + const size_t take = std::min(unit, static_cast(want) - out.size()); + out.insert(out.end(), d->begin(), d->begin() + static_cast(take)); + n = fat[n]; + } + return true; + }; + + if (e.size < c.mini_cutoff && e.type != kCfbfRoot) { + // Small streams live inside the root entry's mini stream, addressed by + // the mini-FAT in 64-byte units. + if (c.minifat.empty()) return false; + CfbfEntry root; + root.type = kCfbfRoot; + root.start = c.mini_start; + root.size = c.mini_size; + std::vector mini; + if (!cfbf_stream(r, base, c, root, mini)) return false; + const size_t mu = c.mini_sector_size; + return read_chain(c.minifat, e.start, mu, e.size, [&](uint32_t n) { + const size_t off = size_t(n) * mu; + if (off + mu > mini.size()) return std::optional>{}; + return std::optional>( + std::span(mini).subspan(off, mu)); + }); + } + + return read_chain(c.fat, e.start, ss, e.size, + [&](uint32_t n) { return r.bytes(base + size_t(n + 1) * ss, ss); }); +} + +} // namespace ft diff --git a/src/cfbf_parse.hpp b/src/cfbf_parse.hpp new file mode 100644 index 0000000..b62ba53 --- /dev/null +++ b/src/cfbf_parse.hpp @@ -0,0 +1,78 @@ +// cfbf_parse.hpp — Compound File Binary Format (CFBF) structure walking. +// +// A compound file is a FAT filesystem in miniature: a 512-byte header, then +// sectors chained through a file allocation table, holding a directory tree of +// storages (directories) and streams (files). MSI installers, .doc/.xls, and +// Outlook .msg are all this format. +// +// This is [MS-CFB] and nothing above it. The COM layer that the format was +// originally built for (structured storage: class-bound objects, \001CompObj, +// property sets) is not parsed or interpreted here — a stream is bytes. +// +// Why moria needs it: a stream is NOT contiguous on disk. An installer's +// Data1.cab lives in a compound-file stream, so a linear scan finds the MSCF +// magic at the stream's first sector and then reads whatever sectors happen to +// follow — usually a few kilobytes in, that is some other stream's bytes. The +// cabinet decodes perfectly right up to the first sector jump and turns to +// noise after it. Reassembling the chain first is the only way to read it. +#pragma once + +#include +#include +#include + +#include "reader.hpp" + +namespace ft { + +// FAT entry markers. Anything <= kCfbfMaxSect is a real sector number. +constexpr uint32_t kCfbfMaxSect = 0xFFFFFFFA; +constexpr uint32_t kCfbfDifSect = 0xFFFFFFFC; +constexpr uint32_t kCfbfFatSect = 0xFFFFFFFD; +constexpr uint32_t kCfbfEndChain = 0xFFFFFFFE; +constexpr uint32_t kCfbfFree = 0xFFFFFFFF; + +constexpr uint32_t kCfbfNoStream = 0xFFFFFFFF; + +constexpr uint8_t kCfbfStorage = 1; +constexpr uint8_t kCfbfStream = 2; +constexpr uint8_t kCfbfRoot = 5; + +struct CfbfEntry { + std::string name; // display name: MSI-demangled when it was encoded + uint8_t type = 0; // kCfbfStorage | kCfbfStream | kCfbfRoot + uint32_t start = 0; // first sector (mini-FAT sector when size < mini_cutoff) + uint64_t size = 0; + bool mangled = false; // the stored name used MSI's code-unit encoding + // Directory tree links: siblings form a red-black tree per storage, `child` + // heads the tree of whatever a storage contains. kCfbfNoStream = absent. + uint32_t left = 0xFFFFFFFF, right = 0xFFFFFFFF, child = 0xFFFFFFFF; +}; + +struct Cfbf { + unsigned sector_size = 512; + unsigned mini_sector_size = 64; + uint32_t mini_cutoff = 4096; + uint16_t ver_major = 3; + uint16_t ver_minor = 0; + std::vector fat; + std::vector minifat; + std::vector entries; // directory order, root first when present + uint32_t mini_start = 0; // root entry's chain: the mini-stream container + uint64_t mini_size = 0; + uint64_t span = 0; // bytes the compound file occupies from `base` + size_t streams = 0; + size_t storages = 0; + bool msi = false; // at least one MSI-encoded name +}; + +// Parse header, FAT, mini-FAT and directory at `base`. False if the bytes are +// not a usable compound file. +bool cfbf_parse(const Reader& r, size_t base, Cfbf& out); + +// Reassemble a stream's bytes by following its sector chain. False if the chain +// is broken, loops, or runs outside the file. +bool cfbf_stream(const Reader& r, size_t base, const Cfbf& c, const CfbfEntry& e, + std::vector& out); + +} // namespace ft diff --git a/src/extract/cfbf.cpp b/src/extract/cfbf.cpp new file mode 100644 index 0000000..2fd3e0c --- /dev/null +++ b/src/extract/cfbf.cpp @@ -0,0 +1,111 @@ +// cfbf.cpp — compound file (CFBF) extractor. See the header. +#include "extract/cfbf.hpp" + +#include +#include +#include + +#include "cfbf_parse.hpp" +#include "extract/safepath.hpp" + +namespace ft { + +namespace { + +// Compound-file names are arbitrary UTF-16; MSI table names decode to things +// like "_StringPool", and the summary stream starts with a control character. +// Keep them recognizable but never path-active. +std::string safe_name(const std::string& name) { + std::string out; + for (char ch : name) { + const unsigned char u = static_cast(ch); + if (u < 0x20 || ch == '/' || ch == '\\' || ch == ':' || ch == '*' || ch == '?' || + ch == '"' || ch == '<' || ch == '>' || ch == '|') + out += '_'; + else + out += ch; + } + while (!out.empty() && (out.back() == '.' || out.back() == ' ')) out.pop_back(); + if (out.empty() || out == "." || out == "..") out = "_unnamed"; + return out; +} + +// Walk the directory tree, recording each entry's path. Siblings form a +// red-black tree inside their storage; a storage's `child` heads the tree of +// what it holds. `seen` breaks the cycles a corrupt directory can describe. +void walk_tree(const Cfbf& c, uint32_t idx, const std::string& prefix, + std::vector& paths, std::vector& seen) { + if (idx == kCfbfNoStream || idx >= c.entries.size() || seen[idx]) return; + seen[idx] = 1; + const CfbfEntry& e = c.entries[idx]; + + walk_tree(c, e.left, prefix, paths, seen); + + const std::string name = safe_name(e.name); + const std::string full = prefix.empty() ? name : prefix + "/" + name; + if (e.type == kCfbfStream) paths[idx] = full; + if (e.type == kCfbfStorage) walk_tree(c, e.child, full, paths, seen); + + walk_tree(c, e.right, prefix, paths, seen); +} + +} // namespace + +bool extract_cfbf(const Reader& r, const Finding& f, SafeRoot& root, const std::string& subdir, + Extracted& out) { + out.offset = f.offset; + out.type = "cfbf"; + out.root = subdir; + + Cfbf c; + if (!cfbf_parse(r, f.offset, c)) { + out.status = "error:header"; + return true; + } + if (!root.make_dir(subdir)) { + out.status = "error:mkdir"; + return true; + } + out.consumed = c.span; + + // Resolve every stream's path through the storage tree. An entry the tree + // does not reach still gets written (flat, under its own name) rather than + // dropped: a damaged directory should cost you the hierarchy, not the data. + std::vector paths(c.entries.size()); + std::vector seen(c.entries.size(), 0); + walk_tree(c, c.entries[0].child, "", paths, seen); + + std::unordered_map used; + for (size_t i = 0; i < c.entries.size(); ++i) { + const CfbfEntry& e = c.entries[i]; + if (e.type != kCfbfStream || e.size == 0) continue; + + std::vector data; + if (!cfbf_stream(r, f.offset, c, e, data)) { + out.warnings.push_back(safe_name(e.name) + ": broken sector chain"); + continue; + } + + std::string rel = paths[i].empty() ? safe_name(e.name) : paths[i]; + const int n = used[rel]++; + if (n) rel += "." + std::to_string(n); // same name in two storages + + if (!root.write_file(subdir + "/" + rel, data, 0644)) { + out.status = "error:write"; + return true; + } + out.files++; + out.bytes += data.size(); + } + + if (out.files == 0) { + if (out.status.empty()) out.status = "error:empty"; + } else if (!out.warnings.empty()) { + out.status = "partial"; + } else { + out.status = "ok"; + } + return true; +} + +} // namespace ft diff --git a/src/extract/cfbf.hpp b/src/extract/cfbf.hpp new file mode 100644 index 0000000..5af132b --- /dev/null +++ b/src/extract/cfbf.hpp @@ -0,0 +1,27 @@ +// cfbf.hpp — Compound File Binary Format (MSI, .doc, .msg) extractor. +// +// Writes each stream out under the storage path that holds it, so a Windows +// Installer package yields its tables plus the payload cabinet as real files. +// MSI-encoded stream names are decoded, which is what turns an unreadable CJK +// name back into "Data1.cab". +// +// The point of extracting at all is that a compound-file stream is not +// contiguous. Recursion then re-scans what lands here, so the cabinet inside an +// installer is read from its reassembled bytes rather than from a linear walk +// off the end of its first sector run. +#pragma once + +#include + +#include "extract/manifest.hpp" +#include "finding.hpp" +#include "reader.hpp" + +namespace ft { + +class SafeRoot; + +bool extract_cfbf(const Reader& r, const Finding& f, SafeRoot& root, const std::string& subdir, + Extracted& out); + +} // namespace ft diff --git a/src/extract/lzx.cpp b/src/extract/lzx.cpp index 097e2f6..b655c6c 100644 --- a/src/extract/lzx.cpp +++ b/src/extract/lzx.cpp @@ -102,6 +102,15 @@ class BitReader { left_ = 0; } + // Discard the bits left in the current 16-bit word. CAB pads the bitstream + // to a word boundary at the end of every 32 KiB output frame, so the next + // frame starts on one; without this the reader stays half a word ahead for + // the rest of the stream. + void align_word() { + if (left_ > 0) ensure(16); + if (left_ & 15) remove(static_cast(left_ & 15)); + } + size_t pos() const { return pos_; } void set_pos(size_t p) { pos_ = p; } size_t size() const { return in_.size(); } @@ -251,6 +260,10 @@ class LzxDecoder { bool intel_started() const { return intel_started_; } int32_t intel_filesize() const { return intel_filesize_; } + // Consume the pad bits that end a CAB output frame. Not used by the CE ROM + // framing, whose blocks are each a self-contained stream. + void align_frame() { bits_.align_word(); } + std::vector out; private: @@ -589,18 +602,32 @@ std::optional> lzx_decompress_cab(std::span LzxDecoder d(src, window_bits, out_len + kMaxMatch); if (!d.read_header()) return std::nullopt; - // Output is produced in 32 KiB frames; the x86 translation is per frame, - // using the frame's absolute position, and stops after 32768 frames. - size_t done = 0; - for (size_t frame = 0; done < out_len; ++frame) { + // Output is produced in 32 KiB frames. Two rules apply at every frame + // boundary: the bitstream is padded to the next 16-bit word, and the x86 + // translation runs over that frame's bytes using their absolute position + // (for the first 32768 frames only). + // + // The translation writes to a copy, never to the decoder's buffer: that + // buffer doubles as the match window, and later matches must see the + // untranslated bytes. + std::vector out; + out.reserve(out_len); + for (size_t frame = 0; out.size() < out_len; ++frame) { + const size_t done = out.size(); const size_t want = std::min(kCabFrameSize, out_len - done); if (!d.decode_until(done + want)) return std::nullopt; + out.insert(out.end(), d.out.begin() + static_cast(done), + d.out.begin() + static_cast(done + want)); if (d.intel_started() && d.intel_filesize() != 0 && frame < 32768) - undo_e8(d.out, done, want, done, d.intel_filesize()); - done += want; + undo_e8(out, done, want, done, d.intel_filesize()); + if (out.size() < out_len) { + // A match that ran past the boundary would leave the pad bits at an + // unpredictable position; a conformant stream never emits one. + if (d.out.size() != done + want) return std::nullopt; + d.align_frame(); + } } - d.out.resize(out_len); - return std::move(d.out); + return out; } } // namespace ft diff --git a/src/extract/manifest.cpp b/src/extract/manifest.cpp index 66b28bb..cefab15 100644 --- a/src/extract/manifest.cpp +++ b/src/extract/manifest.cpp @@ -8,6 +8,7 @@ #include "extract/compressed.hpp" #include "extract/descramble.hpp" #include "extract/cab.hpp" +#include "extract/cfbf.hpp" #include "extract/cpio.hpp" #include "extract/cramfs.hpp" #include "extract/erofs.hpp" @@ -47,6 +48,7 @@ Extractor find_extractor(const std::string& type) { if (is_scheme(type)) return extract_descramble; // vendor-encrypted -> descramble + recurse if (type == "squashfs") return extract_squashfs; if (type == "cab") return extract_cab; + if (type == "cfbf") return extract_cfbf; if (type == "cpio") return extract_cpio; if (type == "ext") return extract_ext; if (type == "jffs2") return extract_jffs2; @@ -148,6 +150,9 @@ std::string manifest_to_json(const Manifest& m) { esc(o, m.cap_reason); o += "\""; } + + if (m.skipped_nested_executables) + o += ",\"skipped_nested_executables\":" + std::to_string(m.skipped_nested_executables); o += "}"; return o; } diff --git a/src/extract/manifest.hpp b/src/extract/manifest.hpp index e81397d..ac16c1a 100644 --- a/src/extract/manifest.hpp +++ b/src/extract/manifest.hpp @@ -41,6 +41,7 @@ struct Manifest { std::string source; // source file path bool capped = false; // a recursion guard (depth/files/bytes/ratio) tripped std::string cap_reason; // which guard, when capped + size_t skipped_nested_executables = 0; // Installer/document containers found below the top level and left packed }; // One extractor: pull `f` out of `r` into `root`, describing results in `out`. diff --git a/src/main.cpp b/src/main.cpp index 1ee4088..b302c14 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -129,6 +129,9 @@ void print_help(std::FILE* out, const char* prog, bool color) { " --list List archive contents without extracting\n" " -C, --outdir Output directory for -e / -c\n" " --depth Max extraction recursion depth (default: 8)\n" + " --unpack-executables\n" + " Also unpack installer/document containers found\n" + " inside other extracted output\n" " --max-files Stop extraction after N files (default: 500000)\n" " --max-bytes Stop extraction after N bytes (default: 4 GiB)\n" " --sigs Load signatures from DIR\n" @@ -158,11 +161,39 @@ struct RecurCtx { size_t max_files; uint64_t max_bytes; bool all = false; // -A/--all: don't suppress interior compressed streams + bool unpack_exe = false; // --unpack-executables: unpack nested installer/document containers // running totals size_t total_files = 0; uint64_t total_bytes = 0; + size_t skipped_nested_executables = 0; // containers left packed by the default policy }; +// Formats that carry a whole application's worth of members and are rarely what +// an analyst is after once they are nested inside something else. An MSI alone +// unpacks to a few hundred database-table streams; two cabinets with an MSI +// apiece bury the findings that matter. So these are unpacked only in the file +// the user pointed at — where they ARE the subject — or on explicit opt-in. +bool nested_opt_in(const std::string& type) { return type == "cfbf"; } + +// Does this produced file start like an executable? Used to decide whether +// extraction may dig *inside* it. +// +// Only the leading magic is read -- no header is followed (an MZ is not chased +// through e_lfanew to its PE signature). That is deliberate: this is a cheap +// carrier test, not an identification. The scan already ran over this file, and +// the cost of a false positive here is a container left unopened behind +// --unpack-executables, not a wrong answer. +bool has_executable_magic(const ft::Reader& r) { + auto b = r.bytes(0, 4); + if (!b) return false; + const uint8_t* p = b->data(); + if (p[0] == 'M' && p[1] == 'Z') return true; // PE / DOS + if (p[0] == 0x7F && p[1] == 'E' && p[2] == 'L' && p[3] == 'F') return true; // ELF + const uint32_t m = uint32_t(p[0]) | (uint32_t(p[1]) << 8) | (uint32_t(p[2]) << 16) | + (uint32_t(p[3]) << 24); + return m == 0xFEEDFACE || m == 0xFEEDFACF || m == 0xCEFAEDFE || m == 0xCFFAEDFE; // Mach-O +} + constexpr uint64_t RATIO_FLOOR = 10u << 20; // only ratio-check outputs above 10 MB constexpr uint64_t RATIO_LIMIT = 1000; // output/input > this = likely bomb constexpr uint64_t MIN_DESCEND = 64; // never descend into a file smaller than this @@ -244,6 +275,11 @@ void descend(const std::string& base_sub, size_t level, RecurCtx& c) { if (ft::find_extractor(f.type) && f.confidence >= static_cast(ft::Confidence::Structural)) keep.push_back(f); if (keep.empty()) continue; + // Skip executables to avoid expansion unless requested + if (!c.unpack_exe && has_executable_magic(r)) { + c.skipped_nested_executables++; + continue; + } extract_findings(r, keep, rel + ".extracted", level, c); } } @@ -287,6 +323,10 @@ void extract_findings(ft::Reader& reader, const std::vector& findin if (f.offset >= exfat_cov) continue; exfat_cov = f.offset; } + if (level > 1 && !c.unpack_exe && nested_opt_in(f.type)) { + c.skipped_nested_executables++; + continue; + } if (c.total_files >= c.max_files) { mark_capped(c, "max-files"); break; } if (c.total_bytes >= c.max_bytes) { mark_capped(c, "max-bytes"); break; } @@ -357,7 +397,7 @@ std::string run_extraction(const std::string& src_path, ft::Reader& reader, const std::vector& findings, const std::vector& sigs, const std::string& outdir, size_t max_depth, size_t max_files, uint64_t max_bytes, bool all, - ft::Manifest& manifest) { + bool unpack_exe, ft::Manifest& manifest) { manifest.source = src_path; ft::SafeRoot root; @@ -365,8 +405,9 @@ std::string run_extraction(const std::string& src_path, ft::Reader& reader, std::fprintf(stderr, "error: cannot create output dir: %s\n", outdir.c_str()); return ""; } - RecurCtx c{root, sigs, manifest, max_depth, max_files, max_bytes, all}; + RecurCtx c{root, sigs, manifest, max_depth, max_files, max_bytes, all, unpack_exe}; extract_findings(reader, findings, "", 1, c); + manifest.skipped_nested_executables = c.skipped_nested_executables; if (manifest.entries.empty()) return ""; std::string json = ft::manifest_to_json(manifest); @@ -381,12 +422,13 @@ int main(int argc, char** argv) { const auto t_start = std::chrono::steady_clock::now(); const char* prog = basename_of(argv[0]); std::string sig_dir_cli, path; - unsigned threads = 0; // 0 -> auto + unsigned threads = 0; // 0 -> auto bool broad = false; - bool json_out = false; // default is the human-readable view - bool show_all = false; // -A: don't collapse compressed-stream swarms - bool verbose = false; // -v: show full labels (no ellipsis truncation) - bool entropy = false; // -E: entropy analysis (unidentified regions + hints) + bool json_out = false; // default is the human-readable view + bool show_all = false; // -A: don't collapse compressed-stream swarms + bool unpack_exe = false; // --unpack-executables: unpack nested installer containers + bool verbose = false; // -v: show full labels (no ellipsis truncation) + bool entropy = false; // -E: entropy analysis (unidentified regions + hints) bool list = false; bool extract = false; bool carve = false; @@ -424,6 +466,8 @@ int main(int argc, char** argv) { broad = true; } else if (!end_of_opts && (std::strcmp(a, "--json") == 0 || std::strcmp(a, "-j") == 0)) { json_out = true; + } else if (!end_of_opts && std::strcmp(a, "--unpack-executables") == 0) { + unpack_exe = true; } else if (!end_of_opts && (std::strcmp(a, "--all") == 0 || std::strcmp(a, "-A") == 0)) { show_all = true; } else if (!end_of_opts && (std::strcmp(a, "--verbose") == 0 || std::strcmp(a, "-v") == 0)) { @@ -606,7 +650,7 @@ int main(int argc, char** argv) { if (extract) { outdir = outdir_cli.empty() ? path + ".extracted" : outdir_cli; extraction = run_extraction(path, reader, findings, sigs.signatures, outdir, rec_depth, - rec_max_files, rec_max_bytes, show_all, manifest); + rec_max_files, rec_max_bytes, show_all, unpack_exe, manifest); } // Carve (`-c`): dump each finding's raw byte range (and the unidentified gaps) // to disk without parsing, so a researcher gets the bytes even when extraction @@ -638,6 +682,12 @@ int main(int argc, char** argv) { char buf[512]; std::string footer; if (!extraction.empty()) footer += "-> extracted to " + outdir + "/\n"; + if (manifest.skipped_nested_executables > 0) + footer += "-> " + std::to_string(manifest.skipped_nested_executables) + + (manifest.skipped_nested_executables == 1 + ? " nested executable/installer container" + : " nested executables/installer containers") + + " identified but not unpacked (--unpack-executables)\n"; if (carve && cr.regions_written > 0) { footer += "-> carved " + std::to_string(cr.regions_written) + (cr.regions_written == 1 ? " region (" : " regions (") + diff --git a/src/validators/cfbf.cpp b/src/validators/cfbf.cpp new file mode 100644 index 0000000..fe5f52d --- /dev/null +++ b/src/validators/cfbf.cpp @@ -0,0 +1,34 @@ +// cfbf.cpp — compound file (CFBF) validator. See the header. +#include "validators/cfbf.hpp" + +#include + +#include "cfbf_parse.hpp" + +namespace ft { + +bool validate_cfbf(ValidatorCtx& ctx) { + Cfbf c; + if (!cfbf_parse(ctx.reader, ctx.offset, c)) return false; + if (c.streams == 0) return false; // a compound file with no stream is noise + + ctx.out.size = static_cast(c.span); + + char ver[16]; + std::snprintf(ver, sizeof(ver), "%u.%u", c.ver_major, c.ver_minor); + ctx.out.version = ver; + + char label[96]; + std::snprintf(label, sizeof(label), "%s%zu storages, %zu streams", + c.msi ? "msi, " : "", c.storages, c.streams); + ctx.out.label = label; + + const uint64_t avail = ctx.reader.size() - ctx.offset; + if (c.span <= avail) + ctx.out.set_confidence(Confidence::Verified, "FAT + directory parse, span within EOF"); + else + ctx.out.set_confidence(Confidence::Consistent, "FAT + directory parse"); + return true; +} + +} // namespace ft diff --git a/src/validators/cfbf.hpp b/src/validators/cfbf.hpp new file mode 100644 index 0000000..ff606d8 --- /dev/null +++ b/src/validators/cfbf.hpp @@ -0,0 +1,11 @@ +#pragma once +#include "signature.hpp" + +namespace ft { +// Validate a compound file (CFBF): build the FAT from the DIFAT, walk the +// directory, and size the finding by the last sector the FAT accounts for. +// Names the flavour (MSI when the stream names use MSI's encoding) and counts +// storages/streams. A compound file whose span fits the bytes available and +// whose directory parses is `verified`. +bool validate_cfbf(ValidatorCtx& ctx); +} // namespace ft diff --git a/src/validators/registry.cpp b/src/validators/registry.cpp index 296071c..194bf8c 100644 --- a/src/validators/registry.cpp +++ b/src/validators/registry.cpp @@ -2,6 +2,7 @@ #include "validators/android_boot.hpp" #include "validators/cab.hpp" +#include "validators/cfbf.hpp" #include "validators/cpio.hpp" #include "validators/deobf.hpp" #include "validators/dtb.hpp" @@ -42,6 +43,7 @@ Validator find_validator(const std::string& name) { if (name == "uimage") return validate_uimage; if (name == "elf") return validate_elf; if (name == "cab") return validate_cab; + if (name == "cfbf") return validate_cfbf; if (name == "cpio") return validate_cpio; if (name == "dtb") return validate_dtb; if (name == "ihex") return validate_ihex; diff --git a/tests/cfbfbuild.py b/tests/cfbfbuild.py new file mode 100644 index 0000000..e3b8391 --- /dev/null +++ b/tests/cfbfbuild.py @@ -0,0 +1,162 @@ +#!/usr/bin/env python3 +"""Minimal compound-file (CFBF) writer, shared by the CFBF regression tests. + +The point of the format, for moria, is that a stream is *not* contiguous: its +sectors are chained through a FAT and interleaved with everything else. So this +writer deliberately round-robins sectors between streams. A reader that walks +linearly from a stream's first sector gets the right bytes for exactly one +sector and garbage after it, which is the bug this fixture exists to catch. +""" +import struct + +SECT = 512 +MINI = 64 +MINI_CUTOFF = 4096 +FREE, ENDOFCHAIN, FATSECT = 0xFFFFFFFF, 0xFFFFFFFE, 0xFFFFFFFD +NOSTREAM = 0xFFFFFFFF +ALPHABET = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz._" + + +def msi_mangle(name): + """Encode a name the way MSI encodes its stream names (two chars per unit).""" + units, i = [], 0 + while i < len(name): + if i + 1 < len(name): + units.append(0x3800 + ALPHABET.index(name[i]) + (ALPHABET.index(name[i + 1]) << 6)) + i += 2 + else: + units.append(0x4800 + ALPHABET.index(name[i])) + i += 1 + return "".join(chr(u) for u in units) + + +def _chunks(data, size): + return [data[i:i + size].ljust(size, b"\0") for i in range(0, len(data), size)] or [] + + +def build(streams, interleave=True): + """streams: list of (name, bytes). Returns the compound file.""" + big = [(n, d) for n, d in streams if len(d) >= MINI_CUTOFF] + small = [(n, d) for n, d in streams if len(d) < MINI_CUTOFF] + + # The mini stream concatenates every small stream, 64 bytes per mini sector. + mini_data, mini_index = b"", {} + for n, d in small: + mini_index[n] = len(mini_data) // MINI + mini_data += b"".join(_chunks(d, MINI)) + + n_dir_entries = 1 + len(streams) + dir_sectors = (n_dir_entries * 128 + SECT - 1) // SECT + mini_sectors = len(mini_data) // MINI + minifat_sectors = ((mini_sectors * 4) + SECT - 1) // SECT if mini_sectors else 0 + ministream_sectors = (len(mini_data) + SECT - 1) // SECT + + # Every chain that needs real sectors, in allocation order. + chains = [("dir", dir_sectors)] + if minifat_sectors: + chains.append(("minifat", minifat_sectors)) + if ministream_sectors: + chains.append(("ministream", ministream_sectors)) + for n, d in big: + chains.append((("big", n), (len(d) + SECT - 1) // SECT)) + + total_payload = sum(c for _, c in chains) + nfat = 1 + while nfat * (SECT // 4) < total_payload + nfat: + nfat += 1 + + # FAT sectors first, then the payload sectors -- round-robin across chains so + # no stream is contiguous. + alloc = {k: [] for k, _ in chains} + nxt = nfat + if interleave: + remaining = {k: c for k, c in chains} + while any(remaining.values()): + for k, _ in chains: + if remaining[k]: + alloc[k].append(nxt) + nxt += 1 + remaining[k] -= 1 + else: + for k, c in chains: + for _ in range(c): + alloc[k].append(nxt) + nxt += 1 + + total_sectors = nxt + fat = [FREE] * (nfat * (SECT // 4)) + for i in range(nfat): + fat[i] = FATSECT + for k, _ in chains: + s = alloc[k] + for a, b in zip(s, s[1:]): + fat[a] = b + fat[s[-1]] = ENDOFCHAIN + + # Directory entries. + def entry(name, typ, start, size, left=NOSTREAM, right=NOSTREAM, child=NOSTREAM): + e = bytearray(128) + u = name.encode("utf-16-le") + b"\0\0" + e[:len(u)] = u + struct.pack_into("= MINI_CUTOFF: + start = alloc[("big", n)][0] + else: + start = mini_index[n] + entries.append(entry(n, 2, start, len(d), right=nxt_sib)) + dir_bytes = b"".join(entries).ljust(dir_sectors * SECT, b"\0") + + minifat = [] + for n, d in small: + base = mini_index[n] + cnt = (len(d) + MINI - 1) // MINI + for k in range(cnt - 1): + minifat.append(base + k + 1) + minifat.append(ENDOFCHAIN) + minifat_bytes = struct.pack("<%dI" % len(minifat), *minifat) if minifat else b"" + minifat_bytes = minifat_bytes.ljust(minifat_sectors * SECT, b"\xff") + + # Lay the sectors down. + body = bytearray(b"\0" * (total_sectors * SECT)) + + def put(sector_list, blob): + for i, s in enumerate(sector_list): + body[s * SECT:(s + 1) * SECT] = blob[i * SECT:(i + 1) * SECT].ljust(SECT, b"\0") + + put(alloc["dir"], dir_bytes) + if minifat_sectors: + put(alloc["minifat"], minifat_bytes) + if ministream_sectors: + put(alloc["ministream"], mini_data.ljust(ministream_sectors * SECT, b"\0")) + for n, d in big: + put(alloc[("big", n)], d) + fat_bytes = struct.pack("<%dI" % len(fat), *fat) + for i in range(nfat): + body[i * SECT:(i + 1) * SECT] = fat_bytes[i * SECT:(i + 1) * SECT] + + hdr = bytearray(b"\0" * SECT) + hdr[0:8] = bytes([0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1]) + struct.pack_into("0}.""" + used = sorted(freqs.items(), key=lambda kv: (kv[1], kv[0])) + n = len(used) + if n == 0: + return {} + if n == 1: + return {used[0][0]: 1} + coins = [(w, (s,)) for s, w in used] + coins.sort(key=lambda t: t[0]) + result = list(coins) + for _ in range(limit - 1): + packaged = [(result[i][0] + result[i + 1][0], result[i][1] + result[i + 1][1]) + for i in range(0, len(result) - 1, 2)] + result = sorted(coins + packaged, key=lambda t: t[0]) + lengths = {s: 0 for s, _ in used} + for _, syms in result[: 2 * n - 2]: + for s in syms: + lengths[s] += 1 + return lengths + + +def code_lengths(freqs, nsyms, limit=16): + """Code lengths over [0, nsyms), always a *complete* code (LZX requires it).""" + lengths = [0] * nsyms + got = package_merge({s: w for s, w in freqs.items() if w > 0}, limit) + for s, ln in got.items(): + lengths[s] = ln + used = [s for s in range(nsyms) if lengths[s]] + if len(used) == 1: + # A lone 1-bit code is an incomplete table; pair it with a filler symbol. + filler = next(s for s in range(nsyms) if s != used[0]) + lengths[used[0]] = lengths[filler] = 1 + return lengths + + +def canonical(lengths): + codes, code = {}, 0 + for bl in range(1, 17): + for sym in (s for s, ln in enumerate(lengths) if ln == bl): + codes[sym] = code + code += 1 + code <<= 1 + return codes + + +def write_lengths(w, lengths, prev, first, last): + """Emit the pretree and the mod-17 length deltas for lengths[first:last].""" + deltas = [(prev[x] - lengths[x]) % 17 for x in range(first, last)] + freqs = {} + for d in deltas: + freqs[d] = freqs.get(d, 0) + 1 + pre_len = code_lengths(freqs, 20, limit=15) # the length field is 4 bits + pre_code = canonical(pre_len) + for x in range(20): + w.put(pre_len[x], 4) + for d in deltas: + w.put(pre_code[d], pre_len[d]) + + +def apply_e8_forward(data, filesize, frame=LZX_FRAME): + """The encoder-side x86 filter: the exact inverse of the decoder's pass.""" + out = bytearray(data) + for base in range(0, len(out), frame): + flen = min(frame, len(out) - base) + if flen <= 10: + continue + i, curpos, limit = 0, base, flen - 10 + while i < limit: + if out[base + i] != 0xE8: + i += 1 + curpos += 1 + continue + off = base + i + 1 + rel = struct.unpack(" window: + continue + n = 0 + cap = min(MAX_MATCH, frame_end - pos) + while n < cap and data[cand + n] == data[pos + n]: + n += 1 + if n > best_len: + best_len, best_off = n, off + if n == cap: + break + if best_len >= MIN_MATCH + 1: + tokens.append((best_off, best_len)) + step = best_len + else: + tokens.append((0, data[pos])) + step = 1 + for k in range(pos, min(pos + step, end - 2)): + index.setdefault(data[k:k + 3], []).append(k) + pos += step + return tokens + + +def _emit_block(w, data, start, end, window_bits, aligned, prev_main, prev_len, on_frame): + """Emit one block. `on_frame(pos)` runs each time output crosses a frame edge.""" + main_elements = NUM_CHARS + POSITION_SLOTS[window_bits] * 8 + window = 1 << window_bits + tokens = _tokenize(data, start, end, window) + + # Resolve each token into its symbols, then count them to build the trees. + plan, main_f, len_f, align_f = [], {}, {}, {} + r0 = r1 = r2 = 1 + for off, val in tokens: + if off == 0: + plan.append(("lit", val, 1)) + main_f[val] = main_f.get(val, 0) + 1 + continue + mlen = val - MIN_MATCH + header = min(mlen, NUM_PRIMARY_LENGTHS) + footer = mlen - NUM_PRIMARY_LENGTHS if header == NUM_PRIMARY_LENGTHS else None + if off == r0: + slot, extra_val = 0, None + else: + fo = off + 2 + slot = max(s for s in range(3, len(POSITION_BASE)) if POSITION_BASE[s] <= fo) + extra_val = fo - POSITION_BASE[slot] + r2, r1, r0 = r1, r0, off + sym = NUM_CHARS + (slot << 3) + header + main_f[sym] = main_f.get(sym, 0) + 1 + if footer is not None: + len_f[footer] = len_f.get(footer, 0) + 1 + if extra_val is not None and aligned and EXTRA_BITS[slot] >= 3: + align_f[extra_val & 7] = align_f.get(extra_val & 7, 0) + 1 + plan.append(("match", sym, footer, slot, extra_val, val)) + + main_len = code_lengths(main_f, main_elements) + lens_len = code_lengths(len_f, 249) + main_code, lens_code = canonical(main_len), canonical(lens_len) + align_len = [3] * 8 + align_code = canonical(align_len) + + w.put(BLOCK_ALIGNED if aligned else BLOCK_VERBATIM, 3) + w.put(end - start, 24) + if aligned: + for x in range(8): + w.put(align_len[x], 3) + write_lengths(w, main_len, prev_main, 0, NUM_CHARS) + write_lengths(w, main_len, prev_main, NUM_CHARS, main_elements) + write_lengths(w, lens_len, prev_len, 0, 249) + + pos = start + for item in plan: + if item[0] == "lit": + s = item[1] + w.put(main_code[s], main_len[s]) + else: + _, sym, footer, slot, extra_val, _n = item + w.put(main_code[sym], main_len[sym]) + if footer is not None: + w.put(lens_code[footer], lens_len[footer]) + if extra_val is not None: + nx = EXTRA_BITS[slot] + if aligned and nx >= 3: + w.put(extra_val >> 3, nx - 3) + w.put(align_code[extra_val & 7], 3) + elif nx: + w.put(extra_val, nx) + pos += item[2] if item[0] == "lit" else item[5] + if pos % LZX_FRAME == 0: + on_frame(pos) + return main_len, lens_len + + +def lzx_compress(data, window_bits=21, intel_filesize=0, block_ends=None, + aligned_blocks=()): + """Compress `data` into per-32-KiB-frame chunks, CAB-style. + + The bitstream is padded to a 16-bit boundary at the end of every output + frame, which is what lets a CAB decoder restart on each CFDATA block. + """ + if intel_filesize: + data = apply_e8_forward(data, intel_filesize) + if block_ends is None: + block_ends = [len(data)] + assert block_ends and block_ends[-1] == len(data) + + w = BitWriter() + w.put(1 if intel_filesize else 0, 1) + if intel_filesize: + w.put((intel_filesize >> 16) & 0xFFFF, 16) + w.put(intel_filesize & 0xFFFF, 16) + + marks = [] + + def on_frame(pos): + if pos < len(data): + w.align_word() + marks.append(len(w.out)) + + prev_main = [0] * (NUM_CHARS + POSITION_SLOTS[window_bits] * 8) + prev_len = [0] * 249 + start = 0 + for bi, end in enumerate(block_ends): + prev_main, prev_len = _emit_block(w, data, start, end, window_bits, + bi in aligned_blocks, prev_main, + prev_len, on_frame) + start = end + w.align_word() + marks.append(len(w.out)) + + blob = bytes(w.out) + frames, prev = [], 0 + for m in marks: + frames.append(blob[prev:m]) + prev = m + return frames diff --git a/tests/run.sh b/tests/run.sh index 7ab1e5d..b89a1e5 100755 --- a/tests/run.sh +++ b/tests/run.sh @@ -43,6 +43,8 @@ python3 tests/test_wince_rom.py python3 tests/test_cab.py # Windows CE registry hive: value-record recovery and its false-positive guard. python3 tests/test_wince_hive.py +# CFBF/MSI compound files: FAT-chained (non-contiguous) stream reassembly. +python3 tests/test_cfbf.py # VBF block extraction + LZSS decode round-trip (self-contained hard gate). python3 tests/test_vbf.py echo diff --git a/tests/test_cab.py b/tests/test_cab.py index 42c8e7a..ce6522a 100644 --- a/tests/test_cab.py +++ b/tests/test_cab.py @@ -16,6 +16,7 @@ """ import json import os +import random import struct import subprocess import sys @@ -26,7 +27,7 @@ sys.path.insert(0, HERE) MORIA = os.path.join(HERE, "..", "build", "moria") -from lzxbuild import lzx_stored_stream # noqa: E402 +from lzxbuild import lzx_compress, lzx_stored_stream # noqa: E402 COMP_NONE, COMP_MSZIP, COMP_LZX = 0, 1, 3 FRAME = 32768 @@ -112,6 +113,39 @@ def lzx_blocks(data, nsplit=4): return list(zip(parts, frames)) +def lzx_real_payload(n=110000): + """Bytes that exercise the whole compressed-LZX path. + + Mixes long back-references (matches, including ones reaching across a + 32 KiB frame boundary), incompressible runs (literals, so the main tree is + genuinely Huffman-coded) and x86 CALL sites (so the E8 filter runs). + """ + rnd = random.Random(1234) + buf = bytearray() + motif = bytes(rnd.randrange(256) for _ in range(900)) + while len(buf) < n: + r = rnd.random() + if r < 0.32: + buf += motif[:rnd.randrange(40, 900)] + elif r < 0.55: + buf += b"\xe8" + struct.pack(" 40000: + src = len(buf) - rnd.randrange(1, min(len(buf), 60000)) + buf += bytes(buf[src:src + rnd.randrange(8, 200)]) or b"\0" + else: + buf += bytes(rnd.randrange(256) for _ in range(rnd.randrange(1, 30))) + return bytes(buf[:n]) + + +def lzx_compressed_blocks(data, window_bits=21): + """A real compressed LZX folder: one CFDATA block per 32 KiB output frame.""" + frames = lzx_compress(data, window_bits=window_bits, intel_filesize=len(data), + block_ends=[50000, 88000, len(data)], aligned_blocks=(1,)) + ulens = [len(data[i:i + FRAME]) for i in range(0, len(data), FRAME)] + assert len(frames) == len(ulens), (len(frames), len(ulens)) + return list(zip(frames, ulens)) + + SETUP_XML = """ @@ -232,7 +266,20 @@ def main(): if rc: return rc - # 4. A Windows CE installer cabinet, rebuilt from its _setup.xml. + # 4. A real *compressed* LZX folder: Huffman-coded verbatim and aligned + # blocks, matches, and the x86 filter, spanning four 32 KiB frames + # with block boundaries deliberately off the frame boundaries. This + # is the path every real-world LZX cabinet takes. + real = lzx_real_payload() + rc = check_roundtrip( + tmp, "lzx-real.cab", + folders=[(COMP_LZX | (21 << 8), lzx_compressed_blocks(real))], + files=[("real.bin", 0, 0, len(real))], + payloads={"real.bin": real}, want_comp="lzx") + if rc: + return rc + + # 5. A Windows CE installer cabinet, rebuilt from its _setup.xml. members = [("_setup.xml", SETUP_XML), ("AGENT~1.001", AGENT), ("HOSTKE~1.002", HOSTKEY), ("SETUP.999", SETUP_DLL), ("HEADER.000", INSTALL_HDR), ("EXTRA~1.500", b"unlisted member")] @@ -269,7 +316,8 @@ def main(): if line not in reg: return fail("CE cab: registry.reg missing %r:\n%s" % (line, reg)) - print("ok: cab identify + stored/MSZIP/LZX extraction + CE install-tree rebuild") + print("ok: cab identify + stored/MSZIP/LZX (stored and compressed) extraction" + " + CE install-tree rebuild") return 0 diff --git a/tests/test_cfbf.py b/tests/test_cfbf.py new file mode 100644 index 0000000..26c8ce0 --- /dev/null +++ b/tests/test_cfbf.py @@ -0,0 +1,258 @@ +#!/usr/bin/env python3 +"""Compound file (CFBF/MSI) identify + extract regression. + +A compound file stores each stream as a chain of sectors interleaved with every +other stream, so a stream's bytes are not contiguous. That is the whole reason +this handler exists: an installer keeps its payload cabinet in a stream, and +reading it linearly from the MSCF magic yields one correct sector run followed +by another stream's bytes. The fixture here is built deliberately fragmented, +and asserts up front that a linear read does NOT reproduce the stream -- so the +test cannot quietly pass against a reader that ignores the FAT. + +Checks: + * the compound file is identified as cfbf at `verified` tier, flagged msi, + * MSI-encoded stream names are decoded ("Data1.cab", not CJK mojibake), + * a large fragmented stream comes back byte-for-byte, + * a small stream (under the 4096-byte cutoff, so it lives in the mini stream + addressed by the mini-FAT) comes back byte-for-byte, + * recursion reads the cabinets inside -- both a compressed (LZX) and a stored + one. + * a compound file nested inside other extracted output is identified but left + packed by default, and unpacked on --unpack-executables. + +Self-contained; no external tools. Exit nonzero on any failure. +""" +import json +import os +import random +import struct +import subprocess +import sys +import tempfile + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, HERE) +MORIA = os.path.join(HERE, "..", "build", "moria") + +import cfbfbuild # noqa: E402 +from test_cab import (build_cab, lzx_compressed_blocks, lzx_real_payload, # noqa: E402 + stored_blocks) + +COMP_NONE, COMP_LZX = 0, 3 + + +def run(path, outdir=None, extra=()): + cmd = [MORIA, "-e", "-j", path, "-C", outdir, *extra] if outdir else [MORIA, "-j", path] + r = subprocess.run(cmd, capture_output=True, timeout=300) + if r.returncode != 0: + raise SystemExit("moria exited %d: %s" % (r.returncode, r.stderr.decode("replace"))) + return json.loads(r.stdout) + + +def fail(msg): + print("FAIL:", msg) + return 1 + + +def find_file(root, name): + for dirpath, _, files in os.walk(root): + if name in files: + return os.path.join(dirpath, name) + return None + + +def main(): + rnd = random.Random(99) + member = lzx_real_payload(90000) + cab = build_cab([(COMP_LZX | (21 << 8), lzx_compressed_blocks(member))], + [("payload.bin", 0, 0, len(member))]) + # A stored cabinet too: its member sits verbatim in the stream, so sectors + # reassembled in the wrong order surface as wrong bytes instead of a codec + # error. Non-repeating content, so a swapped sector cannot go unnoticed. + stored_member = bytes(rnd.randrange(256) for _ in range(30000)) + stored_cab = build_cab([(COMP_NONE, stored_blocks(stored_member))], + [("stored.bin", 0, 0, len(stored_member))]) + big = bytes(rnd.randrange(256) for _ in range(40000)) + small = b"summary stream, under the mini-stream cutoff\n" * 20 + + blob = cfbfbuild.build([ + (cfbfbuild.msi_mangle("Data1.cab"), cab), # compressed, MSI-encoded name + (cfbfbuild.msi_mangle("Data2.cab"), stored_cab), # stored + ("BigStream", big), + ("_SummaryInformation", small), + ]) + + # The fixture must actually be fragmented, else it proves nothing: neither + # cabinet may appear contiguously anywhere in the file, or a reader that + # ignores the FAT would pass this test. + if blob.find(b"MSCF") < 0: + return fail("fixture has no MSCF in it at all") + for label, c in (("Data1.cab", cab), ("Data2.cab", stored_cab)): + if c in blob: + return fail("fixture is contiguous for %s: a linear read would succeed, " + "so the test could not detect a reader that ignores the FAT" + % label) + + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "installer.msi") + with open(path, "wb") as f: + f.write(blob) + out = os.path.join(tmp, "out") + j = run(path, out) + + finds = [x for x in j["findings"] if x["type"] == "cfbf"] + if len(finds) != 1: + return fail("expected one cfbf finding, got %d" % len(finds)) + f0 = finds[0] + if f0["confidence_tier"] != "verified": + return fail("tier is %s, want verified" % f0["confidence_tier"]) + if f0["offset"] != 0: + return fail("cfbf found at offset %d, want 0" % f0["offset"]) + if "msi" not in (f0.get("label") or ""): + return fail("label %r does not flag msi" % f0.get("label")) + if f0.get("size") != len(blob): + return fail("size %r != file size %d" % (f0.get("size"), len(blob))) + print(" PASS identified as cfbf, verified, msi, exact span") + + # A cab finding inside the compound file must not survive as a peer: it + # is the stream's first sector run, and extracting it linearly is what + # produced garbage before this handler existed. + stray = [x for x in j["findings"] if x["type"] == "cab" and x["offset"] != 0] + if stray: + return fail("interior cab finding at 0x%x not suppressed by the container" + % stray[0]["offset"]) + print(" PASS interior cab match suppressed by the enclosing compound file") + + for label, want in (("Data1.cab", cab), ("Data2.cab", stored_cab)): + got = find_file(out, label) + if not got: + return fail("MSI-encoded stream name was not decoded to " + label) + if open(got, "rb").read() != want: + return fail("%s stream did not round-trip byte-for-byte" % label) + print(" PASS MSI names decoded; both cab streams byte-exact " + "(compressed + stored)") + + gb = find_file(out, "BigStream") + if not gb or open(gb, "rb").read() != big: + return fail("BigStream did not round-trip byte-for-byte") + print(" PASS large fragmented stream byte-exact") + + gs = find_file(out, "_SummaryInformation") + if not gs or open(gs, "rb").read() != small: + return fail("mini-stream (under the 4096 cutoff) did not round-trip") + print(" PASS mini-FAT stream byte-exact") + + for label, name, want in (("compressed", "payload.bin", member), + ("stored", "stored.bin", stored_member)): + gm = find_file(out, name) + if not gm: + return fail("recursion did not extract the %s cabinet inside the stream" + % label) + if open(gm, "rb").read() != want: + return fail("%s cab member inside the MSI did not round-trip " + "byte-for-byte" % label) + print(" PASS both cabinets extracted from their streams, members byte-exact") + + rc = nesting_policy() + if rc: + return rc + rc = executable_carrier_policy() + if rc: + return rc + + print("ok: cfbf identify + MSI name decode + fragmented stream reassembly" + " + nesting policy") + return 0 + + +def nesting_policy(): + """A compound file below the top level stays packed unless asked for. + + An MSI unpacks to hundreds of database-table streams. When one is incidental + cargo inside something else -- an installer cabinet shipping a redistributable + -- those streams bury the findings that matter, so they are identified and + counted but not written. The file the user actually pointed at is different: + there the container IS the subject, so it is always unpacked. + """ + inner = cfbfbuild.build([("Payload", b"nested msi payload\n" * 400), + ("_SummaryInformation", b"summary" * 40)]) + # Wrap it in a stored cabinet, so the compound file only appears one level + # down, in what the cabinet extractor produced. + wrapper = build_cab([(0, stored_blocks(inner))], + [("installer.msi", 0, 0, len(inner))]) + + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "outer.cab") + with open(path, "wb") as f: + f.write(wrapper) + + out = os.path.join(tmp, "default") + j = run(path, out) + if find_file(out, "installer.msi") is None: + return fail("the cabinet member itself should still be extracted") + if find_file(out, "Payload") is not None: + return fail("nested compound file was unpacked without --unpack-executables") + if j["extraction"].get("skipped_nested_executables") != 1: + return fail("skipped_nested_executables is %r, want 1" + % j["extraction"].get("skipped_nested_executables")) + print(" PASS nested compound file left packed, and the skip is reported") + + out2 = os.path.join(tmp, "optin") + j2 = run(path, out2, extra=("--unpack-executables",)) + got = find_file(out2, "Payload") + if got is None: + return fail("--unpack-executables did not unpack the nested compound file") + if open(got, "rb").read() != b"nested msi payload\n" * 400: + return fail("nested stream did not round-trip under --unpack-executables") + if j2["extraction"].get("skipped_nested_executables"): + return fail("skipped_nested_executables should be absent once the opt-in " + "is given") + print(" PASS --unpack-executables unpacks it, byte-exact") + return 0 + + +def executable_carrier_policy(): + """Extraction does not dig inside an executable it produced itself. + + Vendor installers ship a redistributable, which ships its own setup.exe, + which carries another cabinet. Following that turns one firmware image into a + tree of everything the vendor ever bundled, so a produced PE/ELF is left + alone -- the gate is on the carrier, not on what is inside it. The file named + on the command line is exempt: it never reaches this path. + """ + buried = b"payload inside a nested executable\n" * 300 + inner_cab = build_cab([(COMP_NONE, stored_blocks(buried))], + [("buried.bin", 0, 0, len(buried))]) + # A PE-shaped carrier: an MZ header, then the cabinet at an offset. + exe = b"MZ" + b"\x90\x00" + bytes(508) + inner_cab + wrapper = build_cab([(COMP_NONE, stored_blocks(exe))], + [("nested.exe", 0, 0, len(exe))]) + + with tempfile.TemporaryDirectory() as tmp: + path = os.path.join(tmp, "outer.cab") + with open(path, "wb") as f: + f.write(wrapper) + + out = os.path.join(tmp, "default") + j = run(path, out) + got = find_file(out, "nested.exe") + if got is None or open(got, "rb").read() != exe: + return fail("the executable member itself should still be extracted") + if find_file(out, "buried.bin") is not None: + return fail("dug into a nested executable without --unpack-executables") + if j["extraction"].get("skipped_nested_executables") != 1: + return fail("skipped_nested_executables is %r, want 1" + % j["extraction"].get("skipped_nested_executables")) + print(" PASS nested executable not dug into, and the skip is reported") + + out2 = os.path.join(tmp, "optin") + run(path, out2, extra=("--unpack-executables",)) + gb = find_file(out2, "buried.bin") + if gb is None or open(gb, "rb").read() != buried: + return fail("--unpack-executables did not reach the cabinet inside the exe") + print(" PASS --unpack-executables reaches it, byte-exact") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/test_wince_rom.py b/tests/test_wince_rom.py index 2049727..cb4709a 100644 --- a/tests/test_wince_rom.py +++ b/tests/test_wince_rom.py @@ -70,7 +70,7 @@ def va(off): f2_name = add(b"packed.bin\0", 1) text_off = add(TEXT) - data_packed = ce_compress_blob(DATA) + data_packed = ce_compress_blob(DATA, compress=True) # Huffman-coded LZX data_off = add(data_packed) f1_off = add(PLAIN) packed = ce_compress_blob(PACKED) From d4ad9b86be319a9181f10bd283b21ababc7341f1 Mon Sep 17 00:00:00 2001 From: Matt Brown Date: Sat, 3 Oct 2026 02:21:05 -0400 Subject: [PATCH 3/3] fix(cfbf): make the directory walk iterative to stop a stack overflow walk_tree recursed over the directory's sibling tree and storage children with only a cycle guard. A compound file's directory is not required to be balanced, so a crafted file can lay its entries out as one long linear left/right chain (entry count bounded only by the 262144 cap). The recursion then overflowed the call stack: moria -e segfaulted on a ~15 MB MSI built this way. Rewrite the walk with an explicit heap work stack so any depth up to the entry cap is safe. Behavior is otherwise unchanged (same paths recorded, same cycle guard). tests/cfbfbuild.build_deep_chain builds such a file (120k-entry chain whose deepest entry carries a payload) and test_cfbf.deep_directory_chain asserts extraction completes and the deepest stream round-trips byte-for-byte. The chain length is sized to exhaust a default 8 MiB stack on the old code, so the test fails if the recursion returns. Also drop an unused variable in extract_cab (-Wall). --- src/extract/cab.cpp | 3 +- src/extract/cfbf.cpp | 60 +++++++++++++++++++++++++--------- tests/cfbfbuild.py | 78 ++++++++++++++++++++++++++++++++++++++++++++ tests/test_cfbf.py | 43 +++++++++++++++++++++++- 4 files changed, 165 insertions(+), 19 deletions(-) diff --git a/src/extract/cab.cpp b/src/extract/cab.cpp index f66de70..c372fb2 100644 --- a/src/extract/cab.cpp +++ b/src/extract/cab.cpp @@ -215,7 +215,7 @@ bool extract_cab(const Reader& r, const Finding& f, SafeRoot& root, const std::s } } - size_t failures = 0, mapped = 0, unmapped = 0; + size_t failures = 0, unmapped = 0; bool made_fs = false, made_unmapped = false; for (size_t fi = 0; fi < folders.size(); ++fi) { @@ -253,7 +253,6 @@ bool extract_cab(const Reader& r, const Finding& f, SafeRoot& root, const std::s out.dirs++; } rel = subdir + "/fs/" + it->second; - ++mapped; } else if (ends_with(lname, ".999")) { // cabwiz stores the setup DLL (Install_Init / Install_Exit) // as the .999 member and the install header as .000. diff --git a/src/extract/cfbf.cpp b/src/extract/cfbf.cpp index 2fd3e0c..a539a32 100644 --- a/src/extract/cfbf.cpp +++ b/src/extract/cfbf.cpp @@ -1,6 +1,7 @@ // cfbf.cpp — compound file (CFBF) extractor. See the header. #include "extract/cfbf.hpp" +#include #include #include #include @@ -30,23 +31,50 @@ std::string safe_name(const std::string& name) { return out; } -// Walk the directory tree, recording each entry's path. Siblings form a +// Walk the directory tree, recording each stream's path. Siblings form a // red-black tree inside their storage; a storage's `child` heads the tree of // what it holds. `seen` breaks the cycles a corrupt directory can describe. -void walk_tree(const Cfbf& c, uint32_t idx, const std::string& prefix, - std::vector& paths, std::vector& seen) { - if (idx == kCfbfNoStream || idx >= c.entries.size() || seen[idx]) return; - seen[idx] = 1; - const CfbfEntry& e = c.entries[idx]; - - walk_tree(c, e.left, prefix, paths, seen); - - const std::string name = safe_name(e.name); - const std::string full = prefix.empty() ? name : prefix + "/" + name; - if (e.type == kCfbfStream) paths[idx] = full; - if (e.type == kCfbfStorage) walk_tree(c, e.child, full, paths, seen); - - walk_tree(c, e.right, prefix, paths, seen); +// +// The walk is iterative with an explicit work stack, not recursive: the entry +// count is bounded only by kMaxDirEntries (262144), and a crafted directory can +// lay those out as one long linear left/right sibling chain (or a deep storage +// nesting). Recursing per node would overflow the call stack on such input; the +// explicit stack lives on the heap and cannot. +void walk_tree(const Cfbf& c, uint32_t root_idx, std::vector& paths, + std::vector& seen) { + struct Node { + uint32_t idx; + const std::string* prefix; + }; + const std::string empty; + // A storage's path is the prefix its children carry, so it must outlive the + // child nodes still on the stack. These are heap-allocated and only ever + // appended, so a pointer into the pool stays valid as it grows. + std::vector> pool; + std::vector stack; + stack.push_back({root_idx, &empty}); + + while (!stack.empty()) { + const Node n = stack.back(); + stack.pop_back(); + const uint32_t idx = n.idx; + if (idx == kCfbfNoStream || idx >= c.entries.size() || seen[idx]) continue; + seen[idx] = 1; + const CfbfEntry& e = c.entries[idx]; + const std::string& prefix = *n.prefix; + + const std::string name = safe_name(e.name); + std::string full = prefix.empty() ? name : prefix + "/" + name; + if (e.type == kCfbfStream) paths[idx] = full; + + // Siblings inherit this node's prefix; a storage's children get `full`. + stack.push_back({e.left, n.prefix}); + stack.push_back({e.right, n.prefix}); + if (e.type == kCfbfStorage) { + pool.push_back(std::make_unique(std::move(full))); + stack.push_back({e.child, pool.back().get()}); + } + } } } // namespace @@ -73,7 +101,7 @@ bool extract_cfbf(const Reader& r, const Finding& f, SafeRoot& root, const std:: // dropped: a damaged directory should cost you the hierarchy, not the data. std::vector paths(c.entries.size()); std::vector seen(c.entries.size(), 0); - walk_tree(c, c.entries[0].child, "", paths, seen); + walk_tree(c, c.entries[0].child, paths, seen); std::unordered_map used; for (size_t i = 0; i < c.entries.size(); ++i) { diff --git a/tests/cfbfbuild.py b/tests/cfbfbuild.py index e3b8391..6e10ce8 100644 --- a/tests/cfbfbuild.py +++ b/tests/cfbfbuild.py @@ -160,3 +160,81 @@ def put(sector_list, blob): for i in range(109): struct.pack_into("= the 4096 mini cutoff, + so it lives in a regular FAT chain) and must come back byte-for-byte, which + proves the walk actually reached the end of the chain rather than giving up. + + Uses 4096-byte sectors so even a large `n` needs only a handful of FAT + sectors, which fit the 109 inline DIFAT slots (no DIFAT-sector chain). + """ + ss = 1 << 12 + assert len(payload) >= MINI_CUTOFF + + def entry(name, typ, left, right, child, start, size): + b = bytearray(128) + u = name.encode("utf-16-le")[:62] + b[:len(u)] = u + struct.pack_into(" entry 1; entry i -> entry i+1 via `left`; the last carries payload. + entries = [entry("Root Entry", 5, NOSTREAM, NOSTREAM, 1, ENDOFCHAIN, 0)] + for i in range(1, n + 1): + nxt = i + 1 if i < n else NOSTREAM + if i == n: + entries.append(entry("deep", 2, nxt, NOSTREAM, NOSTREAM, pay_start, len(payload))) + else: + entries.append(entry("s%d" % i, 2, nxt, NOSTREAM, NOSTREAM, ENDOFCHAIN, 0)) + dir_bytes = b"".join(entries).ljust(dir_sectors * ss, b"\0") + + # FAT: dir chain, payload chain, then the FAT sectors themselves (FATSECT). + nfat = 1 + while True: + total = dir_sectors + pay_sectors + nfat + need = (total * 4 + ss - 1) // ss + if need == nfat: + break + nfat = need + fat_base = dir_sectors + pay_sectors + fat = [FREE] * (fat_base + nfat) + for i in range(dir_sectors): + fat[i] = i + 1 if i + 1 < dir_sectors else ENDOFCHAIN + for i in range(pay_sectors): + fat[pay_start + i] = pay_start + i + 1 if i + 1 < pay_sectors else ENDOFCHAIN + for i in range(nfat): + fat[fat_base + i] = FATSECT + fat_bytes = struct.pack("<%dI" % len(fat), *fat).ljust(nfat * ss, b"\xff") + + body = bytearray((fat_base + nfat) * ss) + body[0:len(dir_bytes)] = dir_bytes + body[pay_start * ss:pay_start * ss + len(payload)] = payload + body[fat_base * ss:fat_base * ss + len(fat_bytes)] = fat_bytes + + hdr = bytearray(ss) + hdr[0:8] = bytes([0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1]) + struct.pack_into("