diff --git a/Cargo.lock b/Cargo.lock index eebb1a434b69..891934f99ff6 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -692,6 +692,7 @@ dependencies = [ "arbtest", "capstone", "cranelift-assembler-x64-meta", + "xed-sys", ] [[package]] @@ -760,7 +761,7 @@ dependencies = [ "similar", "smallvec 1.15.1", "souper-ir", - "target-lexicon", + "target-lexicon 0.13.5", "wasmtime-internal-core", ] @@ -821,7 +822,7 @@ dependencies = [ "serde_derive", "similar", "smallvec 1.15.1", - "target-lexicon", + "target-lexicon 0.13.5", "thiserror 2.0.17", "toml", "wasmtime-internal-unwinder", @@ -838,7 +839,7 @@ dependencies = [ "log", "similar", "smallvec 1.15.1", - "target-lexicon", + "target-lexicon 0.13.5", ] [[package]] @@ -850,7 +851,7 @@ dependencies = [ "cranelift", "cranelift-native", "rand 0.10.1", - "target-lexicon", + "target-lexicon 0.13.5", ] [[package]] @@ -954,7 +955,7 @@ dependencies = [ "log", "memmap2", "region", - "target-lexicon", + "target-lexicon 0.13.5", "wasmtime-internal-jit-icache-coherence", "wasmtime-internal-unwinder", "windows-sys 0.61.2", @@ -978,7 +979,7 @@ version = "0.135.0" dependencies = [ "cranelift-codegen", "libc", - "target-lexicon", + "target-lexicon 0.13.5", ] [[package]] @@ -994,7 +995,7 @@ dependencies = [ "gimli 0.33.0", "log", "object 0.39.0", - "target-lexicon", + "target-lexicon 0.13.5", ] [[package]] @@ -1004,7 +1005,7 @@ dependencies = [ "anyhow", "cranelift-codegen", "smallvec 1.15.1", - "target-lexicon", + "target-lexicon 0.13.5", ] [[package]] @@ -1048,7 +1049,7 @@ dependencies = [ "rustc-hash", "serde", "similar", - "target-lexicon", + "target-lexicon 0.13.5", "thiserror 2.0.17", "toml", "walkdir", @@ -3169,9 +3170,9 @@ dependencies = [ [[package]] name = "regalloc2" -version = "0.15.1" +version = "0.15.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "de2c52737737f8609e94f975dee22854a2d5c125772d4b1cf292120f4d45c186" +checksum = "757712e8e61590d6d4f5d563483755538b5aa13467837a3b41cd9832509a7f85" dependencies = [ "allocator-api2", "bumpalo", @@ -3696,6 +3697,12 @@ dependencies = [ "xattr", ] +[[package]] +name = "target-lexicon" +version = "0.12.16" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "61c41af27dd6d1e27b1b16b489db798443478cef1f06a660c96db617ba5de3b1" + [[package]] name = "target-lexicon" version = "0.13.5" @@ -4627,7 +4634,7 @@ dependencies = [ "serde_derive", "serde_json", "smallvec 1.15.1", - "target-lexicon", + "target-lexicon 0.13.5", "tempfile", "tokio", "wasm-compose", @@ -4659,7 +4666,7 @@ version = "48.0.0" dependencies = [ "clap", "shuffling-allocator", - "target-lexicon", + "target-lexicon 0.13.5", "wasmtime", "wasmtime-cli-flags", "wasmtime-wasi", @@ -4733,7 +4740,7 @@ dependencies = [ "serde_json", "similar", "smallvec 1.15.1", - "target-lexicon", + "target-lexicon 0.13.5", "tempfile", "termcolor", "test-programs-artifacts", @@ -4809,7 +4816,7 @@ dependencies = [ "serde_derive", "sha2", "smallvec 1.15.1", - "target-lexicon", + "target-lexicon 0.13.5", "wasm-encoder 0.254.0", "wasmparser 0.254.0", "wasmprinter", @@ -4856,7 +4863,7 @@ dependencies = [ "quote", "rand 0.10.1", "smallvec 1.15.1", - "target-lexicon", + "target-lexicon 0.13.5", "wasmparser 0.254.0", "wasmtime", "wasmtime-fuzzing", @@ -4881,7 +4888,7 @@ dependencies = [ "serde", "serde_json", "smallvec 1.15.1", - "target-lexicon", + "target-lexicon 0.13.5", "tempfile", "tokio", "v8", @@ -4979,7 +4986,7 @@ dependencies = [ "object 0.39.0", "pulley-interpreter", "smallvec 1.15.1", - "target-lexicon", + "target-lexicon 0.13.5", "thiserror 2.0.17", "wasmparser 0.254.0", "wasmtime-environ", @@ -5009,7 +5016,7 @@ dependencies = [ "serde", "serde_derive", "serde_json", - "target-lexicon", + "target-lexicon 0.13.5", "wasmprinter", "wasmtime", "wasmtime-environ", @@ -5094,7 +5101,7 @@ dependencies = [ "gimli 0.33.0", "log", "object 0.39.0", - "target-lexicon", + "target-lexicon 0.13.5", "wasmparser 0.254.0", "wasmtime-environ", "wasmtime-internal-cranelift", @@ -5140,7 +5147,7 @@ dependencies = [ "rand 0.10.1", "serde", "serde_derive", - "target-lexicon", + "target-lexicon 0.13.5", "toml", "wasmparser 0.254.0", "wasmprinter", @@ -5468,7 +5475,7 @@ dependencies = [ "gimli 0.33.0", "regalloc2", "smallvec 1.15.1", - "target-lexicon", + "target-lexicon 0.13.5", "thiserror 2.0.17", "wasmparser 0.254.0", "wasmtime-environ", @@ -5944,6 +5951,16 @@ dependencies = [ "rustix 1.1.4", ] +[[package]] +name = "xed-sys" +version = "0.6.0+xed-2024.05.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "078a73fb5587d5b3bb21262d2b38cf6a1e821ece8f7cbf9f82e61ca6f8a8df0e" +dependencies = [ + "cc", + "target-lexicon 0.12.16", +] + [[package]] name = "yoke" version = "0.7.5" diff --git a/Cargo.toml b/Cargo.toml index d9a8799b4f7d..0d2ff123720c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -409,6 +409,7 @@ clap = { version = "4.5.48", default-features = false, features = ["std", "deriv clap_complete = "4.5.58" hashbrown = { version = "0.17", default-features = false } capstone = { version = "0.14.0", default-features = false, features = ['full', 'arch_x86', 'arch_riscv', 'arch_arm64', 'arch_sysz'] } +xed-sys = { version = "0.6" } smallvec = { version = "1.15.1", features = ["union"] } tracing = { version = "0.1.41", default-features = false } bitflags = "2.9.4" diff --git a/cranelift/assembler-x64/Cargo.toml b/cranelift/assembler-x64/Cargo.toml index 35a7534b6210..2e87fde098ec 100644 --- a/cranelift/assembler-x64/Cargo.toml +++ b/cranelift/assembler-x64/Cargo.toml @@ -9,6 +9,7 @@ rust-version.workspace = true [dependencies] arbitrary = { workspace = true, features = ["derive"], optional = true } capstone = { workspace = true, optional = true } +xed-sys = { workspace = true, optional = true } [dev-dependencies] arbitrary = { workspace = true, features = ["derive"] } @@ -23,3 +24,8 @@ workspace = true [features] fuzz = ['dep:arbitrary', 'dep:capstone'] +# Adds Intel XED as a second, optional disassembler oracle for the roundtrip +# fuzzer. This is additive on top of `fuzz` (Capstone remains the default +# oracle); XED is only built when this feature is explicitly enabled. Note that +# building XED requires Python 3.9+ and a C compiler. +fuzz-xed = ['fuzz', 'dep:xed-sys'] diff --git a/cranelift/assembler-x64/fuzz/Cargo.toml b/cranelift/assembler-x64/fuzz/Cargo.toml index 384a47e3852a..c041559ec276 100644 --- a/cranelift/assembler-x64/fuzz/Cargo.toml +++ b/cranelift/assembler-x64/fuzz/Cargo.toml @@ -12,9 +12,21 @@ cargo-fuzz = true libfuzzer-sys = { workspace = true } cranelift-assembler-x64 = { path = "..", features = ['fuzz'] } +[features] +# Enable the Intel XED disassembler oracle for the `roundtrip-xed` target. +# This builds XED from source, so it requires a C compiler and Python. +fuzz-xed = ["cranelift-assembler-x64/fuzz-xed"] + [[bin]] name = "roundtrip" path = "fuzz_targets/roundtrip.rs" test = false doc = false bench = false + +[[bin]] +name = "roundtrip-xed" +path = "fuzz_targets/roundtrip-xed.rs" +test = false +doc = false +bench = false diff --git a/cranelift/assembler-x64/fuzz/fuzz_targets/roundtrip-xed.rs b/cranelift/assembler-x64/fuzz/fuzz_targets/roundtrip-xed.rs new file mode 100644 index 000000000000..7317304fdaf2 --- /dev/null +++ b/cranelift/assembler-x64/fuzz/fuzz_targets/roundtrip-xed.rs @@ -0,0 +1,16 @@ +#![no_main] + +use cranelift_assembler_x64::{Inst, fuzz}; +use libfuzzer_sys::fuzz_target; + +// This target drives the Intel XED disassembler oracle instead of Capstone. +// XED understands newer encodings (e.g. APX) that the bundled Capstone does +// not. Building XED from source is only done when the `fuzz-xed` feature is +// enabled; without it this target is a no-op so the default fuzz build does +// not require a C compiler and Python. +fuzz_target!(|inst: Inst| { + #[cfg(feature = "fuzz-xed")] + fuzz::roundtrip_xed(&inst); + #[cfg(not(feature = "fuzz-xed"))] + let _ = inst; +}); diff --git a/cranelift/assembler-x64/src/fuzz.rs b/cranelift/assembler-x64/src/fuzz.rs index 290d888f5a00..d0e918f48a7e 100644 --- a/cranelift/assembler-x64/src/fuzz.rs +++ b/cranelift/assembler-x64/src/fuzz.rs @@ -18,12 +18,46 @@ use capstone::{Capstone, arch::BuildsCapstone, arch::BuildsCapstoneSyntax, arch: /// Take a random assembly instruction and check its encoding and /// pretty-printing against a known-good disassembler. /// +/// This uses Capstone as the disassembler oracle; see [`roundtrip_with`] for +/// the oracle-agnostic core. +/// /// # Panics /// /// This function panics to express failure as expected by the `arbitrary` /// fuzzer infrastructure. It may fail during assembly, disassembly, or when /// comparing the disassembled strings. pub fn roundtrip(inst: &Inst) { + roundtrip_with(inst, "capstone", disassemble_capstone, capstone_matches); +} + +/// Like [`roundtrip`], but uses Intel XED as the disassembler oracle instead of +/// Capstone. +/// +/// XED understands newer encodings (e.g. APX) that the bundled Capstone does +/// not, so this is a useful second oracle. It is only available with the +/// `fuzz-xed` feature (which requires building XED from source). +/// +/// # Panics +/// +/// See [`roundtrip`]. +#[cfg(feature = "fuzz-xed")] +pub fn roundtrip_xed(inst: &Inst) { + roundtrip_with(inst, "xed", disassemble_xed, xed_matches); +} + +/// The oracle-agnostic core of [`roundtrip`]: assemble `inst`, disassemble the +/// resulting bytes with the provided `disassemble` oracle, and check that the +/// oracle's pretty-printed output matches the assembler's own `to_string`, +/// where "matches" is defined by the oracle-specific `matches` predicate +/// (`matches(expected_from_oracle, actual_from_assembler)`). +/// +/// The `oracle` name is only used to label diagnostic output on failure. +fn roundtrip_with( + inst: &Inst, + oracle: &str, + disassemble: impl Fn(&[u8], &Inst) -> String, + matches: impl Fn(&str, &str) -> bool, +) { // Check that we can actually assemble this instruction. let assembled = assemble(inst); let expected = disassemble(&assembled, inst); @@ -32,16 +66,23 @@ pub fn roundtrip(inst: &Inst) { // off the instruction offset first. let expected = expected.split_once(' ').unwrap().1; let actual = inst.to_string(); - if expected != actual && expected.trim() != fix_up(&actual) { + if !matches(expected, &actual) { println!("> {inst}"); println!(" debug: {inst:x?}"); println!(" assembled: {}", pretty_print_hexadecimal(&assembled)); - println!(" expected (capstone): {expected}"); + println!(" expected ({oracle}): {expected}"); println!(" actual (to_string): {actual}"); assert_eq!(expected, &actual); } } +/// Comparison predicate for the Capstone oracle: exact match, or match after +/// applying Capstone-specific normalization ([`fix_up`]) to the assembler +/// output. +fn capstone_matches(expected: &str, actual: &str) -> bool { + expected == actual || expected.trim() == fix_up(actual) +} + /// Use this assembler to emit machine code into a byte buffer. /// /// This will skip any traps or label registrations, but this is fine for the @@ -114,8 +155,11 @@ impl CodeSink for TestCodeSink { } } +/// Disassemble a single instruction with Capstone, returning its AT&T-syntax +/// string. This is the default [`roundtrip`] oracle. +/// /// Building a new `Capstone` each time is suboptimal (TODO). -fn disassemble(assembled: &[u8], original: &Inst) -> String { +fn disassemble_capstone(assembled: &[u8], original: &Inst) -> String { let cs = Capstone::new() .x86() .mode(x86::ArchMode::Mode64) @@ -149,6 +193,79 @@ fn disassemble(assembled: &[u8], original: &Inst) -> String { inst.to_string() } +/// Disassemble a single instruction with Intel XED, returning a string in the +/// same shape as [`disassemble_capstone`] (a leading offset token, a space, +/// then the AT&T-syntax instruction) so that [`roundtrip_with`] can compare it +/// uniformly. +#[cfg(feature = "fuzz-xed")] +fn disassemble_xed(assembled: &[u8], original: &Inst) -> String { + use core::ffi::c_void; + use std::sync::Once; + use xed_sys::*; + + // XED requires a one-time global table initialization before any decode. + static INIT: Once = Once::new(); + // SAFETY: `xed_tables_init` is safe to call; `Once` guarantees it runs + // exactly once even across threads. + INIT.call_once(|| unsafe { xed_tables_init() }); + + // SAFETY: all of the following are standard XED decode/format calls + // operating on stack-allocated, properly initialized structures. + unsafe { + let mut xedd: xed_decoded_inst_t = core::mem::zeroed(); + xed_decoded_inst_zero(&mut xedd); + xed_decoded_inst_set_mode(&mut xedd, XED_MACHINE_MODE_LONG_64, XED_ADDRESS_WIDTH_64b); + + let error = xed_decode( + &mut xedd, + assembled.as_ptr(), + assembled.len() as core::ffi::c_uint, + ); + if error != XED_ERROR_NONE { + println!("> {original}"); + println!(" debug: {original:x?}"); + println!(" assembled: {}", pretty_print_hexadecimal(assembled)); + let name = core::ffi::CStr::from_ptr(xed_error_enum_t2str(error)); + panic!("xed failed to decode: {}", name.to_string_lossy()); + } + + // XED must consume exactly the bytes we emitted; a shorter length means + // trailing bytes were not part of the instruction. + let decoded_len = xed_decoded_inst_get_length(&xedd) as usize; + if decoded_len != assembled.len() { + println!("> {original}"); + println!(" debug: {original:x?}"); + println!(" assembled: {}", pretty_print_hexadecimal(assembled)); + assert_eq!( + decoded_len, + assembled.len(), + "xed did not consume all bytes" + ); + } + + // Format in AT&T syntax to match the assembler's own pretty-printing. + let mut buf = [0i8; 256]; + let ok = xed_format_context( + XED_SYNTAX_ATT, + &xedd, + buf.as_mut_ptr(), + buf.len() as core::ffi::c_int, + 0, + core::ptr::null_mut::(), + None, + ); + assert!(ok != 0, "xed failed to format instruction"); + + let disasm = core::ffi::CStr::from_ptr(buf.as_ptr()) + .to_string_lossy() + .into_owned(); + + // Prepend a fake offset token so the shape matches Capstone's + // `0x0: ` output that `roundtrip_with` expects. + format!("0: {disasm}") + } +} + fn pretty_print_hexadecimal(hex: &[u8]) -> String { use core::fmt::Write; let mut s = String::with_capacity(hex.len() * 2); @@ -268,6 +385,117 @@ fn fix_up(dis: &str) -> alloc::borrow::Cow<'_, str> { replace_signed_immediates(&dis) } +/// Comparison predicate for the Intel XED oracle. +/// +/// XED decodes the same instructions as the assembler but prints them with +/// slightly different conventions. The differences we reconcile here are: +/// +/// - XED omits the AT&T operand-size suffix on the mnemonic (e.g. `adc` +/// instead of `adcw`) when an operand already makes the width unambiguous. +/// - XED appends a vector-length marker (`x`/`y`/`z` for 128/256/512-bit) to +/// the mnemonic of some VEX/EVEX instructions with a memory operand (e.g. +/// `vpalignrx` instead of `vpalignr`). +/// - XED may use different internal whitespace (e.g. a double space after the +/// mnemonic). +/// +/// Rather than blindly stripping suffixes from the assembler mnemonic--which +/// would corrupt mnemonics that legitimately end in those letters, like `mul` +/// or `call`--we use XED's mnemonic as ground truth: a suffix is only dropped +/// if doing so makes the two mnemonics exactly equal. +#[cfg(feature = "fuzz-xed")] +fn xed_matches(expected: &str, actual: &str) -> bool { + let actual = remove_after_semicolon(actual); + + // Normalize runs of whitespace to a single space, and drop spaces that + // follow a comma, so cosmetic spacing differences (XED's double space after + // the mnemonic, and its lack of spaces inside memory operands like + // `(%rsi,%rdx,2)`) don't matter. Also drop an explicit SIB scale of 1, + // which XED prints (`(%rbp,%rsi,1)`) but the assembler omits. + fn normalize_ws(s: &str) -> String { + let collapsed = s.split_whitespace().collect::>().join(" "); + collapsed.replace(", ", ",").replace(",1)", ")") + } + let expected = canonicalize_immediates(&normalize_ws(expected)); + let actual = canonicalize_immediates(&normalize_ws(actual)); + if expected == actual { + return true; + } + + // Split "mnemonic operands" into the leading mnemonic and the remainder. + fn split_mnemonic(s: &str) -> (&str, &str) { + match s.split_once(' ') { + Some((m, rest)) => (m, rest), + None => (s, ""), + } + } + + let (exp_mnemonic, exp_ops) = split_mnemonic(&expected); + let (act_mnemonic, act_ops) = split_mnemonic(&actual); + + if exp_ops != act_ops { + return false; + } + + if act_mnemonic == exp_mnemonic { + return true; + } + + // The assembler mnemonic is the XED mnemonic plus a single trailing + // operand-size suffix (`adcw` vs `adc`). + if act_mnemonic.strip_suffix(['b', 'w', 'l', 'q']) == Some(exp_mnemonic) { + return true; + } + + // The XED mnemonic is the assembler mnemonic plus a trailing vector-length + // marker (`vpalignrx` vs `vpalignr`). + if exp_mnemonic.strip_suffix(['x', 'y', 'z']) == Some(act_mnemonic) { + return true; + } + + false +} + +/// Rewrite every `$`-prefixed immediate in a disassembly string into a single +/// canonical form so that decimal-vs-hex and signedness differences between the +/// assembler and XED don't cause spurious mismatches. +/// +/// The assembler prints small immediates in decimal (`$1`) and larger ones in +/// hex (`$0xb143`), while XED always prints hex (`$0x1`). We parse each +/// immediate's numeric value (handling an optional leading `-` and `0x`) and +/// re-emit it as `$0x{:x}` of its `u64` two's-complement value. +#[cfg(feature = "fuzz-xed")] +fn canonicalize_immediates(dis: &str) -> String { + let mut out = String::with_capacity(dis.len()); + let mut rest = dis; + while let Some(idx) = rest.find('$') { + out.push_str(&rest[..idx]); + // Everything after the '$'. + let after = &rest[idx + 1..]; + let (neg, num) = match after.strip_prefix('-') { + Some(n) => (true, n), + None => (false, after), + }; + let (radix, digits) = match num.strip_prefix("0x") { + Some(d) => (16, d), + None => (10, num), + }; + let n = digits.chars().take_while(|c| c.is_digit(radix)).count(); + if n == 0 { + // Not actually an immediate we can parse; keep the '$' literally. + out.push('$'); + rest = after; + continue; + } + let (value_str, tail) = digits.split_at(n); + let value = u64::from_str_radix(value_str, radix).unwrap_or(0); + let value = if neg { value.wrapping_neg() } else { value }; + out.push_str(&format!("$0x{value:x}")); + rest = tail; + } + out.push_str(rest); + out +} + /// Fuzz-specific registers. /// /// For the fuzzer, we do not need any fancy register types; see [`FuzzReg`]. @@ -412,4 +640,28 @@ mod test { roundtrip(&inst.into()); } } + + /// Same as [`smoke`], but exercises the Intel XED oracle. Only available + /// with the `fuzz-xed` feature. + /// + /// This is `#[ignore]`d for now: XED and the assembler agree on decoding, + /// but reconciling every one of XED's printing conventions is a work in + /// progress. The [`xed_matches`] predicate already handles operand-size + /// suffixes, vector-length markers, whitespace, immediate formatting, and + /// explicit SIB scales; the remaining known gap is condition-code mnemonic + /// aliases (e.g. `cmovnl` vs `cmovge`). Run explicitly with + /// `cargo test --features fuzz-xed -- --ignored smoke_xed`. + #[cfg(feature = "fuzz-xed")] + #[test] + #[ignore = "XED disassembly normalization is a work in progress"] + fn smoke_xed() { + let count = AtomicUsize::new(0); + arbtest(|u| { + let inst: Inst = u.arbitrary()?; + roundtrip_xed(&inst); + println!("#{}: {inst}", count.fetch_add(1, Ordering::SeqCst)); + Ok(()) + }) + .budget_ms(1_000); + } } diff --git a/supply-chain/config.toml b/supply-chain/config.toml index 5dce8044f95d..d51f32d6f79f 100644 --- a/supply-chain/config.toml +++ b/supply-chain/config.toml @@ -475,6 +475,10 @@ criteria = "safe-to-deploy" version = "2.14.0" criteria = "safe-to-deploy" +[[exemptions.target-lexicon]] +version = "0.12.16" +criteria = "safe-to-deploy" + [[exemptions.tempfile]] version = "3.3.0" criteria = "safe-to-deploy" @@ -527,6 +531,10 @@ criteria = "safe-to-deploy" version = "0.4.0" criteria = "safe-to-deploy" +[[exemptions.xed-sys]] +version = "0.6.0+xed-2024.05.20" +criteria = "safe-to-deploy" + [[exemptions.zip]] version = "0.6.6" criteria = "safe-to-deploy"