diff --git a/build/xwasm/code.go b/build/xwasm/code.go new file mode 100644 index 0000000..270e0ee --- /dev/null +++ b/build/xwasm/code.go @@ -0,0 +1,182 @@ +package xwasm + +import ( + "errors" + "io" + "math" +) + +// Func locates one function body in the Code section. +type Func struct { + // Index is the function's index in the module's function index space, + // where imported functions come first: the body's ordinal plus + // [File.NumImportedFuncs]. The name section keys names by this index. + Index uint32 + // Offset is the position of the body's size prefix, measured from the start + // of the Code section's contents. DWARF addresses in a WebAssembly module + // are measured from the same point, but a subprogram's DW_AT_low_pc names + // the body proper: [Func.BodyOffset]. + Offset uint64 + // PrefixSize is the length of the size prefix, 1 to 5 bytes. + PrefixSize uint8 + // Size is the body's size in bytes, excluding the prefix. + Size uint64 +} + +// BodyOffset returns the offset of the body proper -- its local declarations, +// then its code -- from the start of the Code section's contents. +func (fn Func) BodyOffset() uint64 { return fn.Offset + uint64(fn.PrefixSize) } + +// End returns the offset one past the body's last byte, from the start of the +// Code section's contents. +func (fn Func) End() uint64 { return fn.BodyOffset() + fn.Size } + +// CodeReader walks the function bodies of a module's Code section through a +// caller-owned buffer, without holding anything per function. +// +// A CodeReader is reusable: [CodeReader.Reset] points it at another module and +// keeps the buffer. The buffer belongs to the reader from Reset until the next +// Reset; lending it to another reader in between corrupts both walks. +type CodeReader struct { + c cursor + start int64 // File offset of the Code section's contents. + count uint32 + imported uint32 + err error +} + +// Reset points cr at f's Code section, reading through buf. buf only needs to +// hold a few bytes at a time: bodies are skipped, never read. Larger buffers +// mean fewer reads. A module with no Code section yields no functions. +func (cr *CodeReader) Reset(f *File, buf []byte) error { + if len(buf) < maxSectionFrame { + return BufferTooSmallError{Need: maxSectionFrame, Have: len(buf)} + } + imported, err := f.NumImportedFuncs() + if err != nil { + return err + } + *cr = CodeReader{c: cr.c, imported: imported} + sec, err := f.SectionByID(SecCode) + if err != nil { + cr.c.reset(f.r, buf, 0, 0) + return nil + } + sh := sec.SectionHeader() + cr.start = int64(sh.Offset) + cr.c.reset(f.r, buf, cr.start, int64(sh.End())) + // Every body takes at least its one-byte size prefix and one byte of local + // declaration count. + cr.count = cr.c.vecLen(2) + if cr.c.err != nil { + return cr.c.err + } + if uint64(imported)+uint64(cr.count) > math.MaxUint32 { + return makeFormatErr(sh.Offset, "function index space overflows uint32", cr.count) + } + return nil +} + +// NumFuncs returns the number of function bodies in the Code section. +func (cr *CodeReader) NumFuncs() uint32 { return cr.count } + +// NumImportedFuncs returns the number of imported functions, which precede the +// module's own in the function index space. +func (cr *CodeReader) NumImportedFuncs() uint32 { return cr.imported } + +// Err returns why the last [CodeReader.Funcs] walk stopped early, or nil if it +// ran to completion or the caller stopped it. +func (cr *CodeReader) Err() error { return cr.err } + +// Funcs yields each function body in order of index. It may be called again to +// walk the bodies anew. +func (cr *CodeReader) Funcs(yield func(Func) bool) { + c := &cr.c + cr.err = nil + if cr.count == 0 { + return + } + c.err = nil + c.seek(cr.start) + c.u32() // Count, validated by Reset. + for i := uint32(0); i < cr.count; i++ { + prefix := c.pos() + size := c.u32() + if c.err != nil { + break + } + fn := Func{ + Index: cr.imported + i, + Offset: uint64(prefix - cr.start), + PrefixSize: uint8(c.pos() - prefix), + Size: uint64(size), + } + c.skip(fn.Size) + if c.err != nil { + break + } + if !yield(fn) { + return + } + } + if c.err == nil && c.pos() != c.end { + c.fail(makeFormatErr(uint64(c.pos()), "trailing bytes after function bodies", c.end-c.pos())) + } + cr.err = c.err +} + +// Import descriptor kinds. +const ( + importFunc = 0x00 + importTable = 0x01 + importMemory = 0x02 + importGlobal = 0x03 + importTag = 0x04 +) + +// NumImportedFuncs walks the Import section and counts the imported functions. +// A module with no Import section imports none. +func (f *File) NumImportedFuncs() (uint32, error) { + sec, err := f.SectionByID(SecImport) + if errors.Is(err, errNoSection) { + return 0, nil + } + sh := sec.SectionHeader() + var c cursor + c.reset(f.r, f.buf[:], int64(sh.Offset), int64(sh.End())) + // An import is at least two empty names and a kind byte with a one-byte + // operand. + count := c.vecLen(4) + var funcs uint32 + for i := uint32(0); i < count && c.err == nil; i++ { + c.skipName() // Module. + c.skipName() // Field. + switch kind := c.u8(); kind { + case importFunc: + c.u32() // Type index. + funcs++ + case importTable: + c.valType() + c.limits() + case importMemory: + c.limits() + case importGlobal: + c.valType() + c.u8() // Mutability. + case importTag: + c.u8() // Attribute. + c.u32() // Type index. + default: + if c.err == nil { + c.fail(makeFormatErr(uint64(c.pos()-1), "unknown import kind", kind)) + } + } + } + if c.err == nil && c.pos() != c.end { + c.fail(makeFormatErr(uint64(c.pos()), "trailing bytes after imports", c.end-c.pos())) + } + if c.err == io.EOF { + c.err = io.ErrUnexpectedEOF + } + return funcs, c.err +} diff --git a/build/xwasm/data.go b/build/xwasm/data.go new file mode 100644 index 0000000..f09bf4a --- /dev/null +++ b/build/xwasm/data.go @@ -0,0 +1,179 @@ +package xwasm + +// DataSegment locates one segment of the Data section. +type DataSegment struct { + // Index is the segment's index, which the name section keys names by. + Index uint32 + // Offset is the position of the segment's first byte -- its flags -- + // measured from the start of the Data section's contents. + Offset uint64 + // HeaderSize is the bytes the segment spends before its initializer: flags, + // memory index, offset expression and byte count. + HeaderSize uint32 + // Size is the length of the initializer bytes. + Size uint64 + // Passive marks a segment copied into memory only by memory.init, which + // has no address of its own. + Passive bool + // Addr is the linear-memory address an active segment is placed at, valid + // when AddrKnown is set: the offset expression is a single constant. An + // expression reading a global is only resolved at instantiation. + Addr uint64 + AddrKnown bool +} + +// DataOffset returns the offset of the initializer bytes from the start of the +// Data section's contents. +func (s DataSegment) DataOffset() uint64 { return s.Offset + uint64(s.HeaderSize) } + +// End returns the offset one past the segment's last byte, from the start of +// the Data section's contents. +func (s DataSegment) End() uint64 { return s.DataOffset() + s.Size } + +// DataReader walks the segments of a module's Data section through a +// caller-owned buffer, without holding anything per segment. +// +// A DataReader is reusable: [DataReader.Reset] points it at another module and +// keeps the buffer. The buffer belongs to the reader from Reset until the next +// Reset; lending it to another reader in between corrupts both walks. +type DataReader struct { + c cursor + start int64 + count uint32 + err error +} + +// Reset points dr at f's Data section, reading through buf. As with +// [CodeReader], segment contents are skipped rather than read, so buf needs to +// hold only a few bytes. A module with no Data section yields no segments. +func (dr *DataReader) Reset(f *File, buf []byte) error { + if len(buf) < maxSectionFrame { + return BufferTooSmallError{Need: maxSectionFrame, Have: len(buf)} + } + *dr = DataReader{c: dr.c} + sec, err := f.SectionByID(SecData) + if err != nil { + dr.c.reset(f.r, buf, 0, 0) + return nil + } + sh := sec.SectionHeader() + dr.start = int64(sh.Offset) + dr.c.reset(f.r, buf, dr.start, int64(sh.End())) + // The smallest segment is passive and empty: flags and a zero count. + dr.count = dr.c.vecLen(2) + return dr.c.err +} + +// NumSegments returns the number of segments in the Data section. +func (dr *DataReader) NumSegments() uint32 { return dr.count } + +// Err returns why the last [DataReader.Segments] walk stopped early, or nil if +// it ran to completion or the caller stopped it. +func (dr *DataReader) Err() error { return dr.err } + +// Segments yields each data segment in order of index. It may be called again +// to walk the segments anew. +func (dr *DataReader) Segments(yield func(DataSegment) bool) { + c := &dr.c + dr.err = nil + if dr.count == 0 { + return + } + c.err = nil + c.seek(dr.start) + c.u32() // Count, validated by Reset. + for i := uint32(0); i < dr.count; i++ { + first := c.pos() + seg := DataSegment{Index: i, Offset: uint64(first - dr.start)} + switch flags := c.u32(); flags { + case 0: + seg.Addr, seg.AddrKnown = c.constExpr() + case 1: + seg.Passive = true + case 2: + c.u32() // Memory index. + seg.Addr, seg.AddrKnown = c.constExpr() + default: + if c.err == nil { + c.fail(makeFormatErr(uint64(first), "invalid data segment flags", flags)) + } + } + seg.Size = uint64(c.u32()) + if c.err != nil { + break + } + seg.HeaderSize = uint32(c.pos() - first) + c.skip(seg.Size) + if c.err != nil { + break + } + if !yield(seg) { + return + } + } + if c.err == nil && c.pos() != c.end { + c.fail(makeFormatErr(uint64(c.pos()), "trailing bytes after data segments", c.end-c.pos())) + } + dr.err = c.err +} + +// Opcodes that may appear in a constant expression, with the extended-const +// proposal's integer arithmetic. +const ( + opEnd = 0x0b + opGlobalGet = 0x23 + opI32Const = 0x41 + opI64Const = 0x42 + opF32Const = 0x43 + opF64Const = 0x44 + opI32Add = 0x6a + opI32Sub = 0x6b + opI32Mul = 0x6c + opI64Add = 0x7c + opI64Sub = 0x7d + opI64Mul = 0x7e + opRefNull = 0xd0 + opRefFunc = 0xd2 + opPrefixSIMD = 0xfd + simdV128Const = 12 +) + +// constExpr skips a constant expression through its end opcode. It reports the +// expression's value when the expression is a lone integer constant, which is +// how a linker places a data segment. +func (c *cursor) constExpr() (v uint64, known bool) { + start := c.pos() + for n := 0; c.err == nil; n++ { + op := c.u8() + switch op { + case opEnd: + if !known || n != 1 { + return 0, false // Not a lone constant: resolved at instantiation. + } + return v, true + case opI32Const: + v, known = uint64(uint32(c.sleb(32))), true + case opI64Const: + v, known = uint64(c.sleb(64)), true + case opF32Const: + c.skip(4) + case opF64Const: + c.skip(8) + case opGlobalGet, opRefFunc: + c.u32() + case opRefNull: + c.sleb(33) // Heap type. + case opI32Add, opI32Sub, opI32Mul, opI64Add, opI64Sub, opI64Mul: + case opPrefixSIMD: + if sub := c.u32(); sub != simdV128Const && c.err == nil { + c.fail(makeFormatErr(uint64(start), "non-constant SIMD instruction in constant expression", sub)) + } + c.skip(16) + default: + if c.err == nil { + c.fail(makeFormatErr(uint64(c.pos()-1), "non-constant instruction in constant expression", op)) + } + } + } + return 0, false +} diff --git a/build/xwasm/file.go b/build/xwasm/file.go new file mode 100644 index 0000000..f2ba1a5 --- /dev/null +++ b/build/xwasm/file.go @@ -0,0 +1,244 @@ +package xwasm + +import ( + "errors" + "fmt" + "io" +) + +// File is a decoded WebAssembly module: its header and the location of each +// section. Section contents are read on demand through the [io.ReaderAt] given +// to [File.Read], which must outlive the File. +type File struct { + hdr Header + sections []section + r io.ReaderAt + + // buf is scratch for decoding section framing and comparing names. + buf [fileBufSize]byte +} + +// section is deliberately just the header: a module may declare thousands of +// tiny custom sections, and nothing per section is worth holding resident but +// where it is. +type section struct { + SectionHeader +} + +// Read parses the module in r, recording where each section lies. It reads the +// header and the framing of each section, never section contents. The module +// is expected to start at offset 0 and end at the end of r. +// +// Calling Read again on the same File reuses its section table, so parsing a +// sequence of modules allocates only when one has more sections than any +// before it. +func (f *File) Read(r io.ReaderAt) error { + buf := f.buf[:] + n, err := r.ReadAt(buf[:headerSize], 0) + if n < headerSize { + if err == nil || err == io.EOF { + err = io.ErrUnexpectedEOF + } + return err + } + header, _, err := DecodeHeader(buf) + if err != nil { + return err + } + sections := f.sections[:0] + // Leave f describing nothing unless the whole module parses: a half-read + // section table is worse than none. + f.hdr, f.sections, f.r = Header{}, sections, nil + + var lastRank uint8 + for off := uint64(headerSize); ; { + n, err := r.ReadAt(buf[:maxSectionFrame], int64(off)) + if n == 0 { + if err == io.EOF || err == nil { + break // Clean end of module on a section boundary. + } + return err + } else if err != nil && err != io.EOF { + return err + } + sh, _, err := DecodeSectionHeader(buf[:n], off) + if err != nil { + return err + } + if sh.ID != SecCustom { + rank := sectionRank[sh.ID] + if rank <= lastRank { + return makeFormatErr(off, "section out of order or duplicated", sh.ID) + } + lastRank = rank + } + end := sh.End() + if end > 1<<63-1 { + return makeFormatErr(off, errSectionPastEOF.Error(), sh.Size) + } + // Probe the last byte: a section claiming more than the file holds is + // caught here rather than by whichever reader first reaches its tail. + if n, _ := r.ReadAt(buf[:1], int64(end)-1); n != 1 { + return makeFormatErr(off, errSectionPastEOF.Error(), sh.Size) + } + if len(sections) >= maxSections { + return makeFormatErr(off, errTooManySecs.Error(), len(sections)) + } + sections = append(sections, section{SectionHeader: sh}) + off = end + } + f.hdr, f.sections, f.r = header, sections, r + return nil +} + +// Header returns the module header. +func (f *File) Header() Header { return f.hdr } + +// NumSections returns the number of sections in the module, custom ones +// included. +func (f *File) NumSections() int { return len(f.sections) } + +// Section returns the section at sectionIdx, in file order. +func (f *File) Section(sectionIdx int) (FileSection, error) { + if sectionIdx >= len(f.sections) || sectionIdx < 0 { + return FileSection{}, errors.New("OOB/negative section index") + } + return FileSection{f: f, sindex: sectionIdx}, nil +} + +// SectionByID returns the first section with the given id. For any id but +// [SecCustom] that is the only one. +func (f *File) SectionByID(id SectionID) (FileSection, error) { + for i := range f.sections { + if f.sections[i].ID == id { + return FileSection{f: f, sindex: i}, nil + } + } + return FileSection{}, fmt.Errorf("%w: %s", errNoSection, id) +} + +// SectionByName returns the first custom section with the given name, or the +// standard section whose id's [SectionID.String] is name ("code", "data", ...). +func (f *File) SectionByName(name string) (FileSection, error) { + for i := range f.sections { + s := FileSection{f: f, sindex: i} + sh := &f.sections[i] + if sh.ID != SecCustom { + if sh.ID.String() == name { + return s, nil + } + continue + } + ok, err := s.nameEquals(name) + if err != nil { + return FileSection{}, err + } else if ok { + return s, nil + } + } + return FileSection{}, fmt.Errorf("%w: %q", errNoSection, name) +} + +// FileSection is the handle to a module's section. +type FileSection struct { + f *File + sindex int +} + +func (fs FileSection) ptr() *section { return &fs.f.sections[fs.sindex] } + +// Index returns the index of the section within the module, in file order. +func (fs FileSection) Index() int { return fs.sindex } + +// SectionHeader returns the section's header. +func (fs FileSection) SectionHeader() SectionHeader { return fs.ptr().SectionHeader } + +// Size returns the size of the section's contents in bytes, which for a custom +// section excludes its name. +func (fs FileSection) Size() int64 { return int64(fs.ptr().Size) } + +// Open returns a new [io.SectionReader] reading the section's contents. +func (fs FileSection) Open() *io.SectionReader { + s := fs.ptr() + return io.NewSectionReader(fs.f.r, int64(s.Offset), int64(s.Size)) +} + +// AppendData appends the section's contents to dst and returns the result. +func (fs FileSection) AppendData(dst []byte) ([]byte, error) { + s := fs.ptr() + return appendAt(dst, fs.f.r, s.Offset, s.Size) +} + +// AppendName appends the section's name to dst: a custom section's own name, +// or for a standard section its id's [SectionID.String]. +func (fs FileSection) AppendName(dst []byte) ([]byte, error) { + s := fs.ptr() + if s.ID != SecCustom { + return append(dst, s.ID.String()...), nil + } + return appendAt(dst, fs.f.r, s.NameOff, uint64(s.NameLen)) +} + +// Name returns the section's name. See [FileSection.AppendName]. +func (fs FileSection) Name() (string, error) { + s := fs.ptr() + if s.ID != SecCustom { + return s.ID.String(), nil + } + b, err := fs.AppendName(nil) + return string(b), err +} + +// nameEquals compares a custom section's name against name, reading it through +// the File's scratch buffer so the comparison allocates nothing. +func (fs FileSection) nameEquals(name string) (bool, error) { + s := fs.ptr() + if uint64(s.NameLen) != uint64(len(name)) { + return false, nil + } + buf := fs.f.buf[:] + for off := 0; off < len(name); { + chunk := buf[:min(len(buf), len(name)-off)] + n, err := fs.f.r.ReadAt(chunk, int64(s.NameOff)+int64(off)) + if n < len(chunk) { + if err == nil || err == io.EOF { + err = io.ErrUnexpectedEOF + } + return false, err + } + if string(chunk) != name[off:off+n] { // Compiles to a comparison, no copy. + return false, nil + } + off += n + } + return true, nil +} + +func (fs FileSection) String() string { + if fs.f == nil { + return "" + } + name, _ := fs.Name() + sh := fs.SectionHeader() + return fmt.Sprintf("%s, %s off=%#x size=%d", name, sh.ID.String(), sh.Offset, sh.Size) +} + +// appendAt appends the size bytes at off in r to dst. +func appendAt(dst []byte, r io.ReaderAt, off, size uint64) ([]byte, error) { + if size == 0 { + return dst, nil + } + if sliceCapWithSize(1, size) < 0 || size > uint64(1<<63-1-len(dst)) { + return dst, errors.New("section too large") + } + dst = slicesGrow(dst, int(size)) + toRead := dst[len(dst) : len(dst)+int(size)] + n, err := r.ReadAt(toRead, int64(off)) + if n < len(toRead) { + if err == nil || err == io.EOF { + err = io.ErrUnexpectedEOF + } + return dst, err + } + return dst[:len(dst)+int(size)], nil +} diff --git a/build/xwasm/fuzz_test.go b/build/xwasm/fuzz_test.go new file mode 100644 index 0000000..3acefe9 --- /dev/null +++ b/build/xwasm/fuzz_test.go @@ -0,0 +1,96 @@ +package xwasm + +import ( + "bytes" + "os" + "testing" +) + +// fuzzState is everything a fuzz iteration touches, allocated once. The fuzz +// engine calls the target sequentially within a worker process, so one set +// serves every iteration: an iteration allocates nothing but what a module +// with more sections than any before it grows the section table by. +type fuzzState struct { + r bytes.Reader + f File + cr CodeReader + nr NameReader + dr DataReader + cbuf [64]byte + dbuf [64]byte + nbuf [256]byte +} + +func (s *fuzzState) run(data []byte) { + s.r.Reset(data) + if s.f.Read(&s.r) != nil { + return + } + for i := 0; i < s.f.NumSections(); i++ { + sec, _ := s.f.Section(i) + sh := sec.SectionHeader() + if sh.End() > uint64(len(data)) || sh.Offset < sh.Start { + panic("section outside module") + } + } + s.f.SectionByName(".debug_line") + var size uint64 // A module without a Code section yields no functions. + if code, err := s.f.SectionByID(SecCode); err == nil { + size = code.SectionHeader().Size + } + if s.cr.Reset(&s.f, s.cbuf[:]) == nil { + for fn := range s.cr.Funcs { + if fn.End() > size { + panic("function body outside code section") + } + } + } + if s.nr.Reset(&s.f, s.nbuf[:]) == nil { + for range s.nr.FuncNames { + } + for range s.nr.DataNames { + } + } + size = 0 + if data, err := s.f.SectionByID(SecData); err == nil { + size = data.SectionHeader().Size + } + if s.dr.Reset(&s.f, s.dbuf[:]) == nil { + for seg := range s.dr.Segments { + if seg.End() > size { + panic("data segment outside data section") + } + } + } +} + +func FuzzRead(f *testing.F) { + for _, name := range fixtures { + data, err := os.ReadFile(name) + if err != nil { + f.Fatal(err) + } + f.Add(data) + } + f.Add([]byte("\x00asm\x01\x00\x00\x00")) + var s fuzzState + f.Fuzz(func(t *testing.T, data []byte) { + s.run(data) + }) +} + +// BenchmarkFuzzIteration guards the fuzz target's allocation behavior: B/op +// must stay at zero, since the fuzzer runs it flat out on every core. +func BenchmarkFuzzIteration(b *testing.B) { + data, err := os.ReadFile(fixtures[0]) + if err != nil { + b.Fatal(err) + } + var s fuzzState + s.run(data) + b.ReportAllocs() + b.SetBytes(int64(len(data))) + for b.Loop() { + s.run(data) + } +} diff --git a/build/xwasm/leb.go b/build/xwasm/leb.go new file mode 100644 index 0000000..22fd9b3 --- /dev/null +++ b/build/xwasm/leb.go @@ -0,0 +1,69 @@ +package xwasm + +import "io" + +// DecodeULEB128 decodes an unsigned LEB128 value of at most bits bits (1 to +// 64), returning the value and the number of bytes consumed. +// +// WebAssembly is stricter than DWARF about LEB128: an encoding may use at most +// ceil(bits/7) bytes, and the bits of the last byte past the type's width must +// be zero. A violation is an error, not a truncation; in a module it means the +// decoder has lost its place. +func DecodeULEB128(b []byte, bits int) (v uint64, n int, err error) { + maxBytes := (bits + 6) / 7 + var shift uint + for n < len(b) { + c := b[n] + n++ + if n == maxBytes { + if rem := bits - int(shift); rem < 7 && c>>rem != 0 { + return 0, n, errLEBOverflow // Also catches a continuation bit. + } + } + v |= uint64(c&0x7f) << shift + if c&0x80 == 0 { + return v, n, nil + } + if n == maxBytes { + return 0, n, errLEBTooLong + } + shift += 7 + } + return 0, n, io.ErrUnexpectedEOF +} + +// DecodeSLEB128 decodes a signed LEB128 value of at most bits bits (1 to 64) +// under the same rules as [DecodeULEB128]: at most ceil(bits/7) bytes, and the +// unused bits of the last byte must repeat the sign bit. +func DecodeSLEB128(b []byte, bits int) (v int64, n int, err error) { + maxBytes := (bits + 6) / 7 + var shift uint + for n < len(b) { + c := b[n] + n++ + if n == maxBytes { + if c&0x80 != 0 { + return 0, n, errLEBTooLong + } + // The payload's sign bit and every payload bit above the type's + // width must agree: all clear or all set. + rem := bits - int(shift) + top := (c & 0x7f) >> (rem - 1) + if top != 0 && top != 0x7f>>(rem-1) { + return 0, n, errLEBOverflow + } + } + v |= int64(c&0x7f) << shift + shift += 7 + if c&0x80 == 0 { + if shift < 64 && c&0x40 != 0 { + v |= -1 << shift + } + return v, n, nil + } + if n == maxBytes { + return 0, n, errLEBTooLong + } + } + return 0, n, io.ErrUnexpectedEOF +} diff --git a/build/xwasm/names.go b/build/xwasm/names.go new file mode 100644 index 0000000..24509f4 --- /dev/null +++ b/build/xwasm/names.go @@ -0,0 +1,124 @@ +package xwasm + +// Subsection ids of the "name" custom section, from the extended name section +// proposal. Each is optional and at most one of each appears, in order. +const ( + nameSubModule = 0 + nameSubFunction = 1 + nameSubGlobal = 7 + nameSubData = 9 +) + +// NameReader walks the name maps in a module's "name" custom section +// through a caller-owned buffer. Names are presented in place, borrowed from +// the buffer, so a walk allocates nothing. +// +// A NameReader is reusable: [NameReader.Reset] points it at another module and +// keeps the buffer. The buffer belongs to the reader from Reset until the next +// Reset; lending it to another reader in between corrupts both walks. +type NameReader struct { + c cursor + start int64 + ok bool // Module has a name section. + err error +} + +// Reset points nr at f's "name" section, reading through buf. The longest +// function name must fit in buf; a longer one stops the walk with a +// [BufferTooSmallError], and the caller can grow buf and walk again. A module +// with no name section yields no names. +func (nr *NameReader) Reset(f *File, buf []byte) error { + if len(buf) < maxSectionFrame { + return BufferTooSmallError{Need: maxSectionFrame, Have: len(buf)} + } + *nr = NameReader{c: nr.c} + sec, err := f.SectionByName("name") + if err != nil || sec.SectionHeader().ID != SecCustom { + nr.c.reset(f.r, buf, 0, 0) + return nil + } + sh := sec.SectionHeader() + nr.start = int64(sh.Offset) + nr.ok = true + nr.c.reset(f.r, buf, nr.start, int64(sh.End())) + return nil +} + +// Err returns why the last walk stopped early, or nil if it ran to completion +// or the caller stopped it. +func (nr *NameReader) Err() error { return nr.err } + +// FuncNames yields each named function's index and name, in increasing order +// of index as the format requires. name is valid only until the next yield. +// Functions the section does not name are skipped. +func (nr *NameReader) FuncNames(yield func(idx uint32, name []byte) bool) { + nr.nameMap(nameSubFunction, yield) +} + +// DataNames yields each named data segment's index and name, as FuncNames +// does for functions. wasm-ld names segments after the output section they +// came from: ".rodata", ".data". +func (nr *NameReader) DataNames(yield func(idx uint32, name []byte) bool) { + nr.nameMap(nameSubData, yield) +} + +// GlobalNames yields each named global's index and name, as FuncNames does for +// functions. +func (nr *NameReader) GlobalNames(yield func(idx uint32, name []byte) bool) { + nr.nameMap(nameSubGlobal, yield) +} + +// nameMap walks the name map in subsection sub. +func (nr *NameReader) nameMap(sub uint8, yield func(idx uint32, name []byte) bool) { + nr.err = nil + if !nr.ok { + return + } + c := &nr.c + c.err = nil + c.seek(nr.start) + for c.err == nil && c.pos() < c.end { + id := c.u8() + size := c.u32() + subEnd := c.pos() + int64(size) + if c.err != nil { + break + } else if subEnd > c.end { + c.fail(makeFormatErr(uint64(c.pos()), "name subsection exceeds section", size)) + break + } + if id != sub { + c.seek(subEnd) + continue + } + saved := c.end + c.end = subEnd // Bound the name map by its subsection. + // An entry is at least a one-byte index and a one-byte empty name. + count := c.vecLen(2) + var last uint32 + for i := uint32(0); i < count && c.err == nil; i++ { + idx := c.u32() + name := c.view(uint64(c.u32())) + if c.err != nil { + break + } + if i > 0 && idx <= last { + c.fail(makeFormatErr(uint64(c.pos()), "names out of order", idx)) + break + } + last = idx + if !yield(idx, name) { + // Lift the bound here too, or the next walk sees only this + // subsection. + c.end = saved + return + } + } + if c.err == nil && c.pos() != subEnd { + c.fail(makeFormatErr(uint64(c.pos()), "trailing bytes in name map", subEnd-c.pos())) + } + c.end = saved + break // At most one subsection of each kind. + } + nr.err = c.err +} diff --git a/build/xwasm/safeio.go b/build/xwasm/safeio.go new file mode 100644 index 0000000..3b3b4bb --- /dev/null +++ b/build/xwasm/safeio.go @@ -0,0 +1,54 @@ +package xwasm + +import "unsafe" + +// safechunk is an arbitrary limit on how much memory we are willing +// to allocate without concern. +const safechunk = 10 << 20 // 10M + +// sliceCapWithSize returns the capacity to use when allocating a slice. +// After the slice is allocated with the capacity, it should be +// built using append. This will avoid allocating too much memory +// if the capacity is large and incorrect. +// +// A negative result means that the value is always too big. +func sliceCapWithSize(size, c uint64) int { + if int64(c) < 0 || c != uint64(int(c)) { + return -1 + } + if size > 0 && c > (1<<64-1)/size { + return -1 + } + if c*size > safechunk { + c = safechunk / size + if c == 0 { + c = 1 + } + } + return int(c) +} + +// sliceCap is like SliceCapWithSize but using generics. +func sliceCap[E any](c uint64) int { + var v E + size := uint64(unsafe.Sizeof(v)) + return sliceCapWithSize(size, c) +} + +// Grow increases the slice's capacity, if necessary, to guarantee space for +// another n elements. After Grow(n), at least n elements can be appended +// to the slice without another allocation. If n is negative or too large to +// allocate the memory, Grow panics. +func slicesGrow[S ~[]E, E any](s S, n int) S { + if n < 0 { + panic("cannot be negative") + } + if n -= cap(s) - len(s); n > 0 { + s = append(s[:cap(s)], make([]E, n)...)[:len(s)] + } + return s +} + +func aliases[T ~int64 | ~uint64 | ~int](start0, end0, start1, end1 T) bool { + return start0 < end1 && end0 > start1 +} diff --git a/build/xwasm/stream.go b/build/xwasm/stream.go new file mode 100644 index 0000000..9c974f1 --- /dev/null +++ b/build/xwasm/stream.go @@ -0,0 +1,218 @@ +package xwasm + +import ( + "io" + + "github.com/soypat/lexorg" +) + +// cursor reads a section's contents forward through a caller-owned window, +// refusing to run past the section's end. It records the first error it hits +// and then reports zero values, so a decoder can read a whole structure and +// check for failure once at the end. +// +// It is xdwarf's streamCursor adapted to WebAssembly's width-bounded LEB128. +type cursor struct { + wr lexorg.WindowReader + bufLen int // Fill size: the largest span view can present whole. + end int64 // Absolute offset one past the last byte the cursor may read. + err error +} + +// reset binds the cursor to [start, end) of r. It always discards what the +// window holds: the same reader and buffer may now describe different bytes, +// since a File can be re-read from a reader whose contents changed, and a +// caller may have lent the buffer to another reader in between. +func (c *cursor) reset(r io.ReaderAt, buf []byte, start, end int64) { + c.wr.Reset(r, buf, start) + c.wr.Drop() + c.bufLen, c.end, c.err = len(buf), end, nil +} + +// pos reports the absolute offset the next read starts at. +func (c *cursor) pos() int64 { return c.wr.Offset() } + +func (c *cursor) fail(err error) { + if c.err == nil { + c.err = err + } +} + +// seek moves the cursor to off, which costs no read when off is resident. +func (c *cursor) seek(off int64) { + if c.err != nil { + return + } + if off > c.end || off < 0 { + c.fail(makeFormatErr(uint64(c.pos()), "seek past end of section", off)) + return + } + c.wr.Reset(c.wr.ReaderAt(), nil, off) +} + +// skip advances the cursor by n bytes without reading them. +func (c *cursor) skip(n uint64) { + if c.err != nil { + return + } + if n > uint64(c.end-c.pos()) { + c.fail(makeFormatErr(uint64(c.pos()), "skip past end of section", n)) + return + } + c.seek(c.pos() + int64(n)) +} + +func (c *cursor) u8() uint8 { + if c.err != nil { + return 0 + } + if c.pos() >= c.end { + c.fail(makeFormatErr(uint64(c.pos()), "read past end of section", 1)) + return 0 + } + b, err := c.wr.ReadByte() + if err != nil { + if err == io.EOF { + err = io.ErrUnexpectedEOF + } + c.fail(err) + return 0 + } + return b +} + +// uleb reads an unsigned LEB128 of at most bits bits, under the rules of +// [DecodeULEB128]. +func (c *cursor) uleb(bits int) uint64 { + maxBytes := (bits + 6) / 7 + start := c.pos() + var v uint64 + var shift uint + for n := 1; ; n++ { + b := c.u8() + if c.err != nil { + return 0 + } + if n == maxBytes { + if b&0x80 != 0 { + c.fail(makeFormatErr(uint64(start), errLEBTooLong.Error(), bits)) + return 0 + } + if rem := bits - int(shift); rem < 7 && b>>rem != 0 { + c.fail(makeFormatErr(uint64(start), errLEBOverflow.Error(), bits)) + return 0 + } + } + v |= uint64(b&0x7f) << shift + if b&0x80 == 0 { + return v + } + shift += 7 + } +} + +func (c *cursor) u32() uint32 { return uint32(c.uleb(32)) } + +// sleb reads a signed LEB128 of at most bits bits, under the rules of +// [DecodeSLEB128]. +func (c *cursor) sleb(bits int) int64 { + maxBytes := (bits + 6) / 7 + start := c.pos() + var v int64 + var shift uint + for n := 1; ; n++ { + b := c.u8() + if c.err != nil { + return 0 + } + if n == maxBytes { + if b&0x80 != 0 { + c.fail(makeFormatErr(uint64(start), errLEBTooLong.Error(), bits)) + return 0 + } + rem := bits - int(shift) + if top := (b & 0x7f) >> (rem - 1); top != 0 && top != 0x7f>>(rem-1) { + c.fail(makeFormatErr(uint64(start), errLEBOverflow.Error(), bits)) + return 0 + } + } + v |= int64(b&0x7f) << shift + shift += 7 + if b&0x80 == 0 { + if shift < 64 && b&0x40 != 0 { + v |= -1 << shift + } + return v + } + } +} + +// view returns the next n bytes as a subslice of the window, valid until the +// cursor next moves. A span longer than the window fails with a +// [BufferTooSmallError] naming the size that would have held it. +func (c *cursor) view(n uint64) []byte { + if c.err != nil { + return nil + } + if n > uint64(c.end-c.pos()) { + c.fail(makeFormatErr(uint64(c.pos()), "read past end of section", n)) + return nil + } + if n > uint64(c.bufLen) { + c.fail(BufferTooSmallError{Need: int(n), Have: c.bufLen}) + return nil + } + b, err := c.wr.ReadView(int(n)) + if err != nil { + c.fail(err) + return nil + } + return b +} + +// vecLen reads a vector's element count and rejects one that cannot fit in the +// rest of the section when every element takes at least minElem bytes. Counts +// come straight from the file, so this is what stops a loop over a forged count +// from spinning for billions of iterations before it notices the section ended. +func (c *cursor) vecLen(minElem uint64) uint32 { + n := c.u32() + if c.err == nil && uint64(n)*minElem > uint64(c.end-c.pos()) { + c.fail(makeFormatErr(uint64(c.pos()), "vector count exceeds section", n)) + return 0 + } + return n +} + +// skipName skips a length-prefixed byte string. +func (c *cursor) skipName() { + c.skip(uint64(c.u32())) +} + +// limits skips a table or memory limits structure. The flags byte admits the +// shared-memory and memory64 proposals: bit 0 has-max, bit 1 shared, bit 2 +// 64-bit bounds. +func (c *cursor) limits() { + flags := c.u8() + if c.err == nil && flags > 7 { + c.fail(makeFormatErr(uint64(c.pos()-1), "invalid limits flags", flags)) + return + } + bits := 32 + if flags&4 != 0 { + bits = 64 + } + c.uleb(bits) + if flags&1 != 0 { + c.uleb(bits) + } +} + +// valType skips a value or reference type. The typed function references and +// GC proposals prefix a heap type with 0x63 (nullable) or 0x64 (non-null); +// every other type is a single byte. +func (c *cursor) valType() { + switch c.u8() { + case 0x63, 0x64: + c.sleb(33) + } +} diff --git a/build/xwasm/xwasm.go b/build/xwasm/xwasm.go new file mode 100644 index 0000000..f7a31e1 --- /dev/null +++ b/build/xwasm/xwasm.go @@ -0,0 +1,239 @@ +// Package xwasm decodes WebAssembly binary modules in the allocation-conscious +// style of [xelf]. +// +// A module is a header followed by a sequence of sections, each an id and a +// size. There is no section header table: [File.Read] walks the sections once +// and keeps a small fixed-size record of each, never their contents. Everything +// of unbounded size -- section bodies, function bodies, names -- is either read +// through an [io.SectionReader], appended onto a caller-owned buffer, or +// streamed through a caller-owned window by [CodeReader] and [NameReader]. +// +// DWARF in a WebAssembly module lives in custom sections named as in ELF +// (".debug_line", ".debug_info", ...), uncompressed and, in a linked module, +// already relocated. Its addresses are offsets from the start of the Code +// section's payload, the same space [Func.Offset] is expressed in. +// +// [xelf]: github.com/soypat/tinyboot/build/xelf +package xwasm + +import ( + "encoding/binary" + "errors" + "fmt" + "strconv" +) + +const ( + fileBufSize = 512 + + magic uint32 = 0x00 | 'a'<<8 | 's'<<16 | 'm'<<24 + headerSize = 8 // Magic and version. + + // maxSectionFrame is the largest a section's framing can be before its + // payload: id (1), size (5) and, for a custom section, the name length (5). + maxSectionFrame = 1 + 5 + 5 + + // maxSections bounds how many sections [File.Read] will record. A section + // can be as small as two bytes, so the count a file can claim grows with its + // size; real modules carry a dozen standard sections and a handful of + // custom ones. This mirrors xelf's safechunk guard. + maxSections = 1 << 12 +) + +var ( + errBadMagic = errors.New("bad magic number") + errNoSection = errors.New("section not found") + errLEBTooLong = errors.New("LEB128 encoding exceeds maximum length") + errLEBOverflow = errors.New("LEB128 value overflows its type") + errTooManySecs = errors.New("too many sections") + errSectionPastEOF = errors.New("section extends past end of file") +) + +// SectionID identifies the kind of a section. Every id but [SecCustom] may +// appear at most once in a module, in a fixed order. +type SectionID uint8 + +const ( + SecCustom SectionID = 0 + SecType SectionID = 1 + SecImport SectionID = 2 + SecFunction SectionID = 3 + SecTable SectionID = 4 + SecMemory SectionID = 5 + SecGlobal SectionID = 6 + SecExport SectionID = 7 + SecStart SectionID = 8 + SecElement SectionID = 9 + SecCode SectionID = 10 + SecData SectionID = 11 + SecDataCount SectionID = 12 + SecTag SectionID = 13 // Exception handling proposal. +) + +var sectionNames = [...]string{ + SecCustom: "custom", + SecType: "type", + SecImport: "import", + SecFunction: "function", + SecTable: "table", + SecMemory: "memory", + SecGlobal: "global", + SecExport: "export", + SecStart: "start", + SecElement: "element", + SecCode: "code", + SecData: "data", + SecDataCount: "datacount", + SecTag: "tag", +} + +// sectionRank is the position a non-custom section must take in a module. Ids +// were assigned as sections were added to the spec, so they are not in order: +// the tag section sits between memory and global, datacount before code. +var sectionRank = [...]uint8{ + SecType: 1, + SecImport: 2, + SecFunction: 3, + SecTable: 4, + SecMemory: 5, + SecTag: 6, + SecGlobal: 7, + SecExport: 8, + SecStart: 9, + SecElement: 10, + SecDataCount: 11, + SecCode: 12, + SecData: 13, +} + +func (id SectionID) String() string { + if int(id) < len(sectionNames) { + return sectionNames[id] + } + return "SectionID(" + strconv.Itoa(int(id)) + ")" +} + +// Validate reports whether id is a section id this package knows. +func (id SectionID) Validate() error { + if int(id) >= len(sectionNames) { + return makeFormatErr(0, "unknown section id", uint8(id)) + } + return nil +} + +// Header is the module preamble: the magic number, which is checked rather +// than stored, and the binary format version. +type Header struct { + Version uint32 +} + +// DecodeHeader decodes the module preamble at the start of buf. +func DecodeHeader(buf []byte) (h Header, n int, err error) { + if len(buf) < headerSize { + return Header{}, 0, errors.New("too short buffer to decode WASM header") + } + if binary.LittleEndian.Uint32(buf) != magic { + return Header{}, 0, makeFormatErr(0, errBadMagic.Error(), buf[:4]) + } + h.Version = binary.LittleEndian.Uint32(buf[4:]) + if h.Version != 1 { + return Header{}, 0, makeFormatErr(4, "unsupported WASM version", h.Version) + } + return h, headerSize, nil +} + +// Put encodes the module preamble into b. +func (h Header) Put(b []byte) (n int, err error) { + if len(b) < headerSize { + return 0, errors.New("buffer too short to put Header") + } + binary.LittleEndian.PutUint32(b, magic) + binary.LittleEndian.PutUint32(b[4:], h.Version) + return headerSize, nil +} + +// HeaderSize returns the size of the module preamble in bytes. +func (h Header) HeaderSize() int { return headerSize } + +// SectionHeader describes where a section sits in the file. Offsets are +// absolute file offsets. +type SectionHeader struct { + ID SectionID + // Start is the offset of the section's id byte. + Start uint64 + // Offset is the offset of the section's contents: past the id and size + // and, for a custom section, past its name. + Offset uint64 + // Size is the size of the contents in bytes, which for a custom section + // excludes the name. + Size uint64 + // NameOff and NameLen locate a custom section's name. The name is not + // stored: [FileSection.AppendName] reads it when asked, as xelf does with + // string-table names. + NameOff uint64 + NameLen uint32 +} + +// End returns the offset one past the section's last byte. +func (sh SectionHeader) End() uint64 { return sh.Offset + sh.Size } + +// FrameSize returns the bytes the section spends on framing rather than +// contents: its id, size and, for a custom section, its name. +func (sh SectionHeader) FrameSize() uint64 { return sh.Offset - sh.Start } + +// DecodeSectionHeader decodes the framing of the section whose id byte is at +// file offset start and at b[0]. n is the number of framing bytes decoded, +// which excludes a custom section's name: the name need not be in b. +func DecodeSectionHeader(b []byte, start uint64) (sh SectionHeader, n int, err error) { + if len(b) == 0 { + return sh, 0, errors.New("SectionHeader short decode buffer") + } + sh.ID = SectionID(b[0]) + if err = sh.ID.Validate(); err != nil { + return sh, 0, makeFormatErr(start, "unknown section id", b[0]) + } + size, k, err := DecodeULEB128(b[1:], 32) + if err != nil { + return sh, 0, makeFormatErr(start+1, "section size", err) + } + n = 1 + k + sh.Start = start + sh.Offset = start + uint64(n) + sh.Size = size + if sh.ID != SecCustom { + return sh, n, nil + } + namelen, k, err := DecodeULEB128(b[n:], 32) + if err != nil { + return sh, 0, makeFormatErr(start+uint64(n), "custom section name length", err) + } + // The name is part of the payload the size counts. + if uint64(k)+namelen > size { + return sh, 0, makeFormatErr(start+uint64(n), "custom section name exceeds section", namelen) + } + sh.NameOff = sh.Offset + uint64(k) + sh.NameLen = uint32(namelen) + sh.Offset = sh.NameOff + namelen + sh.Size = size - uint64(k) - namelen + return sh, n + k, nil +} + +// BufferTooSmallError reports a caller-owned buffer that cannot hold a single +// item the reader must present whole, and how large it would have to be. A +// caller can grow its buffer to Need and retry. +type BufferTooSmallError struct { + Need int + Have int +} + +func (e BufferTooSmallError) Error() string { + return "xwasm: buffer of " + strconv.Itoa(e.Have) + + " bytes too small, need " + strconv.Itoa(e.Need) +} + +func makeFormatErr(off uint64, msg string, val any) error { + if str, ok := val.(fmt.Stringer); ok { + val = str.String() + } + return fmt.Errorf("WASM format error: %s @ off=%d: %v", msg, off, val) +} diff --git a/build/xwasm/xwasm_test.go b/build/xwasm/xwasm_test.go new file mode 100644 index 0000000..ca69e5e --- /dev/null +++ b/build/xwasm/xwasm_test.go @@ -0,0 +1,531 @@ +package xwasm + +import ( + "bufio" + "bytes" + "errors" + "io" + "os" + "os/exec" + "path/filepath" + "strconv" + "strings" + "testing" +) + +// fixtures is the single TinyGo module the tests read: small, but with a host +// import, data segments, a name section and DWARF. See +// cmd/bindiff/testdata/src/gen.go. +var fixtures = []string{"../../testdata/tiny.wasm"} + +func readFixture(t testing.TB, name string) (*File, []byte) { + t.Helper() + data, err := os.ReadFile(name) + if err != nil { + t.Fatal(err) + } + f := new(File) + if err := f.Read(bytes.NewReader(data)); err != nil { + t.Fatalf("%s: %v", name, err) + } + return f, data +} + +func TestRead_sectionsTileFile(t *testing.T) { + for _, name := range fixtures { + f, data := readFixture(t, name) + // Sections are contiguous from the header to the end of the file. + off := uint64(headerSize) + for i := 0; i < f.NumSections(); i++ { + s, _ := f.Section(i) + sh := s.SectionHeader() + if sh.Start != off { + t.Fatalf("%s: section %d starts at %d, want %d", name, i, sh.Start, off) + } + off = sh.End() + } + if off != uint64(len(data)) { + t.Fatalf("%s: sections end at %d, file is %d bytes", name, off, len(data)) + } + } +} + +func TestRead_sectionNames(t *testing.T) { + f, _ := readFixture(t, fixtures[0]) + for _, name := range []string{"type", "import", "code", "data", "name", ".debug_line", ".debug_str", ".debug_info", "producers"} { + s, err := f.SectionByName(name) + if err != nil { + t.Fatalf("%s: %v", name, err) + } + got, err := s.Name() + if err != nil || got != name { + t.Fatalf("SectionByName(%q).Name() = %q, %v", name, got, err) + } + } + if _, err := f.SectionByName(".debug_nonexistent"); !errors.Is(err, errNoSection) { + t.Fatalf("want errNoSection, got %v", err) + } + code, _ := f.SectionByID(SecCode) + if s, _ := f.SectionByName("code"); s != code { + t.Fatal("SectionByName and SectionByID disagree on code") + } +} + +func TestAppendData(t *testing.T) { + f, data := readFixture(t, fixtures[0]) + for i := 0; i < f.NumSections(); i++ { + s, _ := f.Section(i) + sh := s.SectionHeader() + got, err := s.AppendData([]byte("prefix")) + if err != nil { + t.Fatal(err) + } + want := append([]byte("prefix"), data[sh.Offset:sh.End()]...) + if !bytes.Equal(got, want) { + t.Fatalf("section %d: AppendData mismatch", i) + } + } +} + +// funcNames walks f's functions and joins them with their names. +func funcNames(t testing.TB, f *File) (funcs []Func, names map[uint32]string) { + t.Helper() + var cr CodeReader + var nr NameReader + if err := cr.Reset(f, make([]byte, 64)); err != nil { + t.Fatal(err) + } + if err := nr.Reset(f, make([]byte, 256)); err != nil { + t.Fatal(err) + } + names = make(map[uint32]string) + for idx, name := range nr.FuncNames { + names[idx] = string(name) + } + if err := nr.Err(); err != nil { + t.Fatal(err) + } + for fn := range cr.Funcs { + funcs = append(funcs, fn) + } + if err := cr.Err(); err != nil { + t.Fatal(err) + } + return funcs, names +} + +func TestFuncs_tileCodeSection(t *testing.T) { + for _, name := range fixtures { + f, data := readFixture(t, name) + funcs, names := funcNames(t, f) + if len(funcs) == 0 || len(names) == 0 { + t.Fatalf("%s: %d funcs, %d names", name, len(funcs), len(names)) + } + code, _ := f.SectionByID(SecCode) + sh := code.SectionHeader() + _, countLen, _ := DecodeULEB128(data[sh.Offset:], 32) + // Bodies follow the count back to back and end exactly at the section end. + next := uint64(countLen) + for _, fn := range funcs { + if fn.Offset != next { + t.Fatalf("%s: func %d at %d, want %d", name, fn.Index, fn.Offset, next) + } + next = fn.End() + } + if next != sh.Size { + t.Fatalf("%s: bodies end at %d, code section is %d", name, next, sh.Size) + } + } +} + +// TestFuncs_matchesLLVM cross-checks function placement and naming against +// llvm-nm, which reports a function's address as the file offset of its size +// prefix. +func TestFuncs_matchesLLVM(t *testing.T) { + nm := llvmTool(t, "llvm-nm") + for _, name := range fixtures { + f, _ := readFixture(t, name) + funcs, names := funcNames(t, f) + code, _ := f.SectionByID(SecCode) + base := code.SectionHeader().Offset + + out, err := exec.Command(nm, name).Output() + if err != nil { + t.Fatal(err) + } + want := make(map[string]uint64) + sc := bufio.NewScanner(bytes.NewReader(out)) + for sc.Scan() { + fields := strings.SplitN(sc.Text(), " ", 3) + if len(fields) != 3 || (fields[1] != "t" && fields[1] != "T") { + continue + } + addr, err := strconv.ParseUint(fields[0], 16, 64) + if err != nil { + t.Fatal(err) + } + want[fields[2]] = addr + } + var matched int + for _, fn := range funcs { + fnName, ok := names[fn.Index] + if !ok { + continue + } + addr, ok := want[fnName] + if !ok { + t.Errorf("%s: %s not in llvm-nm output", name, fnName) + continue + } + if got := base + fn.Offset; got != addr { + t.Errorf("%s: %s at %#x, llvm-nm says %#x", name, fnName, got, addr) + } + matched++ + } + if matched != len(want) { + t.Errorf("%s: matched %d functions, llvm-nm lists %d", name, matched, len(want)) + } + } +} + +// llvmTool finds an LLVM tool on PATH or in a sibling TinyGo checkout, and +// skips the test if there is none. +func llvmTool(t *testing.T, name string) string { + t.Helper() + p := llvmToolOptional(name) + if p == "" { + t.Skip(name + " not found") + } + return p +} + +func TestNameReader_bufferTooSmall(t *testing.T) { + f, _ := readFixture(t, fixtures[0]) + var nr NameReader + if err := nr.Reset(f, make([]byte, maxSectionFrame)); err != nil { + t.Fatal(err) + } + for range nr.FuncNames { + } + var small BufferTooSmallError + if !errors.As(nr.Err(), &small) || small.Need <= small.Have { + t.Fatalf("want BufferTooSmallError, got %v", nr.Err()) + } + // Growing to what was asked for gets past that name. + if err := nr.Reset(f, make([]byte, small.Need)); err != nil { + t.Fatal(err) + } + var n int + for range nr.FuncNames { + n++ + } + if n == 0 { + t.Fatal("no names after growing buffer") + } +} + +// A caller breaking out of one name walk must not narrow the next: the name +// map bounds the cursor by its subsection while it walks, and that bound has to +// come off however the walk ends. +func TestNameReader_earlyStop(t *testing.T) { + f, _ := readFixture(t, fixtures[0]) + var nr NameReader + if err := nr.Reset(f, make([]byte, 256)); err != nil { + t.Fatal(err) + } + var want int + for range nr.DataNames { + want++ + } + if nr.Err() != nil || want == 0 { + t.Fatal("fixture has no data names:", nr.Err()) + } + for range nr.FuncNames { + break + } + var got int + for range nr.DataNames { + got++ + } + if nr.Err() != nil || got != want { + t.Fatalf("after early stop got %d data names, want %d (err %v)", got, want, nr.Err()) + } +} + +func TestAllocs(t *testing.T) { + data, err := os.ReadFile(fixtures[0]) + if err != nil { + t.Fatal(err) + } + r := bytes.NewReader(data) + var f File + var cr CodeReader + var nr NameReader + var dr DataReader + cbuf, nbuf, dbuf := make([]byte, 64), make([]byte, 256), make([]byte, 64) + walk := func() { + if err := f.Read(r); err != nil { + t.Fatal(err) + } + if err := cr.Reset(&f, cbuf); err != nil { + t.Fatal(err) + } + if err := nr.Reset(&f, nbuf); err != nil { + t.Fatal(err) + } + for range cr.Funcs { + } + for range nr.FuncNames { + } + for range nr.DataNames { + } + if err := dr.Reset(&f, dbuf); err != nil { + t.Fatal(err) + } + for range dr.Segments { + } + if cr.Err() != nil || nr.Err() != nil || dr.Err() != nil { + t.Fatal(cr.Err(), nr.Err(), dr.Err()) + } + if _, err := f.SectionByName(".debug_line"); err != nil { + t.Fatal(err) + } + } + walk() // Grow the section table. + if allocs := testing.AllocsPerRun(10, walk); allocs != 0 { + t.Fatalf("walk allocates %v times, want 0", allocs) + } +} + +func TestReadRejects(t *testing.T) { + good, err := os.ReadFile(fixtures[0]) + if err != nil { + t.Fatal(err) + } + hdr := []byte("\x00asm\x01\x00\x00\x00") + for _, tc := range []struct { + name string + data []byte + }{ + {"empty", nil}, + {"short header", hdr[:5]}, + {"bad magic", []byte("\x7fELF\x01\x00\x00\x00")}, + {"bad version", []byte("\x00asm\x02\x00\x00\x00")}, + {"unknown section id", append(hdr[:8:8], 14, 0)}, + {"section past EOF", append(hdr[:8:8], 1, 5, 0)}, + {"out of order", append(hdr[:8:8], 3, 0, 1, 0)}, + {"duplicate", append(hdr[:8:8], 1, 0, 1, 0)}, + {"custom name exceeds section", append(hdr[:8:8], 0, 1, 5)}, + {"overlong size LEB", append(hdr[:8:8], 1, 0x80, 0x80, 0x80, 0x80, 0x80, 0)}, + {"truncated", good[:len(good)-1]}, + } { + var f File + if err := f.Read(bytes.NewReader(tc.data)); err == nil { + t.Errorf("%s: no error", tc.name) + } else if f.NumSections() != 0 { + t.Errorf("%s: failed Read left %d sections", tc.name, f.NumSections()) + } + } + // Tag sorts between memory and global despite its id, and datacount + // between element and code. + var f File + ordered := append(hdr[:8:8], 5, 0, 13, 0, 6, 0, 9, 0, 12, 0, 10, 0) + if err := f.Read(bytes.NewReader(ordered)); err != nil { + t.Fatal("spec section order rejected:", err) + } + // A header-only module is valid and empty. + if err := f.Read(bytes.NewReader(hdr)); err != nil || f.NumSections() != 0 { + t.Fatal("empty module:", err, f.NumSections()) + } +} + +func TestHeaderRoundTrip(t *testing.T) { + var b [headerSize]byte + n, err := Header{Version: 1}.Put(b[:]) + if err != nil || n != headerSize { + t.Fatal(n, err) + } + h, n, err := DecodeHeader(b[:]) + if err != nil || n != headerSize || h.Version != 1 { + t.Fatal(h, n, err) + } +} + +func TestDecodeULEB128(t *testing.T) { + for _, tc := range []struct { + b []byte + bits int + v uint64 + n int + err error + }{ + {[]byte{0}, 32, 0, 1, nil}, + {[]byte{0x7f}, 32, 127, 1, nil}, + {[]byte{0x80, 0x01}, 32, 128, 2, nil}, + {[]byte{0xff, 0xff, 0xff, 0xff, 0x0f}, 32, 1<<32 - 1, 5, nil}, + {[]byte{0x80, 0x80, 0x80, 0x80, 0x00}, 32, 0, 5, nil}, // Padded zero is legal. + {[]byte{0xff, 0xff, 0xff, 0xff, 0x1f}, 32, 0, 5, errLEBOverflow}, + {[]byte{0x80, 0x80, 0x80, 0x80, 0x80, 0x00}, 32, 0, 5, errLEBOverflow}, + {[]byte{0x80}, 32, 0, 1, io.ErrUnexpectedEOF}, + {[]byte{0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x01}, 64, 1<<64 - 1, 10, nil}, + {[]byte{0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x02}, 64, 0, 10, errLEBOverflow}, + {[]byte{0x7f}, 7, 127, 1, nil}, + {[]byte{0x80, 0x00}, 7, 0, 1, errLEBTooLong}, + } { + v, n, err := DecodeULEB128(tc.b, tc.bits) + if v != tc.v || n != tc.n || err != tc.err { + t.Errorf("DecodeULEB128(%x, %d) = %d, %d, %v; want %d, %d, %v", tc.b, tc.bits, v, n, err, tc.v, tc.n, tc.err) + } + // The streaming decoder agrees on value and validity. + var c cursor + c.reset(bytes.NewReader(tc.b), make([]byte, 16), 0, int64(len(tc.b))) + sv := c.uleb(tc.bits) + if (c.err == nil) != (tc.err == nil) || sv != tc.v { + t.Errorf("cursor.uleb(%x, %d) = %d, %v; want %d, %v", tc.b, tc.bits, sv, c.err, tc.v, tc.err) + } + } +} + +func TestDecodeSLEB128(t *testing.T) { + for _, tc := range []struct { + b []byte + bits int + v int64 + n int + err error + }{ + {[]byte{0}, 32, 0, 1, nil}, + {[]byte{0x7f}, 32, -1, 1, nil}, + {[]byte{0x3f}, 32, 63, 1, nil}, + {[]byte{0x40}, 32, -64, 1, nil}, + {[]byte{0x80, 0x7f}, 32, -128, 2, nil}, + {[]byte{0xff, 0xff, 0xff, 0xff, 0x07}, 32, 1<<31 - 1, 5, nil}, + {[]byte{0x80, 0x80, 0x80, 0x80, 0x78}, 32, -1 << 31, 5, nil}, + {[]byte{0xff, 0xff, 0xff, 0xff, 0x0f}, 32, 0, 5, errLEBOverflow}, // 2^32-1 as s32. + {[]byte{0x80, 0x80, 0x80, 0x80, 0x70}, 32, 0, 5, errLEBOverflow}, + {[]byte{0x80, 0x80, 0x80, 0x80, 0x80}, 32, 0, 5, errLEBTooLong}, + {[]byte{0x80, 0x80, 0x80, 0x80, 0x70}, 33, -1 << 32, 5, nil}, + {[]byte{0x80, 0x80, 0x80, 0x80, 0x7f}, 33, -1 << 28, 5, nil}, + {[]byte{0x80, 0x80, 0x80, 0x80, 0x60}, 33, 0, 5, errLEBOverflow}, + {[]byte{0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x7f}, 64, -1 << 63, 10, nil}, + {[]byte{0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x00}, 64, 1<<63 - 1, 10, nil}, + {[]byte{0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x01}, 64, 0, 10, errLEBOverflow}, + {[]byte{0xc0}, 32, 0, 1, io.ErrUnexpectedEOF}, + } { + v, n, err := DecodeSLEB128(tc.b, tc.bits) + if v != tc.v || n != tc.n || err != tc.err { + t.Errorf("DecodeSLEB128(%x, %d) = %d, %d, %v; want %d, %d, %v", tc.b, tc.bits, v, n, err, tc.v, tc.n, tc.err) + } + var c cursor + c.reset(bytes.NewReader(tc.b), make([]byte, 16), 0, int64(len(tc.b))) + sv := c.sleb(tc.bits) + if (c.err == nil) != (tc.err == nil) || sv != tc.v { + t.Errorf("cursor.sleb(%x, %d) = %d, %v; want %d, %v", tc.b, tc.bits, sv, c.err, tc.v, tc.err) + } + } +} + +func TestDataSegments(t *testing.T) { + for _, name := range fixtures { + f, data := readFixture(t, name) + var dr DataReader + var nr NameReader + if err := dr.Reset(f, make([]byte, 16)); err != nil { + t.Fatal(err) + } + if err := nr.Reset(f, make([]byte, 256)); err != nil { + t.Fatal(err) + } + names := make(map[uint32]string) + for idx, n := range nr.DataNames { + names[idx] = string(n) + } + if nr.Err() != nil { + t.Fatal(nr.Err()) + } + sec, _ := f.SectionByID(SecData) + sh := sec.SectionHeader() + _, countLen, _ := DecodeULEB128(data[sh.Offset:], 32) + next := uint64(countLen) + addrs := make(map[string]uint64) + for seg := range dr.Segments { + if seg.Offset != next { + t.Fatalf("%s: segment %d at %d, want %d", name, seg.Index, seg.Offset, next) + } + next = seg.End() + if !seg.AddrKnown { + t.Errorf("%s: segment %d has no constant address", name, seg.Index) + } + addrs[names[seg.Index]] = seg.Addr + } + if dr.Err() != nil { + t.Fatal(dr.Err()) + } + if next != sh.Size { + t.Fatalf("%s: segments end at %d, data section is %d", name, next, sh.Size) + } + // wasm-ld names segments after their output sections, and llvm-nm + // reports each at its linear-memory address. + for _, seg := range []string{".rodata", ".data"} { + if _, ok := addrs[seg]; !ok { + t.Errorf("%s: no %s segment among %v", name, seg, names) + } + } + nm := llvmToolOptional("llvm-nm") + if nm == "" { + continue + } + out, err := exec.Command(nm, name).Output() + if err != nil { + t.Fatal(err) + } + sc := bufio.NewScanner(bytes.NewReader(out)) + for sc.Scan() { + fields := strings.SplitN(sc.Text(), " ", 3) + if len(fields) != 3 || fields[1] != "d" { + continue + } + want, _ := strconv.ParseUint(fields[0], 16, 64) + if got, ok := addrs[fields[2]]; ok && got != want { + t.Errorf("%s: segment %s at %#x, llvm-nm says %#x", name, fields[2], got, want) + } + } + } +} + +func TestConstExpr(t *testing.T) { + for _, tc := range []struct { + expr []byte + v uint64 + known bool + fail bool + }{ + {[]byte{opI32Const, 0x80, 0x80, 0x04, opEnd}, 0x10000, true, false}, + {[]byte{opI32Const, 0x7f, opEnd}, 0xffffffff, true, false}, // -1 as an i32 address. + {[]byte{opI64Const, 0x10, opEnd}, 16, true, false}, + {[]byte{opGlobalGet, 0x01, opEnd}, 0, false, false}, + // Extended-const arithmetic is skipped, not evaluated. + {[]byte{opGlobalGet, 0x00, opI32Const, 0x10, opI32Add, opEnd}, 0, false, false}, + {[]byte{opPrefixSIMD, simdV128Const, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, opEnd}, 0, false, false}, + {[]byte{0x20, 0x00, opEnd}, 0, false, true}, // local.get is not constant. + {[]byte{opI32Const, 0x10}, 0, false, true}, // Missing end. + } { + var c cursor + c.reset(bytes.NewReader(tc.expr), make([]byte, 32), 0, int64(len(tc.expr))) + v, known := c.constExpr() + if (c.err != nil) != tc.fail || v != tc.v || known != tc.known { + t.Errorf("constExpr(%x) = %d, %v, %v; want %d, %v, fail=%v", tc.expr, v, known, c.err, tc.v, tc.known, tc.fail) + } + } +} + +func llvmToolOptional(name string) string { + if p, err := exec.LookPath(name); err == nil { + return p + } + p := filepath.Join("..", "..", "..", "tinygo", "llvm-build", "bin", name) + if _, err := os.Stat(p); err == nil { + return p + } + return "" +} diff --git a/cmd/bindiff/bindiff_test.go b/cmd/bindiff/bindiff_test.go index 976a5d6..40be9ad 100644 --- a/cmd/bindiff/bindiff_test.go +++ b/cmd/bindiff/bindiff_test.go @@ -91,6 +91,10 @@ func TestPackageOf(t *testing.T) { {"SystemInit", cPackage}, // The remainder bucket passes through so rollups still reconcile. {Unattributed, Unattributed}, + // Placeholders for unnamed WebAssembly items roll up by kind; their + // bracket must not read as the start of a spelled-out type. + {"[data 12]", "[data]"}, + {"[func 7]", "[func]"}, } { if got := packageOf(tc.sym); got != tc.want { t.Errorf("packageOf(%q)=%q want %q", tc.sym, got, tc.want) diff --git a/cmd/bindiff/dwarf.go b/cmd/bindiff/dwarf.go index 990200e..fb9172c 100644 --- a/cmd/bindiff/dwarf.go +++ b/cmd/bindiff/dwarf.go @@ -72,13 +72,16 @@ func buildLineIndex(sec xdwarf.Sections, size int64, aux []byte) (*lineIndex, er for r := range u.Rows { flush(r.Address) if r.EndSequence { - break + // A unit may hold many sequences -- LLVM's WebAssembly backend + // emits one per function -- so this ends a sequence, not the + // unit. + continue } nameBuf, err = u.AppendFileName(nameBuf[:0], r.File) if err != nil { // A row naming a file outside the table is not worth failing - // the whole binary over; it lands in the remainder instead. - break + // the whole binary over; its bytes land in the remainder. + continue } name := xdwarf.CleanPath(string(nameBuf)) canonical, ok := interned[name] @@ -90,6 +93,9 @@ func buildLineIndex(sec xdwarf.Sections, size int64, aux []byte) (*lineIndex, er havePending = true } flush(0) // A sequence that never ended contributes nothing. + if err := u.Err(); err != nil { + return nil, fmt.Errorf("line unit at %d: %w", off, err) + } off = next } sort.Slice(idx.ranges, func(i, j int) bool { @@ -210,7 +216,12 @@ func profileSource(f *xelf.File, mem, byLine bool) ([]Entry, error) { if err != nil { return nil, err } + return attributeSource(t, idx, byLine), nil +} +// attributeSource splits each placement's bytes among the source ranges that +// cover it. It is the format-independent half of [profileSource]. +func attributeSource(t tiling, idx *lineIndex, byLine bool) []Entry { kind := KindFile if byLine { kind = KindLine @@ -230,6 +241,12 @@ func profileSource(f *xelf.File, mem, byLine bool) ([]Entry, error) { if p.size == 0 { continue } + if t.lineSection != "" && p.section != t.lineSection { + // Not in the line table's address space: its address would alias + // unrelated code. + unmapped[p.section] += p.size + continue + } var covered int64 idx.visitOverlaps(p.addr, p.addr+uint64(p.size), func(r srcRange, n int64) { k := key{file: r.file} @@ -269,7 +286,7 @@ func profileSource(f *xelf.File, mem, byLine bool) ([]Entry, error) { }) } } - return entries, nil + return entries } func sourceName(file string, line uint32, byLine bool) string { diff --git a/cmd/bindiff/main.go b/cmd/bindiff/main.go index f380145..9f2b1cf 100644 --- a/cmd/bindiff/main.go +++ b/cmd/bindiff/main.go @@ -1,5 +1,5 @@ -// Command bindiff characterizes where the bytes of an ELF binary go, and what -// changed between two builds of it. +// Command bindiff characterizes where the bytes of an ELF binary or a +// WebAssembly module go, and what changed between two builds of it. // // It reports at several granularities, from whole segments down to source // lines, and every granularity reconciles: the rows of a report sum to the size @@ -10,6 +10,7 @@ // bindiff -kind=package profile firmware.elf // bindiff -json diff old.elf new.elf // bindiff -threshold=1024 diff old.elf new.elf # exits non-zero on growth +// bindiff -kind=line profile app.wasm package main import ( @@ -105,6 +106,9 @@ func profileFile(path string, flags Flags) ([]Entry, error) { } func profileReader(r io.ReaderAt, size int64, flags Flags) ([]Entry, error) { + if isWasm(r) { + return profileWasm(r, size, flags) + } var f xelf.File if err := f.Read(r); err != nil { return nil, err diff --git a/cmd/bindiff/symbols.go b/cmd/bindiff/symbols.go index 7169661..5497635 100644 --- a/cmd/bindiff/symbols.go +++ b/cmd/bindiff/symbols.go @@ -24,6 +24,74 @@ type tiling struct { remainder map[string]int64 // order preserves section order for deterministic output. order []string + // lineSection, when set, names the only section whose placement addresses + // are in the address space the DWARF line table describes. An ELF binary + // has one address space for everything, and leaves it empty. A WebAssembly + // module does not: line addresses are Code section offsets, and a data + // segment's offset in its own section would alias them. + lineSection string +} + +// claim is one symbol's stake in a section, before overlapping claims are +// resolved. +type claim struct { + name string + start, size uint64 +} + +func newTiling(nsect, nsyms int) tiling { + return tiling{ + placements: make([]placement, 0, nsyms), + remainder: make(map[string]int64, nsect), + order: make([]string, 0, nsect), + } +} + +// tileSection divides a section of size bytes among claims and records what +// they leave over. A section with no claims and no bytes is left out. +func (t *tiling) tileSection(section string, size int64, claims []claim) { + if len(claims) == 0 { + if size > 0 { + t.order = append(t.order, section) + t.remainder[section] = size + } + return + } + t.order = append(t.order, section) + // Tile the section in address order, clamping each symbol so that + // overlapping symbols -- weak aliases and ifunc pairs share an address -- + // are counted once. Without this, a section's symbols can sum past its own + // size and drive the remainder negative. + sort.Slice(claims, func(i, j int) bool { + if claims[i].start != claims[j].start { + return claims[i].start < claims[j].start + } + return claims[i].size > claims[j].size + }) + var cursor uint64 + var attributed int64 + for i, p := range claims { + if i == 0 { + cursor = p.start + } + start, end := p.start, p.start+p.size + if start < cursor { + start = cursor + } + var size int64 + if end > start { + size = int64(end - start) + cursor = end + } + attributed += size + t.placements = append(t.placements, placement{ + name: p.name, + section: section, + addr: start, + size: size, + }) + } + t.remainder[section] = size - attributed } // profileSymbols attributes bytes to individual functions and data objects. @@ -37,6 +105,11 @@ func profileSymbols(f *xelf.File, mem bool) ([]Entry, error) { if err != nil { return nil, err } + return symbolEntries(t), nil +} + +// symbolEntries lists a tiling's placements, then each section's remainder. +func symbolEntries(t tiling) []Entry { entries := make([]Entry, 0, len(t.placements)+len(t.order)) for _, p := range t.placements { entries = append(entries, Entry{ @@ -54,7 +127,7 @@ func profileSymbols(f *xelf.File, mem bool) ([]Entry, error) { }) } } - return entries, nil + return entries } // tileSymbols divides each section's bytes among the symbols that land in it. @@ -92,11 +165,7 @@ func tileSymbols(f *xelf.File, mem bool) (tiling, error) { } // Bucket symbols by section so each section can be tiled independently. - type placed struct { - name string - start, size uint64 - } - bySection := make(map[int][]placed) + bySection := make(map[int][]claim) var strBuf []byte for _, sym := range syms { typ := xelf.SymType(sym.Info & 0xf) @@ -118,60 +187,16 @@ func tileSymbols(f *xelf.File, mem bool) (tiling, error) { if len(strBuf) == 0 { continue } - bySection[idx] = append(bySection[idx], placed{ + bySection[idx] = append(bySection[idx], claim{ name: string(strBuf), start: sym.Value, size: sym.Size, }) } - t.placements = make([]placement, 0, len(syms)) - t.remainder = make(map[string]int64, nsect) - t.order = make([]string, 0, nsect) + t = newTiling(nsect, len(syms)) for idx := 0; idx < nsect; idx++ { - placedSyms := bySection[idx] - if len(placedSyms) == 0 { - if secSizes[idx] > 0 { - t.order = append(t.order, secNames[idx]) - t.remainder[secNames[idx]] = secSizes[idx] - } - continue - } - t.order = append(t.order, secNames[idx]) - // Tile the section in address order, clamping each symbol so that - // overlapping symbols -- weak aliases and ifunc pairs share an address - // -- are counted once. Without this, a section's symbols can sum past - // its own size and drive the remainder negative. - sort.Slice(placedSyms, func(i, j int) bool { - if placedSyms[i].start != placedSyms[j].start { - return placedSyms[i].start < placedSyms[j].start - } - return placedSyms[i].size > placedSyms[j].size - }) - var cursor uint64 - var attributed int64 - for i, p := range placedSyms { - if i == 0 { - cursor = p.start - } - start, end := p.start, p.start+p.size - if start < cursor { - start = cursor - } - var size int64 - if end > start { - size = int64(end - start) - cursor = end - } - attributed += size - t.placements = append(t.placements, placement{ - name: p.name, - section: secNames[idx], - addr: start, - size: size, - }) - } - t.remainder[secNames[idx]] = secSizes[idx] - attributed + t.tileSection(secNames[idx], secSizes[idx], bySection[idx]) } return t, nil } @@ -184,10 +209,15 @@ func profilePackages(f *xelf.File, mem bool) ([]Entry, error) { if err != nil { return nil, err } + return packageEntries(syms, packageOf), nil +} + +// packageEntries rolls symbol entries up by package. +func packageEntries(syms []Entry, pkgOf func(sym string) string) []Entry { order := make([]string, 0, 32) byPkg := make(map[string]*Entry, 32) for _, e := range syms { - name := packageOf(e.Name) + name := pkgOf(e.Name) agg, ok := byPkg[name] if !ok { order = append(order, name) @@ -200,7 +230,7 @@ func profilePackages(f *xelf.File, mem bool) ([]Entry, error) { for _, name := range order { entries = append(entries, *byPkg[name]) } - return entries, nil + return entries } // Buckets for symbols that name no package of their own. @@ -221,8 +251,8 @@ const typePunct = ":{}[](),; " // A package path may itself contain dots ("github.com/soypat/x.Func"), so the // boundary is the first dot after the final slash rather than the last dot. func packageOf(sym string) string { - if sym == Unattributed { - return Unattributed + if pkg, ok := bucketOf(sym); ok { + return pkg } s := sym // The gc linker names its own generated symbols with a colon and no @@ -286,6 +316,22 @@ func packageOf(sym string) string { return pkg } +// bucketOf reports the package of one of bindiff's own synthetic names, which +// are bracketed so no real symbol can collide with them: [unattributed], and +// the placeholders a WebAssembly module's unnamed items get, "[func 12]" and +// "[data 7]". A placeholder's bucket is its kind, so that thousands of unnamed +// items roll up into one row -- "[func]", "[data]" -- rather than each passing +// for a package, or for a spelled-out type because of the bracket. +func bucketOf(sym string) (string, bool) { + if !strings.HasPrefix(sym, "[") { + return "", false + } + if kind, _, ok := strings.Cut(sym, " "); ok { + return kind + "]", true + } + return sym, true +} + func isAllDigits(s string) bool { if s == "" { return false diff --git a/cmd/bindiff/testdata/src/gen.go b/cmd/bindiff/testdata/src/gen.go index cae0b4e..5d7edee 100644 --- a/cmd/bindiff/testdata/src/gen.go +++ b/cmd/bindiff/testdata/src/gen.go @@ -1,4 +1,11 @@ package src -//go:generate tinygo build -o "../blinky-a.elf" -target=pca10040 "./a -//go:generate tinygo build -o "../blinky-b.elf" -target=pca10040 "./b +//go:generate tinygo build -o "../blinky-a.elf" -target=pca10040 "./a" +//go:generate tinygo build -o "../blinky-b.elf" -target=pca10040 "./b" + +// The one WebAssembly fixture, shared with build/xwasm. Every flag after the +// target only shrinks it: the bare wasm-unknown target drops WASI, and the +// rest drop the scheduler, the collector and panic printing. It keeps what the +// tests exercise -- a host import, .rodata and .data segments, the name section +// and DWARF line information -- in about 30 KB, most of it DWARF. +//go:generate tinygo build -o "../../../../testdata/tiny.wasm" -target=wasm-unknown -opt=z -panic=trap -scheduler=none -gc=leaking "./tinywasm" diff --git a/cmd/bindiff/testdata/src/tinywasm/main.go b/cmd/bindiff/testdata/src/tinywasm/main.go new file mode 100644 index 0000000..301828a --- /dev/null +++ b/cmd/bindiff/testdata/src/tinywasm/main.go @@ -0,0 +1,26 @@ +// Package main is the WebAssembly test fixture: small, but with a host +// import, read-only and writable data, and DWARF line information. +package main + +import "unsafe" + +//go:wasmimport env emit +func emit(ptr unsafe.Pointer, n uint32) + +var greetings = [...]string{"hello", "hola", "ciao"} // .rodata + +var counter uint32 = 7 // .data + +func add(a, b uint32) uint32 { return a + b } + +func main() {} + +// run is exported rather than left to main: the bare wasm-unknown target builds +// a library, whose main is never called and would be dropped. +// +//go:wasmexport run +func run() { + counter = add(counter, 1) + s := greetings[counter%uint32(len(greetings))] + emit(unsafe.Pointer(unsafe.StringData(s)), uint32(len(s))) +} diff --git a/cmd/bindiff/wasm.go b/cmd/bindiff/wasm.go new file mode 100644 index 0000000..995d263 --- /dev/null +++ b/cmd/bindiff/wasm.go @@ -0,0 +1,306 @@ +package main + +import ( + "bytes" + "encoding/binary" + "errors" + "fmt" + "io" + "strings" + + "github.com/soypat/tinyboot/build/xdwarf" + "github.com/soypat/tinyboot/build/xwasm" +) + +var errWasmNoDebugLine = errors.New("WebAssembly module has no .debug_line section") + +// isWasm reports whether r holds a WebAssembly module rather than an ELF. +func isWasm(r io.ReaderAt) bool { + var magic [4]byte + n, _ := r.ReadAt(magic[:], 0) + return n == len(magic) && string(magic[:]) == "\x00asm" +} + +// profileWasm is profileReader for a WebAssembly module. +// +// A module has no segments and no load addresses: its code runs from the Code +// section and its linear memory is sized at run time. The segment kind and +// memory mode have no counterpart, and report so rather than guess. +func profileWasm(r io.ReaderAt, size int64, flags Flags) ([]Entry, error) { + var f xwasm.File + if err := f.Read(r); err != nil { + return nil, err + } + if flags.mem { + return nil, errors.New("-mem is not supported for WebAssembly: linear memory is sized at run time") + } + switch flags.kind { + case KindSection: + return profileWasmSections(&f, size) + case KindSymbol, KindPackage, KindFile, KindLine: + default: + return nil, fmt.Errorf("granularity %q does not apply to WebAssembly", flags.kind) + } + t, err := tileWasm(&f) + if err != nil { + return nil, err + } + switch flags.kind { + case KindSymbol: + return symbolEntries(t), nil + case KindPackage: + pkgOf := packageOf + if isGcWasm(&f) { + pkgOf = gcWasmPackageOf + } + return packageEntries(symbolEntries(t), pkgOf), nil + } + idx, err := loadWasmLineIndex(&f) + if err != nil { + return nil, err + } + return attributeSource(t, idx, flags.kind == KindLine), nil +} + +// profileWasmSections attributes a module's bytes to its sections, custom ones +// by their own name. The entries sum exactly to the size of the file: each +// section's framing -- id, size and a custom section's name -- is collected, +// with the module header, under [wasm-headers]. +func profileWasmSections(f *xwasm.File, fileSize int64) ([]Entry, error) { + nsect := f.NumSections() + entries := make([]Entry, 0, nsect+2) + overhead := int64(f.Header().HeaderSize()) + covered := overhead + var name []byte + for i := 0; i < nsect; i++ { + s, err := f.Section(i) + if err != nil { + return nil, err + } + sh := s.SectionHeader() + name, err = s.AppendName(name[:0]) + if err != nil { + return nil, fmt.Errorf("section %d name: %w", i, err) + } + entries = append(entries, Entry{Kind: KindSection, Name: string(name), New: int64(sh.Size)}) + overhead += int64(sh.FrameSize()) + covered += int64(sh.FrameSize() + sh.Size) + } + entries = append(entries, Entry{Kind: KindSection, Name: "[wasm-headers]", New: overhead}) + // Sections tile a module exactly, so this only fires for trailing bytes + // xwasm would have rejected; it is kept so the total cannot silently lie. + if rem := fileSize - covered; rem != 0 { + entries = append(entries, Entry{Kind: KindSection, Name: Unattributed, New: rem}) + } + return entries, nil +} + +// tileWasm divides a module's sections among its functions and data segments. +// Every other section has no symbols and goes wholly to its remainder, as a +// symbol-less ELF section does. +// +// Code placements are Code section offsets and cover a body with its size +// prefix, so only the section's leading function count is left unattributed. +// That is the address space DWARF uses, so the line table can be laid over +// them directly. +func tileWasm(f *xwasm.File) (tiling, error) { + funcNames, err := readNameMap(f, (*xwasm.NameReader).FuncNames) + if err != nil { + return tiling{}, fmt.Errorf("function names: %w", err) + } + dataNames, err := readNameMap(f, (*xwasm.NameReader).DataNames) + if err != nil { + return tiling{}, fmt.Errorf("data segment names: %w", err) + } + var cr xwasm.CodeReader + var dr xwasm.DataReader + // Each reader owns its window; sharing one would let either walk clobber + // the bytes the other believes resident. + var cbuf, dbuf [512]byte + if err := cr.Reset(f, cbuf[:]); err != nil { + return tiling{}, err + } + if err := dr.Reset(f, dbuf[:]); err != nil { + return tiling{}, err + } + + nsect := f.NumSections() + t := newTiling(nsect, int(cr.NumFuncs())+int(dr.NumSegments())) + t.lineSection = xwasm.SecCode.String() + var nameBuf []byte + var claims []claim + for i := 0; i < nsect; i++ { + s, err := f.Section(i) + if err != nil { + return t, err + } + sh := s.SectionHeader() + nameBuf, err = s.AppendName(nameBuf[:0]) + if err != nil { + return t, err + } + claims = claims[:0] + switch sh.ID { + case xwasm.SecCode: + for fn := range cr.Funcs { + claims = append(claims, claim{ + name: nameOr(funcNames, fn.Index, "func"), + start: fn.Offset, + size: fn.End() - fn.Offset, + }) + } + err = cr.Err() + case xwasm.SecData: + for seg := range dr.Segments { + claims = append(claims, claim{ + name: nameOr(dataNames, seg.Index, "data"), + start: seg.Offset, + size: seg.End() - seg.Offset, + }) + } + err = dr.Err() + } + if err != nil { + return t, fmt.Errorf("%s section: %w", sh.ID, err) + } + t.tileSection(string(nameBuf), int64(sh.Size), claims) + } + return t, nil +} + +// nameOr returns the name the name section gives index, or a placeholder +// naming the index space when it gives none. Placeholders are bracketed, like +// bindiff's other synthetic names, so [packageOf] can tell them from symbols. +// The gc toolchain names no data segments at all, so on its modules every +// segment is one of these. +func nameOr(names map[uint32]string, index uint32, space string) string { + if name, ok := names[index]; ok { + return name + } + return fmt.Sprintf("[%s %d]", space, index) +} + +// isGcWasm reports whether f was built by the gc toolchain rather than TinyGo. +// The gc linker writes a go:buildid section when it has a build ID, and names +// itself in the producers section. +func isGcWasm(f *xwasm.File) bool { + if _, err := f.SectionByName("go:buildid"); err == nil { + return true + } + s, err := f.SectionByName("producers") + if err != nil || s.Size() > 4096 { + return false + } + data, err := s.AppendData(nil) + return err == nil && bytes.Contains(data, []byte("Go cmd/compile")) +} + +// gcWasmPackageOf is [packageOf] for a module built by the gc toolchain, whose +// linker rewrites every character of a function name outside [A-Za-z0-9_.] to +// '_' (cmd/link/internal/wasm/asm.go). "github.com/google/gopacket/layers.init" +// arrives as "github.com_google_gopacket_layers.init" and +// "time.(Duration).Truncate" as "time.__Duration_.Truncate", so there is no +// slash left to find the package by. +// +// The package still ends at a recoverable dot. A path keeps dots only in +// elements before its last -- gc escapes the last element's as %2e, arriving +// as "_2e" -- and in practice only in the first, a module's domain. So when a +// name's first '_' comes before its first dot, that dot ends the package: +// "crypto_internal_fips140_nistec_fiat.p521Mul", "net_http.Get". When the dot +// comes first it is ambiguous: "github.com_google_gopacket_layers.init" opens +// with a domain, "time.__Duration_.Truncate" with a whole package. Paths +// without a domain are the standard library's and main, whose roots are a +// short fixed list; anything else is taken to start with a domain and to end +// at the first dot after its first separator. +// +// Paths are reported as mangled rather than restored: '_' is ambiguous between +// a separator and an underscore the path really had. +func gcWasmPackageOf(sym string) string { + if pkg, ok := bucketOf(sym); ok { + return pkg + } + dot := strings.IndexByte(sym, '.') + if dot <= 0 { + return cPackage + } + under := strings.IndexByte(sym, '_') + if under < 0 || under > dot && stdRoots[sym[:dot]] { + return sym[:dot] + } + if under < dot { + return sym[:dot] // Separators precede the first dot: no domain. + } + end := strings.IndexByte(sym[under:], '.') + if end < 0 { + return cPackage + } + return sym[:under+end] +} + +// stdRoots are the first path elements of the standard library, plus main and +// the prefixes of the gc toolchain's own symbols ("type:", "go:"), which arrive +// as "type_" and "go_" and so never reach the lookup with a dot first. +var stdRoots = map[string]bool{ + "archive": true, "arena": true, "bufio": true, "builtin": true, "bytes": true, + "cmp": true, "compress": true, "container": true, "context": true, "crypto": true, + "database": true, "debug": true, "embed": true, "encoding": true, "errors": true, + "expvar": true, "flag": true, "fmt": true, "go": true, "hash": true, "html": true, + "image": true, "index": true, "internal": true, "io": true, "iter": true, "log": true, + "main": true, "maps": true, "math": true, "mime": true, "net": true, "os": true, + "path": true, "plugin": true, "reflect": true, "regexp": true, "runtime": true, + "simd": true, "slices": true, "sort": true, "strconv": true, "strings": true, + "structs": true, "sync": true, "syscall": true, "testing": true, "text": true, + "time": true, "unicode": true, "unique": true, "unsafe": true, "vendor": true, + "weak": true, +} + +// readNameMap collects one of the name section's maps, growing the read +// buffer until it holds the longest name. A module without a name section +// yields an empty map. +func readNameMap(f *xwasm.File, walk func(*xwasm.NameReader, func(uint32, []byte) bool)) (map[uint32]string, error) { + buf := make([]byte, 256) + for { + var nr xwasm.NameReader + if err := nr.Reset(f, buf); err != nil { + return nil, err + } + names := make(map[uint32]string) + walk(&nr, func(idx uint32, name []byte) bool { + names[idx] = string(name) + return true + }) + var small xwasm.BufferTooSmallError + if errors.As(nr.Err(), &small) { + buf = make([]byte, small.Need) + continue + } + return names, nr.Err() + } +} + +// loadWasmLineIndex builds the line index from a module's DWARF custom +// sections. They are stored uncompressed and, in a linked module, already +// relocated, so unlike an ELF they are read in place with no preparation. +func loadWasmLineIndex(f *xwasm.File) (*lineIndex, error) { + line, err := f.SectionByName(".debug_line") + if err != nil { + return nil, errWasmNoDebugLine + } + sec := xdwarf.Sections{Line: line.Open(), ByteOrder: binary.LittleEndian} + if s, err := f.SectionByName(".debug_str"); err == nil { + sec.Str = s.Open() + } + if s, err := f.SectionByName(".debug_line_str"); err == nil { + sec.LineStr = s.Open() + } + aux := make([]byte, defaultAux) + for { + idx, err := buildLineIndex(sec, line.Size(), aux) + var small xdwarf.AuxTooSmallError + if errors.As(err, &small) { + aux = make([]byte, small.Need) + continue + } + return idx, err + } +} diff --git a/cmd/bindiff/wasm_test.go b/cmd/bindiff/wasm_test.go new file mode 100644 index 0000000..edb69ac --- /dev/null +++ b/cmd/bindiff/wasm_test.go @@ -0,0 +1,228 @@ +package main + +import ( + "bufio" + "bytes" + "errors" + "os" + "os/exec" + "path/filepath" + "strconv" + "strings" + "testing" + + "github.com/soypat/tinyboot/build/xwasm" +) + +// wasmFixture is the one WebAssembly module the tests read, a TinyGo build +// with DWARF. See testdata/src/gen.go. What the fixture cannot exercise -- +// a module without DWARF, a gc-built one -- is built in memory instead. +const wasmFixture = "../../testdata/tiny.wasm" + +var wasmKinds = []Kind{KindSection, KindSymbol, KindPackage, KindFile, KindLine} + +// TestWasmProfiles holds the invariants that make a report trustworthy, over +// every kind that applies to a module: +// - sections plus their framing account for every byte of the file; +// - every kind below section describes the section contents, no more or less; +// - symbols tile the code and data sections, leaving only each vector count; +// - a profile diffed against itself changes nothing. +func TestWasmProfiles(t *testing.T) { + sections, size := profileFixture(t, wasmFixture, Flags{kind: KindSection}) + var contents int64 + sectionSize := make(map[string]int64) + for _, e := range sections { + switch e.Name { + case Unattributed: + t.Errorf("%d bytes no section claims", e.New) + case "[wasm-headers]": + default: + contents += e.New + sectionSize[e.Name] = e.New + } + } + if _, total := Total(sections); total != size { + t.Errorf("section profile totals %d, file is %d bytes", total, size) + } + + for _, kind := range wasmKinds { + entries, _ := profileFixture(t, wasmFixture, Flags{kind: kind}) + if kind != KindSection { + if _, total := Total(entries); total != contents { + t.Errorf("kind %v totals %d, section contents total %d", kind, total, contents) + } + } + if n := len(DropUnchanged(Diff(entries, entries))); n != 0 { + t.Errorf("kind %v: self-diff left %d changed rows", kind, n) + } + if kind != KindSymbol { + continue + } + bySection := make(map[string]int64) + var hasMain bool + for _, e := range entries { + bySection[e.Section] += e.New + hasMain = hasMain || strings.HasPrefix(e.Name, "main.") + if e.Name == Unattributed && (e.Section == "code" || e.Section == "data") && e.New > 5 { + t.Errorf("%d bytes of %s unattributed, want only the vector count", e.New, e.Section) + } + } + for name, want := range sectionSize { + if got := bySection[name]; got != want { + t.Errorf("section %q: symbols total %d, section is %d bytes", name, got, want) + } + } + if !hasMain { + t.Error("no main function among symbols") + } + } +} + +// TestWasmLineCoverageMatchesLLVM checks the source kinds attribute exactly the +// code bytes llvm-dwarfdump's line table covers: each row runs to the next row +// of its sequence. +func TestWasmLineCoverageMatchesLLVM(t *testing.T) { + dump := llvmTool(t, "llvm-dwarfdump") + out, err := exec.Command(dump, "--debug-line", wasmFixture).Output() + if err != nil { + t.Fatal(err) + } + var want, prev uint64 + var open bool + sc := bufio.NewScanner(bytes.NewReader(out)) + for sc.Scan() { + line := sc.Text() + if !strings.HasPrefix(line, "0x") || len(line) < 18 || line[18] != ' ' { + continue // Not a row of the line matrix. + } + addr, err := strconv.ParseUint(line[2:18], 16, 64) + if err != nil { + t.Fatal(err) + } + if open && addr > prev { + want += addr - prev + } + prev, open = addr, !strings.Contains(line, "end_sequence") + } + if want == 0 { + t.Fatal("parsed no line table coverage from llvm-dwarfdump") + } + entries, _ := profileFixture(t, wasmFixture, Flags{kind: KindFile}) + var got int64 + for _, e := range entries { + if e.Name != Unattributed { + got += e.New + } + } + if uint64(got) != want { + t.Errorf("file kind attributes %d code bytes, llvm-dwarfdump's line table covers %d", got, want) + } +} + +func llvmTool(t *testing.T, name string) string { + t.Helper() + if p, err := exec.LookPath(name); err == nil { + return p + } + p := filepath.Join("..", "..", "..", "tinygo", "llvm-build", "bin", name) + if _, err := os.Stat(p); err == nil { + return p + } + t.Skip(name + " not found") + return "" +} + +// wasmModule builds a module from the preamble and the given custom sections, +// each a name and payload. +func wasmModule(custom ...string) []byte { + b := []byte("\x00asm\x01\x00\x00\x00") + for i := 0; i+1 < len(custom); i += 2 { + name, payload := custom[i], custom[i+1] + // Lengths here stay under 128, so each LEB128 is one byte. + b = append(b, 0, byte(1+len(name)+len(payload)), byte(len(name))) + b = append(b, name...) + b = append(b, payload...) + } + return b +} + +func TestWasmUnsupported(t *testing.T) { + fixture, err := os.ReadFile(wasmFixture) + if err != nil { + t.Fatal(err) + } + for _, flags := range []Flags{ + {kind: KindSegment}, + {kind: KindSection, mem: true}, + } { + if _, err := profileReader(bytes.NewReader(fixture), 0, flags); err == nil { + t.Errorf("%+v: no error", flags) + } + } + _, err = profileReader(bytes.NewReader(wasmModule()), 0, Flags{kind: KindFile}) + if !errors.Is(err, errWasmNoDebugLine) { + t.Errorf("module without DWARF: got %v, want %v", err, errWasmNoDebugLine) + } +} + +func TestIsGcWasm(t *testing.T) { + fixture, err := os.ReadFile(wasmFixture) + if err != nil { + t.Fatal(err) + } + for _, tc := range []struct { + name string + module []byte + want bool + }{ + {"tinygo fixture", fixture, false}, + {"go:buildid", wasmModule("go:buildid", "abc/def"), true}, + {"producers", wasmModule("producers", "\x01\x0cprocessed-by\x01\x0eGo cmd/compile\x08go1.26.3"), true}, + {"empty", wasmModule(), false}, + } { + var f xwasm.File + if err := f.Read(bytes.NewReader(tc.module)); err != nil { + t.Fatal(tc.name, err) + } + if got := isGcWasm(&f); got != tc.want { + t.Errorf("%s: isGcWasm=%v want %v", tc.name, got, tc.want) + } + } +} + +func TestGcWasmPackageOf(t *testing.T) { + for _, tc := range []struct{ sym, want string }{ + // Single-element paths: the first dot ends the package. + {"runtime.alloc", "runtime"}, + {"time.__Duration_.Truncate", "time"}, + {"runtime.alloc_m.func1", "runtime"}, // '_' after the dot is the identifier's. + // '/' arrives as '_'; the first dot after it ends the path. + {"github.com_google_gopacket_layers.init", "github.com_google_gopacket_layers"}, + {"golang.zx2c4.com_wireguard_device.__Device_.Up", "golang.zx2c4.com_wireguard_device"}, + {"crypto_internal_fips140_nistec_fiat.p521Mul", "crypto_internal_fips140_nistec_fiat"}, + // The last element's dots are escaped as %2e, arriving as _2e. + {"gopkg.in_yaml_2ev3.yaml_emitter_write_double_quoted_scalar", "gopkg.in_yaml_2ev3"}, + {"_rt0_wasm_wasip1", cPackage}, + {"[data 99997]", "[data]"}, + } { + if got := gcWasmPackageOf(tc.sym); got != tc.want { + t.Errorf("gcWasmPackageOf(%q)=%q want %q", tc.sym, got, tc.want) + } + } +} + +// TestGcWasmLocalModule checks a large gc-built module when one is present in +// the untracked local directory. +func TestGcWasmLocalModule(t *testing.T) { + const name = "../../local/netbird.wasm" + if _, err := os.Stat(name); err != nil { + t.Skip("no local module") + } + entries, _ := profileFixture(t, name, Flags{kind: KindPackage}) + for _, e := range entries { + switch e.Name { + case "github", "golang", "google", "gopkg", "[type]": + t.Errorf("package %q (%d bytes): mangled names split at the wrong dot", e.Name, e.New) + } + } +} diff --git a/testdata/tiny.wasm b/testdata/tiny.wasm new file mode 100755 index 0000000..9f28fe4 Binary files /dev/null and b/testdata/tiny.wasm differ