Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
182 changes: 182 additions & 0 deletions build/xwasm/code.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,182 @@
package xwasm

import (
"errors"
"io"
"math"
)

// Func locates one function body in the Code section.
type Func struct {
// Index is the function's index in the module's function index space,
// where imported functions come first: the body's ordinal plus
// [File.NumImportedFuncs]. The name section keys names by this index.
Index uint32
// Offset is the position of the body's size prefix, measured from the start
// of the Code section's contents. DWARF addresses in a WebAssembly module
// are measured from the same point, but a subprogram's DW_AT_low_pc names
// the body proper: [Func.BodyOffset].
Offset uint64
// PrefixSize is the length of the size prefix, 1 to 5 bytes.
PrefixSize uint8
// Size is the body's size in bytes, excluding the prefix.
Size uint64
}

// BodyOffset returns the offset of the body proper -- its local declarations,
// then its code -- from the start of the Code section's contents.
func (fn Func) BodyOffset() uint64 { return fn.Offset + uint64(fn.PrefixSize) }

// End returns the offset one past the body's last byte, from the start of the
// Code section's contents.
func (fn Func) End() uint64 { return fn.BodyOffset() + fn.Size }

// CodeReader walks the function bodies of a module's Code section through a
// caller-owned buffer, without holding anything per function.
//
// A CodeReader is reusable: [CodeReader.Reset] points it at another module and
// keeps the buffer. The buffer belongs to the reader from Reset until the next
// Reset; lending it to another reader in between corrupts both walks.
type CodeReader struct {
c cursor
start int64 // File offset of the Code section's contents.
count uint32
imported uint32
err error
}

// Reset points cr at f's Code section, reading through buf. buf only needs to
// hold a few bytes at a time: bodies are skipped, never read. Larger buffers
// mean fewer reads. A module with no Code section yields no functions.
func (cr *CodeReader) Reset(f *File, buf []byte) error {
if len(buf) < maxSectionFrame {
return BufferTooSmallError{Need: maxSectionFrame, Have: len(buf)}
}
imported, err := f.NumImportedFuncs()
if err != nil {
return err
}
*cr = CodeReader{c: cr.c, imported: imported}
sec, err := f.SectionByID(SecCode)
if err != nil {
cr.c.reset(f.r, buf, 0, 0)
return nil
}
sh := sec.SectionHeader()
cr.start = int64(sh.Offset)
cr.c.reset(f.r, buf, cr.start, int64(sh.End()))
// Every body takes at least its one-byte size prefix and one byte of local
// declaration count.
cr.count = cr.c.vecLen(2)
if cr.c.err != nil {
return cr.c.err
}
if uint64(imported)+uint64(cr.count) > math.MaxUint32 {
return makeFormatErr(sh.Offset, "function index space overflows uint32", cr.count)
}
return nil
}

// NumFuncs returns the number of function bodies in the Code section.
func (cr *CodeReader) NumFuncs() uint32 { return cr.count }

// NumImportedFuncs returns the number of imported functions, which precede the
// module's own in the function index space.
func (cr *CodeReader) NumImportedFuncs() uint32 { return cr.imported }

// Err returns why the last [CodeReader.Funcs] walk stopped early, or nil if it
// ran to completion or the caller stopped it.
func (cr *CodeReader) Err() error { return cr.err }

// Funcs yields each function body in order of index. It may be called again to
// walk the bodies anew.
func (cr *CodeReader) Funcs(yield func(Func) bool) {
c := &cr.c
cr.err = nil
if cr.count == 0 {
return
}
c.err = nil
c.seek(cr.start)
c.u32() // Count, validated by Reset.
for i := uint32(0); i < cr.count; i++ {
prefix := c.pos()
size := c.u32()
if c.err != nil {
break
}
fn := Func{
Index: cr.imported + i,
Offset: uint64(prefix - cr.start),
PrefixSize: uint8(c.pos() - prefix),
Size: uint64(size),
}
c.skip(fn.Size)
if c.err != nil {
break
}
if !yield(fn) {
return
}
}
if c.err == nil && c.pos() != c.end {
c.fail(makeFormatErr(uint64(c.pos()), "trailing bytes after function bodies", c.end-c.pos()))
}
cr.err = c.err
}

// Import descriptor kinds.
const (
importFunc = 0x00
importTable = 0x01
importMemory = 0x02
importGlobal = 0x03
importTag = 0x04
)

// NumImportedFuncs walks the Import section and counts the imported functions.
// A module with no Import section imports none.
func (f *File) NumImportedFuncs() (uint32, error) {
sec, err := f.SectionByID(SecImport)
if errors.Is(err, errNoSection) {
return 0, nil
}
sh := sec.SectionHeader()
var c cursor
c.reset(f.r, f.buf[:], int64(sh.Offset), int64(sh.End()))
// An import is at least two empty names and a kind byte with a one-byte
// operand.
count := c.vecLen(4)
var funcs uint32
for i := uint32(0); i < count && c.err == nil; i++ {
c.skipName() // Module.
c.skipName() // Field.
switch kind := c.u8(); kind {
case importFunc:
c.u32() // Type index.
funcs++
case importTable:
c.valType()
c.limits()
case importMemory:
c.limits()
case importGlobal:
c.valType()
c.u8() // Mutability.
case importTag:
c.u8() // Attribute.
c.u32() // Type index.
default:
if c.err == nil {
c.fail(makeFormatErr(uint64(c.pos()-1), "unknown import kind", kind))
}
}
}
if c.err == nil && c.pos() != c.end {
c.fail(makeFormatErr(uint64(c.pos()), "trailing bytes after imports", c.end-c.pos()))
}
if c.err == io.EOF {
c.err = io.ErrUnexpectedEOF
}
return funcs, c.err
}
179 changes: 179 additions & 0 deletions build/xwasm/data.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,179 @@
package xwasm

// DataSegment locates one segment of the Data section.
type DataSegment struct {
// Index is the segment's index, which the name section keys names by.
Index uint32
// Offset is the position of the segment's first byte -- its flags --
// measured from the start of the Data section's contents.
Offset uint64
// HeaderSize is the bytes the segment spends before its initializer: flags,
// memory index, offset expression and byte count.
HeaderSize uint32
// Size is the length of the initializer bytes.
Size uint64
// Passive marks a segment copied into memory only by memory.init, which
// has no address of its own.
Passive bool
// Addr is the linear-memory address an active segment is placed at, valid
// when AddrKnown is set: the offset expression is a single constant. An
// expression reading a global is only resolved at instantiation.
Addr uint64
AddrKnown bool
}

// DataOffset returns the offset of the initializer bytes from the start of the
// Data section's contents.
func (s DataSegment) DataOffset() uint64 { return s.Offset + uint64(s.HeaderSize) }

// End returns the offset one past the segment's last byte, from the start of
// the Data section's contents.
func (s DataSegment) End() uint64 { return s.DataOffset() + s.Size }

// DataReader walks the segments of a module's Data section through a
// caller-owned buffer, without holding anything per segment.
//
// A DataReader is reusable: [DataReader.Reset] points it at another module and
// keeps the buffer. The buffer belongs to the reader from Reset until the next
// Reset; lending it to another reader in between corrupts both walks.
type DataReader struct {
c cursor
start int64
count uint32
err error
}

// Reset points dr at f's Data section, reading through buf. As with
// [CodeReader], segment contents are skipped rather than read, so buf needs to
// hold only a few bytes. A module with no Data section yields no segments.
func (dr *DataReader) Reset(f *File, buf []byte) error {
if len(buf) < maxSectionFrame {
return BufferTooSmallError{Need: maxSectionFrame, Have: len(buf)}
}
*dr = DataReader{c: dr.c}
sec, err := f.SectionByID(SecData)
if err != nil {
dr.c.reset(f.r, buf, 0, 0)
return nil
}
sh := sec.SectionHeader()
dr.start = int64(sh.Offset)
dr.c.reset(f.r, buf, dr.start, int64(sh.End()))
// The smallest segment is passive and empty: flags and a zero count.
dr.count = dr.c.vecLen(2)
return dr.c.err
}

// NumSegments returns the number of segments in the Data section.
func (dr *DataReader) NumSegments() uint32 { return dr.count }

// Err returns why the last [DataReader.Segments] walk stopped early, or nil if
// it ran to completion or the caller stopped it.
func (dr *DataReader) Err() error { return dr.err }

// Segments yields each data segment in order of index. It may be called again
// to walk the segments anew.
func (dr *DataReader) Segments(yield func(DataSegment) bool) {
c := &dr.c
dr.err = nil
if dr.count == 0 {
return
}
c.err = nil
c.seek(dr.start)
c.u32() // Count, validated by Reset.
for i := uint32(0); i < dr.count; i++ {
first := c.pos()
seg := DataSegment{Index: i, Offset: uint64(first - dr.start)}
switch flags := c.u32(); flags {
case 0:
seg.Addr, seg.AddrKnown = c.constExpr()
case 1:
seg.Passive = true
case 2:
c.u32() // Memory index.
seg.Addr, seg.AddrKnown = c.constExpr()
default:
if c.err == nil {
c.fail(makeFormatErr(uint64(first), "invalid data segment flags", flags))
}
}
seg.Size = uint64(c.u32())
if c.err != nil {
break
}
seg.HeaderSize = uint32(c.pos() - first)
c.skip(seg.Size)
if c.err != nil {
break
}
if !yield(seg) {
return
}
}
if c.err == nil && c.pos() != c.end {
c.fail(makeFormatErr(uint64(c.pos()), "trailing bytes after data segments", c.end-c.pos()))
}
dr.err = c.err
}

// Opcodes that may appear in a constant expression, with the extended-const
// proposal's integer arithmetic.
const (
opEnd = 0x0b
opGlobalGet = 0x23
opI32Const = 0x41
opI64Const = 0x42
opF32Const = 0x43
opF64Const = 0x44
opI32Add = 0x6a
opI32Sub = 0x6b
opI32Mul = 0x6c
opI64Add = 0x7c
opI64Sub = 0x7d
opI64Mul = 0x7e
opRefNull = 0xd0
opRefFunc = 0xd2
opPrefixSIMD = 0xfd
simdV128Const = 12
)

// constExpr skips a constant expression through its end opcode. It reports the
// expression's value when the expression is a lone integer constant, which is
// how a linker places a data segment.
func (c *cursor) constExpr() (v uint64, known bool) {
start := c.pos()
for n := 0; c.err == nil; n++ {
op := c.u8()
switch op {
case opEnd:
if !known || n != 1 {
return 0, false // Not a lone constant: resolved at instantiation.
}
return v, true
case opI32Const:
v, known = uint64(uint32(c.sleb(32))), true
case opI64Const:
v, known = uint64(c.sleb(64)), true
case opF32Const:
c.skip(4)
case opF64Const:
c.skip(8)
case opGlobalGet, opRefFunc:
c.u32()
case opRefNull:
c.sleb(33) // Heap type.
case opI32Add, opI32Sub, opI32Mul, opI64Add, opI64Sub, opI64Mul:
case opPrefixSIMD:
if sub := c.u32(); sub != simdV128Const && c.err == nil {
c.fail(makeFormatErr(uint64(start), "non-constant SIMD instruction in constant expression", sub))
}
c.skip(16)
default:
if c.err == nil {
c.fail(makeFormatErr(uint64(c.pos()-1), "non-constant instruction in constant expression", op))
}
}
}
return 0, false
}
Loading
Loading