diff --git a/Cargo.lock b/Cargo.lock index 863ead9..3a5bb5c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -169,3 +169,7 @@ version = "0.1.0" dependencies = [ "we-encoding", ] + +[[package]] +name = "we-wasm" +version = "0.1.0" diff --git a/Cargo.toml b/Cargo.toml index 0d87a56..2ee3785 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -17,6 +17,7 @@ members = [ "crates/js", "crates/svg", "crates/memory", + "crates/wasm", "crates/browser", "crates/e2e", "tools/generate_icon", diff --git a/crates/wasm/Cargo.toml b/crates/wasm/Cargo.toml new file mode 100644 index 0000000..c7de2ce --- /dev/null +++ b/crates/wasm/Cargo.toml @@ -0,0 +1,7 @@ +[package] +name = "we-wasm" +version = "0.1.0" +edition.workspace = true +license.workspace = true + +[dependencies] diff --git a/crates/wasm/src/decode.rs b/crates/wasm/src/decode.rs new file mode 100644 index 0000000..c9b6393 --- /dev/null +++ b/crates/wasm/src/decode.rs @@ -0,0 +1,363 @@ +//! Opcode → `Instruction` decoder. + +use crate::error::{ParseError, ParseErrorKind}; +use crate::instruction::Instruction; +use crate::reader::Reader; +use crate::types::{read_blocktype, read_reftype, read_valtype_vec, MemArg}; + +const MAX_ALIGN_EXPONENT: u32 = 64; + +fn read_memarg(r: &mut Reader<'_>) -> Result { + let start = r.position(); + let align = r.read_u32()?; + if align > MAX_ALIGN_EXPONENT { + return Err(ParseError::new( + ParseErrorKind::InvalidAlignment(align), + start, + )); + } + let offset = r.read_u32()?; + Ok(MemArg { + align, + offset, + memory: 0, + }) +} + +/// Decode one instruction starting at the reader's current position. Returns +/// the decoded variant; the caller is responsible for tracking structured +/// nesting (`Block`/`Loop`/`If`/`Else`/`End`). +pub fn decode_instruction(r: &mut Reader<'_>) -> Result { + let op_pos = r.position(); + let op = r.read_u8()?; + Ok(match op { + // Control + 0x00 => Instruction::Unreachable, + 0x01 => Instruction::Nop, + 0x02 => Instruction::Block(read_blocktype(r)?), + 0x03 => Instruction::Loop(read_blocktype(r)?), + 0x04 => Instruction::If(read_blocktype(r)?), + 0x05 => Instruction::Else, + 0x0B => Instruction::End, + 0x0C => Instruction::Br(r.read_u32()?), + 0x0D => Instruction::BrIf(r.read_u32()?), + 0x0E => { + let n = r.read_u32()? as usize; + let mut labels = Vec::with_capacity(n); + for _ in 0..n { + labels.push(r.read_u32()?); + } + let default = r.read_u32()?; + Instruction::BrTable { labels, default } + } + 0x0F => Instruction::Return, + 0x10 => Instruction::Call(r.read_u32()?), + 0x11 => { + let type_index = r.read_u32()?; + let table_index = r.read_u32()?; + Instruction::CallIndirect { + type_index, + table_index, + } + } + + // Reference + 0xD0 => Instruction::RefNull(read_reftype(r)?), + 0xD1 => Instruction::RefIsNull, + 0xD2 => Instruction::RefFunc(r.read_u32()?), + + // Parametric + 0x1A => Instruction::Drop, + 0x1B => Instruction::Select, + 0x1C => Instruction::SelectTyped(read_valtype_vec(r)?), + + // Variable + 0x20 => Instruction::LocalGet(r.read_u32()?), + 0x21 => Instruction::LocalSet(r.read_u32()?), + 0x22 => Instruction::LocalTee(r.read_u32()?), + 0x23 => Instruction::GlobalGet(r.read_u32()?), + 0x24 => Instruction::GlobalSet(r.read_u32()?), + + // Table get/set + 0x25 => Instruction::TableGet(r.read_u32()?), + 0x26 => Instruction::TableSet(r.read_u32()?), + + // Memory loads + 0x28 => Instruction::I32Load(read_memarg(r)?), + 0x29 => Instruction::I64Load(read_memarg(r)?), + 0x2A => Instruction::F32Load(read_memarg(r)?), + 0x2B => Instruction::F64Load(read_memarg(r)?), + 0x2C => Instruction::I32Load8S(read_memarg(r)?), + 0x2D => Instruction::I32Load8U(read_memarg(r)?), + 0x2E => Instruction::I32Load16S(read_memarg(r)?), + 0x2F => Instruction::I32Load16U(read_memarg(r)?), + 0x30 => Instruction::I64Load8S(read_memarg(r)?), + 0x31 => Instruction::I64Load8U(read_memarg(r)?), + 0x32 => Instruction::I64Load16S(read_memarg(r)?), + 0x33 => Instruction::I64Load16U(read_memarg(r)?), + 0x34 => Instruction::I64Load32S(read_memarg(r)?), + 0x35 => Instruction::I64Load32U(read_memarg(r)?), + // Memory stores + 0x36 => Instruction::I32Store(read_memarg(r)?), + 0x37 => Instruction::I64Store(read_memarg(r)?), + 0x38 => Instruction::F32Store(read_memarg(r)?), + 0x39 => Instruction::F64Store(read_memarg(r)?), + 0x3A => Instruction::I32Store8(read_memarg(r)?), + 0x3B => Instruction::I32Store16(read_memarg(r)?), + 0x3C => Instruction::I64Store8(read_memarg(r)?), + 0x3D => Instruction::I64Store16(read_memarg(r)?), + 0x3E => Instruction::I64Store32(read_memarg(r)?), + // Memory size/grow + 0x3F => Instruction::MemorySize(r.read_u32()?), + 0x40 => Instruction::MemoryGrow(r.read_u32()?), + + // Numeric consts + 0x41 => Instruction::I32Const(r.read_s32()?), + 0x42 => Instruction::I64Const(r.read_s64()?), + 0x43 => Instruction::F32Const(r.read_f32()?), + 0x44 => Instruction::F64Const(r.read_f64()?), + + // i32 comparisons + 0x45 => Instruction::I32Eqz, + 0x46 => Instruction::I32Eq, + 0x47 => Instruction::I32Ne, + 0x48 => Instruction::I32LtS, + 0x49 => Instruction::I32LtU, + 0x4A => Instruction::I32GtS, + 0x4B => Instruction::I32GtU, + 0x4C => Instruction::I32LeS, + 0x4D => Instruction::I32LeU, + 0x4E => Instruction::I32GeS, + 0x4F => Instruction::I32GeU, + + // i64 comparisons + 0x50 => Instruction::I64Eqz, + 0x51 => Instruction::I64Eq, + 0x52 => Instruction::I64Ne, + 0x53 => Instruction::I64LtS, + 0x54 => Instruction::I64LtU, + 0x55 => Instruction::I64GtS, + 0x56 => Instruction::I64GtU, + 0x57 => Instruction::I64LeS, + 0x58 => Instruction::I64LeU, + 0x59 => Instruction::I64GeS, + 0x5A => Instruction::I64GeU, + + // f32 comparisons + 0x5B => Instruction::F32Eq, + 0x5C => Instruction::F32Ne, + 0x5D => Instruction::F32Lt, + 0x5E => Instruction::F32Gt, + 0x5F => Instruction::F32Le, + 0x60 => Instruction::F32Ge, + + // f64 comparisons + 0x61 => Instruction::F64Eq, + 0x62 => Instruction::F64Ne, + 0x63 => Instruction::F64Lt, + 0x64 => Instruction::F64Gt, + 0x65 => Instruction::F64Le, + 0x66 => Instruction::F64Ge, + + // i32 numeric + 0x67 => Instruction::I32Clz, + 0x68 => Instruction::I32Ctz, + 0x69 => Instruction::I32Popcnt, + 0x6A => Instruction::I32Add, + 0x6B => Instruction::I32Sub, + 0x6C => Instruction::I32Mul, + 0x6D => Instruction::I32DivS, + 0x6E => Instruction::I32DivU, + 0x6F => Instruction::I32RemS, + 0x70 => Instruction::I32RemU, + 0x71 => Instruction::I32And, + 0x72 => Instruction::I32Or, + 0x73 => Instruction::I32Xor, + 0x74 => Instruction::I32Shl, + 0x75 => Instruction::I32ShrS, + 0x76 => Instruction::I32ShrU, + 0x77 => Instruction::I32Rotl, + 0x78 => Instruction::I32Rotr, + + // i64 numeric + 0x79 => Instruction::I64Clz, + 0x7A => Instruction::I64Ctz, + 0x7B => Instruction::I64Popcnt, + 0x7C => Instruction::I64Add, + 0x7D => Instruction::I64Sub, + 0x7E => Instruction::I64Mul, + 0x7F => Instruction::I64DivS, + 0x80 => Instruction::I64DivU, + 0x81 => Instruction::I64RemS, + 0x82 => Instruction::I64RemU, + 0x83 => Instruction::I64And, + 0x84 => Instruction::I64Or, + 0x85 => Instruction::I64Xor, + 0x86 => Instruction::I64Shl, + 0x87 => Instruction::I64ShrS, + 0x88 => Instruction::I64ShrU, + 0x89 => Instruction::I64Rotl, + 0x8A => Instruction::I64Rotr, + + // f32 numeric + 0x8B => Instruction::F32Abs, + 0x8C => Instruction::F32Neg, + 0x8D => Instruction::F32Ceil, + 0x8E => Instruction::F32Floor, + 0x8F => Instruction::F32Trunc, + 0x90 => Instruction::F32Nearest, + 0x91 => Instruction::F32Sqrt, + 0x92 => Instruction::F32Add, + 0x93 => Instruction::F32Sub, + 0x94 => Instruction::F32Mul, + 0x95 => Instruction::F32Div, + 0x96 => Instruction::F32Min, + 0x97 => Instruction::F32Max, + 0x98 => Instruction::F32Copysign, + + // f64 numeric + 0x99 => Instruction::F64Abs, + 0x9A => Instruction::F64Neg, + 0x9B => Instruction::F64Ceil, + 0x9C => Instruction::F64Floor, + 0x9D => Instruction::F64Trunc, + 0x9E => Instruction::F64Nearest, + 0x9F => Instruction::F64Sqrt, + 0xA0 => Instruction::F64Add, + 0xA1 => Instruction::F64Sub, + 0xA2 => Instruction::F64Mul, + 0xA3 => Instruction::F64Div, + 0xA4 => Instruction::F64Min, + 0xA5 => Instruction::F64Max, + 0xA6 => Instruction::F64Copysign, + + // Conversions + 0xA7 => Instruction::I32WrapI64, + 0xA8 => Instruction::I32TruncF32S, + 0xA9 => Instruction::I32TruncF32U, + 0xAA => Instruction::I32TruncF64S, + 0xAB => Instruction::I32TruncF64U, + 0xAC => Instruction::I64ExtendI32S, + 0xAD => Instruction::I64ExtendI32U, + 0xAE => Instruction::I64TruncF32S, + 0xAF => Instruction::I64TruncF32U, + 0xB0 => Instruction::I64TruncF64S, + 0xB1 => Instruction::I64TruncF64U, + 0xB2 => Instruction::F32ConvertI32S, + 0xB3 => Instruction::F32ConvertI32U, + 0xB4 => Instruction::F32ConvertI64S, + 0xB5 => Instruction::F32ConvertI64U, + 0xB6 => Instruction::F32DemoteF64, + 0xB7 => Instruction::F64ConvertI32S, + 0xB8 => Instruction::F64ConvertI32U, + 0xB9 => Instruction::F64ConvertI64S, + 0xBA => Instruction::F64ConvertI64U, + 0xBB => Instruction::F64PromoteF32, + 0xBC => Instruction::I32ReinterpretF32, + 0xBD => Instruction::I64ReinterpretF64, + 0xBE => Instruction::F32ReinterpretI32, + 0xBF => Instruction::F64ReinterpretI64, + + // Sign extension + 0xC0 => Instruction::I32Extend8S, + 0xC1 => Instruction::I32Extend16S, + 0xC2 => Instruction::I64Extend8S, + 0xC3 => Instruction::I64Extend16S, + 0xC4 => Instruction::I64Extend32S, + + // 0xFC-prefixed: saturating conversions + bulk memory + table extensions + 0xFC => { + let sub_pos = r.position(); + let sub = r.read_u32()?; + match sub { + 0 => Instruction::I32TruncSatF32S, + 1 => Instruction::I32TruncSatF32U, + 2 => Instruction::I32TruncSatF64S, + 3 => Instruction::I32TruncSatF64U, + 4 => Instruction::I64TruncSatF32S, + 5 => Instruction::I64TruncSatF32U, + 6 => Instruction::I64TruncSatF64S, + 7 => Instruction::I64TruncSatF64U, + 8 => { + let data_index = r.read_u32()?; + let memory = r.read_u32()?; + Instruction::MemoryInit { data_index, memory } + } + 9 => Instruction::DataDrop(r.read_u32()?), + 10 => { + let dst_memory = r.read_u32()?; + let src_memory = r.read_u32()?; + Instruction::MemoryCopy { + dst_memory, + src_memory, + } + } + 11 => Instruction::MemoryFill(r.read_u32()?), + 12 => { + let elem_index = r.read_u32()?; + let table_index = r.read_u32()?; + Instruction::TableInit { + elem_index, + table_index, + } + } + 13 => Instruction::ElemDrop(r.read_u32()?), + 14 => { + let dst_table = r.read_u32()?; + let src_table = r.read_u32()?; + Instruction::TableCopy { + dst_table, + src_table, + } + } + 15 => Instruction::TableGrow(r.read_u32()?), + 16 => Instruction::TableSize(r.read_u32()?), + 17 => Instruction::TableFill(r.read_u32()?), + other => { + return Err(ParseError::new( + ParseErrorKind::UnknownOpcode { + opcode: 0xFC00 | (other as u16 & 0x00FF), + }, + sub_pos, + )) + } + } + } + + other => { + return Err(ParseError::new( + ParseErrorKind::UnknownOpcode { + opcode: other as u16, + }, + op_pos, + )) + } + }) +} + +/// Decode a sequence of instructions terminated by an `End` opcode that closes +/// the outermost block. Returns the instructions including the terminating +/// `End`. Honors structured nesting so a nested `End` does not terminate the +/// outer sequence. +pub fn decode_expr(r: &mut Reader<'_>) -> Result, ParseError> { + let mut out = Vec::new(); + let mut depth: usize = 1; + loop { + let instr = decode_instruction(r)?; + match &instr { + Instruction::Block(_) | Instruction::Loop(_) | Instruction::If(_) => { + depth += 1; + } + Instruction::End => { + depth -= 1; + out.push(instr); + if depth == 0 { + return Ok(out); + } + continue; + } + _ => {} + } + out.push(instr); + } +} diff --git a/crates/wasm/src/error.rs b/crates/wasm/src/error.rs new file mode 100644 index 0000000..005afcc --- /dev/null +++ b/crates/wasm/src/error.rs @@ -0,0 +1,154 @@ +//! Structured parser errors with byte offsets and diagnostics. + +use core::fmt; + +/// Kinds of failure the binary decoder can produce. +/// +/// Each variant intentionally carries enough context (offsets, magic bytes, +/// expected-vs-actual) that downstream tooling can render a helpful diagnostic +/// without parsing free-form strings. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum ParseErrorKind { + UnexpectedEof, + InvalidMagic([u8; 4]), + UnsupportedVersion(u32), + Leb128Overflow, + Leb128Malformed, + InvalidUtf8, + InvalidValType(u8), + InvalidRefType(u8), + InvalidBlockType(i64), + InvalidFuncTypeTag(u8), + InvalidLimitsTag(u8), + InvalidMutabilityFlag(u8), + InvalidImportDescriptor(u8), + InvalidExportDescriptor(u8), + InvalidElementKind(u8), + InvalidElementSegmentFlags(u32), + InvalidDataSegmentFlags(u32), + UnknownOpcode { + opcode: u16, + }, + UnknownSectionId(u8), + DuplicateSection(u8), + SectionTooShort { + id: u8, + declared: u32, + remaining: u64, + }, + SectionTrailingBytes { + id: u8, + leftover: u64, + }, + DataCountMismatch { + declared: u32, + actual: u32, + }, + InvalidAlignment(u32), + IntegerOverflow, + InvalidName, + InvalidFunctionBody, + Other(&'static str), +} + +impl fmt::Display for ParseErrorKind { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + ParseErrorKind::UnexpectedEof => write!(f, "unexpected end of input"), + ParseErrorKind::InvalidMagic(m) => write!( + f, + "invalid magic bytes {:02x} {:02x} {:02x} {:02x} (expected 00 61 73 6d)", + m[0], m[1], m[2], m[3] + ), + ParseErrorKind::UnsupportedVersion(v) => { + write!(f, "unsupported module version {} (expected 1)", v) + } + ParseErrorKind::Leb128Overflow => write!(f, "LEB128 value overflowed target width"), + ParseErrorKind::Leb128Malformed => write!(f, "LEB128 sequence is malformed"), + ParseErrorKind::InvalidUtf8 => write!(f, "invalid UTF-8 in name"), + ParseErrorKind::InvalidValType(b) => write!(f, "invalid valtype byte 0x{:02x}", b), + ParseErrorKind::InvalidRefType(b) => write!(f, "invalid reftype byte 0x{:02x}", b), + ParseErrorKind::InvalidBlockType(v) => write!(f, "invalid block type encoding {}", v), + ParseErrorKind::InvalidFuncTypeTag(b) => { + write!(f, "invalid functype tag 0x{:02x} (expected 0x60)", b) + } + ParseErrorKind::InvalidLimitsTag(b) => write!(f, "invalid limits tag 0x{:02x}", b), + ParseErrorKind::InvalidMutabilityFlag(b) => { + write!(f, "invalid global mutability flag 0x{:02x}", b) + } + ParseErrorKind::InvalidImportDescriptor(b) => { + write!(f, "invalid import descriptor 0x{:02x}", b) + } + ParseErrorKind::InvalidExportDescriptor(b) => { + write!(f, "invalid export descriptor 0x{:02x}", b) + } + ParseErrorKind::InvalidElementKind(b) => { + write!(f, "invalid element kind 0x{:02x}", b) + } + ParseErrorKind::InvalidElementSegmentFlags(v) => { + write!(f, "invalid element segment flags {}", v) + } + ParseErrorKind::InvalidDataSegmentFlags(v) => { + write!(f, "invalid data segment flags {}", v) + } + ParseErrorKind::UnknownOpcode { opcode } => { + write!(f, "unknown opcode 0x{:04x}", opcode) + } + ParseErrorKind::UnknownSectionId(id) => { + write!(f, "unknown section id {}", id) + } + ParseErrorKind::DuplicateSection(id) => { + write!(f, "duplicate section id {}", id) + } + ParseErrorKind::SectionTooShort { + id, + declared, + remaining, + } => write!( + f, + "section {} declares {} bytes but only {} remain", + id, declared, remaining + ), + ParseErrorKind::SectionTrailingBytes { id, leftover } => write!( + f, + "section {} has {} trailing bytes after its declared content", + id, leftover + ), + ParseErrorKind::DataCountMismatch { declared, actual } => write!( + f, + "data count section declared {} segments but data section has {}", + declared, actual + ), + ParseErrorKind::InvalidAlignment(a) => { + write!(f, "alignment exponent {} is too large", a) + } + ParseErrorKind::IntegerOverflow => write!(f, "integer overflow"), + ParseErrorKind::InvalidName => write!(f, "invalid name encoding"), + ParseErrorKind::InvalidFunctionBody => { + write!(f, "function body did not consume its declared bytes") + } + ParseErrorKind::Other(s) => f.write_str(s), + } + } +} + +/// A parse failure: kind plus the byte offset within the original module. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ParseError { + pub kind: ParseErrorKind, + pub offset: usize, +} + +impl ParseError { + pub const fn new(kind: ParseErrorKind, offset: usize) -> Self { + Self { kind, offset } + } +} + +impl fmt::Display for ParseError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "{} at byte offset {}", self.kind, self.offset) + } +} + +impl std::error::Error for ParseError {} diff --git a/crates/wasm/src/instruction.rs b/crates/wasm/src/instruction.rs new file mode 100644 index 0000000..2c84508 --- /dev/null +++ b/crates/wasm/src/instruction.rs @@ -0,0 +1,251 @@ +//! Decoded WebAssembly instruction representation. +//! +//! Covers all numeric, parametric, variable, table, memory, control, and +//! reference instructions from the MVP plus the post-MVP proposals listed in +//! `PLAN.md` (sign-extension, non-trapping float-to-int, reference types, bulk +//! memory). SIMD (`0xFD` prefix) and threads (`0xFE` prefix) are intentionally +//! out of scope here — the parser rejects them as unknown opcodes. + +use crate::types::{BlockType, MemArg, RefType, ValType}; + +/// One decoded instruction. Variants with operands inline their immediates +/// directly; everything else carries an empty payload. +#[derive(Debug, Clone, PartialEq)] +pub enum Instruction { + // ---- Control instructions --------------------------------------------- + Unreachable, + Nop, + Block(BlockType), + Loop(BlockType), + If(BlockType), + Else, + End, + Br(u32), + BrIf(u32), + BrTable { labels: Vec, default: u32 }, + Return, + Call(u32), + CallIndirect { type_index: u32, table_index: u32 }, + + // ---- Reference instructions ------------------------------------------- + RefNull(RefType), + RefIsNull, + RefFunc(u32), + + // ---- Parametric instructions ------------------------------------------ + Drop, + Select, + SelectTyped(Vec), + + // ---- Variable instructions -------------------------------------------- + LocalGet(u32), + LocalSet(u32), + LocalTee(u32), + GlobalGet(u32), + GlobalSet(u32), + + // ---- Table instructions ----------------------------------------------- + TableGet(u32), + TableSet(u32), + TableInit { elem_index: u32, table_index: u32 }, + ElemDrop(u32), + TableCopy { dst_table: u32, src_table: u32 }, + TableGrow(u32), + TableSize(u32), + TableFill(u32), + + // ---- Memory instructions ---------------------------------------------- + I32Load(MemArg), + I64Load(MemArg), + F32Load(MemArg), + F64Load(MemArg), + I32Load8S(MemArg), + I32Load8U(MemArg), + I32Load16S(MemArg), + I32Load16U(MemArg), + I64Load8S(MemArg), + I64Load8U(MemArg), + I64Load16S(MemArg), + I64Load16U(MemArg), + I64Load32S(MemArg), + I64Load32U(MemArg), + I32Store(MemArg), + I64Store(MemArg), + F32Store(MemArg), + F64Store(MemArg), + I32Store8(MemArg), + I32Store16(MemArg), + I64Store8(MemArg), + I64Store16(MemArg), + I64Store32(MemArg), + MemorySize(u32), + MemoryGrow(u32), + MemoryInit { data_index: u32, memory: u32 }, + DataDrop(u32), + MemoryCopy { dst_memory: u32, src_memory: u32 }, + MemoryFill(u32), + + // ---- Numeric constants ------------------------------------------------ + I32Const(i32), + I64Const(i64), + F32Const(f32), + F64Const(f64), + + // ---- i32 comparisons -------------------------------------------------- + I32Eqz, + I32Eq, + I32Ne, + I32LtS, + I32LtU, + I32GtS, + I32GtU, + I32LeS, + I32LeU, + I32GeS, + I32GeU, + + // ---- i64 comparisons -------------------------------------------------- + I64Eqz, + I64Eq, + I64Ne, + I64LtS, + I64LtU, + I64GtS, + I64GtU, + I64LeS, + I64LeU, + I64GeS, + I64GeU, + + // ---- f32 comparisons -------------------------------------------------- + F32Eq, + F32Ne, + F32Lt, + F32Gt, + F32Le, + F32Ge, + + // ---- f64 comparisons -------------------------------------------------- + F64Eq, + F64Ne, + F64Lt, + F64Gt, + F64Le, + F64Ge, + + // ---- i32 numeric ------------------------------------------------------ + I32Clz, + I32Ctz, + I32Popcnt, + I32Add, + I32Sub, + I32Mul, + I32DivS, + I32DivU, + I32RemS, + I32RemU, + I32And, + I32Or, + I32Xor, + I32Shl, + I32ShrS, + I32ShrU, + I32Rotl, + I32Rotr, + + // ---- i64 numeric ------------------------------------------------------ + I64Clz, + I64Ctz, + I64Popcnt, + I64Add, + I64Sub, + I64Mul, + I64DivS, + I64DivU, + I64RemS, + I64RemU, + I64And, + I64Or, + I64Xor, + I64Shl, + I64ShrS, + I64ShrU, + I64Rotl, + I64Rotr, + + // ---- f32 numeric ------------------------------------------------------ + F32Abs, + F32Neg, + F32Ceil, + F32Floor, + F32Trunc, + F32Nearest, + F32Sqrt, + F32Add, + F32Sub, + F32Mul, + F32Div, + F32Min, + F32Max, + F32Copysign, + + // ---- f64 numeric ------------------------------------------------------ + F64Abs, + F64Neg, + F64Ceil, + F64Floor, + F64Trunc, + F64Nearest, + F64Sqrt, + F64Add, + F64Sub, + F64Mul, + F64Div, + F64Min, + F64Max, + F64Copysign, + + // ---- Conversions ------------------------------------------------------ + I32WrapI64, + I32TruncF32S, + I32TruncF32U, + I32TruncF64S, + I32TruncF64U, + I64ExtendI32S, + I64ExtendI32U, + I64TruncF32S, + I64TruncF32U, + I64TruncF64S, + I64TruncF64U, + F32ConvertI32S, + F32ConvertI32U, + F32ConvertI64S, + F32ConvertI64U, + F32DemoteF64, + F64ConvertI32S, + F64ConvertI32U, + F64ConvertI64S, + F64ConvertI64U, + F64PromoteF32, + I32ReinterpretF32, + I64ReinterpretF64, + F32ReinterpretI32, + F64ReinterpretI64, + + // ---- Sign-extension (post-MVP) ---------------------------------------- + I32Extend8S, + I32Extend16S, + I64Extend8S, + I64Extend16S, + I64Extend32S, + + // ---- Saturating float-to-int conversions (0xFC prefix, post-MVP) ------ + I32TruncSatF32S, + I32TruncSatF32U, + I32TruncSatF64S, + I32TruncSatF64U, + I64TruncSatF32S, + I64TruncSatF32U, + I64TruncSatF64S, + I64TruncSatF64U, +} diff --git a/crates/wasm/src/lib.rs b/crates/wasm/src/lib.rs new file mode 100644 index 0000000..34222f0 --- /dev/null +++ b/crates/wasm/src/lib.rs @@ -0,0 +1,55 @@ +//! WebAssembly binary-format parser. +//! +//! Decodes a `.wasm` byte stream into an in-memory [`Module`] following the +//! WebAssembly Core Specification 2.0, §5 (Binary Format). Validation and +//! execution live in sibling modules (added in follow-up issues); this crate +//! is intentionally limited to lossless decoding and structural rejection of +//! malformed input. +//! +//! Coverage: +//! - Module preamble (magic + version). +//! - All standard sections (0–12). Unknown custom sections are preserved +//! verbatim as `(name, payload)` byte slices. +//! - All numeric, parametric, variable, table, memory, control, and reference +//! opcodes from the MVP plus the post-MVP proposals listed in `PLAN.md` +//! (sign-extension, non-trapping float-to-int / saturating conversions, +//! reference types, bulk memory, multi-value blocks). +//! - SIMD (`0xFD`) and threads (`0xFE`) opcodes are intentionally rejected as +//! `UnknownOpcode`. +//! +//! ``` +//! use we_wasm::parse; +//! +//! let bytes = &[ +//! 0x00, 0x61, 0x73, 0x6d, // magic +//! 0x01, 0x00, 0x00, 0x00, // version 1 +//! ]; +//! let module = parse(bytes).unwrap(); +//! assert!(module.types.is_empty()); +//! ``` + +#![forbid(unsafe_code)] +#![deny(rust_2018_idioms)] + +mod decode; +mod error; +mod instruction; +mod module; +mod parser; +mod reader; +mod sections; +mod types; + +pub use decode::{decode_expr, decode_instruction}; +pub use error::{ParseError, ParseErrorKind}; +pub use instruction::Instruction; +pub use module::{ + ConstExpr, CustomSection, Data, DataMode, Element, ElementItems, ElementMode, Export, + ExportDesc, FunctionBody, Global, Import, ImportDesc, Module, +}; +pub use parser::{parse, MAGIC, VERSION}; +pub use reader::Reader; +pub use types::{ + BlockType, FuncType, GlobalType, Limits, MemArg, MemoryType, Mutability, NumType, RefType, + TableType, ValType, +}; diff --git a/crates/wasm/src/module.rs b/crates/wasm/src/module.rs new file mode 100644 index 0000000..fbe8bb2 --- /dev/null +++ b/crates/wasm/src/module.rs @@ -0,0 +1,112 @@ +//! In-memory representation of a parsed module. + +use crate::instruction::Instruction; +use crate::types::{FuncType, GlobalType, MemoryType, RefType, TableType, ValType}; + +/// A constant initializer expression: a sequence of instructions terminated by +/// `End`. The terminating `End` is included in the vector to keep instruction +/// indices stable. +pub type ConstExpr = Vec; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum ImportDesc { + Func(u32), + Table(TableType), + Memory(MemoryType), + Global(GlobalType), +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Import { + pub module: String, + pub name: String, + pub desc: ImportDesc, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum ExportDesc { + Func(u32), + Table(u32), + Memory(u32), + Global(u32), +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Export { + pub name: String, + pub desc: ExportDesc, +} + +#[derive(Debug, Clone, PartialEq)] +pub struct Global { + pub ty: GlobalType, + pub init: ConstExpr, +} + +/// Initializer values an element segment carries: either function indices +/// (`RefFunc`/`RefNull` shorthand) or full constant expressions. +#[derive(Debug, Clone, PartialEq)] +pub enum ElementItems { + FunctionIndices(Vec), + Expressions(Vec), +} + +#[derive(Debug, Clone, PartialEq)] +pub enum ElementMode { + Passive, + Active { table: u32, offset: ConstExpr }, + Declarative, +} + +#[derive(Debug, Clone, PartialEq)] +pub struct Element { + pub ref_type: RefType, + pub items: ElementItems, + pub mode: ElementMode, +} + +#[derive(Debug, Clone, PartialEq)] +pub enum DataMode { + Passive, + Active { memory: u32, offset: ConstExpr }, +} + +#[derive(Debug, Clone, PartialEq)] +pub struct Data<'a> { + pub init: &'a [u8], + pub mode: DataMode, +} + +/// One function body: locals declarations + the instruction stream. The +/// terminating `End` is included in `body` to keep instruction indices stable. +#[derive(Debug, Clone, PartialEq)] +pub struct FunctionBody { + pub locals: Vec<(u32, ValType)>, + pub body: Vec, +} + +/// Custom section preserved verbatim for downstream tools (debug info, names). +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct CustomSection<'a> { + pub name: String, + pub payload: &'a [u8], +} + +/// A parsed module. Borrowed slices reference back into the input bytes so the +/// parser is zero-copy for data segments and custom sections. +#[derive(Debug, Default, Clone, PartialEq)] +pub struct Module<'a> { + pub types: Vec, + pub imports: Vec, + pub functions: Vec, + pub tables: Vec, + pub memories: Vec, + pub globals: Vec, + pub exports: Vec, + pub start: Option, + pub elements: Vec, + pub code: Vec, + pub data: Vec>, + pub data_count: Option, + pub custom_sections: Vec>, +} diff --git a/crates/wasm/src/parser.rs b/crates/wasm/src/parser.rs new file mode 100644 index 0000000..5fca523 --- /dev/null +++ b/crates/wasm/src/parser.rs @@ -0,0 +1,82 @@ +//! Top-level binary-module driver: preamble, then each section in turn. + +use crate::error::{ParseError, ParseErrorKind}; +use crate::module::Module; +use crate::reader::Reader; +use crate::sections::{parse_section_payload, SectionSeenSet}; + +pub const MAGIC: [u8; 4] = [0x00, 0x61, 0x73, 0x6d]; +pub const VERSION: u32 = 1; + +/// Parse a complete binary WebAssembly module. +pub fn parse(bytes: &[u8]) -> Result, ParseError> { + let mut r = Reader::new(bytes); + + // Preamble. + if r.remaining() < 8 { + return Err(ParseError::new(ParseErrorKind::UnexpectedEof, 0)); + } + let magic_pos = r.position(); + let magic_bytes = r.read_bytes(4)?; + let mut magic = [0u8; 4]; + magic.copy_from_slice(magic_bytes); + if magic != MAGIC { + return Err(ParseError::new( + ParseErrorKind::InvalidMagic(magic), + magic_pos, + )); + } + let version_pos = r.position(); + let version = r.read_u32_le()?; + if version != VERSION { + return Err(ParseError::new( + ParseErrorKind::UnsupportedVersion(version), + version_pos, + )); + } + + let mut module = Module::default(); + let mut seen = SectionSeenSet::default(); + while !r.eof() { + let id = r.read_u8()?; + let size_pos = r.position(); + let size = r.read_u32()? as u64; + if size > r.remaining() as u64 { + return Err(ParseError::new( + ParseErrorKind::SectionTooShort { + id, + declared: size as u32, + remaining: r.remaining() as u64, + }, + size_pos, + )); + } + let payload_start = r.position(); + let section_bytes = r.read_bytes(size as usize)?; + let mut section_reader = Reader::new(section_bytes); + parse_section_payload(id, &mut module, &mut section_reader, &mut seen) + .map_err(|e| ParseError::new(e.kind, payload_start + e.offset))?; + if !section_reader.eof() { + return Err(ParseError::new( + ParseErrorKind::SectionTrailingBytes { + id, + leftover: section_reader.remaining() as u64, + }, + payload_start + section_reader.position(), + )); + } + } + + // Validate data count if both sections are present. + if let Some(declared) = module.data_count { + let actual = module.data.len() as u32; + if declared != actual { + return Err(ParseError::new( + ParseErrorKind::DataCountMismatch { declared, actual }, + bytes.len(), + )); + } + } + + Ok(module) +} diff --git a/crates/wasm/src/reader.rs b/crates/wasm/src/reader.rs new file mode 100644 index 0000000..bab2d35 --- /dev/null +++ b/crates/wasm/src/reader.rs @@ -0,0 +1,348 @@ +//! Low-level byte readers for the WebAssembly binary format. +//! +//! This module is the only place that touches raw bytes. Section and +//! instruction parsing builds on top of it. + +use crate::error::{ParseError, ParseErrorKind}; + +/// A bounds-checked cursor over a binary module slice. +pub struct Reader<'a> { + bytes: &'a [u8], + pos: usize, +} + +impl<'a> Reader<'a> { + pub fn new(bytes: &'a [u8]) -> Self { + Self { bytes, pos: 0 } + } + + pub fn position(&self) -> usize { + self.pos + } + + pub fn remaining(&self) -> usize { + self.bytes.len() - self.pos + } + + pub fn eof(&self) -> bool { + self.pos >= self.bytes.len() + } + + fn err(&self, kind: ParseErrorKind) -> ParseError { + ParseError::new(kind, self.pos) + } + + pub fn read_u8(&mut self) -> Result { + if self.pos >= self.bytes.len() { + return Err(self.err(ParseErrorKind::UnexpectedEof)); + } + let b = self.bytes[self.pos]; + self.pos += 1; + Ok(b) + } + + pub fn peek_u8(&self) -> Result { + if self.pos >= self.bytes.len() { + return Err(self.err(ParseErrorKind::UnexpectedEof)); + } + Ok(self.bytes[self.pos]) + } + + pub fn read_bytes(&mut self, n: usize) -> Result<&'a [u8], ParseError> { + if self.pos.saturating_add(n) > self.bytes.len() { + return Err(self.err(ParseErrorKind::UnexpectedEof)); + } + let slice = &self.bytes[self.pos..self.pos + n]; + self.pos += n; + Ok(slice) + } + + pub fn read_u32_le(&mut self) -> Result { + let b = self.read_bytes(4)?; + Ok(u32::from_le_bytes([b[0], b[1], b[2], b[3]])) + } + + pub fn read_f32(&mut self) -> Result { + let b = self.read_bytes(4)?; + Ok(f32::from_le_bytes([b[0], b[1], b[2], b[3]])) + } + + pub fn read_f64(&mut self) -> Result { + let b = self.read_bytes(8)?; + Ok(f64::from_le_bytes([ + b[0], b[1], b[2], b[3], b[4], b[5], b[6], b[7], + ])) + } + + /// Unsigned LEB128 with a width cap. `bits` selects 32 or 64. + fn read_u_leb128(&mut self, bits: u32) -> Result { + let max_bytes = bits.div_ceil(7) as usize; + let mut result: u64 = 0; + let mut shift: u32 = 0; + for byte_idx in 0..max_bytes { + let byte = self.read_u8()?; + let low7 = (byte & 0x7f) as u64; + let last_byte = byte_idx == max_bytes - 1; + if last_byte { + let used = bits - shift; // in 1..=7 + let mask: u64 = if used == 0 { 0 } else { (1u64 << used) - 1 }; + if low7 != (low7 & mask) { + return Err(self.err(ParseErrorKind::Leb128Overflow)); + } + if (byte & 0x80) != 0 { + return Err(self.err(ParseErrorKind::Leb128Overflow)); + } + } + result |= low7 << shift; + if (byte & 0x80) == 0 { + return Ok(result); + } + shift += 7; + } + Err(self.err(ParseErrorKind::Leb128Overflow)) + } + + /// Signed LEB128 with a width cap. `bits` selects 32 or 64. + fn read_s_leb128(&mut self, bits: u32) -> Result { + let max_bytes = bits.div_ceil(7) as usize; + let mut result: i64 = 0; + let mut shift: u32 = 0; + for byte_idx in 0..max_bytes { + let byte = self.read_u8()?; + let low7 = (byte & 0x7f) as i64; + let last_byte = byte_idx == max_bytes - 1; + if last_byte { + let value_bits_used = bits - shift; // in 1..=7 + let sign_bit = (low7 >> (value_bits_used - 1)) & 1; + let upper = low7 >> value_bits_used; + let upper_width = 7 - value_bits_used; + let expected_upper = if sign_bit == 1 { + if upper_width == 0 { + 0 + } else { + (1i64 << upper_width) - 1 + } + } else { + 0 + }; + if upper != expected_upper { + return Err(self.err(ParseErrorKind::Leb128Overflow)); + } + if (byte & 0x80) != 0 { + return Err(self.err(ParseErrorKind::Leb128Overflow)); + } + } + result |= low7 << shift; + if (byte & 0x80) == 0 { + shift += 7; + if shift < 64 && (byte & 0x40) != 0 { + result |= !0i64 << shift; + } + return Ok(result); + } + shift += 7; + } + Err(self.err(ParseErrorKind::Leb128Overflow)) + } + + pub fn read_u32(&mut self) -> Result { + Ok(self.read_u_leb128(32)? as u32) + } + + pub fn read_u64(&mut self) -> Result { + self.read_u_leb128(64) + } + + pub fn read_s32(&mut self) -> Result { + let v = self.read_s_leb128(32)?; + if !(i32::MIN as i64..=i32::MAX as i64).contains(&v) { + return Err(self.err(ParseErrorKind::Leb128Overflow)); + } + Ok(v as i32) + } + + pub fn read_s33(&mut self) -> Result { + self.read_s_leb128(33) + } + + pub fn read_s64(&mut self) -> Result { + self.read_s_leb128(64) + } + + /// Read a length-prefixed UTF-8 name. + pub fn read_name(&mut self) -> Result { + let len = self.read_u32()? as usize; + let raw = self.read_bytes(len)?; + match core::str::from_utf8(raw) { + Ok(s) => Ok(s.to_string()), + Err(_) => Err(ParseError::new(ParseErrorKind::InvalidUtf8, self.pos - len)), + } + } + + /// Read a length-prefixed byte vector (data, custom section payload). + pub fn read_byte_vec(&mut self) -> Result<&'a [u8], ParseError> { + let len = self.read_u32()? as usize; + self.read_bytes(len) + } + + /// Read a sub-reader covering exactly `len` bytes of input and advance past + /// them. Errors if the input is too short. + pub fn split_off(&mut self, len: usize) -> Result, ParseError> { + let bytes = self.read_bytes(len)?; + Ok(Reader::new(bytes)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn read_u32(bytes: &[u8]) -> Result { + Reader::new(bytes).read_u32() + } + + fn read_u64(bytes: &[u8]) -> Result { + Reader::new(bytes).read_u64() + } + + fn read_s32(bytes: &[u8]) -> Result { + Reader::new(bytes).read_s32() + } + + fn read_s64(bytes: &[u8]) -> Result { + Reader::new(bytes).read_s64() + } + + #[test] + fn leb128_unsigned_basic() { + assert_eq!(read_u32(&[0x00]).unwrap(), 0); + assert_eq!(read_u32(&[0x7f]).unwrap(), 127); + assert_eq!(read_u32(&[0x80, 0x01]).unwrap(), 128); + assert_eq!(read_u32(&[0xe5, 0x8e, 0x26]).unwrap(), 624485); + } + + #[test] + fn leb128_unsigned_max32() { + assert_eq!(read_u32(&[0xff, 0xff, 0xff, 0xff, 0x0f]).unwrap(), u32::MAX); + } + + #[test] + fn leb128_unsigned_overflow32() { + // 0x10 in the top byte sets bit 32, which is outside u32. + assert!(matches!( + read_u32(&[0xff, 0xff, 0xff, 0xff, 0x10]), + Err(ParseError { + kind: ParseErrorKind::Leb128Overflow, + .. + }) + )); + } + + #[test] + fn leb128_unsigned_overlong_overflow() { + // Five 0x80 bytes then 0x00 is overlong / overflow. + assert!(matches!( + read_u32(&[0x80, 0x80, 0x80, 0x80, 0x80, 0x00]), + Err(ParseError { + kind: ParseErrorKind::Leb128Overflow, + .. + }) + )); + } + + #[test] + fn leb128_unsigned_eof() { + assert!(matches!( + read_u32(&[0x80]), + Err(ParseError { + kind: ParseErrorKind::UnexpectedEof, + .. + }) + )); + } + + #[test] + fn leb128_signed_basic() { + assert_eq!(read_s32(&[0x00]).unwrap(), 0); + assert_eq!(read_s32(&[0x7f]).unwrap(), -1); + assert_eq!(read_s32(&[0x40]).unwrap(), -64); + assert_eq!(read_s32(&[0xc0, 0x00]).unwrap(), 64); + assert_eq!(read_s32(&[0xff, 0x7e]).unwrap(), -129); + } + + #[test] + fn leb128_signed_max32() { + assert_eq!(read_s32(&[0xff, 0xff, 0xff, 0xff, 0x07]).unwrap(), i32::MAX); + assert_eq!(read_s32(&[0x80, 0x80, 0x80, 0x80, 0x78]).unwrap(), i32::MIN); + } + + #[test] + fn leb128_signed_overflow32() { + // One past max. + assert!(matches!( + read_s32(&[0x80, 0x80, 0x80, 0x80, 0x08]), + Err(ParseError { + kind: ParseErrorKind::Leb128Overflow, + .. + }) + )); + // One below min. + assert!(matches!( + read_s32(&[0xff, 0xff, 0xff, 0xff, 0x77]), + Err(ParseError { + kind: ParseErrorKind::Leb128Overflow, + .. + }) + )); + } + + #[test] + fn leb128_signed_max64() { + assert_eq!( + read_s64(&[0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x00]).unwrap(), + i64::MAX + ); + assert_eq!( + read_s64(&[0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x7f]).unwrap(), + i64::MIN + ); + } + + #[test] + fn leb128_unsigned_max64() { + assert_eq!( + read_u64(&[0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x01]).unwrap(), + u64::MAX + ); + } + + #[test] + fn read_name_utf8() { + let mut r = Reader::new(b"\x05hello extra"); + assert_eq!(r.read_name().unwrap(), "hello"); + assert_eq!(r.position(), 6); + } + + #[test] + fn read_name_invalid_utf8() { + // Length 2, body is an incomplete UTF-8 sequence. + let mut r = Reader::new(&[0x02, 0xc3, 0x28]); + assert!(matches!( + r.read_name(), + Err(ParseError { + kind: ParseErrorKind::InvalidUtf8, + .. + }) + )); + } + + #[test] + fn floats_round_trip() { + let mut buf = Vec::new(); + buf.extend_from_slice(&1.5f32.to_le_bytes()); + buf.extend_from_slice(&(-2.5f64).to_le_bytes()); + let mut r = Reader::new(&buf); + assert_eq!(r.read_f32().unwrap(), 1.5); + assert_eq!(r.read_f64().unwrap(), -2.5); + } +} diff --git a/crates/wasm/src/sections.rs b/crates/wasm/src/sections.rs new file mode 100644 index 0000000..4f2b9ca --- /dev/null +++ b/crates/wasm/src/sections.rs @@ -0,0 +1,471 @@ +//! Section parsers. Each function reads exactly one section's payload and +//! populates the corresponding `Module` field. + +use crate::decode::decode_expr; +use crate::error::{ParseError, ParseErrorKind}; +use crate::module::{ + CustomSection, Data, DataMode, Element, ElementItems, ElementMode, Export, ExportDesc, + FunctionBody, Global, Import, ImportDesc, Module, +}; +use crate::reader::Reader; +use crate::types::{ + read_functype, read_globaltype, read_limits, read_reftype, read_tabletype, MemoryType, RefType, +}; + +const SECTION_CUSTOM: u8 = 0; +const SECTION_TYPE: u8 = 1; +const SECTION_IMPORT: u8 = 2; +const SECTION_FUNCTION: u8 = 3; +const SECTION_TABLE: u8 = 4; +const SECTION_MEMORY: u8 = 5; +const SECTION_GLOBAL: u8 = 6; +const SECTION_EXPORT: u8 = 7; +const SECTION_START: u8 = 8; +const SECTION_ELEMENT: u8 = 9; +const SECTION_CODE: u8 = 10; +const SECTION_DATA: u8 = 11; +const SECTION_DATA_COUNT: u8 = 12; + +pub(crate) fn parse_type_section<'a>( + module: &mut Module<'a>, + r: &mut Reader<'a>, +) -> Result<(), ParseError> { + let n = r.read_u32()? as usize; + module.types.reserve(n); + for _ in 0..n { + module.types.push(read_functype(r)?); + } + Ok(()) +} + +pub(crate) fn parse_import_section<'a>( + module: &mut Module<'a>, + r: &mut Reader<'a>, +) -> Result<(), ParseError> { + let n = r.read_u32()? as usize; + module.imports.reserve(n); + for _ in 0..n { + let mod_name = r.read_name()?; + let item_name = r.read_name()?; + let desc_pos = r.position(); + let tag = r.read_u8()?; + let desc = match tag { + 0x00 => ImportDesc::Func(r.read_u32()?), + 0x01 => ImportDesc::Table(read_tabletype(r)?), + 0x02 => ImportDesc::Memory(MemoryType(read_limits(r)?)), + 0x03 => ImportDesc::Global(read_globaltype(r)?), + other => { + return Err(ParseError::new( + ParseErrorKind::InvalidImportDescriptor(other), + desc_pos, + )) + } + }; + module.imports.push(Import { + module: mod_name, + name: item_name, + desc, + }); + } + Ok(()) +} + +pub(crate) fn parse_function_section<'a>( + module: &mut Module<'a>, + r: &mut Reader<'a>, +) -> Result<(), ParseError> { + let n = r.read_u32()? as usize; + module.functions.reserve(n); + for _ in 0..n { + module.functions.push(r.read_u32()?); + } + Ok(()) +} + +pub(crate) fn parse_table_section<'a>( + module: &mut Module<'a>, + r: &mut Reader<'a>, +) -> Result<(), ParseError> { + let n = r.read_u32()? as usize; + module.tables.reserve(n); + for _ in 0..n { + module.tables.push(read_tabletype(r)?); + } + Ok(()) +} + +pub(crate) fn parse_memory_section<'a>( + module: &mut Module<'a>, + r: &mut Reader<'a>, +) -> Result<(), ParseError> { + let n = r.read_u32()? as usize; + module.memories.reserve(n); + for _ in 0..n { + module.memories.push(MemoryType(read_limits(r)?)); + } + Ok(()) +} + +pub(crate) fn parse_global_section<'a>( + module: &mut Module<'a>, + r: &mut Reader<'a>, +) -> Result<(), ParseError> { + let n = r.read_u32()? as usize; + module.globals.reserve(n); + for _ in 0..n { + let ty = read_globaltype(r)?; + let init = decode_expr(r)?; + module.globals.push(Global { ty, init }); + } + Ok(()) +} + +pub(crate) fn parse_export_section<'a>( + module: &mut Module<'a>, + r: &mut Reader<'a>, +) -> Result<(), ParseError> { + let n = r.read_u32()? as usize; + module.exports.reserve(n); + for _ in 0..n { + let name = r.read_name()?; + let desc_pos = r.position(); + let tag = r.read_u8()?; + let idx = r.read_u32()?; + let desc = match tag { + 0x00 => ExportDesc::Func(idx), + 0x01 => ExportDesc::Table(idx), + 0x02 => ExportDesc::Memory(idx), + 0x03 => ExportDesc::Global(idx), + other => { + return Err(ParseError::new( + ParseErrorKind::InvalidExportDescriptor(other), + desc_pos, + )) + } + }; + module.exports.push(Export { name, desc }); + } + Ok(()) +} + +pub(crate) fn parse_start_section<'a>( + module: &mut Module<'a>, + r: &mut Reader<'a>, +) -> Result<(), ParseError> { + module.start = Some(r.read_u32()?); + Ok(()) +} + +pub(crate) fn parse_element_section<'a>( + module: &mut Module<'a>, + r: &mut Reader<'a>, +) -> Result<(), ParseError> { + let n = r.read_u32()? as usize; + module.elements.reserve(n); + for _ in 0..n { + let flags_pos = r.position(); + let flags = r.read_u32()?; + let elem = parse_one_element_segment(r, flags, flags_pos)?; + module.elements.push(elem); + } + Ok(()) +} + +fn parse_elem_kind(r: &mut Reader<'_>) -> Result { + // The "elemkind" byte is currently always 0x00 (funcref) in the spec. + let pos = r.position(); + let b = r.read_u8()?; + if b == 0x00 { + Ok(RefType::FuncRef) + } else { + Err(ParseError::new(ParseErrorKind::InvalidElementKind(b), pos)) + } +} + +fn read_func_index_vec(r: &mut Reader<'_>) -> Result, ParseError> { + let n = r.read_u32()? as usize; + let mut v = Vec::with_capacity(n); + for _ in 0..n { + v.push(r.read_u32()?); + } + Ok(v) +} + +fn read_expr_vec( + r: &mut Reader<'_>, +) -> Result>, ParseError> { + let n = r.read_u32()? as usize; + let mut v = Vec::with_capacity(n); + for _ in 0..n { + v.push(decode_expr(r)?); + } + Ok(v) +} + +fn parse_one_element_segment( + r: &mut Reader<'_>, + flags: u32, + flags_pos: usize, +) -> Result { + match flags { + 0 => { + let offset = decode_expr(r)?; + let funcs = read_func_index_vec(r)?; + Ok(Element { + ref_type: RefType::FuncRef, + items: ElementItems::FunctionIndices(funcs), + mode: ElementMode::Active { table: 0, offset }, + }) + } + 1 => { + let elem_kind = parse_elem_kind(r)?; + let funcs = read_func_index_vec(r)?; + Ok(Element { + ref_type: elem_kind, + items: ElementItems::FunctionIndices(funcs), + mode: ElementMode::Passive, + }) + } + 2 => { + let table = r.read_u32()?; + let offset = decode_expr(r)?; + let elem_kind = parse_elem_kind(r)?; + let funcs = read_func_index_vec(r)?; + Ok(Element { + ref_type: elem_kind, + items: ElementItems::FunctionIndices(funcs), + mode: ElementMode::Active { table, offset }, + }) + } + 3 => { + let elem_kind = parse_elem_kind(r)?; + let funcs = read_func_index_vec(r)?; + Ok(Element { + ref_type: elem_kind, + items: ElementItems::FunctionIndices(funcs), + mode: ElementMode::Declarative, + }) + } + 4 => { + let offset = decode_expr(r)?; + let exprs = read_expr_vec(r)?; + Ok(Element { + ref_type: RefType::FuncRef, + items: ElementItems::Expressions(exprs), + mode: ElementMode::Active { table: 0, offset }, + }) + } + 5 => { + let ref_type = read_reftype(r)?; + let exprs = read_expr_vec(r)?; + Ok(Element { + ref_type, + items: ElementItems::Expressions(exprs), + mode: ElementMode::Passive, + }) + } + 6 => { + let table = r.read_u32()?; + let offset = decode_expr(r)?; + let ref_type = read_reftype(r)?; + let exprs = read_expr_vec(r)?; + Ok(Element { + ref_type, + items: ElementItems::Expressions(exprs), + mode: ElementMode::Active { table, offset }, + }) + } + 7 => { + let ref_type = read_reftype(r)?; + let exprs = read_expr_vec(r)?; + Ok(Element { + ref_type, + items: ElementItems::Expressions(exprs), + mode: ElementMode::Declarative, + }) + } + other => Err(ParseError::new( + ParseErrorKind::InvalidElementSegmentFlags(other), + flags_pos, + )), + } +} + +pub(crate) fn parse_code_section<'a>( + module: &mut Module<'a>, + r: &mut Reader<'a>, +) -> Result<(), ParseError> { + let n = r.read_u32()? as usize; + module.code.reserve(n); + for _ in 0..n { + let size = r.read_u32()? as usize; + let body_start = r.position(); + let mut body_reader = r.split_off(size)?; + let locals_count = body_reader.read_u32()? as usize; + let mut locals = Vec::with_capacity(locals_count); + for _ in 0..locals_count { + let count = body_reader.read_u32()?; + let vt = crate::types::read_valtype(&mut body_reader)?; + locals.push((count, vt)); + } + let body = decode_expr(&mut body_reader)?; + if !body_reader.eof() { + return Err(ParseError::new( + ParseErrorKind::InvalidFunctionBody, + body_start + body_reader.position(), + )); + } + module.code.push(FunctionBody { locals, body }); + } + Ok(()) +} + +pub(crate) fn parse_data_section<'a>( + module: &mut Module<'a>, + r: &mut Reader<'a>, +) -> Result<(), ParseError> { + let n = r.read_u32()? as usize; + module.data.reserve(n); + for _ in 0..n { + let flags_pos = r.position(); + let flags = r.read_u32()?; + let data = match flags { + 0 => { + let offset = decode_expr(r)?; + let init = r.read_byte_vec()?; + Data { + init, + mode: DataMode::Active { memory: 0, offset }, + } + } + 1 => { + let init = r.read_byte_vec()?; + Data { + init, + mode: DataMode::Passive, + } + } + 2 => { + let memory = r.read_u32()?; + let offset = decode_expr(r)?; + let init = r.read_byte_vec()?; + Data { + init, + mode: DataMode::Active { memory, offset }, + } + } + other => { + return Err(ParseError::new( + ParseErrorKind::InvalidDataSegmentFlags(other), + flags_pos, + )) + } + }; + module.data.push(data); + } + Ok(()) +} + +pub(crate) fn parse_data_count_section<'a>( + module: &mut Module<'a>, + r: &mut Reader<'a>, +) -> Result<(), ParseError> { + module.data_count = Some(r.read_u32()?); + Ok(()) +} + +pub(crate) fn parse_custom_section<'a>( + module: &mut Module<'a>, + r: &mut Reader<'a>, +) -> Result<(), ParseError> { + let name = r.read_name()?; + let payload = r.read_bytes(r.remaining())?; + module.custom_sections.push(CustomSection { name, payload }); + Ok(()) +} + +/// Drive the section dispatch. `r` covers the section payload exactly. +pub(crate) fn parse_section_payload<'a>( + id: u8, + module: &mut Module<'a>, + r: &mut Reader<'a>, + seen: &mut SectionSeenSet, +) -> Result<(), ParseError> { + match id { + SECTION_CUSTOM => { + // Custom sections may appear any number of times. + parse_custom_section(module, r) + } + SECTION_TYPE => { + seen.mark_once(SECTION_TYPE, r.position())?; + parse_type_section(module, r) + } + SECTION_IMPORT => { + seen.mark_once(SECTION_IMPORT, r.position())?; + parse_import_section(module, r) + } + SECTION_FUNCTION => { + seen.mark_once(SECTION_FUNCTION, r.position())?; + parse_function_section(module, r) + } + SECTION_TABLE => { + seen.mark_once(SECTION_TABLE, r.position())?; + parse_table_section(module, r) + } + SECTION_MEMORY => { + seen.mark_once(SECTION_MEMORY, r.position())?; + parse_memory_section(module, r) + } + SECTION_GLOBAL => { + seen.mark_once(SECTION_GLOBAL, r.position())?; + parse_global_section(module, r) + } + SECTION_EXPORT => { + seen.mark_once(SECTION_EXPORT, r.position())?; + parse_export_section(module, r) + } + SECTION_START => { + seen.mark_once(SECTION_START, r.position())?; + parse_start_section(module, r) + } + SECTION_ELEMENT => { + seen.mark_once(SECTION_ELEMENT, r.position())?; + parse_element_section(module, r) + } + SECTION_CODE => { + seen.mark_once(SECTION_CODE, r.position())?; + parse_code_section(module, r) + } + SECTION_DATA => { + seen.mark_once(SECTION_DATA, r.position())?; + parse_data_section(module, r) + } + SECTION_DATA_COUNT => { + seen.mark_once(SECTION_DATA_COUNT, r.position())?; + parse_data_count_section(module, r) + } + other => Err(ParseError::new( + ParseErrorKind::UnknownSectionId(other), + r.position(), + )), + } +} + +/// Tracks which non-custom sections have been seen so duplicates are rejected. +/// IDs 0..=12 fit comfortably in a small bitset. +#[derive(Default)] +pub(crate) struct SectionSeenSet(u32); + +impl SectionSeenSet { + fn mark_once(&mut self, id: u8, offset: usize) -> Result<(), ParseError> { + let mask = 1u32 << id; + if (self.0 & mask) != 0 { + return Err(ParseError::new( + ParseErrorKind::DuplicateSection(id), + offset, + )); + } + self.0 |= mask; + Ok(()) + } +} diff --git a/crates/wasm/src/types.rs b/crates/wasm/src/types.rs new file mode 100644 index 0000000..b7c66bf --- /dev/null +++ b/crates/wasm/src/types.rs @@ -0,0 +1,223 @@ +//! Value, function, and entity types from the WebAssembly type system. + +use crate::error::{ParseError, ParseErrorKind}; +use crate::reader::Reader; + +/// Reference type (subset of valtype). +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum RefType { + FuncRef, + ExternRef, +} + +/// Number type. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum NumType { + I32, + I64, + F32, + F64, +} + +/// Value type carried on the operand stack. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum ValType { + Num(NumType), + Ref(RefType), +} + +impl ValType { + pub const I32: Self = ValType::Num(NumType::I32); + pub const I64: Self = ValType::Num(NumType::I64); + pub const F32: Self = ValType::Num(NumType::F32); + pub const F64: Self = ValType::Num(NumType::F64); + pub const FUNCREF: Self = ValType::Ref(RefType::FuncRef); + pub const EXTERNREF: Self = ValType::Ref(RefType::ExternRef); +} + +pub(crate) fn read_valtype(r: &mut Reader<'_>) -> Result { + let start = r.position(); + let b = r.read_u8()?; + match b { + 0x7F => Ok(ValType::I32), + 0x7E => Ok(ValType::I64), + 0x7D => Ok(ValType::F32), + 0x7C => Ok(ValType::F64), + 0x70 => Ok(ValType::FUNCREF), + 0x6F => Ok(ValType::EXTERNREF), + other => Err(ParseError::new( + ParseErrorKind::InvalidValType(other), + start, + )), + } +} + +pub(crate) fn read_reftype(r: &mut Reader<'_>) -> Result { + let start = r.position(); + let b = r.read_u8()?; + match b { + 0x70 => Ok(RefType::FuncRef), + 0x6F => Ok(RefType::ExternRef), + other => Err(ParseError::new( + ParseErrorKind::InvalidRefType(other), + start, + )), + } +} + +/// Block type encoded inline on `block`/`loop`/`if`. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BlockType { + Empty, + Value(ValType), + /// Index into the type section, naming a function signature. + TypeIndex(u32), +} + +pub(crate) fn read_blocktype(r: &mut Reader<'_>) -> Result { + let start = r.position(); + let b = r.peek_u8()?; + if b == 0x40 { + r.read_u8()?; + return Ok(BlockType::Empty); + } + if matches!(b, 0x7F | 0x7E | 0x7D | 0x7C | 0x70 | 0x6F) { + let vt = read_valtype(r)?; + return Ok(BlockType::Value(vt)); + } + // Otherwise, signed LEB128 in 33 bits. + let idx = r.read_s33()?; + if idx < 0 { + return Err(ParseError::new( + ParseErrorKind::InvalidBlockType(idx), + start, + )); + } + if idx > u32::MAX as i64 { + return Err(ParseError::new(ParseErrorKind::IntegerOverflow, start)); + } + Ok(BlockType::TypeIndex(idx as u32)) +} + +/// Function type (parameters + results). +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct FuncType { + pub params: Vec, + pub results: Vec, +} + +pub(crate) fn read_functype(r: &mut Reader<'_>) -> Result { + let tag_pos = r.position(); + let tag = r.read_u8()?; + if tag != 0x60 { + return Err(ParseError::new( + ParseErrorKind::InvalidFuncTypeTag(tag), + tag_pos, + )); + } + let params = read_valtype_vec(r)?; + let results = read_valtype_vec(r)?; + Ok(FuncType { params, results }) +} + +pub(crate) fn read_valtype_vec(r: &mut Reader<'_>) -> Result, ParseError> { + let n = r.read_u32()? as usize; + let mut v = Vec::with_capacity(n); + for _ in 0..n { + v.push(read_valtype(r)?); + } + Ok(v) +} + +/// `{ min, max? }` resource limits used by memories and tables. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Limits { + pub min: u32, + pub max: Option, +} + +pub(crate) fn read_limits(r: &mut Reader<'_>) -> Result { + let tag_pos = r.position(); + let tag = r.read_u8()?; + match tag { + 0x00 => { + let min = r.read_u32()?; + Ok(Limits { min, max: None }) + } + 0x01 => { + let min = r.read_u32()?; + let max = r.read_u32()?; + Ok(Limits { + min, + max: Some(max), + }) + } + other => Err(ParseError::new( + ParseErrorKind::InvalidLimitsTag(other), + tag_pos, + )), + } +} + +/// Memory type: linear-memory limits. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct MemoryType(pub Limits); + +/// Table type: element reftype + limits on element count. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct TableType { + pub elem: RefType, + pub limits: Limits, +} + +pub(crate) fn read_tabletype(r: &mut Reader<'_>) -> Result { + let elem = read_reftype(r)?; + let limits = read_limits(r)?; + Ok(TableType { elem, limits }) +} + +/// Mutability of a global. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Mutability { + Const, + Var, +} + +pub(crate) fn read_mutability(r: &mut Reader<'_>) -> Result { + let pos = r.position(); + let b = r.read_u8()?; + match b { + 0x00 => Ok(Mutability::Const), + 0x01 => Ok(Mutability::Var), + other => Err(ParseError::new( + ParseErrorKind::InvalidMutabilityFlag(other), + pos, + )), + } +} + +/// Global type: value type + mutability. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct GlobalType { + pub valtype: ValType, + pub mutability: Mutability, +} + +pub(crate) fn read_globaltype(r: &mut Reader<'_>) -> Result { + let valtype = read_valtype(r)?; + let mutability = read_mutability(r)?; + Ok(GlobalType { + valtype, + mutability, + }) +} + +/// Memory access immediate for load/store instructions: alignment exponent and +/// a static byte offset added to the dynamic address. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct MemArg { + pub align: u32, + pub offset: u32, + /// Index of the linear memory the access targets (always 0 in MVP). + pub memory: u32, +} diff --git a/crates/wasm/tests/fixtures/wasm_fixture.wasm b/crates/wasm/tests/fixtures/wasm_fixture.wasm new file mode 100755 index 0000000000000000000000000000000000000000..07963b59e714bc3e11b77c0eecf04f914fe2d35a GIT binary patch literal 525 zcmZQbEY4+QU|?VrW=>$LuV<`JV611XOJJ@Cv6)$z85o&ZSQ!f#85vob85soFB$?|O z9UB@BFmTs1pwjI2Y;3uyx%owvObpD4DJcvL%xRfP42;~xsX3|1CGjPx#U%_(T=DTK zi6x2gsd*{PjNI|@8L5c{@kxorsmx4Vfz0d-3XBR2S=_daEOkr{Ob!YRybMkZii|vr z+)j*&Oim1n%*+l7%nD33iYy9Dii`?Oip&ZuicAU&j`dke47?27+zO1`oFG$}6d0tr z8QfX&5_3}-gapeIi*w`CGAl|-i&FJK97YLdurnD!u4F=XB)bqJ1AlycaYp8 zW?o5Z5rZ#hK~a86X>w{&F%w5lVqSV_VtOhgOHgTX2?I|-QGRl2adB#jZc-&9Yf))& zNwTb=o~5atfr5sqrG>+C;@p!o4+VEEwv~$FF94Wq_QBj PSev6bGd)i?wW0(7aY>EB literal 0 HcmV?d00001 diff --git a/crates/wasm/tests/parser_tests.rs b/crates/wasm/tests/parser_tests.rs new file mode 100644 index 0000000..ea16056 --- /dev/null +++ b/crates/wasm/tests/parser_tests.rs @@ -0,0 +1,469 @@ +//! Integration tests for the binary-format parser. +//! +//! Exercises preamble handling, each section type built up from hand-authored +//! fixtures, instruction decoding, and a real Rust-compiled module from +//! `target wasm32-unknown-unknown`. + +use we_wasm::{ + parse, BlockType, DataMode, ElementItems, ElementMode, ExportDesc, ImportDesc, Instruction, + Mutability, ParseErrorKind, RefType, ValType, +}; + +const PREAMBLE: [u8; 8] = [0x00, 0x61, 0x73, 0x6d, 0x01, 0x00, 0x00, 0x00]; + +// --------------------------------------------------------------------------- +// Preamble. +// --------------------------------------------------------------------------- + +#[test] +fn parses_empty_module() { + let m = parse(&PREAMBLE).unwrap(); + assert!(m.types.is_empty()); + assert!(m.functions.is_empty()); + assert!(m.code.is_empty()); +} + +#[test] +fn rejects_wrong_magic() { + let mut bytes = PREAMBLE; + bytes[0] = 0xff; + let err = parse(&bytes).unwrap_err(); + assert!(matches!(err.kind, ParseErrorKind::InvalidMagic(_))); + assert_eq!(err.offset, 0); +} + +#[test] +fn rejects_unsupported_version() { + let mut bytes = PREAMBLE; + bytes[4] = 0x02; + let err = parse(&bytes).unwrap_err(); + assert!(matches!(err.kind, ParseErrorKind::UnsupportedVersion(2))); + assert_eq!(err.offset, 4); +} + +#[test] +fn rejects_truncated_preamble() { + let err = parse(&[0x00, 0x61, 0x73]).unwrap_err(); + assert_eq!(err.kind, ParseErrorKind::UnexpectedEof); +} + +// --------------------------------------------------------------------------- +// Hand-authored section coverage. +// --------------------------------------------------------------------------- + +/// Build a section: id, LEB128 size of payload, payload bytes. +fn section(id: u8, payload: &[u8]) -> Vec { + let mut out = vec![id]; + push_leb_u32(&mut out, payload.len() as u32); + out.extend_from_slice(payload); + out +} + +fn push_leb_u32(out: &mut Vec, mut value: u32) { + loop { + let byte = (value & 0x7f) as u8; + value >>= 7; + if value == 0 { + out.push(byte); + return; + } + out.push(byte | 0x80); + } +} + +fn module_with(sections: &[Vec]) -> Vec { + let mut bytes = PREAMBLE.to_vec(); + for s in sections { + bytes.extend_from_slice(s); + } + bytes +} + +#[test] +fn type_section_round_trip() { + // One functype: (i32, i32) -> i32, one: () -> (i64, f64) + let payload = vec![ + 0x02, // count + 0x60, 0x02, 0x7f, 0x7f, 0x01, 0x7f, // (i32 i32) -> i32 + 0x60, 0x00, 0x02, 0x7e, 0x7c, // () -> (i64 f64) (multi-value) + ]; + let bytes = module_with(&[section(1, &payload)]); + let m = parse(&bytes).unwrap(); + assert_eq!(m.types.len(), 2); + assert_eq!(m.types[0].params, vec![ValType::I32, ValType::I32]); + assert_eq!(m.types[0].results, vec![ValType::I32]); + assert_eq!(m.types[1].results, vec![ValType::I64, ValType::F64]); +} + +#[test] +fn import_section_round_trip() { + // type: () -> () + let type_payload = vec![0x01, 0x60, 0x00, 0x00]; + let import_payload = vec![ + 0x04, // 4 imports + // env.host_func (type 0) + 0x03, b'e', b'n', b'v', 0x09, b'h', b'o', b's', b't', b'_', b'f', b'u', b'n', b'c', 0x00, + 0x00, // + // env.mem (memory limits {min:1}) + 0x03, b'e', b'n', b'v', 0x03, b'm', b'e', b'm', 0x02, 0x00, 0x01, // + // env.tbl (table funcref, {min:1, max:2}) + 0x03, b'e', b'n', b'v', 0x03, b't', b'b', b'l', 0x01, 0x70, 0x01, 0x01, 0x02, // + // env.g (mutable i32) + 0x03, b'e', b'n', b'v', 0x01, b'g', 0x03, 0x7f, 0x01, // + ]; + let bytes = module_with(&[section(1, &type_payload), section(2, &import_payload)]); + let m = parse(&bytes).unwrap(); + assert_eq!(m.imports.len(), 4); + assert_eq!(m.imports[0].module, "env"); + assert_eq!(m.imports[0].name, "host_func"); + assert!(matches!(m.imports[0].desc, ImportDesc::Func(0))); + assert!(matches!(m.imports[1].desc, ImportDesc::Memory(_))); + if let ImportDesc::Table(t) = &m.imports[2].desc { + assert_eq!(t.elem, RefType::FuncRef); + assert_eq!(t.limits.min, 1); + assert_eq!(t.limits.max, Some(2)); + } else { + panic!("expected Table import"); + } + if let ImportDesc::Global(g) = &m.imports[3].desc { + assert_eq!(g.valtype, ValType::I32); + assert_eq!(g.mutability, Mutability::Var); + } else { + panic!("expected Global import"); + } +} + +#[test] +fn function_table_memory_global_export_start_round_trip() { + // type: () -> i32 + let type_payload = vec![0x01, 0x60, 0x00, 0x01, 0x7f]; + let func_payload = vec![0x01, 0x00]; + let table_payload = vec![0x01, 0x70, 0x00, 0x02]; + let memory_payload = vec![0x01, 0x01, 0x02, 0x05]; + // global i32 const, init = i32.const 42 end + let global_payload = vec![0x01, 0x7f, 0x00, 0x41, 0x2a, 0x0b]; + // export "f" func 0 + let export_payload = vec![0x01, 0x01, b'f', 0x00, 0x00]; + // start: index 0 + let start_payload = vec![0x00]; + // code: empty locals, body = i32.const 7 end + let code_payload = vec![0x01, 0x04, 0x00, 0x41, 0x07, 0x0b]; + + let bytes = module_with(&[ + section(1, &type_payload), + section(3, &func_payload), + section(4, &table_payload), + section(5, &memory_payload), + section(6, &global_payload), + section(7, &export_payload), + section(8, &start_payload), + section(10, &code_payload), + ]); + let m = parse(&bytes).unwrap(); + assert_eq!(m.functions, vec![0]); + assert_eq!(m.tables.len(), 1); + assert_eq!(m.memories.len(), 1); + assert_eq!(m.memories[0].0.min, 2); + assert_eq!(m.memories[0].0.max, Some(5)); + assert_eq!(m.globals.len(), 1); + assert_eq!(m.globals[0].ty.valtype, ValType::I32); + assert_eq!(m.globals[0].ty.mutability, Mutability::Const); + assert_eq!(m.globals[0].init.len(), 2); + assert_eq!(m.globals[0].init[0], Instruction::I32Const(42)); + assert_eq!(m.globals[0].init[1], Instruction::End); + assert_eq!(m.exports.len(), 1); + assert_eq!(m.exports[0].name, "f"); + assert!(matches!(m.exports[0].desc, ExportDesc::Func(0))); + assert_eq!(m.start, Some(0)); + assert_eq!(m.code.len(), 1); + assert_eq!(m.code[0].locals.len(), 0); + assert_eq!(m.code[0].body[0], Instruction::I32Const(7)); + assert_eq!(m.code[0].body[1], Instruction::End); +} + +#[test] +fn element_section_all_modes() { + // type: () -> () + let type_payload = vec![0x01, 0x60, 0x00, 0x00]; + // function: one local function of type 0 + let func_payload = vec![0x01, 0x00]; + // table: 1 funcref table, min 4 + let table_payload = vec![0x01, 0x70, 0x00, 0x04]; + // memory needed because some tests use it; skip. + // 8 element segments — one for each flag value 0..=7 + let mut elem_payload = vec![0x08]; // 8 segments + // flags=0: active, offset = i32.const 0, vec + elem_payload.extend_from_slice(&[ + 0x00, // flags + 0x41, 0x00, 0x0b, // offset expr: i32.const 0 end + 0x01, 0x00, // 1 funcidx + ]); + // flags=1: passive, elem kind 0, vec + elem_payload.extend_from_slice(&[0x01, 0x00, 0x01, 0x00]); + // flags=2: active, tableidx 0, offset expr, elem kind 0, vec + elem_payload.extend_from_slice(&[0x02, 0x00, 0x41, 0x00, 0x0b, 0x00, 0x01, 0x00]); + // flags=3: declarative, elem kind 0, vec + elem_payload.extend_from_slice(&[0x03, 0x00, 0x01, 0x00]); + // flags=4: active, offset expr, vec + elem_payload.extend_from_slice(&[ + 0x04, 0x41, 0x00, 0x0b, 0x01, // 1 expr + 0xd2, 0x00, 0x0b, // ref.func 0 end + ]); + // flags=5: passive, reftype, vec + elem_payload.extend_from_slice(&[0x05, 0x70, 0x01, 0xd2, 0x00, 0x0b]); + // flags=6: active, tableidx, offset expr, reftype, vec + elem_payload.extend_from_slice(&[0x06, 0x00, 0x41, 0x00, 0x0b, 0x70, 0x01, 0xd2, 0x00, 0x0b]); + // flags=7: declarative, reftype, vec + elem_payload.extend_from_slice(&[0x07, 0x6f, 0x01, 0xd0, 0x6f, 0x0b]); + // code: one body, empty + let code_payload = vec![0x01, 0x02, 0x00, 0x0b]; + let bytes = module_with(&[ + section(1, &type_payload), + section(3, &func_payload), + section(4, &table_payload), + section(9, &elem_payload), + section(10, &code_payload), + ]); + let m = parse(&bytes).unwrap(); + assert_eq!(m.elements.len(), 8); + assert!(matches!( + m.elements[0].mode, + ElementMode::Active { table: 0, .. } + )); + assert!(matches!( + m.elements[0].items, + ElementItems::FunctionIndices(_) + )); + assert!(matches!(m.elements[1].mode, ElementMode::Passive)); + assert!(matches!( + m.elements[2].mode, + ElementMode::Active { table: 0, .. } + )); + assert!(matches!(m.elements[3].mode, ElementMode::Declarative)); + assert!(matches!(m.elements[4].items, ElementItems::Expressions(_))); + assert!(matches!(m.elements[5].mode, ElementMode::Passive)); + assert!(matches!( + m.elements[6].mode, + ElementMode::Active { table: 0, .. } + )); + assert_eq!(m.elements[7].ref_type, RefType::ExternRef); +} + +#[test] +fn data_section_round_trip() { + let memory_payload = vec![0x01, 0x00, 0x01]; // 1 memory, min=1 + let mut data_payload = vec![0x02]; // 2 segments + // flags=0: active mem 0, offset expr, bytes "hi" + data_payload.extend_from_slice(&[0x00, 0x41, 0x00, 0x0b, 0x02, b'h', b'i']); + // flags=1: passive, bytes "ok" + data_payload.extend_from_slice(&[0x01, 0x02, b'o', b'k']); + // datacount section comes first by spec order, but for parsing order doesn't matter + let data_count_payload = vec![0x02]; + let bytes = module_with(&[ + section(5, &memory_payload), + section(12, &data_count_payload), + section(11, &data_payload), + ]); + let m = parse(&bytes).unwrap(); + assert_eq!(m.data.len(), 2); + assert_eq!(m.data[0].init, b"hi"); + assert!(matches!(m.data[0].mode, DataMode::Active { memory: 0, .. })); + assert_eq!(m.data[1].init, b"ok"); + assert!(matches!(m.data[1].mode, DataMode::Passive)); + assert_eq!(m.data_count, Some(2)); +} + +#[test] +fn custom_section_preserved() { + // Custom section with name "we" and payload [1, 2, 3] + let mut payload = vec![0x02, b'w', b'e']; + payload.extend_from_slice(&[1, 2, 3]); + let bytes = module_with(&[section(0, &payload)]); + let m = parse(&bytes).unwrap(); + assert_eq!(m.custom_sections.len(), 1); + assert_eq!(m.custom_sections[0].name, "we"); + assert_eq!(m.custom_sections[0].payload, &[1, 2, 3]); +} + +#[test] +fn data_count_mismatch_rejected() { + let memory_payload = vec![0x01, 0x00, 0x01]; + let data_count_payload = vec![0x05]; // claim 5 segments + let data_payload = vec![0x01, 0x01, 0x00]; // only 1 + let bytes = module_with(&[ + section(5, &memory_payload), + section(12, &data_count_payload), + section(11, &data_payload), + ]); + let err = parse(&bytes).unwrap_err(); + assert!(matches!( + err.kind, + ParseErrorKind::DataCountMismatch { + declared: 5, + actual: 1 + } + )); +} + +#[test] +fn duplicate_type_section_rejected() { + let payload = vec![0x00]; + let bytes = module_with(&[section(1, &payload), section(1, &payload)]); + let err = parse(&bytes).unwrap_err(); + assert!(matches!(err.kind, ParseErrorKind::DuplicateSection(1))); +} + +#[test] +fn unknown_section_rejected() { + // Section ids 0..=12 are known. + let bytes = module_with(&[section(99, &[0x00])]); + let err = parse(&bytes).unwrap_err(); + assert!(matches!(err.kind, ParseErrorKind::UnknownSectionId(99))); +} + +#[test] +fn unknown_opcode_rejected() { + // Custom-style section with one type ()->() and one body containing 0xFD (SIMD). + let type_payload = vec![0x01, 0x60, 0x00, 0x00]; + let func_payload = vec![0x01, 0x00]; + let code_payload = vec![0x01, 0x03, 0x00, 0xFD, 0x0b]; + let bytes = module_with(&[ + section(1, &type_payload), + section(3, &func_payload), + section(10, &code_payload), + ]); + let err = parse(&bytes).unwrap_err(); + assert!(matches!( + err.kind, + ParseErrorKind::UnknownOpcode { opcode: 0xfd } + )); +} + +// --------------------------------------------------------------------------- +// Instruction decoding coverage (control / structured blocks / 0xFC prefix). +// --------------------------------------------------------------------------- + +#[test] +fn decodes_structured_control_flow() { + let type_payload = vec![0x01, 0x60, 0x00, 0x00]; + let func_payload = vec![0x01, 0x00]; + // body: block (empty) loop (empty) br 1 end end end + let code_payload = vec![ + 0x01, 0x0a, // count + size + 0x00, // 0 local groups + 0x02, 0x40, // block empty + 0x03, 0x40, // loop empty + 0x0c, 0x01, // br 1 + 0x0b, 0x0b, 0x0b, // end end end + ]; + let bytes = module_with(&[ + section(1, &type_payload), + section(3, &func_payload), + section(10, &code_payload), + ]); + let m = parse(&bytes).unwrap(); + let body = &m.code[0].body; + assert_eq!(body.len(), 6); + assert_eq!(body[0], Instruction::Block(BlockType::Empty)); + assert_eq!(body[1], Instruction::Loop(BlockType::Empty)); + assert_eq!(body[2], Instruction::Br(1)); + assert_eq!(body[3], Instruction::End); + assert_eq!(body[4], Instruction::End); + assert_eq!(body[5], Instruction::End); +} + +#[test] +fn decodes_bulk_memory_table_extensions() { + let type_payload = vec![0x01, 0x60, 0x00, 0x00]; + let func_payload = vec![0x01, 0x00]; + let mem_payload = vec![0x01, 0x00, 0x01]; + // We need data segments to satisfy data.drop usage; declare 1 passive. + let data_count_payload = vec![0x01]; + let data_payload = vec![0x01, 0x01, 0x01, 0xff]; + // body uses: memory.fill 0; memory.copy 0 0; data.drop 0; i32.trunc_sat_f32_s + let code_payload = vec![ + 0x01, // count + 0x10, // size + 0x00, // locals + // memory.fill: 0xFC 11 0 + 0x41, 0x00, 0x41, 0x00, 0x41, 0x00, // push three i32s + 0xFC, 0x0B, 0x00, // memory.fill 0 + 0xFC, 0x09, 0x00, // data.drop 0 + 0xFC, 0x00, // i32.trunc_sat_f32_s + 0x0b, // end + ]; + let bytes = module_with(&[ + section(1, &type_payload), + section(3, &func_payload), + section(5, &mem_payload), + section(12, &data_count_payload), + section(10, &code_payload), + section(11, &data_payload), + ]); + let m = parse(&bytes).unwrap(); + let body = &m.code[0].body; + assert!(body.contains(&Instruction::MemoryFill(0))); + assert!(body.contains(&Instruction::DataDrop(0))); + assert!(body.contains(&Instruction::I32TruncSatF32S)); +} + +#[test] +fn decodes_call_indirect_and_typed_select() { + let type_payload = vec![0x01, 0x60, 0x00, 0x00]; + let func_payload = vec![0x01, 0x00]; + let table_payload = vec![0x01, 0x70, 0x00, 0x01]; + let code_payload = vec![ + 0x01, // count + 0x0c, // size + 0x00, // locals + // i32.const 1; i32.const 2; i32.const 0; select t [i32]; drop + 0x41, 0x01, 0x41, 0x02, 0x41, 0x00, 0x1c, 0x01, 0x7f, 0x1a, 0x0b, + ]; + let bytes = module_with(&[ + section(1, &type_payload), + section(3, &func_payload), + section(4, &table_payload), + section(10, &code_payload), + ]); + let m = parse(&bytes).unwrap(); + let body = &m.code[0].body; + let has_select_typed = body + .iter() + .any(|i| matches!(i, Instruction::SelectTyped(v) if v.as_slice() == [ValType::I32])); + assert!(has_select_typed); +} + +// --------------------------------------------------------------------------- +// Real-world: Rust-compiled module. +// --------------------------------------------------------------------------- + +const FIXTURE: &[u8] = include_bytes!("fixtures/wasm_fixture.wasm"); + +#[test] +fn parses_rust_compiled_module() { + let m = parse(FIXTURE).unwrap(); + // Cargo + rustc compile our cdylib with: + // exports: "memory" + the three #[no_mangle] functions + some __helpers + let names: Vec<&str> = m.exports.iter().map(|e| e.name.as_str()).collect(); + assert!(names.contains(&"add")); + assert!(names.contains(&"fib")); + assert!(names.contains(&"select_test")); + assert!(names.contains(&"memory")); + // Memory and at least one custom section ("name" or "producers") present. + assert!(!m.memories.is_empty()); + assert!(!m.custom_sections.is_empty()); + // Function bodies decoded without error. + assert!(!m.code.is_empty()); + for body in &m.code { + assert!(body.body.iter().any(|i| matches!(i, Instruction::End))); + } +} + +#[test] +fn rust_compiled_module_has_name_custom_section() { + let m = parse(FIXTURE).unwrap(); + assert!(m + .custom_sections + .iter() + .any(|c| c.name == "name" && !c.payload.is_empty())); +}