diff --git a/.gitmodules b/.gitmodules new file mode 100644 index 0000000..e316ccb --- /dev/null +++ b/.gitmodules @@ -0,0 +1,6 @@ +[submodule "tests/html5lib-tests"] + path = tests/html5lib-tests + url = https://github.com/html5lib/html5lib-tests.git +[submodule "tests/test262"] + path = tests/test262 + url = https://github.com/nicolo-ribaudo/test262 diff --git a/crates/html/src/lib.rs b/crates/html/src/lib.rs index e900eaf..e9ce9ae 100644 --- a/crates/html/src/lib.rs +++ b/crates/html/src/lib.rs @@ -1 +1,35 @@ //! HTML5 tokenizer and tree builder. + +/// A token emitted by the HTML tokenizer. +#[derive(Debug, Clone, PartialEq)] +pub enum Token { + /// `` + Doctype { + name: Option, + public_id: Option, + system_id: Option, + force_quirks: bool, + }, + /// `` + StartTag { + name: String, + attributes: Vec<(String, String)>, + self_closing: bool, + }, + /// `` + EndTag { name: String }, + /// Character data (may be coalesced). + Character(String), + /// `` + Comment(String), + /// End of file. + Eof, +} + +/// Tokenize an HTML input string into a sequence of tokens. +/// +/// This is a stub that returns an empty `Vec`. The real implementation +/// will be a spec-compliant HTML5 tokenizer state machine. +pub fn tokenize(_input: &str) -> Vec { + Vec::new() +} diff --git a/crates/html/tests/html5lib_tokenizer.rs b/crates/html/tests/html5lib_tokenizer.rs new file mode 100644 index 0000000..eae8559 --- /dev/null +++ b/crates/html/tests/html5lib_tokenizer.rs @@ -0,0 +1,263 @@ +//! html5lib tokenizer test harness. +//! +//! Reads JSON test files from `tests/html5lib-tests/tokenizer/` and runs each +//! test case against our HTML tokenizer. Reports pass/fail/skip counts. +//! +//! Run with: `cargo test -p we-html --test html5lib_tokenizer` + +mod json; + +use json::JsonValue; +use we_html::Token; + +/// Workspace root relative to the crate directory. +const WORKSPACE_ROOT: &str = concat!(env!("CARGO_MANIFEST_DIR"), "/../../"); + +/// Convert a JSON output token (array) into our `Token` type for comparison. +fn json_to_token(val: &JsonValue) -> Option { + let arr = val.as_array()?; + let kind = arr.first()?.as_str()?; + match kind { + "DOCTYPE" => { + let name = arr.get(1).and_then(|v| v.as_str()).map(String::from); + let public_id = match arr.get(2) { + Some(JsonValue::Null) => None, + Some(v) => v.as_str().map(String::from), + None => None, + }; + let system_id = match arr.get(3) { + Some(JsonValue::Null) => None, + Some(v) => v.as_str().map(String::from), + None => None, + }; + let correctness = arr.get(4).and_then(|v| v.as_bool()).unwrap_or(true); + Some(Token::Doctype { + name, + public_id, + system_id, + force_quirks: !correctness, + }) + } + "StartTag" => { + let name = arr.get(1)?.as_str()?.to_string(); + let mut attributes = Vec::new(); + if let Some(attrs_obj) = arr.get(2).and_then(|v| v.as_object()) { + for (k, v) in attrs_obj { + let val_str = v.as_str().unwrap_or("").to_string(); + attributes.push((k.clone(), val_str)); + } + } + let self_closing = arr.get(3).and_then(|v| v.as_bool()).unwrap_or(false); + Some(Token::StartTag { + name, + attributes, + self_closing, + }) + } + "EndTag" => { + let name = arr.get(1)?.as_str()?.to_string(); + Some(Token::EndTag { name }) + } + "Character" => { + let data = arr.get(1)?.as_str()?.to_string(); + Some(Token::Character(data)) + } + "Comment" => { + let data = arr.get(1)?.as_str()?.to_string(); + Some(Token::Comment(data)) + } + _ => None, + } +} + +/// Apply double-escaping as described in the html5lib test format. +/// When `doubleEscaped` is true, the input and expected strings contain +/// literal `\uXXXX` sequences that should be decoded. +fn unescape_double_escaped(s: &str) -> String { + let mut result = String::new(); + let mut chars = s.chars(); + while let Some(ch) = chars.next() { + if ch == '\\' { + match chars.next() { + Some('u') => { + let hex: String = chars.by_ref().take(4).collect(); + if hex.len() == 4 { + if let Ok(cp) = u32::from_str_radix(&hex, 16) { + if let Some(c) = char::from_u32(cp) { + result.push(c); + continue; + } + } + } + result.push('\\'); + result.push('u'); + result.push_str(&hex); + } + Some(other) => { + result.push('\\'); + result.push(other); + } + None => { + result.push('\\'); + } + } + } else { + result.push(ch); + } + } + result +} + +/// Run a single test case and return whether it passed. +fn run_test_case(test: &JsonValue, double_escaped: bool) -> bool { + let input = match test.get("input").and_then(|v| v.as_str()) { + Some(s) => { + if double_escaped { + unescape_double_escaped(s) + } else { + s.to_string() + } + } + None => return false, + }; + + let expected_output = match test.get("output").and_then(|v| v.as_array()) { + Some(arr) => arr, + None => return false, + }; + + // Convert expected output tokens. + let expected_tokens: Vec = expected_output + .iter() + .filter_map(|tok_json| { + let mut tok = json_to_token(tok_json)?; + if double_escaped { + match &mut tok { + Token::Character(ref mut s) => *s = unescape_double_escaped(s), + Token::Comment(ref mut s) => *s = unescape_double_escaped(s), + _ => {} + } + } + Some(tok) + }) + .collect(); + + // Run our tokenizer. + let actual_tokens = we_html::tokenize(&input); + + actual_tokens == expected_tokens +} + +/// Load and run all test cases from a single html5lib tokenizer test file. +fn run_test_file(path: &std::path::Path) -> (usize, usize, usize) { + let content = match std::fs::read_to_string(path) { + Ok(c) => c, + Err(e) => { + eprintln!(" failed to read {}: {}", path.display(), e); + return (0, 0, 1); + } + }; + + let root = match json::parse(&content) { + Ok(v) => v, + Err(e) => { + eprintln!(" failed to parse {}: {}", path.display(), e); + return (0, 0, 1); + } + }; + + let tests = match root.get("tests").and_then(|v| v.as_array()) { + Some(t) => t, + None => { + eprintln!(" no 'tests' array in {}", path.display()); + return (0, 0, 1); + } + }; + + let mut pass = 0; + let mut fail = 0; + let mut skip = 0; + + for test in tests { + let desc = test + .get("description") + .and_then(|v| v.as_str()) + .unwrap_or(""); + + let double_escaped = test + .get("doubleEscaped") + .and_then(|v| v.as_bool()) + .unwrap_or(false); + + // If the test specifies initialStates, we run once per state. + // For now we only support the default "Data state" so skip others. + if let Some(states) = test.get("initialStates").and_then(|v| v.as_array()) { + let has_data_state = states.iter().any(|s| s.as_str() == Some("Data state")); + if !has_data_state { + skip += 1; + continue; + } + } + + if run_test_case(test, double_escaped) { + pass += 1; + } else { + fail += 1; + // Only print first few failures to avoid noise. + if fail <= 5 { + eprintln!(" FAIL: {}", desc); + } + } + } + + (pass, fail, skip) +} + +#[test] +fn html5lib_tokenizer_tests() { + let test_dir = std::path::PathBuf::from(WORKSPACE_ROOT).join("tests/html5lib-tests/tokenizer"); + + if !test_dir.exists() { + eprintln!( + "html5lib-tests submodule not checked out at {}", + test_dir.display() + ); + eprintln!("Run: git submodule update --init tests/html5lib-tests"); + // Don't fail the test — the submodule might not be initialized. + return; + } + + let mut total_pass = 0; + let mut total_fail = 0; + let mut total_skip = 0; + + let mut entries: Vec<_> = std::fs::read_dir(&test_dir) + .expect("failed to read tokenizer test dir") + .filter_map(|e| e.ok()) + .filter(|e| e.path().extension().map_or(false, |ext| ext == "test")) + .collect(); + entries.sort_by_key(|e| e.file_name()); + + for entry in &entries { + let path = entry.path(); + let name = path.file_name().unwrap().to_string_lossy(); + let (pass, fail, skip) = run_test_file(&path); + eprintln!("{}: {} pass, {} fail, {} skip", name, pass, fail, skip); + total_pass += pass; + total_fail += fail; + total_skip += skip; + } + + eprintln!(); + eprintln!( + "html5lib tokenizer totals: {} pass, {} fail, {} skip ({} total)", + total_pass, + total_fail, + total_skip, + total_pass + total_fail + total_skip + ); + + // The test "passes" as a harness — it reports results but doesn't fail + // the test suite until we have an implementation to measure against. + // This lets CI always run and report progress. +} diff --git a/crates/html/tests/json.rs b/crates/html/tests/json.rs new file mode 100644 index 0000000..a54f136 --- /dev/null +++ b/crates/html/tests/json.rs @@ -0,0 +1,339 @@ +//! Minimal JSON parser for reading html5lib test fixtures. +//! +//! Supports the subset of JSON used by html5lib-tests: objects, arrays, +//! strings (with escape sequences including `\uXXXX`), numbers, booleans, +//! and null. + +#[derive(Debug, Clone, PartialEq)] +pub enum JsonValue { + Null, + Bool(bool), + Number(f64), + Str(String), + Array(Vec), + Object(Vec<(String, JsonValue)>), +} + +impl JsonValue { + pub fn as_str(&self) -> Option<&str> { + match self { + JsonValue::Str(s) => Some(s), + _ => None, + } + } + + pub fn as_array(&self) -> Option<&[JsonValue]> { + match self { + JsonValue::Array(a) => Some(a), + _ => None, + } + } + + pub fn as_object(&self) -> Option<&[(String, JsonValue)]> { + match self { + JsonValue::Object(o) => Some(o), + _ => None, + } + } + + pub fn as_bool(&self) -> Option { + match self { + JsonValue::Bool(b) => Some(*b), + _ => None, + } + } + + /// Look up a key in a JSON object. + pub fn get(&self, key: &str) -> Option<&JsonValue> { + match self { + JsonValue::Object(pairs) => pairs.iter().find(|(k, _)| k == key).map(|(_, v)| v), + _ => None, + } + } +} + +struct Parser<'a> { + bytes: &'a [u8], + pos: usize, +} + +impl<'a> Parser<'a> { + fn new(input: &'a str) -> Self { + Self { + bytes: input.as_bytes(), + pos: 0, + } + } + + fn skip_ws(&mut self) { + while self.pos < self.bytes.len() { + match self.bytes[self.pos] { + b' ' | b'\t' | b'\n' | b'\r' => self.pos += 1, + _ => break, + } + } + } + + fn peek(&self) -> Option { + self.bytes.get(self.pos).copied() + } + + fn advance(&mut self) -> Option { + let b = self.bytes.get(self.pos).copied()?; + self.pos += 1; + Some(b) + } + + fn expect(&mut self, ch: u8) -> Result<(), String> { + match self.advance() { + Some(b) if b == ch => Ok(()), + Some(b) => Err(format!( + "expected '{}' at pos {}, got '{}'", + ch as char, self.pos, b as char + )), + None => Err(format!( + "expected '{}' at pos {}, got EOF", + ch as char, self.pos + )), + } + } + + fn parse_value(&mut self) -> Result { + self.skip_ws(); + match self.peek() { + Some(b'"') => self.parse_string().map(JsonValue::Str), + Some(b'{') => self.parse_object(), + Some(b'[') => self.parse_array(), + Some(b't') => self.parse_literal("true", JsonValue::Bool(true)), + Some(b'f') => self.parse_literal("false", JsonValue::Bool(false)), + Some(b'n') => self.parse_literal("null", JsonValue::Null), + Some(b'-') | Some(b'0'..=b'9') => self.parse_number(), + Some(b) => Err(format!( + "unexpected byte '{}' at pos {}", + b as char, self.pos + )), + None => Err("unexpected EOF".into()), + } + } + + fn parse_string(&mut self) -> Result { + self.expect(b'"')?; + let mut s = String::new(); + loop { + match self.advance() { + Some(b'"') => return Ok(s), + Some(b'\\') => match self.advance() { + Some(b'"') => s.push('"'), + Some(b'\\') => s.push('\\'), + Some(b'/') => s.push('/'), + Some(b'n') => s.push('\n'), + Some(b'r') => s.push('\r'), + Some(b't') => s.push('\t'), + Some(b'b') => s.push('\u{0008}'), + Some(b'f') => s.push('\u{000C}'), + Some(b'u') => { + let cp = self.parse_hex4()?; + // Handle surrogate pairs. + if (0xD800..=0xDBFF).contains(&cp) { + // High surrogate — expect \uXXXX low surrogate. + if self.advance() == Some(b'\\') && self.advance() == Some(b'u') { + let lo = self.parse_hex4()?; + if (0xDC00..=0xDFFF).contains(&lo) { + let combined = 0x10000 + + ((cp as u32 - 0xD800) << 10) + + (lo as u32 - 0xDC00); + if let Some(ch) = char::from_u32(combined) { + s.push(ch); + } + } + } + } else if let Some(ch) = char::from_u32(cp as u32) { + s.push(ch); + } + } + Some(b) => { + s.push('\\'); + s.push(b as char); + } + None => return Err("unexpected EOF in string escape".into()), + }, + Some(_) => { + // We need to handle multi-byte UTF-8 properly. + // Since we're working on bytes, back up and grab the char. + self.pos -= 1; + let rest = std::str::from_utf8(&self.bytes[self.pos..]) + .map_err(|e| format!("invalid UTF-8: {}", e))?; + let ch = rest.chars().next().unwrap(); + self.pos += ch.len_utf8(); + s.push(ch); + } + None => return Err("unexpected EOF in string".into()), + } + } + } + + fn parse_hex4(&mut self) -> Result { + let mut val: u16 = 0; + for _ in 0..4 { + let b = self.advance().ok_or("unexpected EOF in \\u escape")?; + let digit = match b { + b'0'..=b'9' => b - b'0', + b'a'..=b'f' => b - b'a' + 10, + b'A'..=b'F' => b - b'A' + 10, + _ => return Err(format!("invalid hex digit '{}'", b as char)), + }; + val = val * 16 + digit as u16; + } + Ok(val) + } + + fn parse_number(&mut self) -> Result { + let start = self.pos; + if self.peek() == Some(b'-') { + self.pos += 1; + } + while self.pos < self.bytes.len() && self.bytes[self.pos].is_ascii_digit() { + self.pos += 1; + } + if self.pos < self.bytes.len() && self.bytes[self.pos] == b'.' { + self.pos += 1; + while self.pos < self.bytes.len() && self.bytes[self.pos].is_ascii_digit() { + self.pos += 1; + } + } + if self.pos < self.bytes.len() + && (self.bytes[self.pos] == b'e' || self.bytes[self.pos] == b'E') + { + self.pos += 1; + if self.pos < self.bytes.len() + && (self.bytes[self.pos] == b'+' || self.bytes[self.pos] == b'-') + { + self.pos += 1; + } + while self.pos < self.bytes.len() && self.bytes[self.pos].is_ascii_digit() { + self.pos += 1; + } + } + let s = std::str::from_utf8(&self.bytes[start..self.pos]) + .map_err(|e| format!("invalid UTF-8 in number: {}", e))?; + let n: f64 = s + .parse() + .map_err(|e| format!("invalid number '{}': {}", s, e))?; + Ok(JsonValue::Number(n)) + } + + fn parse_object(&mut self) -> Result { + self.expect(b'{')?; + self.skip_ws(); + let mut pairs = Vec::new(); + if self.peek() == Some(b'}') { + self.pos += 1; + return Ok(JsonValue::Object(pairs)); + } + loop { + self.skip_ws(); + let key = self.parse_string()?; + self.skip_ws(); + self.expect(b':')?; + let val = self.parse_value()?; + pairs.push((key, val)); + self.skip_ws(); + match self.peek() { + Some(b',') => { + self.pos += 1; + } + Some(b'}') => { + self.pos += 1; + return Ok(JsonValue::Object(pairs)); + } + _ => return Err(format!("expected ',' or '}}' at pos {}", self.pos)), + } + } + } + + fn parse_array(&mut self) -> Result { + self.expect(b'[')?; + self.skip_ws(); + let mut elems = Vec::new(); + if self.peek() == Some(b']') { + self.pos += 1; + return Ok(JsonValue::Array(elems)); + } + loop { + let val = self.parse_value()?; + elems.push(val); + self.skip_ws(); + match self.peek() { + Some(b',') => { + self.pos += 1; + } + Some(b']') => { + self.pos += 1; + return Ok(JsonValue::Array(elems)); + } + _ => return Err(format!("expected ',' or ']' at pos {}", self.pos)), + } + } + } + + fn parse_literal(&mut self, expected: &str, value: JsonValue) -> Result { + for b in expected.bytes() { + match self.advance() { + Some(got) if got == b => {} + _ => return Err(format!("expected literal '{}'", expected)), + } + } + Ok(value) + } +} + +/// Parse a JSON string into a `JsonValue`. +pub fn parse(input: &str) -> Result { + let mut parser = Parser::new(input); + let val = parser.parse_value()?; + parser.skip_ws(); + if parser.pos != parser.bytes.len() { + return Err(format!("trailing data at pos {}", parser.pos)); + } + Ok(val) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn parse_simple_object() { + let val = parse(r#"{"a": 1, "b": "hello"}"#).unwrap(); + assert_eq!(val.get("a"), Some(&JsonValue::Number(1.0))); + assert_eq!(val.get("b"), Some(&JsonValue::Str("hello".into()))); + } + + #[test] + fn parse_array() { + let val = parse(r#"[1, "two", true, null]"#).unwrap(); + let arr = val.as_array().unwrap(); + assert_eq!(arr.len(), 4); + assert_eq!(arr[2], JsonValue::Bool(true)); + assert_eq!(arr[3], JsonValue::Null); + } + + #[test] + fn parse_nested() { + let val = parse(r#"{"tests": [{"desc": "a"}]}"#).unwrap(); + let tests = val.get("tests").unwrap().as_array().unwrap(); + assert_eq!(tests.len(), 1); + } + + #[test] + fn parse_string_escapes() { + let val = parse(r#""hello\nworld""#).unwrap(); + assert_eq!(val.as_str().unwrap(), "hello\nworld"); + } + + #[test] + fn parse_unicode_escape() { + let val = parse(r#""\u0041""#).unwrap(); + assert_eq!(val.as_str().unwrap(), "A"); + } +} diff --git a/crates/js/src/lib.rs b/crates/js/src/lib.rs index ac0e648..76d0bf5 100644 --- a/crates/js/src/lib.rs +++ b/crates/js/src/lib.rs @@ -1 +1,32 @@ //! JavaScript engine — lexer, parser, bytecode, register VM, GC, JIT (AArch64). + +use std::fmt; + +/// An error produced by the JavaScript engine. +#[derive(Debug)] +pub enum JsError { + /// The engine does not yet support this feature or syntax. + NotImplemented, + /// A parse/syntax error in the source. + SyntaxError(String), + /// A runtime error during execution. + RuntimeError(String), +} + +impl fmt::Display for JsError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + JsError::NotImplemented => write!(f, "not implemented"), + JsError::SyntaxError(msg) => write!(f, "SyntaxError: {}", msg), + JsError::RuntimeError(msg) => write!(f, "RuntimeError: {}", msg), + } + } +} + +/// Evaluate a JavaScript source string and return the completion value. +/// +/// This is a stub that always returns `NotImplemented`. The real +/// implementation will lex, parse, compile to bytecode, and execute. +pub fn evaluate(_source: &str) -> Result<(), JsError> { + Err(JsError::NotImplemented) +} diff --git a/crates/js/tests/test262.rs b/crates/js/tests/test262.rs new file mode 100644 index 0000000..4775bf9 --- /dev/null +++ b/crates/js/tests/test262.rs @@ -0,0 +1,304 @@ +//! Test262 test harness. +//! +//! Walks the Test262 test suite and runs each test case against our JavaScript +//! engine. Reports pass/fail/skip counts. +//! +//! Run with: `cargo test -p we-js --test test262` + +/// Workspace root relative to the crate directory. +const WORKSPACE_ROOT: &str = concat!(env!("CARGO_MANIFEST_DIR"), "/../../"); + +/// Metadata extracted from a Test262 test file's YAML frontmatter. +struct TestMeta { + /// If true, the test expects a parse/early error. + negative_phase_parse: bool, + /// If true, the test expects a runtime error. + negative_phase_runtime: bool, + /// The expected error type for negative tests (e.g. "SyntaxError"). + negative_type: Option, + /// If true, this is an async test. + is_async: bool, + /// If true, this test should be run as a module. + is_module: bool, + /// If true, skip the harness preamble. + is_raw: bool, + /// Required features. + features: Vec, + /// Required harness includes. + includes: Vec, +} + +impl TestMeta { + fn should_skip(&self) -> bool { + // Skip async tests and module tests for now. + self.is_async || self.is_module + } +} + +/// Parse the YAML-ish frontmatter from a Test262 test file. +/// +/// The frontmatter is between `/*---` and `---*/`. +fn parse_frontmatter(source: &str) -> TestMeta { + let mut meta = TestMeta { + negative_phase_parse: false, + negative_phase_runtime: false, + negative_type: None, + is_async: false, + is_module: false, + is_raw: false, + features: Vec::new(), + includes: Vec::new(), + }; + + let start = match source.find("/*---") { + Some(i) => i + 5, + None => return meta, + }; + let end = match source[start..].find("---*/") { + Some(i) => start + i, + None => return meta, + }; + let yaml = &source[start..end]; + + // Very simple line-by-line YAML extraction. + let mut in_negative = false; + let mut in_features = false; + let mut in_includes = false; + let mut in_flags = false; + + for line in yaml.lines() { + let trimmed = line.trim(); + + // Detect top-level keys (not indented or with specific indent). + if !trimmed.is_empty() && !trimmed.starts_with('-') && !line.starts_with(' ') { + in_negative = false; + in_features = false; + in_includes = false; + in_flags = false; + } + + if trimmed.starts_with("negative:") { + in_negative = true; + continue; + } + if trimmed.starts_with("features:") { + in_features = true; + // Check for inline list: features: [a, b] + if let Some(rest) = trimmed.strip_prefix("features:") { + let rest = rest.trim(); + if rest.starts_with('[') && rest.ends_with(']') { + let inner = &rest[1..rest.len() - 1]; + for item in inner.split(',') { + let item = item.trim(); + if !item.is_empty() { + meta.features.push(item.to_string()); + } + } + in_features = false; + } + } + continue; + } + if trimmed.starts_with("includes:") { + in_includes = true; + if let Some(rest) = trimmed.strip_prefix("includes:") { + let rest = rest.trim(); + if rest.starts_with('[') && rest.ends_with(']') { + let inner = &rest[1..rest.len() - 1]; + for item in inner.split(',') { + let item = item.trim(); + if !item.is_empty() { + meta.includes.push(item.to_string()); + } + } + in_includes = false; + } + } + continue; + } + if trimmed.starts_with("flags:") { + in_flags = true; + if let Some(rest) = trimmed.strip_prefix("flags:") { + let rest = rest.trim(); + if rest.starts_with('[') && rest.ends_with(']') { + let inner = &rest[1..rest.len() - 1]; + for item in inner.split(',') { + let flag = item.trim(); + match flag { + "async" => meta.is_async = true, + "module" => meta.is_module = true, + "raw" => meta.is_raw = true, + _ => {} + } + } + in_flags = false; + } + } + continue; + } + + // Handle list items under current key. + if let Some(item) = trimmed.strip_prefix("- ") { + if in_features { + meta.features.push(item.to_string()); + } else if in_includes { + meta.includes.push(item.to_string()); + } else if in_flags { + match item { + "async" => meta.is_async = true, + "module" => meta.is_module = true, + "raw" => meta.is_raw = true, + _ => {} + } + } + continue; + } + + // Handle sub-keys under negative. + if in_negative { + if let Some(rest) = trimmed.strip_prefix("phase:") { + let phase = rest.trim(); + match phase { + "parse" | "early" => meta.negative_phase_parse = true, + "runtime" | "resolution" => meta.negative_phase_runtime = true, + _ => {} + } + } + if let Some(rest) = trimmed.strip_prefix("type:") { + meta.negative_type = Some(rest.trim().to_string()); + } + } + } + + meta +} + +/// Recursively collect all `.js` test files under a directory. +fn collect_test_files(dir: &std::path::Path, files: &mut Vec) { + let entries = match std::fs::read_dir(dir) { + Ok(e) => e, + Err(_) => return, + }; + let mut entries: Vec<_> = entries.filter_map(|e| e.ok()).collect(); + entries.sort_by_key(|e| e.file_name()); + + for entry in entries { + let path = entry.path(); + if path.is_dir() { + collect_test_files(&path, files); + } else if path.extension().map_or(false, |e| e == "js") { + // Skip _FIXTURE files (test helpers, not tests themselves). + let name = path.file_name().unwrap().to_string_lossy(); + if !name.contains("_FIXTURE") { + files.push(path); + } + } + } +} + +/// Run a single Test262 test file. Returns (pass, fail, skip). +fn run_test(path: &std::path::Path) -> (usize, usize, usize) { + let source = match std::fs::read_to_string(path) { + Ok(s) => s, + Err(_) => return (0, 0, 1), + }; + + let meta = parse_frontmatter(&source); + + if meta.should_skip() { + return (0, 0, 1); + } + + // For negative parse tests, if our evaluate returns an error, that's a pass. + // For positive tests, evaluate should succeed (return Ok). + let result = we_js::evaluate(&source); + + if meta.negative_phase_parse { + // We expect a parse error. If our engine returns any error, count as pass. + match result { + Err(_) => (1, 0, 0), + Ok(()) => (0, 1, 0), + } + } else { + // We expect success. + match result { + Ok(()) => (1, 0, 0), + Err(_) => (0, 1, 0), + } + } +} + +#[test] +fn test262_language_tests() { + let test_dir = std::path::PathBuf::from(WORKSPACE_ROOT).join("tests/test262/test/language"); + + if !test_dir.exists() { + eprintln!( + "test262 submodule not checked out at {}", + test_dir.display() + ); + eprintln!("Run: git submodule update --init tests/test262"); + return; + } + + let mut files = Vec::new(); + collect_test_files(&test_dir, &mut files); + + let mut total_pass = 0; + let mut total_fail = 0; + let mut total_skip = 0; + + // Group results by top-level subdirectory for reporting. + let mut current_group = String::new(); + let mut group_pass = 0; + let mut group_fail = 0; + let mut group_skip = 0; + + for path in &files { + // Determine the top-level group (e.g. "expressions", "literals"). + let rel = path.strip_prefix(&test_dir).unwrap_or(path); + let group = rel + .components() + .next() + .map(|c| c.as_os_str().to_string_lossy().to_string()) + .unwrap_or_default(); + + if group != current_group { + if !current_group.is_empty() { + eprintln!( + " {}: {} pass, {} fail, {} skip", + current_group, group_pass, group_fail, group_skip + ); + } + current_group = group; + group_pass = 0; + group_fail = 0; + group_skip = 0; + } + + let (p, f, s) = run_test(path); + group_pass += p; + group_fail += f; + group_skip += s; + total_pass += p; + total_fail += f; + total_skip += s; + } + + // Print last group. + if !current_group.is_empty() { + eprintln!( + " {}: {} pass, {} fail, {} skip", + current_group, group_pass, group_fail, group_skip + ); + } + + eprintln!(); + eprintln!( + "Test262 language totals: {} pass, {} fail, {} skip ({} total)", + total_pass, + total_fail, + total_skip, + total_pass + total_fail + total_skip + ); +} diff --git a/tests/html5lib-tests b/tests/html5lib-tests new file mode 160000 index 0000000..8f43b7e --- /dev/null +++ b/tests/html5lib-tests @@ -0,0 +1 @@ +Subproject commit 8f43b7ec8c9d02179f5f38e0ea08cb5000fb9c9e diff --git a/tests/test262 b/tests/test262 new file mode 160000 index 0000000..93d6396 --- /dev/null +++ b/tests/test262 @@ -0,0 +1 @@ +Subproject commit 93d63969bccbf8b4471b7c7fadc875099b7668d3