diff --git a/src/srd/verify/normalize.rs b/src/srd/verify/normalize.rs index 93eacfa..d5c82a5 100644 --- a/src/srd/verify/normalize.rs +++ b/src/srd/verify/normalize.rs @@ -2,12 +2,13 @@ //! //! A word stream drops everything markdown uses for structure (headings, //! emphasis, table pipes and alignment rows, list bullets, horizontal -//! rules) and keeps everything else, so a healed line break disappears -//! but a changed, dropped, or added word does not. Structural characters -//! become spaces rather than being deleted outright, because the vendored -//! source sometimes runs a value straight into the next bold label with no -//! space (`9**Languages**`); deleting the markers would fuse the two words -//! into one and hide the boundary that whitespace-splitting depends on. +//! rules, leading blockquote markers) and keeps everything else, so a +//! healed line break disappears but a changed, dropped, or added word does +//! not. Structural characters become spaces rather than being deleted +//! outright, because the vendored source sometimes runs a value straight +//! into the next bold label with no space (`9**Languages**`); deleting the +//! markers would fuse the two words into one and hide the boundary that +//! whitespace-splitting depends on. use std::sync::LazyLock; @@ -21,6 +22,8 @@ static LEADING_BULLET: LazyLock = static HEADING_MARKER: LazyLock = LazyLock::new(|| Regex::new(r"^#{1,6}\s+").unwrap()); +static BLOCKQUOTE_MARKER: LazyLock = LazyLock::new(|| Regex::new(r"^(?:>\s*)+").unwrap()); + /// Splits `text` into its normalized word stream. pub fn words(text: &str) -> Vec { let mut buffer = String::new(); @@ -29,7 +32,8 @@ pub fn words(text: &str) -> Vec { if trimmed.is_empty() || is_horizontal_rule(trimmed) || is_table_rule(trimmed) { continue; } - let without_heading = HEADING_MARKER.replace(line, ""); + let without_quote = BLOCKQUOTE_MARKER.replace(line, ""); + let without_heading = HEADING_MARKER.replace(&without_quote, ""); let without_bullet = LEADING_BULLET.replace(&without_heading, ""); for ch in without_bullet.chars() { buffer.push(normalize_char(ch)); @@ -173,4 +177,22 @@ mod tests { fn ignores_blank_lines() { assert_eq!(words("first\n\n\nsecond"), vec!["first", "second"]); } + + #[test] + fn strips_a_leading_blockquote_marker() { + assert_eq!(words("> Quoted text"), vec!["Quoted", "text"]); + } + + #[test] + fn strips_nested_leading_blockquote_markers() { + assert_eq!(words("> > Quoted text"), vec!["Quoted", "text"]); + } + + #[test] + fn keeps_a_mid_line_greater_than_sign_significant() { + assert_eq!( + words("HP > 10 means bloodied"), + vec!["HP", ">", "10", "means", "bloodied"] + ); + } } diff --git a/src/srd/verify/readme_meta.rs b/src/srd/verify/readme_meta.rs index 82a7db3..489b709 100644 --- a/src/srd/verify/readme_meta.rs +++ b/src/srd/verify/readme_meta.rs @@ -62,8 +62,13 @@ fn attribution_paragraph(layer_root: &Path) -> Result { .ok_or_else(|| format!("{LEGAL_PATH} has no attribution paragraph to check against")) } +/// Removes every whitespace character from `text`. The vendored `Legal.md` +/// has a stray space inside its license URL ("by/4.0/ legalcode"), so the +/// containment check below cannot line up on whitespace at all: it can only +/// confirm that the same words, in the same order, are present on both +/// sides. fn normalize_whitespace(text: &str) -> String { - text.split_whitespace().collect::>().join(" ") + text.chars().filter(|ch| !ch.is_whitespace()).collect() } /// Checks `meta.yaml` parses and records a version and both source pins @@ -203,6 +208,29 @@ mod tests { ); } + #[test] + fn check_readme_tolerates_a_url_the_source_wrote_with_a_stray_space() { + // Legal.md's own attribution paragraph has a stray space inside the + // license URL ("by/4.0/ legalcode"). A README that writes the URL + // correctly, with no space, must still be accepted. + let source_attribution = format!( + "{ATTRIBUTION} It is licensed under CC-BY-4.0, available at \ + https://creativecommons.org/licenses/by/4.0/ legalcode." + ); + let readme_attribution = + source_attribution.replace("by/4.0/ legalcode", "by/4.0/legalcode"); + let layer_root = layer_with_legal(&source_attribution); + write_file( + &layer_root.join(README_PATH), + &format!("# SRD\n\n{readme_attribution}\n\nSee {SOURCE_REPO_URL}.\n"), + ); + let mut failures = Vec::new(); + + check_readme(&layer_root, &mut failures); + + assert_eq!(failures, vec![]); + } + #[test] fn check_readme_reports_a_missing_source_link() { let layer_root = layer_with_legal(ATTRIBUTION); diff --git a/src/srd/verify/slug.rs b/src/srd/verify/slug.rs index a1b322e..98b9f5d 100644 --- a/src/srd/verify/slug.rs +++ b/src/srd/verify/slug.rs @@ -1,19 +1,16 @@ //! The deterministic address every corpus file's name must produce. -/// Lowercases `name`, turns `+` into `plus`, and collapses every other run -/// of non-alphanumeric characters into a single `-`, with no leading or -/// trailing `-`. +/// Lowercases `name`, turns `+` into the word `plus`, and collapses every +/// other run of non-alphanumeric characters into a single `-`, with no +/// leading or trailing `-`. `+` is a word in its own right, so it gets a +/// `-` on both sides just like any other word: `+1` becomes `plus-1`, not +/// `plus1`. pub fn slugify(name: &str) -> String { - let mut slug = String::with_capacity(name.len()); + let expanded = name.replace('+', " plus "); + let mut slug = String::with_capacity(expanded.len()); let mut pending_dash = false; - for ch in name.chars() { - if ch == '+' { - if pending_dash && !slug.is_empty() { - slug.push('-'); - } - pending_dash = false; - slug.push_str("plus"); - } else if ch.is_ascii_alphanumeric() { + for ch in expanded.chars() { + if ch.is_ascii_alphanumeric() { if pending_dash && !slug.is_empty() { slug.push('-'); } @@ -46,13 +43,21 @@ mod tests { } #[test] - fn turns_plus_into_plus() { - assert_eq!(slugify("+1 Armor"), "plus1-armor"); + fn turns_plus_into_the_word_plus_as_its_own_word() { + assert_eq!(slugify("+1 Armor"), "plus-1-armor"); + } + + #[test] + fn separates_repeated_pluses_like_the_brief_example() { + assert_eq!( + slugify("Armor, +1, +2, or +3"), + "armor-plus-1-plus-2-or-plus-3" + ); } #[test] fn dashes_before_a_mid_string_plus() { - assert_eq!(slugify("Ammunition, +1"), "ammunition-plus1"); + assert_eq!(slugify("Ammunition, +1"), "ammunition-plus-1"); } #[test] @@ -61,7 +66,7 @@ mod tests { } #[test] - fn substitutes_plus_in_place_without_forcing_a_word_boundary() { - assert_eq!(slugify("Boots+Cloak"), "bootspluscloak"); + fn treats_plus_as_a_word_boundary_even_with_no_surrounding_spaces() { + assert_eq!(slugify("Boots+Cloak"), "boots-plus-cloak"); } }