From ac257c31cb8a636337ce610521b56568875ea9cc Mon Sep 17 00:00:00 2001 From: Chris Guidry Date: Thu, 30 Jul 2026 11:04:37 -0400 Subject: [PATCH] Fix three verify normalization gaps: plus slugs, README URL, blockquotes The slug rule fused a leading '+' straight into the following digit ("+1" became "plus1" instead of "plus-1"), so slugify now expands '+' to " plus " before running its normal separator-collapse pass, giving '+' the same word-boundary behavior as any other separator. The README attribution check required a byte-exact match against the vendored Legal.md paragraph, but Legal.md itself has a stray space inside its license URL ("by/4.0/ legalcode"). A README that writes the URL correctly, with no space, failed the check. The whitespace normalization on both sides of the containment check now strips all whitespace instead of collapsing it, so the check verifies the same words in the same order and ignores where the whitespace falls. The word-stream normalizer stripped headings, bullets, and other markdown furniture as separators but left blockquote markers alone, so a quoted line's leading '>' (or nested '> >') counted as a word and broke fidelity comparisons between corpus and source. normalize now strips line-leading '>' runs the same way it strips other markdown structure, while leaving '>' elsewhere on a line untouched. Co-Authored-By: Claude Fable 5 --- src/srd/verify/normalize.rs | 36 +++++++++++++++++++++++++------- src/srd/verify/readme_meta.rs | 30 ++++++++++++++++++++++++++- src/srd/verify/slug.rs | 39 ++++++++++++++++++++--------------- 3 files changed, 80 insertions(+), 25 deletions(-) diff --git a/src/srd/verify/normalize.rs b/src/srd/verify/normalize.rs index 93eacfa..d5c82a5 100644 --- a/src/srd/verify/normalize.rs +++ b/src/srd/verify/normalize.rs @@ -2,12 +2,13 @@ //! //! A word stream drops everything markdown uses for structure (headings, //! emphasis, table pipes and alignment rows, list bullets, horizontal -//! rules) and keeps everything else, so a healed line break disappears -//! but a changed, dropped, or added word does not. Structural characters -//! become spaces rather than being deleted outright, because the vendored -//! source sometimes runs a value straight into the next bold label with no -//! space (`9**Languages**`); deleting the markers would fuse the two words -//! into one and hide the boundary that whitespace-splitting depends on. +//! rules, leading blockquote markers) and keeps everything else, so a +//! healed line break disappears but a changed, dropped, or added word does +//! not. Structural characters become spaces rather than being deleted +//! outright, because the vendored source sometimes runs a value straight +//! into the next bold label with no space (`9**Languages**`); deleting the +//! markers would fuse the two words into one and hide the boundary that +//! whitespace-splitting depends on. use std::sync::LazyLock; @@ -21,6 +22,8 @@ static LEADING_BULLET: LazyLock = static HEADING_MARKER: LazyLock = LazyLock::new(|| Regex::new(r"^#{1,6}\s+").unwrap()); +static BLOCKQUOTE_MARKER: LazyLock = LazyLock::new(|| Regex::new(r"^(?:>\s*)+").unwrap()); + /// Splits `text` into its normalized word stream. pub fn words(text: &str) -> Vec { let mut buffer = String::new(); @@ -29,7 +32,8 @@ pub fn words(text: &str) -> Vec { if trimmed.is_empty() || is_horizontal_rule(trimmed) || is_table_rule(trimmed) { continue; } - let without_heading = HEADING_MARKER.replace(line, ""); + let without_quote = BLOCKQUOTE_MARKER.replace(line, ""); + let without_heading = HEADING_MARKER.replace(&without_quote, ""); let without_bullet = LEADING_BULLET.replace(&without_heading, ""); for ch in without_bullet.chars() { buffer.push(normalize_char(ch)); @@ -173,4 +177,22 @@ mod tests { fn ignores_blank_lines() { assert_eq!(words("first\n\n\nsecond"), vec!["first", "second"]); } + + #[test] + fn strips_a_leading_blockquote_marker() { + assert_eq!(words("> Quoted text"), vec!["Quoted", "text"]); + } + + #[test] + fn strips_nested_leading_blockquote_markers() { + assert_eq!(words("> > Quoted text"), vec!["Quoted", "text"]); + } + + #[test] + fn keeps_a_mid_line_greater_than_sign_significant() { + assert_eq!( + words("HP > 10 means bloodied"), + vec!["HP", ">", "10", "means", "bloodied"] + ); + } } diff --git a/src/srd/verify/readme_meta.rs b/src/srd/verify/readme_meta.rs index 82a7db3..489b709 100644 --- a/src/srd/verify/readme_meta.rs +++ b/src/srd/verify/readme_meta.rs @@ -62,8 +62,13 @@ fn attribution_paragraph(layer_root: &Path) -> Result { .ok_or_else(|| format!("{LEGAL_PATH} has no attribution paragraph to check against")) } +/// Removes every whitespace character from `text`. The vendored `Legal.md` +/// has a stray space inside its license URL ("by/4.0/ legalcode"), so the +/// containment check below cannot line up on whitespace at all: it can only +/// confirm that the same words, in the same order, are present on both +/// sides. fn normalize_whitespace(text: &str) -> String { - text.split_whitespace().collect::>().join(" ") + text.chars().filter(|ch| !ch.is_whitespace()).collect() } /// Checks `meta.yaml` parses and records a version and both source pins @@ -203,6 +208,29 @@ mod tests { ); } + #[test] + fn check_readme_tolerates_a_url_the_source_wrote_with_a_stray_space() { + // Legal.md's own attribution paragraph has a stray space inside the + // license URL ("by/4.0/ legalcode"). A README that writes the URL + // correctly, with no space, must still be accepted. + let source_attribution = format!( + "{ATTRIBUTION} It is licensed under CC-BY-4.0, available at \ + https://creativecommons.org/licenses/by/4.0/ legalcode." + ); + let readme_attribution = + source_attribution.replace("by/4.0/ legalcode", "by/4.0/legalcode"); + let layer_root = layer_with_legal(&source_attribution); + write_file( + &layer_root.join(README_PATH), + &format!("# SRD\n\n{readme_attribution}\n\nSee {SOURCE_REPO_URL}.\n"), + ); + let mut failures = Vec::new(); + + check_readme(&layer_root, &mut failures); + + assert_eq!(failures, vec![]); + } + #[test] fn check_readme_reports_a_missing_source_link() { let layer_root = layer_with_legal(ATTRIBUTION); diff --git a/src/srd/verify/slug.rs b/src/srd/verify/slug.rs index a1b322e..98b9f5d 100644 --- a/src/srd/verify/slug.rs +++ b/src/srd/verify/slug.rs @@ -1,19 +1,16 @@ //! The deterministic address every corpus file's name must produce. -/// Lowercases `name`, turns `+` into `plus`, and collapses every other run -/// of non-alphanumeric characters into a single `-`, with no leading or -/// trailing `-`. +/// Lowercases `name`, turns `+` into the word `plus`, and collapses every +/// other run of non-alphanumeric characters into a single `-`, with no +/// leading or trailing `-`. `+` is a word in its own right, so it gets a +/// `-` on both sides just like any other word: `+1` becomes `plus-1`, not +/// `plus1`. pub fn slugify(name: &str) -> String { - let mut slug = String::with_capacity(name.len()); + let expanded = name.replace('+', " plus "); + let mut slug = String::with_capacity(expanded.len()); let mut pending_dash = false; - for ch in name.chars() { - if ch == '+' { - if pending_dash && !slug.is_empty() { - slug.push('-'); - } - pending_dash = false; - slug.push_str("plus"); - } else if ch.is_ascii_alphanumeric() { + for ch in expanded.chars() { + if ch.is_ascii_alphanumeric() { if pending_dash && !slug.is_empty() { slug.push('-'); } @@ -46,13 +43,21 @@ mod tests { } #[test] - fn turns_plus_into_plus() { - assert_eq!(slugify("+1 Armor"), "plus1-armor"); + fn turns_plus_into_the_word_plus_as_its_own_word() { + assert_eq!(slugify("+1 Armor"), "plus-1-armor"); + } + + #[test] + fn separates_repeated_pluses_like_the_brief_example() { + assert_eq!( + slugify("Armor, +1, +2, or +3"), + "armor-plus-1-plus-2-or-plus-3" + ); } #[test] fn dashes_before_a_mid_string_plus() { - assert_eq!(slugify("Ammunition, +1"), "ammunition-plus1"); + assert_eq!(slugify("Ammunition, +1"), "ammunition-plus-1"); } #[test] @@ -61,7 +66,7 @@ mod tests { } #[test] - fn substitutes_plus_in_place_without_forcing_a_word_boundary() { - assert_eq!(slugify("Boots+Cloak"), "bootspluscloak"); + fn treats_plus_as_a_word_boundary_even_with_no_surrounding_spaces() { + assert_eq!(slugify("Boots+Cloak"), "boots-plus-cloak"); } } -- 2.51.2