diff --git a/README.md b/README.md index 63d9a58..981bb48 100644 --- a/README.md +++ b/README.md @@ -25,9 +25,13 @@ Scan text from stdin: printf 'Let us delve into this robust ecosystem.' | cargo run -q -p tropius-cli ``` -Scan article text extracted from a live URL with [lectito](https://lectito.stormlightlabs.org/): +Scan article text extracted from a live URL with +[lectito](https://lectito.stormlightlabs.org/): ```sh +# Install lectito +cargo install lectito-cli + lectito 'https://www.solo.io/blog/what-is-agent-identity-human-workload-a-new-layer' \ --format text \ | cargo run -q -p tropius-cli @@ -41,11 +45,12 @@ Color output respects [`NO_COLOR`](https://no-color.org/). ## Coverage -Current coverage includes: - -- phrase patterns for 22 of 33 trope.fyi sections +- an implementation path for every source section in + [`tropes.md`](https://tropes.fyi/tropes-md) +- phrase patterns for literal trope signals - structural detectors for sentence and paragraph shape - repetition detectors for repeated metaphor terms and duplicated content +- markdown-aware detection for bold-first bullets - character-class detection for Unicode decoration ## Inspiration @@ -62,11 +67,13 @@ I got nerd-sniped on [BlueSky](https://bsky.app/profile/samuel.fm/post/3mp3l3cxg > [Samuel - @samuel.fm](https://bsky.app/profile/samuel.fm) **2026-06-25 04:22** > -> smells of claude, e.g. short, punchy yet needlessly florid prose, excessive claudeisms, it’s not just x it’s y etc +> smells of claude, e.g. short, punchy yet needlessly florid prose, excessive claudeisms, +> it’s not just x it’s y etc > [owais - @desertthunder.dev](https://bsky.app/profile/desertthunder.dev) **2026-06-25 04:26** > -> You know that feeling of nerd-sniping about to happen? I gotta use aho-corasick + [tropes.fyi](https://tropes.fyi) in some way +> You know that feeling of nerd-sniping about to happen? I gotta use aho-corasick + +> [tropes.fyi](https://tropes.fyi) in some way ## Further Reading diff --git a/crates/core/src/detector.rs b/crates/core/src/detector.rs index 8b09916..458658d 100644 --- a/crates/core/src/detector.rs +++ b/crates/core/src/detector.rs @@ -1,6 +1,7 @@ //! Text detectors for phrase-based and structural trope signals. pub mod char_class; +pub mod markdown; pub mod repetition; pub mod structural; @@ -56,6 +57,7 @@ impl Detector { pub fn scan(&self, text: &str) -> Vec { let mut findings = self.scan_phrases(text); findings.extend(char_class::scan_unicode_decoration(text)); + findings.extend(markdown::scan_markdown(text)); findings.extend(structural::scan_structural(text)); findings.extend(repetition::scan_repetition(text)); findings.sort_by_key(|finding| finding.span.start()); @@ -100,17 +102,11 @@ pub struct Finding { } impl Finding { - pub fn structural( - rule_id: &str, - rule_name: &str, - severity: Severity, - text: &str, - span: Span, - ) -> Finding { + pub fn structural(rule: (&str, &str), text: &str, span: Span) -> Finding { Finding { - rule_id: rule_id.to_owned(), - rule_name: rule_name.to_owned(), - severity, + rule_id: rule.0.to_owned(), + rule_name: rule.1.to_owned(), + severity: Severity::Medium, kind: FindingKind::Structural, matched: text[span.start()..span.end()].to_owned(), span, @@ -118,22 +114,28 @@ impl Finding { } /// Builds a repetition finding from a byte range in the scanned text. - pub fn repetition( - rule_id: &str, - rule_name: &str, - severity: Severity, - text: &str, - span: Span, - ) -> Finding { + pub fn repetition(rule: (&str, &str), severity: Severity, text: &str, span: Span) -> Finding { Finding { - rule_id: rule_id.to_owned(), - rule_name: rule_name.to_owned(), + rule_id: rule.0.to_owned(), + rule_name: rule.1.to_owned(), severity, kind: FindingKind::Repetition, matched: text[span.start()..span.end()].to_owned(), span, } } + + /// Builds a markdown-aware finding from a byte range in the scanned text. + pub fn markdown(rule: (&str, &str), severity: Severity, text: &str, span: Span) -> Finding { + Finding { + rule_id: rule.0.to_owned(), + rule_name: rule.1.to_owned(), + severity, + kind: FindingKind::Markdown, + matched: text[span.start()..span.end()].to_owned(), + span, + } + } } /// The kind of detector that produced a finding. @@ -147,6 +149,8 @@ pub enum FindingKind { Structural, /// Repeated document content matched by repetition detectors. Repetition, + /// Markdown syntax matched by markdown-aware detectors. + Markdown, } impl Display for FindingKind { @@ -156,6 +160,7 @@ impl Display for FindingKind { FindingKind::CharacterClass => "char", FindingKind::Structural => "struct", FindingKind::Repetition => "repeat", + FindingKind::Markdown => "markdown", }) } } diff --git a/crates/core/src/detector/markdown.rs b/crates/core/src/detector/markdown.rs new file mode 100644 index 0000000..d70b921 --- /dev/null +++ b/crates/core/src/detector/markdown.rs @@ -0,0 +1,111 @@ +//! Markdown-aware detectors for formatting tropes. + +use crate::patterns::Severity; + +use super::{Finding, Span}; + +/// Finds markdown-specific trope signals. +pub fn scan_markdown(text: &str) -> Vec { + line_spans(text) + .into_iter() + .filter(|line| starts_with_bold_list_item(&text[line.start()..line.end()])) + .map(|line| { + Finding::markdown( + ("formatting.bold_first_bullets", "Bold-First Bullets"), + Severity::Medium, + text, + line, + ) + }) + .collect() +} + +fn starts_with_bold_list_item(line: &str) -> bool { + match list_item_body(line.trim_start()) { + Some(after_marker) => starts_with_closed_bold(after_marker.trim_start()), + None => false, + } +} + +fn list_item_body(line: &str) -> Option<&str> { + if let Some(rest) = line + .strip_prefix("- ") + .or_else(|| line.strip_prefix("* ")) + .or_else(|| line.strip_prefix("+ ")) + { + return Some(rest); + } + + let (digits, rest) = line.split_at(line.find(|character: char| !character.is_ascii_digit())?); + + if digits.is_empty() { + return None; + } + + rest.strip_prefix(". ") +} + +fn starts_with_closed_bold(value: &str) -> bool { + value + .strip_prefix("**") + .and_then(|rest| rest.find("**").map(|index| index > 0)) + .unwrap_or(false) + || value + .strip_prefix("__") + .and_then(|rest| rest.find("__").map(|index| index > 0)) + .unwrap_or(false) +} + +fn line_spans(text: &str) -> Vec { + let mut spans = Vec::new(); + let mut start = 0; + + for (index, character) in text.char_indices() { + if character == '\n' { + push_line_span(text, &mut spans, start, index); + start = index + character.len_utf8(); + } + } + + push_line_span(text, &mut spans, start, text.len()); + spans +} + +fn push_line_span(text: &str, spans: &mut Vec, start: usize, end: usize) { + let line = &text[start..end]; + if !line.trim().is_empty() { + let trailing = line.len() - line.trim_end().len(); + spans.push(Span(start, end - trailing)); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn detects_unordered_bold_first_bullets() { + let findings = scan_markdown("- **Security**: Environment-based configuration"); + + assert_eq!(findings.len(), 1); + assert_eq!(findings[0].rule_id, "formatting.bold_first_bullets"); + } + + #[test] + fn detects_numbered_bold_first_bullets() { + let findings = scan_markdown("1. __Performance__: Lazy loading"); + + assert_eq!(findings.len(), 1); + assert_eq!(findings[0].matched, "1. __Performance__: Lazy loading"); + } + + #[test] + fn ignores_bold_later_in_bullets() { + assert!(scan_markdown("- The **security** setting").is_empty()); + } + + #[test] + fn ignores_unclosed_bold() { + assert!(scan_markdown("- **Security: Environment").is_empty()); + } +} diff --git a/crates/core/src/detector/repetition.rs b/crates/core/src/detector/repetition.rs index 3ed887f..43e72d0 100644 --- a/crates/core/src/detector/repetition.rs +++ b/crates/core/src/detector/repetition.rs @@ -56,8 +56,7 @@ fn scan_dead_metaphor(text: &str) -> Vec { .filter(|spans| spans.len() >= 5) .map(|spans| { Finding::repetition( - "composition.dead_metaphor", - "The Dead Metaphor", + ("composition.dead_metaphor", "The Dead Metaphor"), Severity::Medium, text, super::Span(spans[0].start(), spans[spans.len() - 1].end()), @@ -86,8 +85,7 @@ fn scan_one_point_dilution(text: &str, paragraphs: &[super::Span]) -> Vec Vec { .filter(|window| window[0].1 == window[1].1 && window[1].1 == window[2].1) .map(|window| { Finding::structural( - "sentence_structure.anaphora_abuse", - "Anaphora Abuse", - Severity::Medium, + ("sentence_structure.anaphora_abuse", "Anaphora Abuse"), text, super::Span(window[0].0.start(), window[2].0.end()), ) @@ -50,9 +46,7 @@ fn scan_tricolon(text: &str, sentences: &[super::Span]) -> Vec { }) .map(|sentence| { Finding::structural( - "sentence_structure.tricolon_abuse", - "Tricolon Abuse", - Severity::Medium, + ("sentence_structure.tricolon_abuse", "Tricolon Abuse"), text, super::Span(sentence.start(), sentence.end()), ) @@ -74,9 +68,10 @@ fn scan_short_punchy_fragments(text: &str, sentences: &[super::Span]) -> Vec= 3 { findings.push(Finding::structural( - "paragraph_structure.short_punchy_fragments", - "Short Punchy Fragments", - Severity::Medium, + ( + "paragraph_structure.short_punchy_fragments", + "Short Punchy Fragments", + ), text, super::Span(run_start.unwrap(), run_end), )); @@ -89,9 +84,10 @@ fn scan_short_punchy_fragments(text: &str, sentences: &[super::Span]) -> Vec= 3 { findings.push(Finding::structural( - "paragraph_structure.short_punchy_fragments", - "Short Punchy Fragments", - Severity::Medium, + ( + "paragraph_structure.short_punchy_fragments", + "Short Punchy Fragments", + ), text, super::Span(run_start.unwrap(), run_end), )); @@ -113,9 +109,10 @@ fn scan_listicle_in_trench_coat(text: &str, paragraphs: &[super::Span]) -> Vec Vec Ve }) .map(|window| { Finding::structural( - "composition.historical_analogy_stacking", - "Historical Analogy Stacking", - Severity::Medium, + ( + "composition.historical_analogy_stacking", + "Historical Analogy Stacking", + ), text, super::Span(window[0].start(), window[2].end()), ) diff --git a/todo.md b/todo.md index de434d8..6bb432e 100644 --- a/todo.md +++ b/todo.md @@ -1,151 +1,32 @@ -# CLI / Matcher Plan +# TODO -## Shape +## Fixtures -- `meta/tropes.md` is source material only. -- Pattern dictionaries live in `crates/core/src/patterns/*.toml` so contributors - can add focused rule files without editing one giant dictionary. -- `crates/core` owns: - - loading all TOML pattern files - - validating pattern ids and phrases - - building one Aho-Corasick matcher from all phrase patterns - - scanning text and returning plain finding structs -- `crates/cli` owns: - - `clap` argument parsing - - reading a file or stdin - - display with the bundled core patterns by default - - display with `owo_colors`, respecting `NO_COLOR` - - exit code behavior - -## Pattern TOML - -Example file: `crates/core/src/patterns/word-choice.toml` - -```toml -[[patterns]] -id = "word_choice.delve" -name = "Delve and Friends" -severity = "medium" -phrases = [ - "delve into", - "delving deeper", - "certainly", - "utilize", - "leverage", - "robust", - "streamline", - "harness", -] -``` - -## Trope Coverage Checklist - -Current phrase coverage: 22 of 33 source sections. -Implemented non-Aho detectors: 10. - -- [x] Quietly and Other Magic Adverbs -- [x] Delve and Friends -- [x] Tapestry and Landscape -- [x] The Serves As Dodge -- [x] Negative Parallelism -- [x] Not X. Not Y. Just Z. -- [x] The X? A Y. -- [x] Anaphora Abuse - structural detector -- [x] Tricolon Abuse - structural detector -- [x] It's Worth Noting -- [x] Superficial Analyses -- [x] False Ranges -- [x] Short Punchy Fragments - structural detector -- [x] Listicle in a Trench Coat - structural detector -- [x] Here's the Kicker -- [x] Think of It As -- [x] Imagine a World Where -- [x] False Vulnerability -- [x] The Truth Is Simple -- [x] Grandiose Stakes Inflation -- [x] Let's Break This Down -- [x] Vague Attributions -- [x] Invented Concept Labels -- [x] Em-Dash Addiction -- [ ] Bold-First Bullets - markdown-aware detector -- [x] Unicode Decoration - character-class detector -- [x] Fractal Summaries - structural detector -- [x] The Dead Metaphor - repetition detector -- [x] Historical Analogy Stacking - structural detector -- [x] One-Point Dilution - repetition detector -- [x] Content Duplication - repetition detector -- [x] The Signposted Conclusion -- [x] Despite Its Challenges - -## Non-Aho-Corasick Rules - -Aho-Corasick is for literal phrase signals. - -Separate rule types are for when the trope depends on structure, repetition, -markdown syntax, or document-level shape. - -- Anaphora Abuse: split into sentences and flag repeated sentence starts within - a short window. -- Tricolon Abuse: detect repeated clause patterns and dense comma/semicolon - triples, not just fixed phrases. -- Short Punchy Fragments: measure runs of very short sentences or paragraph - fragments. -- Listicle in a Trench Coat: detect paragraph openings like `The first`, `The -second`, `The third` across adjacent paragraphs. -- Bold-First Bullets: parse markdown list items and flag bullets that start with - bold text. -- Unicode Decoration: scan for configured Unicode punctuation and symbols. - Actual Unicode em dashes belong here, not in phrase TOML. Keep ASCII `" -- "` - as a phrase proxy for typed em-dash style until Em-Dash Addiction gets a count - or density detector. -- Fractal Summaries: detect repeated summary/conclusion signposts at section - boundaries. -- The Dead Metaphor: count repeated uncommon nouns or configured metaphor terms - across a document. -- Historical Analogy Stacking: detect runs of named examples and comparison - verbs across adjacent sentences. -- One-Point Dilution: likely needs repetition or semantic similarity scoring. -- Content Duplication: compare normalized paragraphs or sentence shingles. - -## CLI - -Initial command: - -```text -tropius [FILE] -``` - -Behavior: - -- read stdin when `FILE` is omitted -- print each finding with pattern name, matched phrase, and byte range -- return `0` when no findings are found -- return `1` when findings are found -- return `2` for usage/configuration errors - -## Test Bed - -Use the `lectito` CLI to extract article text into fixtures when useful: +- Use `lectito` to turn selected URLs from `meta/examples.txt` into checked-in + plain text fixtures. ```text lectito --format text > meta/examples/clean/example.txt ``` -### Unit Tests +- Add clean examples that produce no findings, or only expected low-noise + findings. +- Add slop examples that produce expected pattern ids. + +## Tests -- loading multiple `crates/core/src/patterns/*.toml` files -- detecting phrases across files with one matcher -- case-insensitive matching -- duplicate pattern ids fail validation -- duplicate phrases fail validation -- empty pattern ids, names, phrase lists, and phrases fail validation -- clean examples produce no findings, or only expected low-noise findings -- slop examples produce expected pattern ids +- Add CLI integration tests for file input. +- Add CLI integration tests for stdin input. +- Add CLI integration tests for `NO_COLOR`. +- Add CLI integration tests for exit code `0` on clean input. +- Add CLI integration tests for exit code `1` on matched input. +- Add snapshot-style tests for multiline CLI report formatting. -### Integration Tests +## Detector Quality -- file input -- stdin input -- `NO_COLOR` -- exit code `0` for clean input -- exit code `1` for matched input +- Tune detector thresholds against checked-in fixtures. +- Consider grouping or suppressing overlapping findings when one text span + triggers multiple heuristic detectors. +- Consider summarizing repeated Unicode Decoration findings to reduce report noise. +- Consider turning Em-Dash Addiction into a count or density detector instead + of relying on the ASCII `" -- "` phrase proxy.