diff --git a/scripts/chapter_lint.py b/scripts/chapter_lint.py index d5e849c..1e2f719 100755 --- a/scripts/chapter_lint.py +++ b/scripts/chapter_lint.py @@ -98,6 +98,16 @@ ING_CLAUSE_LIMIT = 2 FILLER_LIMIT = 1 PASSIVE_LIMIT = 15 +# A paragraph over this many words is exceptionally long (Rule 6.5: one topic +# per paragraph). Median paragraph is 38 words, p90 is 75; 120 catches the +# worst offenders. +PARAGRAPH_LIMIT = 120 + +# A bullet list is flagged when at least this many consecutive items start +# with bold text followed by a colon or a dash (an inline glossary/definition +# list that STE would render as a table or prose). +BOLD_LIST_MIN = 3 + # --------------------------------------------------------------------------- # Structural checks (fail on violation) # --------------------------------------------------------------------------- @@ -126,12 +136,16 @@ class Parsed: paragraphs: text of paragraph body tokens only (for sentence length). fences: list of (info, content) for every fenced code block. headings: text of every heading. + list_items: text of every list item (bold markers included). + sections: list of (title, paragraph_count, code_count) per heading block. """ prose: list[tuple[str, str]] paragraphs: list[str] fences: list[tuple[str, str]] headings: list[str] + list_items: list[str] + sections: list[tuple[str, int, int]] def analyze(text: str) -> Parsed: @@ -145,9 +159,15 @@ def analyze(text: str) -> Parsed: paragraphs: list[str] = [] fences: list[tuple[str, str]] = [] headings: list[str] = [] + list_items: list[str] = [] + sections: list[tuple[str, int, int]] = [] in_paragraph = False in_heading = False + in_list_item = False + # Current section: (title, paragraph_count, code_count). None outside a + # heading block. + section: list = None for tok in tokens: if tok.type == "paragraph_open": in_paragraph = True @@ -155,14 +175,30 @@ def analyze(text: str) -> Parsed: in_paragraph = False elif tok.type == "heading_open": in_heading = True + # A heading starts a new section; close the previous one. + if section is not None: + sections.append(tuple(section)) + section = ["", 0, 0] elif tok.type == "heading_close": in_heading = False + elif tok.type == "list_item_open": + in_list_item = True + elif tok.type == "list_item_close": + in_list_item = False elif tok.type == "fence": fences.append((tok.info, tok.content)) + if section is not None: + section[2] += 1 elif tok.type == "inline": # Only literal text children: emphasis/strong are markers, inline # code is code. Both are excluded. chunk = "".join(c.content for c in (tok.children or []) if c.type == "text") + if in_heading: + headings.append(chunk) + if section is not None: + section[0] += chunk + if in_list_item: + list_items.append(chunk) if not chunk.strip(): continue if in_paragraph: @@ -178,9 +214,11 @@ def analyze(text: str) -> Parsed: prose.append((kind, chunk)) if in_paragraph: paragraphs.append(chunk) - if in_heading: - headings.append(chunk) - return Parsed(prose, paragraphs, fences, headings) + if section is not None: + section[1] += 1 + if section is not None: + sections.append(tuple(section)) + return Parsed(prose, paragraphs, fences, headings, list_items, sections) def split_sentences(text: str) -> list[tuple[int, str]]: @@ -233,6 +271,12 @@ def check_judgment(prose: list[tuple[str, str]], paragraphs: list[str], res: Res if count > DESCRIPTIVE_LIMIT: res.warnings.append(f"sentence {count} words (> {DESCRIPTIVE_LIMIT}): {sentence!r}") + # Paragraph length (Rule 6.5: one topic per paragraph). + for para in paragraphs: + count = len(para.split()) + if count > PARAGRAPH_LIMIT: + res.warnings.append(f"paragraph {count} words (> {PARAGRAPH_LIMIT}): {para[:80]!r}") + # Patterns that warn always and hard-fail above a count threshold. thresholded = [ ("'-ing' verb clause after a comma", ING_CLAUSE, ING_CLAUSE_LIMIT), @@ -283,6 +327,35 @@ def check_structure(parsed: Parsed, text: str, res: Result) -> None: res.hard_failures.append("an {{#include}} is outside a ```cpp fence") break + # 4. Bolted-led bullet list (inline glossary): three or more consecutive + # items that start with bold text followed by a colon or a dash. + bold_lead = re.compile(r"^\*\*.+?\*\*\s*(?:[:—–]| - )", re.DOTALL) + run = 0 + for item in parsed.list_items: + if bold_lead.match(item.strip()): + run += 1 + if run >= BOLD_LIST_MIN: + res.hard_failures.append( + f"bold-led bullet list: {run} consecutive items start with bold + colon/dash" + ) + break + else: + run = 0 + + # 5. Two or more one-paragraph sections in a row. A code block in a + # section counts as a second paragraph. + run = 0 + for title, paras, code in parsed.sections: + if paras <= 1 and code == 0: + run += 1 + if run >= 2: + res.hard_failures.append( + "two or more one-paragraph sections in a row (no code block)" + ) + break + else: + run = 0 + def main(argv: list[str]) -> int: if len(argv) < 2: