Something went wrong. Try again.
A Tour of C++ for experienced programmers, as if C++26 is the only version that ever existed.
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403#!/usr/bin/env python3"""Pre-submission lint for a single book chapter.
The lint parses the chapter with a real Markdown parser (markdown-it-py) sothat it inspects only prose, never code. It then checks the prose against themechanical rules of ASD-STE100 Simplified Technical English (Issue 9), plus aban on em/en dashes and semicolons. A chapter agent must run this and must NOTdeclare the chapter done until it exits 0.
Usage: python3 scripts/chapter_lint.py [--no-length] <chapter.md>
Exit codes: 0 lint passed 1 a hard check failed (the chapter is not done) 2 usage error (missing file, bad arguments)
--no-length skip the sentence/paragraph maximum length checks. Front matter (preface, appendices) is exempt from length checks, but it still obeys every other rule.
Hard checks fail the gate: the objective STE violations, plus the judgmentpatterns (-ing verb clauses, filler words, passive voice) when a chapterexceeds the configured count for that pattern. Below the threshold thosepatterns warn so the writer fixes them without blocking the build."""
from __future__ import annotations
import reimport sysfrom dataclasses import dataclass, field
from markdown_it import MarkdownIt
# CommonMark plus GFM tables. The book uses tables for reference material.MD = MarkdownIt("commonmark").enable("table")
# ---------------------------------------------------------------------------# Hard-fail patterns (objective, unambiguous STE violations)# ---------------------------------------------------------------------------
# Unapproved modals. STE allows only can, will, must (Rule 3.2).BANNED_MODALS = re.compile(r"\b(?:may|might|could|should|would)\b", re.IGNORECASE)
# Contractions (Rule 4.2). STE forbids all of them.CONTRACTIONS = re.compile( r"\b(?:" r"it's|that's|let's|who's|what's|there's|here's|he's|she's|we're|they're|" r"you're|don't|can't|won't|isn't|aren't|wasn't|weren't|doesn't|didn't|" r"hasn't|haven't|hadn't|wouldn't|couldn't|shouldn't|mustn't|i'm|we've|" r"they've|you've|i've|i'd|we'd|you'd|they'd|he'd|she'd|it'd|we'll|you'll|" r"they'll|he'll|she'll|it'll|i'll" r")\b", re.IGNORECASE,)
# Semicolons and em/en dashes (Rule 8.1). Also catch a space-hyphen-space,# which is how a hyphen is misused as a dash.DASH_SEMICOLON = re.compile(r"[—–;]| - ")
# Latin abbreviations (GR-6): e.g., i.e., etc.LATIN_ABBREV = re.compile(r"\b(?:e\.g\.|i\.e\.|etc\.)\b", re.IGNORECASE)
# Progressive passive (Rules 3.4, 3.5): "is being", "are being", etc.PROGRESSIVE_PASSIVE = re.compile(r"\b(?:is|are|was|were|has been|have been|had been) being\b", re.IGNORECASE)
# ---------------------------------------------------------------------------# Judgment patterns (warn only below threshold)# ---------------------------------------------------------------------------
# Present/past perfect: has/have/had + past participle. This is noisy because# "has" is also the main verb "to have" and past participles double as# adjectives ("has undefined behavior"). Warn, do not fail.PERFECT_PREFIX = re.compile(r"\b(?:has|have|had)\s+(?:been|[a-z]+(?:ed|en))\b", re.IGNORECASE)
# "-ing" clause used as a verb after a comma (Rule 3.5): ", making", ", ensuring".ING_CLAUSE = re.compile(r",\s*(?:making|allowing|enabling|ensuring|giving|causing|providing|using|turning|resulting|leading|creating|producing|letting|forcing|preventing|allowing)\b", re.IGNORECASE)
# Filler words that carry no fact (STE slop list).FILLER = re.compile( r"\b(?:simply|easily|seamlessly|robust|comprehensive|powerful|leverage|" r"utilize|utilise|in order to|prior to|due to the fact that|and/or|" r"it is worth noting|dive into|delve into|when it comes to|" r"out of the box|under the hood|blazingly fast|state-of-the-art|" r"plethora|myriad|streamline)\b", re.IGNORECASE,)
# Passive voice heuristic: be + past participle (Rule 3.6). Warn only below# threshold; passive is legal in descriptive text when the agent is unknown.PASSIVE = re.compile(r"\b(?:is|are|was|were|been)\s+[a-z]+(?:ed|en)\b", re.IGNORECASE)
# Sentence-length limit for descriptive prose (Rule 6.3). Procedural limit is# 20, but chapters are descriptive, so 25 is the gate.DESCRIPTIVE_LIMIT = 25
# A chapter fails when it exceeds these counts for a judgment pattern. Below# the threshold the same pattern warns. The passive heuristic is noisy (it# counts legal forms like "is undefined" and "is destroyed"), so its bar is# high; -ing clauses and filler words are more clearly violations.ING_CLAUSE_LIMIT = 2FILLER_LIMIT = 1PASSIVE_LIMIT = 15
# A paragraph over this many words is exceptionally long (Rule 6.5: one topic# per paragraph). Median paragraph is 38 words, p90 is 75; 120 catches the# worst offenders.PARAGRAPH_LIMIT = 120
# A bullet list is flagged when at least this many consecutive items start# with bold text followed by a colon or a dash (an inline glossary/definition# list that STE would render as a table or prose).BOLD_LIST_MIN = 3
# ---------------------------------------------------------------------------# Structural checks (fail on violation)# ---------------------------------------------------------------------------
APPARATUS_HEADING = re.compile( r"^(?:summary of key points|key points|summary|references|" r"core-?guideline citations|end of chapter|further reading)$", re.IGNORECASE,)
@dataclassclass Result: """Accumulates lint findings."""
hard_failures: list[str] = field(default_factory=list) warnings: list[str] = field(default_factory=list)
@dataclassclass Parsed: """The prose and structure extracted from a chapter by the Markdown parser.
prose: list of (kind, text) for every inline token with code removed. kind is one of "paragraph", "heading", "table", "list", "other". paragraphs: text of paragraph body tokens only (for sentence length). fences: list of (info, content) for every fenced code block. headings: text of every heading. list_items: text of every list item (bold markers included). sections: list of (title, paragraph_count, code_count) per heading block. """
prose: list[tuple[str, str]] paragraphs: list[str] fences: list[tuple[str, str]] headings: list[str] list_items: list[str] sections: list[tuple[str, int, int]]
def analyze(text: str) -> Parsed: """Parse the chapter and return its prose and structure.
Only the text children of inline tokens are kept, so emphasis markers, inline code, and fenced code never reach the STE checks. """ tokens = MD.parse(text) prose: list[tuple[str, str]] = [] paragraphs: list[str] = [] fences: list[tuple[str, str]] = [] headings: list[str] = [] list_items: list[str] = [] sections: list[tuple[str, int, int]] = []
in_paragraph = False in_heading = False in_list_item = False # Current section: (title, paragraph_count, code_count). None outside a # heading block. section: list = None for tok in tokens: if tok.type == "paragraph_open": in_paragraph = True elif tok.type == "paragraph_close": in_paragraph = False elif tok.type == "heading_open": in_heading = True # A heading starts a new section; close the previous one. if section is not None: sections.append(tuple(section)) section = ["", 0, 0] elif tok.type == "heading_close": in_heading = False elif tok.type == "list_item_open": in_list_item = True elif tok.type == "list_item_close": in_list_item = False elif tok.type == "fence": fences.append((tok.info, tok.content)) if section is not None: section[2] += 1 elif tok.type == "inline": # Only literal text children: emphasis/strong are markers, inline # code is code. Both are excluded. chunk = "".join(c.content for c in (tok.children or []) if c.type == "text") if in_heading: headings.append(chunk) if section is not None: section[0] += chunk if in_list_item: list_items.append(chunk) if not chunk.strip(): continue if in_paragraph: kind = "paragraph" elif in_heading: kind = "heading" elif tok.tag in ("td", "th"): kind = "table" elif tok.tag == "li": kind = "list" else: kind = "other" prose.append((kind, chunk)) if in_paragraph: paragraphs.append(chunk) if section is not None: section[1] += 1 if section is not None: sections.append(tuple(section)) return Parsed(prose, paragraphs, fences, headings, list_items, sections)
def split_sentences(text: str) -> list[tuple[int, str]]: """Split a paragraph into (word_count, sentence) pairs.
Dotted identifiers and decimals are protected so they do not split a sentence. A hyphenated compound counts as one word (Rule 8.7). """ protected = re.sub(r"\d\.\d", "N", text) protected = re.sub(r"\b[A-Za-z]\.", "A", protected) out: list[tuple[int, str]] = [] for part in re.split(r"[.!?](?:\s+|$)", protected): part = part.strip() if not part: continue tokens = part.split() hyphens = sum(1 for t in tokens if re.search(r"[-–]$", t)) out.append((len(tokens) - hyphens, part)) return out
def _snippet(text: str, start: int, end: int) -> str: """Return a short context window around a match.""" lo = max(0, start - 30) hi = min(len(text), end + 30) return text[lo:hi]
def check_hard(prose: list[tuple[str, str]], res: Result) -> None: """Run the objective checks that fail the gate.""" checks = [ ("banned modal (may/might/could/should/would)", BANNED_MODALS), ("contraction", CONTRACTIONS), ("em/en dash or semicolon", DASH_SEMICOLON), ("Latin abbreviation (e.g./i.e./etc.)", LATIN_ABBREV), ("progressive passive (is/are/was being)", PROGRESSIVE_PASSIVE), ] for _, chunk in prose: for label, pattern in checks: for m in pattern.finditer(chunk): res.hard_failures.append(f"{label}: {_snippet(chunk, m.start(), m.end())!r}")
def check_judgment(prose: list[tuple[str, str]], paragraphs: list[str], res: Result) -> None: """Run the judgment checks. Each pattern warns; three also hard-fail when the chapter exceeds the configured count for that pattern.""" # Sentence length (Rule 6.3), on paragraph body text only. for para in paragraphs: for count, sentence in split_sentences(para): if count > DESCRIPTIVE_LIMIT: res.warnings.append(f"sentence {count} words (> {DESCRIPTIVE_LIMIT}): {sentence!r}")
# Paragraph length (Rule 6.5: one topic per paragraph). for para in paragraphs: count = len(para.split()) if count > PARAGRAPH_LIMIT: res.warnings.append(f"paragraph {count} words (> {PARAGRAPH_LIMIT}): {para[:80]!r}")
# Patterns that warn always and hard-fail above a count threshold. thresholded = [ ("'-ing' verb clause after a comma", ING_CLAUSE, ING_CLAUSE_LIMIT), ("filler word", FILLER, FILLER_LIMIT), ("passive voice (be + past participle)", PASSIVE, PASSIVE_LIMIT), ] for label, pattern, limit in thresholded: matches = [m for _, chunk in prose for m in pattern.finditer(chunk)] for m in matches: # Recover the chunk that produced this match for context. for _, chunk in prose: if m.string is chunk: res.warnings.append(f"{label}: {_snippet(chunk, m.start(), m.end())!r}") break if len(matches) > limit: res.hard_failures.append( f"{label}: {len(matches)} occurrences exceed the limit of {limit}" )
# Patterns that warn only (no threshold). for label, pattern in [ ("present/past perfect (has/have/had + participle)", PERFECT_PREFIX), ]: for _, chunk in prose: for m in pattern.finditer(chunk): res.warnings.append(f"{label}: {_snippet(chunk, m.start(), m.end())!r}")
def check_structure(parsed: Parsed, text: str, res: Result) -> None: """Run the structural checks on the parsed chapter.""" # 1. Balanced code fences (raw-text parity check). fences = len(re.findall(r"^```", text, flags=re.MULTILINE)) if fences % 2 != 0: res.hard_failures.append(f"unbalanced code fences ({fences})")
# 2. No appended apparatus (standalone headings only). for heading in parsed.headings: if APPARATUS_HEADING.match(heading.strip()): res.hard_failures.append(f"apparatus heading present: {heading!r}") break
# 3. Every {{#include}} must sit inside a ```cpp fence. for info, content in parsed.fences: if "{{#include" in content and info != "cpp": res.hard_failures.append("an {{#include}} is in a non-cpp fence") for _, chunk in parsed.prose: if "{{#include" in chunk: res.hard_failures.append("an {{#include}} is outside a ```cpp fence") break
# 4. Bolted-led bullet list (inline glossary): three or more consecutive # items that start with bold text followed by a colon or a dash. bold_lead = re.compile(r"^\*\*.+?\*\*\s*(?:[:—–]| - )", re.DOTALL) run = 0 for item in parsed.list_items: if bold_lead.match(item.strip()): run += 1 if run >= BOLD_LIST_MIN: res.hard_failures.append( f"bold-led bullet list: {run} consecutive items start with bold + colon/dash" ) break else: run = 0
# 5. Two or more one-paragraph sections in a row. A code block in a # section counts as a second paragraph. run = 0 for title, paras, code in parsed.sections: if paras <= 1 and code == 0: run += 1 if run >= 2: res.hard_failures.append( "two or more one-paragraph sections in a row (no code block)" ) break else: run = 0
def main(argv: list[str]) -> int: no_length = "--no-length" in argv argv = [a for a in argv if a != "--no-length"] if len(argv) < 2: print(f"usage: {argv[0]} <chapter.md>", file=sys.stderr) return 2 path = argv[1]
try: with open(path, encoding="utf-8") as fh: text = fh.read() except OSError as exc: print(f"FAIL: cannot read {path}: {exc}", file=sys.stderr) return 2
res = Result() parsed = analyze(text) check_structure(parsed, text, res) check_hard(parsed.prose, res) if not no_length: check_judgment(parsed.prose, parsed.paragraphs, res)
words = len(text.split())
# Report. for w in res.warnings: print(f"WARN: {w}") if res.hard_failures: print("FAIL:") for f in res.hard_failures: print(f" - {f}") return 1
print(f"OK: {words} words, lint passed ({len(res.warnings)} warnings)") return 0
if __name__ == "__main__": sys.exit(main(sys.argv))