Something went wrong. Try again.
zig langref cli nate.tngl.io/zigman
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331//! HTML fragment -> markdown renderer for the zig language reference.//!//! Scope: this renders a *known-clean* fragment (one langref section, sliced by//! id anchor). It is NOT a general web extractor — there is no boilerplate to//! strip inside a section, so it just preserves structure: headings, paragraphs,//! lists, tables, code, and inline code. `render` is a clean seam, so a real//! content extractor could front it later if zigman ever read arbitrary pages.
const std = @import("std");const Writer = std.Io.Writer;
/// Render an HTML section fragment to markdown. Caller owns the returned slice.pub fn render(allocator: std.mem.Allocator, html: []const u8) ![]u8 { var out: Writer.Allocating = .init(allocator); defer out.deinit(); var e: Emitter = .{ .w = &out.writer };
var i: usize = 0; while (i < html.len) { if (html[i] == '<') { const tag = parseTag(html[i..]) orelse { try e.text("<"); i += 1; continue; }; if (!tag.close and eql(tag.name, "a") and contains(tag.raw, "class=\"hdr\"")) { i += skipUntilClose(html[i..], "a"); continue; } if (!tag.close and eql(tag.name, "figcaption")) { i += skipUntilClose(html[i..], "figcaption"); continue; } if (!tag.close and eql(tag.name, "table")) { const span = skipUntilClose(html[i..], "table"); e.newlines(2); try e.flushWs(); try renderTable(allocator, e.w, html[i .. i + span]); e.wrote = true; e.newlines(2); i += span; continue; } try applyTag(&e, tag); i += tag.len; continue; } const end = std.mem.indexOfScalarPos(u8, html, i, '<') orelse html.len; try e.text(html[i..end]); i = end; }
return allocator.dupe(u8, std.mem.trim(u8, out.written(), " \n"));}
/// Coalesces whitespace so block tags can freely request blank lines / spaces/// without producing runs of empty lines or leading indentation. Inside a/// `<pre>`, bytes pass through verbatim.const Emitter = struct { w: *Writer, pending_nl: u8 = 0, // requested newlines, capped at 2, flushed before content pending_sp: bool = false, wrote: bool = false, in_pre: bool = false,
fn newlines(e: *Emitter, n: u8) void { if (n > e.pending_nl) e.pending_nl = @min(n, 2); e.pending_sp = false; } fn space(e: *Emitter) void { e.pending_sp = true; }
fn flushWs(e: *Emitter) !void { if (!e.wrote) { e.pending_nl = 0; e.pending_sp = false; return; } if (e.pending_nl > 0) { try e.w.splatByteAll('\n', e.pending_nl); } else if (e.pending_sp) { try e.w.writeByte(' '); } e.pending_nl = 0; e.pending_sp = false; }
/// raw literal content (already markdown), flushing pending whitespace first. fn raw(e: *Emitter, s: []const u8) !void { if (s.len == 0) return; try e.flushWs(); try e.w.writeAll(s); e.wrote = true; }
/// HTML text run: decode entities; outside <pre>, collapse whitespace and /// drop the § pilcrow; inside <pre>, preserve everything verbatim. fn text(e: *Emitter, run: []const u8) !void { if (e.in_pre) { // emit pending block boundary (the fence opener), then raw code try e.flushWs(); try decodeEntities(e.w, run); if (run.len > 0) e.wrote = true; return; } var k: usize = 0; while (k < run.len) { const c = run[k]; if (c == ' ' or c == '\n' or c == '\t' or c == '\r') { e.space(); k += 1; continue; } if (c == 0xC2 and k + 1 < run.len and run[k + 1] == 0xA7) { // § k += 2; continue; } try e.flushWs(); if (c == '&') { k += try writeEntity(e.w, run[k..]); } else { try e.w.writeByte(c); k += 1; } e.wrote = true; } }};
fn applyTag(e: *Emitter, tag: Tag) !void { const n = tag.name; if (headingLevel(n)) |lvl| { if (!tag.close) { e.newlines(2); try e.flushWs(); try e.w.splatByteAll('#', lvl); try e.w.writeByte(' '); e.wrote = true; } else e.newlines(2); } else if (eql(n, "p") or eql(n, "ul") or eql(n, "ol") or eql(n, "blockquote")) { e.newlines(2); } else if (eql(n, "li")) { if (!tag.close) { e.newlines(1); try e.raw("- "); } } else if (eql(n, "pre")) { if (!tag.close) { e.newlines(2); try e.raw("```zig\n"); e.in_pre = true; } else { e.in_pre = false; try e.raw("\n```"); e.newlines(2); } } else if (eql(n, "code") and !e.in_pre) { try e.raw("`"); } else if (eql(n, "br")) { e.newlines(1); } // span, a, em, i, b, figure, cite, … : structurally dropped}
const Tag = struct { name: []const u8, close: bool, len: usize, raw: []const u8,};
fn parseTag(s: []const u8) ?Tag { if (s.len < 2 or s[0] != '<') return null; const gt = std.mem.indexOfScalar(u8, s, '>') orelse return null; var j: usize = 1; const close = s[j] == '/'; if (close) j += 1; const start = j; while (j < gt and isNameChar(s[j])) j += 1; return .{ .name = s[start..j], .close = close, .len = gt + 1, .raw = s[0 .. gt + 1] };}
fn decodeEntities(w: *Writer, raw: []const u8) !void { var k: usize = 0; while (k < raw.len) { if (raw[k] == '&') { k += try writeEntity(w, raw[k..]); } else { try w.writeByte(raw[k]); k += 1; } }}
fn writeEntity(w: *Writer, s: []const u8) !usize { const map = .{ .{ """, "\"" }, .{ "&", "&" }, .{ "<", "<" }, .{ ">", ">" }, .{ "'", "'" }, .{ "'", "'" }, .{ " ", " " }, }; inline for (map) |m| { if (std.mem.startsWith(u8, s, m[0])) { try w.writeAll(m[1]); return m[0].len; } } try w.writeByte('&'); return 1;}
/// Render a `<table>…</table>` region as a markdown table. The first row is/// treated as the header (langref tables lead with `<th>`).fn renderTable(allocator: std.mem.Allocator, w: *Writer, table: []const u8) !void { var first = true; var ncols: usize = 0; var ri: usize = 0; while (std.mem.indexOfPos(u8, table, ri, "<tr")) |tr| { const tr_end = tr + skipUntilClose(table[tr..], "tr"); const row = table[tr..tr_end];
var cells: usize = 0; try w.writeAll("|"); var ci: usize = 0; while (nextCell(row, ci)) |cell| { const md = try renderCell(allocator, cell.inner); defer allocator.free(md); try w.print(" {s} |", .{md}); cells += 1; ci = cell.next; } try w.writeByte('\n'); if (first) { ncols = cells; try w.writeAll("|"); for (0..ncols) |_| try w.writeAll(" --- |"); try w.writeByte('\n'); first = false; } ri = tr_end; }}
const Cell = struct { inner: []const u8, next: usize };
fn nextCell(row: []const u8, from: usize) ?Cell { var i = from; while (std.mem.indexOfScalarPos(u8, row, i, '<')) |lt| { const tag = parseTag(row[lt..]) orelse { i = lt + 1; continue; }; if (!tag.close and (eql(tag.name, "td") or eql(tag.name, "th"))) { const inner_start = lt + tag.len; const close = skipUntilClose(row[inner_start..], tag.name); // close includes the closing tag; trim it back to inner content const inner = trimClose(row[inner_start .. inner_start + close], tag.name); return .{ .inner = inner, .next = inner_start + close }; } i = lt + tag.len; } return null;}
fn trimClose(s: []const u8, name: []const u8) []const u8 { // s ends with "</name>"; drop it const tail = std.mem.lastIndexOf(u8, s, "</") orelse return s; _ = name; return s[0..tail];}
/// Render one table cell's inner HTML to a single inline string (no newlines,/// `|` escaped). Reuses the Emitter for code/entity/whitespace handling.fn renderCell(allocator: std.mem.Allocator, inner: []const u8) ![]u8 { var tmp: Writer.Allocating = .init(allocator); defer tmp.deinit(); var e: Emitter = .{ .w = &tmp.writer }; var i: usize = 0; while (i < inner.len) { if (inner[i] == '<') { const tag = parseTag(inner[i..]) orelse { try e.text("<"); i += 1; continue; }; if (eql(tag.name, "code")) try e.raw("`"); i += tag.len; continue; } const end = std.mem.indexOfScalarPos(u8, inner, i, '<') orelse inner.len; try e.text(inner[i..end]); i = end; } const flat = std.mem.trim(u8, tmp.written(), " \n"); // escape pipes so cells don't break the table var buf: std.ArrayList(u8) = .empty; errdefer buf.deinit(allocator); for (flat) |c| { if (c == '|') try buf.append(allocator, '\\'); if (c == '\n') { try buf.append(allocator, ' '); } else try buf.append(allocator, c); } return buf.toOwnedSlice(allocator);}
fn skipUntilClose(s: []const u8, name: []const u8) usize { var i: usize = 0; while (std.mem.indexOfScalarPos(u8, s, i, '<')) |lt| { if (parseTag(s[lt..])) |t| { if (t.close and eql(t.name, name)) return lt + t.len; i = lt + t.len; } else i = lt + 1; } return s.len;}
fn headingLevel(n: []const u8) ?u3 { if (n.len == 2 and n[0] == 'h' and n[1] >= '1' and n[1] <= '6') return @intCast(n[1] - '0'); return null;}fn isNameChar(c: u8) bool { return (c >= 'a' and c <= 'z') or (c >= 'A' and c <= 'Z') or (c >= '0' and c <= '9');}fn eql(a: []const u8, b: []const u8) bool { return std.ascii.eqlIgnoreCase(a, b);}fn contains(h: []const u8, n: []const u8) bool { return std.mem.indexOf(u8, h, n) != null;}