//! HTML fragment -> markdown renderer for the zig language reference. //! //! Scope: this renders a *known-clean* fragment (one langref section, sliced by //! id anchor). It is NOT a general web extractor — there is no boilerplate to //! strip inside a section, so it just preserves structure: headings, paragraphs, //! lists, tables, code, and inline code. `render` is a clean seam, so a real //! content extractor could front it later if zigman ever read arbitrary pages. const std = @import("std"); const Writer = std.Io.Writer; /// Render an HTML section fragment to markdown. Caller owns the returned slice. pub fn render(allocator: std.mem.Allocator, html: []const u8) ![]u8 { var out: Writer.Allocating = .init(allocator); defer out.deinit(); var e: Emitter = .{ .w = &out.writer }; var i: usize = 0; while (i < html.len) { if (html[i] == '<') { const tag = parseTag(html[i..]) orelse { try e.text("<"); i += 1; continue; }; if (!tag.close and eql(tag.name, "a") and contains(tag.raw, "class=\"hdr\"")) { i += skipUntilClose(html[i..], "a"); continue; } if (!tag.close and eql(tag.name, "figcaption")) { i += skipUntilClose(html[i..], "figcaption"); continue; } if (!tag.close and eql(tag.name, "table")) { const span = skipUntilClose(html[i..], "table"); e.newlines(2); try e.flushWs(); try renderTable(allocator, e.w, html[i .. i + span]); e.wrote = true; e.newlines(2); i += span; continue; } try applyTag(&e, tag); i += tag.len; continue; } const end = std.mem.indexOfScalarPos(u8, html, i, '<') orelse html.len; try e.text(html[i..end]); i = end; } return allocator.dupe(u8, std.mem.trim(u8, out.written(), " \n")); } /// Coalesces whitespace so block tags can freely request blank lines / spaces /// without producing runs of empty lines or leading indentation. Inside a /// `
`, bytes pass through verbatim.
const Emitter = struct {
w: *Writer,
pending_nl: u8 = 0, // requested newlines, capped at 2, flushed before content
pending_sp: bool = false,
wrote: bool = false,
in_pre: bool = false,
fn newlines(e: *Emitter, n: u8) void {
if (n > e.pending_nl) e.pending_nl = @min(n, 2);
e.pending_sp = false;
}
fn space(e: *Emitter) void {
e.pending_sp = true;
}
fn flushWs(e: *Emitter) !void {
if (!e.wrote) {
e.pending_nl = 0;
e.pending_sp = false;
return;
}
if (e.pending_nl > 0) {
try e.w.splatByteAll('\n', e.pending_nl);
} else if (e.pending_sp) {
try e.w.writeByte(' ');
}
e.pending_nl = 0;
e.pending_sp = false;
}
/// raw literal content (already markdown), flushing pending whitespace first.
fn raw(e: *Emitter, s: []const u8) !void {
if (s.len == 0) return;
try e.flushWs();
try e.w.writeAll(s);
e.wrote = true;
}
/// HTML text run: decode entities; outside , collapse whitespace and
/// drop the § pilcrow; inside , preserve everything verbatim.
fn text(e: *Emitter, run: []const u8) !void {
if (e.in_pre) {
// emit pending block boundary (the fence opener), then raw code
try e.flushWs();
try decodeEntities(e.w, run);
if (run.len > 0) e.wrote = true;
return;
}
var k: usize = 0;
while (k < run.len) {
const c = run[k];
if (c == ' ' or c == '\n' or c == '\t' or c == '\r') {
e.space();
k += 1;
continue;
}
if (c == 0xC2 and k + 1 < run.len and run[k + 1] == 0xA7) { // §
k += 2;
continue;
}
try e.flushWs();
if (c == '&') {
k += try writeEntity(e.w, run[k..]);
} else {
try e.w.writeByte(c);
k += 1;
}
e.wrote = true;
}
}
};
fn applyTag(e: *Emitter, tag: Tag) !void {
const n = tag.name;
if (headingLevel(n)) |lvl| {
if (!tag.close) {
e.newlines(2);
try e.flushWs();
try e.w.splatByteAll('#', lvl);
try e.w.writeByte(' ');
e.wrote = true;
} else e.newlines(2);
} else if (eql(n, "p") or eql(n, "ul") or eql(n, "ol") or eql(n, "blockquote")) {
e.newlines(2);
} else if (eql(n, "li")) {
if (!tag.close) {
e.newlines(1);
try e.raw("- ");
}
} else if (eql(n, "pre")) {
if (!tag.close) {
e.newlines(2);
try e.raw("```zig\n");
e.in_pre = true;
} else {
e.in_pre = false;
try e.raw("\n```");
e.newlines(2);
}
} else if (eql(n, "code") and !e.in_pre) {
try e.raw("`");
} else if (eql(n, "br")) {
e.newlines(1);
}
// span, a, em, i, b, figure, cite, … : structurally dropped
}
const Tag = struct {
name: []const u8,
close: bool,
len: usize,
raw: []const u8,
};
fn parseTag(s: []const u8) ?Tag {
if (s.len < 2 or s[0] != '<') return null;
const gt = std.mem.indexOfScalar(u8, s, '>') orelse return null;
var j: usize = 1;
const close = s[j] == '/';
if (close) j += 1;
const start = j;
while (j < gt and isNameChar(s[j])) j += 1;
return .{ .name = s[start..j], .close = close, .len = gt + 1, .raw = s[0 .. gt + 1] };
}
fn decodeEntities(w: *Writer, raw: []const u8) !void {
var k: usize = 0;
while (k < raw.len) {
if (raw[k] == '&') {
k += try writeEntity(w, raw[k..]);
} else {
try w.writeByte(raw[k]);
k += 1;
}
}
}
fn writeEntity(w: *Writer, s: []const u8) !usize {
const map = .{
.{ """, "\"" }, .{ "&", "&" }, .{ "<", "<" },
.{ ">", ">" }, .{ "'", "'" }, .{ "'", "'" },
.{ " ", " " },
};
inline for (map) |m| {
if (std.mem.startsWith(u8, s, m[0])) {
try w.writeAll(m[1]);
return m[0].len;
}
}
try w.writeByte('&');
return 1;
}
/// Render a `