//! HTML fragment -> markdown renderer for the zig language reference. //! //! Scope: this renders a *known-clean* fragment (one langref section, sliced by //! id anchor). It is NOT a general web extractor — there is no boilerplate to //! strip inside a section, so it just preserves structure: headings, paragraphs, //! lists, tables, code, and inline code. `render` is a clean seam, so a real //! content extractor could front it later if zigman ever read arbitrary pages. const std = @import("std"); const Writer = std.Io.Writer; /// Render an HTML section fragment to markdown. Caller owns the returned slice. pub fn render(allocator: std.mem.Allocator, html: []const u8) ![]u8 { var out: Writer.Allocating = .init(allocator); defer out.deinit(); var e: Emitter = .{ .w = &out.writer }; var i: usize = 0; while (i < html.len) { if (html[i] == '<') { const tag = parseTag(html[i..]) orelse { try e.text("<"); i += 1; continue; }; if (!tag.close and eql(tag.name, "a") and contains(tag.raw, "class=\"hdr\"")) { i += skipUntilClose(html[i..], "a"); continue; } if (!tag.close and eql(tag.name, "figcaption")) { i += skipUntilClose(html[i..], "figcaption"); continue; } if (!tag.close and eql(tag.name, "table")) { const span = skipUntilClose(html[i..], "table"); e.newlines(2); try e.flushWs(); try renderTable(allocator, e.w, html[i .. i + span]); e.wrote = true; e.newlines(2); i += span; continue; } try applyTag(&e, tag); i += tag.len; continue; } const end = std.mem.indexOfScalarPos(u8, html, i, '<') orelse html.len; try e.text(html[i..end]); i = end; } return allocator.dupe(u8, std.mem.trim(u8, out.written(), " \n")); } /// Coalesces whitespace so block tags can freely request blank lines / spaces /// without producing runs of empty lines or leading indentation. Inside a /// `
`, bytes pass through verbatim.
const Emitter = struct {
    w: *Writer,
    pending_nl: u8 = 0, // requested newlines, capped at 2, flushed before content
    pending_sp: bool = false,
    wrote: bool = false,
    in_pre: bool = false,

    fn newlines(e: *Emitter, n: u8) void {
        if (n > e.pending_nl) e.pending_nl = @min(n, 2);
        e.pending_sp = false;
    }
    fn space(e: *Emitter) void {
        e.pending_sp = true;
    }

    fn flushWs(e: *Emitter) !void {
        if (!e.wrote) {
            e.pending_nl = 0;
            e.pending_sp = false;
            return;
        }
        if (e.pending_nl > 0) {
            try e.w.splatByteAll('\n', e.pending_nl);
        } else if (e.pending_sp) {
            try e.w.writeByte(' ');
        }
        e.pending_nl = 0;
        e.pending_sp = false;
    }

    /// raw literal content (already markdown), flushing pending whitespace first.
    fn raw(e: *Emitter, s: []const u8) !void {
        if (s.len == 0) return;
        try e.flushWs();
        try e.w.writeAll(s);
        e.wrote = true;
    }

    /// HTML text run: decode entities; outside 
, collapse whitespace and
    /// drop the § pilcrow; inside 
, preserve everything verbatim.
    fn text(e: *Emitter, run: []const u8) !void {
        if (e.in_pre) {
            // emit pending block boundary (the fence opener), then raw code
            try e.flushWs();
            try decodeEntities(e.w, run);
            if (run.len > 0) e.wrote = true;
            return;
        }
        var k: usize = 0;
        while (k < run.len) {
            const c = run[k];
            if (c == ' ' or c == '\n' or c == '\t' or c == '\r') {
                e.space();
                k += 1;
                continue;
            }
            if (c == 0xC2 and k + 1 < run.len and run[k + 1] == 0xA7) { // §
                k += 2;
                continue;
            }
            try e.flushWs();
            if (c == '&') {
                k += try writeEntity(e.w, run[k..]);
            } else {
                try e.w.writeByte(c);
                k += 1;
            }
            e.wrote = true;
        }
    }
};

fn applyTag(e: *Emitter, tag: Tag) !void {
    const n = tag.name;
    if (headingLevel(n)) |lvl| {
        if (!tag.close) {
            e.newlines(2);
            try e.flushWs();
            try e.w.splatByteAll('#', lvl);
            try e.w.writeByte(' ');
            e.wrote = true;
        } else e.newlines(2);
    } else if (eql(n, "p") or eql(n, "ul") or eql(n, "ol") or eql(n, "blockquote")) {
        e.newlines(2);
    } else if (eql(n, "li")) {
        if (!tag.close) {
            e.newlines(1);
            try e.raw("- ");
        }
    } else if (eql(n, "pre")) {
        if (!tag.close) {
            e.newlines(2);
            try e.raw("```zig\n");
            e.in_pre = true;
        } else {
            e.in_pre = false;
            try e.raw("\n```");
            e.newlines(2);
        }
    } else if (eql(n, "code") and !e.in_pre) {
        try e.raw("`");
    } else if (eql(n, "br")) {
        e.newlines(1);
    }
    // span, a, em, i, b, figure, cite, … : structurally dropped
}

const Tag = struct {
    name: []const u8,
    close: bool,
    len: usize,
    raw: []const u8,
};

fn parseTag(s: []const u8) ?Tag {
    if (s.len < 2 or s[0] != '<') return null;
    const gt = std.mem.indexOfScalar(u8, s, '>') orelse return null;
    var j: usize = 1;
    const close = s[j] == '/';
    if (close) j += 1;
    const start = j;
    while (j < gt and isNameChar(s[j])) j += 1;
    return .{ .name = s[start..j], .close = close, .len = gt + 1, .raw = s[0 .. gt + 1] };
}

fn decodeEntities(w: *Writer, raw: []const u8) !void {
    var k: usize = 0;
    while (k < raw.len) {
        if (raw[k] == '&') {
            k += try writeEntity(w, raw[k..]);
        } else {
            try w.writeByte(raw[k]);
            k += 1;
        }
    }
}

fn writeEntity(w: *Writer, s: []const u8) !usize {
    const map = .{
        .{ """, "\"" }, .{ "&", "&" }, .{ "<", "<" },
        .{ ">", ">" },    .{ "'", "'" }, .{ "'", "'" },
        .{ " ", " " },
    };
    inline for (map) |m| {
        if (std.mem.startsWith(u8, s, m[0])) {
            try w.writeAll(m[1]);
            return m[0].len;
        }
    }
    try w.writeByte('&');
    return 1;
}

/// Render a `…
` region as a markdown table. The first row is /// treated as the header (langref tables lead with ``). fn renderTable(allocator: std.mem.Allocator, w: *Writer, table: []const u8) !void { var first = true; var ncols: usize = 0; var ri: usize = 0; while (std.mem.indexOfPos(u8, table, ri, ""; drop it const tail = std.mem.lastIndexOf(u8, s, "= '1' and n[1] <= '6') return @intCast(n[1] - '0'); return null; } fn isNameChar(c: u8) bool { return (c >= 'a' and c <= 'z') or (c >= 'A' and c <= 'Z') or (c >= '0' and c <= '9'); } fn eql(a: []const u8, b: []const u8) bool { return std.ascii.eqlIgnoreCase(a, b); } fn contains(h: []const u8, n: []const u8) bool { return std.mem.indexOf(u8, h, n) != null; }