Something went wrong. Try again.
zig langref cli nate.tngl.io/zigman
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315//! zigman library: parse the zig language reference (a single static HTML page//! with stable per-section `id` anchors) into sections and clean markdown.//!//! The CLI in `main.zig` is a thin shell over this; everything here is pure//! (no I/O), so it is all unit-testable.
const std = @import("std");
pub const render = @import("render.zig").render;
/// A heading in the reference's table of contents.pub const Section = struct { level: u8, // 1..6 id: []const u8, // anchor, e.g. "Sentinel-Terminated-Slices" title: []const u8, // visible text, e.g. "Sentinel-Terminated Slices"};
/// Iterates `<hN id="...">title</hN>` headings in document order.pub const Sections = struct { html: []const u8, pos: usize = 0, buf: [256]u8 = undefined, // backing store for the cleaned title
pub fn init(html: []const u8) Sections { return .{ .html = html }; }
pub fn next(it: *Sections) ?Section { while (std.mem.indexOfPos(u8, it.html, it.pos, "<h")) |h| { it.pos = h + 2; const level = headingLevel(it.html[h..]) orelse continue; const gt = std.mem.indexOfScalarPos(u8, it.html, h, '>') orelse return null; const id = attr(it.html[h..gt], "id") orelse continue; const close = std.mem.indexOfPos(u8, it.html, gt, "</h") orelse return null; const title = cleanText(&it.buf, it.html[gt + 1 .. close]); it.pos = close; return .{ .level = level, .id = id, .title = title }; } return null; }};
/// Returns the HTML fragment for the section with the given anchor: from its/// heading up to the next heading of equal-or-higher rank. `null` if absent.pub fn sliceSection(html: []const u8, anchor: []const u8) ?[]const u8 { var buf: [128]u8 = undefined; const needle = std.fmt.bufPrint(&buf, "id=\"{s}\"", .{anchor}) catch return null; const at = std.mem.indexOf(u8, html, needle) orelse return null; const tag_open = std.mem.lastIndexOfScalar(u8, html[0..at], '<') orelse return null; const level = headingLevel(html[tag_open..]) orelse return null; var i = at; while (std.mem.indexOfPos(u8, html, i, "<h")) |h| { if (h > tag_open) { if (headingLevel(html[h..])) |l| { if (l <= level) return html[tag_open..h]; } } i = h + 2; } return html[tag_open..];}
/// Normalize a string to lowercase alphanumerics only (drops spaces, hyphens,/// punctuation) so "Error Union Type", "error-union-type", and "error union"/// compare on equal footing. Writes into `buf`, returns the used slice.fn normalize(buf: []u8, s: []const u8) []const u8 { var n: usize = 0; for (s) |c| { if (std.ascii.isAlphanumeric(c) and n < buf.len) { buf[n] = std.ascii.toLower(c); n += 1; } } return buf[0..n];}
/// equal, or equal after dropping a trailing plural 's' on either side/// ("optional" ~ "optionals").fn eqlOrPlural(a: []const u8, b: []const u8) bool { if (std.mem.eql(u8, a, b)) return true; if (a.len > 1 and a[a.len - 1] == 's' and std.mem.eql(u8, a[0 .. a.len - 1], b)) return true; if (b.len > 1 and b[b.len - 1] == 's' and std.mem.eql(u8, b[0 .. b.len - 1], a)) return true; return false;}
/// The "obvious primary" section for a query, when there is one — so a fuzzy/// query opens the right section instead of dumping a candidate list. Tiers:/// 1. a section whose normalized title/id equals the query (modulo plural),/// preferring the shallowest such section;/// 2. else, if exactly one section's normalized title *starts with* the query./// Returns that section's id, or null (let the caller fall back to listing).pub fn primaryMatch(html: []const u8, query: []const u8) ?[]const u8 { var qbuf: [128]u8 = undefined; const nq = normalize(&qbuf, query); if (nq.len == 0) return null;
// tier 1: exact (or plural) title/id equality, shallowest wins var best_id: ?[]const u8 = null; var best_level: u8 = 255; // tier 2: unique title prefix var prefix_id: ?[]const u8 = null; var prefix_count: usize = 0;
var it = Sections.init(html); var nbuf: [256]u8 = undefined; while (it.next()) |s| { const nt = normalize(&nbuf, s.title); if (eqlOrPlural(nt, nq)) { if (s.level < best_level) { best_level = s.level; best_id = s.id; } continue; } // id equality is a strong signal too var ibuf: [256]u8 = undefined; if (eqlOrPlural(normalize(&ibuf, s.id), nq)) { if (s.level < best_level) { best_level = s.level; best_id = s.id; } continue; } if (std.mem.startsWith(u8, nt, nq)) { prefix_count += 1; prefix_id = s.id; } } if (best_id) |id| return id; if (prefix_count == 1) return prefix_id; return null;}
/// A langref version is used in both a URL and a cache file path, so constrain/// it to characters that can't escape either: "master" or `[0-9A-Za-z.-]+`.pub fn validVersion(v: []const u8) bool { if (v.len == 0 or v.len > 32) return false; for (v) |c| { const ok = (c >= '0' and c <= '9') or (c >= 'a' and c <= 'z') or (c >= 'A' and c <= 'Z') or c == '.' or c == '-'; if (!ok) return false; } return true;}
/// The langref URL for a version, optionally deep-linked to a section anchor./// Caller owns the returned string.pub fn url(a: std.mem.Allocator, ver: []const u8, anchor: ?[]const u8) ![]u8 { if (anchor) |id| return std.fmt.allocPrint(a, "https://ziglang.org/documentation/{s}/#{s}", .{ ver, id }); return std.fmt.allocPrint(a, "https://ziglang.org/documentation/{s}/", .{ver});}
/// A query matches a section if it is a case-insensitive substring of either/// the anchor or the visible title — so `slice`, `Slice`, and `terminated` all/// find `Sentinel-Terminated-Slices`.pub fn matchesQuery(s: Section, query: []const u8) bool { return containsIgnoreCase(s.id, query) or containsIgnoreCase(s.title, query);}
pub fn containsIgnoreCase(haystack: []const u8, needle: []const u8) bool { if (needle.len == 0) return true; if (needle.len > haystack.len) return false; var i: usize = 0; while (i + needle.len <= haystack.len) : (i += 1) { if (std.ascii.eqlIgnoreCase(haystack[i .. i + needle.len], needle)) return true; } return false;}
pub fn headingLevel(s: []const u8) ?u8 { if (s.len < 3 or s[0] != '<' or s[1] != 'h') return null; if (s[2] < '1' or s[2] > '6') return null; return s[2] - '0';}
pub fn attr(tag: []const u8, name: []const u8) ?[]const u8 { var key_buf: [32]u8 = undefined; const key = std.fmt.bufPrint(&key_buf, "{s}=\"", .{name}) catch return null; const start = std.mem.indexOf(u8, tag, key) orelse return null; const vstart = start + key.len; const vend = std.mem.indexOfScalarPos(u8, tag, vstart, '"') orelse return null; return tag[vstart..vend];}
/// Visible text of a heading: drop tags and the § pilcrow. Truncates to `buf`.fn cleanText(buf: []u8, inner: []const u8) []const u8 { var n: usize = 0; var k: usize = 0; var in_tag = false; while (k < inner.len and n < buf.len) { const c = inner[k]; if (c == '<') { in_tag = true; } else if (c == '>') { in_tag = false; } else if (!in_tag) { if (c == 0xC2 and k + 1 < inner.len and inner[k + 1] == 0xA7) { k += 2; continue; } buf[n] = c; n += 1; } k += 1; } return std.mem.trim(u8, buf[0..n], " ");}
// ----------------------------------------------------------------------------// tests// ----------------------------------------------------------------------------
const testing = std.testing;
test "render: prose with inline code and entities" { const md = try render(testing.allocator, "<p>a < b and <code>len</code></p>"); defer testing.allocator.free(md); try testing.expectEqualStrings("a < b and `len`", md);}
test "render: fenced code block, spans and entities decoded" { const html = \\<pre><code><span class="tok-kw">const</span> x = <span class="tok-str">"hi"</span>;</code></pre> ; const md = try render(testing.allocator, html); defer testing.allocator.free(md); try testing.expectEqualStrings("```zig\nconst x = \"hi\";\n```", md);}
test "render: table becomes markdown rows" { const html = "<table><tr><th>A</th><th>B</th></tr><tr><td>1</td><td>2</td></tr></table>"; const md = try render(testing.allocator, html); defer testing.allocator.free(md); try testing.expectEqualStrings("| A | B |\n| --- | --- |\n| 1 | 2 |", md);}
test "sliceSection: bounded by next equal-rank heading" { const html = "<h2 id=\"a\">A</h2><p>x</p><h2 id=\"b\">B</h2>"; const sec = sliceSection(html, "a").?; try testing.expectEqualStrings("<h2 id=\"a\">A</h2><p>x</p>", sec); try testing.expect(sliceSection(html, "missing") == null);}
test "sliceSection: subsections stay within parent" { const html = "<h2 id=\"p\">P</h2><h3 id=\"c\">C</h3><p>y</p><h2 id=\"q\">Q</h2>"; const sec = sliceSection(html, "p").?; try testing.expectEqualStrings("<h2 id=\"p\">P</h2><h3 id=\"c\">C</h3><p>y</p>", sec);}
test "Sections: yields level, id, cleaned title" { const html = "<h2 id=\"Slices\"><a href=\"#toc\">Slices</a> <a class=\"hdr\">§</a></h2>" ++ "<h3 id=\"Sub\">Sub Thing</h3>"; var it = Sections.init(html); const a = it.next().?; try testing.expectEqual(@as(u8, 2), a.level); try testing.expectEqualStrings("Slices", a.id); try testing.expectEqualStrings("Slices", a.title); const b = it.next().?; try testing.expectEqual(@as(u8, 3), b.level); try testing.expectEqualStrings("Sub Thing", b.title); try testing.expect(it.next() == null);}
test "validVersion: accepts releases and master, rejects path escapes" { try testing.expect(validVersion("master")); try testing.expect(validVersion("0.15.1")); try testing.expect(!validVersion("../../etc/passwd")); try testing.expect(!validVersion("a/b")); try testing.expect(!validVersion(""));}
test "url: with and without an anchor" { const a = testing.allocator; const root = try url(a, "master", null); defer a.free(root); try testing.expectEqualStrings("https://ziglang.org/documentation/master/", root); const deep = try url(a, "0.15.1", "comptime"); defer a.free(deep); try testing.expectEqualStrings("https://ziglang.org/documentation/0.15.1/#comptime", deep);}
test "primaryMatch: fuzzy query resolves to the obvious section" { const html = "<h3 id=\"while-with-Optionals\">while with Optionals</h3>" ++ "<h3 id=\"Optional-Type\">Optional Type</h3>" ++ "<h2 id=\"Optionals\">Optionals</h2>" ++ "<h3 id=\"Error-Union-Type\">Error Union Type</h3>" ++ "<h3 id=\"while-with-Error-Unions\">while with Error Unions</h3>" ++ "<h2 id=\"Blocks\">Blocks</h2>"; // tier 1: plural-normalized title equality, even though substrings match more try testing.expectEqualStrings("Optionals", primaryMatch(html, "optional").?); try testing.expectEqualStrings("Optionals", primaryMatch(html, "Optionals").?); try testing.expectEqualStrings("Blocks", primaryMatch(html, "blocks").?); // tier 2: unique title prefix ("error union" prefixes only "Error Union Type") try testing.expectEqualStrings("Error-Union-Type", primaryMatch(html, "error union").?); // genuinely ambiguous prefix -> null (let caller list) try testing.expect(primaryMatch(html, "while") == null); try testing.expect(primaryMatch(html, "xyzzy") == null);}
test "matchesQuery: anchor or title, case-insensitive" { const s: Section = .{ .level = 3, .id = "Sentinel-Terminated-Slices", .title = "Sentinel-Terminated Slices" }; try testing.expect(matchesQuery(s, "slice")); // anchor + title try testing.expect(matchesQuery(s, "terminated")); // title word try testing.expect(matchesQuery(s, "SENTINEL")); // case-insensitive try testing.expect(!matchesQuery(s, "comptime"));}
test "containsIgnoreCase" { try testing.expect(containsIgnoreCase("Slices", "slice")); try testing.expect(containsIgnoreCase("Optionals", "OPT")); try testing.expect(!containsIgnoreCase("Errors", "slice")); try testing.expect(containsIgnoreCase("anything", ""));}