//! zigman library: parse the zig language reference (a single static HTML page //! with stable per-section `id` anchors) into sections and clean markdown. //! //! The CLI in `main.zig` is a thin shell over this; everything here is pure //! (no I/O), so it is all unit-testable. const std = @import("std"); pub const render = @import("render.zig").render; /// A heading in the reference's table of contents. pub const Section = struct { level: u8, // 1..6 id: []const u8, // anchor, e.g. "Sentinel-Terminated-Slices" title: []const u8, // visible text, e.g. "Sentinel-Terminated Slices" }; /// Iterates `title` headings in document order. pub const Sections = struct { html: []const u8, pos: usize = 0, buf: [256]u8 = undefined, // backing store for the cleaned title pub fn init(html: []const u8) Sections { return .{ .html = html }; } pub fn next(it: *Sections) ?Section { while (std.mem.indexOfPos(u8, it.html, it.pos, "') orelse return null; const id = attr(it.html[h..gt], "id") orelse continue; const close = std.mem.indexOfPos(u8, it.html, gt, " tag_open) { if (headingLevel(html[h..])) |l| { if (l <= level) return html[tag_open..h]; } } i = h + 2; } return html[tag_open..]; } /// Normalize a string to lowercase alphanumerics only (drops spaces, hyphens, /// punctuation) so "Error Union Type", "error-union-type", and "error union" /// compare on equal footing. Writes into `buf`, returns the used slice. fn normalize(buf: []u8, s: []const u8) []const u8 { var n: usize = 0; for (s) |c| { if (std.ascii.isAlphanumeric(c) and n < buf.len) { buf[n] = std.ascii.toLower(c); n += 1; } } return buf[0..n]; } /// equal, or equal after dropping a trailing plural 's' on either side /// ("optional" ~ "optionals"). fn eqlOrPlural(a: []const u8, b: []const u8) bool { if (std.mem.eql(u8, a, b)) return true; if (a.len > 1 and a[a.len - 1] == 's' and std.mem.eql(u8, a[0 .. a.len - 1], b)) return true; if (b.len > 1 and b[b.len - 1] == 's' and std.mem.eql(u8, b[0 .. b.len - 1], a)) return true; return false; } /// The "obvious primary" section for a query, when there is one — so a fuzzy /// query opens the right section instead of dumping a candidate list. Tiers: /// 1. a section whose normalized title/id equals the query (modulo plural), /// preferring the shallowest such section; /// 2. else, if exactly one section's normalized title *starts with* the query. /// Returns that section's id, or null (let the caller fall back to listing). pub fn primaryMatch(html: []const u8, query: []const u8) ?[]const u8 { var qbuf: [128]u8 = undefined; const nq = normalize(&qbuf, query); if (nq.len == 0) return null; // tier 1: exact (or plural) title/id equality, shallowest wins var best_id: ?[]const u8 = null; var best_level: u8 = 255; // tier 2: unique title prefix var prefix_id: ?[]const u8 = null; var prefix_count: usize = 0; var it = Sections.init(html); var nbuf: [256]u8 = undefined; while (it.next()) |s| { const nt = normalize(&nbuf, s.title); if (eqlOrPlural(nt, nq)) { if (s.level < best_level) { best_level = s.level; best_id = s.id; } continue; } // id equality is a strong signal too var ibuf: [256]u8 = undefined; if (eqlOrPlural(normalize(&ibuf, s.id), nq)) { if (s.level < best_level) { best_level = s.level; best_id = s.id; } continue; } if (std.mem.startsWith(u8, nt, nq)) { prefix_count += 1; prefix_id = s.id; } } if (best_id) |id| return id; if (prefix_count == 1) return prefix_id; return null; } /// A langref version is used in both a URL and a cache file path, so constrain /// it to characters that can't escape either: "master" or `[0-9A-Za-z.-]+`. pub fn validVersion(v: []const u8) bool { if (v.len == 0 or v.len > 32) return false; for (v) |c| { const ok = (c >= '0' and c <= '9') or (c >= 'a' and c <= 'z') or (c >= 'A' and c <= 'Z') or c == '.' or c == '-'; if (!ok) return false; } return true; } /// The langref URL for a version, optionally deep-linked to a section anchor. /// Caller owns the returned string. pub fn url(a: std.mem.Allocator, ver: []const u8, anchor: ?[]const u8) ![]u8 { if (anchor) |id| return std.fmt.allocPrint(a, "https://ziglang.org/documentation/{s}/#{s}", .{ ver, id }); return std.fmt.allocPrint(a, "https://ziglang.org/documentation/{s}/", .{ver}); } /// A query matches a section if it is a case-insensitive substring of either /// the anchor or the visible title — so `slice`, `Slice`, and `terminated` all /// find `Sentinel-Terminated-Slices`. pub fn matchesQuery(s: Section, query: []const u8) bool { return containsIgnoreCase(s.id, query) or containsIgnoreCase(s.title, query); } pub fn containsIgnoreCase(haystack: []const u8, needle: []const u8) bool { if (needle.len == 0) return true; if (needle.len > haystack.len) return false; var i: usize = 0; while (i + needle.len <= haystack.len) : (i += 1) { if (std.ascii.eqlIgnoreCase(haystack[i .. i + needle.len], needle)) return true; } return false; } pub fn headingLevel(s: []const u8) ?u8 { if (s.len < 3 or s[0] != '<' or s[1] != 'h') return null; if (s[2] < '1' or s[2] > '6') return null; return s[2] - '0'; } pub fn attr(tag: []const u8, name: []const u8) ?[]const u8 { var key_buf: [32]u8 = undefined; const key = std.fmt.bufPrint(&key_buf, "{s}=\"", .{name}) catch return null; const start = std.mem.indexOf(u8, tag, key) orelse return null; const vstart = start + key.len; const vend = std.mem.indexOfScalarPos(u8, tag, vstart, '"') orelse return null; return tag[vstart..vend]; } /// Visible text of a heading: drop tags and the § pilcrow. Truncates to `buf`. fn cleanText(buf: []u8, inner: []const u8) []const u8 { var n: usize = 0; var k: usize = 0; var in_tag = false; while (k < inner.len and n < buf.len) { const c = inner[k]; if (c == '<') { in_tag = true; } else if (c == '>') { in_tag = false; } else if (!in_tag) { if (c == 0xC2 and k + 1 < inner.len and inner[k + 1] == 0xA7) { k += 2; continue; } buf[n] = c; n += 1; } k += 1; } return std.mem.trim(u8, buf[0..n], " "); } // ---------------------------------------------------------------------------- // tests // ---------------------------------------------------------------------------- const testing = std.testing; test "render: prose with inline code and entities" { const md = try render(testing.allocator, "

a < b and len

"); defer testing.allocator.free(md); try testing.expectEqualStrings("a < b and `len`", md); } test "render: fenced code block, spans and entities decoded" { const html = \\
const x = "hi";
; const md = try render(testing.allocator, html); defer testing.allocator.free(md); try testing.expectEqualStrings("```zig\nconst x = \"hi\";\n```", md); } test "render: table becomes markdown rows" { const html = "
AB
12
"; const md = try render(testing.allocator, html); defer testing.allocator.free(md); try testing.expectEqualStrings("| A | B |\n| --- | --- |\n| 1 | 2 |", md); } test "sliceSection: bounded by next equal-rank heading" { const html = "

A

x

B

"; const sec = sliceSection(html, "a").?; try testing.expectEqualStrings("

A

x

", sec); try testing.expect(sliceSection(html, "missing") == null); } test "sliceSection: subsections stay within parent" { const html = "

P

C

y

Q

"; const sec = sliceSection(html, "p").?; try testing.expectEqualStrings("

P

C

y

", sec); } test "Sections: yields level, id, cleaned title" { const html = "

Slices §

" ++ "

Sub Thing

"; var it = Sections.init(html); const a = it.next().?; try testing.expectEqual(@as(u8, 2), a.level); try testing.expectEqualStrings("Slices", a.id); try testing.expectEqualStrings("Slices", a.title); const b = it.next().?; try testing.expectEqual(@as(u8, 3), b.level); try testing.expectEqualStrings("Sub Thing", b.title); try testing.expect(it.next() == null); } test "validVersion: accepts releases and master, rejects path escapes" { try testing.expect(validVersion("master")); try testing.expect(validVersion("0.15.1")); try testing.expect(!validVersion("../../etc/passwd")); try testing.expect(!validVersion("a/b")); try testing.expect(!validVersion("")); } test "url: with and without an anchor" { const a = testing.allocator; const root = try url(a, "master", null); defer a.free(root); try testing.expectEqualStrings("https://ziglang.org/documentation/master/", root); const deep = try url(a, "0.15.1", "comptime"); defer a.free(deep); try testing.expectEqualStrings("https://ziglang.org/documentation/0.15.1/#comptime", deep); } test "primaryMatch: fuzzy query resolves to the obvious section" { const html = "

while with Optionals

" ++ "

Optional Type

" ++ "

Optionals

" ++ "

Error Union Type

" ++ "

while with Error Unions

" ++ "

Blocks

"; // tier 1: plural-normalized title equality, even though substrings match more try testing.expectEqualStrings("Optionals", primaryMatch(html, "optional").?); try testing.expectEqualStrings("Optionals", primaryMatch(html, "Optionals").?); try testing.expectEqualStrings("Blocks", primaryMatch(html, "blocks").?); // tier 2: unique title prefix ("error union" prefixes only "Error Union Type") try testing.expectEqualStrings("Error-Union-Type", primaryMatch(html, "error union").?); // genuinely ambiguous prefix -> null (let caller list) try testing.expect(primaryMatch(html, "while") == null); try testing.expect(primaryMatch(html, "xyzzy") == null); } test "matchesQuery: anchor or title, case-insensitive" { const s: Section = .{ .level = 3, .id = "Sentinel-Terminated-Slices", .title = "Sentinel-Terminated Slices" }; try testing.expect(matchesQuery(s, "slice")); // anchor + title try testing.expect(matchesQuery(s, "terminated")); // title word try testing.expect(matchesQuery(s, "SENTINEL")); // case-insensitive try testing.expect(!matchesQuery(s, "comptime")); } test "containsIgnoreCase" { try testing.expect(containsIgnoreCase("Slices", "slice")); try testing.expect(containsIgnoreCase("Optionals", "OPT")); try testing.expect(!containsIgnoreCase("Errors", "slice")); try testing.expect(containsIgnoreCase("anything", "")); }