//! zigman library: parse the zig language reference (a single static HTML page
//! with stable per-section `id` anchors) into sections and clean markdown.
//!
//! The CLI in `main.zig` is a thin shell over this; everything here is pure
//! (no I/O), so it is all unit-testable.
const std = @import("std");
pub const render = @import("render.zig").render;
/// A heading in the reference's table of contents.
pub const Section = struct {
level: u8, // 1..6
id: []const u8, // anchor, e.g. "Sentinel-Terminated-Slices"
title: []const u8, // visible text, e.g. "Sentinel-Terminated Slices"
};
/// Iterates `title` headings in document order.
pub const Sections = struct {
html: []const u8,
pos: usize = 0,
buf: [256]u8 = undefined, // backing store for the cleaned title
pub fn init(html: []const u8) Sections {
return .{ .html = html };
}
pub fn next(it: *Sections) ?Section {
while (std.mem.indexOfPos(u8, it.html, it.pos, "') orelse return null;
const id = attr(it.html[h..gt], "id") orelse continue;
const close = std.mem.indexOfPos(u8, it.html, gt, " tag_open) {
if (headingLevel(html[h..])) |l| {
if (l <= level) return html[tag_open..h];
}
}
i = h + 2;
}
return html[tag_open..];
}
/// Normalize a string to lowercase alphanumerics only (drops spaces, hyphens,
/// punctuation) so "Error Union Type", "error-union-type", and "error union"
/// compare on equal footing. Writes into `buf`, returns the used slice.
fn normalize(buf: []u8, s: []const u8) []const u8 {
var n: usize = 0;
for (s) |c| {
if (std.ascii.isAlphanumeric(c) and n < buf.len) {
buf[n] = std.ascii.toLower(c);
n += 1;
}
}
return buf[0..n];
}
/// equal, or equal after dropping a trailing plural 's' on either side
/// ("optional" ~ "optionals").
fn eqlOrPlural(a: []const u8, b: []const u8) bool {
if (std.mem.eql(u8, a, b)) return true;
if (a.len > 1 and a[a.len - 1] == 's' and std.mem.eql(u8, a[0 .. a.len - 1], b)) return true;
if (b.len > 1 and b[b.len - 1] == 's' and std.mem.eql(u8, b[0 .. b.len - 1], a)) return true;
return false;
}
/// The "obvious primary" section for a query, when there is one — so a fuzzy
/// query opens the right section instead of dumping a candidate list. Tiers:
/// 1. a section whose normalized title/id equals the query (modulo plural),
/// preferring the shallowest such section;
/// 2. else, if exactly one section's normalized title *starts with* the query.
/// Returns that section's id, or null (let the caller fall back to listing).
pub fn primaryMatch(html: []const u8, query: []const u8) ?[]const u8 {
var qbuf: [128]u8 = undefined;
const nq = normalize(&qbuf, query);
if (nq.len == 0) return null;
// tier 1: exact (or plural) title/id equality, shallowest wins
var best_id: ?[]const u8 = null;
var best_level: u8 = 255;
// tier 2: unique title prefix
var prefix_id: ?[]const u8 = null;
var prefix_count: usize = 0;
var it = Sections.init(html);
var nbuf: [256]u8 = undefined;
while (it.next()) |s| {
const nt = normalize(&nbuf, s.title);
if (eqlOrPlural(nt, nq)) {
if (s.level < best_level) {
best_level = s.level;
best_id = s.id;
}
continue;
}
// id equality is a strong signal too
var ibuf: [256]u8 = undefined;
if (eqlOrPlural(normalize(&ibuf, s.id), nq)) {
if (s.level < best_level) {
best_level = s.level;
best_id = s.id;
}
continue;
}
if (std.mem.startsWith(u8, nt, nq)) {
prefix_count += 1;
prefix_id = s.id;
}
}
if (best_id) |id| return id;
if (prefix_count == 1) return prefix_id;
return null;
}
/// A langref version is used in both a URL and a cache file path, so constrain
/// it to characters that can't escape either: "master" or `[0-9A-Za-z.-]+`.
pub fn validVersion(v: []const u8) bool {
if (v.len == 0 or v.len > 32) return false;
for (v) |c| {
const ok = (c >= '0' and c <= '9') or (c >= 'a' and c <= 'z') or
(c >= 'A' and c <= 'Z') or c == '.' or c == '-';
if (!ok) return false;
}
return true;
}
/// The langref URL for a version, optionally deep-linked to a section anchor.
/// Caller owns the returned string.
pub fn url(a: std.mem.Allocator, ver: []const u8, anchor: ?[]const u8) ![]u8 {
if (anchor) |id| return std.fmt.allocPrint(a, "https://ziglang.org/documentation/{s}/#{s}", .{ ver, id });
return std.fmt.allocPrint(a, "https://ziglang.org/documentation/{s}/", .{ver});
}
/// A query matches a section if it is a case-insensitive substring of either
/// the anchor or the visible title — so `slice`, `Slice`, and `terminated` all
/// find `Sentinel-Terminated-Slices`.
pub fn matchesQuery(s: Section, query: []const u8) bool {
return containsIgnoreCase(s.id, query) or containsIgnoreCase(s.title, query);
}
pub fn containsIgnoreCase(haystack: []const u8, needle: []const u8) bool {
if (needle.len == 0) return true;
if (needle.len > haystack.len) return false;
var i: usize = 0;
while (i + needle.len <= haystack.len) : (i += 1) {
if (std.ascii.eqlIgnoreCase(haystack[i .. i + needle.len], needle)) return true;
}
return false;
}
pub fn headingLevel(s: []const u8) ?u8 {
if (s.len < 3 or s[0] != '<' or s[1] != 'h') return null;
if (s[2] < '1' or s[2] > '6') return null;
return s[2] - '0';
}
pub fn attr(tag: []const u8, name: []const u8) ?[]const u8 {
var key_buf: [32]u8 = undefined;
const key = std.fmt.bufPrint(&key_buf, "{s}=\"", .{name}) catch return null;
const start = std.mem.indexOf(u8, tag, key) orelse return null;
const vstart = start + key.len;
const vend = std.mem.indexOfScalarPos(u8, tag, vstart, '"') orelse return null;
return tag[vstart..vend];
}
/// Visible text of a heading: drop tags and the § pilcrow. Truncates to `buf`.
fn cleanText(buf: []u8, inner: []const u8) []const u8 {
var n: usize = 0;
var k: usize = 0;
var in_tag = false;
while (k < inner.len and n < buf.len) {
const c = inner[k];
if (c == '<') {
in_tag = true;
} else if (c == '>') {
in_tag = false;
} else if (!in_tag) {
if (c == 0xC2 and k + 1 < inner.len and inner[k + 1] == 0xA7) {
k += 2;
continue;
}
buf[n] = c;
n += 1;
}
k += 1;
}
return std.mem.trim(u8, buf[0..n], " ");
}
// ----------------------------------------------------------------------------
// tests
// ----------------------------------------------------------------------------
const testing = std.testing;
test "render: prose with inline code and entities" {
const md = try render(testing.allocator, "
a < b and len
");
defer testing.allocator.free(md);
try testing.expectEqualStrings("a < b and `len`", md);
}
test "render: fenced code block, spans and entities decoded" {
const html =
\\const x = "hi";
;
const md = try render(testing.allocator, html);
defer testing.allocator.free(md);
try testing.expectEqualStrings("```zig\nconst x = \"hi\";\n```", md);
}
test "render: table becomes markdown rows" {
const html = "";
const md = try render(testing.allocator, html);
defer testing.allocator.free(md);
try testing.expectEqualStrings("| A | B |\n| --- | --- |\n| 1 | 2 |", md);
}
test "sliceSection: bounded by next equal-rank heading" {
const html = "A
x
B
";
const sec = sliceSection(html, "a").?;
try testing.expectEqualStrings("A
x
", sec);
try testing.expect(sliceSection(html, "missing") == null);
}
test "sliceSection: subsections stay within parent" {
const html = "P
C
y
Q
";
const sec = sliceSection(html, "p").?;
try testing.expectEqualStrings("P
C
y
", sec);
}
test "Sections: yields level, id, cleaned title" {
const html = "" ++
"Sub Thing
";
var it = Sections.init(html);
const a = it.next().?;
try testing.expectEqual(@as(u8, 2), a.level);
try testing.expectEqualStrings("Slices", a.id);
try testing.expectEqualStrings("Slices", a.title);
const b = it.next().?;
try testing.expectEqual(@as(u8, 3), b.level);
try testing.expectEqualStrings("Sub Thing", b.title);
try testing.expect(it.next() == null);
}
test "validVersion: accepts releases and master, rejects path escapes" {
try testing.expect(validVersion("master"));
try testing.expect(validVersion("0.15.1"));
try testing.expect(!validVersion("../../etc/passwd"));
try testing.expect(!validVersion("a/b"));
try testing.expect(!validVersion(""));
}
test "url: with and without an anchor" {
const a = testing.allocator;
const root = try url(a, "master", null);
defer a.free(root);
try testing.expectEqualStrings("https://ziglang.org/documentation/master/", root);
const deep = try url(a, "0.15.1", "comptime");
defer a.free(deep);
try testing.expectEqualStrings("https://ziglang.org/documentation/0.15.1/#comptime", deep);
}
test "primaryMatch: fuzzy query resolves to the obvious section" {
const html =
"while with Optionals
" ++
"Optional Type
" ++
"Optionals
" ++
"Error Union Type
" ++
"while with Error Unions
" ++
"Blocks
";
// tier 1: plural-normalized title equality, even though substrings match more
try testing.expectEqualStrings("Optionals", primaryMatch(html, "optional").?);
try testing.expectEqualStrings("Optionals", primaryMatch(html, "Optionals").?);
try testing.expectEqualStrings("Blocks", primaryMatch(html, "blocks").?);
// tier 2: unique title prefix ("error union" prefixes only "Error Union Type")
try testing.expectEqualStrings("Error-Union-Type", primaryMatch(html, "error union").?);
// genuinely ambiguous prefix -> null (let caller list)
try testing.expect(primaryMatch(html, "while") == null);
try testing.expect(primaryMatch(html, "xyzzy") == null);
}
test "matchesQuery: anchor or title, case-insensitive" {
const s: Section = .{ .level = 3, .id = "Sentinel-Terminated-Slices", .title = "Sentinel-Terminated Slices" };
try testing.expect(matchesQuery(s, "slice")); // anchor + title
try testing.expect(matchesQuery(s, "terminated")); // title word
try testing.expect(matchesQuery(s, "SENTINEL")); // case-insensitive
try testing.expect(!matchesQuery(s, "comptime"));
}
test "containsIgnoreCase" {
try testing.expect(containsIgnoreCase("Slices", "slice"));
try testing.expect(containsIgnoreCase("Optionals", "OPT"));
try testing.expect(!containsIgnoreCase("Errors", "slice"));
try testing.expect(containsIgnoreCase("anything", ""));
}