Something went wrong. Try again.
Parse/validate ISBNs in Zig
Something went wrong. Try again.
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257// SPDX-FileCopyrightText: © 2026 Jeffrey C. Ollie// SPDX-License-Identifier: MIT
//! Generates src/ranges.zig from the International ISBN Agency's range//! data, fetched from https://www.isbn-international.org/.//!//! Usage://! zig build update-ranges//!//! or directly, optionally with a previously downloaded range file//! instead of the network://! zig run tools/generate_ranges.zig -- src/ranges.zig [RangeMessage.xml]
const std = @import("std");const xml = @import("zxml");
const range_message_url = "https://www.isbn-international.org/export_rangemessage.xml";
const Rule = struct { min: u32, max: u32, len: u8,};
const Group = struct { ean: u16, digits: []const u8, rules: []const Rule,};
fn splitOnDash(s: []const u8) ?struct { []const u8, []const u8 } { const dash = std.mem.indexOfScalar(u8, s, '-') orelse return null; return .{ s[0..dash], s[dash + 1 ..] };}
/// Walks the document and collects every `<Group>` under/// `<RegistrationGroups>`. The `<EAN.UCCPrefixes>` section describes the 978/// and 979 prefixes themselves and carries no registrant rules, so it is/// skipped whole.////// This is a structural walk rather than a search for tag text: the range/// message carries an internal DTD subset that mentions `<!ELEMENT Group ...>`,/// and matching on substrings cannot tell that apart from an element.fn parseGroups(arena: std.mem.Allocator, document: []const u8) ![]Group { var groups: std.ArrayList(Group) = .empty; var r: xml.Reader = .init(try xml.toUtf8(arena, document));
while (true) { switch (try r.next()) { .eof => break, .start_element => |e| { if (std.mem.eql(u8, e.name, "RegistrationGroups")) continue; if (std.mem.eql(u8, e.name, "Group")) { try groups.append(arena, try parseGroup(arena, &r)); } else if (!std.mem.eql(u8, e.name, "ISBNRangeMessage")) { try r.skipElement(); } }, else => {}, } } return groups.toOwnedSlice(arena);}
fn parseGroup(arena: std.mem.Allocator, r: *xml.Reader) !Group { const target = r.depth; var ean: ?u16 = null; var digits: []const u8 = ""; var rules: std.ArrayList(Rule) = .empty;
while (true) { switch (try r.next()) { .eof => return error.MalformedXml, .end_element => if (r.depth < target) break, .start_element => |e| { if (std.mem.eql(u8, e.name, "Prefix")) { const prefix = try r.trimmedTextAlloc(arena, .lenient); const ean_str, const rest = splitOnDash(prefix) orelse return error.MalformedXml; ean = try std.fmt.parseInt(u16, ean_str, 10); digits = rest; } else if (std.mem.eql(u8, e.name, "Rule")) { try rules.append(arena, try parseRule(arena, r)); } else if (!std.mem.eql(u8, e.name, "Rules")) { try r.skipElement(); } }, else => {}, } }
const prefix = ean orelse return error.MalformedXml; if (prefix != 978 and prefix != 979) return error.MalformedXml; return .{ .ean = prefix, .digits = digits, .rules = try rules.toOwnedSlice(arena) };}
fn parseRule(arena: std.mem.Allocator, r: *xml.Reader) !Rule { const target = r.depth; var min: ?u32 = null; var max: ?u32 = null; var len: ?u8 = null;
while (true) { switch (try r.next()) { .eof => return error.MalformedXml, .end_element => if (r.depth < target) break, .start_element => |e| { if (std.mem.eql(u8, e.name, "Range")) { const range = try r.trimmedTextAlloc(arena, .lenient); const lo, const hi = splitOnDash(range) orelse return error.MalformedXml; if (lo.len != 7 or hi.len != 7) return error.MalformedXml; min = try std.fmt.parseInt(u32, lo, 10); max = try std.fmt.parseInt(u32, hi, 10); } else if (std.mem.eql(u8, e.name, "Length")) { const length = try r.trimmedTextAlloc(arena, .lenient); len = try std.fmt.parseInt(u8, length, 10); } else { try r.skipElement(); } }, else => {}, } }
return .{ .min = min orelse return error.MalformedXml, .max = max orelse return error.MalformedXml, .len = len orelse return error.MalformedXml, };}
pub fn main(init: std.process.Init) !void { const arena = init.arena.allocator(); const io = init.io;
var it: std.process.Args.Iterator = .init(init.minimal.args); _ = it.next(); const output_path = it.next(); const input_path = it.next(); if (output_path == null or it.next() != null) { std.debug.print("usage: generate_ranges <ranges.zig> [RangeMessage.xml]\n", .{}); return error.InvalidArguments; }
const document = if (input_path) |path| try std.Io.Dir.cwd().readFileAlloc(io, path, arena, .limited(16 * 1024 * 1024)) else document: { var client: std.http.Client = .{ .allocator = init.gpa, .io = io }; defer client.deinit(); var body: std.Io.Writer.Allocating = .init(arena); const result = try client.fetch(.{ .location = .{ .url = range_message_url }, .response_writer = &body.writer, }); if (result.status != .ok) { std.debug.print("HTTP {d} fetching {s}\n", .{ @intFromEnum(result.status), range_message_url }); return error.HttpRequestFailed; } break :document body.written(); };
const groups = try parseGroups(arena, document); if (groups.len == 0) return error.MalformedXml;
std.mem.sort(Group, groups, {}, struct { fn lessThan(_: void, a: Group, b: Group) bool { if (a.ean != b.ean) return a.ean < b.ean; return std.mem.order(u8, a.digits, b.digits) == .lt; } }.lessThan);
// Group lookup relies on group prefixes being prefix-free per EAN // prefix. for (groups) |a| { for (groups) |b| { if (a.ean == b.ean and !std.mem.eql(u8, a.digits, b.digits) and std.mem.startsWith(u8, b.digits, a.digits)) { std.debug.print("group {d}-{s} is a prefix of {d}-{s}\n", .{ a.ean, a.digits, b.ean, b.digits }); return error.GroupsNotPrefixFree; } } }
// The body is emitted first so that it can be fingerprinted, and the // header written around it afterwards. var body: std.Io.Writer.Allocating = .init(arena); const w = &body.writer; try w.writeAll("/// Determines the length of the registrant element from the value of\n"); try w.writeAll("/// the seven digits that follow the registration group.\n"); try w.writeAll("pub const Rule = struct {\n"); try w.writeAll(" /// Inclusive lower bound on the seven-digit value.\n"); try w.writeAll(" min: u32,\n"); try w.writeAll(" /// Inclusive upper bound on the seven-digit value.\n"); try w.writeAll(" max: u32,\n"); try w.writeAll(" /// Number of digits in the registrant element; 0 means the range\n"); try w.writeAll(" /// has not been assigned and cannot be hyphenated.\n"); try w.writeAll(" len: u8,\n"); try w.writeAll("};\n"); try w.writeAll("\n"); try w.writeAll("/// A registration group (a country, region, or language area).\n"); try w.writeAll("pub const Group = struct {\n"); try w.writeAll(" /// The EAN prefix, 978 or 979.\n"); try w.writeAll(" ean: u16,\n"); try w.writeAll(" /// The registration group digits that follow the EAN prefix.\n"); try w.writeAll(" digits: []const u8,\n"); try w.writeAll(" /// Registrant length rules for this group.\n"); try w.writeAll(" rules: []const Rule,\n"); try w.writeAll("};\n"); try w.writeAll("\n"); try w.writeAll("pub const groups: []const Group = &.{\n"); for (groups) |group| { try w.writeAll(" .{\n"); try w.print(" .ean = {d},\n", .{group.ean}); try w.print(" .digits = \"{s}\",\n", .{group.digits}); try w.writeAll(" .rules = &.{\n"); for (group.rules) |rule| { try w.print(" .{{ .min = {d}, .max = {d}, .len = {d} }},\n", .{ rule.min, rule.max, rule.len }); } try w.writeAll(" },\n"); try w.writeAll(" },\n"); } try w.writeAll("};\n");
var rule_count: usize = 0; for (groups) |g| rule_count += g.rules.len;
var digest: [std.crypto.hash.sha2.Sha256.digest_length]u8 = undefined; std.crypto.hash.sha2.Sha256.hash(body.written(), &digest, .{});
var out: std.Io.Writer.Allocating = .init(arena); const hw = &out.writer; // REUSE-IgnoreStart try hw.writeAll("// SPDX-FileCopyrightText: © 2026 Jeffrey C. Ollie\n"); try hw.writeAll("// SPDX-License-Identifier: MIT\n"); // REUSE-IgnoreEnd try hw.writeAll("\n"); try hw.writeAll("//! ISBN registration group and registrant ranges, generated by\n"); try hw.writeAll("//! tools/generate_ranges.zig from the International ISBN Agency's\n"); try hw.writeAll("//! range data (https://www.isbn-international.org/export_rangemessage.xml).\n"); try hw.writeAll("//! Do not edit by hand.\n"); try hw.writeAll("//!\n"); try hw.print("//! {d} registration groups, {d} rules.\n", .{ groups.len, rule_count }); try hw.print("//! Data fingerprint: {x}\n", .{&digest}); try hw.writeAll("//!\n"); try hw.writeAll("//! The fingerprint covers the ranges below and nothing else. The agency\n"); try hw.writeAll("//! stamps a fresh MessageSerialNumber and MessageDate on every download,\n"); try hw.writeAll("//! so recording either would make this file differ on every regeneration\n"); try hw.writeAll("//! and hide whether the ranges actually moved.\n"); try hw.writeAll("\n"); try hw.writeAll(body.written());
try std.Io.Dir.cwd().writeFile(io, .{ .sub_path = output_path.?, .data = out.written(), });}