Something went wrong. Try again.
sentence embeddings in pure zig: bge-small with an HF-exact tokenizer and an SME matmul
Something went wrong. Try again.
1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677//! Kernel microbenchmark: GFLOP/s of each encoder matmul shape and of attention.//!//! usage: embedz-bench [ROWS]
const std = @import("std");const embedz = @import("embedz");const ops = embedz.ops;const Io = std.Io;
pub fn main(init: std.process.Init) !void { const gpa = init.gpa; const io = init.io; const argv = try init.minimal.args.toSlice(init.arena.allocator()); const rows: usize = if (argv.len > 1) try std.fmt.parseInt(usize, argv[1], 10) else 2048; if (@import("builtin").os.tag == .macos) { const f = struct { extern "c" fn pthread_set_qos_class_self_np(qos: c_uint, priority: c_int) c_int; }; _ = f.pthread_set_qos_class_self_np(0x21, 0); }
var prng = std.Random.DefaultPrng.init(42); const rand = prng.random();
const bf16_ok = embedz.bf16.selfTest(); const int8_ok = embedz.int8.selfTest(); std.debug.print("avx512-bf16: available={} self_test={}; int8 avx2 self_test={}\n", .{ embedz.bf16.available(), bf16_ok, int8_ok });
const shapes = [_][2]usize{ .{ 384, 1152 }, .{ 384, 384 }, .{ 384, 1536 }, .{ 1536, 384 } }; for ([_]embedz.Precision{ .f32, .bf16, .int8 }) |prec| for (shapes) |shape| { if (prec == .bf16 and !bf16_ok) break; if (prec == .int8 and !int8_ok) break; const in, const out = shape; const w = try gpa.alloc(f32, in * out); defer gpa.free(w); const b = try gpa.alloc(f32, out); defer gpa.free(b); for (w) |*v| v.* = rand.float(f32) - 0.5; for (b) |*v| v.* = rand.float(f32); var lin = switch (prec) { .bf16 => try ops.Linear.packBf16(gpa, w, b, in, out, true), .int8 => try ops.Linear.packInt8(gpa, w, b, in, out, true), else => try ops.Linear.pack(gpa, w, b, in, out), }; defer lin.deinit(gpa);
const x = try gpa.alloc(f32, rows * in); defer gpa.free(x); const y = try gpa.alloc(f32, rows * out); defer gpa.free(y); for (x) |*v| v.* = rand.float(f32) - 0.5;
lin.forward(x, y, rows); const reps = 20; const t = Io.Timestamp.now(io, .awake); for (0..reps) |_| lin.forward(x, y, rows); const ns: f64 = @floatFromInt(t.untilNow(io, .awake).nanoseconds); const flop: f64 = @floatFromInt(2 * rows * in * out * reps); std.debug.print("linear {t:<4} {d:>4}x{d:<4} rows={d}: {d:.1} GFLOP/s\n", .{ prec, in, out, rows, flop / ns }); };
const len = 33; const seqs = rows / len; const qkv = try gpa.alloc(f32, rows * 3 * 384); defer gpa.free(qkv); const ctx = try gpa.alloc(f32, rows * 384); defer gpa.free(ctx); var scores: [ops.attentionScratch(512)]f32 = undefined; for (qkv) |*v| v.* = rand.float(f32) - 0.5; const t = Io.Timestamp.now(io, .awake); const reps = 20; for (0..reps) |_| for (0..seqs) |s| ops.attention(qkv[s * len * 3 * 384 ..], ctx[s * len * 384 ..], &scores, len, 384, 12, len); const ns: f64 = @floatFromInt(t.untilNow(io, .awake).nanoseconds); const flop: f64 = @floatFromInt(reps * seqs * 4 * len * len * 384); std.debug.print("attention len={d} seqs={d}: {d:.1} GFLOP/s, {d:.1} us/seq\n", .{ len, seqs, flop / ns, ns / 1000 / @as(f64, @floatFromInt(reps * seqs)) });}