# /// script # requires-python = ">=3.12" # dependencies = ["sentence-transformers>=5", "torch>=2.6", "numpy"] # /// """Dump the reference implementation's view of a corpus, and time it. usage: uv run scripts/reference.py CORPUS.jsonl OUT_DIR [--model HF_ID] [--bench] writes, in OUT_DIR: model_path.txt local snapshot directory holding model.safetensors + tokenizer.json ids.jsonl one JSON array of token ids per post (what the model actually sees) embeddings.f32 N x DIM little-endian float32, row per post, as the model outputs them (normalized only if its modules include Normalize) meta.json model id, dims, count, torch version, thread count, timings """ import json import sys import time from pathlib import Path import numpy as np import torch from huggingface_hub import snapshot_download from sentence_transformers import SentenceTransformer DEFAULT_MODEL = "BAAI/bge-small-en-v1.5" BATCH = 64 def main() -> None: corpus, out_dir = Path(sys.argv[1]), Path(sys.argv[2]) bench = "--bench" in sys.argv model_id = sys.argv[sys.argv.index("--model") + 1] if "--model" in sys.argv else DEFAULT_MODEL out_dir.mkdir(parents=True, exist_ok=True) texts = [json.loads(line)["text"] for line in corpus.read_text().splitlines()] path = snapshot_download(model_id, allow_patterns=["*.json", "*.safetensors", "*.txt", "1_Pooling/*"]) (out_dir / "model_path.txt").write_text(path) model = SentenceTransformer(model_id, device="cpu") tok = model.tokenizer max_len = model.max_seq_length with (out_dir / "ids.jsonl").open("w") as f: for t in texts: ids = tok(t, truncation=True, max_length=max_len)["input_ids"] f.write(json.dumps(ids) + "\n") emb = model.encode(texts, batch_size=BATCH, convert_to_numpy=True) emb.astype("