export function porcentage(min: number, max: number, v: number) { return ((v - min) / (max - min)) * 100; } /** Converts a cell (string, number, whatever) to a number. */ export function toNumber(value: unknown): number { return typeof value === "number" ? value : parseFloat(String(value)); } /* Shuffles an array, used in Hover when multiple tooltips */ export function shuffle(arr: T[]): T[] { const result = [...arr]; for (let i = result.length - 1; i > 0; i--) { const j = Math.floor(Math.random() * (i + 1)); [result[i], result[j]] = [result[j], result[i]]; } return result; } /** Minimal RFC4180-style split: handles quoted fields that contain the * delimiter or escaped quotes (""). */ export function splitCSVLine(line: string): string[] { const result: string[] = []; let cur = ""; let inQuotes = false; for (let i = 0; i < line.length; i++) { const c = line[i]; if (inQuotes) { if (c === '"') { if (line[i + 1] === '"') { cur += '"'; i++; } else inQuotes = false; } else { cur += c; } } else { if (c === '"') inQuotes = true; else if (c === ",") { result.push(cur.trim()); cur = ""; } else cur += c; } } result.push(cur.trim()); return result; } /** * FORTRAN-style format codes VizieR uses: * F/E/D = floating point * I = integer * A/a = text */ function classifyFormat(format: string): "number" | "string" { const letter = format[0]?.toUpperCase(); return letter === "F" || letter === "E" || letter === "D" || letter === "I" ? "number" : "string"; } /** VizieR prefixes a column's description with "?" to mark it as * nullable, in both its .tsv comment lines and its .vot FIELD * descriptions. Not part of the actual description, so strip it. */ function stripNullableMarker(description: string): string { return description.startsWith("?") ? description.slice(1).trim() : description; } /** * Reads the `#Column NAME (FORMAT) DESCRIPTION [ucd=...]` lines VizieR * puts above the table. These are always tab-separated regardless of the * table's own delimiter, which is what makes them a reliable source for * descriptions and types even when we're not sure how the table below * is delimited yet. */ export function parseColumnComments(rawRows: string[]): Map { const docs = new Map(); for (const line of rawRows) { if (!line.startsWith("#Column\t")) continue; const parts = line.split("\t"); if (parts.length < 4) continue; const name = parts[1].trim(); const format = parts[2].trim().replace(/^\(|\)$/g, ""); const description = stripNullableMarker(parts[3].trim()); docs.set(name, { description, type: classifyFormat(format) }); } return docs; } /** A column is "number" only if every non-empty value in it parses as * a finite number; empty/missing values don't disqualify it. */ export function inferType(values: string[]): "number" | "string" { const nonEmpty = values.map(v => v.trim()).filter(v => v.length > 0); if (nonEmpty.length === 0) return "string"; return nonEmpty.every(v => Number.isFinite(Number(v))) ? "number" : "string"; } /** Picks whichever candidate splits the header row into the most fields. */ export function detectDelimiter(headerLine: string): string { const candidates = ["\t", "|", ";", ","]; let best = candidates[0]; let bestCount = 1; for (const d of candidates) { const count = headerLine.split(d).length; if (count > bestCount) { bestCount = count; best = d; } } return best; } /** True for VizieR's "-------\t-------\t..." column-width separator row. */ export function isDashRow(line: string): boolean { return line.includes("-") && /^[-\s|;,\t]+$/.test(line); } /** Finds the "error column" for `name` (VizieR convention: e_). */ export function findErrCol(headers: string[], name: string): number | null { const eCol = headers.findIndex((v) => v === "e_" + name) if (eCol > 0) return eCol; return null } /** Maps header names to their column index in a row, keeping the first * occurrence when a name repeats. */ export function buildColumnIndex(headers: string[]): Map { const index = new Map(); headers.forEach((h, i) => { if (!index.has(h)) index.set(h, i); }); return index; } /** * Min/max for a column's raw cell values, computed once at parse time * over the whole file. Returns nulls for a non-numeric column, or a * numeric one with no parseable values. */ export function computeMinMax(values: string[], type: "number" | "string"): { min: number | null; max: number | null } { if (type !== "number") return { min: null, max: null }; let min = Infinity; let max = -Infinity; for (const v of values) { const n = toNumber(v); if (!Number.isFinite(n)) continue; if (n < min) min = n; if (n > max) max = n; } return Number.isFinite(min) ? { min, max } : { min: null, max: null }; } /// FITS /////////////////////////////////////////////////////////////////////////////////////////// /** * Minimal FITS (Flexible Image Transport System) reader * just enough to locate the table extension and hand its header cards + byte layout back * to the caller. * callers must use `file.arrayBuffer()` for `.fit` files). */ /** One parsed FITS header: its cards, and the byte offset (into the * whole file) where this HDU's data block begins. */ export type FitsHeader = { cards: Map; dataStart: number; }; /** FITS headers are pure ASCII, so bytes map 1:1 to char codes. */ export function bytesToAscii(bytes: Uint8Array): string { let s = ""; for (let i = 0; i < bytes.length; i++) s += String.fromCharCode(bytes[i]); return s; } /** Splits one 80-byte header card into its keyword, string or null value and comment. */ function parseFitsCard(card: string): { keyword: string; rawValue: string | null; comment: string } { const keyword = card.slice(0, 8).trim(); if (card.slice(8, 10) !== "= ") { return { keyword, rawValue: null, comment: card.slice(8).trim() }; } const rest = card.slice(10); let value: string; let comment = ""; if (rest.trimStart()[0] === "'") { // Quoted string value. A doubled '' inside the quotes is an escaped // literal quote, so we can't just jump to the next "'". const start = rest.indexOf("'"); let end = start + 1; let str = ""; while (end < rest.length) { if (rest[end] === "'") { if (rest[end + 1] === "'") { str += "'"; end += 2; continue; } break; } str += rest[end]; end++; } value = str.replace(/\s+$/, ""); // FITS right-pads string values with spaces const afterQuote = rest.slice(end + 1); const slashIdx = afterQuote.indexOf("/"); if (slashIdx >= 0) comment = afterQuote.slice(slashIdx + 1).trim(); } else { const slashIdx = rest.indexOf("/"); value = (slashIdx >= 0 ? rest.slice(0, slashIdx) : rest).trim(); if (slashIdx >= 0) comment = rest.slice(slashIdx + 1).trim(); } return { keyword, rawValue: value, comment }; } /** Reads one header starting at `offset`: every card up to (and * including) `END`, then rounds up to the next 2880-byte boundary so * `dataStart` lands exactly where this HDU's data block begins. */ export function readFitsHeader(bytes: Uint8Array, offset: number): FitsHeader { const cards = new Map(); let pos = offset; for (; ;) { if (pos + 80 > bytes.length) { throw new Error("Unexpected end of file while reading a FITS header."); } const { keyword, rawValue, comment } = parseFitsCard(bytesToAscii(bytes.subarray(pos, pos + 80))); pos += 80; if (keyword === "END") break; if (rawValue !== null) cards.set(keyword, { value: rawValue, comment }); } const headerBytes = pos - offset; return { cards, dataStart: offset + fitsPaddedLength(headerBytes) }; } /** Rounds a byte count up to the next multiple of 2880 (the FITS block size). */ export function fitsPaddedLength(nBytes: number): number { return Math.ceil(nBytes / 2880) * 2880; } /** Size in bytes of an HDU's data block, per the FITS standard: * `(|BITPIX| / 8) * GCOUNT * (PCOUNT + NAXIS1*NAXIS2*...*NAXISn)`. * For a table HDU this is just `NAXIS1 * NAXIS2` (row width * row count) * plus `PCOUNT` bytes of "heap" for any variable-length array columns. */ export function fitsDataLength(cards: FitsHeader["cards"]): number { const naxis = parseInt(cards.get("NAXIS")?.value ?? "0", 10); if (naxis === 0) return 0; // NAXIS=0 means "no data array at all", not a product of 1 let product = 1; for (let i = 1; i <= naxis; i++) { product *= parseInt(cards.get(`NAXIS${i}`)?.value ?? "0", 10); } const bitpix = Math.abs(parseInt(cards.get("BITPIX")?.value ?? "8", 10)); const pcount = parseInt(cards.get("PCOUNT")?.value ?? "0", 10); const gcount = parseInt(cards.get("GCOUNT")?.value ?? "1", 10); return (bitpix / 8) * gcount * (pcount + product); } /** * Parses a BINTABLE column's `TFORMn` code: an optional repeat count * followed by a single type letter ("1D", "20A", "J"). */ export function parseBinaryTForm(tform: string): { repeat: number; type: string } { const m = tform.trim().match(/^(\d*)([LXBIJKAEDCMPQ])/i); if (!m) throw new Error(`Unrecognized TFORM code: "${tform}"`); return { repeat: m[1] ? parseInt(m[1], 10) : 1, type: m[2].toUpperCase() }; } /** Byte size of a single element of a BINTABLE type letter. `X` (bit * arrays) doesn't have a fixed per-element byte size, so callers should * branch on `type === "X"` before calling this. */ export function binaryTypeSize(type: string): number { switch (type) { case "L": case "B": case "A": return 1; case "I": return 2; case "J": case "E": return 4; case "K": case "D": case "C": return 8; case "M": return 16; case "P": return 8; // array descriptor: one (int32 length, int32 heap offset) pair case "Q": return 16; // array descriptor: one (int64 length, int64 heap offset) pair default: throw new Error(`Unsupported TFORM type: "${type}"`); } } /** Total width in bytes of a column with a given repeat count and type. */ export function binaryColumnWidth(repeat: number, type: string): number { if (type === "X") return Math.ceil(repeat / 8); if (type === "P" || type === "Q") return binaryTypeSize(type); return repeat * binaryTypeSize(type); } //////////////////////////////////////////////////////////////////////////////////////////////////// /// VOTABLE //////////////////////////////////////////////////////////////////////////////////////// /** * Minimal VOTable (XML) reader: just enough to pull the columns and * rows out of what VizieR actually emits (one with a * flat list of s and a body). We use targeted * regexes instead of a real XML/DOM parser so this can run in a Worker * on files tens of megabytes large without needing a DOM implementation, * and doesn't choke on VizieR's habit of putting several FIELDs or an * entire data row on one line. */ /** One 's metadata, in the order it appears in the header - that * order is what lines a FIELD up with its column in the data rows. */ export type VotableField = { name: string; /** Raw VOTable datatype attribute (e.g. "double", "char"), or null * if the FIELD didn't declare one. */ datatype: string | null; /** Raw arraysize attribute (e.g. "28*", "3x4", "*"), or null for a * scalar field. Only matters for BINARY/BINARY2 streams: a * TABLEDATA cell is just text either way. */ arraysize: string | null; unit: string | null; description: string | null; /** The FIELD's `` sentinel, if it declared one. * Needed because XML has no way to leave e.g. an "int" cell empty, * so VizieR writes this exact string into the cell instead. */ nullValue: string | null; }; /** VOTable datatype names -> our two-value type system. */ export function classifyVotableType(datatype: string): "number" | "string" { switch (datatype) { case "short": case "int": case "long": case "float": case "double": case "unsignedByte": return "number"; default: // "char", "boolean", "unicodeChar", "floatComplex", "doubleComplex", ... return "string"; } } /** Un-escapes the handful of XML entities VizieR actually emits, plus * numeric character references. `&` is handled last so a literal * "<" written as "&lt;" doesn't get double-unescaped. */ export function decodeXmlEntities(text: string): string { return text .replace(/&#x([0-9a-fA-F]+);/g, (_, hex) => String.fromCodePoint(parseInt(hex, 16))) .replace(/&#(\d+);/g, (_, dec) => String.fromCodePoint(parseInt(dec, 10))) .replace(/</g, "<") .replace(/>/g, ">") .replace(/"/g, '"') .replace(/'/g, "'") .replace(/&/g, "&"); } /** Pulls one attribute's value out of an opening tag, e.g. `name="Plx"` * from ``. */ export function getXmlAttr(tag: string, attr: string): string | null { const m = tag.match(new RegExp(attr + '="([^"]*)"')); return m ? decodeXmlEntities(m[1]) : null; } /** * VizieR's "-tsv" output option and its "VOTable" output option both * end up as a `.tsv`-ish file: one is the plain delimited-text export * (`asu.tsv`, handled by `parseTSV`), the other is a full VOTable XML * document whose `` happens to be a `` block instead of the * default `` (`vizier_votable.tsv`, handled by `parseVOT`). * Extensions alone can't tell these apart, so callers sniff the first * non-blank bytes instead: real VizieR tsv exports start with a `#` * comment block, while every VOTable starts with an XML declaration * and/or a `...` element in the `
` header * (the slice of the file before ``). Assumes FIELDs are * always written with a closing tag, which holds for VizieR's exports * (a bare `` would just be skipped). */ export function parseVotableFields(headerXml: string): VotableField[] { const fields: VotableField[] = []; const fieldRe = /]*)>([\s\S]*?)<\/FIELD>/g; let m: RegExpExecArray | null; while ((m = fieldRe.exec(headerXml)) !== null) { const [, attrs, body] = m; const name = getXmlAttr(attrs, "name"); if (!name) continue; // malformed FIELD, nothing to key it by const descMatch = body.match(/([\s\S]*?)<\/DESCRIPTION>/); const nullMatch = body.match(/]*\bnull="([^"]*)"/); fields.push({ name, datatype: getXmlAttr(attrs, "datatype"), arraysize: getXmlAttr(attrs, "arraysize"), unit: getXmlAttr(attrs, "unit"), description: descMatch ? stripNullableMarker(decodeXmlEntities(descMatch[1].trim())) : null, nullValue: nullMatch ? decodeXmlEntities(nullMatch[1]) : null, }); } return fields; } /** * Splits one `...` row into its cell strings. VOTable cells * aren't quoted/escaped beyond standard XML entities, so this just * grabs the text between each `` pair (or "" for a * self-closed `
`/``). */ export function parseVotableRow(trContent: string): string[] { const cells: string[] = []; const cellRe = /]*)?(?:\/>|>([\s\S]*?)<\/TD>)/g; let m: RegExpExecArray | null; while ((m = cellRe.exec(trContent)) !== null) { cells.push(m[1] !== undefined ? decodeXmlEntities(m[1].trim()) : ""); } return cells; } /// VOTABLE BINARY/BINARY2 (VizieR's .b64/.b264 exports) ////////////////////////////////////////// /** * A FIELD's `arraysize` attribute (e.g. "28*", "3x4", "*"), parsed into * a fixed element count plus whether that count is actually variable. * A trailing "*" means the real per-row count is written as a 4-byte * big-endian integer right before the field's data in a BINARY/BINARY2 * stream - `fixedCount` in that case is just VizieR's declared upper * bound (or 0 for a bare "*"), not the real length, and callers must * read the real count from the stream instead of trusting it. * `null` (no arraysize at all) means a plain scalar: one element. */ export function parseVotableArraySize(arraysize: string | null): { fixedCount: number; variable: boolean } { if (arraysize === null) return { fixedCount: 1, variable: false }; const variable = arraysize.endsWith("*"); const dims = arraysize.replace(/\*$/, "").split("x").filter(d => d.length > 0); const fixedCount = dims.length === 0 ? 0 : dims.reduce((product, d) => product * parseInt(d, 10), 1); return { fixedCount, variable }; } /** Byte size of a single element of a VOTable BINARY/BINARY2 stream, * by `datatype`. `bit` and the complex types aren't handled since * VizieR doesn't emit them in its .b64/.b264 exports. */ export function votableElementSize(datatype: string): number { switch (datatype) { case "boolean": case "unsignedByte": case "char": return 1; case "short": case "unicodeChar": return 2; case "int": case "float": return 4; case "long": case "double": return 8; default: throw new Error(`Unsupported VOTable BINARY datatype: "${datatype}"`); } } /** Decodes a base64 string into raw bytes. `atob` follows the * "forgiving-base64" algorithm, which strips embedded ASCII whitespace * before decoding - exactly what's needed here, since VizieR wraps a * 's base64 content across many lines. */ export function base64ToBytes(b64: string): Uint8Array { const binary = atob(b64); const bytes = new Uint8Array(binary.length); for (let i = 0; i < binary.length; i++) bytes[i] = binary.charCodeAt(i); return bytes; } ////////////////////////////////////////////////////////////////////////////////////////////////////