import { detectDelimiter, inferType, isDashRow, parseColumnComments, splitCSVLine } from "./utils"; /// DATA /////////////////////////////////////////////////////////////////////////////////////////// /** parse result extended with a column index and per-column metadata, * generated from the file. */ export type Data = { _mainData: string[][], _columnIndex: ColumnIndex, _columnMeta: Map, col: (name: string[]) => number | null, row: (index: number) => string[] | null, find: (header: string, value: string) => number | null, meta: (name: string[]) => ColumnMeta | null getStarRecors: (index: number) => StarData | null, }; export type FileType = ".csv" | ".tsv" export function buildData(raw: string, type: FileType): Data | null { const rows = raw.split("\n") const parsed = type === ".csv" ? parseCSV(rows) : parseTSV(rows); if (parsed === null) return null const data: Data = { _mainData: parsed.mainData, _columnIndex: parsed.columnIndex, _columnMeta: parsed.columnMeta, col: (names) => col(parsed.columnIndex, names), row: (index) => getRow(data, index), find: (header, value) => find(data, header, value), meta: (names) => meta(data, names), getStarRecors: (index) => getStarRecord(data, index) }; return data } //////////////////////////////////////////////////////////////////////////////////////////////////// /// PARSE ////////////////////////////////////////////////////////////////////////////////////////// /** Common shape both format-specific parsers produce.*/ type ParsedTable = { mainData: string[][]; columnIndex: ColumnIndex; columnMeta: Map; }; /** * Parses a comma-separated file, * Assumes: first non-blank row = headers, every row after that = data. */ function parseCSV(rawRows: string[]): ParsedTable | null { const lines = rawRows.filter(r => r.trim().length > 0); if (lines.length === 0) return null; const headers = splitCSVLine(lines[0]); const mainData = lines.slice(1).map(splitCSVLine); const columnIndex = buildColumnIndex(headers); const columnMeta = new Map(); headers.forEach((h, i) => { const values = mainData.map(row => row[i] ?? ""); columnMeta.set(h, { type: inferType(values), description: null, unit: null, }); }); return { mainData, columnIndex, columnMeta }; } /** * Parses VizieR tsv files. * Structure: * - a block of `#` with metadata lines. The `#Column` lines describe * every output column. * - a blank line * - header row * - units row * - a row of dashes * - the data rows. * * VizieR lets users pick a preferred field separator * so the delimiter is auto-detected from the header row. */ function parseTSV(rawRows: string[]): ParsedTable | null { const columnDocs = parseColumnComments(rawRows); const tableLines = rawRows.filter(r => !r.startsWith("#")); let i = 0; while (i < tableLines.length && tableLines[i].trim() === "") i++; if (i >= tableLines.length) return null; const delimiter = detectDelimiter(tableLines[i]); const headers = tableLines[i].split(delimiter).map(h => h.trim()); i++; const unitsLine = i < tableLines.length ? tableLines[i] : ""; const units = unitsLine.split(delimiter).map(u => u.trim()); i++; // row of dashes marking column widths -- present in every VizieR // export we've seen, but skip it defensively rather than assuming. if (i < tableLines.length && isDashRow(tableLines[i])) i++; const mainData: string[][] = []; for (; i < tableLines.length; i++) { if (tableLines[i].trim() === "") break; // end of table / next #RESOURCE block mainData.push(tableLines[i].split(delimiter).map(v => v.trim())); } const columnIndex = buildColumnIndex(headers); const columnMeta = new Map(); headers.forEach((h, idx) => { const doc = columnDocs.get(h); const unit = units[idx] && units[idx].trim() !== "" ? units[idx].trim() : null; const values = mainData.map(row => row[idx] ?? ""); columnMeta.set(h, { type: doc?.type ?? inferType(values), description: doc?.description ?? null, unit, }); }); return { mainData, columnIndex, columnMeta }; } //////////////////////////////////////////////////////////////////////////////////////////////////// /// COLUMN INDEX /////////////////////////////////////////////////////////////////////////////////// /** Maps normalized column names to their index in a CSV row */ export type ColumnIndex = Map; function buildColumnIndex(headers: string[]): ColumnIndex { const index: ColumnIndex = new Map(); headers.forEach((h, i) => { const key = h; if (!index.has(key)) index.set(key, i); }); return index; } function col(index: ColumnIndex, names: string[]): number | null { for (const name of names) { const i = index.get(name); if (i !== undefined) return i; } return null; } //////////////////////////////////////////////////////////////////////////////////////////////////// /// COLUMN METADATA //////////////////////////////////////////////////////////////////////////////// /** Everything we know about a single column of a loaded file. */ type ColumnMeta = { /** Value type, inferred from the file's own data */ type: "number" | "string"; /** From the source file when it carries one, else a prettified `name` */ description: string | null; /** e.g. "mas", "deg". `null` when the file didn't carry one */ unit: string | null; }; function meta(data: Data, names: string[]): ColumnMeta | null { for (const name of names) { const meta = data._columnMeta.get(name) if (meta) return meta } return null } //////////////////////////////////////////////////////////////////////////////////////////////////// /// GET STAR ///////////////////////////////////////////////////////////////////////////////////// function find(data: Data, header: string, value: string): number | null { const colIndex = data.col([header]); if (colIndex === null) return null; const target = value.trim(); for (let i = 0; i < data._mainData.length; i++) { if (data._mainData[i][colIndex] === target) return i; } return null; } function getRow(data: Data, index: number): string[] | null { const r = data._mainData[index]; return r !== undefined ? r : null; } //////////////////////////////////////////////////////////////////////////////////////////////////// /// STAR DATA ////////////////////////////////////////////////////////////////////////////////// /** A row's worth of the fields we usually care about, already typed and * with `null` for anything the source file didn't have. */ export type StarData = { Source: string | null; GMag: string | null; Plx: string | null; Ra: string | null; Dej: string | null; Pm: string | null; }; const FIELD_ALIASES: Record = { Source: ["source", "Source"], GMag: ["Gmag", "phot_g_mean_mag", "G"], Plx: ["Plx", "plx", "parallax"], Ra: ["RA_ICRS", "ra", "RAJ2000"], Dej: ["DE_ICRS", "dej", "DEJ2000"], Pm: ["PM", "pm"], }; function rawFieldValue(data: Data, row: string[], field: keyof StarData): string | null { const idx = data.col(FIELD_ALIASES[field]); if (idx === null) return null; const idxMeta = data.meta(FIELD_ALIASES[field]) const value = row[idx]; if (value === undefined) return null; if (value.trim() === "") return null; if (!idxMeta!.unit) return value; return value.trim() + " " + idxMeta!.unit } export function getStarRecord(data: Data, rowIndex: number): StarData | null { const row = data._mainData[rowIndex]; if (!row) return null; return { Source: rawFieldValue(data, row, "Source"), GMag: rawFieldValue(data, row, "GMag"), Plx: rawFieldValue(data, row, "Plx"), Ra: rawFieldValue(data, row, "Ra"), Dej: rawFieldValue(data, row, "Dej"), Pm: rawFieldValue(data, row, "Pm"), }; } ////////////////////////////////////////////////////////////////////////////////////////////////////