import { CompileError } from '../diagnostic.ts'; import type { OffsetMap, OffsetRun, Span } from '../span.ts'; import type { Comment, Entry, Metadata, Resource, Section } from './data-model.ts'; // draft W3C grammar: https://github.com/w3c/i18n-discuss/blob/gh-pages/explainers/message-resource.abnf /** thrown when source violates the message resource grammar. */ export class ResourceParseError extends CompileError { constructor(message: string, span: Span) { super(message, { span }); this.name = 'ResourceParseError'; } } // #region lines interface Lines { lines: string[]; /** source offset of each line. */ starts: number[]; } const splitLines = (source: string): Lines => { const lines: string[] = []; const starts = [0]; let last = 0; for (const match of source.matchAll(/\r\n|\n/g)) { lines.push(source.slice(last, match.index)); last = match.index + match[0].length; starts.push(last); } lines.push(source.slice(last)); return { lines, starts }; }; const columnSpan = (src: Lines, i: number, from: number, to: number): Span => ({ end: src.starts[i] + to, start: src.starts[i] + from, }); const lineSpan = (src: Lines, i: number): Span => columnSpan(src, i, 0, src.lines[i].length); // #endregion // #region line classification const isBlank = (line: string): boolean => /^[ \t]*$/.test(line); const isFrontmatter = (line: string): boolean => /^---[ \t]*$/.test(line); const isSectionHead = (line: string): boolean => line.startsWith('['); const trimComment = (line: string): string => line .slice(1) .replace(/^[ \t]/, '') .replace(/[ \t]+$/, ''); // #endregion // #region escapes const HEX = /^[0-9a-fA-F]+$/; const fromHex = (raw: string, at: number, length: number, span: Span): string => { const hex = raw.slice(at + 2, at + 2 + length); if (hex.length !== length || !HEX.test(hex)) { throw new ResourceParseError(`invalid \\${raw[at + 1]} escape`, span); } const code = parseInt(hex, 16); if (code > 0x10ffff) { throw new ResourceParseError(`escape \\${raw[at + 1]} is out of range`, span); } return String.fromCodePoint(code); }; interface Decoded { offsets: OffsetMap; value: string; } // preserve MF2 escapes while decoding resource escapes. // `at[i]` is raw[i]'s source offset; `end` is the source offset after the raw text. const unescapeValue = (raw: string, at: readonly number[], end: number): Decoded => { let value = ''; const runs: OffsetRun[] = []; const copy = (i: number): void => { const run = runs.at(-1); if ( run === undefined || run.escapeEnd !== undefined || run.source + value.length - run.derived !== at[i] ) { runs.push({ derived: value.length, source: at[i] }); } value += raw[i]; }; const escapeSpan = (from: number, to: number): Span => ({ end: at[Math.min(to, raw.length - 1)] + 1, start: at[from], }); const decode = (text: string, from: number, to: number): void => { const { end: escapeEnd, start: source } = escapeSpan(from, to); runs.push({ derived: value.length, escapeEnd, source }); value += text; }; for (let i = 0; i < raw.length; i++) { if (raw[i] !== '\\') { copy(i); continue; } const next = raw[i + 1]; switch (next) { case '\\': case '{': case '|': case '}': { copy(i); copy(i + 1); i++; break; } case '\n': { i++; break; } case ' ': { decode(' ', i, i + 1); i++; break; } case 'n': { decode('\n', i, i + 1); i++; break; } case 'r': { decode('\r', i, i + 1); i++; break; } case 't': { decode('\t', i, i + 1); i++; break; } case 'x': { decode(fromHex(raw, i, 2, escapeSpan(i, i + 3)), i, i + 3); i += 3; break; } case 'u': { decode(fromHex(raw, i, 4, escapeSpan(i, i + 5)), i, i + 5); i += 5; break; } case 'U': { decode(fromHex(raw, i, 6, escapeSpan(i, i + 7)), i, i + 7); i += 7; break; } default: { throw new ResourceParseError(`invalid escape \\${next ?? ''}`, escapeSpan(i, i + 1)); } } } return { offsets: { end, length: value.length, runs }, value }; }; // #endregion // #region identifiers // split an identifier without treating escaped dots as separators. const parseId = (raw: string, span: Span): string[] => { const segments: string[] = []; let current = ''; for (let i = 0; i < raw.length; i++) { const c = raw[i]; if (c === '\\') { const next = raw[i + 1]; if (next === undefined) { throw new ResourceParseError('trailing backslash in identifier', span); } current += next; i++; continue; } if (c === '.') { segments.push(current.trim()); current = ''; continue; } current += c; } segments.push(current.trim()); return segments.map((segment) => { if (segment === '') { throw new ResourceParseError('empty identifier segment', span); } // canonically equivalent spellings name the same message. return segment.normalize('NFC'); }); }; /** * format a resource identifier, escaping delimiters and backslashes. * @param path the identifier's segments * @returns the dotted identifier */ export const formatId = (path: string[]): string => path.map((segment) => segment.replace(/[\\.=[\]]/g, '\\$&')).join('.'); // #endregion // #region values interface ParsedValue extends Decoded { next: number; } // join indented continuation lines before unescaping. const parseValue = (src: Lines, start: number, column: number): ParsedValue => { const parts = [{ offset: src.starts[start] + column, text: src.lines[start].slice(column) }]; let next = start + 1; for (; next < src.lines.length; next++) { const line = src.lines[next]; if (!/^[ \t]/.test(line) || isBlank(line)) { break; } const text = line.replace(/^[ \t]+/, ''); parts.push({ offset: src.starts[next] + line.length - text.length, text }); } // omit the empty segment before a value that starts on the next line. const segments = parts.length > 1 && parts[0].text === '' ? parts.slice(1) : parts; let raw = ''; const at: number[] = []; for (const [index, segment] of segments.entries()) { if (index > 0) { // a join maps to the line break it replaces. const previous = segments[index - 1]; raw += '\n'; at.push(previous.offset + previous.text.length); } raw += segment.text; for (let k = 0; k < segment.text.length; k++) { at.push(segment.offset + k); } } const last = segments[segments.length - 1]; return { next, ...unescapeValue(raw, at, last.offset + last.text.length) }; }; // #endregion // #region metadata interface ParsedMetadata { meta: Metadata; next: number; } const parseMetadata = (src: Lines, start: number): ParsedMetadata => { const line = src.lines[start]; const body = line.slice(1); const space = body.search(/[ \t]/); if (space === -1) { return { meta: { key: body, value: '' }, next: start + 1 }; } let column = 1 + space; while (column < line.length && (line[column] === ' ' || line[column] === '\t')) { column++; } const parsed = parseValue(src, start, column); return { meta: { key: body.slice(0, space), value: parsed.value }, next: parsed.next }; }; // #endregion // #region sections and entries const parseSectionHead = (src: Lines, i: number): string[] => { const match = /^(\[[ \t]*)(.*?)[ \t]*\][ \t]*$/.exec(src.lines[i]); if (match === null) { throw new ResourceParseError('invalid section header', lineSpan(src, i)); } const from = match[1].length; return parseId(match[2], columnSpan(src, i, from, from + match[2].length)); }; interface EntryHead { id: string[]; idSpan: Span; valueColumn: number; } const parseEntryHead = (src: Lines, i: number): EntryHead => { const line = src.lines[i]; let equals = 0; for (; equals < line.length; equals++) { const c = line[equals]; if (c === '\\') { equals++; continue; } if (c === '=') { break; } } if (equals >= line.length) { throw new ResourceParseError('expected `=` in entry', lineSpan(src, i)); } const raw = line.slice(0, equals); const id = raw.trim(); const from = raw.length - raw.trimStart().length; const idSpan = columnSpan(src, i, from, from + id.length); let valueColumn = equals + 1; while (valueColumn < line.length && (line[valueColumn] === ' ' || line[valueColumn] === '\t')) { valueColumn++; } return { id: parseId(id, idSpan), idSpan, valueColumn }; }; // #endregion // #region driver const emptySection = (): Section => ({ comment: '', entries: [], id: [], meta: [] }); const isMeaningful = (section: Section): boolean => section.id.length > 0 || section.entries.length > 0 || section.comment !== '' || section.meta.length > 0; /** * parse a message resource. * @param source the resource source * @returns the parsed resource with resource escapes decoded and MF2 escapes preserved * @throws ResourceParseError when the source violates the grammar, with a span relative to `source` */ export const parseResource = (source: string): Resource => { const src = splitLines(source); const { lines } = src; const resource: Resource = { comment: '', meta: [], sections: [] }; let start = 0; const frontmatter = lines.findIndex(isFrontmatter); if (frontmatter !== -1) { const comment: string[] = []; let i = 0; while (i < frontmatter) { const line = lines[i]; if (isBlank(line)) { i++; } else if (line.startsWith('#')) { comment.push(trimComment(line)); i++; } else if (line.startsWith('@')) { const parsed = parseMetadata(src, i); resource.meta.push(parsed.meta); i = parsed.next; } else { throw new ResourceParseError('unexpected content before frontmatter separator', lineSpan(src, i)); } } resource.comment = comment.join('\n'); start = frontmatter + 1; } let section = emptySection(); let pendingComment: string[] = []; // retain line numbers to locate orphaned metadata. let pendingMeta: { line: number; meta: Metadata }[] = []; const flushPending = (): void => { if (pendingMeta.length > 0) { throw new ResourceParseError( 'metadata must attach to an entry or section', lineSpan(src, pendingMeta[0].line), ); } if (pendingComment.length > 0) { const comment: Comment = { comment: pendingComment.join('\n'), type: 'comment' }; section.entries.push(comment); pendingComment = []; } }; for (let i = start; i < lines.length; i++) { const line = lines[i]; if (isBlank(line)) { flushPending(); continue; } if (line.startsWith('#')) { pendingComment.push(trimComment(line)); continue; } if (line.startsWith('@')) { const parsed = parseMetadata(src, i); pendingMeta.push({ line: i, meta: parsed.meta }); i = parsed.next - 1; continue; } if (isSectionHead(line)) { if (isMeaningful(section)) { resource.sections.push(section); } section = { comment: pendingComment.join('\n'), entries: [], id: parseSectionHead(src, i), meta: pendingMeta.map(({ meta }) => meta), }; pendingComment = []; pendingMeta = []; continue; } const head = parseEntryHead(src, i); const parsed = parseValue(src, i, head.valueColumn); const entry: Entry = { comment: pendingComment.join('\n'), id: head.id, idSpan: head.idSpan, meta: pendingMeta.map(({ meta }) => meta), type: 'entry', value: parsed.value, valueOffsets: parsed.offsets, }; section.entries.push(entry); pendingComment = []; pendingMeta = []; i = parsed.next - 1; } flushPending(); if (isMeaningful(section)) { resource.sections.push(section); } return resource; }; // #endregion