diff --git a/packages/rdf/.npmignore b/packages/rdf/.npmignore new file mode 100644 index 0000000..db25063 --- /dev/null +++ b/packages/rdf/.npmignore @@ -0,0 +1,4 @@ +*_test.ts +*_bench.ts +_memory_test.ts +*.map diff --git a/packages/rdf/README.md b/packages/rdf/README.md new file mode 100644 index 0000000..f38d763 --- /dev/null +++ b/packages/rdf/README.md @@ -0,0 +1,68 @@ +# `@okikio/rdf` + +RDF programming model for TypeScript runtimes. + +```ts +import * as rdf from '@okikio/rdf' + +const schema = rdf.namespace('https://schema.org/') +const product = rdf.namedNode('https://example.com/product/1') +const data = rdf.dataset([ + rdf.quad(product, schema('name'), rdf.literal('Widget')), +]) +``` + +## Core model + +The root owns RDF 1.2 terms, RDF/JS interoperability, Dataset indexes, namespaces, source contracts, and shared serialization primitives. RDF 1.2 triple terms are represented as default-graph quads when embedded as objects. + +The native streaming contracts use JavaScript/Web primitives: + +```text +Iterable +AsyncIterable +ReadableStream +AbortSignal +``` + +## Formats and semantics + +Use explicit subpaths: + +```ts +import * as nquads from '@okikio/rdf/nquads' +import * as turtle from '@okikio/rdf/turtle' +import * as jsonld from '@okikio/rdf/jsonld' +import * as ontology from '@okikio/rdf/ontology' +import * as shape from '@okikio/rdf/shape' +``` + +Available public subpaths include: + +```text +ntriples +nquads +turtle +trig +jsonld +xml +rdfa +microdata +canon +ontology +shape +``` + +The root module does not import or initialize the focused third-party processors used by JSON-LD, RDFC-1.0, RDF/XML, RDFa, or Microdata. The npm package manifest is package-scoped, however, so installing `@okikio/rdf` currently installs those dependencies. + +## Parser lifecycle + +Project-owned streaming parsers use bounded source windows and cancel pending Web Stream reads when an operation stops. Tolerant parsing withholds statement-local semantic output until the current statement is known to be valid, so recovery does not leak partial RDF. + +## Ontologies and shapes + +`@okikio/rdf/ontology` interprets generic named RDFS/OWL relationships while retaining unsupported assertions. + +`@okikio/rdf/shape` reads loss-preserving SHACL shape structure. Ontology domain/range semantics are not treated as closed-world JSON requiredness. + +See the repository architecture and testing guides for the current standards/version posture and release gates. diff --git a/packages/rdf/canon/mod.ts b/packages/rdf/canon/mod.ts new file mode 100644 index 0000000..51a9edf --- /dev/null +++ b/packages/rdf/canon/mod.ts @@ -0,0 +1,174 @@ +/** RDFC-1.0 RDF dataset canonicalization with explicit complexity controls. @module */ + +import { parse as parseNQuads, write as writeNQuads } from '../nquads/mod.ts' +import type { Literal, ObjectTerm, Quad } from '../term.ts' +import type { CanonizerOptionsType, CanonizerType } from './types.ts' + +export type { CanonizerOptionsType, CanonizerType } from './types.ts' + +/** Default max quads used when the caller does not provide an override. */ +const DEFAULT_MAX_QUADS = 1_000_000 +/** Default max work factor used when the caller does not provide an override. */ +const DEFAULT_MAX_WORK_FACTOR = 1 + +/** RDFC-1.0 canonicalization options. */ +export interface OptionsType { + /** Optional implementation injection for tests or alternate conforming RDFC-1.0 engines. */ + readonly canonizer?: CanonizerType + /** Hash algorithm used internally by RDFC-1.0. */ + readonly messageDigestAlgorithm?: 'sha256' | 'sha384' | 'sha512' + /** Complexity limit passed to the deep blank-node comparison algorithm. Defaults to 1, or O(n). */ + readonly maxWorkFactor?: number + /** Exact deep-iteration limit. When supplied, this overrides `maxWorkFactor`. */ + readonly maxDeepIterations?: number + /** Maximum number of input quads materialized for one canonicalization. */ + readonly maxQuads?: number + /** Cooperative cancellation checked by this facade and the canonicalizer. */ + readonly signal?: AbortSignal +} + +/** Options for hashing the resulting canonical N-Quads document. */ +export interface HashOptionsType extends OptionsType { + /** Digest applied to the final canonical N-Quads bytes. This is separate from RDFC's internal hash. */ + readonly digest?: 'SHA-256' | 'SHA-384' | 'SHA-512' +} + +/** + * Produces the canonical N-Quads representation defined by RDFC-1.0. + * + * RDFC-1.0 is defined over the RDF 1.1 dataset model. RDF 1.2 triple terms and + * directional language-tagged strings are therefore rejected instead of being + * silently lowered to a representation whose canonicalization is unspecified. + */ +export async function canonicalize( + source: Iterable | AsyncIterable, + options: OptionsType = {}, +): Promise { + abort(options.signal) + const quads = await collect(source, options.maxQuads ?? DEFAULT_MAX_QUADS, options.signal) + for (const value of quads) validate(value) + + const canonizer = options.canonizer ?? await defaultCanonizer() + const settings = settingsFor(options) + const output = await canonizer.canonize(writeNQuads(quads), settings) + abort(options.signal) + return output +} + +/** Canonicalizes a dataset and parses the canonical N-Quads back into native RDF terms. */ +export async function canonicalizeQuads( + source: Iterable | AsyncIterable, + options: OptionsType = {}, +): Promise { + const text = await canonicalize(source, options) + const quads: Quad[] = [] + const parseOptions = options.signal ? { signal: options.signal } : {} + for await (const value of parseNQuads(text, parseOptions)) quads.push(value) + return quads +} + +/** Hashes the canonical N-Quads bytes with one explicit Web Crypto digest. */ +export async function hash( + source: Iterable | AsyncIterable, + options: HashOptionsType = {}, +): Promise { + const text = await canonicalize(source, options) + abort(options.signal) + const bytes = new TextEncoder().encode(text) + const digest = await crypto.subtle.digest(options.digest ?? 'SHA-256', bytes) + abort(options.signal) + return hex(new Uint8Array(digest)) +} + +/** Returns true when two datasets canonicalize to the same RDFC-1.0 representation. */ +export async function isomorphic( + left: Iterable | AsyncIterable, + right: Iterable | AsyncIterable, + options: OptionsType = {}, +): Promise { + const leftValue = await canonicalize(left, options) + const rightValue = await canonicalize(right, options) + return leftValue === rightValue +} + +/** Creates the exact option object passed to the external implementation. */ +function settingsFor(options: OptionsType): CanonizerOptionsType { + const maxWorkFactor = nonNegativeFiniteOrInfinity(options.maxWorkFactor ?? DEFAULT_MAX_WORK_FACTOR, 'maxWorkFactor') + const settings: CanonizerOptionsType = { + algorithm: 'RDFC-1.0', + inputFormat: 'application/n-quads', + format: 'application/n-quads', + messageDigestAlgorithm: options.messageDigestAlgorithm ?? 'sha256', + maxWorkFactor, + rejectURDNA2015: true, + ...(options.maxDeepIterations === undefined + ? {} + : { maxDeepIterations: nonNegativeFiniteOrInfinity(options.maxDeepIterations, 'maxDeepIterations') }), + ...(options.signal ? { signal: options.signal } : {}), + } + return settings +} + +/** Materializes one canonicalization input with an explicit cardinality limit. */ +async function collect( + source: Iterable | AsyncIterable, + maxQuads: number, + signal?: AbortSignal, +): Promise { + if (!Number.isSafeInteger(maxQuads) || maxQuads <= 0) throw new RangeError('maxQuads must be a positive safe integer.') + const quads: Quad[] = [] + for await (const value of source) { + abort(signal) + if (quads.length >= maxQuads) throw new RangeError(`RDFC-1.0 input exceeds maxQuads (${maxQuads}).`) + quads.push(value) + } + return quads +} + +/** Rejects RDF 1.2 terms that RDFC-1.0 does not currently define. */ +function validate(value: Quad): void { + validateObject(value.object) +} + +/** Validates one graph object for RDFC-1.0 compatibility. */ +function validateObject(value: ObjectTerm): void { + if (value.termType === 'Quad') { + throw new TypeError('RDFC-1.0 does not define canonicalization for RDF 1.2 triple terms.') + } + if (value.termType === 'Literal') validateLiteral(value) +} + +/** Rejects RDF 1.2 directional language-tagged strings from RDFC-1.0 input. */ +function validateLiteral(value: Literal): void { + if (value.direction !== '') { + throw new TypeError('RDFC-1.0 does not define canonicalization for RDF 1.2 directional language-tagged strings.') + } +} + +/** Lazily resolved RDFC-1.0 processor shared across calls after the caller first requests canonicalization. */ +let canonizerPromise: Promise | undefined + +/** Lazily imports the canonicalizer only when this focused subpath performs work. */ +async function defaultCanonizer(): Promise { + canonizerPromise ??= import('rdf-canonize').then((module) => module as unknown as CanonizerType) + return await canonizerPromise +} + +/** Formats digest bytes as lowercase hexadecimal. */ +function hex(bytes: Uint8Array): string { + let result = '' + for (const value of bytes) result += value.toString(16).padStart(2, '0') + return result +} + +/** Validates canonicalization work limits while allowing Infinity only where the upstream contract permits it. */ +function nonNegativeFiniteOrInfinity(value: number, name: string): number { + if (value === Infinity) return value + if (!Number.isSafeInteger(value) || value < 0) throw new RangeError(`${name} must be a non-negative safe integer or Infinity.`) + return value +} + +/** Throws the caller supplied abort reason when cancellation has been requested. */ +function abort(signal?: AbortSignal): void { + if (signal?.aborted) throw signal.reason ?? new DOMException('Aborted', 'AbortError') +} diff --git a/packages/rdf/canon/mod_test.ts b/packages/rdf/canon/mod_test.ts new file mode 100644 index 0000000..2a324e8 --- /dev/null +++ b/packages/rdf/canon/mod_test.ts @@ -0,0 +1,38 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { blankNode, literal, namedNode, quad } from '../mod.ts' +import { canonicalize, canonicalizeQuads, hash, isomorphic, type CanonizerType } from './mod.ts' + +const canonical = ' "value" .\n' +const canonizer: CanonizerType = { + async canonize(_input, options) { + expect(options.algorithm).toBe('RDFC-1.0') + expect(options.rejectURDNA2015).toBe(true) + return canonical + }, +} + +describe('@okikio/rdf/canon', () => { + it('owns the RDFC-1.0 option contract and native quad conversion', async () => { + const input = [quad(blankNode('input'), namedNode('https://example.test/p'), literal('value'))] + expect(await canonicalize(input, { canonizer })).toBe(canonical) + const values = await canonicalizeQuads(input, { canonizer }) + expect(values).toHaveLength(1) + expect(values[0]?.subject.value).toBe('https://example.test/s') + }) + + it('rejects RDF 1.2 terms that RDFC-1.0 does not define', async () => { + const directional = quad( + namedNode('https://example.test/s'), + namedNode('https://example.test/p'), + literal('bonjour', { language: 'fr', direction: 'ltr' }), + ) + await expect(canonicalize([directional], { canonizer })).rejects.toThrow('directional') + }) + + it('hashes canonical bytes and compares canonical representations', async () => { + const input = [quad(namedNode('https://example.test/a'), namedNode('https://example.test/p'), literal('x'))] + expect(/^[0-9a-f]{64}$/u.test(await hash(input, { canonizer }))).toBe(true) + expect(await isomorphic(input, input, { canonizer })).toBe(true) + }) +}) diff --git a/packages/rdf/canon/types.ts b/packages/rdf/canon/types.ts new file mode 100644 index 0000000..4cf4b92 --- /dev/null +++ b/packages/rdf/canon/types.ts @@ -0,0 +1,18 @@ +/** Structural contract implemented by an RDF dataset canonicalizer. @module */ + +/** Options forwarded to the RDFC-1.0 implementation. */ +export interface CanonizerOptionsType { + readonly algorithm: 'RDFC-1.0' + readonly inputFormat: 'application/n-quads' + readonly format: 'application/n-quads' + readonly messageDigestAlgorithm: 'sha256' | 'sha384' | 'sha512' + readonly maxWorkFactor: number + readonly maxDeepIterations?: number + readonly signal?: AbortSignal + readonly rejectURDNA2015: true +} + +/** Minimal external canonicalizer contract used by this subpath. */ +export interface CanonizerType { + canonize(input: string, options: CanonizerOptionsType): Promise +} diff --git a/packages/rdf/compact.ts b/packages/rdf/compact.ts new file mode 100644 index 0000000..4e89b4b --- /dev/null +++ b/packages/rdf/compact.ts @@ -0,0 +1,1383 @@ +/** Shared streaming RDF 1.2 Turtle/TriG scanner and semantic parser. @module */ + +import { blankNode, defaultGraph, literal, namedNode, quad, triple } from './factory.ts' +import { chunks, throwIfAborted, type TextSource } from './text.ts' +import { RDF, XSD, type Graph, type Literal, type NamedNode, type ObjectTerm, type Predicate, type Quad, type Subject } from './term.ts' + +/** Source range expressed in UTF-16 code-unit offsets and one-based line/column positions. */ +export interface CompactRange { + readonly start: number + readonly end: number + readonly line: number + readonly column: number +} + +/** Recoverable Turtle/TriG diagnostic. */ +export interface CompactDiagnostic { + readonly code: string + readonly message: string + readonly range: CompactRange +} + +/** RDF version labels accepted by RDF 1.2 Turtle and TriG. */ +export type CompactVersion = '1.1' | '1.2-basic' | '1.2' + +/** Streaming parser controls for Turtle and TriG. */ +export interface CompactOptions { + /** Retrieval/base IRI used before an in-document BASE directive appears. */ + readonly baseIri?: string + /** Emit diagnostics and resume at the next statement where safe. */ + readonly tolerant?: boolean + /** Maximum decoded lexical token length. */ + readonly maxTokenLength?: number + /** Maximum nested collection/property-list/triple-term depth. */ + readonly maxDepth?: number + /** Maximum semantic events buffered for one invalidatable statement in tolerant mode. */ + readonly maxStatementEvents?: number + readonly signal?: AbortSignal +} + +/** Parser event common to Turtle and TriG. */ +export type CompactEvent = + | { readonly kind: 'quad'; readonly quad: Quad; readonly range: CompactRange } + | { readonly kind: 'prefix'; readonly prefix: string; readonly iri: string; readonly range: CompactRange } + | { readonly kind: 'base'; readonly iri: string; readonly range: CompactRange } + | { readonly kind: 'version'; readonly version: CompactVersion; readonly range: CompactRange } + | { readonly kind: 'diagnostic'; readonly diagnostic: CompactDiagnostic } + +/** Default max token length used when the caller does not provide an override. */ +const DEFAULT_MAX_TOKEN_LENGTH = 8 * 1024 * 1024 +/** Default max depth used when the caller does not provide an override. */ +const DEFAULT_MAX_DEPTH = 128 +/** Default max statement events used when the caller does not provide an override. */ +const DEFAULT_MAX_STATEMENT_EVENTS = 1_000_000 +/** Consumed scanner bytes required before slicing the retained source buffer to cap memory. */ +const COMPACT_THRESHOLD = 64 * 1024 + +/** Transient scanner token kinds. Tokens are fields on one scanner object, not allocated AST nodes. */ +const Kind = { + Eof: 0, + Iri: 1, + PName: 2, + Blank: 3, + String: 4, + Number: 5, + Lang: 6, + True: 7, + False: 8, + A: 9, + Prefix: 10, + Base: 11, + Version: 12, + Graph: 13, + Dot: 14, + Semicolon: 15, + Comma: 16, + LBracket: 17, + RBracket: 18, + LParen: 19, + RParen: 20, + LBrace: 21, + RBrace: 22, + HatHat: 23, + Tilde: 24, + TripleStart: 25, + TripleEnd: 26, + ReifiedStart: 27, + ReifiedEnd: 28, + AnnotationStart: 29, + AnnotationEnd: 30, + Unknown: 31, +} as const + +/** Numeric compact-syntax token kind used only inside the allocation-light scanner/parser state machine. */ +type Kind = (typeof Kind)[keyof typeof Kind] + +/** Position-aware parser failure used by strict mode and converted to diagnostics in tolerant mode. */ +class CompactError extends SyntaxError { + readonly code: string + readonly range: CompactRange + + /** Creates a position-aware Turtle/TriG syntax failure that tolerant mode can convert to a diagnostic. */ + constructor(code: string, message: string, range: CompactRange) { + super(message) + this.name = 'RdfCompactParseError' + this.code = code + this.range = range + } +} + +/** + * Incremental UTF-8 lexical scanner. + * + * The scanner keeps one mutable token record (`kind`, `value`, `raw`, range) + * and compacts consumed source. This avoids a token-array/AST allocation layer + * while still giving the semantic parser one-token lookahead. + */ +class Scanner { + readonly signal: AbortSignal | undefined + readonly maxTokenLength: number + + kind: Kind = Kind.Eof + value = '' + raw = '' + start = 0 + end = 0 + line = 1 + column = 1 + + #source: AsyncGenerator + #decoder = new TextDecoder('utf-8', { fatal: true }) + #buffer = '' + #index = 0 + #absolute = 0 + #line = 1 + #column = 1 + #done = false + + /** Creates one incremental scanner over bounded source chunks without materializing a token array. */ + constructor(source: TextSource, options: CompactOptions) { + this.signal = options.signal + this.maxTokenLength = options.maxTokenLength ?? DEFAULT_MAX_TOKEN_LENGTH + this.#source = chunks(source, options.signal) + } + + /** Advances to the next significant token. */ + async next(): Promise { + throwIfAborted(this.signal) + await this.#skipSpace() + + this.value = '' + this.raw = '' + this.start = this.#absolute + this.line = this.#line + this.column = this.#column + + const first = await this.#peek() + if (first === undefined) { + this.kind = Kind.Eof + this.end = this.#absolute + return + } + + const three = `${first}${await this.#peek(1) ?? ''}${await this.#peek(2) ?? ''}` + const two = three.slice(0, 2) + + if (three === '<<(') return await this.#punct(Kind.TripleStart, 3) + if (three === ')>>') return await this.#punct(Kind.TripleEnd, 3) + if (two === '<<') return await this.#punct(Kind.ReifiedStart, 2) + if (two === '>>') return await this.#punct(Kind.ReifiedEnd, 2) + if (two === '{|') return await this.#punct(Kind.AnnotationStart, 2) + if (two === '|}') return await this.#punct(Kind.AnnotationEnd, 2) + if (two === '^^') return await this.#punct(Kind.HatHat, 2) + + switch (first) { + case '.': { + const next = await this.#peek(1) + if (next !== undefined && /[0-9]/.test(next)) return await this.#number() + return await this.#punct(Kind.Dot, 1) + } + case ';': return await this.#punct(Kind.Semicolon, 1) + case ',': return await this.#punct(Kind.Comma, 1) + case '[': return await this.#punct(Kind.LBracket, 1) + case ']': return await this.#punct(Kind.RBracket, 1) + case '(': return await this.#punct(Kind.LParen, 1) + case ')': return await this.#punct(Kind.RParen, 1) + case '{': return await this.#punct(Kind.LBrace, 1) + case '}': return await this.#punct(Kind.RBrace, 1) + case '~': return await this.#punct(Kind.Tilde, 1) + case '<': return await this.#iri() + case '"': + case "'": return await this.#string(first) + case '@': return await this.#at() + case ':': return await this.#pname() + case '+': + case '-': return await this.#numberOrUnknown() + default: + if (/[0-9]/.test(first)) return await this.#number() + if (two === '_:') return await this.#blank() + if (isNameStart(first)) return await this.#wordOrPname() + await this.#take() + this.kind = Kind.Unknown + this.raw = first + this.value = first + this.end = this.#absolute + } + } + + /** Returns a range covering the current scanner token. */ + range(): CompactRange { + return { start: this.start, end: this.end, line: this.line, column: this.column } + } + + /** Constructs a parser error at the current token. */ + error(code: string, message: string): CompactError { + return new CompactError(code, message, this.range()) + } + + /** Skips space in the current parser or scanner state. */ + async #skipSpace(): Promise { + while (true) { + const char = await this.#peek() + if (char === undefined) return + if (isWhitespace(char)) { + await this.#take() + continue + } + if (char === '#') { + while (true) { + const item = await this.#peek() + if (item === undefined || item === '\n' || item === '\r') break + await this.#take() + } + continue + } + return + } + } + + /** Punct as one isolated step of the Scanner state machine. */ + async #punct(kind: Kind, width: number): Promise { + let raw = '' + for (let i = 0; i < width; i++) raw += await this.#take() ?? '' + this.kind = kind + this.raw = raw + this.value = raw + this.end = this.#absolute + } + + /** Iri as one isolated step of the Scanner state machine. */ + async #iri(): Promise { + const mark = this.#absolute + await this.#take() + let value = '' + let raw = '<' + while (true) { + const char = await this.#peek() + if (char === undefined) throw this.error('turtle-iri-end', 'Unterminated IRI reference.') + if (char === '>') { + raw += await this.#take() + this.#guard(mark) + this.kind = Kind.Iri + this.value = value + this.raw = raw + this.end = this.#absolute + return + } + if (char === '\\') { + raw += await this.#take() + const escape = await this.#unicodeEscape() + raw += escape.raw + value += escape.value + continue + } + if (char <= ' ' || /[<>"{}|^`]/.test(char)) { + throw this.error('turtle-iri-char', 'IRI reference contains a forbidden character.') + } + raw += await this.#take() + value += char + this.#guard(mark) + } + } + + /** String as one isolated step of the Scanner state machine. */ + async #string(quote: string): Promise { + const mark = this.#absolute + const long = await this.#peek(1) === quote && await this.#peek(2) === quote + const width = long ? 3 : 1 + let raw = '' + for (let i = 0; i < width; i++) raw += await this.#take() ?? '' + let value = '' + + while (true) { + const char = await this.#peek() + if (char === undefined) throw this.error('turtle-string-end', 'Unterminated Turtle string literal.') + if (char === quote) { + if (long) { + if (await this.#peek(1) === quote && await this.#peek(2) === quote) { + for (let i = 0; i < 3; i++) raw += await this.#take() ?? '' + break + } + } else { + raw += await this.#take() + break + } + } + if (!long && (char === '\n' || char === '\r')) { + throw this.error('turtle-string-line', 'Short Turtle string literals cannot contain line breaks.') + } + if (char === '\\') { + raw += await this.#take() + const next = await this.#peek() + if (next === 'u' || next === 'U') { + const escape = await this.#unicodeEscape() + raw += escape.raw + value += escape.value + continue + } + if (next === undefined || !'tbnrf"\'\\'.includes(next)) { + throw this.error('turtle-string-escape', 'Invalid Turtle string escape.') + } + raw += await this.#take() + value += escapeValue(next) + continue + } + raw += await this.#take() + value += char + this.#guard(mark) + } + + this.#guard(mark) + this.kind = Kind.String + this.value = value + this.raw = raw + this.end = this.#absolute + } + + /** At as one isolated step of the Scanner state machine. */ + async #at(): Promise { + const mark = this.#absolute + let raw = await this.#take() ?? '' + while (true) { + const char = await this.#peek() + if (char === undefined || !/[A-Za-z0-9-]/.test(char)) break + raw += await this.#take() + this.#guard(mark) + } + + if (raw === '@prefix') this.kind = Kind.Prefix + else if (raw === '@base') this.kind = Kind.Base + else if (raw === '@version') this.kind = Kind.Version + else this.kind = Kind.Lang + this.raw = raw + this.value = raw.slice(1) + this.end = this.#absolute + } + + /** Blank as one isolated step of the Scanner state machine. */ + async #blank(): Promise { + const mark = this.#absolute + let raw = `${await this.#take() ?? ''}${await this.#take() ?? ''}` + const first = await this.#peek() + if (first === undefined || !isBlankStart(first)) throw this.error('turtle-blank', 'Invalid blank-node label.') + while (true) { + const char = await this.#peek() + if (char === undefined || !isBlankChar(char)) break + raw += await this.#take() + this.#guard(mark) + } + if (raw.endsWith('.')) { + this.#rewindOne('.') + raw = raw.slice(0, -1) + } + this.kind = Kind.Blank + this.raw = raw + this.value = raw.slice(2) + this.end = this.#absolute + } + + /** Number or unknown as one isolated step of the Scanner state machine. */ + async #numberOrUnknown(): Promise { + const next = await this.#peek(1) + const after = await this.#peek(2) + if (next !== undefined && (/[0-9]/.test(next) || (next === '.' && after !== undefined && /[0-9]/.test(after)))) { + return await this.#number() + } + const first = await this.#take() ?? '' + this.kind = Kind.Unknown + this.raw = first + this.value = first + this.end = this.#absolute + } + + /** Number as one isolated step of the Scanner state machine. */ + async #number(): Promise { + const mark = this.#absolute + let raw = '' + let char = await this.#peek() + if (char === '+' || char === '-') { + raw += await this.#take() + char = await this.#peek() + } + + while (char !== undefined && /[0-9]/.test(char)) { + raw += await this.#take() + char = await this.#peek() + this.#guard(mark) + } + if (char === '.') { + const next = await this.#peek(1) + if (next !== undefined && (/[0-9]/.test(next) || next === 'e' || next === 'E')) { + raw += await this.#take() + char = await this.#peek() + while (char !== undefined && /[0-9]/.test(char)) { + raw += await this.#take() + char = await this.#peek() + this.#guard(mark) + } + } + } + if (char === 'e' || char === 'E') { + raw += await this.#take() + char = await this.#peek() + if (char === '+' || char === '-') { + raw += await this.#take() + char = await this.#peek() + } + if (char === undefined || !/[0-9]/.test(char)) throw this.error('turtle-number', 'Exponent requires at least one digit.') + while (char !== undefined && /[0-9]/.test(char)) { + raw += await this.#take() + char = await this.#peek() + this.#guard(mark) + } + } + + if (!numericKind(raw)) throw this.error('turtle-number', `Invalid numeric literal '${raw}'.`) + this.kind = Kind.Number + this.raw = raw + this.value = raw + this.end = this.#absolute + } + + /** Word or pname as one isolated step of the Scanner state machine. */ + async #wordOrPname(): Promise { + const mark = this.#absolute + let raw = '' + while (true) { + const char = await this.#peek() + if (char === undefined || !isPrefixChar(char)) break + raw += await this.#take() + this.#guard(mark) + } + + if (await this.#peek() === ':') { + raw += await this.#take() + while (true) { + const char = await this.#peek() + if (char === undefined) break + if (char === '\\') { + raw += await this.#take() + const escaped = await this.#peek() + if (escaped === undefined || !isLocalEscape(escaped)) throw this.error('turtle-pname-escape', 'Invalid prefixed-name escape.') + raw += await this.#take() + continue + } + if (char === '%') { + const a = await this.#peek(1) + const b = await this.#peek(2) + if (a !== undefined && b !== undefined && /[0-9A-Fa-f]/.test(a) && /[0-9A-Fa-f]/.test(b)) { + raw += `${await this.#take()}${await this.#take()}${await this.#take()}` + continue + } + break + } + if (!isLocalChar(char)) break + raw += await this.#take() + this.#guard(mark) + } + while (raw.endsWith('.')) { + this.#rewindOne('.') + raw = raw.slice(0, -1) + } + this.kind = Kind.PName + this.raw = raw + this.value = raw + this.end = this.#absolute + return + } + + const upper = raw.toUpperCase() + if (raw === 'a') this.kind = Kind.A + else if (raw === 'true') this.kind = Kind.True + else if (raw === 'false') this.kind = Kind.False + else if (upper === 'PREFIX') this.kind = Kind.Prefix + else if (upper === 'BASE') this.kind = Kind.Base + else if (upper === 'VERSION') this.kind = Kind.Version + else if (upper === 'GRAPH') this.kind = Kind.Graph + else this.kind = Kind.Unknown + this.raw = raw + this.value = raw + this.end = this.#absolute + } + + /** Pname as one isolated step of the Scanner state machine. */ + async #pname(): Promise { + const mark = this.#absolute + let raw = await this.#take() ?? '' + while (true) { + const char = await this.#peek() + if (char === undefined) break + if (char === '\\') { + raw += await this.#take() + const escaped = await this.#peek() + if (escaped === undefined || !isLocalEscape(escaped)) throw this.error('turtle-pname-escape', 'Invalid prefixed-name escape.') + raw += await this.#take() + continue + } + if (char === '%') { + const a = await this.#peek(1) + const b = await this.#peek(2) + if (a !== undefined && b !== undefined && /[0-9A-Fa-f]/.test(a) && /[0-9A-Fa-f]/.test(b)) { + raw += `${await this.#take()}${await this.#take()}${await this.#take()}` + continue + } + break + } + if (!isLocalChar(char)) break + raw += await this.#take() + this.#guard(mark) + } + while (raw.endsWith('.')) { + this.#rewindOne('.') + raw = raw.slice(0, -1) + } + this.kind = Kind.PName + this.raw = raw + this.value = raw + this.end = this.#absolute + } + + /** Unicode escape as one isolated step of the Scanner state machine. */ + async #unicodeEscape(): Promise<{ readonly raw: string; readonly value: string }> { + const kind = await this.#take() + if (kind !== 'u' && kind !== 'U') throw this.error('turtle-unicode', 'Expected Unicode escape.') + const width = kind === 'u' ? 4 : 8 + let hex = '' + for (let i = 0; i < width; i++) { + const char = await this.#take() + if (char === undefined || !/[0-9A-Fa-f]/.test(char)) throw this.error('turtle-unicode', 'Invalid Unicode escape.') + hex += char + } + const point = Number.parseInt(hex, 16) + if (point > 0x10ffff || (point >= 0xd800 && point <= 0xdfff)) { + throw this.error('turtle-unicode', 'Unicode escape is not a Unicode scalar value.') + } + return { raw: `${kind}${hex}`, value: String.fromCodePoint(point) } + } + + /** Reads the next buffered source value without consuming it. */ + async #peek(offset = 0): Promise { + await this.#fill(offset + 1) + return this.#buffer[this.#index + offset] + } + + /** Consumes and returns the next buffered source value. */ + async #take(): Promise { + await this.#fill(1) + const char = this.#buffer[this.#index] + if (char === undefined) return undefined + this.#index++ + this.#absolute++ + if (char === '\n') { + this.#line++ + this.#column = 1 + } else { + this.#column++ + } + if (this.#index >= COMPACT_THRESHOLD) this.#compact() + return char + } + + /** Refills the buffered source window only when the current window is exhausted. */ + async #fill(required: number): Promise { + while (!this.#done && this.#buffer.length - this.#index < required) { + const item = await this.#source.next() + if (item.done) { + this.#buffer += this.#decoder.decode() + this.#done = true + return + } + const chunk = item.value + this.#buffer += typeof chunk === 'string' ? chunk : this.#decoder.decode(chunk, { stream: true }) + } + } + + /** Compacts consumed source data while preserving every unread token byte. */ + #compact(): void { + this.#buffer = this.#buffer.slice(this.#index) + this.#index = 0 + } + + /** Rewinds the scanner by the one token position required by grammar lookahead. */ + #rewindOne(expected: string): void { + // Rewind is only used for a trailing dot in names/blank labels. Such a dot + // cannot be a newline, so line tracking remains unchanged. + if (this.#index === 0 || this.#buffer[this.#index - 1] !== expected) { + throw new Error('Scanner rewind invariant failed.') + } + this.#index-- + this.#absolute-- + this.#column-- + } + + /** Releases the underlying chunk iterator so Web Streams cancel on early parser return. */ + async close(): Promise { + await this.#source.return(undefined) + } + + /** Applies configured parser resource limits before accepting more input. */ + #guard(start: number): void { + if (this.#absolute - start > this.maxTokenLength) { + throw this.error('turtle-token-limit', `Token exceeds maxTokenLength (${this.maxTokenLength}).`) + } + } +} + +/** RDF 1.2 Turtle/TriG semantic parser over the transient scanner state. */ +class Parser { + readonly scanner: Scanner + readonly options: CompactOptions + readonly allowGraphs: boolean + readonly prefixes = new Map() + readonly maxDepth: number + readonly maxStatementEvents: number + baseIri: string | undefined + version: CompactVersion | undefined + #generated = 0 + #started = false + + /** Creates semantic Turtle/TriG parser state with isolated prefixes, base IRI, graph, and statement buffers. */ + constructor(source: TextSource, options: CompactOptions, allowGraphs: boolean) { + this.scanner = new Scanner(source, options) + this.options = options + this.allowGraphs = allowGraphs + this.baseIri = options.baseIri + this.maxDepth = options.maxDepth ?? DEFAULT_MAX_DEPTH + this.maxStatementEvents = options.maxStatementEvents ?? DEFAULT_MAX_STATEMENT_EVENTS + } + + /** Parses the complete document and yields semantic/directive events. */ + async *events(): AsyncGenerator { + try { + if (!this.#started) { + this.#started = true + await this.scanner.next() + } + + while (this.#kind() !== Kind.Eof) { + throwIfAborted(this.options.signal) + if (isDirective(this.#kind())) { + try { + yield await this.#directive() + } catch (error) { + if (!this.options.tolerant) throw error + yield { kind: 'diagnostic', diagnostic: this.#diagnostic(error) } + await this.#recoverTop() + } + continue + } + + if (this.allowGraphs && this.#kind() === Kind.Graph) { + await this.#advance() + const graph = await this.#graphLabel(0) + if (this.#kind() !== Kind.LBrace) { + const error = this.scanner.error('trig-graph-open', "Expected '{' after GRAPH label.") + if (!this.options.tolerant) throw error + yield { kind: 'diagnostic', diagnostic: this.#diagnostic(error) } + await this.#recoverTop() + continue + } + yield* this.#graphBlock(graph) + continue + } + + if (this.allowGraphs && this.#kind() === Kind.LBrace) { + yield* this.#graphBlock(defaultGraph()) + continue + } + + if (this.allowGraphs && this.#kind() === Kind.LBracket) { + yield* this.#trigBracket() + continue + } + + if (this.allowGraphs && isGraphLabelStart(this.#kind())) { + const range = this.scanner.range() + const lead = await this.#graphLabel(0) + if (this.#kind() === Kind.LBrace) { + yield* this.#graphBlock(lead) + continue + } + yield* this.#statement(defaultGraph(), lead, range) + continue + } + + yield* this.#statement(defaultGraph()) + } + } finally { + await this.scanner.close() + } + } + + /** + * Parses a top-level TriG `[` construct without confusing a blank-node + * property list with the empty anonymous blank node allowed as a graph name. + * + * TriG graph labels may use `[]`, but `[ :p :o ]` is a Turtle subject. The + * distinction is only visible after the opening bracket, so it cannot be + * decided by the one-token lookahead in {@link Scanner}. + */ + async *#trigBracket(): AsyncGenerator { + const start = this.scanner.range() + await this.#advance() + const node = this.#fresh() + + if (this.#kind() === Kind.RBracket) { + await this.#advance() + if (this.#kind() === Kind.LBrace) { + yield* this.#graphBlock(node) + return + } + yield* this.#statement(defaultGraph(), node, start) + return + } + + if (!this.options.tolerant) { + yield* this.#trigPropertyStatement(node, start) + return + } + + const buffered: CompactEvent[] = [] + try { + for await (const event of this.#trigPropertyStatement(node, start)) { + buffered.push(event) + if (buffered.length > this.maxStatementEvents) { + throw this.scanner.error('turtle-event-limit', `Statement exceeds maxStatementEvents (${this.maxStatementEvents}).`) + } + } + yield* buffered + } catch (error) { + yield { kind: 'diagnostic', diagnostic: this.#diagnostic(error) } + await this.#recoverStatement() + } + } + + /** Parses the remainder of a non-empty top-level blank-node property-list statement. */ + async *#trigPropertyStatement(node: Subject, start: CompactRange): AsyncGenerator { + yield* this.#predicateObjectList(node, defaultGraph(), 1, start, Kind.RBracket) + if (this.#kind() !== Kind.RBracket) { + throw this.scanner.error('turtle-property-list-end', "Expected ']' to close blank-node property list.") + } + await this.#advance() + if (this.#kind() !== Kind.Dot) { + yield* this.#predicateObjectList(node, defaultGraph(), 0, start) + } + await this.#expectDot() + } + + /** Graph block as one isolated step of the Parser state machine. */ + async *#graphBlock(graph: Graph): AsyncGenerator { + if (this.#kind() !== Kind.LBrace) throw this.scanner.error('trig-graph-open', "Expected '{' to start graph block.") + await this.#advance() + + while (this.#kind() !== Kind.RBrace && this.#kind() !== Kind.Eof) { + yield* this.#statement(graph) + } + + if (this.#kind() !== Kind.RBrace) throw this.scanner.error('trig-graph-end', "Expected '}' to close graph block.") + await this.#advance() + } + + /** Statement as one isolated step of the Parser state machine. */ + async *#statement(graph: Graph, lead?: Subject, leadRange?: CompactRange): AsyncGenerator { + if (!this.options.tolerant) { + yield* this.#statementStrict(graph, lead, leadRange) + return + } + + const buffered: CompactEvent[] = [] + try { + for await (const event of this.#statementStrict(graph, lead, leadRange)) { + buffered.push(event) + if (buffered.length > this.maxStatementEvents) { + throw this.scanner.error('turtle-event-limit', `Statement exceeds maxStatementEvents (${this.maxStatementEvents}).`) + } + } + yield* buffered + } catch (error) { + yield { kind: 'diagnostic', diagnostic: this.#diagnostic(error) } + await this.#recoverStatement() + } + } + + /** Statement strict as one isolated step of the Parser state machine. */ + async *#statementStrict(graph: Graph, lead?: Subject, leadRange?: CompactRange): AsyncGenerator { + const range = leadRange ?? this.scanner.range() + let subject: Subject + + if (lead !== undefined) { + subject = lead + } else if (this.#kind() === Kind.LBracket) { + subject = yield* this.#blankPropertyList(graph, 0) + if (this.#kind() !== Kind.Dot) { + yield* this.#predicateObjectList(subject, graph, 0) + } + await this.#expectDot() + return + } else if (this.#kind() === Kind.ReifiedStart) { + subject = yield* this.#reified(graph, 0) + if (this.#kind() !== Kind.Dot) yield* this.#predicateObjectList(subject, graph, 0) + await this.#expectDot() + return + } else { + subject = yield* this.#subject(graph, 0) + } + + yield* this.#predicateObjectList(subject, graph, 0, range) + await this.#expectDot() + } + + /** Predicate object list as one isolated step of the Parser state machine. */ + async *#predicateObjectList( + subject: Subject, + graph: Graph, + depth: number, + statementRange?: CompactRange, + terminator?: Kind, + ): AsyncGenerator { + this.#depth(depth) + while (true) { + const predicate = await this.#verb() + yield* this.#objectList(subject, predicate, graph, depth + 1, statementRange) + + if (this.#kind() !== Kind.Semicolon) return + do await this.#advance() + while (this.#kind() === Kind.Semicolon) + if (terminator !== undefined && this.#kind() === terminator) return + if (this.#kind() === Kind.Dot || this.#kind() === Kind.RBracket || this.#kind() === Kind.AnnotationEnd) return + } + } + + /** Object list as one isolated step of the Parser state machine. */ + async *#objectList( + subject: Subject, + predicate: Predicate, + graph: Graph, + depth: number, + statementRange?: CompactRange, + ): AsyncGenerator { + while (true) { + const start = statementRange ?? this.scanner.range() + const object = yield* this.#object(graph, depth) + const asserted = quad(subject, predicate, object, graph) + yield { kind: 'quad', quad: asserted, range: mergeRange(start, this.scanner.range()) } + yield* this.#annotations(asserted, graph, depth + 1) + if (this.#kind() !== Kind.Comma) return + await this.#advance() + } + } + + /** Annotations as one isolated step of the Parser state machine. */ + async *#annotations(asserted: Quad, graph: Graph, depth: number): AsyncGenerator { + this.#depth(depth) + const tripleTerm = triple(asserted.subject, asserted.predicate, asserted.object) + let activeReifier: Subject | undefined + + while (this.#kind() === Kind.Tilde || this.#kind() === Kind.AnnotationStart) { + if (this.#kind() === Kind.Tilde) { + const start = this.scanner.range() + await this.#advance() + activeReifier = isIriStart(this.#kind()) || isBlankStartKind(this.#kind()) + ? await this.#reifierTerm() + : this.#fresh() + yield { + kind: 'quad', + quad: quad(activeReifier, namedNode(RDF.reifies), tripleTerm, graph), + range: mergeRange(start, this.scanner.range()), + } + if (this.#kind() !== Kind.AnnotationStart) { + activeReifier = undefined + continue + } + } + + if (this.#kind() === Kind.AnnotationStart) { + const start = this.scanner.range() + const reifier = activeReifier ?? this.#fresh() + if (activeReifier === undefined) { + yield { + kind: 'quad', + quad: quad(reifier, namedNode(RDF.reifies), tripleTerm, graph), + range: start, + } + } + await this.#advance() + yield* this.#predicateObjectList(reifier, graph, depth + 1, start, Kind.AnnotationEnd) + if (this.#kind() !== Kind.AnnotationEnd) { + throw this.scanner.error('turtle-annotation-end', "Expected '|}' to close annotation block.") + } + await this.#advance() + activeReifier = undefined + } + } + } + + /** Subject as one isolated step of the Parser state machine. */ + async *#subject(graph: Graph, depth: number): AsyncGenerator { + this.#depth(depth) + if (isIriStart(this.#kind())) return await this.#iri() + if (this.#kind() === Kind.Blank) return await this.#labelledBlank() + if (this.#kind() === Kind.LParen) return yield* this.#collection(graph, depth + 1) + throw this.scanner.error('turtle-subject', 'Expected IRI, blank node, or collection as Turtle subject.') + } + + /** Object as one isolated step of the Parser state machine. */ + async *#object(graph: Graph, depth: number): AsyncGenerator { + this.#depth(depth) + if (isIriStart(this.#kind())) return await this.#iri() + if (this.#kind() === Kind.Blank) return await this.#labelledBlank() + if (this.#kind() === Kind.LBracket) return yield* this.#blankPropertyList(graph, depth + 1) + if (this.#kind() === Kind.LParen) return yield* this.#collection(graph, depth + 1) + if (this.#kind() === Kind.String || this.#kind() === Kind.Number || this.#kind() === Kind.True || this.#kind() === Kind.False) { + return await this.#literal() + } + if (this.#kind() === Kind.TripleStart) return yield* this.#tripleTerm(graph, depth + 1) + if (this.#kind() === Kind.ReifiedStart) return yield* this.#reified(graph, depth + 1) + throw this.scanner.error('turtle-object', 'Expected Turtle RDF object.') + } + + /** Collection as one isolated step of the Parser state machine. */ + async *#collection(graph: Graph, depth: number): AsyncGenerator { + this.#depth(depth) + const start = this.scanner.range() + if (this.#kind() !== Kind.LParen) throw this.scanner.error('turtle-collection', "Expected '(' to start collection.") + await this.#advance() + if (this.#kind() === Kind.RParen) { + await this.#advance() + return namedNode(RDF.nil) + } + + const head = this.#fresh() + let current = head + while (this.#kind() !== Kind.RParen) { + if (this.#kind() === Kind.Eof) throw this.scanner.error('turtle-collection-end', "Expected ')' to close collection.") + const object = yield* this.#object(graph, depth + 1) + yield { kind: 'quad', quad: quad(current, namedNode(RDF.first), object, graph), range: start } + if (this.#kind() === Kind.RParen) { + yield { kind: 'quad', quad: quad(current, namedNode(RDF.rest), namedNode(RDF.nil), graph), range: start } + break + } + const next = this.#fresh() + yield { kind: 'quad', quad: quad(current, namedNode(RDF.rest), next, graph), range: start } + current = next + } + await this.#advance() + return head + } + + /** Blank property list as one isolated step of the Parser state machine. */ + async *#blankPropertyList(graph: Graph, depth: number): AsyncGenerator { + this.#depth(depth) + if (this.#kind() !== Kind.LBracket) throw this.scanner.error('turtle-property-list', "Expected '['.") + const start = this.scanner.range() + await this.#advance() + const node = this.#fresh() + if (this.#kind() === Kind.RBracket) { + await this.#advance() + return node + } + yield* this.#predicateObjectList(node, graph, depth + 1, start, Kind.RBracket) + if (this.#kind() !== Kind.RBracket) throw this.scanner.error('turtle-property-list-end', "Expected ']' to close blank-node property list.") + await this.#advance() + return node + } + + /** Triple term as one isolated step of the Parser state machine. */ + async *#tripleTerm(graph: Graph, depth: number): AsyncGenerator { + this.#depth(depth) + if (this.#kind() !== Kind.TripleStart) throw this.scanner.error('turtle-triple-term', "Expected '<<('.") + await this.#advance() + const subject = await this.#tripleSubject() + const predicate = await this.#verb() + const object = yield* this.#tripleObject(graph, depth + 1) + if (this.#kind() !== Kind.TripleEnd) throw this.scanner.error('turtle-triple-term-end', "Expected ')>>' after triple term.") + await this.#advance() + return triple(subject, predicate, object) + } + + /** Reified as one isolated step of the Parser state machine. */ + async *#reified(graph: Graph, depth: number): AsyncGenerator { + this.#depth(depth) + const start = this.scanner.range() + if (this.#kind() !== Kind.ReifiedStart) throw this.scanner.error('turtle-reified', "Expected '<<'.") + await this.#advance() + const subject = yield* this.#reifiedSubject(graph, depth + 1) + const predicate = await this.#verb() + const object = yield* this.#reifiedObject(graph, depth + 1) + let reifier: Subject | undefined + if (this.#kind() === Kind.Tilde) { + await this.#advance() + reifier = isIriStart(this.#kind()) || isBlankStartKind(this.#kind()) + ? await this.#reifierTerm() + : this.#fresh() + } + if (this.#kind() !== Kind.ReifiedEnd) throw this.scanner.error('turtle-reified-end', "Expected '>>' after reified triple.") + await this.#advance() + const value = reifier ?? this.#fresh() + yield { + kind: 'quad', + quad: quad(value, namedNode(RDF.reifies), triple(subject, predicate, object), graph), + range: mergeRange(start, this.scanner.range()), + } + return value + } + + /** Reified subject as one isolated step of the Parser state machine. */ + async *#reifiedSubject(graph: Graph, depth: number): AsyncGenerator { + if (isIriStart(this.#kind())) return await this.#iri() + if (isBlankStartKind(this.#kind())) return await this.#reifierTerm() + if (this.#kind() === Kind.ReifiedStart) return yield* this.#reified(graph, depth + 1) + throw this.scanner.error('turtle-reified-subject', 'Expected IRI, blank node, or nested reified triple.') + } + + /** Reified object as one isolated step of the Parser state machine. */ + async *#reifiedObject(graph: Graph, depth: number): AsyncGenerator { + if (isIriStart(this.#kind())) return await this.#iri() + if (isBlankStartKind(this.#kind())) return await this.#reifierTerm() + if (this.#kind() === Kind.String || this.#kind() === Kind.Number || this.#kind() === Kind.True || this.#kind() === Kind.False) return await this.#literal() + if (this.#kind() === Kind.TripleStart) return yield* this.#tripleTerm(graph, depth + 1) + if (this.#kind() === Kind.ReifiedStart) return yield* this.#reified(graph, depth + 1) + throw this.scanner.error('turtle-reified-object', 'Expected RDF term allowed in a reified triple object.') + } + + /** Triple object as one isolated step of the Parser state machine. */ + async *#tripleObject(graph: Graph, depth: number): AsyncGenerator { + if (isIriStart(this.#kind())) return await this.#iri() + if (this.#kind() === Kind.Blank) return await this.#labelledBlank() + if (this.#kind() === Kind.LBracket) return await this.#anonymous() + if (this.#kind() === Kind.String || this.#kind() === Kind.Number || this.#kind() === Kind.True || this.#kind() === Kind.False) return await this.#literal() + if (this.#kind() === Kind.TripleStart) return yield* this.#tripleTerm(graph, depth + 1) + throw this.scanner.error('turtle-triple-object', 'Expected RDF term allowed in a triple-term object.') + } + + /** Triple subject as one isolated step of the Parser state machine. */ + async #tripleSubject(): Promise { + if (isIriStart(this.#kind())) return await this.#iri() + if (this.#kind() === Kind.Blank) return await this.#labelledBlank() + if (this.#kind() === Kind.LBracket) return await this.#anonymous() + throw this.scanner.error('turtle-triple-subject', 'Expected IRI or blank node in triple term.') + } + + /** Verb as one isolated step of the Parser state machine. */ + async #verb(): Promise { + if (this.#kind() === Kind.A) { + await this.#advance() + return namedNode(RDF.type) + } + return await this.#iri() + } + + /** Literal as one isolated step of the Parser state machine. */ + async #literal(): Promise { + if (this.#kind() === Kind.True || this.#kind() === Kind.False) { + const raw = this.scanner.raw + await this.#advance() + return literal(raw, namedNode(XSD.boolean)) + } + if (this.#kind() === Kind.Number) { + const raw = this.scanner.raw + const datatype = numericKind(raw) + if (!datatype) throw this.scanner.error('turtle-number', `Invalid numeric literal '${raw}'.`) + await this.#advance() + return literal(raw, namedNode(datatype)) + } + if (this.#kind() !== Kind.String) throw this.scanner.error('turtle-literal', 'Expected RDF literal.') + const value = this.scanner.value + await this.#advance() + if (this.#kind() === Kind.Lang) { + const raw = this.scanner.value + await this.#advance() + const marker = raw.lastIndexOf('--') + if (marker > 0) { + const language = raw.slice(0, marker) + const direction = raw.slice(marker + 2).toLowerCase() + if (direction !== 'ltr' && direction !== 'rtl') { + throw this.scanner.error('turtle-direction', `Initial text direction must be ltr or rtl, got '${direction}'.`) + } + return literal(value, { language, direction }) + } + return literal(value, raw) + } + if (this.#kind() === Kind.HatHat) { + await this.#advance() + return literal(value, await this.#iri()) + } + return literal(value) + } + + /** Directive as one isolated step of the Parser state machine. */ + async #directive(): Promise> { + const kind = this.#kind() + const raw = this.scanner.raw + const start = this.scanner.range() + const oldStyle = raw.startsWith('@') + await this.#advance() + + if (kind === Kind.Prefix) { + if (this.#kind() !== Kind.PName || !this.scanner.raw.endsWith(':')) { + throw this.scanner.error('turtle-prefix-name', 'PREFIX requires a prefix label ending in colon.') + } + const prefix = this.scanner.raw.slice(0, -1) + await this.#advance() + if (this.#kind() !== Kind.Iri) throw this.scanner.error('turtle-prefix-iri', 'PREFIX requires an IRI reference.') + const iri = this.#resolve(this.scanner.value) + await this.#advance() + if (oldStyle) await this.#expectDot() + this.prefixes.set(prefix, iri) + return { kind: 'prefix', prefix, iri, range: mergeRange(start, this.scanner.range()) } + } + + if (kind === Kind.Base) { + if (this.#kind() !== Kind.Iri) throw this.scanner.error('turtle-base-iri', 'BASE requires an IRI reference.') + const iri = this.#resolve(this.scanner.value) + await this.#advance() + if (oldStyle) await this.#expectDot() + this.baseIri = iri + return { kind: 'base', iri, range: mergeRange(start, this.scanner.range()) } + } + + if (kind === Kind.Version) { + if (this.#kind() !== Kind.String) throw this.scanner.error('turtle-version', 'VERSION requires a quoted RDF version label.') + const version = this.scanner.value + if (version !== '1.1' && version !== '1.2-basic' && version !== '1.2') { + throw this.scanner.error('turtle-version', `Unsupported RDF version '${version}'.`) + } + await this.#advance() + if (oldStyle) await this.#expectDot() + this.version = version + return { kind: 'version', version, range: mergeRange(start, this.scanner.range()) } + } + + throw this.scanner.error('turtle-directive', 'Unsupported Turtle directive.') + } + + /** Iri as one isolated step of the Parser state machine. */ + async #iri(): Promise { + if (this.#kind() === Kind.Iri) { + const value = this.#resolve(this.scanner.value) + await this.#advance() + return namedNode(value) + } + if (this.#kind() === Kind.PName) { + const raw = this.scanner.raw + const colon = raw.indexOf(':') + const prefix = raw.slice(0, colon) + const local = decodeLocal(raw.slice(colon + 1)) + const base = this.prefixes.get(prefix) + if (base === undefined) throw this.scanner.error('turtle-prefix', `Prefix '${prefix}' is not defined.`) + await this.#advance() + return namedNode(`${base}${local}`) + } + throw this.scanner.error('turtle-iri', 'Expected IRI reference or prefixed name.') + } + + /** Graph label as one isolated step of the Parser state machine. */ + async #graphLabel(depth: number): Promise { + this.#depth(depth) + if (isIriStart(this.#kind())) return await this.#iri() + if (this.#kind() === Kind.Blank) return await this.#labelledBlank() + if (this.#kind() === Kind.LBracket) return await this.#anonymous() + throw this.scanner.error('trig-graph-label', 'Expected IRI or blank node as TriG graph label.') + } + + /** Reifier term as one isolated step of the Parser state machine. */ + async #reifierTerm(): Promise { + if (isIriStart(this.#kind())) return await this.#iri() + if (this.#kind() === Kind.Blank) return await this.#labelledBlank() + if (this.#kind() === Kind.LBracket) return await this.#anonymous() + throw this.scanner.error('turtle-reifier', 'Expected IRI or blank node as reifier.') + } + + /** Labelled blank as one isolated step of the Parser state machine. */ + async #labelledBlank(): Promise { + if (this.#kind() !== Kind.Blank) throw this.scanner.error('turtle-blank', 'Expected blank node.') + const value = `l${this.scanner.value.length}:${this.scanner.value}` + await this.#advance() + return blankNode(value) + } + + /** Anonymous as one isolated step of the Parser state machine. */ + async #anonymous(): Promise { + if (this.#kind() !== Kind.LBracket) throw this.scanner.error('turtle-anon', "Expected '['.") + await this.#advance() + if (this.#kind() !== Kind.RBracket) throw this.scanner.error('turtle-anon', "Expected ']' for anonymous blank node.") + await this.#advance() + return this.#fresh() + } + + /** Expect dot as one isolated step of the Parser state machine. */ + async #expectDot(): Promise { + if (this.#kind() !== Kind.Dot) throw this.scanner.error('turtle-period', "Expected '.' after Turtle statement.") + await this.#advance() + } + + /** Kind as one isolated step of the Parser state machine. */ + #kind(): Kind { + return this.scanner.kind + } + + /** Advance as one isolated step of the Parser state machine. */ + async #advance(): Promise { + await this.scanner.next() + } + + /** Resolve as one isolated step of the Parser state machine. */ + #resolve(reference: string): string { + try { + if (this.baseIri !== undefined) return new URL(reference, this.baseIri).href + return new URL(reference).href + } catch { + throw this.scanner.error('turtle-relative-iri', `Relative IRI '${reference}' requires a base IRI.`) + } + } + + /** Fresh as one isolated step of the Parser state machine. */ + #fresh(): Subject { + return blankNode(`g:${++this.#generated}`) + } + + /** Depth as one isolated step of the Parser state machine. */ + #depth(depth: number): void { + if (depth > this.maxDepth) throw this.scanner.error('turtle-depth', `Nested Turtle syntax exceeds maxDepth (${this.maxDepth}).`) + } + + /** Diagnostic as one isolated step of the Parser state machine. */ + #diagnostic(error: unknown): CompactDiagnostic { + if (error instanceof CompactError) return { code: error.code, message: error.message, range: error.range } + return { + code: 'turtle-syntax', + message: error instanceof Error ? error.message : String(error), + range: this.scanner.range(), + } + } + + /** Recover statement as one isolated step of the Parser state machine. */ + async #recoverStatement(): Promise { + let square = 0 + let paren = 0 + let annotation = 0 + while (this.#kind() !== Kind.Eof) { + if (this.#kind() === Kind.LBracket) square++ + else if (this.#kind() === Kind.RBracket) square = Math.max(0, square - 1) + else if (this.#kind() === Kind.LParen) paren++ + else if (this.#kind() === Kind.RParen) paren = Math.max(0, paren - 1) + else if (this.#kind() === Kind.AnnotationStart) annotation++ + else if (this.#kind() === Kind.AnnotationEnd) annotation = Math.max(0, annotation - 1) + else if (this.#kind() === Kind.Dot && square === 0 && paren === 0 && annotation === 0) { + await this.#advance() + return + } else if (this.allowGraphs && this.#kind() === Kind.RBrace && square === 0 && paren === 0 && annotation === 0) { + return + } + await this.#advance() + } + } + + /** Recover top as one isolated step of the Parser state machine. */ + async #recoverTop(): Promise { + while (this.#kind() !== Kind.Eof) { + if (this.#kind() === Kind.Dot) { + await this.#advance() + return + } + if (this.allowGraphs && this.#kind() === Kind.RBrace) { + await this.#advance() + return + } + await this.#advance() + } + } +} + +/** Parses Turtle/TriG events; `allowGraphs` selects TriG graph syntax. */ +export function parseCompact(source: TextSource, options: CompactOptions, allowGraphs: boolean): AsyncGenerator { + return new Parser(source, options, allowGraphs).events() +} + +/** Resolves one lexer numeric token to its RDF datatype. */ +function numericKind(raw: string): string | undefined { + if (/^[+-]?[0-9]+$/.test(raw)) return XSD.integer + if (/^[+-]?(?:[0-9]*\.[0-9]+)$/.test(raw)) return XSD.decimal + if (/^[+-]?(?:(?:[0-9]+(?:\.[0-9]*)?)|(?:\.[0-9]+))[eE][+-]?[0-9]+$/.test(raw)) return XSD.double + return undefined +} + +/** Returns whether the supplied value satisfies the directive contract. */ +function isDirective(kind: Kind): boolean { + return kind === Kind.Prefix || kind === Kind.Base || kind === Kind.Version +} + +/** Returns whether the supplied value satisfies the iri start contract. */ +function isIriStart(kind: Kind): boolean { + return kind === Kind.Iri || kind === Kind.PName +} + +/** Returns whether the supplied value satisfies the blank start kind contract. */ +function isBlankStartKind(kind: Kind): boolean { + return kind === Kind.Blank || kind === Kind.LBracket +} + +/** Returns whether the supplied value satisfies the graph label start contract. */ +function isGraphLabelStart(kind: Kind): boolean { + return isIriStart(kind) || kind === Kind.Blank +} + +/** Returns whether the supplied value satisfies the whitespace contract. */ +function isWhitespace(char: string): boolean { + return char === ' ' || char === '\t' || char === '\n' || char === '\r' +} + +/** Returns whether the supplied value satisfies the name start contract. */ +function isNameStart(char: string): boolean { + return /[A-Za-z_\u00C0-\uFFFF]/u.test(char) +} + +/** Returns whether the supplied value satisfies the prefix char contract. */ +function isPrefixChar(char: string): boolean { + return /[A-Za-z0-9_.\-\u00B7-\uFFFF]/u.test(char) +} + +/** Returns whether the supplied value satisfies the local char contract. */ +function isLocalChar(char: string): boolean { + return /[A-Za-z0-9_.:\-\u00B7-\uFFFF]/u.test(char) +} + +/** Returns whether the supplied value satisfies the blank start contract. */ +function isBlankStart(char: string): boolean { + return /[A-Za-z0-9_\u00C0-\uFFFF]/u.test(char) +} + +/** Returns whether the supplied value satisfies the blank char contract. */ +function isBlankChar(char: string): boolean { + return /[A-Za-z0-9_.\-\u00B7-\uFFFF]/u.test(char) +} + +/** Returns whether the supplied value satisfies the local escape contract. */ +function isLocalEscape(char: string): boolean { + return "_~.-!$&'()*+,;=/?#@%".includes(char) +} + +/** Decodes percent and PN_LOCAL escapes in a prefixed-name local part without changing the namespace IRI. */ +function decodeLocal(value: string): string { + return value.replace(/\\([_~.\-!$&'()*+,;=/?#@%])/g, '$1') +} + +/** Decodes Turtle backslash escapes used in local names and string-like scanner values. */ +function escapeValue(value: string): string { + switch (value) { + case 't': return '\t' + case 'b': return '\b' + case 'n': return '\n' + case 'r': return '\r' + case 'f': return '\f' + case '"': return '"' + case "'": return "'" + case '\\': return '\\' + default: return value + } +} + +/** Spans two source ranges so emitted semantic events retain the full originating syntax range. */ +function mergeRange(start: CompactRange, end: CompactRange): CompactRange { + return { start: start.start, end: Math.max(start.end, end.end), line: start.line, column: start.column } +} diff --git a/packages/rdf/dataset.ts b/packages/rdf/dataset.ts new file mode 100644 index 0000000..20b54c0 --- /dev/null +++ b/packages/rdf/dataset.ts @@ -0,0 +1,264 @@ +/** + * In-memory RDF dataset with exact-term indexes. + * + * The public API follows RDF/JS DatasetCore semantics. Four indexes accelerate + * common exact-term match patterns without changing the semantic object model. + * The indexes are deliberately private so a future packed representation can + * replace them without changing callers. + * + * @module + */ + +import { key } from './term.ts' +import type { Graph, ObjectTerm, Predicate, Quad, Subject, Term } from './term.ts' +import { iterate } from './source.ts' + +/** RDF dataset match pattern. `null` and `undefined` mean wildcard. */ +export interface MatchOptions { + readonly subject?: Subject | null + readonly predicate?: Predicate | null + readonly object?: ObjectTerm | null + readonly graph?: Graph | null +} + +/** Mutable RDF/JS-style DatasetCore implementation. */ +export class Dataset implements Iterable { + readonly #quads = new Map() + readonly #subject = new Map() + readonly #predicate = new Map() + readonly #object = new Map() + readonly #graph = new Map() + + /** Seeds the dataset through `addAll` so initial quads and all exact-term indexes use the normal deduplication path. */ + constructor(quads?: Iterable) { + if (quads) this.addAll(quads) + } + + /** Number of unique quads currently stored. */ + get size(): number { + return this.#quads.size + } + + /** Adds one quad and returns this dataset. Existing equal quads are ignored. */ + add(quad: Quad): this { + const keys = quadKeys(quad) + if (this.#quads.has(keys.quad)) return this + this.#quads.set(keys.quad, quad) + addIndex(this.#subject, keys.subject, keys.quad) + addIndex(this.#predicate, keys.predicate, keys.quad) + addIndex(this.#object, keys.object, keys.quad) + addIndex(this.#graph, keys.graph, keys.quad) + return this + } + + /** Adds all quads from a synchronous source. */ + addAll(quads: Iterable): this { + for (const quad of quads) this.add(quad) + return this + } + + /** Imports a sync or async source with cooperative cancellation. */ + async import(source: Iterable | AsyncIterable, options: { readonly signal?: AbortSignal } = {}): Promise { + for await (const quad of iterate(source)) { + if (options.signal?.aborted) throw options.signal.reason ?? new DOMException('Aborted', 'AbortError') + this.add(quad) + } + return this + } + + /** Deletes one equal quad and returns this dataset. */ + delete(quad: Quad): this { + const keys = quadKeys(quad) + const stored = this.#quads.get(keys.quad) + if (!stored) return this + this.#quads.delete(keys.quad) + deleteIndex(this.#subject, keys.subject, keys.quad) + deleteIndex(this.#predicate, keys.predicate, keys.quad) + deleteIndex(this.#object, keys.object, keys.quad) + deleteIndex(this.#graph, keys.graph, keys.quad) + return this + } + + /** Returns whether this dataset contains an equal quad. */ + has(quad: Quad): boolean { + return this.#quads.has(key(quad)) + } + + /** Removes every quad. */ + clear(): void { + this.#quads.clear() + this.#subject.clear() + this.#predicate.clear() + this.#object.clear() + this.#graph.clear() + } + + /** + * Returns a new dataset containing quads that match the RDF/JS pattern. + * + * The implementation starts from the smallest available exact-term index and + * verifies the remaining terms semantically. This avoids a full scan for the + * dominant selective lookup shapes while retaining correct wildcard behavior. + */ + match( + subject: Subject | null = null, + predicate: Predicate | null = null, + object: ObjectTerm | null = null, + graph: Graph | null = null, + ): Dataset { + return new Dataset(this.matchIter({ subject, predicate, object, graph })) + } + + /** Lazily iterates matching quads without materializing another dataset. */ + *matchIter(options: MatchOptions = {}): Generator { + const { subject = null, predicate = null, object = null, graph = null } = options + const indexed: IndexBucket[] = [] + if (subject) { const value = this.#subject.get(key(subject)); if (!value) return; indexed.push(value) } + if (predicate) { const value = this.#predicate.get(key(predicate)); if (!value) return; indexed.push(value) } + if (object) { const value = this.#object.get(key(object)); if (!value) return; indexed.push(value) } + if (graph) { const value = this.#graph.get(key(graph)); if (!value) return; indexed.push(value) } + + const candidates = indexed.length === 0 ? this.#quads.keys() : bucketValues(smallest(indexed)) + + for (const quadKey of candidates) { + const quad = this.#quads.get(quadKey) + if (!quad) continue + if (subject && !quad.subject.equals(subject)) continue + if (predicate && !quad.predicate.equals(predicate)) continue + if (object && !quad.object.equals(object)) continue + if (graph && !quad.graph.equals(graph)) continue + yield quad + } + } + + /** Deletes all quads matching a pattern and returns this dataset. */ + deleteMatches( + subject: Subject | null = null, + predicate: Predicate | null = null, + object: ObjectTerm | null = null, + graph: Graph | null = null, + ): this { + for (const quad of [...this.matchIter({ subject, predicate, object, graph })]) this.delete(quad) + return this + } + + /** Returns an inexpensive exact-term cardinality estimate when an index exists. */ + estimate(options: MatchOptions = {}): number | undefined { + const counts: number[] = [] + if (options.subject) counts.push(bucketSize(this.#subject.get(key(options.subject)))) + if (options.predicate) counts.push(bucketSize(this.#predicate.get(key(options.predicate)))) + if (options.object) counts.push(bucketSize(this.#object.get(key(options.object)))) + if (options.graph) counts.push(bucketSize(this.#graph.get(key(options.graph)))) + if (counts.length === 0) return this.size + return Math.min(...counts) + } + + /** Returns the dataset quads in insertion order. */ + [Symbol.iterator](): Iterator { + return this.#quads.values() + } +} + +/** Exact-term index bucket that stores a singleton quad key directly and promotes to a Set only after a second match. */ +type IndexBucket = string | Set + +/** Canonical semantic keys computed once per quad for deduplication plus subject/predicate/object/graph indexing. */ +interface QuadKeysType { + readonly quad: string + readonly subject: string + readonly predicate: string + readonly object: string + readonly graph: string +} + +/** Computes component keys once so indexed insertion does not repeat RDF term serialization. */ +function quadKeys(quad: Quad): QuadKeysType { + const subject = key(quad.subject) + const predicate = key(quad.predicate) + const object = key(quad.object) + const graph = key(quad.graph) + return { quad: `Q${part(subject)}${part(predicate)}${part(object)}${part(graph)}`, subject, predicate, object, graph } +} + +/** Length-prefixes one already-serialized term key exactly like the public quad key format. */ +function part(value: string): string { + return `${value.length}:${value}` +} + +/** Creates an in-memory dataset. */ +export function dataset(quads?: Iterable): Dataset { + return new Dataset(quads) +} + +/** Adds one quad key, allocating a Set only after a term has multiple quads. */ +function addIndex(index: Map, termKey: string, quadKey: string): void { + const current = index.get(termKey) + if (current === undefined) { + index.set(termKey, quadKey) + return + } + if (typeof current === 'string') { + if (current !== quadKey) index.set(termKey, new Set([current, quadKey])) + return + } + current.add(quadKey) +} + +/** Removes one quad key and demotes two-entry Sets back to singleton strings. */ +function deleteIndex(index: Map, termKey: string, quadKey: string): void { + const current = index.get(termKey) + if (current === undefined) return + if (typeof current === 'string') { + if (current === quadKey) index.delete(termKey) + return + } + current.delete(quadKey) + if (current.size === 0) index.delete(termKey) + else if (current.size === 1) index.set(termKey, current.values().next().value!) +} + +/** Returns the cardinality of one optional compact index bucket. */ +function bucketSize(value: IndexBucket | undefined): number { + if (value === undefined) return 0 + return typeof value === 'string' ? 1 : value.size +} + +/** Iterates singleton and multi-quad buckets through one allocation-free lookup shape. */ +function* bucketValues(value: IndexBucket): Generator { + if (typeof value === 'string') yield value + else yield* value +} + +/** Chooses the narrowest exact-term index before semantic verification. */ +function smallest(values: readonly IndexBucket[]): IndexBucket { + let selected = values[0]! + let selectedSize = bucketSize(selected) + for (let index = 1; index < values.length; index++) { + const value = values[index]! + const size = bucketSize(value) + if (size < selectedSize) { + selected = value + selectedSize = size + } + } + return selected +} + +/** Returns whether two datasets contain exactly the same RDF terms. */ +export function datasetEquals(left: Iterable, right: Iterable): boolean { + const other = right instanceof Dataset ? right : new Dataset(right) + let count = 0 + for (const quad of left) { + count++ + if (!other.has(quad)) return false + } + return count === other.size +} + +/** Returns a stable semantic digest input by sorting term keys. */ +export function datasetKey(value: Iterable): string { + return [...value].map((quad) => key(quad)).sort().join('\n') +} + +/** Re-exports Term in TSDoc references without adding another public data model. */ +export type { Term } diff --git a/packages/rdf/dataset_bench.ts b/packages/rdf/dataset_bench.ts new file mode 100644 index 0000000..068d84f --- /dev/null +++ b/packages/rdf/dataset_bench.ts @@ -0,0 +1,46 @@ +/** Decision benchmark for exact-term dataset indexes. @module */ + +import { bench, do_not_optimize, group, run } from 'mitata' +import { Dataset } from './dataset.ts' +import { literal, namedNode, quad } from './factory.ts' +import type { Quad, Subject } from './term.ts' + +const SIZE = 50_000 +const SUBJECTS = 5_000 +const predicate = namedNode('https://example.com/p') +const quads = Array.from({ length: SIZE }, (_, index) => quad( + namedNode(`https://example.com/s/${index % SUBJECTS}`), + predicate, + literal(`value-${index}`), +)) +const dataset = new Dataset(quads) +const target = namedNode('https://example.com/s/1729') +const expected = scan(quads, target) +const indexed = count(dataset.matchIter({ subject: target })) +if (indexed !== expected) throw new Error(`Dataset benchmark oracle failed: ${indexed} != ${expected}.`) + +/** Baseline full scan over the same semantic quad values. */ +function scan(values: readonly Quad[], subject: Subject): number { + let matches = 0 + for (const value of values) if (value.subject.equals(subject)) matches++ + return matches +} + +/** Counts a lazy match result without materializing another Dataset. */ +function count(values: Iterable): number { + let matches = 0 + for (const _value of values) matches++ + return matches +} + +group('rdf Dataset exact subject match: 50k quads', () => { + bench('semantic full scan baseline', () => { + do_not_optimize(scan(quads, target)) + }) + + bench('four-index Dataset.matchIter', () => { + do_not_optimize(count(dataset.matchIter({ subject: target }))) + }) +}) + +await run() diff --git a/packages/rdf/dataset_test.ts b/packages/rdf/dataset_test.ts new file mode 100644 index 0000000..cc0c2fe --- /dev/null +++ b/packages/rdf/dataset_test.ts @@ -0,0 +1,66 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { dataset, datasetEquals, datasetKey, literal, namedNode, quad } from './mod.ts' + +const p = namedNode('urn:p') +const g = namedNode('urn:g') +const a = quad(namedNode('urn:a'), p, literal('one')) +const b = quad(namedNode('urn:b'), p, literal('two'), g) +const c = quad(namedNode('urn:a'), namedNode('urn:q'), literal('three'), g) + +describe('@okikio/rdf Dataset', () => { + it('matches by semantic RDF term equality instead of object identity', () => { + const graph = dataset([a]) + expect([...graph.matchIter({ subject: namedNode('urn:a') })]).toHaveLength(1) + }) + + it('deduplicates equal quads and preserves insertion order', () => { + const graph = dataset([a, quad(namedNode('urn:a'), p, literal('one')), b]) + expect(graph.size).toBe(2) + expect([...graph].map((value) => value.subject.value)).toEqual(['urn:a', 'urn:b']) + }) + + it('matches intersections across subject, predicate, object, and graph indexes', () => { + const graph = dataset([a, b, c]) + expect([...graph.matchIter({ subject: namedNode('urn:a') })]).toHaveLength(2) + expect([...graph.matchIter({ predicate: p, graph: g })]).toHaveLength(1) + expect([...graph.matchIter({ object: literal('three'), graph: g })]).toHaveLength(1) + expect([...graph.matchIter({ subject: namedNode('urn:missing') })]).toHaveLength(0) + }) + + it('updates compact index buckets after deletes and deleteMatches', () => { + const graph = dataset([a, b, c]) + graph.delete(a) + expect(graph.estimate({ predicate: p })).toBe(1) + graph.deleteMatches(null, null, null, g) + expect(graph.size).toBe(0) + expect(graph.estimate({ graph: g })).toBe(0) + }) + + it('imports asynchronous sources and honors cancellation', async () => { + async function* values() { + yield a + yield b + } + const graph = dataset() + await graph.import(values()) + expect(graph.size).toBe(2) + + const controller = new AbortController() + controller.abort(new Error('stop-import')) + await expect(dataset().import(values(), { signal: controller.signal })).rejects.toThrow('stop-import') + }) + + it('compares datasets semantically and builds insertion-order-independent keys', () => { + expect(datasetEquals([a, b], [b, a])).toBe(true) + expect(datasetEquals([a], [a, b])).toBe(false) + expect(datasetKey([a, b])).toBe(datasetKey([b, a])) + }) + + it('clears all data and exact-term estimates', () => { + const graph = dataset([a, b]) + graph.clear() + expect(graph.size).toBe(0) + expect(graph.estimate()).toBe(0) + }) +}) diff --git a/packages/rdf/deno.json b/packages/rdf/deno.json new file mode 100644 index 0000000..8207f3f --- /dev/null +++ b/packages/rdf/deno.json @@ -0,0 +1,26 @@ +{ + "name": "@okikio/rdf", + "version": "0.1.0", + "license": "MIT", + "exports": { + ".": "./mod.ts", + "./ntriples": "./ntriples/mod.ts", + "./nquads": "./nquads/mod.ts", + "./turtle": "./turtle/mod.ts", + "./trig": "./trig/mod.ts", + "./jsonld": "./jsonld/mod.ts", + "./canon": "./canon/mod.ts", + "./xml": "./xml/mod.ts", + "./rdfa": "./rdfa/mod.ts", + "./microdata": "./microdata/mod.ts", + "./shape": "./shape/mod.ts", + "./ontology": "./ontology/mod.ts" + }, + "imports": { + "jsonld": "npm:jsonld@^9.0.0", + "rdf-canonize": "npm:rdf-canonize@^5.0.0", + "rdfxml-streaming-parser": "npm:rdfxml-streaming-parser@^3.2.0", + "rdfa-streaming-parser": "npm:rdfa-streaming-parser@^3.0.2", + "microdata-rdf-streaming-parser": "npm:microdata-rdf-streaming-parser@^3.0.0" + } +} diff --git a/packages/rdf/factory.ts b/packages/rdf/factory.ts new file mode 100644 index 0000000..ca83b26 --- /dev/null +++ b/packages/rdf/factory.ts @@ -0,0 +1,156 @@ +/** RDF term factories and RDF/JS conversion helpers. @module */ + +import { + BlankNodeValue, + DefaultGraphValue, + type DirectionalLanguage, + type Graph, + type NamedNode, + NamedNodeValue, + type ObjectTerm, + type Predicate, + type Quad, + QuadValue, + RDF, + type Subject, + type Term, + type TermType, + type Variable, + VariableValue, + type Literal, + LiteralValue, + XSD, +} from './term.ts' + +/** Default graph used when the caller does not provide an override. */ +const DEFAULT_GRAPH = new DefaultGraphValue() +/** Monotonic process-local suffix used only when the caller requests an anonymous blank-node identifier. */ +let blankNodeSequence = 0 + +/** Creates an RDF named node. */ +export function namedNode(value: string): NamedNode { + return new NamedNodeValue(value) +} + +/** Creates an RDF blank node, allocating a process-local identifier when omitted. */ +export function blankNode(value?: string): ReturnType { + return createBlankNode(value ?? `b${++blankNodeSequence}`) +} + +/** Create blank node without acquiring unrelated global resources. */ +function createBlankNode(value: string): BlankNodeValue { + return new BlankNodeValue(value) +} + +/** Creates an RDF query variable without a leading `?` or `$`. */ +export function variable(value: string): Variable { + const name = value.replace(/^[?$]/, '') + if (!name) throw new TypeError('Variable name must not be empty.') + return new VariableValue(name) +} + +/** Returns the immutable default-graph singleton. */ +export function defaultGraph(): DefaultGraphValue { + return DEFAULT_GRAPH +} + +/** + * Creates an RDF literal. + * + * A string second argument is interpreted as a language tag. A named node is a + * datatype. Directional language input uses RDF 1.2 `rdf:dirLangString`. + */ +export function literal( + value: string, + languageOrDatatype?: string | NamedNode | DirectionalLanguage, +): Literal { + if (typeof languageOrDatatype === 'string') { + const language = normalizeLanguage(languageOrDatatype) + return new LiteralValue(value, namedNode(RDF.langString), language) + } + + if (languageOrDatatype !== undefined && 'termType' in languageOrDatatype) { + if (languageOrDatatype.termType !== 'NamedNode') throw new TypeError('Literal datatype must be a named node.') + return new LiteralValue(value, languageOrDatatype) + } + + if (languageOrDatatype !== undefined) { + const language = normalizeLanguage(languageOrDatatype.language) + const direction = languageOrDatatype.direction ?? '' + if (direction !== '' && direction !== 'ltr' && direction !== 'rtl') { + throw new TypeError(`Unsupported literal direction '${String(direction)}'.`) + } + return new LiteralValue( + value, + namedNode(direction === '' ? RDF.langString : RDF.dirLangString), + language, + direction, + ) + } + + return new LiteralValue(value, namedNode(XSD.string)) +} + +/** Creates a quad or, with the default graph, an RDF 1.2 triple term. */ +export function quad(subject: Subject, predicate: Predicate, object: ObjectTerm, graph: Graph = DEFAULT_GRAPH): Quad { + if (object.termType === 'Quad' && object.graph.termType !== 'DefaultGraph') { + throw new TypeError('An RDF 1.2 triple term cannot contain a named graph.') + } + return new QuadValue(subject, predicate, object, graph) +} + +/** Creates a triple term explicitly. */ +export function triple(subject: Subject, predicate: Predicate, object: ObjectTerm): Quad { + return quad(subject, predicate, object) +} + +/** Copies an RDF/JS-compatible term into this implementation. */ +export function fromTerm(original: Term): TermType { + switch (original.termType) { + case 'NamedNode': + return namedNode(original.value) + case 'BlankNode': + return blankNode(original.value) + case 'Variable': + return variable(original.value) + case 'DefaultGraph': + return defaultGraph() + case 'Literal': { + const value = original as Literal + if (value.direction) return literal(value.value, { language: value.language, direction: value.direction }) + if (value.language) return literal(value.value, value.language) + return literal(value.value, namedNode(value.datatype.value)) + } + case 'Quad': + return fromQuad(original as Quad) + } +} + +/** Copies an RDF/JS-compatible quad recursively. */ +export function fromQuad(original: Quad): Quad { + return quad( + fromTerm(original.subject) as Subject, + fromTerm(original.predicate) as Predicate, + fromTerm(original.object) as ObjectTerm, + fromTerm(original.graph) as Graph, + ) +} + +/** RDF/JS-compatible data factory object for APIs that expect one. */ +export const factory = { + namedNode, + blankNode, + literal, + variable, + defaultGraph, + quad, + fromTerm, + fromQuad, +} as const + +/** Trims and lowercases a required BCP47-style language token without guessing invalid empty values. */ +function normalizeLanguage(language: string): string { + const normalized = language.trim().toLowerCase() + if (!normalized) throw new TypeError('Language tag must not be empty.') + return normalized +} diff --git a/packages/rdf/jsonld/loader.ts b/packages/rdf/jsonld/loader.ts new file mode 100644 index 0000000..6041dde --- /dev/null +++ b/packages/rdf/jsonld/loader.ts @@ -0,0 +1,276 @@ +/** Bounded JSON-LD remote document loading. @module */ + +import type { DocumentLoaderType, JsonLdValueType, RemoteDocumentType } from './types.ts' + +/** Default max documents used when the caller does not provide an override. */ +const DEFAULT_MAX_DOCUMENTS = 32 +/** Default max bytes used when the caller does not provide an override. */ +const DEFAULT_MAX_BYTES = 2 * 1024 * 1024 +/** Default max redirects used when the caller does not provide an override. */ +const DEFAULT_MAX_REDIRECTS = 5 +/** Default timeout ms used when the caller does not provide an override. */ +const DEFAULT_TIMEOUT_MS = 10_000 +/** Link relation used by JSON-LD document loading to locate an external context. */ +const JSON_LD_CONTEXT_REL = 'http://www.w3.org/ns/json-ld#context' + +/** Shared cache contract for caller-owned remote JSON-LD documents. */ +export interface DocumentCacheType { + get(url: string): RemoteDocumentType | undefined + set(url: string, value: RemoteDocumentType): void +} + +/** Policy for one JSON-LD processing operation's remote-document loader. */ +export interface LoaderOptionsType { + /** Caller-owned loader. When omitted, network access remains disabled unless `remote` is true. */ + readonly loadDocument?: DocumentLoaderType + /** Fetch implementation for the built-in HTTP(S) loader. */ + readonly fetch?: typeof fetch + /** Explicitly allow built-in remote HTTP(S) loading. Defaults to false. */ + readonly remote?: boolean + /** URL policy checked before every request and redirect. */ + readonly allowUrl?: (url: URL) => boolean | Promise + /** Optional caller-owned cross-operation cache. */ + readonly cache?: DocumentCacheType + readonly maxDocuments?: number + readonly maxBytes?: number + readonly maxRedirects?: number + readonly timeoutMs?: number + readonly signal?: AbortSignal +} + +/** + * Creates one bounded, deduplicating document loader. + * + * Network loading is denied by default. Applications such as Kaiju should pass + * their existing crawl-aware loader or an explicit URL policy rather than let + * JSON-LD processing acquire arbitrary network access implicitly. + */ +export function createDocumentLoader(options: LoaderOptionsType = {}): DocumentLoaderType { + const maxDocuments = positive(options.maxDocuments ?? DEFAULT_MAX_DOCUMENTS, 'maxDocuments') + const maxBytes = positive(options.maxBytes ?? DEFAULT_MAX_BYTES, 'maxBytes') + const maxRedirects = nonNegative(options.maxRedirects ?? DEFAULT_MAX_REDIRECTS, 'maxRedirects') + const timeoutMs = nonNegative(options.timeoutMs ?? DEFAULT_TIMEOUT_MS, 'timeoutMs') + const inflight = new Map>() + const local = new Map() + let documents = 0 + + return async (url) => { + abort(options.signal) + const cached = local.get(url) ?? options.cache?.get(url) + if (cached) return cached + const pending = inflight.get(url) + if (pending) return await pending + if (++documents > maxDocuments) throw new JsonLdLoadError('document-limit', `JSON-LD remote document count exceeds maxDocuments (${maxDocuments}).`, url) + + const promise = load(url) + inflight.set(url, promise) + try { + const document = await promise + const bytes = measure(document.document) + if (bytes > maxBytes) throw new JsonLdLoadError('document-size', `JSON-LD remote document exceeds maxBytes (${maxBytes}).`, url) + local.set(url, document) + local.set(document.documentUrl, document) + options.cache?.set(url, document) + if (document.documentUrl !== url) options.cache?.set(document.documentUrl, document) + return document + } finally { + inflight.delete(url) + } + } + + /** Resolves one remote JSON-LD document through the bounded cache/deduplication and redirect policy. */ + async function load(url: string): Promise { + if (options.loadDocument) return await options.loadDocument(url) + if (!options.remote) throw new JsonLdLoadError('remote-disabled', 'Remote JSON-LD document loading is disabled.', url) + const fetchOptions: FetchOptionsType = { + fetch: options.fetch ?? fetch, + maxBytes, + maxRedirects, + timeoutMs, + ...(options.allowUrl ? { allowUrl: options.allowUrl } : {}), + ...(options.signal ? { signal: options.signal } : {}), + } + return await fetchDocument(url, fetchOptions) + } +} + +/** Stable failure from the bounded remote-document layer. */ +export class JsonLdLoadError extends Error { + readonly kind: 'remote-disabled' | 'url' | 'document-limit' | 'document-size' | 'redirect-limit' | 'http' | 'json' | 'timeout' | 'abort' + readonly url: string + + /** Creates a stable JSON-LD loading failure with the requested URL and underlying cause. */ + constructor(kind: JsonLdLoadError['kind'], message: string, url: string, cause?: unknown) { + super(message, cause === undefined ? undefined : { cause }) + this.name = 'JsonLdLoadError' + this.kind = kind + this.url = url + } +} + +/** Fully resolved remote-fetch policy passed through every redirect so limits and URL approval cannot be bypassed mid-chain. */ +interface FetchOptionsType { + readonly fetch: typeof fetch + readonly allowUrl?: (url: URL) => boolean | Promise + readonly maxBytes: number + readonly maxRedirects: number + readonly timeoutMs: number + readonly signal?: AbortSignal +} + +/** Fetches one JSON-LD document while applying redirect, byte, timeout, and URL policy. */ +async function fetchDocument(input: string, options: FetchOptionsType): Promise { + let current = toHttpUrl(input) + for (let redirects = 0; ; redirects++) { + abort(options.signal) + if (options.allowUrl && !(await options.allowUrl(current))) { + throw new JsonLdLoadError('url', `JSON-LD remote URL is not allowed: ${current.href}`, current.href) + } + + const timed = timeout(options.signal, options.timeoutMs) + let response: Response + try { + response = await options.fetch(current, { + headers: { Accept: 'application/ld+json, application/json;q=0.9' }, + redirect: 'manual', + signal: timed.signal, + }) + } catch (error) { + if (options.signal?.aborted) throw new JsonLdLoadError('abort', 'JSON-LD remote load was aborted.', current.href, error) + if (timed.expired()) throw new JsonLdLoadError('timeout', `JSON-LD remote load exceeded ${options.timeoutMs}ms.`, current.href, error) + throw error + } finally { + timed.dispose() + } + + if (response.status >= 300 && response.status < 400) { + if (redirects >= options.maxRedirects) throw new JsonLdLoadError('redirect-limit', `JSON-LD redirects exceed maxRedirects (${options.maxRedirects}).`, current.href) + const location = response.headers.get('location') + if (!location) throw new JsonLdLoadError('http', `JSON-LD redirect ${response.status} has no Location header.`, current.href) + current = toHttpUrl(new URL(location, current).href) + continue + } + if (!response.ok) throw new JsonLdLoadError('http', `JSON-LD remote load returned HTTP ${response.status}.`, current.href) + + const contentLength = Number(response.headers.get('content-length')) + if (Number.isFinite(contentLength) && contentLength > options.maxBytes) { + throw new JsonLdLoadError('document-size', `JSON-LD remote document exceeds maxBytes (${options.maxBytes}).`, current.href) + } + const bytes = new Uint8Array(await response.arrayBuffer()) + if (bytes.byteLength > options.maxBytes) throw new JsonLdLoadError('document-size', `JSON-LD remote document exceeds maxBytes (${options.maxBytes}).`, current.href) + + let document: JsonLdValueType + try { + document = JSON.parse(new TextDecoder('utf-8', { fatal: true }).decode(bytes)) as JsonLdValueType + } catch (error) { + throw new JsonLdLoadError('json', 'Remote JSON-LD document is not valid UTF-8 JSON.', current.href, error) + } + + return { + contextUrl: contextLink(response.headers.get('link'), response.headers.get('content-type'), current), + documentUrl: response.url || current.href, + document, + } + } +} + +/** Extracts the JSON-LD context Link relation used for non-JSON-LD response types. */ +function contextLink(header: string | null, contentType: string | null, base: URL): string | null { + if (!header || contentType?.toLowerCase().includes('application/ld+json')) return null + let context: string | null = null + for (const value of splitLinks(header)) { + const match = /^\s*<([^>]+)>\s*(.*)$/u.exec(value) + if (!match) continue + const parameters = match[2] ?? '' + const rel = /(?:^|;)\s*rel\s*=\s*(?:"([^"]*)"|([^;\s]+))/iu.exec(parameters) + const relations = (rel?.[1] ?? rel?.[2] ?? '').split(/\s+/u) + if (!relations.includes(JSON_LD_CONTEXT_REL)) continue + if (context !== null) throw new JsonLdLoadError('http', 'Remote document contains more than one JSON-LD context Link relation.', base.href) + context = new URL(match[1]!, base).href + } + return context +} + +/** Splits an HTTP Link field without treating commas inside quoted strings or IRIs as separators. */ +function splitLinks(value: string): string[] { + const result: string[] = [] + let start = 0 + let quoted = false + let angle = false + for (let index = 0; index < value.length; index++) { + const char = value[index]! + if (char === '"' && value[index - 1] !== '\\') quoted = !quoted + else if (!quoted && char === '<') angle = true + else if (!quoted && char === '>') angle = false + else if (!quoted && !angle && char === ',') { + result.push(value.slice(start, index)) + start = index + 1 + } + } + result.push(value.slice(start)) + return result +} + +/** Converts one input into an allowed built-in HTTP(S) URL. */ +function toHttpUrl(value: string): URL { + let url: URL + try { + url = new URL(value) + } catch (error) { + throw new JsonLdLoadError('url', `Invalid JSON-LD remote URL '${value}'.`, value, error) + } + if (url.protocol !== 'http:' && url.protocol !== 'https:') { + throw new JsonLdLoadError('url', `Unsupported JSON-LD remote URL protocol '${url.protocol}'.`, url.href) + } + return url +} + +/** Estimates already-loaded document bytes to enforce custom-loader output limits too. */ +function measure(value: JsonLdValueType): number { + return new TextEncoder().encode(JSON.stringify(value)).byteLength +} + +/** Creates a disposable timeout signal without transferring ownership of the caller signal. */ +function timeout(signal: AbortSignal | undefined, timeoutMs: number): { + readonly signal: AbortSignal + expired(): boolean + dispose(): void +} { + const controller = new AbortController() + let didExpire = false + const abort = () => controller.abort(signal?.reason) + if (signal?.aborted) abort() + else signal?.addEventListener('abort', abort, { once: true }) + const timer = timeoutMs > 0 + ? setTimeout(() => { + didExpire = true + controller.abort(new DOMException('Timed out', 'TimeoutError')) + }, timeoutMs) + : undefined + return { + signal: controller.signal, + expired: () => didExpire, + /** Releases the operation timeout when the composed loader signal is no longer needed. */ + dispose() { + if (timer !== undefined) clearTimeout(timer) + signal?.removeEventListener('abort', abort) + }, + } +} + +/** Throws the caller supplied abort reason when cancellation has been requested. */ +function abort(signal?: AbortSignal): void { + if (signal?.aborted) throw signal.reason ?? new DOMException('Aborted', 'AbortError') +} + +/** Validates a positive finite loader limit such as bytes, redirects, or context count. */ +function positive(value: number, name: string): number { + if (!Number.isSafeInteger(value) || value <= 0) throw new RangeError(`${name} must be a positive safe integer.`) + return value +} + +/** Validates a non-negative finite loader limit that may be explicitly disabled with zero. */ +function nonNegative(value: number, name: string): number { + if (!Number.isSafeInteger(value) || value < 0) throw new RangeError(`${name} must be a non-negative safe integer.`) + return value +} diff --git a/packages/rdf/jsonld/mod.ts b/packages/rdf/jsonld/mod.ts new file mode 100644 index 0000000..2091c9f --- /dev/null +++ b/packages/rdf/jsonld/mod.ts @@ -0,0 +1,132 @@ +/** Full JSON-LD processing facade with bounded document loading. @module */ + +import { parse as parseNQuads, write as writeNQuads } from '../nquads/mod.ts' +import type { Quad } from '../term.ts' +import { createDocumentLoader, type LoaderOptionsType } from './loader.ts' +import type { JsonLdValueType, ProcessorType } from './types.ts' + +export { createDocumentLoader, JsonLdLoadError } from './loader.ts' +export type { DocumentCacheType, LoaderOptionsType } from './loader.ts' +export type { DocumentLoaderType, JsonLdValueType, ProcessorType, RemoteDocumentType } from './types.ts' + +/** Default max rdf bytes used when the caller does not provide an override. */ +const DEFAULT_MAX_RDF_BYTES = 64 * 1024 * 1024 + +/** Processing options shared by JSON-LD operations. */ +export interface OptionsType extends LoaderOptionsType { + /** Optional processor injection for tests, custom builds, or alternate conforming engines. */ + readonly processor?: ProcessorType + /** Base IRI forwarded to the JSON-LD processor. */ + readonly base?: string + /** Maximum intermediary N-Quads bytes for JSON-LD/RDF conversion. */ + readonly maxRdfBytes?: number +} + +/** JSON text serialization options. */ +export interface SerializeOptionsType extends OptionsType { + readonly space?: number +} + +/** Expands JSON-LD using the JSON-LD 1.1 processing algorithm. */ +export async function expand(input: unknown, options: OptionsType = {}): Promise { + const { processor, settings } = await prepare(options) + const result = await processor.expand(input, settings) + abort(options.signal) + return result +} + +/** Compact as a focused public package operation. */ +export async function compact(input: unknown, context: unknown, options: OptionsType = {}): Promise { + const { processor, settings } = await prepare(options) + const result = await processor.compact(input, context, settings) + abort(options.signal) + return result +} + +/** Flattens JSON-LD, optionally compacting it with a context. */ +export async function flatten(input: unknown, context?: unknown, options: OptionsType = {}): Promise { + const { processor, settings } = await prepare(options) + const result = await processor.flatten(input, context, settings) + abort(options.signal) + return result +} + +/** Frames JSON-LD with one JSON-LD frame. */ +export async function frame(input: unknown, value: unknown, options: OptionsType = {}): Promise { + const { processor, settings } = await prepare(options) + const result = await processor.frame(input, value, settings) + abort(options.signal) + return result +} + +/** Converts JSON-LD into native `@okikio/rdf` quads through standards N-Quads. */ +export async function toRdf(input: unknown, options: OptionsType = {}): Promise { + const { processor, settings } = await prepare(options) + const result = await processor.toRDF(input, { ...settings, format: 'application/n-quads' }) + if (typeof result !== 'string') throw new TypeError('JSON-LD processor did not return N-Quads for application/n-quads.') + limit(result, options.maxRdfBytes ?? DEFAULT_MAX_RDF_BYTES) + const quads: Quad[] = [] + const parseOptions = options.signal ? { signal: options.signal } : {} + for await (const value of parseNQuads(result, parseOptions)) quads.push(value) + abort(options.signal) + return quads +} + +/** Parses JSON-LD and emits native RDF quads. Processing is materialized by jsonld.js before emission. */ +export async function* parse(input: unknown, options: OptionsType = {}): AsyncGenerator { + for (const value of await toRdf(input, options)) yield value +} + +/** Converts native RDF quads to expanded JSON-LD. */ +export async function fromRdf(source: Iterable, options: OptionsType = {}): Promise { + abort(options.signal) + const text = writeNQuads(source) + limit(text, options.maxRdfBytes ?? DEFAULT_MAX_RDF_BYTES) + const { processor, settings } = await prepare(options) + const result = await processor.fromRDF(text, { ...settings, format: 'application/n-quads' }) + abort(options.signal) + return result +} + +/** Serializes native RDF quads as JSON-LD JSON text. */ +export async function serialize(source: Iterable, options: SerializeOptionsType = {}): Promise { + return `${JSON.stringify(await fromRdf(source, options), null, options.space)}\n` +} + +/** Prepares one operation with an operation-local bounded document loader. */ +async function prepare(options: OptionsType): Promise<{ + readonly processor: ProcessorType + readonly settings: Readonly> +}> { + abort(options.signal) + const processor = options.processor ?? await defaultProcessor() + const documentLoader = createDocumentLoader(options) + return { + processor, + settings: options.base === undefined ? { documentLoader } : { base: options.base, documentLoader }, + } +} + +/** Lazily resolved JSON-LD processor shared across operations without making the RDF root import processor code. */ +let processorPromise: Promise | undefined + +/** Lazily imports jsonld.js only when this subpath actually performs processing. */ +async function defaultProcessor(): Promise { + processorPromise ??= import('jsonld').then((module) => { + const value = 'default' in module ? module.default : module + return value as unknown as ProcessorType + }) + return await processorPromise +} + +/** Increments one JSON-LD operation counter and fails before the configured work limit is exceeded. */ +function limit(value: string, maxBytes: number): void { + if (!Number.isSafeInteger(maxBytes) || maxBytes <= 0) throw new RangeError('maxRdfBytes must be a positive safe integer.') + const bytes = new TextEncoder().encode(value).byteLength + if (bytes > maxBytes) throw new RangeError(`JSON-LD RDF intermediary exceeds maxRdfBytes (${maxBytes}).`) +} + +/** Throws the caller supplied abort reason when cancellation has been requested. */ +function abort(signal?: AbortSignal): void { + if (signal?.aborted) throw signal.reason ?? new DOMException('Aborted', 'AbortError') +} diff --git a/packages/rdf/jsonld/mod_test.ts b/packages/rdf/jsonld/mod_test.ts new file mode 100644 index 0000000..8f4ea75 --- /dev/null +++ b/packages/rdf/jsonld/mod_test.ts @@ -0,0 +1,56 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { namedNode, quad } from '../mod.ts' +import { compact, createDocumentLoader, fromRdf, toRdf, type ProcessorType } from './mod.ts' + +const processor: ProcessorType = { + async expand(input) { return [input as never] }, + async compact(input, _context, options) { + expect(typeof options?.documentLoader).toBe('function') + return input as never + }, + async flatten(input) { return input as never }, + async frame(input) { return input as never }, + async toRDF() { return ' .\n' }, + async fromRDF(input) { + expect(typeof input).toBe('string') + return { '@id': 'https://example.com/s' } + }, +} + +describe('@okikio/rdf/jsonld', () => { + it('keeps remote document loading disabled unless the caller explicitly supplies or enables it', async () => { + const load = createDocumentLoader() + await expect(load('https://schema.org/')).rejects.toThrow('disabled') + }) + + it('deduplicates concurrent caller-owned remote document loads', async () => { + let loads = 0 + const load = createDocumentLoader({ + async loadDocument(url) { + loads++ + await Promise.resolve() + return { contextUrl: null, documentUrl: url, document: { '@context': {} } } + }, + }) + const [left, right] = await Promise.all([load('https://example.com/context'), load('https://example.com/context')]) + expect(loads).toBe(1) + expect(left).toBe(right) + }) + + it('owns JSON-LD/RDF conversion while preserving native RDF terms', async () => { + const values = await toRdf({ '@id': 'https://example.com/s' }, { processor }) + expect(values).toHaveLength(1) + expect(values[0]?.subject.value).toBe('https://example.com/s') + + const json = await fromRdf([ + quad(namedNode('https://example.com/s'), namedNode('https://example.com/p'), namedNode('https://example.com/o')), + ], { processor }) + expect(json).toEqual({ '@id': 'https://example.com/s' }) + }) + + it('passes the bounded loader to non-RDF JSON-LD operations too', async () => { + const value = { '@context': { name: 'https://schema.org/name' }, name: 'Widget' } + expect(await compact(value, {}, { processor })).toBe(value) + }) +}) diff --git a/packages/rdf/jsonld/types.ts b/packages/rdf/jsonld/types.ts new file mode 100644 index 0000000..1f18c5c --- /dev/null +++ b/packages/rdf/jsonld/types.ts @@ -0,0 +1,27 @@ +/** JSON-LD processor and remote-document contracts. @module */ + +/** JSON-compatible JSON-LD input/output value. */ +export type JsonLdValueType = null | boolean | number | string | JsonLdValueType[] | { [key: string]: JsonLdValueType } + +/** Remote document shape required by the JSON-LD processing algorithms. */ +export interface RemoteDocumentType { + readonly contextUrl: string | null + readonly documentUrl: string + readonly document: JsonLdValueType +} + +/** JSON-LD document loader compatible with jsonld.js. */ +export type DocumentLoaderType = ( + url: string, + options?: Readonly>, +) => Promise + +/** Minimal processing API used by `@okikio/rdf/jsonld`. */ +export interface ProcessorType { + expand(input: unknown, options?: Readonly>): Promise + compact(input: unknown, context: unknown, options?: Readonly>): Promise + flatten(input: unknown, context?: unknown, options?: Readonly>): Promise + frame(input: unknown, frame: unknown, options?: Readonly>): Promise + toRDF(input: unknown, options?: Readonly>): Promise + fromRDF(input: unknown, options?: Readonly>): Promise +} diff --git a/packages/rdf/line.ts b/packages/rdf/line.ts new file mode 100644 index 0000000..1011d95 --- /dev/null +++ b/packages/rdf/line.ts @@ -0,0 +1,462 @@ +/** Shared streaming scanner for RDF 1.2 N-Triples and N-Quads. @module */ + +import { blankNode, defaultGraph, literal, namedNode, quad } from './factory.ts' +import type { Graph, ObjectTerm, Predicate, Quad, Subject } from './term.ts' +import { chunks, throwIfAborted, type TextSource } from './text.ts' + +export type { TextSource } from './text.ts' + +/** Source range expressed in UTF-16 code-unit offsets and one-based line/column positions. */ +export interface SourceRange { + readonly start: number + readonly end: number + readonly line: number + readonly column: number +} + +/** Recoverable parser diagnostic. */ +export interface Diagnostic { + readonly code: string + readonly message: string + readonly range: SourceRange +} + +/** Parser controls for hostile input and tolerant analysis. */ +export interface ParseOptions { + readonly tolerant?: boolean + readonly maxLineLength?: number + readonly maxTripleDepth?: number + readonly signal?: AbortSignal +} + +/** RDF 1.2 version labels understood by the line syntaxes. */ +export type RdfVersion = '1.1' | '1.2-basic' | '1.2' + +/** Event stream emitted by N-Triples/N-Quads analysis. */ +export type ParseEvent = + | { readonly kind: 'quad'; readonly quad: Quad; readonly range: SourceRange } + | { readonly kind: 'version'; readonly version: RdfVersion; readonly range: SourceRange } + | { readonly kind: 'diagnostic'; readonly diagnostic: Diagnostic } + +/** Default max line length used when the caller does not provide an override. */ +const DEFAULT_MAX_LINE_LENGTH = 8 * 1024 * 1024 +/** Default max triple depth used when the caller does not provide an override. */ +const DEFAULT_MAX_TRIPLE_DEPTH = 64 + +/** Internal line record retaining absolute source offsets. */ +interface LineRecord { + /** Decoded source window that contains this logical line. */ + readonly source: string + /** Inclusive line start within `source`. */ + readonly from: number + /** Exclusive line end within `source`. */ + readonly to: number + readonly line: number + /** Absolute document offset corresponding to `from`. */ + readonly start: number +} + +/** Incrementally yields logical lines and cancels a Web Stream on early return. */ +export async function* lines(source: TextSource, options: ParseOptions = {}): AsyncGenerator { + const maxLineLength = options.maxLineLength ?? DEFAULT_MAX_LINE_LENGTH + const decoder = new TextDecoder() + let buffered = '' + let index = 0 + let base = 0 + let line = 1 + + /** Emits every complete line currently available without repeatedly slicing the unconsumed suffix. */ + function* drain(final: boolean): Generator { + while (true) { + const boundary = nextLineBreak(buffered, index, final) + if (!boundary) return + const length = boundary.index - index + if (length > maxLineLength) throw new SyntaxError(`RDF line exceeds maxLineLength (${maxLineLength}).`) + yield { source: buffered, from: index, to: boundary.index, line, start: base + index } + index = boundary.index + boundary.length + line++ + if (index >= 64 * 1024) { + buffered = buffered.slice(index) + base += index + index = 0 + } + } + } + + for await (const chunk of chunks(source, options.signal)) { + throwIfAborted(options.signal) + buffered += typeof chunk === 'string' ? chunk : decoder.decode(chunk, { stream: true }) + yield* drain(false) + + if (buffered.length - index > maxLineLength && !hasLineBreak(buffered, index)) { + throw new SyntaxError(`RDF line exceeds maxLineLength (${maxLineLength}).`) + } + } + + buffered += decoder.decode() + yield* drain(true) + const remaining = buffered.length - index + if (remaining > maxLineLength) throw new SyntaxError(`RDF line exceeds maxLineLength (${maxLineLength}).`) + if (remaining > 0) yield { source: buffered, from: index, to: buffered.length, line, start: base + index } +} + +/** Parses one N-Triples/N-Quads line into a semantic event. */ +export function parseLine(record: LineRecord, allowGraph: boolean, options: ParseOptions = {}): ParseEvent | undefined { + const cursor = new Cursor(record.source, record.from, record.to, record.start, record.line, options.maxTripleDepth ?? DEFAULT_MAX_TRIPLE_DEPTH) + cursor.space() + if (cursor.done || cursor.peek() === '#') return undefined + + const rangeStart = cursor.absolute + if (cursor.word('VERSION')) { + cursor.requiredSpace('Expected whitespace after VERSION.') + const value = cursor.string() + cursor.space() + cursor.commentOrEnd() + if (value !== '1.1' && value !== '1.2-basic' && value !== '1.2') { + throw cursor.error('rdf-version', `Unsupported RDF version '${value}'.`) + } + return { + kind: 'version', + version: value, + range: cursor.range(rangeStart, cursor.absolute), + } + } + + const subject = cursor.subject() + cursor.requiredSpace('Expected whitespace after RDF subject.') + const predicate = cursor.iri() as Predicate + cursor.requiredSpace('Expected whitespace after RDF predicate.') + const object = cursor.object(0) + cursor.space() + + let graph: Graph = defaultGraph() + if (allowGraph && cursor.peek() !== '.') { + graph = cursor.graph() + cursor.space() + } + + if (!cursor.take('.')) throw cursor.error('rdf-period', "Expected '.' after RDF statement.") + cursor.space() + cursor.commentOrEnd() + return { + kind: 'quad', + quad: quad(subject, predicate, object, graph), + range: cursor.range(rangeStart, cursor.absolute), + } +} + +/** Converts a parser exception into a source-ranged diagnostic. */ +export function diagnostic(error: unknown, record: LineRecord): Diagnostic { + if (error instanceof ParseError) return { code: error.code, message: error.message, range: error.range } + return { + code: 'rdf-syntax', + message: error instanceof Error ? error.message : String(error), + range: { start: record.start, end: record.start + record.to - record.from, line: record.line, column: 1 }, + } +} + +/** Position-aware parser error used internally and surfaced as diagnostics in tolerant mode. */ +class ParseError extends SyntaxError { + readonly code: string + readonly range: SourceRange + + /** Creates one source-ranged line-syntax failure for strict throwing or tolerant diagnostic conversion. */ + constructor(code: string, message: string, range: SourceRange) { + super(message) + this.name = 'RdfParseError' + this.code = code + this.range = range + } +} + +/** Data-oriented cursor over one line. It emits RDF terms directly and builds no token objects. */ +class Cursor { + #index: number + readonly source: string + readonly from: number + readonly to: number + readonly sourceStart: number + readonly line: number + readonly maxTripleDepth: number + + /** Creates a cursor over one logical RDF line with absolute source offsets and a bounded RDF 1.2 triple depth. */ + constructor(source: string, from: number, to: number, sourceStart: number, line: number, maxTripleDepth: number) { + this.source = source + this.from = from + this.to = to + this.#index = from + this.sourceStart = sourceStart + this.line = line + this.maxTripleDepth = maxTripleDepth + } + + /** Reports whether the cursor has consumed every code unit in the current logical line. */ + get done(): boolean { + return this.#index >= this.to + } + + /** Returns the absolute document offset corresponding to the current line-local cursor position. */ + get absolute(): number { + return this.sourceStart + this.#index - this.from + } + + /** Reads a code unit relative to the current cursor without advancing it. */ + peek(offset = 0): string | undefined { + const index = this.#index + offset + return index < this.to ? this.source[index] : undefined + } + + /** Consumes an exact lexical token only when it fits inside this logical line. */ + take(value: string): boolean { + if (this.#index + value.length > this.to || !this.source.startsWith(value, this.#index)) return false + this.#index += value.length + return true + } + + /** Tests a lexical prefix without reading past this logical line. */ + starts(value: string): boolean { + return this.#index + value.length <= this.to && this.source.startsWith(value, this.#index) + } + + /** Consumes a directive word only when followed by line whitespace or end-of-line, rewinding otherwise. */ + word(value: string): boolean { + const start = this.#index + if (!this.take(value)) return false + const next = this.peek() + if (next !== undefined && !/[ \t]/.test(next)) { + this.#index = start + return false + } + return true + } + + /** Consumes N-Triples/N-Quads horizontal whitespace. */ + space(): void { + while (this.peek() === ' ' || this.peek() === '\t') this.#index++ + } + + /** Requires at least one horizontal whitespace character between grammar terms. */ + requiredSpace(message: string): void { + const start = this.#index + this.space() + if (this.#index === start) throw this.error('rdf-whitespace', message) + } + + /** Reads a legal line-format RDF subject: IRI, blank node, or RDF 1.2 triple term where permitted. */ + subject(): Subject { + if (this.peek() === '<') return this.iri() + if (this.starts('_:')) return this.blank() + throw this.error('rdf-subject', 'Expected IRI or blank node as RDF subject.') + } + + /** Reads a legal N-Quads graph label without allowing the default graph token in source syntax. */ + graph(): Graph { + if (this.peek() === '<') return this.iri() + if (this.starts('_:')) return this.blank() + throw this.error('rdf-graph', 'Expected IRI or blank node as RDF graph label.') + } + + /** Reads an RDF object, including bounded RDF 1.2 nested triple terms. */ + object(depth: number): ObjectTerm { + if (depth > this.maxTripleDepth) { + throw this.error('rdf-depth', `Triple term nesting exceeds maxTripleDepth (${this.maxTripleDepth}).`) + } + if (this.starts('<<(')) return this.triple(depth + 1) + if (this.peek() === '<') return this.iri() + if (this.starts('_:')) return this.blank() + if (this.peek() === '"') return this.literal() + throw this.error('rdf-object', 'Expected IRI, blank node, literal, or RDF 1.2 triple term as object.') + } + + /** Decodes one `` while rejecting forbidden characters and invalid Unicode escapes. */ + iri(): ReturnType { + const start = this.#index + if (!this.take('<')) throw this.error('rdf-iri', "Expected '<' to start IRI.") + let value = '' + while (!this.done) { + const char = this.peek()! + if (char === '>') { + this.#index++ + return namedNode(value) + } + if (char === '\\') { + value += this.unicodeEscape() + continue + } + if (char <= ' ' || /[<>"{}|^`]/.test(char)) { + throw this.errorAt('rdf-iri-char', `Invalid character in IRI at column ${this.#index - this.from + 1}.`, start) + } + value += char + this.#index++ + } + throw this.errorAt('rdf-iri-end', 'Unterminated IRI.', start) + } + + /** Reads one blank-node label while keeping a trailing statement period outside the label. */ + blank(): ReturnType { + const start = this.#index + this.#index += 2 + const first = this.peek() + if (!first || !/[A-Za-z0-9_\u00C0-\uFFFF]/u.test(first)) { + throw this.errorAt('rdf-blank', 'Invalid blank-node label.', start) + } + let value = '' + while (!this.done) { + const char = this.peek()! + if (!/[A-Za-z0-9_.-\u00B7-\uFFFF]/u.test(char)) break + value += char + this.#index++ + } + if (value.endsWith('.')) { + this.#index-- + value = value.slice(0, -1) + } + return blankNode(value) + } + + /** Reads a lexical string plus datatype, language, and optional RDF 1.2 direction into one literal. */ + literal(): ReturnType { + const value = this.string() + if (this.take('^^')) return literal(value, this.iri()) + if (this.take('@')) { + const languageStart = this.#index + while (/[A-Za-z0-9-]/.test(this.peek() ?? '')) this.#index++ + const raw = this.source.slice(languageStart, this.#index) + if (!raw) throw this.error('rdf-language', 'Expected language tag after @.') + const marker = raw.lastIndexOf('--') + if (marker > 0) { + const language = raw.slice(0, marker) + const direction = raw.slice(marker + 2) + if (direction !== 'ltr' && direction !== 'rtl') { + throw this.error('rdf-direction', `Unsupported base direction '${direction}'.`) + } + return literal(value, { language, direction }) + } + return literal(value, raw) + } + return literal(value) + } + + /** Decodes one N-Triples quoted string and rejects raw line breaks. */ + string(): string { + const start = this.#index + if (!this.take('"')) throw this.error('rdf-string', 'Expected string literal.') + let value = '' + while (!this.done) { + const char = this.peek()! + if (char === '"') { + this.#index++ + return value + } + if (char === '\\') { + const escaped = this.peek(1) + if (escaped === 't' || escaped === 'b' || escaped === 'n' || escaped === 'r' || escaped === 'f' || escaped === '"' || escaped === "'" || escaped === '\\') { + this.#index += 2 + value += escapeValue(escaped) + continue + } + value += this.unicodeEscape() + continue + } + if (char === '\n' || char === '\r') throw this.errorAt('rdf-string-line', 'Line break is not allowed in an N-Triples literal.', start) + value += char + this.#index++ + } + throw this.errorAt('rdf-string-end', 'Unterminated string literal.', start) + } + + /** Reads RDF 1.2 `<<( subject predicate object )>>` recursively under `maxTripleDepth`. */ + triple(depth: number): Quad { + const start = this.#index + this.#index += 3 + this.space() + const subject = this.subject() + this.requiredSpace('Expected whitespace in triple term after subject.') + const predicate = this.iri() as Predicate + this.requiredSpace('Expected whitespace in triple term after predicate.') + const object = this.object(depth) + this.space() + if (!this.take(')>>')) throw this.errorAt('rdf-triple-end', "Expected ')>>' after triple term.", start) + return quad(subject, predicate, object) + } + + /** Decodes `\u`/`\U` escapes only when the result is a Unicode scalar value. */ + unicodeEscape(): string { + const start = this.#index + this.#index++ + const kind = this.peek() + if (kind !== 'u' && kind !== 'U') throw this.errorAt('rdf-escape', 'Expected Unicode escape.', start) + this.#index++ + const width = kind === 'u' ? 4 : 8 + const text = this.source.slice(this.#index, Math.min(this.#index + width, this.to)) + if (!new RegExp(`^[0-9A-Fa-f]{${width}}$`).test(text)) { + throw this.errorAt('rdf-unicode', 'Invalid Unicode escape.', start) + } + this.#index += width + const codePoint = Number.parseInt(text, 16) + if (codePoint > 0x10ffff || (codePoint >= 0xd800 && codePoint <= 0xdfff)) { + throw this.errorAt('rdf-unicode', 'Unicode escape is not a Unicode scalar value.', start) + } + return String.fromCodePoint(codePoint) + } + + /** Accepts only a trailing comment or logical end after the statement period. */ + commentOrEnd(): void { + if (this.peek() === '#') { + this.#index = this.to + return + } + if (!this.done) throw this.error('rdf-trailing', 'Unexpected content after RDF statement.') + } + + /** Converts absolute offsets into a one-line source range with the correct one-based column. */ + range(start: number, end: number): SourceRange { + return { start, end, line: this.line, column: start - this.sourceStart + 1 } + } + + /** Creates a one-code-unit parse error at the current cursor. */ + error(code: string, message: string): ParseError { + return new ParseError(code, message, this.range(this.absolute, this.absolute + 1)) + } + + /** Creates a parse error spanning a saved lexical start through the current cursor. */ + errorAt(code: string, message: string, start: number): ParseError { + return new ParseError(code, message, this.range(this.sourceStart + start, this.absolute + 1)) + } +} + +/** Returns whether the unconsumed source contains a complete line separator. */ +function hasLineBreak(value: string, start: number): boolean { + return value.indexOf('\n', start) !== -1 || value.indexOf('\r', start) !== -1 +} + +/** Finds the next line separator without rescanning or copying the consumed prefix. */ +function nextLineBreak( + value: string, + start: number, + final: boolean, +): { readonly index: number; readonly length: number } | undefined { + const lf = value.indexOf('\n', start) + const cr = value.indexOf('\r', start) + if (lf === -1 && cr === -1) return undefined + if (cr !== -1 && (lf === -1 || cr < lf)) { + if (!final && cr + 1 === value.length) return undefined + return { index: cr, length: value[cr + 1] === '\n' ? 2 : 1 } + } + return { index: lf, length: 1 } +} + +/** Maps an N-Triples single-character escape to its decoded code point. */ +function escapeValue(value: string): string { + switch (value) { + case 't': return '\t' + case 'b': return '\b' + case 'n': return '\n' + case 'r': return '\r' + case 'f': return '\f' + case '"': return '"' + case "'": return "'" + case '\\': return '\\' + default: return value + } +} diff --git a/packages/rdf/microdata/mod.ts b/packages/rdf/microdata/mod.ts new file mode 100644 index 0000000..cd7737c --- /dev/null +++ b/packages/rdf/microdata/mod.ts @@ -0,0 +1,64 @@ +/** HTML Microdata-to-RDF parsing behind native RDF and Web-oriented source contracts. @module */ + +import { factory } from '../factory.ts' +import type { Graph, Quad } from '../term.ts' +import { throwIfAborted, type TextSource } from '../text.ts' +import { parseTransform } from '../transform.ts' +import type { ParserConstructorType } from './types.ts' + +export type { ParserConstructorType, ParserType } from './types.ts' + +/** Vocabulary-registry entry used by the Microdata-to-RDF conversion algorithm. */ +export interface VocabularyType { + readonly properties?: Readonly>>> + readonly [key: string]: unknown +} + +/** Microdata vocabulary registry keyed by vocabulary IRI prefix. */ +export type VocabularyRegistryType = Readonly> + +/** Options for Microdata-to-RDF parsing. */ +export interface ParseOptionsType { + readonly base?: string + readonly graph?: Graph + /** Parse the input as strict XML/XHTML instead of HTML. */ + readonly xml?: boolean + /** Replaces the processor's standard Microdata vocabulary registry when supplied. */ + readonly vocabularies?: VocabularyRegistryType + /** External parser injection used by tests or alternate conforming implementations. */ + readonly parser?: ParserConstructorType + readonly signal?: AbortSignal +} + +/** + * Parses HTML Microdata incrementally into native `@okikio/rdf` quads. + * + * Conversion follows the W3C Microdata-to-RDF algorithm implemented by the + * external parser. The external Transform and HTML parser remain subpath-only. + */ +export async function* parse(source: TextSource, options: ParseOptionsType = {}): AsyncGenerator { + throwIfAborted(options.signal) + const Parser = options.parser ?? await defaultParser() + const parser = new Parser(parserOptions(options)) + yield* parseTransform(parser, source, { label: 'Microdata parser', ...(options.signal ? { signal: options.signal } : {}) }) +} + +/** Builds the upstream Microdata options while omitting absent optional fields. */ +function parserOptions(options: ParseOptionsType): Readonly> { + return { + dataFactory: factory, + xmlMode: options.xml ?? false, + ...(options.base === undefined ? {} : { baseIRI: options.base }), + ...(options.graph === undefined ? {} : { defaultGraph: options.graph }), + ...(options.vocabularies === undefined ? {} : { vocabRegistry: options.vocabularies }), + } +} + +/** Lazily resolved Microdata parser constructor so importing the subpath does not initialize the optional processor. */ +let parserPromise: Promise | undefined + +/** Lazily imports the Microdata implementation only when this subpath is used. */ +async function defaultParser(): Promise { + parserPromise ??= import('microdata-rdf-streaming-parser').then((module) => module.MicrodataRdfParser as unknown as ParserConstructorType) + return await parserPromise +} diff --git a/packages/rdf/microdata/mod_test.ts b/packages/rdf/microdata/mod_test.ts new file mode 100644 index 0000000..07e1c27 --- /dev/null +++ b/packages/rdf/microdata/mod_test.ts @@ -0,0 +1,49 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { literal, namedNode, quad, type Quad } from '../mod.ts' +import { parse, type ParserType } from './mod.ts' + +const fixture = quad(namedNode('https://example.test/s'), namedNode('https://example.test/p'), literal('value')) +type EventType = 'drain' | 'close' | 'error' +type ListenerType = (...args: unknown[]) => void + +class TestParser implements ParserType { + static options: Readonly> | undefined + private readonly listeners = new Map>() + private ended = false + private resolveEnd: (() => void) | undefined + constructor(options: Readonly>) { TestParser.options = options } + write(_value: string | Uint8Array): boolean { return true } + end(): void { this.ended = true; this.resolveEnd?.() } + destroy(error?: Error): void { if (error) this.emit('error', error); this.ended = true; this.resolveEnd?.(); this.emit('close') } + once(event: EventType, listener: ListenerType): this { const values = this.listeners.get(event) ?? new Set(); values.add(listener); this.listeners.set(event, values); return this } + off(event: EventType, listener: ListenerType): this { this.listeners.get(event)?.delete(listener); return this } + private emit(event: EventType, ...args: unknown[]): void { for (const listener of this.listeners.get(event) ?? []) listener(...args) } + async *[Symbol.asyncIterator](): AsyncIterator { if (!this.ended) await new Promise((resolve) => { this.resolveEnd = resolve }); yield fixture } +} + +describe('@okikio/rdf/microdata', () => { + it('adapts Microdata parser output back to native RDF terms', async () => { + const values: Quad[] = [] + for await (const value of parse('
', { parser: TestParser })) values.push(value) + expect(values).toHaveLength(1) + expect(values[0]?.equals(fixture)).toBe(true) + }) + + it('forwards Microdata base, XML mode, vocabulary registry, and native RDF factory', async () => { + const vocabularies = { 'https://schema.org/': {} } + for await (const _value of parse('
', { + parser: TestParser, + base: 'https://example.test/base/', + xml: true, + vocabularies, + })) { /* drain */ } + + expect(TestParser.options?.baseIRI).toBe('https://example.test/base/') + expect(TestParser.options?.xmlMode).toBe(true) + expect(TestParser.options?.vocabRegistry).toBe(vocabularies) + const factory = TestParser.options?.dataFactory as { namedNode(value: string): { value: string } } + expect(factory.namedNode('urn:test').value).toBe('urn:test') + }) + +}) diff --git a/packages/rdf/microdata/types.ts b/packages/rdf/microdata/types.ts new file mode 100644 index 0000000..966a2d5 --- /dev/null +++ b/packages/rdf/microdata/types.ts @@ -0,0 +1,11 @@ +/** Structural Microdata parser contracts used to isolate the external implementation. @module */ + +import type { TransformParserType } from '../transform.ts' + +/** Microdata parser stream shape required by the adapter. */ +export type ParserType = TransformParserType + +/** Constructor contract for a Microdata parser implementation. */ +export interface ParserConstructorType { + new (options: Readonly>): ParserType +} diff --git a/packages/rdf/mod.ts b/packages/rdf/mod.ts new file mode 100644 index 0000000..d3a66f3 --- /dev/null +++ b/packages/rdf/mod.ts @@ -0,0 +1,45 @@ +/** + * Complete RDF programming model for TypeScript runtimes. + * + * The root keeps the core term, dataset, namespace, and source contracts light. + * Concrete syntaxes live on explicit subpaths so importing `@okikio/rdf` does + * not initialize parser-specific state. + * + * @example + * ```ts + * import * as rdf from '@okikio/rdf' + * + * const subject = rdf.namedNode('https://example.com/product/1') + * const predicate = rdf.namedNode('https://schema.org/name') + * const graph = rdf.dataset([ + * rdf.quad(subject, predicate, rdf.literal('Widget')), + * ]) + * ``` + * + * @module + */ + +export { dataset, Dataset, datasetEquals, datasetKey } from './dataset.ts' +export type { MatchOptions } from './dataset.ts' +export { blankNode, defaultGraph, factory, fromQuad, fromTerm, literal, namedNode, quad, triple, variable } from './factory.ts' +export { namespace } from './namespace.ts' +export type { Namespace } from './namespace.ts' +export { equals, isTerm, key, RDF, XSD } from './term.ts' +export type { + BlankNode, + DefaultGraph, + Direction, + DirectionalLanguage, + Graph, + Literal, + NamedNode, + ObjectTerm, + Predicate, + Quad, + Subject, + Term, + TermType, + Variable, +} from './term.ts' +export { iterate } from './source.ts' +export type { AsyncSource, Sink, Source } from './source.ts' diff --git a/packages/rdf/namespace.ts b/packages/rdf/namespace.ts new file mode 100644 index 0000000..9e2ad55 --- /dev/null +++ b/packages/rdf/namespace.ts @@ -0,0 +1,17 @@ +/** Namespace helpers for building RDF named nodes without global registration. @module */ + +import { namedNode } from './factory.ts' +import type { NamedNode } from './term.ts' + +/** A callable namespace that expands local names into RDF named nodes. */ +export interface Namespace { + (local: string): NamedNode + readonly iri: string +} + +/** Creates an import-safe namespace expansion function. */ +export function namespace(iri: string): Namespace { + const expand = ((local: string) => namedNode(`${iri}${local}`)) as Namespace + Object.defineProperty(expand, 'iri', { value: iri, enumerable: true }) + return expand +} diff --git a/packages/rdf/namespace_test.ts b/packages/rdf/namespace_test.ts new file mode 100644 index 0000000..96f2953 --- /dev/null +++ b/packages/rdf/namespace_test.ts @@ -0,0 +1,12 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { namespace } from './namespace.ts' + +describe('@okikio/rdf namespaces', () => { + it('keeps the base IRI inspectable while expanding local names to named nodes', () => { + const schema = namespace('https://schema.org/') + expect(schema.iri).toBe('https://schema.org/') + expect(schema('Product').value).toBe('https://schema.org/Product') + expect(Object.keys(schema)).toEqual(['iri']) + }) +}) diff --git a/packages/rdf/nquads/mod.ts b/packages/rdf/nquads/mod.ts new file mode 100644 index 0000000..127eae7 --- /dev/null +++ b/packages/rdf/nquads/mod.ts @@ -0,0 +1,31 @@ +/** RDF 1.2 N-Quads parser and serializer. @module */ + +import { diagnostic, lines, parseLine, type ParseEvent, type ParseOptions, type TextSource } from '../line.ts' +import type { Quad } from '../term.ts' +import { writeQuad } from '../write.ts' + +/** Emits source-ranged N-Quads semantic events. */ +export async function* analyze(source: TextSource, options: ParseOptions = {}): AsyncGenerator { + for await (const record of lines(source, options)) { + try { + const event = parseLine(record, true, options) + if (event) yield event + } catch (error) { + if (!options.tolerant) throw error + yield { kind: 'diagnostic', diagnostic: diagnostic(error, record) } + } + } +} + +/** Parses N-Quads incrementally. */ +export async function* parse(source: TextSource, options: ParseOptions = {}): AsyncGenerator { + for await (const event of analyze(source, options)) if (event.kind === 'quad') yield event.quad +} + +/** Serializes RDF quads using canonical-layout-compatible line formatting. */ +export function write(quads: Iterable): string { + const output = [...quads].map((quad) => writeQuad(quad, true)) + return output.length === 0 ? '' : `${output.join('\n')}\n` +} + +export type { Diagnostic, ParseEvent, ParseOptions, SourceRange, TextSource } from '../line.ts' diff --git a/packages/rdf/nquads/parse_bench.ts b/packages/rdf/nquads/parse_bench.ts new file mode 100644 index 0000000..2fbd71d --- /dev/null +++ b/packages/rdf/nquads/parse_bench.ts @@ -0,0 +1,46 @@ +/** Decision benchmark for N-Quads whole-source versus chunked streaming overhead. @module */ + +import { bench, do_not_optimize, group, run } from 'mitata' +import { datasetKey } from '../dataset.ts' +import { parse } from './mod.ts' + +const COUNT = 10_000 +const text = Array.from( + { length: COUNT }, + (_, index) => ` "value-${index}" .`, +).join('\n') +const chunks = split(text, 4096) +const expected = await read(text) +const expectedDigest = datasetKey(expected) +const chunked = await read(chunks) +if (chunked.length !== COUNT || datasetKey(chunked) !== expectedDigest) { + throw new Error('Chunked N-Quads benchmark oracle does not match whole-source output.') +} + +/** Splits one deterministic fixture without adding work to the timed callback. */ +function split(value: string, size: number): string[] { + const result: string[] = [] + for (let offset = 0; offset < value.length; offset += size) result.push(value.slice(offset, offset + size)) + return result +} + +/** Fully consumes one parser result so parser work cannot be optimized away. */ +async function read(source: string | Iterable) { + const result = [] + for await (const value of parse(source)) result.push(value) + return result +} + +group('N-Quads parse: 10k quads', () => { + bench('whole string', async () => { + const values = await read(text) + do_not_optimize(values.length) + }).gc('inner') + + bench('4 KiB chunks', async () => { + const values = await read(chunks) + do_not_optimize(values.length) + }).gc('inner') +}) + +await run() diff --git a/packages/rdf/nquads/parse_test.ts b/packages/rdf/nquads/parse_test.ts new file mode 100644 index 0000000..bcf6aee --- /dev/null +++ b/packages/rdf/nquads/parse_test.ts @@ -0,0 +1,58 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { analyze, parse, write } from './mod.ts' +import type { Quad } from '../mod.ts' + +describe('@okikio/rdf/nquads', () => { + it('round-trips RDF 1.2 directional literals and triple terms', async () => { + const source = [ + 'VERSION "1.2"', + ' "bonjour"@fr--ltr .', + ' <<( "x" )>> .', + '', + ].join('\n') + + const quads: Quad[] = [] + for await (const quad of parse(source)) quads.push(quad) + expect(quads.length).toBe(2) + expect(quads[0]?.object.termType).toBe('Literal') + if (quads[0]?.object.termType === 'Literal') { + expect(quads[0].object.language).toBe('fr') + expect(quads[0].object.direction).toBe('ltr') + } + expect(quads[1]?.object.termType).toBe('Quad') + + const reparsed: Quad[] = [] + for await (const quad of parse(write(quads))) reparsed.push(quad) + expect(quads.every((quad, index) => quad.equals(reparsed[index]!))).toBe(true) + }) + + it('emits a source-ranged diagnostic and resumes at the next record in tolerant mode', async () => { + const source = ' "ok" .\nnot rdf\n "ok" .\n' + const events = [] + for await (const event of analyze(source, { tolerant: true })) events.push(event) + expect(events.map((event) => event.kind)).toEqual(['quad', 'diagnostic', 'quad']) + const problem = events[1] + expect(problem?.kind).toBe('diagnostic') + if (problem?.kind === 'diagnostic') expect(problem.diagnostic.range.line).toBe(2) + }) + + it('cancels a ReadableStream when the consumer returns early', async () => { + let cancelled = false + const bytes = new TextEncoder().encode( + ' "one" .\n "two" .\n', + ) + const stream = new ReadableStream({ + start(controller) { + controller.enqueue(bytes.slice(0, bytes.length / 2)) + controller.enqueue(bytes.slice(bytes.length / 2)) + }, + cancel() { + cancelled = true + }, + }) + + for await (const _quad of parse(stream)) break + expect(cancelled).toBe(true) + }) +}) diff --git a/packages/rdf/ntriples/mod.ts b/packages/rdf/ntriples/mod.ts new file mode 100644 index 0000000..dc4ca8b --- /dev/null +++ b/packages/rdf/ntriples/mod.ts @@ -0,0 +1,35 @@ +/** RDF 1.2 N-Triples parser and serializer. @module */ + +import { diagnostic, lines, parseLine, type ParseEvent, type ParseOptions, type TextSource } from '../line.ts' +import type { Quad } from '../term.ts' +import { writeQuad } from '../write.ts' + +/** Emits source-ranged semantic events. Tolerant mode reports malformed lines and continues. */ +export async function* analyze(source: TextSource, options: ParseOptions = {}): AsyncGenerator { + for await (const record of lines(source, options)) { + try { + const event = parseLine(record, false, options) + if (event) yield event + } catch (error) { + if (!options.tolerant) throw error + yield { kind: 'diagnostic', diagnostic: diagnostic(error, record) } + } + } +} + +/** Parses N-Triples incrementally and yields RDF quads in the default graph. */ +export async function* parse(source: TextSource, options: ParseOptions = {}): AsyncGenerator { + for await (const event of analyze(source, options)) if (event.kind === 'quad') yield event.quad +} + +/** Serializes RDF triples using canonical-layout-compatible line formatting. */ +export function write(quads: Iterable): string { + const lines: string[] = [] + for (const quad of quads) { + if (quad.graph.termType !== 'DefaultGraph') throw new TypeError('N-Triples cannot serialize named graphs.') + lines.push(writeQuad(quad, false)) + } + return lines.length === 0 ? '' : `${lines.join('\n')}\n` +} + +export type { Diagnostic, ParseEvent, ParseOptions, SourceRange, TextSource } from '../line.ts' diff --git a/packages/rdf/ntriples/parse_test.ts b/packages/rdf/ntriples/parse_test.ts new file mode 100644 index 0000000..4e8991e --- /dev/null +++ b/packages/rdf/ntriples/parse_test.ts @@ -0,0 +1,25 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { namedNode, quad, literal, type Quad } from '../mod.ts' +import { parse, write } from './mod.ts' + +describe('@okikio/rdf/ntriples', () => { + it('parses triples into the default graph and round-trips them', async () => { + const source = ' "value" .\n' + const values: Quad[] = [] + for await (const value of parse(source)) values.push(value) + expect(values.length).toBe(1) + expect(values[0]?.graph.termType).toBe('DefaultGraph') + expect(write(values)).toBe(source) + }) + + it('rejects named graphs during serialization', () => { + const value = quad( + namedNode('https://example/s'), + namedNode('https://example/p'), + literal('value'), + namedNode('https://example/g'), + ) + expect(() => write([value])).toThrow('N-Triples cannot serialize named graphs') + }) +}) diff --git a/packages/rdf/ontology/index.ts b/packages/rdf/ontology/index.ts new file mode 100644 index 0000000..56738a3 --- /dev/null +++ b/packages/rdf/ontology/index.ts @@ -0,0 +1,66 @@ +/** Cycle-safe lookup and transitive hierarchy operations for ontology models. @module */ + +import type { ClassType, ModelType, PropertyType } from './model.ts' + +/** Read-only indexes over one ontology model. */ +export class OntologyIndex { + readonly #classes: ReadonlyMap + readonly #properties: ReadonlyMap + + /** Builds immutable lookup maps over a parsed ontology model without performing entailment. */ + constructor(model: ModelType) { + this.#classes = new Map(model.classes.map((value) => [value.iri, value])) + this.#properties = new Map(model.properties.map((value) => [value.iri, value])) + } + + /** Gets one named class declaration. */ + getClass(iri: string): ClassType | undefined { + return this.#classes.get(iri) + } + + /** Gets one named property declaration. */ + getProperty(iri: string): PropertyType | undefined { + return this.#properties.get(iri) + } + + /** Returns transitive superclasses without duplicating cycles. */ + superClasses(iri: string): readonly string[] { + return closure(iri, (value) => this.#classes.get(value)?.superClasses ?? []) + } + + /** Returns transitive superproperties without duplicating cycles. */ + superProperties(iri: string): readonly string[] { + return closure(iri, (value) => this.#properties.get(value)?.superProperties ?? []) + } + + /** + * Returns properties whose declared domain contains a class or superclass. + * + * This is a vocabulary lookup operation, not a claim that the property is + * required or that its range forms a closed validation rule. + */ + propertiesForClass(iri: string): readonly PropertyType[] { + const classes = new Set([iri, ...this.superClasses(iri)]) + return [...this.#properties.values()] + .filter((property) => property.domains.some((domain) => classes.has(domain))) + .sort((left, right) => left.iri.localeCompare(right.iri)) + } +} + +/** Creates hierarchy indexes without changing the ontology model. */ +export function index(model: ModelType): OntologyIndex { + return new OntologyIndex(model) +} + +/** Computes cycle-safe transitive closure over one named ontology relationship map. */ +function closure(start: string, getParents: (value: string) => readonly string[]): string[] { + const result = new Set() + const pending = [...getParents(start)] + while (pending.length) { + const value = pending.pop()! + if (value === start || result.has(value)) continue + result.add(value) + pending.push(...getParents(value)) + } + return [...result].sort() +} diff --git a/packages/rdf/ontology/mod.ts b/packages/rdf/ontology/mod.ts new file mode 100644 index 0000000..e0d879d --- /dev/null +++ b/packages/rdf/ontology/mod.ts @@ -0,0 +1,24 @@ +/** + * Generic RDFS and OWL ontology interpretation helpers. + * + * The reader records named declarations and relationships without performing + * entailment. Anonymous OWL expressions and unsupported axioms are retained as + * assertions so a reasoner or future parser can interpret them later. + * + * @module + */ + +export { index, OntologyIndex } from './index.ts' +export { read } from './read.ts' +export type { OntologySourceType, ReadOptions } from './read.ts' +export type { + AssertionType, + ClassType, + DiagnosticType, + ModelType, + PropertyCharacteristicType, + PropertyKindType, + PropertyType, + SourceType, + TextType, +} from './model.ts' diff --git a/packages/rdf/ontology/model.ts b/packages/rdf/ontology/model.ts new file mode 100644 index 0000000..0bd65f0 --- /dev/null +++ b/packages/rdf/ontology/model.ts @@ -0,0 +1,92 @@ +/** + * Serializable RDFS/OWL ontology model. + * + * This model represents named ontology declarations and relationships that can + * be interpreted without OWL reasoning. Anonymous class expressions and other + * unsupported axioms remain in {@link AssertionType} so later OWL layers do not + * lose the source graph. + * + * @module + */ + +/** Localized ontology text. */ +export interface TextType { + readonly value: string + readonly language?: string + readonly direction?: 'ltr' | 'rtl' +} + +/** Source metadata attached to ontology assertions. */ +export interface SourceType { + readonly id: string + readonly iri?: string + readonly version?: string + readonly hash?: string +} + +/** Named ontology class and directly interpretable RDFS/OWL relationships. */ +export interface ClassType { + readonly iri: string + readonly labels: readonly TextType[] + readonly comments: readonly TextType[] + readonly superClasses: readonly string[] + readonly equivalentClasses: readonly string[] + readonly disjointClasses: readonly string[] + readonly deprecated: boolean +} + +/** Declared property flavor. A property can have more than one RDF/OWL type. */ +export type PropertyKindType = 'rdf' | 'object' | 'data' | 'annotation' + +/** OWL characteristics that can be stated directly on a named property. */ +export type PropertyCharacteristicType = + | 'functional' + | 'inverseFunctional' + | 'transitive' + | 'symmetric' + | 'asymmetric' + | 'reflexive' + | 'irreflexive' + +/** Named RDF/OWL property. */ +export interface PropertyType { + readonly iri: string + readonly kinds: readonly PropertyKindType[] + readonly labels: readonly TextType[] + readonly comments: readonly TextType[] + readonly domains: readonly string[] + readonly ranges: readonly string[] + readonly superProperties: readonly string[] + readonly equivalentProperties: readonly string[] + readonly inverseOf: readonly string[] + readonly disjointProperties: readonly string[] + readonly characteristics: readonly PropertyCharacteristicType[] + readonly deprecated: boolean +} + +/** RDF assertion retained when this reader does not interpret it. */ +export interface AssertionType { + readonly sourceId: string + readonly subject: string + readonly predicate: string + readonly object: string + readonly graph: string +} + +/** Ontology-reader problem that does not require discarding the source graph. */ +export interface DiagnosticType { + readonly code: 'unclassified-text' | 'unclassified-deprecation' | 'invalid-deprecation' + readonly message: string + readonly sourceId: string + readonly term?: string +} + +/** Stable ontology intermediate representation. */ +export interface ModelType { + readonly sources: readonly SourceType[] + readonly classes: readonly ClassType[] + readonly properties: readonly PropertyType[] + readonly datatypes: readonly string[] + readonly assertions: readonly AssertionType[] + readonly diagnostics: readonly DiagnosticType[] +} diff --git a/packages/rdf/ontology/read.ts b/packages/rdf/ontology/read.ts new file mode 100644 index 0000000..f3b677b --- /dev/null +++ b/packages/rdf/ontology/read.ts @@ -0,0 +1,474 @@ +/** RDFS and directly interpretable OWL ontology reader. @module */ + +import { iterate } from '../source.ts' +import { key, XSD } from '../term.ts' +import type { Literal, Quad } from '../term.ts' +import type { + AssertionType, + ClassType, + DiagnosticType, + ModelType, + PropertyCharacteristicType, + PropertyKindType, + PropertyType, + SourceType, + TextType, +} from './model.ts' + +/** `rdf:type` predicate used to classify ontology resources. */ +const RDF_TYPE = 'http://www.w3.org/1999/02/22-rdf-syntax-ns#type' +/** `rdf:Property` class used to recognize generic RDF properties. */ +const RDF_PROPERTY = 'http://www.w3.org/1999/02/22-rdf-syntax-ns#Property' +/** `rdfs:Class` IRI used to recognize declared RDFS classes. */ +const RDFS_CLASS = 'http://www.w3.org/2000/01/rdf-schema#Class' +/** `rdfs:Datatype` IRI used to recognize declared RDF datatypes. */ +const RDFS_DATATYPE = 'http://www.w3.org/2000/01/rdf-schema#Datatype' +/** `rdfs:subClassOf` predicate used to build direct class inheritance edges. */ +const RDFS_SUBCLASS = 'http://www.w3.org/2000/01/rdf-schema#subClassOf' +/** `rdfs:subPropertyOf` predicate used to build direct property inheritance edges. */ +const RDFS_SUBPROPERTY = 'http://www.w3.org/2000/01/rdf-schema#subPropertyOf' +/** `rdfs:domain` predicate retained as ontology inference metadata, not JSON requiredness. */ +const RDFS_DOMAIN = 'http://www.w3.org/2000/01/rdf-schema#domain' +/** `rdfs:range` predicate retained as ontology inference/range metadata. */ +const RDFS_RANGE = 'http://www.w3.org/2000/01/rdf-schema#range' +/** `rdfs:label` predicate used for human-readable ontology names. */ +const RDFS_LABEL = 'http://www.w3.org/2000/01/rdf-schema#label' +/** `rdfs:comment` predicate used for generated documentation when available. */ +const RDFS_COMMENT = 'http://www.w3.org/2000/01/rdf-schema#comment' +/** OWL namespace prefix used to derive directly interpreted OWL terms. */ +const OWL = 'http://www.w3.org/2002/07/owl#' +/** `owl:Class` IRI used to recognize declared OWL classes. */ +const OWL_CLASS = `${OWL}Class` +/** `owl:ObjectProperty` class mapped to the ontology object-property kind. */ +const OWL_OBJECT_PROPERTY = `${OWL}ObjectProperty` +/** `owl:DatatypeProperty` class mapped to the ontology data-property kind. */ +const OWL_DATATYPE_PROPERTY = `${OWL}DatatypeProperty` +/** `owl:AnnotationProperty` class mapped to the ontology annotation-property kind. */ +const OWL_ANNOTATION_PROPERTY = `${OWL}AnnotationProperty` +/** OWL class that marks a property functional. */ +const OWL_FUNCTIONAL_PROPERTY = `${OWL}FunctionalProperty` +/** OWL class that marks a property inverse-functional. */ +const OWL_INVERSE_FUNCTIONAL_PROPERTY = `${OWL}InverseFunctionalProperty` +/** OWL class that marks a property transitive. */ +const OWL_TRANSITIVE_PROPERTY = `${OWL}TransitiveProperty` +/** OWL class that marks a property symmetric. */ +const OWL_SYMMETRIC_PROPERTY = `${OWL}SymmetricProperty` +/** OWL class that marks a property asymmetric. */ +const OWL_ASYMMETRIC_PROPERTY = `${OWL}AsymmetricProperty` +/** OWL class that marks a property reflexive. */ +const OWL_REFLEXIVE_PROPERTY = `${OWL}ReflexiveProperty` +/** OWL class that marks a property irreflexive. */ +const OWL_IRREFLEXIVE_PROPERTY = `${OWL}IrreflexiveProperty` +/** `owl:equivalentClass` predicate used for named-class equivalence edges. */ +const OWL_EQUIVALENT_CLASS = `${OWL}equivalentClass` +/** `owl:disjointWith` predicate retained for named-class disjointness. */ +const OWL_DISJOINT_CLASS = `${OWL}disjointWith` +/** `owl:equivalentProperty` predicate used for named-property equivalence edges. */ +const OWL_EQUIVALENT_PROPERTY = `${OWL}equivalentProperty` +/** `owl:propertyDisjointWith` predicate retained for named-property disjointness. */ +const OWL_PROPERTY_DISJOINT = `${OWL}propertyDisjointWith` +/** `owl:inverseOf` predicate used for directly named inverse-property relationships. */ +const OWL_INVERSE = `${OWL}inverseOf` +/** `owl:deprecated` predicate used to mark generated ontology symbols deprecated. */ +const OWL_DEPRECATED = `${OWL}deprecated` + +/** Maps recognized RDF/OWL property class IRIs to the normalized property-kind model. */ +const PROPERTY_TYPES = new Map([ + [RDF_PROPERTY, 'rdf'], + [OWL_OBJECT_PROPERTY, 'object'], + [OWL_DATATYPE_PROPERTY, 'data'], + [OWL_ANNOTATION_PROPERTY, 'annotation'], +]) + +/** Maps recognized OWL characteristic classes to normalized property-characteristic values. */ +const PROPERTY_CHARACTERISTICS = new Map([ + [OWL_FUNCTIONAL_PROPERTY, 'functional'], + [OWL_INVERSE_FUNCTIONAL_PROPERTY, 'inverseFunctional'], + [OWL_TRANSITIVE_PROPERTY, 'transitive'], + [OWL_SYMMETRIC_PROPERTY, 'symmetric'], + [OWL_ASYMMETRIC_PROPERTY, 'asymmetric'], + [OWL_REFLEXIVE_PROPERTY, 'reflexive'], + [OWL_IRREFLEXIVE_PROPERTY, 'irreflexive'], +]) + +/** One RDF source contributing to an ontology model. */ +export interface OntologySourceType extends SourceType { + readonly quads: Iterable | AsyncIterable +} + +/** Extension predicates and resource limits for ontology ingestion. */ +export interface ReadOptions { + /** Additional predicates that behave like vocabulary-domain declarations. */ + readonly domainPredicates?: readonly string[] + /** Additional predicates that behave like vocabulary-range declarations. */ + readonly rangePredicates?: readonly string[] + /** Maximum quads across all sources. Default is 5,000,000. */ + readonly maxQuads?: number + readonly signal?: AbortSignal +} + +/** Mutable accumulation record for one named class before deterministic sets/metadata are frozen into the public ontology model. */ +interface MutableClass { + readonly iri: string + readonly labels: TextType[] + readonly comments: TextType[] + readonly superClasses: Set + readonly equivalentClasses: Set + readonly disjointClasses: Set + deprecated: boolean +} + +/** Mutable accumulation record for one named property before relationship sets and characteristics are normalized. */ +interface MutableProperty { + readonly iri: string + readonly kinds: Set + readonly labels: TextType[] + readonly comments: TextType[] + readonly domains: Set + readonly ranges: Set + readonly superProperties: Set + readonly equivalentProperties: Set + readonly inverseOf: Set + readonly disjointProperties: Set + readonly characteristics: Set + deprecated: boolean +} + +/** Label/comment assertion deferred until its subject is known to be a modeled class, property, or datatype. */ +interface PendingText { + readonly sourceId: string + readonly field: 'labels' | 'comments' + readonly quad: Quad +} + +/** `owl:deprecated` assertion deferred until the subject kind is known, avoiding premature lossy classification. */ +interface PendingDeprecated { + readonly sourceId: string + readonly quad: Quad +} + +/** + * Reads named RDFS/OWL declarations into a deterministic ontology model. + * + * This operation does not perform RDFS or OWL entailment. It records declared + * named relationships and preserves every unsupported assertion so reasoning + * engines or future readers can interpret richer class expressions later. + */ +export async function read( + sources: readonly OntologySourceType[], + options: ReadOptions = {}, +): Promise { + const classes = new Map() + const properties = new Map() + const datatypes = new Set() + const assertions: AssertionType[] = [] + const diagnostics: DiagnosticType[] = [] + const pendingText: PendingText[] = [] + const pendingDeprecated: PendingDeprecated[] = [] + const domainPredicates = new Set([RDFS_DOMAIN, ...(options.domainPredicates ?? [])]) + const rangePredicates = new Set([RDFS_RANGE, ...(options.rangePredicates ?? [])]) + const maxQuads = options.maxQuads ?? 5_000_000 + let count = 0 + + for (const source of sources) { + for await (const quad of iterate(source.quads)) { + if (options.signal?.aborted) throw options.signal.reason ?? new DOMException('Aborted', 'AbortError') + if (++count > maxQuads) throw new RangeError(`Ontology input exceeds the configured ${maxQuads} quad limit.`) + if (quad.subject.termType !== 'NamedNode') { + assertions.push(toAssertion(quad, source.id)) + continue + } + + const subject = quad.subject.value + const predicate = quad.predicate.value + const objectIri = quad.object.termType === 'NamedNode' ? quad.object.value : undefined + + if (predicate === RDF_TYPE && objectIri) { + if (objectIri === RDFS_CLASS || objectIri === OWL_CLASS) { + getClass(classes, subject) + continue + } + const propertyKind = PROPERTY_TYPES.get(objectIri) + if (propertyKind) { + getProperty(properties, subject).kinds.add(propertyKind) + continue + } + const characteristic = PROPERTY_CHARACTERISTICS.get(objectIri) + if (characteristic) { + getProperty(properties, subject).characteristics.add(characteristic) + continue + } + if (objectIri === RDFS_DATATYPE || objectIri.startsWith(`${XSD.string.slice(0, XSD.string.lastIndexOf('#') + 1)}`)) { + datatypes.add(subject) + continue + } + } + + if (predicate === RDFS_SUBCLASS && objectIri) { + getClass(classes, subject).superClasses.add(objectIri) + continue + } + if (predicate === OWL_EQUIVALENT_CLASS && objectIri) { + getClass(classes, subject).equivalentClasses.add(objectIri) + continue + } + if (predicate === OWL_DISJOINT_CLASS && objectIri) { + getClass(classes, subject).disjointClasses.add(objectIri) + continue + } + if (predicate === RDFS_SUBPROPERTY && objectIri) { + getProperty(properties, subject).superProperties.add(objectIri) + continue + } + if (predicate === OWL_EQUIVALENT_PROPERTY && objectIri) { + getProperty(properties, subject).equivalentProperties.add(objectIri) + continue + } + if (predicate === OWL_PROPERTY_DISJOINT && objectIri) { + getProperty(properties, subject).disjointProperties.add(objectIri) + continue + } + if (predicate === OWL_INVERSE && objectIri) { + getProperty(properties, subject).inverseOf.add(objectIri) + continue + } + if (domainPredicates.has(predicate) && objectIri) { + getProperty(properties, subject).domains.add(objectIri) + continue + } + if (rangePredicates.has(predicate) && objectIri) { + getProperty(properties, subject).ranges.add(objectIri) + continue + } + if (predicate === RDFS_LABEL && quad.object.termType === 'Literal') { + pendingText.push({ sourceId: source.id, field: 'labels', quad }) + continue + } + if (predicate === RDFS_COMMENT && quad.object.termType === 'Literal') { + pendingText.push({ sourceId: source.id, field: 'comments', quad }) + continue + } + if (predicate === OWL_DEPRECATED) { + pendingDeprecated.push({ sourceId: source.id, quad }) + continue + } + + assertions.push(toAssertion(quad, source.id)) + } + } + + attachText(classes, properties, pendingText, assertions, diagnostics) + attachDeprecation(classes, properties, pendingDeprecated, assertions, diagnostics) + + return { + sources: sources.map(stripQuads), + classes: [...classes.values()].map(freezeClass).sort(byIri), + properties: [...properties.values()].map(freezeProperty).sort(byIri), + datatypes: [...datatypes].sort(), + assertions: assertions.sort(compareAssertion), + diagnostics: diagnostics.sort(compareDiagnostic), + } +} + +/** Attaches labels/comments only to classified named ontology resources and records otherwise-unclassified text. */ +function attachText( + classes: ReadonlyMap, + properties: ReadonlyMap, + pending: readonly PendingText[], + assertions: AssertionType[], + diagnostics: DiagnosticType[], +): void { + for (const entry of pending) { + const subject = entry.quad.subject + const object = entry.quad.object + if (subject.termType !== 'NamedNode' || object.termType !== 'Literal') continue + const target = classes.get(subject.value) ?? properties.get(subject.value) + if (target) { + target[entry.field].push(toText(object)) + continue + } + assertions.push(toAssertion(entry.quad, entry.sourceId)) + diagnostics.push({ + code: 'unclassified-text', + message: `Text metadata retained for unclassified ontology term ${subject.value}.`, + sourceId: entry.sourceId, + term: subject.value, + }) + } +} + +/** Interprets supported deprecation literals while retaining malformed or unclassified declarations diagnostically. */ +function attachDeprecation( + classes: ReadonlyMap, + properties: ReadonlyMap, + pending: readonly PendingDeprecated[], + assertions: AssertionType[], + diagnostics: DiagnosticType[], +): void { + for (const entry of pending) { + const subject = entry.quad.subject + const object = entry.quad.object + if (subject.termType !== 'NamedNode' || object.termType !== 'Literal') { + assertions.push(toAssertion(entry.quad, entry.sourceId)) + continue + } + const deprecated = booleanLiteral(object) + if (deprecated === undefined) { + assertions.push(toAssertion(entry.quad, entry.sourceId)) + diagnostics.push({ + code: 'invalid-deprecation', + message: `owl:deprecated for ${subject.value} is not a valid boolean literal.`, + sourceId: entry.sourceId, + term: subject.value, + }) + continue + } + const target = classes.get(subject.value) ?? properties.get(subject.value) + if (target) { + target.deprecated = deprecated + continue + } + assertions.push(toAssertion(entry.quad, entry.sourceId)) + diagnostics.push({ + code: 'unclassified-deprecation', + message: `Deprecation metadata retained for unclassified ontology term ${subject.value}.`, + sourceId: entry.sourceId, + term: subject.value, + }) + } +} + +/** Returns or creates the mutable accumulator for one named ontology class. */ +function getClass(values: Map, iri: string): MutableClass { + let value = values.get(iri) + if (!value) { + value = { + iri, + labels: [], + comments: [], + superClasses: new Set(), + equivalentClasses: new Set(), + disjointClasses: new Set(), + deprecated: false, + } + values.set(iri, value) + } + return value +} + +/** Returns or creates the mutable accumulator for one named ontology property. */ +function getProperty(values: Map, iri: string): MutableProperty { + let value = values.get(iri) + if (!value) { + value = { + iri, + kinds: new Set(), + labels: [], + comments: [], + domains: new Set(), + ranges: new Set(), + superProperties: new Set(), + equivalentProperties: new Set(), + inverseOf: new Set(), + disjointProperties: new Set(), + characteristics: new Set(), + deprecated: false, + } + values.set(iri, value) + } + return value +} + +/** Converts a mutable class accumulator into stable sorted serializable ontology output. */ +function freezeClass(value: MutableClass): ClassType { + return { + iri: value.iri, + labels: sortText(value.labels), + comments: sortText(value.comments), + superClasses: [...value.superClasses].sort(), + equivalentClasses: [...value.equivalentClasses].sort(), + disjointClasses: [...value.disjointClasses].sort(), + deprecated: value.deprecated, + } +} + +/** Converts a mutable property accumulator into stable sorted serializable ontology output. */ +function freezeProperty(value: MutableProperty): PropertyType { + return { + iri: value.iri, + kinds: [...value.kinds].sort(), + labels: sortText(value.labels), + comments: sortText(value.comments), + domains: [...value.domains].sort(), + ranges: [...value.ranges].sort(), + superProperties: [...value.superProperties].sort(), + equivalentProperties: [...value.equivalentProperties].sort(), + inverseOf: [...value.inverseOf].sort(), + disjointProperties: [...value.disjointProperties].sort(), + characteristics: [...value.characteristics].sort(), + deprecated: value.deprecated, + } +} + +/** Converts the supplied value to text without changing semantic identity. */ +function toText(value: Literal): TextType { + const text: { value: string; language?: string; direction?: 'ltr' | 'rtl' } = { value: value.value } + if (value.language) text.language = value.language + if (value.direction) text.direction = value.direction + return text +} + +/** Reads recognized RDF boolean lexical forms without treating arbitrary strings as booleans. */ +function booleanLiteral(value: Literal): boolean | undefined { + if (value.value === 'true' || value.value === '1') return true + if (value.value === 'false' || value.value === '0') return false + return undefined +} + +/** Converts the supplied value to assertion without changing semantic identity. */ +function toAssertion(value: Quad, sourceId: string): AssertionType { + return { + sourceId, + subject: key(value.subject), + predicate: value.predicate.value, + object: key(value.object), + graph: key(value.graph), + } +} + +/** Converts retained RDF assertions to stable string records without carrying runtime term objects. */ +function stripQuads(source: OntologySourceType): SourceType { + const value: { id: string; iri?: string; version?: string; hash?: string } = { id: source.id } + if (source.iri !== undefined) value.iri = source.iri + if (source.version !== undefined) value.version = source.version + if (source.hash !== undefined) value.hash = source.hash + return value +} + +/** Sorts localized ontology text deterministically for byte-stable downstream generation. */ +function sortText(values: readonly TextType[]): TextType[] { + return [...values].sort((left, right) => + `${left.language ?? ''}\u0000${left.direction ?? ''}\u0000${left.value}`.localeCompare( + `${right.language ?? ''}\u0000${right.direction ?? ''}\u0000${right.value}`, + ) + ) +} + +/** Compare assertion using deterministic semantic ordering. */ +function compareAssertion(left: AssertionType, right: AssertionType): number { + return `${left.sourceId}\u0000${left.subject}\u0000${left.predicate}\u0000${left.object}`.localeCompare( + `${right.sourceId}\u0000${right.subject}\u0000${right.predicate}\u0000${right.object}`, + ) +} + +/** Compare diagnostic using deterministic semantic ordering. */ +function compareDiagnostic(left: DiagnosticType, right: DiagnosticType): number { + return `${left.sourceId}\u0000${left.code}\u0000${left.term ?? ''}`.localeCompare( + `${right.sourceId}\u0000${right.code}\u0000${right.term ?? ''}`, + ) +} + +/** Orders named ontology resources by IRI so source file order does not affect the model. */ +function byIri(left: T, right: T): number { + return left.iri.localeCompare(right.iri) +} diff --git a/packages/rdf/ontology/read_test.ts b/packages/rdf/ontology/read_test.ts new file mode 100644 index 0000000..3888c15 --- /dev/null +++ b/packages/rdf/ontology/read_test.ts @@ -0,0 +1,44 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { parse } from '../turtle/mod.ts' +import { index, read } from './mod.ts' + +const SCHEMA_DOMAIN = 'https://schema.org/domainIncludes' +const SCHEMA_RANGE = 'https://schema.org/rangeIncludes' + +describe('@okikio/rdf/ontology', () => { + it('reads named RDFS/OWL relationships and configurable vocabulary aliases', async () => { + const source = ` + @prefix rdf: . + @prefix rdfs: . + @prefix owl: . + @prefix schema: . + @prefix ex: . + + ex:Thing a rdfs:Class . + ex:Product a owl:Class ; rdfs:subClassOf ex:Thing . + ex:name a rdf:Property, owl:FunctionalProperty ; + schema:domainIncludes ex:Thing ; schema:rangeIncludes schema:Text . + ` + const model = await read([{ id: 'test', quads: parse(source) }], { + domainPredicates: [SCHEMA_DOMAIN], + rangePredicates: [SCHEMA_RANGE], + }) + expect(model.classes).toHaveLength(2) + expect(model.properties[0]?.characteristics.includes('functional')).toBe(true) + expect(index(model).superClasses('https://example.com/Product')).toHaveLength(1) + expect(index(model).propertiesForClass('https://example.com/Product')).toHaveLength(1) + }) + + it('retains anonymous OWL expressions instead of silently flattening them', async () => { + const source = ` + @prefix rdfs: . + @prefix owl: . + @prefix ex: . + ex:Product a owl:Class ; rdfs:subClassOf [ a owl:Restriction ; owl:onProperty ex:name ] . + ` + const model = await read([{ id: 'test', quads: parse(source) }]) + expect(model.classes).toHaveLength(1) + expect(model.assertions.length > 0).toBe(true) + }) +}) diff --git a/packages/rdf/package.json b/packages/rdf/package.json new file mode 100644 index 0000000..074c32d --- /dev/null +++ b/packages/rdf/package.json @@ -0,0 +1,37 @@ +{ + "name": "@okikio/rdf", + "version": "0.1.0", + "type": "module", + "sideEffects": false, + "exports": { + ".": "./mod.ts", + "./ntriples": "./ntriples/mod.ts", + "./nquads": "./nquads/mod.ts", + "./turtle": "./turtle/mod.ts", + "./trig": "./trig/mod.ts", + "./jsonld": "./jsonld/mod.ts", + "./canon": "./canon/mod.ts", + "./xml": "./xml/mod.ts", + "./rdfa": "./rdfa/mod.ts", + "./microdata": "./microdata/mod.ts", + "./shape": "./shape/mod.ts", + "./ontology": "./ontology/mod.ts" + }, + "dependencies": { + "jsonld": "^9.0.0", + "rdf-canonize": "^5.0.0", + "rdfxml-streaming-parser": "^3.2.0", + "rdfa-streaming-parser": "^3.0.2", + "microdata-rdf-streaming-parser": "^3.0.0" + }, + "description": "Complete RDF programming model for TypeScript runtimes.", + "license": "MIT", + "repository": { + "type": "git", + "url": "git+https://github.com/okikio/sparql-client.git", + "directory": "packages/rdf" + }, + "publishConfig": { + "access": "public" + } +} diff --git a/packages/rdf/rdfa/mod.ts b/packages/rdf/rdfa/mod.ts new file mode 100644 index 0000000..4741e13 --- /dev/null +++ b/packages/rdf/rdfa/mod.ts @@ -0,0 +1,76 @@ +/** RDFa 1.1 parsing behind native RDF and Web-oriented source contracts. @module */ + +import { factory } from '../factory.ts' +import type { Graph, Quad } from '../term.ts' +import { throwIfAborted, type TextSource } from '../text.ts' +import { parseTransform } from '../transform.ts' +import type { ParserConstructorType } from './types.ts' + +export type { ParserConstructorType, ParserType } from './types.ts' + +/** RDFa profiles implemented by the external RDFa 1.1 processor. */ +export type ProfileType = '' | 'core' | 'html' | 'xhtml' | 'svg' | 'xml' + +/** Known content types that select an RDFa host-language profile. */ +export type ContentType = + | 'text/html' + | 'application/xhtml+xml' + | 'application/xml' + | 'text/xml' + | 'image/svg+xml' + +/** RDFa feature switches exposed by the upstream processor. */ +export type FeatureType = Readonly> + +/** Options for RDFa 1.1 parsing. */ +export interface ParseOptionsType { + readonly base?: string + readonly graph?: Graph + readonly language?: string + readonly vocab?: string + /** Host-language content type. Prefer this over a manual profile when known. */ + readonly contentType?: ContentType + /** Explicit RDFa host-language profile when a content type is unavailable. */ + readonly profile?: ProfileType + /** Fine-grained processor features for specialist integrations. */ + readonly features?: FeatureType + /** External parser injection used by tests or alternate conforming implementations. */ + readonly parser?: ParserConstructorType + readonly signal?: AbortSignal +} + +/** + * Parses RDFa 1.1 incrementally into native `@okikio/rdf` quads. + * + * The RDFa implementation stays behind this subpath because its HTML parser and + * Node-style stream dependencies should not enter the root RDF module graph. + */ +export async function* parse(source: TextSource, options: ParseOptionsType = {}): AsyncGenerator { + throwIfAborted(options.signal) + const Parser = options.parser ?? await defaultParser() + const parser = new Parser(parserOptions(options)) + yield* parseTransform(parser, source, { label: 'RDFa parser', ...(options.signal ? { signal: options.signal } : {}) }) +} + +/** Builds the upstream RDFa options while omitting absent optional fields. */ +function parserOptions(options: ParseOptionsType): Readonly> { + return { + dataFactory: factory, + ...(options.base === undefined ? {} : { baseIRI: options.base }), + ...(options.graph === undefined ? {} : { defaultGraph: options.graph }), + ...(options.language === undefined ? {} : { language: options.language }), + ...(options.vocab === undefined ? {} : { vocab: options.vocab }), + ...(options.contentType === undefined ? {} : { contentType: options.contentType }), + ...(options.profile === undefined ? {} : { profile: options.profile }), + ...(options.features === undefined ? {} : { features: options.features }), + } +} + +/** Lazily resolved RDFa parser constructor so importing the subpath does not initialize the optional processor. */ +let parserPromise: Promise | undefined + +/** Lazily imports the RDFa implementation only when this subpath is used. */ +async function defaultParser(): Promise { + parserPromise ??= import('rdfa-streaming-parser').then((module) => module.RdfaParser as unknown as ParserConstructorType) + return await parserPromise +} diff --git a/packages/rdf/rdfa/mod_test.ts b/packages/rdf/rdfa/mod_test.ts new file mode 100644 index 0000000..b701c50 --- /dev/null +++ b/packages/rdf/rdfa/mod_test.ts @@ -0,0 +1,50 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { literal, namedNode, quad, type Quad } from '../mod.ts' +import { parse, type ParserType } from './mod.ts' + +const fixture = quad(namedNode('https://example.test/s'), namedNode('https://example.test/p'), literal('value')) +type EventType = 'drain' | 'close' | 'error' +type ListenerType = (...args: unknown[]) => void + +class TestParser implements ParserType { + static options: Readonly> | undefined + private readonly listeners = new Map>() + private ended = false + private resolveEnd: (() => void) | undefined + constructor(options: Readonly>) { TestParser.options = options } + write(_value: string | Uint8Array): boolean { return true } + end(): void { this.ended = true; this.resolveEnd?.() } + destroy(error?: Error): void { if (error) this.emit('error', error); this.ended = true; this.resolveEnd?.(); this.emit('close') } + once(event: EventType, listener: ListenerType): this { const values = this.listeners.get(event) ?? new Set(); values.add(listener); this.listeners.set(event, values); return this } + off(event: EventType, listener: ListenerType): this { this.listeners.get(event)?.delete(listener); return this } + private emit(event: EventType, ...args: unknown[]): void { for (const listener of this.listeners.get(event) ?? []) listener(...args) } + async *[Symbol.asyncIterator](): AsyncIterator { if (!this.ended) await new Promise((resolve) => { this.resolveEnd = resolve }); yield fixture } +} + +describe('@okikio/rdf/rdfa', () => { + it('adapts RDFa parser output back to native RDF terms', async () => { + const values: Quad[] = [] + for await (const value of parse('
', { parser: TestParser, contentType: 'text/html' })) values.push(value) + expect(values).toHaveLength(1) + expect(values[0]?.equals(fixture)).toBe(true) + }) + + it('forwards RDFa host-language options and the native RDF factory', async () => { + for await (const _value of parse('
', { + parser: TestParser, + base: 'https://example.test/base/', + contentType: 'text/html', + language: 'en', + vocab: 'https://schema.org/', + })) { /* drain */ } + + expect(TestParser.options?.baseIRI).toBe('https://example.test/base/') + expect(TestParser.options?.contentType).toBe('text/html') + expect(TestParser.options?.language).toBe('en') + expect(TestParser.options?.vocab).toBe('https://schema.org/') + const factory = TestParser.options?.dataFactory as { namedNode(value: string): { value: string } } + expect(factory.namedNode('urn:test').value).toBe('urn:test') + }) + +}) diff --git a/packages/rdf/rdfa/types.ts b/packages/rdf/rdfa/types.ts new file mode 100644 index 0000000..5188939 --- /dev/null +++ b/packages/rdf/rdfa/types.ts @@ -0,0 +1,11 @@ +/** Structural RDFa parser contracts used to isolate the external implementation. @module */ + +import type { TransformParserType } from '../transform.ts' + +/** RDFa parser stream shape required by the adapter. */ +export type ParserType = TransformParserType + +/** Constructor contract for an RDFa parser implementation. */ +export interface ParserConstructorType { + new (options: Readonly>): ParserType +} diff --git a/packages/rdf/shape/index.ts b/packages/rdf/shape/index.ts new file mode 100644 index 0000000..e5eafc4 --- /dev/null +++ b/packages/rdf/shape/index.ts @@ -0,0 +1,65 @@ +/** Indexed view over one materialized SHACL shapes graph. @module */ + +import { key } from '../term.ts' +import type { ObjectTerm, Quad, Subject, Term } from '../term.ts' + +/** Subject/predicate index used by the SHACL reader and property-path parser. */ +export class ShapeIndex { + readonly #subjects = new Map() + readonly #values = new Map>() + readonly #quads = new Map() + + /** Adds one quad to the index. */ + add(quad: Quad): void { + const subjectKey = key(quad.subject) + this.#subjects.set(subjectKey, quad.subject) + + let predicates = this.#values.get(subjectKey) + if (!predicates) { + predicates = new Map() + this.#values.set(subjectKey, predicates) + } + let values = predicates.get(quad.predicate.value) + if (!values) { + values = [] + predicates.set(quad.predicate.value, values) + } + values.push(quad.object) + + let quads = this.#quads.get(subjectKey) + if (!quads) { + quads = [] + this.#quads.set(subjectKey, quads) + } + quads.push(quad) + } + + /** Returns each indexed subject. */ + subjects(): Iterable { + return this.#subjects.values() + } + + /** Returns predicate values for one RDF subject. */ + get(subject: Subject, predicate: string): readonly ObjectTerm[] { + return this.#values.get(key(subject))?.get(predicate) ?? [] + } + + /** Returns all quads for one RDF subject. */ + quads(subject: Subject): readonly Quad[] { + return this.#quads.get(key(subject)) ?? [] + } + + /** Returns whether a node is an RDF list cell. */ + isList(term: Term): boolean { + if (term.termType !== 'NamedNode' && term.termType !== 'BlankNode') return false + const subject = term as Subject + return this.get(subject, RDF_FIRST).length > 0 || this.get(subject, RDF_REST).length > 0 + } +} + +/** RDF collection predicate that points at the current list member. */ +export const RDF_FIRST = 'http://www.w3.org/1999/02/22-rdf-syntax-ns#first' +/** RDF collection predicate that points at the remaining list cell. */ +export const RDF_REST = 'http://www.w3.org/1999/02/22-rdf-syntax-ns#rest' +/** RDF collection terminator used by list and SHACL-path traversal. */ +export const RDF_NIL = 'http://www.w3.org/1999/02/22-rdf-syntax-ns#nil' diff --git a/packages/rdf/shape/list.ts b/packages/rdf/shape/list.ts new file mode 100644 index 0000000..b7b5e7f --- /dev/null +++ b/packages/rdf/shape/list.ts @@ -0,0 +1,85 @@ +/** SHACL list reader with cycle and cardinality diagnostics. @module */ + +import { key } from '../term.ts' +import type { ObjectTerm, Subject } from '../term.ts' +import { RDF_FIRST, RDF_NIL, RDF_REST, ShapeIndex } from './index.ts' +import type { DiagnosticType, IdType } from './model.ts' + +/** Limits for one SHACL list traversal. */ +export interface ListOptions { + readonly maxItems: number + readonly diagnostics: DiagnosticType[] + readonly shape?: IdType + readonly predicate?: string +} + +/** Read list from the supplied source while preserving caller ownership. */ +export function readList( + index: ShapeIndex, + start: ObjectTerm, + options: ListOptions, +): readonly ObjectTerm[] | undefined { + if (start.termType === 'NamedNode' && start.value === RDF_NIL) return [] + if (start.termType !== 'NamedNode' && start.termType !== 'BlankNode') { + addInvalid(options, 'SHACL list head must be an IRI or blank node.') + return undefined + } + + const values: ObjectTerm[] = [] + const seen = new Set() + let cursor: Subject = start + + while (!(cursor.termType === 'NamedNode' && cursor.value === RDF_NIL)) { + const cursorKey = key(cursor) + if (seen.has(cursorKey)) { + options.diagnostics.push(diagnostic('list-cycle', 'SHACL list contains an rdf:rest cycle.', options)) + return undefined + } + seen.add(cursorKey) + + if (values.length >= options.maxItems) { + addInvalid(options, `SHACL list exceeds the configured ${options.maxItems} item limit.`) + return undefined + } + + const first = index.get(cursor, RDF_FIRST) + const rest = index.get(cursor, RDF_REST) + if (first.length !== 1 || rest.length !== 1) { + addInvalid(options, 'Each SHACL list cell must have exactly one rdf:first and one rdf:rest value.') + return undefined + } + + values.push(first[0]!) + const next = rest[0]! + if (next.termType !== 'NamedNode' && next.termType !== 'BlankNode') { + addInvalid(options, 'rdf:rest must reference another SHACL list node.') + return undefined + } + cursor = next + } + + return values +} + +/** Records a malformed RDF-list node so shape parsing can continue without silently accepting it. */ +function addInvalid(options: ListOptions, message: string): void { + options.diagnostics.push(diagnostic('invalid-list', message, options)) +} + +/** Appends one source-aware RDF-list diagnostic to the caller-owned collection. */ +function diagnostic( + code: DiagnosticType['code'], + message: string, + options: ListOptions, +): DiagnosticType { + const value: { + code: DiagnosticType['code'] + severity: 'error' + message: string + shape?: IdType + predicate?: string + } = { code, severity: 'error', message } + if (options.shape) value.shape = options.shape + if (options.predicate) value.predicate = options.predicate + return value +} diff --git a/packages/rdf/shape/mod.ts b/packages/rdf/shape/mod.ts new file mode 100644 index 0000000..975ad6f --- /dev/null +++ b/packages/rdf/shape/mod.ts @@ -0,0 +1,42 @@ +/** + * SHACL shape and property-path infrastructure. + * + * This subpath reads shapes graphs into a loss-preserving, versioned IR. It is + * intentionally separate from ontology interpretation and does not claim full + * SHACL validation. SHACL 1.2 extension specifications can add evaluators over + * the retained RDF term records without changing the Core model. + * + * @example + * ```ts + * import * as turtle from '@okikio/rdf/turtle' + * import * as shape from '@okikio/rdf/shape' + * + * const graph = await shape.read(turtle.parse(source), { version: '1.2' }) + * const person = graph.shapes.find((value) => value.id.value.endsWith('PersonShape')) + * ``` + * + * @module + */ + +export { read } from './read.ts' +export type { ReadOptions } from './read.ts' +export { readPath } from './path.ts' +export type { PathOptions } from './path.ts' +export type { + AssertionType, + BlankType, + ConstraintType, + DiagnosticType, + GraphType, + IdType, + IriType, + LiteralType, + MetadataType, + PathType, + ShapeType, + TargetType, + TermType, + TextType, + TripleType, + VersionType, +} from './model.ts' diff --git a/packages/rdf/shape/model.ts b/packages/rdf/shape/model.ts new file mode 100644 index 0000000..044e15a --- /dev/null +++ b/packages/rdf/shape/model.ts @@ -0,0 +1,200 @@ +/** + * Versioned, serializable SHACL shape model. + * + * The model captures SHACL Core structure without making validation claims. + * Values that belong to newer extension specifications, such as node + * expressions, remain lossless RDF term records until a focused evaluator owns + * their semantics. + * + * @module + */ + +import type { Direction } from '../term.ts' + +/** SHACL language family understood by the reader. */ +export type VersionType = '1.0' | '1.2' + +/** Stable serializable reference to an RDF IRI. */ +export interface IriType { + readonly kind: 'iri' + readonly value: string +} + +/** Stable serializable reference to an RDF blank node. */ +export interface BlankType { + readonly kind: 'blank' + readonly value: string +} + +/** RDF graph node that can identify a SHACL shape. */ +export type IdType = IriType | BlankType + +/** Serializable RDF literal retained by the shape model. */ +export interface LiteralType { + readonly kind: 'literal' + readonly value: string + readonly datatype: string + readonly language?: string + readonly direction?: Direction +} + +/** Serializable RDF 1.2 triple term retained by the shape model. */ +export interface TripleType { + readonly kind: 'triple' + readonly subject: IdType + readonly predicate: string + readonly object: TermType +} + +/** Serializable RDF term used by constraints, metadata, and extensions. */ +export type TermType = IriType | BlankType | LiteralType | TripleType + +/** Localized human-facing SHACL text. */ +export interface TextType { + readonly value: string + readonly language?: string + readonly direction?: Direction + readonly datatype: string +} + +/** SHACL property path with the complete Core path constructor family. */ +export type PathType = + | { readonly kind: 'predicate'; readonly iri: string } + | { readonly kind: 'sequence'; readonly items: readonly PathType[] } + | { readonly kind: 'alternative'; readonly items: readonly PathType[] } + | { readonly kind: 'inverse'; readonly path: PathType } + | { readonly kind: 'zeroOrMore'; readonly path: PathType } + | { readonly kind: 'oneOrMore'; readonly path: PathType } + | { readonly kind: 'zeroOrOne'; readonly path: PathType } + | { readonly kind: 'unknown'; readonly value: TermType } + +/** Explicit target attached to a shape. */ +export type TargetType = + | { readonly kind: 'node'; readonly value: TermType } + | { readonly kind: 'class'; readonly iri: string } + | { readonly kind: 'subjectsOf'; readonly iri: string } + | { readonly kind: 'objectsOf'; readonly iri: string } + | { readonly kind: 'where'; readonly expression: TermType } + | { readonly kind: 'shape'; readonly shape: IdType } + +/** Known SHACL Core constraint retained in semantic form. */ +export type ConstraintType = + | { readonly kind: 'class'; readonly choices: readonly string[] } + | { readonly kind: 'datatype'; readonly choices: readonly string[] } + | { readonly kind: 'nodeKind'; readonly choices: readonly string[] } + | { readonly kind: 'minCount'; readonly count: number } + | { readonly kind: 'maxCount'; readonly count: number } + | { readonly kind: 'minExclusive'; readonly value: LiteralType } + | { readonly kind: 'minInclusive'; readonly value: LiteralType } + | { readonly kind: 'maxExclusive'; readonly value: LiteralType } + | { readonly kind: 'maxInclusive'; readonly value: LiteralType } + | { readonly kind: 'minLength'; readonly length: number } + | { readonly kind: 'maxLength'; readonly length: number } + | { readonly kind: 'pattern'; readonly pattern: string; readonly flags?: string } + | { readonly kind: 'singleLine'; readonly value: boolean } + | { readonly kind: 'languageIn'; readonly languages: readonly string[] } + | { readonly kind: 'uniqueLang'; readonly value: boolean } + | { readonly kind: 'memberShape'; readonly shape: IdType } + | { readonly kind: 'minListLength'; readonly length: number } + | { readonly kind: 'maxListLength'; readonly length: number } + | { readonly kind: 'uniqueMembers'; readonly value: boolean } + | { readonly kind: 'equals'; readonly path: PathType } + | { readonly kind: 'disjoint'; readonly path: PathType } + | { readonly kind: 'subsetOf'; readonly path: PathType } + | { readonly kind: 'lessThan'; readonly path: PathType } + | { readonly kind: 'lessThanOrEquals'; readonly path: PathType } + | { readonly kind: 'not'; readonly shape: IdType } + | { readonly kind: 'and'; readonly shapes: readonly IdType[] } + | { readonly kind: 'or'; readonly shapes: readonly IdType[] } + | { readonly kind: 'xone'; readonly shapes: readonly IdType[] } + | { readonly kind: 'node'; readonly shape: IdType } + | { readonly kind: 'property'; readonly shape: IdType } + | { readonly kind: 'someValue'; readonly shape: IdType } + | { + readonly kind: 'qualified' + readonly shape: IdType + readonly minCount?: number + readonly maxCount?: number + readonly disjoint?: boolean + } + | { readonly kind: 'reifierShape'; readonly shape: IdType } + | { readonly kind: 'reificationRequired'; readonly value: boolean } + | { + readonly kind: 'closed' + readonly mode: boolean | 'byTypes' + readonly ignoredProperties: readonly string[] + } + | { readonly kind: 'hasValue'; readonly value: TermType } + | { readonly kind: 'in'; readonly values: readonly TermType[] } + | { readonly kind: 'rootClass'; readonly iri: string } + | { readonly kind: 'uniqueValuesFor'; readonly paths: readonly PathType[] } + +/** Non-validating and extension-facing metadata retained on one shape. */ +export interface MetadataType { + readonly names: readonly TextType[] + readonly descriptions: readonly TextType[] + readonly intents: readonly TextType[] + readonly agentInstructions: readonly TextType[] + readonly codeIdentifiers: readonly string[] + readonly units: readonly TermType[] + readonly order: readonly LiteralType[] + readonly groups: readonly IdType[] + readonly values: readonly TermType[] + readonly defaultValues: readonly TermType[] +} + +/** One assertion the current Core reader intentionally does not interpret. */ +export interface AssertionType { + readonly subject: IdType + readonly predicate: string + readonly object: TermType + readonly graph?: IdType +} + +/** Structured shape-reader diagnostic. */ +export interface DiagnosticType { + readonly code: + | 'invalid-shape-id' + | 'invalid-value' + | 'invalid-list' + | 'list-cycle' + | 'invalid-path' + | 'path-cycle' + | 'invalid-cardinality' + | 'unsupported-version' + readonly severity: 'warning' | 'error' + readonly message: string + readonly shape?: IdType + readonly predicate?: string +} + +/** One node or property shape. */ +export interface ShapeType { + readonly id: IdType + readonly kind: 'node' | 'property' + /** Every named rdf:type declared for this shape, including extension types. */ + readonly types: readonly string[] + readonly path?: PathType + readonly targets: readonly TargetType[] + readonly severity?: string + readonly messages: readonly TextType[] + /** + * Raw deactivation expression values. + * + * SHACL 1.0 normally uses one boolean. SHACL 1.2 generalizes this field to + * node-expression machinery, so the reader preserves the RDF values instead + * of pretending they are all booleans. + */ + readonly deactivated: readonly TermType[] + readonly constraints: readonly ConstraintType[] + readonly metadata: MetadataType + readonly assertions: readonly AssertionType[] +} + +/** Loss-preserving SHACL shapes graph intermediate representation. */ +export interface GraphType { + readonly version: VersionType + readonly shapes: readonly ShapeType[] + readonly diagnostics: readonly DiagnosticType[] + readonly assertions: readonly AssertionType[] +} diff --git a/packages/rdf/shape/path.ts b/packages/rdf/shape/path.ts new file mode 100644 index 0000000..c8ef2da --- /dev/null +++ b/packages/rdf/shape/path.ts @@ -0,0 +1,130 @@ +/** SHACL Core property-path reader. @module */ + +import { key } from '../term.ts' +import type { ObjectTerm, Subject } from '../term.ts' +import { ShapeIndex } from './index.ts' +import { readList } from './list.ts' +import type { DiagnosticType, IdType, PathType } from './model.ts' +import { term } from './value.ts' + +/** SHACL namespace used to recognize Core property-path predicates. */ +const SH = 'http://www.w3.org/ns/shacl#' +/** SHACL predicates whose blank-node objects encode compound property-path operators. */ +const PATH_PREDICATES = [ + `${SH}alternativePath`, + `${SH}inversePath`, + `${SH}zeroOrMorePath`, + `${SH}oneOrMorePath`, + `${SH}zeroOrOnePath`, +] as const + +/** Limits and diagnostics shared by recursive path parsing. */ +export interface PathOptions { + readonly maxDepth: number + readonly maxListItems: number + readonly diagnostics: DiagnosticType[] + readonly shape?: IdType + readonly predicate?: string +} + +/** Parses one RDF term as a SHACL Core property path. */ +export function readPath(index: ShapeIndex, value: ObjectTerm, options: PathOptions): PathType { + return readPathAt(index, value, options, new Set(), 0) +} + +/** Read path at from the supplied source while preserving caller ownership. */ +function readPathAt( + index: ShapeIndex, + value: ObjectTerm, + options: PathOptions, + active: Set, + depth: number, +): PathType { + if (value.termType === 'NamedNode' && !index.isList(value) && !hasConstructor(index, value)) { + return { kind: 'predicate', iri: value.value } + } + + if (value.termType !== 'NamedNode' && value.termType !== 'BlankNode') return unknown(value, options, 'SHACL path must be an IRI or blank node.') + if (depth >= options.maxDepth) return unknown(value, options, `SHACL path exceeds the configured depth limit of ${options.maxDepth}.`) + + const subject = value as Subject + const subjectKey = key(subject) + if (active.has(subjectKey)) { + options.diagnostics.push(pathDiagnostic('path-cycle', 'SHACL property path contains a cycle.', options)) + return unknownValue(value) + } + const nextActive = new Set(active) + nextActive.add(subjectKey) + + if (index.isList(subject)) { + const values = readList(index, subject, listOptions(options)) + if (!values || values.length < 2) return unknown(value, options, 'A SHACL sequence path must contain at least two path members.') + return { kind: 'sequence', items: values.map((item) => readPathAt(index, item, options, nextActive, depth + 1)) } + } + + const constructors = PATH_PREDICATES.flatMap((predicate) => index.get(subject, predicate).map((object) => ({ predicate, object }))) + if (constructors.length !== 1) return unknown(value, options, 'A blank-node SHACL path must have exactly one Core path constructor.') + const constructor = constructors[0]! + + if (constructor.predicate === `${SH}alternativePath`) { + const values = readList(index, constructor.object, listOptions(options)) + if (!values || values.length < 2) return unknown(value, options, 'sh:alternativePath must reference a list with at least two members.') + return { kind: 'alternative', items: values.map((item) => readPathAt(index, item, options, nextActive, depth + 1)) } + } + + const child = readPathAt(index, constructor.object, options, nextActive, depth + 1) + if (constructor.predicate === `${SH}inversePath`) return { kind: 'inverse', path: child } + if (constructor.predicate === `${SH}zeroOrMorePath`) return { kind: 'zeroOrMore', path: child } + if (constructor.predicate === `${SH}oneOrMorePath`) return { kind: 'oneOrMore', path: child } + return { kind: 'zeroOrOne', path: child } +} + +/** Detects whether a blank-node path uses one of the SHACL path constructor predicates. */ +function hasConstructor(index: ShapeIndex, subject: Subject): boolean { + return PATH_PREDICATES.some((predicate) => index.get(subject, predicate).length > 0) +} + +/** Retains an unrecognized path node as a loss-preserving unknown path record. */ +function unknown(value: ObjectTerm, options: PathOptions, message: string): PathType { + options.diagnostics.push(pathDiagnostic('invalid-path', message, options)) + return unknownValue(value) +} + +/** Retains a path assertion whose value cannot be normalized by the current SHACL profile. */ +function unknownValue(value: ObjectTerm): PathType { + const record = term(value) + if (!record) throw new TypeError('SHACL path term could not be represented.') + return { kind: 'unknown', value: record } +} + + +/** Projects path-reader state into the shared RDF-list traversal options. */ +function listOptions(options: PathOptions) { + const value: { + maxItems: number + diagnostics: DiagnosticType[] + shape?: IdType + predicate?: string + } = { maxItems: options.maxListItems, diagnostics: options.diagnostics } + if (options.shape) value.shape = options.shape + if (options.predicate) value.predicate = options.predicate + return value +} + +/** Records a path-specific diagnostic without discarding the underlying SHACL assertion. */ +function pathDiagnostic( + code: 'invalid-path' | 'path-cycle', + message: string, + options: PathOptions, +): DiagnosticType { + const value: { + code: 'invalid-path' | 'path-cycle' + severity: 'error' + message: string + shape?: IdType + predicate?: string + } = { code, severity: 'error', message } + if (options.shape) value.shape = options.shape + if (options.predicate) value.predicate = options.predicate + return value +} diff --git a/packages/rdf/shape/read.ts b/packages/rdf/shape/read.ts new file mode 100644 index 0000000..f230e05 --- /dev/null +++ b/packages/rdf/shape/read.ts @@ -0,0 +1,952 @@ +/** + * Loss-preserving SHACL Core shapes-graph reader. + * + * The reader materializes the shapes graph because SHACL lists and property + * paths require random access. It does not validate data graphs and it does not + * evaluate SHACL 1.2 Node Expressions. Known Core statements are normalized + * into the semantic model; unsupported or malformed statements remain in the + * assertion set with diagnostics. + * + * @module + */ + +import { iterate } from '../source.ts' +import { key, RDF, XSD } from '../term.ts' +import type { Literal, ObjectTerm, Quad, Subject } from '../term.ts' +import { ShapeIndex } from './index.ts' +import { readList } from './list.ts' +import type { + AssertionType, + ConstraintType, + DiagnosticType, + GraphType, + IdType, + LiteralType, + MetadataType, + PathType, + ShapeType, + TargetType, + TermType, + TextType, + VersionType, +} from './model.ts' +import { readPath } from './path.ts' +import { assertion, id, literal, term, text } from './value.ts' + +/** SHACL namespace used to identify Core shapes, targets, constraints, and metadata. */ +const SH = 'http://www.w3.org/ns/shacl#' +/** RDFS namespace used for label/comment metadata understood by the shape reader. */ +const RDFS = 'http://www.w3.org/2000/01/rdf-schema#' + +/** IRI identifying explicit SHACL node shapes. */ +const NODE_SHAPE = `${SH}NodeShape` +/** IRI identifying explicit SHACL property shapes. */ +const PROPERTY_SHAPE = `${SH}PropertyShape` +/** SHACL 1.2 class used for resources that act as both classes and shapes. */ +const SHAPE_CLASS = `${SH}ShapeClass` +/** SHACL 1.2 `sh:ByTypes` value accepted by the expanded closed-shape constraint. */ +const BY_TYPES = `${SH}ByTypes` + +/** Known SHACL predicates that are sufficient evidence to discover an implicit shape resource. */ +const SHAPE_PREDICATES = new Set([ + `${SH}path`, `${SH}targetNode`, `${SH}targetClass`, `${SH}targetSubjectsOf`, `${SH}targetObjectsOf`, + `${SH}targetWhere`, `${SH}shape`, `${SH}severity`, `${SH}message`, `${SH}deactivated`, + `${SH}class`, `${SH}datatype`, `${SH}nodeKind`, `${SH}minCount`, `${SH}maxCount`, + `${SH}minExclusive`, `${SH}minInclusive`, `${SH}maxExclusive`, `${SH}maxInclusive`, + `${SH}minLength`, `${SH}maxLength`, `${SH}pattern`, `${SH}flags`, `${SH}singleLine`, + `${SH}languageIn`, `${SH}uniqueLang`, `${SH}memberShape`, `${SH}minListLength`, `${SH}maxListLength`, + `${SH}uniqueMembers`, `${SH}equals`, `${SH}disjoint`, `${SH}subsetOf`, `${SH}lessThan`, + `${SH}lessThanOrEquals`, `${SH}not`, `${SH}and`, `${SH}or`, `${SH}xone`, `${SH}node`, `${SH}property`, + `${SH}someValue`, `${SH}qualifiedValueShape`, `${SH}qualifiedMinCount`, `${SH}qualifiedMaxCount`, + `${SH}qualifiedValueShapesDisjoint`, `${SH}reifierShape`, `${SH}reificationRequired`, `${SH}closed`, + `${SH}ignoredProperties`, `${SH}hasValue`, `${SH}in`, `${SH}rootClass`, `${SH}uniqueValuesFor`, + `${SH}name`, `${SH}description`, `${SH}intent`, `${SH}agentInstruction`, `${SH}codeIdentifier`, + `${SH}unit`, `${SH}order`, `${SH}group`, `${SH}values`, `${SH}defaultValue`, +]) + +/** Core predicates that require an `unsupported-version` diagnostic when read in SHACL 1.0 mode. */ +const SHACL_12_PREDICATES = new Set([ + `${SH}targetWhere`, `${SH}shape`, `${SH}singleLine`, `${SH}memberShape`, `${SH}minListLength`, + `${SH}maxListLength`, `${SH}uniqueMembers`, `${SH}subsetOf`, `${SH}someValue`, `${SH}reifierShape`, + `${SH}reificationRequired`, `${SH}rootClass`, `${SH}uniqueValuesFor`, `${SH}intent`, `${SH}agentInstruction`, + `${SH}codeIdentifier`, `${SH}values`, `${SH}defaultValue`, +]) + +/** Reader resource limits and draft-version interpretation. */ +export interface ReadOptions { + /** Core vocabulary generation to interpret. Default is the current 1.2 draft. */ + readonly version?: VersionType + /** Maximum number of quads to materialize. Default is 1,000,000. */ + readonly maxQuads?: number + /** Maximum members followed from one SHACL list. Default is 100,000. */ + readonly maxListItems?: number + /** Maximum nested property-path depth. Default is 256. */ + readonly maxPathDepth?: number + readonly signal?: AbortSignal +} + +/** Shared materialized-shape read state carrying version policy, random-access index, diagnostics, and recursion/list limits. */ +interface StateType { + readonly version: VersionType + readonly index: ShapeIndex + readonly diagnostics: DiagnosticType[] + readonly maxListItems: number + readonly maxPathDepth: number +} + +/** Read from the supplied source while preserving caller ownership. */ +export async function read( + source: Iterable | AsyncIterable, + options: ReadOptions = {}, +): Promise { + const version = options.version ?? '1.2' + if (version !== '1.0' && version !== '1.2') throw new TypeError(`Unsupported SHACL version '${String(version)}'.`) + const maxQuads = options.maxQuads ?? 1_000_000 + const index = new ShapeIndex() + const quads: Quad[] = [] + + for await (const quad of iterate(source)) { + if (options.signal?.aborted) throw options.signal.reason ?? new DOMException('Aborted', 'AbortError') + if (quads.length >= maxQuads) throw new RangeError(`SHACL shapes graph exceeds the configured ${maxQuads} quad limit.`) + quads.push(quad) + index.add(quad) + } + + const diagnostics: DiagnosticType[] = [] + const state: StateType = { + version, + index, + diagnostics, + maxListItems: options.maxListItems ?? 100_000, + maxPathDepth: options.maxPathDepth ?? 256, + } + const candidates = discoverShapes(index) + const shapes: ShapeType[] = [] + const shapeKeys = new Set() + + for (const subject of candidates) { + const shape = readShape(state, subject) + if (!shape) continue + shapes.push(shape) + shapeKeys.add(key(subject)) + } + + const graphAssertions: AssertionType[] = [] + for (const quad of quads) { + if (shapeKeys.has(key(quad.subject))) continue + if (!quad.predicate.value.startsWith(SH)) continue + const value = assertion(quad) + if (value) graphAssertions.push(value) + } + + shapes.sort((left, right) => idKey(left.id).localeCompare(idKey(right.id))) + graphAssertions.sort(compareAssertion) + diagnostics.sort(compareDiagnostic) + return { version, shapes, diagnostics, assertions: graphAssertions } +} + +/** Discovers resources with SHACL type, target, path, or constraint evidence without treating arbitrary labelled ontology resources as shapes. */ +function discoverShapes(index: ShapeIndex): Subject[] { + const values = new Map() + for (const subject of index.subjects()) { + const types = index.get(subject, RDF.type) + const explicit = types.some((value) => + value.termType === 'NamedNode' && + (value.value === NODE_SHAPE || value.value === PROPERTY_SHAPE || value.value === SHAPE_CLASS) + ) + const hasShapePredicate = index.quads(subject).some((quad) => SHAPE_PREDICATES.has(quad.predicate.value)) + if (explicit || hasShapePredicate) values.set(key(subject), subject) + } + return [...values.values()].sort((left, right) => key(left).localeCompare(key(right))) +} + +/** Read shape from the supplied source while preserving caller ownership. */ +function readShape(state: StateType, subject: Subject): ShapeType | undefined { + const shapeId = id(subject) + if (!shapeId) { + state.diagnostics.push({ code: 'invalid-shape-id', severity: 'error', message: 'SHACL shape identifier must be an IRI or blank node.' }) + return undefined + } + + recordVersionDiagnostics(state, subject, shapeId) + const consumed = new Set([RDF.type]) + const types = state.index.get(subject, RDF.type) + .filter((value) => value.termType === 'NamedNode') + .map((value) => value.value) + .sort() + const pathValues = state.index.get(subject, `${SH}path`) + const explicitProperty = state.index.get(subject, RDF.type).some((value) => value.termType === 'NamedNode' && value.value === PROPERTY_SHAPE) + const kind = explicitProperty || pathValues.length > 0 ? 'property' : 'node' + let path: PathType | undefined + if (pathValues.length > 0) { + consumed.add(`${SH}path`) + if (pathValues.length !== 1) addCardinality(state, shapeId, `${SH}path`, 'A property shape must have at most one sh:path value.') + path = readPath(state.index, pathValues[0]!, pathOptions(state, shapeId, `${SH}path`)) + } + + const targets = readTargets(state, subject, shapeId, consumed) + const severity = readIriSingleton(state, subject, shapeId, `${SH}severity`, consumed) + const messages = readTextValues(state, subject, shapeId, `${SH}message`, consumed) + const deactivated = readTermValues(state, subject, `${SH}deactivated`, consumed) + const constraints = readConstraints(state, subject, shapeId, consumed) + const metadata = readMetadata(state, subject, shapeId, consumed) + const assertions = readAssertions(state, subject, consumed) + + const value: { + id: IdType + kind: 'node' | 'property' + types: readonly string[] + path?: PathType + targets: readonly TargetType[] + severity?: string + messages: readonly TextType[] + deactivated: readonly TermType[] + constraints: readonly ConstraintType[] + metadata: MetadataType + assertions: readonly AssertionType[] + } = { id: shapeId, kind, types, targets, messages, deactivated, constraints, metadata, assertions } + if (path) value.path = path + if (severity) value.severity = severity + return value +} + +/** Read targets from the supplied source while preserving caller ownership. */ +function readTargets( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, +): TargetType[] { + const targets: TargetType[] = [] + for (const value of state.index.get(subject, `${SH}targetNode`)) { + const record = term(value) + if (record) targets.push({ kind: 'node', value: record }) + else addInvalid(state, shape, `${SH}targetNode`, 'sh:targetNode value cannot be represented as a shape term.') + } + consumeWhenPresent(state, subject, consumed, `${SH}targetNode`) + pushIriTargets(state, subject, shape, consumed, `${SH}targetClass`, 'class', targets) + pushIriTargets(state, subject, shape, consumed, `${SH}targetSubjectsOf`, 'subjectsOf', targets) + pushIriTargets(state, subject, shape, consumed, `${SH}targetObjectsOf`, 'objectsOf', targets) + + for (const value of state.index.get(subject, `${SH}targetWhere`)) { + const record = term(value) + if (record) targets.push({ kind: 'where', expression: record }) + else addInvalid(state, shape, `${SH}targetWhere`, 'sh:targetWhere value cannot be represented as an RDF term.') + } + consumeWhenPresent(state, subject, consumed, `${SH}targetWhere`) + + for (const value of state.index.get(subject, `${SH}shape`)) { + const shapeValue = id(value) + if (shapeValue) targets.push({ kind: 'shape', shape: shapeValue }) + else addInvalid(state, shape, `${SH}shape`, 'sh:shape target must reference an IRI or blank node.') + } + consumeWhenPresent(state, subject, consumed, `${SH}shape`) + return targets +} + +/** Reads every IRI-valued target assertion, retaining invalid values through diagnostics instead of silently coercing them. */ +function pushIriTargets( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + predicate: string, + kind: 'class' | 'subjectsOf' | 'objectsOf', + targets: TargetType[], +): void { + for (const value of state.index.get(subject, predicate)) { + if (value.termType === 'NamedNode') targets.push({ kind, iri: value.value }) + else addInvalid(state, shape, predicate, `${local(predicate)} must reference an IRI.`) + } + consumeWhenPresent(state, subject, consumed, predicate) +} + +/** Read constraints from the supplied source while preserving caller ownership. */ +function readConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, +): ConstraintType[] { + const constraints: ConstraintType[] = [] + + pushChoiceConstraints(state, subject, shape, consumed, `${SH}class`, 'class', constraints) + pushChoiceConstraints(state, subject, shape, consumed, `${SH}datatype`, 'datatype', constraints) + pushChoiceConstraints(state, subject, shape, consumed, `${SH}nodeKind`, 'nodeKind', constraints) + pushIntegerConstraints(state, subject, shape, consumed, `${SH}minCount`, 'minCount', 'count', constraints) + pushIntegerConstraints(state, subject, shape, consumed, `${SH}maxCount`, 'maxCount', 'count', constraints) + pushLiteralConstraints(state, subject, shape, consumed, `${SH}minExclusive`, 'minExclusive', constraints) + pushLiteralConstraints(state, subject, shape, consumed, `${SH}minInclusive`, 'minInclusive', constraints) + pushLiteralConstraints(state, subject, shape, consumed, `${SH}maxExclusive`, 'maxExclusive', constraints) + pushLiteralConstraints(state, subject, shape, consumed, `${SH}maxInclusive`, 'maxInclusive', constraints) + pushIntegerConstraints(state, subject, shape, consumed, `${SH}minLength`, 'minLength', 'length', constraints) + pushIntegerConstraints(state, subject, shape, consumed, `${SH}maxLength`, 'maxLength', 'length', constraints) + pushPatternConstraints(state, subject, shape, consumed, constraints) + pushBooleanConstraints(state, subject, shape, consumed, `${SH}singleLine`, 'singleLine', constraints) + pushLanguageConstraints(state, subject, shape, consumed, constraints) + pushBooleanConstraints(state, subject, shape, consumed, `${SH}uniqueLang`, 'uniqueLang', constraints) + pushShapeConstraints(state, subject, shape, consumed, `${SH}memberShape`, 'memberShape', constraints) + pushIntegerConstraints(state, subject, shape, consumed, `${SH}minListLength`, 'minListLength', 'length', constraints) + pushIntegerConstraints(state, subject, shape, consumed, `${SH}maxListLength`, 'maxListLength', 'length', constraints) + pushBooleanConstraints(state, subject, shape, consumed, `${SH}uniqueMembers`, 'uniqueMembers', constraints) + + for (const [predicate, kind] of [ + [`${SH}equals`, 'equals'], [`${SH}disjoint`, 'disjoint'], [`${SH}subsetOf`, 'subsetOf'], + [`${SH}lessThan`, 'lessThan'], [`${SH}lessThanOrEquals`, 'lessThanOrEquals'], + ] as const) pushPathConstraints(state, subject, shape, consumed, predicate, kind, constraints) + + pushShapeConstraints(state, subject, shape, consumed, `${SH}not`, 'not', constraints) + pushShapeListConstraints(state, subject, shape, consumed, `${SH}and`, 'and', constraints) + pushShapeListConstraints(state, subject, shape, consumed, `${SH}or`, 'or', constraints) + pushShapeListConstraints(state, subject, shape, consumed, `${SH}xone`, 'xone', constraints) + pushShapeConstraints(state, subject, shape, consumed, `${SH}node`, 'node', constraints) + pushShapeConstraints(state, subject, shape, consumed, `${SH}property`, 'property', constraints) + pushShapeConstraints(state, subject, shape, consumed, `${SH}someValue`, 'someValue', constraints) + pushQualifiedConstraints(state, subject, shape, consumed, constraints) + pushShapeConstraints(state, subject, shape, consumed, `${SH}reifierShape`, 'reifierShape', constraints) + pushBooleanConstraints(state, subject, shape, consumed, `${SH}reificationRequired`, 'reificationRequired', constraints) + pushClosedConstraints(state, subject, shape, consumed, constraints) + + for (const value of state.index.get(subject, `${SH}hasValue`)) { + const record = term(value) + if (record) constraints.push({ kind: 'hasValue', value: record }) + else addInvalid(state, shape, `${SH}hasValue`, 'sh:hasValue cannot be represented as a shape term.') + } + consumeWhenPresent(state, subject, consumed, `${SH}hasValue`) + pushTermListConstraints(state, subject, shape, consumed, `${SH}in`, constraints) + pushIriConstraints(state, subject, shape, consumed, `${SH}rootClass`, 'rootClass', constraints) + pushUniqueValuesFor(state, subject, shape, consumed, constraints) + + return constraints +} + +/** Reads class/datatype/node-kind constraints as either one IRI or a SHACL 1.2 IRI choice list. */ +function pushChoiceConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + predicate: string, + kind: 'class' | 'datatype' | 'nodeKind', + constraints: ConstraintType[], +): void { + for (const value of state.index.get(subject, predicate)) { + const choices = readIriChoices(state, value, shape, predicate) + if (choices) constraints.push({ kind, choices }) + } + consumeWhenPresent(state, subject, consumed, predicate) +} + +/** Read iri choices from the supplied source while preserving caller ownership. */ +function readIriChoices( + state: StateType, + value: ObjectTerm, + shape: IdType, + predicate: string, +): readonly string[] | undefined { + if ((value.termType === 'NamedNode' || value.termType === 'BlankNode') && state.index.isList(value)) { + const members = readList(state.index, value, listOptions(state, shape, predicate)) + if (!members) return undefined + const choices: string[] = [] + for (const member of members) { + if (member.termType !== 'NamedNode') { + addInvalid(state, shape, predicate, `${local(predicate)} list members must be IRIs.`) + return undefined + } + choices.push(member.value) + } + return choices + } + if (value.termType === 'NamedNode') return [value.value] + addInvalid(state, shape, predicate, `${local(predicate)} must be an IRI or SHACL list of IRIs.`) + return undefined +} + +/** Reads non-negative integer constraints and records malformed/cardinality violations without dropping their source assertions. */ +function pushIntegerConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + predicate: string, + kind: 'minCount' | 'maxCount' | 'minLength' | 'maxLength' | 'minListLength' | 'maxListLength', + field: 'count' | 'length', + constraints: ConstraintType[], +): void { + for (const value of state.index.get(subject, predicate)) { + const integer = integerValue(value) + if (integer === undefined || integer < 0) { + addInvalid(state, shape, predicate, `${local(predicate)} must be a non-negative xsd:integer literal.`) + continue + } + constraints.push(field === 'count' ? { kind: kind as 'minCount' | 'maxCount', count: integer } : { kind: kind as 'minLength' | 'maxLength' | 'minListLength' | 'maxListLength', length: integer }) + } + consumeWhenPresent(state, subject, consumed, predicate) +} + +/** Preserves literal-valued comparison constraints as RDF terms so datatype ordering remains a validator concern. */ +function pushLiteralConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + predicate: string, + kind: 'minExclusive' | 'minInclusive' | 'maxExclusive' | 'maxInclusive', + constraints: ConstraintType[], +): void { + for (const value of state.index.get(subject, predicate)) { + if (value.termType === 'Literal') constraints.push({ kind, value: literal(value) }) + else addInvalid(state, shape, predicate, `${local(predicate)} must be a literal.`) + } + consumeWhenPresent(state, subject, consumed, predicate) +} + +/** Combines `sh:pattern` with its optional `sh:flags` value while diagnosing duplicate or non-string values. */ +function pushPatternConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + constraints: ConstraintType[], +): void { + const flags = state.index.get(subject, `${SH}flags`) + let flag: string | undefined + if (flags.length > 1) addCardinality(state, shape, `${SH}flags`, 'A shape must have at most one sh:flags value.') + if (flags[0]) { + if (isStringLiteral(flags[0])) flag = flags[0].value + else addInvalid(state, shape, `${SH}flags`, 'sh:flags must be an xsd:string literal.') + } + if (flags.length) consumed.add(`${SH}flags`) + + for (const value of state.index.get(subject, `${SH}pattern`)) { + if (!isStringLiteral(value)) { + addInvalid(state, shape, `${SH}pattern`, 'sh:pattern must be an xsd:string literal.') + continue + } + constraints.push(flag === undefined ? { kind: 'pattern', pattern: value.value } : { kind: 'pattern', pattern: value.value, flags: flag }) + } + consumeWhenPresent(state, subject, consumed, `${SH}pattern`) +} + +/** Reads language-list constraints from RDF lists and preserves their declared member order. */ +function pushLanguageConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + constraints: ConstraintType[], +): void { + for (const value of state.index.get(subject, `${SH}languageIn`)) { + const members = readList(state.index, value, listOptions(state, shape, `${SH}languageIn`)) + if (!members) continue + const languages: string[] = [] + let valid = true + for (const member of members) { + if (!isStringLiteral(member)) { + addInvalid(state, shape, `${SH}languageIn`, 'sh:languageIn list members must be xsd:string literals.') + valid = false + break + } + languages.push(member.value) + } + if (valid) constraints.push({ kind: 'languageIn', languages }) + } + consumeWhenPresent(state, subject, consumed, `${SH}languageIn`) +} + +/** Reads singleton `xsd:boolean` constraints using RDF lexical boolean rules. */ +function pushBooleanConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + predicate: string, + kind: 'singleLine' | 'uniqueLang' | 'uniqueMembers' | 'reificationRequired', + constraints: ConstraintType[], +): void { + for (const value of state.index.get(subject, predicate)) { + const boolean = booleanValue(value) + if (boolean === undefined) { + addInvalid(state, shape, predicate, `${local(predicate)} must be an xsd:boolean literal.`) + continue + } + constraints.push({ kind, value: boolean }) + } + consumeWhenPresent(state, subject, consumed, predicate) +} + +/** Reads constraints that reference exactly one named or blank-node shape and diagnoses non-shape terms. */ +function pushShapeConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + predicate: string, + kind: 'memberShape' | 'not' | 'node' | 'property' | 'someValue' | 'reifierShape', + constraints: ConstraintType[], +): void { + for (const value of state.index.get(subject, predicate)) { + const reference = id(value) + if (reference) constraints.push({ kind, shape: reference }) + else addInvalid(state, shape, predicate, `${local(predicate)} must reference a shape IRI or blank node.`) + } + consumeWhenPresent(state, subject, consumed, predicate) +} + +/** Resolves SHACL shape lists such as `and`, `or`, and `xone` under the configured list limits. */ +function pushShapeListConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + predicate: string, + kind: 'and' | 'or' | 'xone', + constraints: ConstraintType[], +): void { + for (const value of state.index.get(subject, predicate)) { + const members = readList(state.index, value, listOptions(state, shape, predicate)) + if (!members) continue + const shapes: IdType[] = [] + let valid = true + for (const member of members) { + const reference = id(member) + if (!reference) { + addInvalid(state, shape, predicate, `${local(predicate)} list members must reference shapes.`) + valid = false + break + } + shapes.push(reference) + } + if (valid) constraints.push({ kind, shapes }) + } + consumeWhenPresent(state, subject, consumed, predicate) +} + +/** Reads property-pair constraints using the full SHACL path reader rather than restricting them to predicate IRIs. */ +function pushPathConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + predicate: string, + kind: 'equals' | 'disjoint' | 'subsetOf' | 'lessThan' | 'lessThanOrEquals', + constraints: ConstraintType[], +): void { + for (const value of state.index.get(subject, predicate)) { + const path = readPath(state.index, value, pathOptions(state, shape, predicate)) + constraints.push({ kind, path }) + } + consumeWhenPresent(state, subject, consumed, predicate) +} + +/** Aggregates qualified shape, min/max count, and disjointness assertions into one qualified-value constraint. */ +function pushQualifiedConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + constraints: ConstraintType[], +): void { + const shapes = state.index.get(subject, `${SH}qualifiedValueShape`) + const mins = state.index.get(subject, `${SH}qualifiedMinCount`) + const maxes = state.index.get(subject, `${SH}qualifiedMaxCount`) + const disjoints = state.index.get(subject, `${SH}qualifiedValueShapesDisjoint`) + for (const predicate of [`${SH}qualifiedValueShape`, `${SH}qualifiedMinCount`, `${SH}qualifiedMaxCount`, `${SH}qualifiedValueShapesDisjoint`]) { + consumeWhenPresent(state, subject, consumed, predicate) + } + if (!shapes.length && !mins.length && !maxes.length && !disjoints.length) return + if (shapes.length !== 1) { + addInvalid(state, shape, `${SH}qualifiedValueShape`, 'Qualified cardinality requires exactly one sh:qualifiedValueShape.') + return + } + const reference = id(shapes[0]!) + if (!reference) { + addInvalid(state, shape, `${SH}qualifiedValueShape`, 'sh:qualifiedValueShape must reference a shape.') + return + } + const minCount = optionalInteger(state, mins, shape, `${SH}qualifiedMinCount`) + const maxCount = optionalInteger(state, maxes, shape, `${SH}qualifiedMaxCount`) + const disjoint = optionalBoolean(state, disjoints, shape, `${SH}qualifiedValueShapesDisjoint`) + if (minCount === undefined && maxCount === undefined) { + addInvalid(state, shape, `${SH}qualifiedValueShape`, 'Qualified cardinality requires sh:qualifiedMinCount or sh:qualifiedMaxCount.') + return + } + const value: { + kind: 'qualified' + shape: IdType + minCount?: number + maxCount?: number + disjoint?: boolean + } = { kind: 'qualified', shape: reference } + if (minCount !== undefined) value.minCount = minCount + if (maxCount !== undefined) value.maxCount = maxCount + if (disjoint !== undefined) value.disjoint = disjoint + constraints.push(value) +} + +/** Reads `sh:closed` as boolean or SHACL 1.2 `sh:ByTypes` and resolves the optional ignored-properties list. */ +function pushClosedConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + constraints: ConstraintType[], +): void { + const ignored = readIriList(state, state.index.get(subject, `${SH}ignoredProperties`)[0], shape, `${SH}ignoredProperties`) ?? [] + consumeWhenPresent(state, subject, consumed, `${SH}ignoredProperties`) + for (const value of state.index.get(subject, `${SH}closed`)) { + const boolean = booleanValue(value) + if (boolean !== undefined) { + constraints.push({ kind: 'closed', mode: boolean, ignoredProperties: ignored }) + continue + } + if (state.version === '1.2' && value.termType === 'NamedNode' && value.value === BY_TYPES) { + constraints.push({ kind: 'closed', mode: 'byTypes', ignoredProperties: ignored }) + continue + } + addInvalid(state, shape, `${SH}closed`, 'sh:closed must be xsd:boolean or sh:ByTypes in SHACL 1.2.') + } + consumeWhenPresent(state, subject, consumed, `${SH}closed`) +} + +/** Resolves RDF-list term constraints such as `sh:in` while retaining each RDF term without JSON coercion. */ +function pushTermListConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + predicate: string, + constraints: ConstraintType[], +): void { + for (const value of state.index.get(subject, predicate)) { + const members = readList(state.index, value, listOptions(state, shape, predicate)) + if (!members) continue + const values = members.map(term) + if (values.some((entry) => entry === undefined)) { + addInvalid(state, shape, predicate, `${local(predicate)} contains an unsupported RDF term.`) + continue + } + constraints.push({ kind: 'in', values: values as TermType[] }) + } + consumeWhenPresent(state, subject, consumed, predicate) +} + +/** Reads singleton IRI-valued constraints such as `sh:rootClass` and diagnoses non-IRI values. */ +function pushIriConstraints( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + predicate: string, + kind: 'rootClass', + constraints: ConstraintType[], +): void { + for (const value of state.index.get(subject, predicate)) { + if (value.termType === 'NamedNode') constraints.push({ kind, iri: value.value }) + else addInvalid(state, shape, predicate, `${local(predicate)} must reference an IRI.`) + } + consumeWhenPresent(state, subject, consumed, predicate) +} + +/** Reads SHACL 1.2 `sh:uniqueValuesFor` as a non-empty list of property paths. */ +function pushUniqueValuesFor( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, + constraints: ConstraintType[], +): void { + for (const value of state.index.get(subject, `${SH}uniqueValuesFor`)) { + const members = state.index.isList(value) ? readList(state.index, value, listOptions(state, shape, `${SH}uniqueValuesFor`)) : [value] + if (!members) continue + const paths = members.map((member) => readPath(state.index, member, pathOptions(state, shape, `${SH}uniqueValuesFor`))) + constraints.push({ kind: 'uniqueValuesFor', paths }) + } + consumeWhenPresent(state, subject, consumed, `${SH}uniqueValuesFor`) +} + +/** Read metadata from the supplied source while preserving caller ownership. */ +function readMetadata( + state: StateType, + subject: Subject, + shape: IdType, + consumed: Set, +): MetadataType { + const names = [ + ...readTextValues(state, subject, shape, `${SH}name`, consumed), + ...readTextValues(state, subject, shape, `${RDFS}label`, consumed), + ] + const descriptions = [ + ...readTextValues(state, subject, shape, `${SH}description`, consumed), + ...readTextValues(state, subject, shape, `${RDFS}comment`, consumed), + ] + const intents = readTextValues(state, subject, shape, `${SH}intent`, consumed) + const agentInstructions = readTextValues(state, subject, shape, `${SH}agentInstruction`, consumed) + const codeIdentifiers = readStringValues(state, subject, shape, `${SH}codeIdentifier`, consumed) + const units = readTermValues(state, subject, `${SH}unit`, consumed) + const order = readLiteralValues(state, subject, shape, `${SH}order`, consumed) + const groups = readIdValues(state, subject, shape, `${SH}group`, consumed) + const values = readTermValues(state, subject, `${SH}values`, consumed) + const defaultValues = readTermValues(state, subject, `${SH}defaultValue`, consumed) + return { names, descriptions, intents, agentInstructions, codeIdentifiers, units, order, groups, values, defaultValues } +} + +/** Read assertions from the supplied source while preserving caller ownership. */ +function readAssertions(state: StateType, subject: Subject, consumed: ReadonlySet): AssertionType[] { + const values: AssertionType[] = [] + const invalidPredicates = new Set( + state.diagnostics + .filter((diagnostic) => diagnostic.shape && idKey(diagnostic.shape) === idKey(id(subject)!) && diagnostic.predicate) + .map((diagnostic) => diagnostic.predicate!), + ) + for (const quad of state.index.quads(subject)) { + if (consumed.has(quad.predicate.value) && !invalidPredicates.has(quad.predicate.value)) continue + const value = assertion(quad) + if (value) values.push(value) + } + values.sort(compareAssertion) + return values +} + +/** Read text values from the supplied source while preserving caller ownership. */ +function readTextValues( + state: StateType, + subject: Subject, + shape: IdType, + predicate: string, + consumed: Set, +): TextType[] { + const values: TextType[] = [] + for (const value of state.index.get(subject, predicate)) { + if (value.termType === 'Literal') values.push(text(value)) + else addInvalid(state, shape, predicate, `${local(predicate)} must be a literal.`) + } + consumeWhenPresent(state, subject, consumed, predicate) + return values +} + +/** Read string values from the supplied source while preserving caller ownership. */ +function readStringValues( + state: StateType, + subject: Subject, + shape: IdType, + predicate: string, + consumed: Set, +): string[] { + const values: string[] = [] + for (const value of state.index.get(subject, predicate)) { + if (isStringLiteral(value)) values.push(value.value) + else addInvalid(state, shape, predicate, `${local(predicate)} must be an xsd:string literal.`) + } + consumeWhenPresent(state, subject, consumed, predicate) + return values +} + +/** Read term values from the supplied source while preserving caller ownership. */ +function readTermValues(state: StateType, subject: Subject, predicate: string, consumed: Set): TermType[] { + const values: TermType[] = [] + for (const value of state.index.get(subject, predicate)) { + const record = term(value) + if (record) values.push(record) + } + consumeWhenPresent(state, subject, consumed, predicate) + return values +} + +/** Read literal values from the supplied source while preserving caller ownership. */ +function readLiteralValues( + state: StateType, + subject: Subject, + shape: IdType, + predicate: string, + consumed: Set, +): LiteralType[] { + const values: LiteralType[] = [] + for (const value of state.index.get(subject, predicate)) { + if (value.termType === 'Literal') values.push(literal(value)) + else addInvalid(state, shape, predicate, `${local(predicate)} must be a literal.`) + } + consumeWhenPresent(state, subject, consumed, predicate) + return values +} + +/** Read id values from the supplied source while preserving caller ownership. */ +function readIdValues( + state: StateType, + subject: Subject, + shape: IdType, + predicate: string, + consumed: Set, +): IdType[] { + const values: IdType[] = [] + for (const value of state.index.get(subject, predicate)) { + const reference = id(value) + if (reference) values.push(reference) + else addInvalid(state, shape, predicate, `${local(predicate)} must reference an IRI or blank node.`) + } + consumeWhenPresent(state, subject, consumed, predicate) + return values +} + +/** Read iri singleton from the supplied source while preserving caller ownership. */ +function readIriSingleton( + state: StateType, + subject: Subject, + shape: IdType, + predicate: string, + consumed: Set, +): string | undefined { + const values = state.index.get(subject, predicate) + consumeWhenPresent(state, subject, consumed, predicate) + if (values.length > 1) addCardinality(state, shape, predicate, `${local(predicate)} must have at most one value.`) + const value = values[0] + if (!value) return undefined + if (value.termType === 'NamedNode') return value.value + addInvalid(state, shape, predicate, `${local(predicate)} must reference an IRI.`) + return undefined +} + +/** Read iri list from the supplied source while preserving caller ownership. */ +function readIriList( + state: StateType, + value: ObjectTerm | undefined, + shape: IdType, + predicate: string, +): readonly string[] | undefined { + if (!value) return undefined + const members = readList(state.index, value, listOptions(state, shape, predicate)) + if (!members) return undefined + const values: string[] = [] + for (const member of members) { + if (member.termType !== 'NamedNode') { + addInvalid(state, shape, predicate, `${local(predicate)} list members must be IRIs.`) + return undefined + } + values.push(member.value) + } + return values +} + +/** Reads an optional singleton non-negative `xsd:integer`, diagnosing duplicate or invalid lexical values. */ +function optionalInteger( + state: StateType, + values: readonly ObjectTerm[], + shape: IdType, + predicate: string, +): number | undefined { + if (values.length > 1) addCardinality(state, shape, predicate, `${local(predicate)} must have at most one value.`) + if (!values[0]) return undefined + const value = integerValue(values[0]) + if (value === undefined || value < 0) addInvalid(state, shape, predicate, `${local(predicate)} must be a non-negative xsd:integer.`) + return value !== undefined && value >= 0 ? value : undefined +} + +/** Reads an optional singleton `xsd:boolean`, accepting canonical and numeric RDF boolean lexical forms. */ +function optionalBoolean( + state: StateType, + values: readonly ObjectTerm[], + shape: IdType, + predicate: string, +): boolean | undefined { + if (values.length > 1) addCardinality(state, shape, predicate, `${local(predicate)} must have at most one value.`) + if (!values[0]) return undefined + const value = booleanValue(values[0]) + if (value === undefined) addInvalid(state, shape, predicate, `${local(predicate)} must be xsd:boolean.`) + return value +} + +/** Decodes a safe JavaScript integer only from an `xsd:integer` lexical form. */ +function integerValue(value: ObjectTerm): number | undefined { + if (value.termType !== 'Literal' || value.datatype.value !== XSD.integer || !/^[+-]?\d+$/.test(value.value)) return undefined + const integer = Number(value.value) + return Number.isSafeInteger(integer) ? integer : undefined +} + +/** Decodes RDF `xsd:boolean` lexical forms without accepting JavaScript truthiness. */ +function booleanValue(value: ObjectTerm): boolean | undefined { + if (value.termType !== 'Literal' || value.datatype.value !== XSD.boolean) return undefined + if (value.value === 'true' || value.value === '1') return true + if (value.value === 'false' || value.value === '0') return false + return undefined +} + +/** Returns whether the supplied value satisfies the string literal contract. */ +function isStringLiteral(value: ObjectTerm): value is Literal { + return value.termType === 'Literal' && value.datatype.value === XSD.string +} + +/** Marks a predicate consumed only when the source graph actually contains at least one value for it. */ +function consumeWhenPresent(state: StateType, subject: Subject, consumed: Set, predicate: string): void { + if (state.index.get(subject, predicate).length) consumed.add(predicate) +} + +/** Creates bounded RDF-list read options that attach diagnostics to the owning shape and predicate. */ +function listOptions(state: StateType, shape: IdType, predicate: string) { + return { maxItems: state.maxListItems, diagnostics: state.diagnostics, shape, predicate } +} + +/** Creates bounded SHACL-path read options that attach recursion diagnostics to the owning assertion. */ +function pathOptions(state: StateType, shape: IdType, predicate: string) { + return { + maxDepth: state.maxPathDepth, + maxListItems: state.maxListItems, + diagnostics: state.diagnostics, + shape, + predicate, + } +} + +/** Records SHACL 1.2-only terms encountered in 1.0 mode while leaving those assertions available for loss-preserving reads. */ +function recordVersionDiagnostics(state: StateType, subject: Subject, shape: IdType): void { + if (state.version !== '1.0') return + const types = state.index.get(subject, RDF.type) + if (types.some((value) => value.termType === 'NamedNode' && value.value === SHAPE_CLASS)) { + state.diagnostics.push({ + code: 'unsupported-version', + severity: 'warning', + message: 'sh:ShapeClass is a SHACL 1.2 feature and is retained while reading in 1.0 mode.', + shape, + predicate: RDF.type, + }) + } + for (const predicate of SHACL_12_PREDICATES) { + if (!state.index.get(subject, predicate).length) continue + state.diagnostics.push({ + code: 'unsupported-version', + severity: 'warning', + message: `${local(predicate)} is a SHACL 1.2 Core feature and is retained while reading in 1.0 mode.`, + shape, + predicate, + }) + } +} + +/** Records a SHACL source-cardinality error without mutating or discarding the original RDF assertion. */ +function addCardinality(state: StateType, shape: IdType, predicate: string, message: string): void { + state.diagnostics.push({ code: 'invalid-cardinality', severity: 'error', message, shape, predicate }) +} + +/** Records an invalid known SHACL value; raw assertions remain available to newer or external interpreters. */ +function addInvalid(state: StateType, shape: IdType, predicate: string, message: string): void { + state.diagnostics.push({ code: 'invalid-value', severity: 'error', message, shape, predicate }) +} + +/** Produces a compact SHACL term label for diagnostics only; it is never used as semantic identity. */ +function local(predicate: string): string { + return predicate.startsWith(SH) ? `sh:${predicate.slice(SH.length)}` : predicate +} + +/** Produces a stable sort/deduplication key that distinguishes named and blank-node shape identifiers. */ +function idKey(value: IdType): string { + return `${value.kind}:${value.value}` +} + +/** Compare assertion using deterministic semantic ordering. */ +function compareAssertion(left: AssertionType, right: AssertionType): number { + return `${idKey(left.subject)}\u0000${left.predicate}\u0000${JSON.stringify(left.object)}`.localeCompare(`${idKey(right.subject)}\u0000${right.predicate}\u0000${JSON.stringify(right.object)}`) +} + +/** Compare diagnostic using deterministic semantic ordering. */ +function compareDiagnostic(left: DiagnosticType, right: DiagnosticType): number { + return `${left.code}\u0000${left.predicate ?? ''}\u0000${left.message}`.localeCompare(`${right.code}\u0000${right.predicate ?? ''}\u0000${right.message}`) +} diff --git a/packages/rdf/shape/read_test.ts b/packages/rdf/shape/read_test.ts new file mode 100644 index 0000000..0abccf5 --- /dev/null +++ b/packages/rdf/shape/read_test.ts @@ -0,0 +1,95 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { blankNode, literal, namedNode, quad, RDF, XSD } from '../mod.ts' +import { parse } from '../turtle/mod.ts' +import { read } from './mod.ts' + +const SH = 'http://www.w3.org/ns/shacl#' + +function getShape(graph: Awaited>, suffix: string) { + return graph.shapes.find((shape) => shape.id.value.endsWith(suffix)) +} + +describe('@okikio/rdf/shape', () => { + it('reads SHACL 1.2 Core paths, constraints, metadata, and extension assertions', async () => { + const source = ` + @prefix sh: . + @prefix ex: . + @prefix rdf: . + @prefix xsd: . + + ex:PersonShape a sh:NodeShape, ex:Profile ; + sh:targetClass ex:Person ; + sh:closed sh:ByTypes ; + sh:ignoredProperties ( rdf:type ) ; + sh:uniqueValuesFor ( ex:id ex:tenant ) ; + sh:property [ + a sh:PropertyShape ; + sh:path [ sh:alternativePath ( ex:name [ sh:inversePath ex:label ] ) ] ; + sh:class ( ex:Person ex:Organization ) ; + sh:minCount 1 ; + sh:pattern "^[a-z]+$" ; + sh:flags "i" ; + sh:name "Display name"@en ; + ex:customConstraint "retained" + ] . + ` + + const graph = await read(parse(source), { version: '1.2' }) + expect(graph.diagnostics).toHaveLength(0) + + const person = getShape(graph, 'PersonShape') + expect(person?.types.includes('https://example.com/Profile')).toBe(true) + expect(person?.constraints.some((value) => value.kind === 'closed' && value.mode === 'byTypes')).toBe(true) + expect(person?.constraints.some((value) => value.kind === 'uniqueValuesFor' && value.paths.length === 2)).toBe(true) + + const property = graph.shapes.find((shape) => shape.kind === 'property' && shape.id.kind === 'blank') + expect(property?.path?.kind).toBe('alternative') + if (property?.path?.kind === 'alternative') { + expect(property.path.items).toHaveLength(2) + expect(property.path.items[1]?.kind).toBe('inverse') + } + expect(property?.constraints.some((value) => value.kind === 'class' && value.choices.length === 2)).toBe(true) + expect(property?.metadata.names[0]?.value).toBe('Display name') + expect(property?.assertions.some((value) => value.predicate === 'https://example.com/customConstraint')).toBe(true) + }) + + it('reports cyclic property paths without recursive overflow', async () => { + const shape = namedNode('https://example.com/Shape') + const path = blankNode('path') + const values = [ + quad(shape, namedNode(RDF.type), namedNode(`${SH}PropertyShape`)), + quad(shape, namedNode(`${SH}path`), path), + quad(path, namedNode(`${SH}inversePath`), path), + ] + + const graph = await read(values) + expect(graph.diagnostics.some((value) => value.code === 'path-cycle')).toBe(true) + expect(graph.shapes[0]?.path?.kind).toBe('inverse') + }) + + it('retains malformed known values as diagnostics instead of weakening them silently', async () => { + const shape = namedNode('https://example.com/Shape') + const values = [ + quad(shape, namedNode(RDF.type), namedNode(`${SH}NodeShape`)), + quad(shape, namedNode(`${SH}minCount`), literal('-1', namedNode(XSD.integer))), + quad(shape, namedNode(`${SH}closed`), literal('not-a-boolean', namedNode(XSD.boolean))), + ] + + const graph = await read(values) + expect(graph.shapes[0]?.constraints).toHaveLength(0) + expect(graph.diagnostics.filter((value) => value.code === 'invalid-value')).toHaveLength(2) + expect(graph.shapes[0]?.assertions.some((value) => value.predicate === `${SH}minCount`)).toBe(true) + expect(graph.shapes[0]?.assertions.some((value) => value.predicate === `${SH}closed`)).toBe(true) + }) + + it('warns when 1.2-only Core terms are read through the 1.0 interpretation mode', async () => { + const shape = namedNode('https://example.com/Shape') + const graph = await read([ + quad(shape, namedNode(RDF.type), namedNode(`${SH}NodeShape`)), + quad(shape, namedNode(`${SH}singleLine`), literal('true', namedNode(XSD.boolean))), + ], { version: '1.0' }) + expect(graph.diagnostics.some((value) => value.code === 'unsupported-version')).toBe(true) + expect(graph.shapes[0]?.constraints.some((value) => value.kind === 'singleLine')).toBe(true) + }) +}) diff --git a/packages/rdf/shape/value.ts b/packages/rdf/shape/value.ts new file mode 100644 index 0000000..3c66dde --- /dev/null +++ b/packages/rdf/shape/value.ts @@ -0,0 +1,85 @@ +/** Serializable RDF term conversion for the SHACL model. @module */ + +import type { Graph, Literal, Quad, Term } from '../term.ts' +import type { AssertionType, IdType, LiteralType, TermType, TextType } from './model.ts' + +/** Converts an RDF graph node into a serializable SHACL identifier. */ +export function id(term: Term): IdType | undefined { + if (term.termType === 'NamedNode') return { kind: 'iri', value: term.value } + if (term.termType === 'BlankNode') return { kind: 'blank', value: term.value } + return undefined +} + +/** Converts an RDF term into the serializable shape representation. */ +export function term(value: Term): TermType | undefined { + switch (value.termType) { + case 'NamedNode': + return { kind: 'iri', value: value.value } + case 'BlankNode': + return { kind: 'blank', value: value.value } + case 'Literal': + return literal(value as Literal) + case 'Quad': { + const triple = value as Quad + const subject = id(triple.subject) + const object = term(triple.object) + if (!subject || !object) return undefined + return { kind: 'triple', subject, predicate: triple.predicate.value, object } + } + case 'Variable': + case 'DefaultGraph': + return undefined + } +} + +/** Converts an RDF literal into a serializable SHACL literal. */ +export function literal(value: Literal): LiteralType { + const record: { + kind: 'literal' + value: string + datatype: string + language?: string + direction?: Literal['direction'] extends '' ? never : 'ltr' | 'rtl' + } = { + kind: 'literal', + value: value.value, + datatype: value.datatype.value, + } + if (value.language) record.language = value.language + if (value.direction) record.direction = value.direction + return record +} + +/** Converts an RDF literal into localized human-facing text. */ +export function text(value: Literal): TextType { + const record: { + value: string + datatype: string + language?: string + direction?: 'ltr' | 'rtl' + } = { value: value.value, datatype: value.datatype.value } + if (value.language) record.language = value.language + if (value.direction) record.direction = value.direction + return record +} + +/** Converts an unhandled quad into a loss-preserving assertion record. */ +export function assertion(quad: Quad): AssertionType | undefined { + const object = term(quad.object) + if (!object) return undefined + const subject = id(quad.subject) + if (!subject) return undefined + const record: { subject: IdType; predicate: string; object: TermType; graph?: IdType } = { + subject, + predicate: quad.predicate.value, + object, + } + const graph = graphId(quad.graph) + if (graph) record.graph = graph + return record +} + +/** Returns a serializable graph identifier when a quad belongs to a named graph. */ +function graphId(graph: Graph): IdType | undefined { + return graph.termType === 'DefaultGraph' ? undefined : id(graph) +} diff --git a/packages/rdf/source.ts b/packages/rdf/source.ts new file mode 100644 index 0000000..851537a --- /dev/null +++ b/packages/rdf/source.ts @@ -0,0 +1,23 @@ +/** Incremental RDF source and sink contracts. @module */ + +import type { Quad } from './term.ts' + +/** Synchronous RDF quad source. */ +export interface Source extends Iterable {} + +/** Asynchronous RDF quad source. */ +export interface AsyncSource extends AsyncIterable {} + +/** A sink that consumes an RDF quad sequence without taking source ownership. */ +export interface Sink { + import(source: Iterable | AsyncIterable, options?: { readonly signal?: AbortSignal }): Promise +} + +/** Converts sync or async quad input into one async iteration contract. */ +export async function* iterate(source: Iterable | AsyncIterable): AsyncGenerator { + if (Symbol.asyncIterator in Object(source)) { + for await (const quad of source as AsyncIterable) yield quad + return + } + for (const quad of source as Iterable) yield quad +} diff --git a/packages/rdf/source_test.ts b/packages/rdf/source_test.ts new file mode 100644 index 0000000..51b7ecd --- /dev/null +++ b/packages/rdf/source_test.ts @@ -0,0 +1,23 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { literal, namedNode, quad } from './mod.ts' +import { iterate } from './source.ts' + +const value = quad(namedNode('urn:s'), namedNode('urn:p'), literal('o')) + +describe('@okikio/rdf source iteration', () => { + it('adapts synchronous and asynchronous quad sources to one async contract', async () => { + const sync = [] + for await (const item of iterate([value])) sync.push(item) + + async function* asyncSource() { + yield value + } + const asyncValues = [] + for await (const item of iterate(asyncSource())) asyncValues.push(item) + + expect(sync).toHaveLength(1) + expect(asyncValues).toHaveLength(1) + expect(sync[0]?.equals(asyncValues[0])).toBe(true) + }) +}) diff --git a/packages/rdf/term.ts b/packages/rdf/term.ts new file mode 100644 index 0000000..7ea659b --- /dev/null +++ b/packages/rdf/term.ts @@ -0,0 +1,259 @@ +/** + * RDF 1.2-compatible term contracts with RDF/JS interoperability. + * + * The public objects stay semantic and immutable. Storage engines may map them + * to integer IDs or packed records internally, but those representations must + * not leak through this module. + * + * @module + */ + +/** Initial text direction attached to an RDF 1.2 directional language string. */ +export type Direction = 'ltr' | 'rtl' + +/** RDF/JS-compatible base term. */ +export interface Term { + readonly termType: 'NamedNode' | 'BlankNode' | 'Literal' | 'Variable' | 'DefaultGraph' | 'Quad' + readonly value: string + equals(other?: Term | null): boolean +} + +/** An RDF IRI term. */ +export interface NamedNode extends Term { + readonly termType: 'NamedNode' +} + +/** An RDF blank node. */ +export interface BlankNode extends Term { + readonly termType: 'BlankNode' +} + +/** An RDF query variable used by RDF/JS-compatible query surfaces. */ +export interface Variable extends Term { + readonly termType: 'Variable' +} + +/** The default graph name. */ +export interface DefaultGraph extends Term { + readonly termType: 'DefaultGraph' + readonly value: '' +} + +/** RDF 1.2 literal, including optional language direction. */ +export interface Literal extends Term { + readonly termType: 'Literal' + readonly language: string + readonly direction: Direction | '' + readonly datatype: NamedNode +} + +/** RDF triple subject. RDF 1.2 triple terms are object terms, not graph subjects. */ +export type Subject = NamedNode | BlankNode + +/** RDF triple predicate. */ +export type Predicate = NamedNode + +/** RDF triple object, including RDF 1.2 triple terms. */ +export type ObjectTerm = NamedNode | BlankNode | Literal | Quad + +/** RDF dataset graph name. */ +export type Graph = DefaultGraph | NamedNode | BlankNode + +/** + * RDF/JS-compatible quad. + * + * A triple term is represented by a Quad whose graph is the default graph. The + * factory rejects non-default graphs when a Quad is embedded as another + * triple's object so the native model remains consistent with RDF 1.2. + */ +export interface Quad extends Term { + readonly termType: 'Quad' + readonly value: '' + readonly subject: Subject + readonly predicate: Predicate + readonly object: ObjectTerm + readonly graph: Graph +} + +/** Any public RDF term. */ +export type TermType = NamedNode | BlankNode | Literal | Variable | DefaultGraph | Quad + +/** RDF/JS-compatible directional-language factory input. */ +export interface DirectionalLanguage { + readonly language: string + readonly direction?: Direction | null +} + +/** RDF namespace constants used by the core term model. */ +export const RDF = { + type: 'http://www.w3.org/1999/02/22-rdf-syntax-ns#type', + statement: 'http://www.w3.org/1999/02/22-rdf-syntax-ns#Statement', + subject: 'http://www.w3.org/1999/02/22-rdf-syntax-ns#subject', + predicate: 'http://www.w3.org/1999/02/22-rdf-syntax-ns#predicate', + object: 'http://www.w3.org/1999/02/22-rdf-syntax-ns#object', + langString: 'http://www.w3.org/1999/02/22-rdf-syntax-ns#langString', + dirLangString: 'http://www.w3.org/1999/02/22-rdf-syntax-ns#dirLangString', + first: 'http://www.w3.org/1999/02/22-rdf-syntax-ns#first', + rest: 'http://www.w3.org/1999/02/22-rdf-syntax-ns#rest', + reifies: 'http://www.w3.org/1999/02/22-rdf-syntax-ns#reifies', + nil: 'http://www.w3.org/1999/02/22-rdf-syntax-ns#nil', +} as const + +/** XML Schema namespace constants needed by the core literal factory. */ +export const XSD = { + string: 'http://www.w3.org/2001/XMLSchema#string', + boolean: 'http://www.w3.org/2001/XMLSchema#boolean', + integer: 'http://www.w3.org/2001/XMLSchema#integer', + decimal: 'http://www.w3.org/2001/XMLSchema#decimal', + double: 'http://www.w3.org/2001/XMLSchema#double', + date: 'http://www.w3.org/2001/XMLSchema#date', + dateTime: 'http://www.w3.org/2001/XMLSchema#dateTime', +} as const + +/** Base immutable term implementation. */ +abstract class BaseTerm implements Term { + abstract readonly termType: Term['termType'] + + readonly value: string + + /** Stores the immutable lexical value shared by concrete RDF term implementations. */ + constructor(value: string) { + this.value = value + } + + /** Compares simple RDF terms by term kind and lexical value. */ + equals(other?: Term | null): boolean { + return other !== null && other !== undefined && this.termType === other.termType && this.value === other.value + } +} + +/** Immutable named-node implementation. */ +export class NamedNodeValue extends BaseTerm implements NamedNode { + readonly termType = 'NamedNode' as const +} + +/** Immutable blank-node implementation. */ +export class BlankNodeValue extends BaseTerm implements BlankNode { + readonly termType = 'BlankNode' as const +} + +/** Immutable variable implementation. */ +export class VariableValue extends BaseTerm implements Variable { + readonly termType = 'Variable' as const +} + +/** Immutable default-graph singleton implementation. */ +export class DefaultGraphValue extends BaseTerm implements DefaultGraph { + readonly termType = 'DefaultGraph' as const + readonly value = '' as const + + /** Creates the RDF default graph singleton value with the required empty lexical form. */ + constructor() { + super('') + } + + /** Treats every RDF/JS-compatible DefaultGraph term as term-equal. */ + override equals(other?: Term | null): boolean { + return other?.termType === 'DefaultGraph' + } +} + +/** Immutable RDF literal implementation. */ +export class LiteralValue extends BaseTerm implements Literal { + readonly termType = 'Literal' as const + readonly datatype: NamedNode + readonly language: string + readonly direction: Direction | '' + + /** Stores literal lexical form, datatype, language, and RDF 1.2 base direction without coercion. */ + constructor(value: string, datatype: NamedNode, language = '', direction: Direction | '' = '') { + super(value) + this.datatype = datatype + this.language = language + this.direction = direction + } + + /** Applies RDF literal term equality, including case-insensitive language tags and exact direction. */ + override equals(other?: Term | null): boolean { + if (other?.termType !== 'Literal') return false + const literal = other as Literal + return this.value === literal.value && + this.datatype.equals(literal.datatype) && + this.language.toLowerCase() === literal.language.toLowerCase() && + this.direction === literal.direction + } +} + +/** Immutable quad and triple-term implementation. */ +export class QuadValue extends BaseTerm implements Quad { + readonly termType = 'Quad' as const + readonly value = '' as const + readonly subject: Subject + readonly predicate: Predicate + readonly object: ObjectTerm + readonly graph: Graph + + /** Stores one immutable quad; default-graph quads can also represent RDF 1.2 triple terms. */ + constructor(subject: Subject, predicate: Predicate, object: ObjectTerm, graph: Graph) { + super('') + this.subject = subject + this.predicate = predicate + this.object = object + this.graph = graph + } + + /** Compares every quad component recursively using RDF term equality. */ + override equals(other?: Term | null): boolean { + if (other?.termType !== 'Quad') return false + const quad = other as Quad + return this.subject.equals(quad.subject) && + this.predicate.equals(quad.predicate) && + this.object.equals(quad.object) && + this.graph.equals(quad.graph) + } +} + +/** Returns whether a value follows the RDF/JS term contract. */ +export function isTerm(value: unknown): value is TermType { + if (typeof value !== 'object' || value === null) return false + const candidate = value as Partial + return typeof candidate.termType === 'string' && typeof candidate.value === 'string' && + typeof candidate.equals === 'function' +} + +/** Returns whether two RDF terms are term-equal. */ +export function equals(left: Term, right: Term): boolean { + return left.equals(right) +} + +/** + * Returns a collision-safe semantic key for one RDF term. + * + * Length prefixes prevent delimiter collisions and make the key suitable for + * Maps and persistent dictionaries without depending on one serialization. + */ +export function key(term: Term): string { + switch (term.termType) { + case 'NamedNode': + return atom('N', term.value) + case 'BlankNode': + return atom('B', term.value) + case 'Variable': + return atom('V', term.value) + case 'DefaultGraph': + return 'D' + case 'Literal': { + const literal = term as Literal + return `L${atom('', literal.value)}${atom('', literal.datatype.value)}${atom('', literal.language.toLowerCase())}${atom('', literal.direction)}` + } + case 'Quad': { + const quad = term as Quad + return `Q${atom('', key(quad.subject))}${atom('', key(quad.predicate))}${atom('', key(quad.object))}${atom('', key(quad.graph))}` + } + } +} + +/** Length-prefixes a string so nested term keys remain unambiguous. */ +function atom(prefix: string, value: string): string { + return `${prefix}${value.length}:${value}` +} diff --git a/packages/rdf/term_test.ts b/packages/rdf/term_test.ts new file mode 100644 index 0000000..1dfb2f2 --- /dev/null +++ b/packages/rdf/term_test.ts @@ -0,0 +1,74 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { + RDF, + XSD, + blankNode, + defaultGraph, + equals, + fromQuad, + key, + literal, + namedNode, + quad, + triple, + variable, +} from './mod.ts' + +describe('@okikio/rdf terms and factories', () => { + it('normalizes variables and language tags without changing RDF identity rules', () => { + expect(variable('$name').value).toBe('name') + expect(() => variable('?')).toThrow() + + const left = literal('bonjour', 'FR') + const right = literal('bonjour', 'fr') + expect(left.language).toBe('fr') + expect(left.datatype.value).toBe(RDF.langString) + expect(equals(left, right)).toBe(true) + }) + + it('creates RDF 1.2 directional language strings', () => { + const value = literal('مرحبا', { language: 'AR', direction: 'rtl' }) + expect(value.language).toBe('ar') + expect(value.direction).toBe('rtl') + expect(value.datatype.value).toBe(RDF.dirLangString) + }) + + it('uses xsd:string only for plain strings', () => { + const value = literal('Widget') + expect(value.datatype.value).toBe(XSD.string) + expect(value.language).toBe('') + }) + + it('represents triple terms as default-graph quads and rejects named-graph embedded quads', () => { + const s = namedNode('urn:s') + const p = namedNode('urn:p') + const embedded = triple(s, p, literal('value')) + expect(embedded.graph.termType).toBe('DefaultGraph') + expect(quad(s, p, embedded).object.equals(embedded)).toBe(true) + + const named = quad(s, p, literal('value'), namedNode('urn:g')) + expect(() => quad(s, p, named)).toThrow('triple term') + }) + + it('creates collision-safe semantic keys for nested terms', () => { + const first = literal('a:1', namedNode('urn:type')) + const second = literal('a', namedNode('1:urn:type')) + expect(key(first) === key(second)).toBe(false) + + const embedded = triple(blankNode('s'), namedNode('urn:p'), first) + expect(key(embedded).startsWith('Q')).toBe(true) + }) + + it('copies RDF/JS-compatible quads recursively and keeps the default graph singleton', () => { + const source = quad( + namedNode('urn:s'), + namedNode('urn:p'), + triple(namedNode('urn:a'), namedNode('urn:b'), literal('c')), + ) + const copy = fromQuad(source) + expect(copy === source).toBe(false) + expect(copy.equals(source)).toBe(true) + expect(defaultGraph() === defaultGraph()).toBe(true) + }) +}) diff --git a/packages/rdf/text.ts b/packages/rdf/text.ts new file mode 100644 index 0000000..a9b77cd --- /dev/null +++ b/packages/rdf/text.ts @@ -0,0 +1,114 @@ +/** Incremental UTF-8/text source helpers shared by RDF syntax parsers. @module */ + +const DIRECT_CHUNK_SIZE = 16 * 1024 + +/** Byte/text source accepted by streaming RDF parsers. */ +export type TextSource = + | string + | Uint8Array + | Iterable + | AsyncIterable + | ReadableStream + +/** + * Iterates source chunks without taking ownership of ordinary iterables. + * + * A Web `ReadableStream` reader is cancelled when the consumer returns before + * source completion. This prevents an upstream producer from continuing work + * after a parser or its caller has stopped reading. + */ +export async function* chunks(source: TextSource, signal?: AbortSignal): AsyncGenerator { + if (typeof source === 'string') { + for (let offset = 0; offset < source.length; offset += DIRECT_CHUNK_SIZE) { + throwIfAborted(signal) + yield source.slice(offset, offset + DIRECT_CHUNK_SIZE) + } + return + } + + if (source instanceof Uint8Array) { + for (let offset = 0; offset < source.byteLength; offset += DIRECT_CHUNK_SIZE) { + throwIfAborted(signal) + yield source.subarray(offset, offset + DIRECT_CHUNK_SIZE) + } + return + } + + if (source instanceof ReadableStream) { + const reader = source.getReader() + let complete = false + try { + while (true) { + throwIfAborted(signal) + const item = await read(reader, signal) + if (item.done) { + complete = true + return + } + yield item.value + } + } finally { + if (!complete) await reader.cancel('RDF parser consumer stopped before source completion').catch(() => undefined) + reader.releaseLock() + } + } + + if (Symbol.asyncIterator in Object(source)) { + for await (const chunk of source as AsyncIterable) { + throwIfAborted(signal) + yield chunk + } + return + } + + for (const chunk of source as Iterable) { + throwIfAborted(signal) + yield chunk + } +} + + +/** Read from the supplied source while preserving caller ownership. */ +function read( + reader: ReadableStreamDefaultReader, + signal?: AbortSignal, +): Promise> { + if (!signal) return reader.read() + if (signal.aborted) { + void reader.cancel(signal.reason).catch(() => undefined) + return Promise.reject(signal.reason ?? new DOMException('Aborted', 'AbortError')) + } + + return new Promise((resolve, reject) => { + let settled = false + const finish = () => signal.removeEventListener('abort', onAbort) + const onAbort = () => { + if (settled) return + settled = true + finish() + void reader.cancel(signal.reason).catch(() => undefined) + reject(signal.reason ?? new DOMException('Aborted', 'AbortError')) + } + + signal.addEventListener('abort', onAbort, { once: true }) + reader.read().then( + (value) => { + if (settled) return + settled = true + finish() + resolve(value) + }, + (error) => { + if (settled) return + settled = true + finish() + reject(error) + }, + ) + }) +} + +/** Throws the caller's abort reason before additional parsing work is accepted. */ +export function throwIfAborted(signal?: AbortSignal): void { + if (signal?.aborted) throw signal.reason ?? new DOMException('Aborted', 'AbortError') +} diff --git a/packages/rdf/text_test.ts b/packages/rdf/text_test.ts new file mode 100644 index 0000000..d51e02c --- /dev/null +++ b/packages/rdf/text_test.ts @@ -0,0 +1,49 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { chunks, throwIfAborted } from './text.ts' + +/** Collects parser source chunks as decoded strings for source-contract tests. */ +async function collect(source: Parameters[0], signal?: AbortSignal): Promise { + const decoder = new TextDecoder() + const values: string[] = [] + for await (const value of chunks(source, signal)) values.push(typeof value === 'string' ? value : decoder.decode(value)) + return values +} + +describe('@okikio/rdf text sources', () => { + it('windows direct strings and bytes instead of exposing one unbounded parser window', async () => { + const text = 'x'.repeat(40 * 1024) + const strings = await collect(text) + const bytes = await collect(new TextEncoder().encode(text)) + expect(strings.length > 1).toBe(true) + expect(bytes.length > 1).toBe(true) + expect(strings.join('')).toBe(text) + expect(bytes.join('')).toBe(text) + }) + + it('cancels a pending Web Stream read when the signal aborts', async () => { + let cancelled = false + const stream = new ReadableStream({ + cancel() { + cancelled = true + }, + }) + const controller = new AbortController() + const iterator = chunks(stream, controller.signal) + const pending = iterator.next() + controller.abort(new Error('stop')) + await expect(pending).rejects.toThrow('stop') + expect(cancelled).toBe(true) + }) + + it('throws the caller abort reason before accepting more work', () => { + const controller = new AbortController() + controller.abort(new Error('cancelled')) + try { + throwIfAborted(controller.signal) + throw new Error('Expected abort failure.') + } catch (error) { + expect(error instanceof Error ? error.message : String(error)).toBe('cancelled') + } + }) +}) diff --git a/packages/rdf/transform.ts b/packages/rdf/transform.ts new file mode 100644 index 0000000..21b3602 --- /dev/null +++ b/packages/rdf/transform.ts @@ -0,0 +1,95 @@ +/** Internal bridge from Node-style streaming RDF parsers to Web-oriented RDF sources. @module */ + +import { fromQuad } from './factory.ts' +import type { Quad } from './term.ts' +import { chunks, throwIfAborted, type TextSource } from './text.ts' + +/** Minimal stream surface required from an external streaming RDF parser. */ +export interface TransformParserType extends AsyncIterable { + write(chunk: string | Uint8Array): boolean + end(): void + once(event: 'drain' | 'close' | 'error', listener: (...args: unknown[]) => void): this + off(event: 'drain' | 'close' | 'error', listener: (...args: unknown[]) => void): this + destroy(error?: Error): void +} + +/** + * Reads an external Transform parser through the project's async-iterable RDF contract. + * + * The parser resource is owned by this operation. The source is borrowed. Returning + * from iteration early aborts the source pump and destroys the parser so upstream + * network, file, or Web Stream work cannot continue without a consumer. + */ +export async function* parseTransform( + parser: TransformParserType, + source: TextSource, + options: { readonly label: string; readonly signal?: AbortSignal }, +): AsyncGenerator { + throwIfAborted(options.signal) + const lifecycle = new AbortController() + const signal = options.signal ? AbortSignal.any([options.signal, lifecycle.signal]) : lifecycle.signal + let complete = false + const pumping = pump(parser, source, signal, options.label) + + try { + for await (const value of parser) { + throwIfAborted(signal) + yield fromQuad(value) + } + await pumping + complete = true + } finally { + if (!complete) { + lifecycle.abort(new DOMException(`${options.label} consumer stopped before source completion`, 'AbortError')) + parser.destroy() + } + await pumping.catch(() => undefined) + } +} + +/** Feeds source chunks into the parser while respecting writable backpressure. */ +async function pump( + parser: TransformParserType, + source: TextSource, + signal: AbortSignal, + label: string, +): Promise { + try { + for await (const value of chunks(source, signal)) { + throwIfAborted(signal) + if (!parser.write(value)) await drain(parser, signal, label) + } + throwIfAborted(signal) + parser.end() + } catch (error) { + parser.destroy(toError(error)) + throw error + } +} + +/** Waits for parser capacity while remaining interruptible by caller cancellation. */ +function drain(parser: TransformParserType, signal: AbortSignal, label: string): Promise { + return new Promise((resolve, reject) => { + const cleanup = () => { + signal.removeEventListener('abort', onAbort) + parser.off('drain', onDrain) + parser.off('close', onClose) + parser.off('error', onError) + } + const onDrain = () => { cleanup(); resolve() } + const onClose = () => { cleanup(); reject(new Error(`${label} closed while waiting for writable capacity.`)) } + const onError = (value: unknown) => { cleanup(); reject(toError(value)) } + const onAbort = () => { cleanup(); reject(signal.reason ?? new DOMException('Aborted', 'AbortError')) } + + if (signal.aborted) return onAbort() + signal.addEventListener('abort', onAbort, { once: true }) + parser.once('drain', onDrain) + parser.once('close', onClose) + parser.once('error', onError) + }) +} + +/** Converts the supplied value to error without changing semantic identity. */ +function toError(value: unknown): Error { + return value instanceof Error ? value : new Error(String(value)) +} diff --git a/packages/rdf/trig/mod.ts b/packages/rdf/trig/mod.ts new file mode 100644 index 0000000..9bc122b --- /dev/null +++ b/packages/rdf/trig/mod.ts @@ -0,0 +1,37 @@ +/** RDF 1.2 TriG parser and conservative streaming-friendly serializer. @module */ + +import { parseCompact, type CompactEvent, type CompactOptions } from '../compact.ts' +import type { TextSource } from '../text.ts' +import type { Quad } from '../term.ts' +import { writeQuad, writeTerm } from '../write.ts' + +export type { CompactDiagnostic as Diagnostic, CompactEvent as ParseEvent, CompactOptions as ParseOptions, CompactRange as SourceRange } from '../compact.ts' +export type { TextSource } from '../text.ts' + +/** Emits TriG directives, semantic quads, and optional tolerant diagnostics. */ +export function events(source: TextSource, options: CompactOptions = {}): AsyncGenerator { + return parseCompact(source, options, true) +} + +/** Parses TriG incrementally and emits semantic RDF dataset quads. */ +export async function* parse(source: TextSource, options: CompactOptions = {}): AsyncGenerator { + for await (const event of events(source, options)) { + if (event.kind === 'quad') yield event.quad + } +} + +/** + * Serializes a dataset as valid TriG without grouping named graphs in memory. + * + * Repeated graph blocks are legal TriG and let the serializer retain an O(1) + * working set for an arbitrary input iteration order. + */ +export function serialize(source: Iterable, options: { readonly version?: boolean } = {}): string { + const lines: string[] = [] + if (options.version) lines.push('VERSION "1.2"') + for (const value of source) { + if (value.graph.termType === 'DefaultGraph') lines.push(writeQuad(value, false)) + else lines.push(`${writeTerm(value.graph)} { ${writeQuad(value, false)} }`) + } + return lines.length === 0 ? '' : `${lines.join('\n')}\n` +} diff --git a/packages/rdf/trig/parse_test.ts b/packages/rdf/trig/parse_test.ts new file mode 100644 index 0000000..c775a65 --- /dev/null +++ b/packages/rdf/trig/parse_test.ts @@ -0,0 +1,27 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { parse } from './mod.ts' + +async function collect(source: AsyncIterable): Promise { + const values: T[] = [] + for await (const value of source) values.push(value) + return values +} + +describe('@okikio/rdf/trig', () => { + it('retains default and named graph identity', async () => { + const values = await collect(parse('PREFIX : \n:a :p :d .\n:g { :a :p :n . }')) + expect(values.map((value) => value.graph.value)).toEqual(['', 'https://example.com/g']) + }) + it('distinguishes a top-level property list from the empty anonymous graph label', async () => { + const property = await collect(parse('@prefix : . [ :inside :value ] :outside :tail .')) + expect(property).toHaveLength(2) + expect(property.every((item) => item.graph.termType === 'DefaultGraph')).toBe(true) + expect(property[0]?.subject.value).toBe(property[1]?.subject.value) + + const graph = await collect(parse('@prefix : . [] { :s :p :o . }')) + expect(graph).toHaveLength(1) + expect(graph[0]?.graph.termType).toBe('BlankNode') + }) + +}) diff --git a/packages/rdf/turtle/mod.ts b/packages/rdf/turtle/mod.ts new file mode 100644 index 0000000..da82f66 --- /dev/null +++ b/packages/rdf/turtle/mod.ts @@ -0,0 +1,38 @@ +/** RDF 1.2 Turtle parser and conservative serializer. @module */ + +import { parseCompact, type CompactEvent, type CompactOptions } from '../compact.ts' +import type { TextSource } from '../text.ts' +import type { Quad } from '../term.ts' +import { writeQuad } from '../write.ts' + +export type { CompactDiagnostic as Diagnostic, CompactEvent as ParseEvent, CompactOptions as ParseOptions, CompactRange as SourceRange } from '../compact.ts' +export type { TextSource } from '../text.ts' + +/** Emits Turtle directives, semantic quads, and optional tolerant diagnostics. */ +export function events(source: TextSource, options: CompactOptions = {}): AsyncGenerator { + return parseCompact(source, options, false) +} + +/** Parses Turtle incrementally and emits semantic RDF quads. */ +export async function* parse(source: TextSource, options: CompactOptions = {}): AsyncGenerator { + for await (const event of events(source, options)) { + if (event.kind === 'quad') yield event.quad + } +} + +/** + * Serializes an RDF graph using the explicit-IRI subset of Turtle. + * + * This intentionally optimizes for deterministic correctness rather than pretty + * prefix compaction. Named-graph quads are rejected because Turtle represents a + * graph, while TriG represents a dataset. + */ +export function serialize(source: Iterable, options: { readonly version?: boolean } = {}): string { + const lines: string[] = [] + if (options.version) lines.push('VERSION "1.2"') + for (const value of source) { + if (value.graph.termType !== 'DefaultGraph') throw new TypeError('Turtle serialization cannot contain named-graph quads.') + lines.push(writeQuad(value, false)) + } + return lines.length === 0 ? '' : `${lines.join('\n')}\n` +} diff --git a/packages/rdf/turtle/parse_bench.ts b/packages/rdf/turtle/parse_bench.ts new file mode 100644 index 0000000..0bca348 --- /dev/null +++ b/packages/rdf/turtle/parse_bench.ts @@ -0,0 +1,43 @@ +/** Decision benchmark for compact Turtle syntax versus explicit N-Quads representation. @module */ + +import { bench, do_not_optimize, group, run } from 'mitata' +import { datasetKey } from '../dataset.ts' +import { parse as parseNQuads } from '../nquads/mod.ts' +import { parse as parseTurtle } from './mod.ts' + +const COUNT = 10_000 +const turtle = [ + 'PREFIX : ', + ...Array.from({ length: COUNT }, (_, index) => `:s${index} :p "value-${index}" .`), +].join('\n') +const nquads = Array.from( + { length: COUNT }, + (_, index) => ` "value-${index}" .`, +).join('\n') + +const turtleExpected = await read(parseTurtle(turtle)) +const nquadsExpected = await read(parseNQuads(nquads)) +if (turtleExpected.length !== COUNT || datasetKey(turtleExpected) !== datasetKey(nquadsExpected)) { + throw new Error('Turtle benchmark semantic oracle does not match equivalent N-Quads data.') +} + +/** Fully consumes a parser result for a semantic count/digest oracle. */ +async function read(source: AsyncIterable): Promise { + const result: T[] = [] + for await (const value of source) result.push(value) + return result +} + +group('RDF text parse representation cost: 10k equivalent quads', () => { + bench('N-Quads explicit baseline', async () => { + const values = await read(parseNQuads(nquads)) + do_not_optimize(values.length) + }).gc('inner') + + bench('Turtle compact parser', async () => { + const values = await read(parseTurtle(turtle)) + do_not_optimize(values.length) + }).gc('inner') +}) + +await run() diff --git a/packages/rdf/turtle/parse_test.ts b/packages/rdf/turtle/parse_test.ts new file mode 100644 index 0000000..3b475fe --- /dev/null +++ b/packages/rdf/turtle/parse_test.ts @@ -0,0 +1,32 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { RDF } from '../mod.ts' +import { events, parse } from './mod.ts' + +async function collect(source: AsyncIterable): Promise { + const values: T[] = [] + for await (const value of source) values.push(value) + return values +} + +describe('@okikio/rdf/turtle', () => { + it('expands RDF 1.2 annotations without constructing a syntax tree', async () => { + const source = 'PREFIX : \n:a :p :b ~ :r {| :source :c |} .' + const quads = await collect(parse(source)) + expect(quads).toHaveLength(3) + expect(quads.some((value) => value.predicate.value === RDF.reifies)).toBe(true) + }) + + it('accepts RDF 1.2 double literals whose decimal point is followed by an exponent', async () => { + const quads = await collect(parse('@prefix : . :s :p 1.e3 .')) + expect(quads).toHaveLength(1) + expect(quads[0]?.object.termType).toBe('Literal') + expect(quads[0]?.object.value).toBe('1.e3') + }) + + it('buffers a malformed statement before tolerant diagnostics are released', async () => { + const values = await collect(events('PREFIX : \n:a :p [ :q :r ; BROKEN ] .\n:b :p :c .', { tolerant: true })) + expect(values.filter((value) => value.kind === 'quad')).toHaveLength(1) + expect(values.filter((value) => value.kind === 'diagnostic')).toHaveLength(1) + }) +}) diff --git a/packages/rdf/write.ts b/packages/rdf/write.ts new file mode 100644 index 0000000..86ae913 --- /dev/null +++ b/packages/rdf/write.ts @@ -0,0 +1,48 @@ +/** Shared RDF 1.2 N-Triples/N-Quads term serialization. @module */ + +import type { Literal, Quad, Term } from './term.ts' +import { XSD } from './term.ts' + +/** Serializes one quad in line RDF syntax. */ +export function writeQuad(quad: Quad, includeGraph: boolean): string { + const values = [writeTerm(quad.subject), writeTerm(quad.predicate), writeTerm(quad.object)] + if (includeGraph && quad.graph.termType !== 'DefaultGraph') values.push(writeTerm(quad.graph)) + return `${values.join(' ')} .` +} + +/** Serializes one RDF term in RDF 1.2 line syntax. */ +export function writeTerm(term: Term): string { + switch (term.termType) { + case 'NamedNode': return `<${escapeIri(term.value)}>` + case 'BlankNode': return `_:${term.value}` + case 'DefaultGraph': return '' + case 'Variable': return `?${term.value}` + case 'Literal': return writeLiteral(term as Literal) + case 'Quad': { + const quad = term as Quad + if (quad.graph.termType !== 'DefaultGraph') throw new TypeError('Embedded RDF triple term must use the default graph.') + return `<<( ${writeTerm(quad.subject)} ${writeTerm(quad.predicate)} ${writeTerm(quad.object)} )>>` + } + } +} + +/** Write literal deterministically to the caller-owned output. */ +function writeLiteral(literal: Literal): string { + const lexical = `"${escapeString(literal.value)}"` + if (literal.language) return `${lexical}@${literal.language}${literal.direction ? `--${literal.direction}` : ''}` + if (literal.datatype.value === XSD.string) return lexical + return `${lexical}^^${writeTerm(literal.datatype)}` +} + +/** Escapes the control characters required by N-Triples/N-Quads string literal syntax. */ +function escapeString(value: string): string { + return value.replace(/\\/g, '\\\\').replace(/"/g, '\\"').replace(/\t/g, '\\t').replace(/\n/g, '\\n').replace(/\r/g, '\\r') +} + +/** Escapes characters forbidden directly inside N-Triples/N-Quads IRI references. */ +function escapeIri(value: string): string { + return value.replace(/[<>"{}|^`\\\u0000-\u0020]/g, (char) => { + const point = char.codePointAt(0)! + return point <= 0xffff ? `\\u${point.toString(16).padStart(4, '0').toUpperCase()}` : `\\U${point.toString(16).padStart(8, '0').toUpperCase()}` + }) +} diff --git a/packages/rdf/write_test.ts b/packages/rdf/write_test.ts new file mode 100644 index 0000000..321d7c2 --- /dev/null +++ b/packages/rdf/write_test.ts @@ -0,0 +1,22 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { literal, namedNode, quad, triple } from './mod.ts' +import { writeQuad, writeTerm } from './write.ts' + +describe('@okikio/rdf line serializer primitives', () => { + it('escapes literals and IRIs deterministically', () => { + expect(writeTerm(literal('a\n"b"'))).toBe('"a\\n\\"b\\""') + expect(writeTerm(namedNode('urn:a b'))).toBe('') + }) + + it('serializes directional literals and RDF 1.2 triple terms', () => { + expect(writeTerm(literal('bonjour', { language: 'fr', direction: 'ltr' }))).toBe('"bonjour"@fr--ltr') + expect(writeTerm(triple(namedNode('urn:s'), namedNode('urn:p'), literal('o')))).toBe('<<( "o" )>>') + }) + + it('includes named graphs only for N-Quads output', () => { + const value = quad(namedNode('urn:s'), namedNode('urn:p'), literal('o'), namedNode('urn:g')) + expect(writeQuad(value, false)).toBe(' "o" .') + expect(writeQuad(value, true)).toBe(' "o" .') + }) +}) diff --git a/packages/rdf/xml/mod.ts b/packages/rdf/xml/mod.ts new file mode 100644 index 0000000..cf60c44 --- /dev/null +++ b/packages/rdf/xml/mod.ts @@ -0,0 +1,87 @@ +/** Streaming RDF 1.1/1.2 XML parsing behind Web-oriented source contracts. @module */ + +import { blankNode, defaultGraph, literal, namedNode, quad, variable } from '../factory.ts' +import type { Graph, NamedNode, Quad } from '../term.ts' +import { throwIfAborted, type TextSource } from '../text.ts' +import { parseTransform } from '../transform.ts' +import type { ParserConstructorType } from './types.ts' + +export type { ParserConstructorType, ParserType } from './types.ts' + +/** Options for RDF/XML parsing. */ +export interface ParseOptionsType { + /** Initial base IRI used before xml:base declarations are encountered. */ + readonly base?: string + /** Graph assigned to parsed RDF/XML triples. RDF/XML itself serializes one graph. */ + readonly graph?: Graph + /** Reject malformed XML instead of accepting the parser's lenient compatibility mode. Defaults to true. */ + readonly strict?: boolean + /** Include line/column information in parser errors where supported. */ + readonly trackPosition?: boolean + /** Permit repeated rdf:ID values. Defaults to false. */ + readonly allowDuplicateRdfIds?: boolean + /** Validate RDF IRIs. Defaults to true. */ + readonly validateIri?: boolean + /** Parse unsupported rdf:version values instead of rejecting them. Defaults to false. */ + readonly parseUnsupportedVersions?: boolean + /** Version provided by an application/rdf+xml media-type parameter. */ + readonly version?: '1.1' | '1.2-basic' | '1.2' + /** External parser injection used by tests or alternate conforming implementations. */ + readonly parser?: ParserConstructorType + readonly signal?: AbortSignal +} + +/** + * Parses RDF/XML incrementally into native `@okikio/rdf` quads. + * + * The public source contract stays on strings, byte chunks, async iterables, + * and Web `ReadableStream`s. The external parser's Node-style Transform is an + * implementation detail and is destroyed when the caller stops early. + */ +export async function* parse(source: TextSource, options: ParseOptionsType = {}): AsyncGenerator { + throwIfAborted(options.signal) + const Parser = options.parser ?? await defaultParser() + const parser = new Parser(parserOptions(options)) + yield* parseTransform(parser, source, { label: 'RDF/XML parser', ...(options.signal ? { signal: options.signal } : {}) }) +} + +/** Converts project RDF factory calls into the RDF/JS DataFactory shape expected by the parser. */ +const dataFactory = { + namedNode, + blankNode, + defaultGraph, + variable, + quad, + /** Adapts the RDF/JS literal factory signature while preserving RDF 1.2 directional language literals when an upstream parser supplies direction. */ + literal(value: string, languageOrDatatype?: string | NamedNode, direction?: 'ltr' | 'rtl'): ReturnType { + if (direction !== undefined) { + if (typeof languageOrDatatype !== 'string') throw new TypeError('Directional RDF/XML literal requires a language tag.') + return literal(value, { language: languageOrDatatype, direction }) + } + return literal(value, languageOrDatatype) + }, +} as const + +/** Builds the external parser options without serializing absent optional fields as undefined. */ +function parserOptions(options: ParseOptionsType): Readonly> { + return { + dataFactory, + strict: options.strict ?? true, + trackPosition: options.trackPosition ?? true, + allowDuplicateRdfIds: options.allowDuplicateRdfIds ?? false, + validateUri: options.validateIri ?? true, + parseUnsupportedVersions: options.parseUnsupportedVersions ?? false, + ...(options.base === undefined ? {} : { baseIRI: options.base }), + ...(options.graph === undefined ? {} : { defaultGraph: options.graph }), + ...(options.version === undefined ? {} : { version: options.version }), + } +} + +/** Lazily resolved RDF/XML parser constructor so importing the subpath does not initialize the optional processor. */ +let parserPromise: Promise | undefined + +/** Lazily imports the RDF/XML implementation only when the subpath is used. */ +async function defaultParser(): Promise { + parserPromise ??= import('rdfxml-streaming-parser').then((module) => module.RdfXmlParser as unknown as ParserConstructorType) + return await parserPromise +} diff --git a/packages/rdf/xml/mod_test.ts b/packages/rdf/xml/mod_test.ts new file mode 100644 index 0000000..9979ddf --- /dev/null +++ b/packages/rdf/xml/mod_test.ts @@ -0,0 +1,138 @@ +import { describe, it } from 'node:test' +import { expect } from '@std/expect' +import { literal, namedNode, quad, type Quad } from '../mod.ts' +import { parse, type ParserType } from './mod.ts' + +const fixture = quad(namedNode('https://example.test/s'), namedNode('https://example.test/p'), literal('value')) +type EventType = 'drain' | 'close' | 'error' +type ListenerType = (...args: unknown[]) => void + +/** Minimal event surface for testing the Node Transform bridge without owning a host emitter dependency. */ +class Emitter { + private readonly listeners = new Map>() + + once(event: EventType, listener: ListenerType): this { + const wrapper: ListenerType = (...args) => { this.off(event, wrapper); listener(...args) } + const values = this.listeners.get(event) ?? new Set() + values.add(wrapper) + this.listeners.set(event, values) + return this + } + + off(event: EventType, listener: ListenerType): this { + this.listeners.get(event)?.delete(listener) + return this + } + + protected emit(event: EventType, ...args: unknown[]): void { + for (const listener of [...(this.listeners.get(event) ?? [])]) listener(...args) + } +} + +/** Parser fixture that records constructor options and exercises writable backpressure. */ +class TestParser extends Emitter implements ParserType { + static options: Readonly> | undefined + private ended = false + private resolveEnd: (() => void) | undefined + private first = true + + constructor(options: Readonly>) { + super() + TestParser.options = options + } + + write(_value: string | Uint8Array): boolean { + if (this.first) { + this.first = false + queueMicrotask(() => this.emit('drain')) + return false + } + return true + } + + end(): void { this.ended = true; this.resolveEnd?.() } + destroy(error?: Error): void { + if (error) this.emit('error', error) + this.ended = true + this.resolveEnd?.() + this.emit('close') + } + + async *[Symbol.asyncIterator](): AsyncIterator { + if (!this.ended) await new Promise((resolve) => { this.resolveEnd = resolve }) + yield fixture + } +} + +/** Parser fixture that leaves output open so early iterator return must destroy it. */ +class EarlyParser extends Emitter implements ParserType { + static destroyed = false + private available = false + private resolveValue: (() => void) | undefined + + constructor(_options: Readonly>) { + super() + EarlyParser.destroyed = false + } + + write(_value: string | Uint8Array): boolean { + this.available = true + this.resolveValue?.() + return true + } + + end(): void {} + destroy(error?: Error): void { + EarlyParser.destroyed = true + if (error) this.emit('error', error) + this.emit('close') + } + + async *[Symbol.asyncIterator](): AsyncIterator { + if (!this.available) await new Promise((resolve) => { this.resolveValue = resolve }) + yield fixture + await new Promise(() => {}) + } +} + +describe('@okikio/rdf/xml', () => { + it('forwards RDF/XML options and supplies the native RDF 1.2 factory', async () => { + const values: Quad[] = [] + for await (const value of parse('', { + parser: TestParser, + base: 'https://example.test/base/', + version: '1.2', + })) values.push(value) + + expect(values).toHaveLength(1) + expect(values[0]?.equals(fixture)).toBe(true) + expect(TestParser.options?.strict).toBe(true) + expect(TestParser.options?.trackPosition).toBe(true) + expect(TestParser.options?.baseIRI).toBe('https://example.test/base/') + expect(TestParser.options?.version).toBe('1.2') + const factory = TestParser.options?.dataFactory as { + literal(value: string, language?: string, direction?: 'ltr' | 'rtl'): ReturnType + } + const directional = factory.literal('مرحبا', 'ar', 'rtl') + expect(directional.language).toBe('ar') + expect(directional.direction).toBe('rtl') + }) + + it('honors writable backpressure before ending the external parser', async () => { + const values: Quad[] = [] + for await (const value of parse([''], { parser: TestParser })) values.push(value) + expect(values).toHaveLength(1) + }) + + it('cancels a pending Web Stream read and destroys the parser on early return', async () => { + let cancelled = false + const stream = new ReadableStream({ + start(controller) { controller.enqueue(new TextEncoder().encode('')) }, + cancel() { cancelled = true }, + }) + + for await (const _value of parse(stream, { parser: EarlyParser })) break + expect(EarlyParser.destroyed).toBe(true) + expect(cancelled).toBe(true) + }) +}) diff --git a/packages/rdf/xml/types.ts b/packages/rdf/xml/types.ts new file mode 100644 index 0000000..fea5137 --- /dev/null +++ b/packages/rdf/xml/types.ts @@ -0,0 +1,11 @@ +/** Structural RDF/XML parser contracts used to hide the external stream implementation. @module */ + +import type { TransformParserType } from '../transform.ts' + +/** RDF/XML parser stream shape required by the adapter. */ +export type ParserType = TransformParserType + +/** Constructor contract for an RDF/XML parser implementation. */ +export interface ParserConstructorType { + new (options: Readonly>): ParserType +}