From b5d33a4ce15412b8868502b2da1cbd5a4fac8bfc Mon Sep 17 00:00:00 2001 From: Okiki Ojo Date: Tue, 18 Aug 2026 11:20:47 -0400 Subject: [PATCH] feat(rdf): add native standards subpaths and loss-preserving inspectors --- packages/rdf/README.md | 27 +- packages/rdf/canon/mod.ts | 498 +++++++--- packages/rdf/canon/mod_test.ts | 82 +- packages/rdf/canon/types.ts | 32 +- packages/rdf/compact.ts | 878 ++++++++++++------ packages/rdf/dataset.ts | 116 ++- packages/rdf/dataset_test.ts | 4 +- packages/rdf/deno.json | 22 +- packages/rdf/factory.ts | 45 +- packages/rdf/jsonld/compact.ts | 346 +++++++ packages/rdf/jsonld/context.ts | 485 ++++++++++ packages/rdf/jsonld/expand.ts | 451 +++++++++ packages/rdf/jsonld/frame.ts | 125 +++ packages/rdf/jsonld/loader.ts | 158 +++- packages/rdf/jsonld/mod.ts | 432 ++++++--- packages/rdf/jsonld/mod_test.ts | 184 +++- packages/rdf/jsonld/node.ts | 182 ++++ packages/rdf/jsonld/rdf.ts | 487 ++++++++++ packages/rdf/jsonld/types.ts | 43 +- packages/rdf/line.ts | 221 ++++- packages/rdf/markup.ts | 604 ++++++++++++ packages/rdf/microdata/mod.ts | 441 +++++++-- packages/rdf/microdata/mod_test.ts | 85 +- packages/rdf/microdata/types.ts | 11 - packages/rdf/mod.ts | 31 +- packages/rdf/namespace.ts | 2 + packages/rdf/nquads/mod.ts | 33 +- packages/rdf/nquads/parse_test.ts | 3 +- packages/rdf/ntriples/mod.ts | 37 +- packages/rdf/ntriples/parse_test.ts | 2 +- packages/rdf/ontology/index.ts | 2 + packages/rdf/ontology/{read.ts => inspect.ts} | 139 ++- .../{read_test.ts => inspect_test.ts} | 8 +- packages/rdf/ontology/mod.ts | 8 +- packages/rdf/ontology/model.ts | 45 +- packages/rdf/package.json | 16 +- packages/rdf/rdfa/mod.ts | 576 ++++++++++-- packages/rdf/rdfa/mod_test.ts | 86 +- packages/rdf/rdfa/types.ts | 11 - packages/rdf/shape/index.ts | 19 +- packages/rdf/shape/{read.ts => inspect.ts} | 760 +++++++++++---- .../shape/{read_test.ts => inspect_test.ts} | 47 +- packages/rdf/shape/list.ts | 40 +- packages/rdf/shape/mod.ts | 12 +- packages/rdf/shape/model.ts | 454 +++++++-- packages/rdf/shape/path.ts | 111 ++- packages/rdf/shape/value.ts | 24 +- packages/rdf/source.ts | 6 +- packages/rdf/stream.ts | 150 +++ packages/rdf/term.ts | 100 +- packages/rdf/term_test.ts | 4 +- packages/rdf/text.ts | 14 +- packages/rdf/text_test.ts | 9 +- packages/rdf/transform.ts | 95 -- packages/rdf/trig/mod.ts | 28 +- packages/rdf/trig/parse_test.ts | 9 +- packages/rdf/turtle/mod.ts | 32 +- packages/rdf/turtle/parse_test.ts | 6 +- packages/rdf/write.ts | 36 +- packages/rdf/write_test.ts | 8 +- packages/rdf/xml/mod.ts | 455 +++++++-- packages/rdf/xml/mod_test.ts | 175 ++-- packages/rdf/xml/types.ts | 11 - 63 files changed, 7781 insertions(+), 1782 deletions(-) create mode 100644 packages/rdf/jsonld/compact.ts create mode 100644 packages/rdf/jsonld/context.ts create mode 100644 packages/rdf/jsonld/expand.ts create mode 100644 packages/rdf/jsonld/frame.ts create mode 100644 packages/rdf/jsonld/node.ts create mode 100644 packages/rdf/jsonld/rdf.ts create mode 100644 packages/rdf/markup.ts delete mode 100644 packages/rdf/microdata/types.ts rename packages/rdf/ontology/{read.ts => inspect.ts} (77%) rename packages/rdf/ontology/{read_test.ts => inspect_test.ts} (85%) delete mode 100644 packages/rdf/rdfa/types.ts rename packages/rdf/shape/{read.ts => inspect.ts} (61%) rename packages/rdf/shape/{read_test.ts => inspect_test.ts} (73%) create mode 100644 packages/rdf/stream.ts delete mode 100644 packages/rdf/transform.ts delete mode 100644 packages/rdf/xml/types.ts diff --git a/packages/rdf/README.md b/packages/rdf/README.md index f38d763..b3893f1 100644 --- a/packages/rdf/README.md +++ b/packages/rdf/README.md @@ -27,33 +27,28 @@ AbortSignal ## Formats and semantics -Use explicit subpaths: +Project-owned syntax and semantic capabilities use explicit `@okikio/rdf` subpaths: ```ts import * as nquads from '@okikio/rdf/nquads' import * as turtle from '@okikio/rdf/turtle' -import * as jsonld from '@okikio/rdf/jsonld' import * as ontology from '@okikio/rdf/ontology' import * as shape from '@okikio/rdf/shape' ``` -Available public subpaths include: +The package exports: ```text -ntriples -nquads -turtle -trig -jsonld -xml -rdfa -microdata -canon -ontology -shape +@okikio/rdf/ntriples +@okikio/rdf/nquads +@okikio/rdf/turtle +@okikio/rdf/trig +@okikio/rdf/ontology +@okikio/rdf/shape +@okikio/rdf/stream ``` -The root module does not import or initialize the focused third-party processors used by JSON-LD, RDFC-1.0, RDF/XML, RDFa, or Microdata. The npm package manifest is package-scoped, however, so installing `@okikio/rdf` currently installs those dependencies. +`@okikio/rdf` has no third-party runtime implementation dependency. JSON-LD, RDFC-1.0, RDF/XML, RDFa, and Microdata are implemented natively at `@okikio/rdf/jsonld`, `@okikio/rdf/canon`, `@okikio/rdf/xml`, `@okikio/rdf/rdfa`, and `@okikio/rdf/microdata`. Tests, conformance suites, and benchmarks can use external implementations only as independent correctness and performance references. ## Parser lifecycle @@ -63,6 +58,6 @@ Project-owned streaming parsers use bounded source windows and cancel pending We `@okikio/rdf/ontology` interprets generic named RDFS/OWL relationships while retaining unsupported assertions. -`@okikio/rdf/shape` reads loss-preserving SHACL shape structure. Ontology domain/range semantics are not treated as closed-world JSON requiredness. +`@okikio/rdf/shape` inspects loss-preserving SHACL shape structure. Ontology domain/range semantics are not treated as closed-world JSON requiredness. See the repository architecture and testing guides for the current standards/version posture and release gates. diff --git a/packages/rdf/canon/mod.ts b/packages/rdf/canon/mod.ts index 51a9edf..0a986d3 100644 --- a/packages/rdf/canon/mod.ts +++ b/packages/rdf/canon/mod.ts @@ -1,174 +1,406 @@ -/** RDFC-1.0 RDF dataset canonicalization with explicit complexity controls. @module */ - -import { parse as parseNQuads, write as writeNQuads } from '../nquads/mod.ts' -import type { Literal, ObjectTerm, Quad } from '../term.ts' -import type { CanonizerOptionsType, CanonizerType } from './types.ts' - -export type { CanonizerOptionsType, CanonizerType } from './types.ts' - -/** Default max quads used when the caller does not provide an override. */ -const DEFAULT_MAX_QUADS = 1_000_000 -/** Default max work factor used when the caller does not provide an override. */ -const DEFAULT_MAX_WORK_FACTOR = 1 - +/** Native RDFC-1.0 RDF dataset canonicalization with explicit complexity controls. @module */ +import { blankNode, quad } from '../factory.ts' +import { parse as parseNQuads } from '../nquads/mod.ts' +import type { GraphTermType, Literal, ObjectTermType, Quad } from '../term.ts' +import { writeQuad } from '../write.ts' +import type { CryptoDigestType, DegreeResultType, DigestType } from './types.ts' +export type { CryptoDigestType, DegreeResultType, DigestType } from './types.ts' +/** Default maximum dataset size admitted by canonicalization when the caller does not set `maxQuads`. */ +const DEFAULT_MAX_QUADS = 1_000_000, DEFAULT_MAX_WORK_FACTOR = 8, MIN_WORK = 64 /** RDFC-1.0 canonicalization options. */ export interface OptionsType { - /** Optional implementation injection for tests or alternate conforming RDFC-1.0 engines. */ - readonly canonizer?: CanonizerType - /** Hash algorithm used internally by RDFC-1.0. */ - readonly messageDigestAlgorithm?: 'sha256' | 'sha384' | 'sha512' - /** Complexity limit passed to the deep blank-node comparison algorithm. Defaults to 1, or O(n). */ - readonly maxWorkFactor?: number - /** Exact deep-iteration limit. When supplied, this overrides `maxWorkFactor`. */ - readonly maxDeepIterations?: number - /** Maximum number of input quads materialized for one canonicalization. */ - readonly maxQuads?: number - /** Cooperative cancellation checked by this facade and the canonicalizer. */ - readonly signal?: AbortSignal -} - -/** Options for hashing the resulting canonical N-Quads document. */ -export interface HashOptionsType extends OptionsType { - /** Digest applied to the final canonical N-Quads bytes. This is separate from RDFC's internal hash. */ - readonly digest?: 'SHA-256' | 'SHA-384' | 'SHA-512' -} - + /** Internal RDFC digest algorithm. */ readonly messageDigestAlgorithm?: DigestType + /** Multiplier used to derive the N-degree work limit. */ readonly maxWorkFactor?: number + /** Exact recursive/permutation work limit. */ readonly maxDeepIterations?: number + /** Maximum materialized input quads. */ readonly maxQuads?: number + /** Caller-owned output map from source blank ids to canonical ids. */ readonly canonicalIdMap?: + Map + /** Caller-owned cancellation signal. */ readonly signal?: AbortSignal +} +/** Options for hashing the final canonical bytes. */ export interface HashOptionsType + extends OptionsType { + /** Web Crypto digest used to hash the final canonical N-Quads bytes. */ + readonly digest?: CryptoDigestType +} +/** Quad position occupied by a related blank node during first-degree and N-degree hashing. */ +type PositionType = 's' | 'o' | 'g' +/** Deterministic blank-node identifier issuer. */ +class Issuer { + /** Canonical identifier prefix prepended to identifiers allocated by this issuer. */ + readonly prefix: string + /** Blank-node identifier map owned by this issuer. */ + readonly ids: Map + /** Next numeric suffix allocated by this issuer. */ + #next: number + /** Creates one Issuer instance with operation-local state. */ + constructor(prefix: string, ids: ReadonlyMap = new Map(), next = 0) { + this.prefix = prefix + this.ids = new Map(ids) + this.#next = next + } + /** Returns the stable identifier for a blank node, allocating the next issuer identifier when needed. */ + issue(id: string) { + const found = this.ids.get(id) + if (found !== undefined) return found + const value = `${this.prefix}${this.#next++}` + this.ids.set(id, value) + return value + } + /** Returns the previously issued identifier without changing issuer ordering state. */ + get(id: string) { + return this.ids.get(id) + } + /** Tests whether this issuer already assigned an identifier to the blank node. */ + has(id: string) { + return this.ids.has(id) + } + /** Clones the issuer so recursive canonicalization can explore a candidate path without mutating sibling candidates. */ + copy() { + return new Issuer(this.prefix, this.ids, this.#next) + } +} +/** Mutable state shared by the canonicalization algorithms. */ +interface StateType { + /** Quads indexed by every blank-node identifier they mention. */ + readonly quads: Map + /** First-degree digest cache. */ readonly first: Map + /** Canonical c14n issuer. */ readonly canonical: Issuer + /** Internal Web Crypto digest. */ readonly digest: CryptoDigestType + /** Maximum adversarial work. */ readonly maxWork: number + /** Caller cancellation. */ readonly signal?: AbortSignal + /** Work already consumed. */ work: number +} /** * Produces the canonical N-Quads representation defined by RDFC-1.0. * - * RDFC-1.0 is defined over the RDF 1.1 dataset model. RDF 1.2 triple terms and - * directional language-tagged strings are therefore rejected instead of being - * silently lowered to a representation whose canonicalization is unspecified. + * The W3C algorithm is implemented directly. No external canonicalizer is used. */ export async function canonicalize( source: Iterable | AsyncIterable, options: OptionsType = {}, ): Promise { abort(options.signal) - const quads = await collect(source, options.maxQuads ?? DEFAULT_MAX_QUADS, options.signal) - for (const value of quads) validate(value) - - const canonizer = options.canonizer ?? await defaultCanonizer() - const settings = settingsFor(options) - const output = await canonizer.canonize(writeNQuads(quads), settings) - abort(options.signal) - return output + const quads = await collect( + source, + positive(options.maxQuads ?? DEFAULT_MAX_QUADS, 'maxQuads'), + options.signal, + ) + quads.forEach(validate) + const byNode = index(quads) + const factor = nonNegative(options.maxWorkFactor ?? DEFAULT_MAX_WORK_FACTOR, 'maxWorkFactor') + const maxWork = options.maxDeepIterations === undefined + ? (factor === Infinity ? Infinity : Math.max(MIN_WORK, factor * Math.max(1, byNode.size))) + : nonNegative(options.maxDeepIterations, 'maxDeepIterations') + const state: StateType = { + quads: byNode, + first: new Map(), + canonical: new Issuer('c14n'), + digest: cryptoDigest(options.messageDigestAlgorithm ?? 'sha256'), + maxWork, + ...(options.signal ? { signal: options.signal } : {}), + work: 0, + } + const byHash = new Map() + for (const id of byNode.keys()) add(byHash, await first(id, state), id) + for (const h of ordered(byHash.keys())) { + const ids = byHash.get(h)! + if (ids.length === 1) { + state.canonical.issue(ids[0]!) + byHash.delete(h) + } + } + for (const h of ordered(byHash.keys())) { + const results: DegreeResultType[] = [] + for (const id of byHash.get(h)!) { + if (state.canonical.has(id)) continue + const issuer = new Issuer('b') + issuer.issue(id) + results.push(await degree(id, issuer, state)) + } + results.sort((a, b) => compare(a.hash, b.hash)) + for (const result of results) { + for (const id of result.issuer.ids.keys()) state.canonical.issue(id) + } + } + options.canonicalIdMap?.clear() + for (const [a, b] of state.canonical.ids) options.canonicalIdMap?.set(a, b) + const lines = quads.map((v) => canonicalQuad(v, state.canonical)).sort(compare) + return lines.length ? `${lines.join('\n')}\n` : '' } - -/** Canonicalizes a dataset and parses the canonical N-Quads back into native RDF terms. */ -export async function canonicalizeQuads( +/** Parses canonical bytes back into native quads. */ export async function canonicalizeQuads( source: Iterable | AsyncIterable, options: OptionsType = {}, ): Promise { - const text = await canonicalize(source, options) - const quads: Quad[] = [] - const parseOptions = options.signal ? { signal: options.signal } : {} - for await (const value of parseNQuads(text, parseOptions)) quads.push(value) - return quads -} - -/** Hashes the canonical N-Quads bytes with one explicit Web Crypto digest. */ -export async function hash( + const values: Quad[] = [] + for await ( + const value of parseNQuads( + await canonicalize(source, options), + options.signal ? { signal: options.signal } : {}, + ) + ) values.push(value) + return values +} +/** Hashes canonical N-Quads bytes. */ export async function hash( source: Iterable | AsyncIterable, options: HashOptionsType = {}, ): Promise { - const text = await canonicalize(source, options) - abort(options.signal) - const bytes = new TextEncoder().encode(text) - const digest = await crypto.subtle.digest(options.digest ?? 'SHA-256', bytes) - abort(options.signal) - return hex(new Uint8Array(digest)) + return digest(await canonicalize(source, options), options.digest ?? 'SHA-256') } - -/** Returns true when two datasets canonicalize to the same RDFC-1.0 representation. */ -export async function isomorphic( +/** Tests dataset isomorphism through canonical equality. */ export async function isomorphic( left: Iterable | AsyncIterable, right: Iterable | AsyncIterable, options: OptionsType = {}, ): Promise { - const leftValue = await canonicalize(left, options) - const rightValue = await canonicalize(right, options) - return leftValue === rightValue -} - -/** Creates the exact option object passed to the external implementation. */ -function settingsFor(options: OptionsType): CanonizerOptionsType { - const maxWorkFactor = nonNegativeFiniteOrInfinity(options.maxWorkFactor ?? DEFAULT_MAX_WORK_FACTOR, 'maxWorkFactor') - const settings: CanonizerOptionsType = { - algorithm: 'RDFC-1.0', - inputFormat: 'application/n-quads', - format: 'application/n-quads', - messageDigestAlgorithm: options.messageDigestAlgorithm ?? 'sha256', - maxWorkFactor, - rejectURDNA2015: true, - ...(options.maxDeepIterations === undefined - ? {} - : { maxDeepIterations: nonNegativeFiniteOrInfinity(options.maxDeepIterations, 'maxDeepIterations') }), - ...(options.signal ? { signal: options.signal } : {}), + return await canonicalize(left, options) === await canonicalize(right, options) +} +/** Computes and caches the first-degree hash for one blank node. */ async function first( + id: string, + state: StateType, +) { + const cached = state.first.get(id) + if (cached !== undefined) return cached + const lines = (state.quads.get(id) ?? []).map((q) => firstQuad(q, id)).sort(compare) + const value = await digest(lines.map((x) => `${x}\n`).join(''), state.digest) + state.first.set(id, value) + return value +} +/** Serializes one first-degree quad with _:a and _:z placeholders. */ function firstQuad( + value: Quad, + id: string, +) { + const replace = (term: Quad['subject'] | Quad['object'] | Quad['graph']) => + term.termType === 'BlankNode' ? blankNode(term.value === id ? 'a' : 'z') : term + return writeQuad( + quad( + replace(value.subject) as Quad['subject'], + value.predicate, + replace(value.object) as Quad['object'], + replace(value.graph) as GraphTermType, + ), + true, + ) +} +/** Implements recursive Hash N-Degree Quads. */ async function degree( + id: string, + issuer: Issuer, + state: StateType, +): Promise> { + work(state) + const related = new Map() + for (const q of state.quads.get(id) ?? []) { + for (const [term, pos] of components(q)) { + if (term.termType === 'BlankNode' && term.value !== id) { + add(related, await relatedHash(term.value, q, issuer, pos, state), term.value) + } + } + } + let data = '', selected = issuer + for (const h of ordered(related.keys())) { + data += h + let chosen = '', chosenIssuer: Issuer | undefined + for (const perm of permutations(related.get(h)!)) { + work(state) + let current = selected.copy(), path = '' + const recurse: string[] = [] + let rejected = false + for (const rid of perm) { + const canonical = state.canonical.get(rid) + if (canonical !== undefined) path += `_:${canonical}` + else { + if (!current.has(rid)) recurse.push(rid) + path += `_:${current.issue(rid)}` + } + if (worse(path, chosen)) { + rejected = true + break + } + } + if (rejected) continue + for (const rid of recurse) { + const result = await degree(rid, current, state) + path += `_:${current.issue(rid)}<${result.hash}>` + current = result.issuer + if (worse(path, chosen)) { + rejected = true + break + } + } + if (!rejected && (chosen === '' || compare(path, chosen) < 0)) { + chosen = path + chosenIssuer = current + } + } + if (!chosenIssuer) throw new Error('RDFC-1.0 could not choose a deterministic blank-node path.') + data += chosen + selected = chosenIssuer + } + return { hash: await digest(data, state.digest), issuer: selected } +} +/** Computes one related blank-node hash. */ async function relatedHash( + id: string, + q: Quad, + issuer: Issuer, + pos: PositionType, + state: StateType, +) { + let input = pos + if (pos !== 'g') input += `<${q.predicate.value}>` + const known = state.canonical.get(id) ?? issuer.get(id) + input += known === undefined ? await first(id, state) : `_:${known}` + return digest(input, state.digest) +} +/** Returns canonicalization-relevant quad components. */ function components( + q: Quad, +): ReadonlyArray { + return [[q.subject, 's'], [q.object, 'o'], [q.graph, 'g']] +} +/** Indexes quads by unique blank nodes they mention. */ function index(quads: readonly Quad[]) { + const result = new Map() + for (const q of quads) { + const seen = new Set() + for (const [term] of components(q)) { + if (term.termType === 'BlankNode' && !seen.has(term.value)) { + seen.add(term.value) + add(result, term.value, q) + } + } + } + return result +} +/** Serializes one quad with canonical blank labels. */ function canonicalQuad( + value: Quad, + issuer: Issuer, +) { + const replace = (term: Quad['subject'] | Quad['object'] | Quad['graph']) => { + if (term.termType !== 'BlankNode') return term + const id = issuer.get(term.value) + if (id === undefined) { + throw new Error(`Missing canonical identifier for blank node '${term.value}'.`) + } + return blankNode(id) + } + return writeQuad( + quad( + replace(value.subject) as Quad['subject'], + value.predicate, + replace(value.object) as Quad['object'], + replace(value.graph) as GraphTermType, + ), + true, + ) +} +/** Lazily yields every positional permutation. */ function* permutations( + values: readonly string[], +): Generator { + const source = [...values], used = new Array(source.length).fill(false), current: string[] = [] + /** Recursively fills the next permutation slot. */ function* visit(): Generator { + if (current.length === source.length) { + yield [...current] + return + } + for (let i = 0; i < source.length; i++) { + if (used[i]) continue + used[i] = true + current.push(source[i]!) + yield* visit() + current.pop() + used[i] = false + } + } + yield* visit() +} +/** Rejects a candidate path once it cannot beat the selected path. */ function worse( + path: string, + chosen: string, +) { + return chosen !== '' && path.length >= chosen.length && compare(path, chosen) > 0 +} +/** Appends a value to a map-of-arrays. */ function add( + map: Map, + key: string, + value: T, +) { + const list = map.get(key) + if (list) list.push(value) + else map.set(key, [value]) +} +/** Sorts strings by Unicode code point. */ function ordered(values: Iterable) { + return [...values].sort(compare) +} +/** Unicode scalar-value comparator. */ function compare(a: string, b: string) { + if (a === b) return 0 + const ai = a[Symbol.iterator](), bi = b[Symbol.iterator]() + while (true) { + const av = ai.next(), bv = bi.next() + if (av.done) return bv.done ? 0 : -1 + if (bv.done) return 1 + const ac = av.value.codePointAt(0)!, bc = bv.value.codePointAt(0)! + if (ac !== bc) return ac < bc ? -1 : 1 } - return settings } - -/** Materializes one canonicalization input with an explicit cardinality limit. */ -async function collect( +/** Hashes UTF-8 text to lowercase hex. */ async function digest( + value: string, + algorithm: CryptoDigestType, +) { + const output = await crypto.subtle.digest(algorithm, new TextEncoder().encode(value)) + return [...new Uint8Array(output)].map((v) => v.toString(16).padStart(2, '0')).join('') +} +/** Maps public digest spelling to Web Crypto. */ function cryptoDigest( + value: DigestType, +): CryptoDigestType { + return value === 'sha256' ? 'SHA-256' : value === 'sha384' ? 'SHA-384' : 'SHA-512' +} +/** Materializes bounded input. */ async function collect( source: Iterable | AsyncIterable, - maxQuads: number, + max: number, signal?: AbortSignal, -): Promise { - if (!Number.isSafeInteger(maxQuads) || maxQuads <= 0) throw new RangeError('maxQuads must be a positive safe integer.') - const quads: Quad[] = [] - for await (const value of source) { +) { + const out: Quad[] = [] + for await (const q of source) { abort(signal) - if (quads.length >= maxQuads) throw new RangeError(`RDFC-1.0 input exceeds maxQuads (${maxQuads}).`) - quads.push(value) + if (out.length >= max) throw new RangeError(`RDFC-1.0 input exceeds maxQuads (${max}).`) + out.push(q) } - return quads + return out } - -/** Rejects RDF 1.2 terms that RDFC-1.0 does not currently define. */ -function validate(value: Quad): void { - validateObject(value.object) +/** Rejects RDF 1.2 forms outside RDFC-1.0. */ function validate(q: Quad) { + validateObject(q.object) } - -/** Validates one graph object for RDFC-1.0 compatibility. */ -function validateObject(value: ObjectTerm): void { +/** Validates one RDFC object term. */ function validateObject(value: ObjectTermType) { if (value.termType === 'Quad') { throw new TypeError('RDFC-1.0 does not define canonicalization for RDF 1.2 triple terms.') } if (value.termType === 'Literal') validateLiteral(value) } - -/** Rejects RDF 1.2 directional language-tagged strings from RDFC-1.0 input. */ -function validateLiteral(value: Literal): void { +/** Rejects RDF 1.2 directional language strings. */ function validateLiteral(value: Literal) { if (value.direction !== '') { - throw new TypeError('RDFC-1.0 does not define canonicalization for RDF 1.2 directional language-tagged strings.') - } -} - -/** Lazily resolved RDFC-1.0 processor shared across calls after the caller first requests canonicalization. */ -let canonizerPromise: Promise | undefined - -/** Lazily imports the canonicalizer only when this focused subpath performs work. */ -async function defaultCanonizer(): Promise { - canonizerPromise ??= import('rdf-canonize').then((module) => module as unknown as CanonizerType) - return await canonizerPromise -} - -/** Formats digest bytes as lowercase hexadecimal. */ -function hex(bytes: Uint8Array): string { - let result = '' - for (const value of bytes) result += value.toString(16).padStart(2, '0') - return result + throw new TypeError( + 'RDFC-1.0 does not define canonicalization for RDF 1.2 directional language-tagged strings.', + ) + } } - -/** Validates canonicalization work limits while allowing Infinity only where the upstream contract permits it. */ -function nonNegativeFiniteOrInfinity(value: number, name: string): number { - if (value === Infinity) return value - if (!Number.isSafeInteger(value) || value < 0) throw new RangeError(`${name} must be a non-negative safe integer or Infinity.`) - return value +/** Increments bounded recursive/permutation work. */ function work(state: StateType) { + abort(state.signal) + if (++state.work > state.maxWork) { + throw new RangeError(`RDFC-1.0 exceeded its N-degree work limit (${state.maxWork}).`) + } +} +/** Validates positive safe integers. */ function positive(v: number, n: string) { + if (!Number.isSafeInteger(v) || v <= 0) { + throw new RangeError(`${n} must be a positive safe integer.`) + } + return v +} +/** Validates nonnegative work limits, permitting Infinity. */ function nonNegative( + v: number, + n: string, +) { + if (v === Infinity) return v + if (!Number.isSafeInteger(v) || v < 0) { + throw new RangeError(`${n} must be a non-negative safe integer or Infinity.`) + } + return v } - -/** Throws the caller supplied abort reason when cancellation has been requested. */ -function abort(signal?: AbortSignal): void { +/** Throws caller cancellation. */ function abort(signal?: AbortSignal) { if (signal?.aborted) throw signal.reason ?? new DOMException('Aborted', 'AbortError') } diff --git a/packages/rdf/canon/mod_test.ts b/packages/rdf/canon/mod_test.ts index 2a324e8..37f61ca 100644 --- a/packages/rdf/canon/mod_test.ts +++ b/packages/rdf/canon/mod_test.ts @@ -1,38 +1,70 @@ import { describe, it } from 'node:test' import { expect } from '@std/expect' import { blankNode, literal, namedNode, quad } from '../mod.ts' -import { canonicalize, canonicalizeQuads, hash, isomorphic, type CanonizerType } from './mod.ts' +import { canonicalize, canonicalizeQuads, hash, isomorphic } from './mod.ts' -const canonical = ' "value" .\n' -const canonizer: CanonizerType = { - async canonize(_input, options) { - expect(options.algorithm).toBe('RDFC-1.0') - expect(options.rejectURDNA2015).toBe(true) - return canonical - }, -} +const ex = 'http://example.com/#' describe('@okikio/rdf/canon', () => { - it('owns the RDFC-1.0 option contract and native quad conversion', async () => { - const input = [quad(blankNode('input'), namedNode('https://example.test/p'), literal('value'))] - expect(await canonicalize(input, { canonizer })).toBe(canonical) - const values = await canonicalizeQuads(input, { canonizer }) - expect(values).toHaveLength(1) - expect(values[0]?.subject.value).toBe('https://example.test/s') + it('canonicalizes a dataset with uniquely hashed blank nodes', async () => { + const input = [ + quad(namedNode(`${ex}p`), namedNode(`${ex}q`), blankNode('e0')), + quad(namedNode(`${ex}p`), namedNode(`${ex}r`), blankNode('e1')), + quad(blankNode('e0'), namedNode(`${ex}s`), namedNode(`${ex}u`)), + quad(blankNode('e1'), namedNode(`${ex}t`), namedNode(`${ex}u`)), + ] + expect(await canonicalize(input)).toBe( + `<${ex}p> <${ex}q> _:c14n0 .\n` + + `<${ex}p> <${ex}r> _:c14n1 .\n` + + `_:c14n0 <${ex}s> <${ex}u> .\n` + + `_:c14n1 <${ex}t> <${ex}u> .\n`, + ) }) - it('rejects RDF 1.2 terms that RDFC-1.0 does not define', async () => { - const directional = quad( - namedNode('https://example.test/s'), - namedNode('https://example.test/p'), - literal('bonjour', { language: 'fr', direction: 'ltr' }), + it('uses N-degree hashing to order blank nodes with shared first-degree hashes', async () => { + const ids = new Map() + const input = [ + quad(namedNode(`${ex}p`), namedNode(`${ex}q`), blankNode('e1')), + quad(namedNode(`${ex}p`), namedNode(`${ex}q`), blankNode('e0')), + quad(blankNode('e2'), namedNode(`${ex}r`), blankNode('e3')), + quad(blankNode('e1'), namedNode(`${ex}p`), blankNode('e3')), + quad(blankNode('e0'), namedNode(`${ex}p`), blankNode('e2')), + ] + const value = await canonicalize(input, { canonicalIdMap: ids, maxWorkFactor: 64 }) + expect(value).toBe( + `<${ex}p> <${ex}q> _:c14n2 .\n` + + `<${ex}p> <${ex}q> _:c14n3 .\n` + + `_:c14n0 <${ex}r> _:c14n1 .\n` + + `_:c14n2 <${ex}p> _:c14n1 .\n` + + `_:c14n3 <${ex}p> _:c14n0 .\n`, ) - await expect(canonicalize([directional], { canonizer })).rejects.toThrow('directional') + expect(Object.fromEntries(ids)).toEqual({ e2: 'c14n0', e3: 'c14n1', e1: 'c14n2', e0: 'c14n3' }) + }) + + it('returns canonical native quads and stable digests', async () => { + const input = [quad(blankNode('x'), namedNode(`${ex}p`), literal('value'))] + const values = await canonicalizeQuads(input) + expect(values).toHaveLength(1) + expect(values[0]?.subject.value).toBe('c14n0') + expect(await hash(input)).toMatch(/^[0-9a-f]{64}$/u) + expect( + await isomorphic(input, [quad(blankNode('other'), namedNode(`${ex}p`), literal('value'))]), + ).toBe(true) }) - it('hashes canonical bytes and compares canonical representations', async () => { - const input = [quad(namedNode('https://example.test/a'), namedNode('https://example.test/p'), literal('x'))] - expect(/^[0-9a-f]{64}$/u.test(await hash(input, { canonizer }))).toBe(true) - expect(await isomorphic(input, input, { canonizer })).toBe(true) + it('rejects RDF 1.2 directional literals and bounded N-degree work exhaustion', async () => { + await expect(canonicalize([ + quad( + namedNode(`${ex}s`), + namedNode(`${ex}p`), + literal('bonjour', { language: 'fr', direction: 'ltr' }), + ), + ])).rejects.toThrow('directional') + + const input = [ + quad(blankNode('a'), namedNode(`${ex}p`), blankNode('b')), + quad(blankNode('b'), namedNode(`${ex}p`), blankNode('a')), + ] + await expect(canonicalize(input, { maxDeepIterations: 0 })).rejects.toThrow('work limit') }) }) diff --git a/packages/rdf/canon/types.ts b/packages/rdf/canon/types.ts index 4cf4b92..58647f6 100644 --- a/packages/rdf/canon/types.ts +++ b/packages/rdf/canon/types.ts @@ -1,18 +1,16 @@ -/** Structural contract implemented by an RDF dataset canonicalizer. @module */ - -/** Options forwarded to the RDFC-1.0 implementation. */ -export interface CanonizerOptionsType { - readonly algorithm: 'RDFC-1.0' - readonly inputFormat: 'application/n-quads' - readonly format: 'application/n-quads' - readonly messageDigestAlgorithm: 'sha256' | 'sha384' | 'sha512' - readonly maxWorkFactor: number - readonly maxDeepIterations?: number - readonly signal?: AbortSignal - readonly rejectURDNA2015: true -} - -/** Minimal external canonicalizer contract used by this subpath. */ -export interface CanonizerType { - canonize(input: string, options: CanonizerOptionsType): Promise +/** Native RDFC-1.0 canonicalization contracts. @module */ +/** Hash algorithms accepted by RDFC-1.0 internal hashing. */ export type DigestType = + | 'sha256' + | 'sha384' + | 'sha512' +/** Web Crypto digest names used internally. */ export type CryptoDigestType = + | 'SHA-256' + | 'SHA-384' + | 'SHA-512' +/** Result of one recursive N-degree blank-node comparison. */ export interface DegreeResultType< + Issuer, +> { + /** N-degree hash selected for this canonicalization candidate. */ + readonly hash: string + /** Issuer preserving selected traversal order. */ readonly issuer: Issuer } diff --git a/packages/rdf/compact.ts b/packages/rdf/compact.ts index 4e89b4b..66e30e5 100644 --- a/packages/rdf/compact.ts +++ b/packages/rdf/compact.ts @@ -1,29 +1,46 @@ /** Shared streaming RDF 1.2 Turtle/TriG scanner and semantic parser. @module */ import { blankNode, defaultGraph, literal, namedNode, quad, triple } from './factory.ts' -import { chunks, throwIfAborted, type TextSource } from './text.ts' -import { RDF, XSD, type Graph, type Literal, type NamedNode, type ObjectTerm, type Predicate, type Quad, type Subject } from './term.ts' +import { chunks, type TextSourceType, throwIfAborted } from './text.ts' +import { + type GraphTermType, + type Literal, + type NamedNode, + type ObjectTermType, + type PredicateTermType, + type Quad, + RDF, + type SubjectTermType, + XSD, +} from './term.ts' /** Source range expressed in UTF-16 code-unit offsets and one-based line/column positions. */ -export interface CompactRange { +export interface CompactRangeType { + /** Zero-based source offset where this record starts. */ readonly start: number + /** Exclusive zero-based source offset where this record ends. */ readonly end: number + /** One-based source line containing the start of this record. */ readonly line: number + /** One-based source column containing the start of this record. */ readonly column: number } /** Recoverable Turtle/TriG diagnostic. */ -export interface CompactDiagnostic { +export interface CompactDiagnosticType { + /** Stable machine-readable code used to classify this diagnostic or failure. */ readonly code: string + /** Human-readable explanation of the diagnostic or failure. */ readonly message: string - readonly range: CompactRange + /** Source range that locates the related token, statement, feature, or diagnostic. */ + readonly range: CompactRangeType } /** RDF version labels accepted by RDF 1.2 Turtle and TriG. */ -export type CompactVersion = '1.1' | '1.2-basic' | '1.2' +export type CompactVersionType = '1.1' | '1.2-basic' | '1.2' /** Streaming parser controls for Turtle and TriG. */ -export interface CompactOptions { +export interface CompactOptionsType { /** Retrieval/base IRI used before an in-document BASE directive appears. */ readonly baseIri?: string /** Emit diagnostics and resume at the next statement where safe. */ @@ -34,16 +51,52 @@ export interface CompactOptions { readonly maxDepth?: number /** Maximum semantic events buffered for one invalidatable statement in tolerant mode. */ readonly maxStatementEvents?: number + /** Caller-owned abort signal checked before expensive work and between long-running steps. */ readonly signal?: AbortSignal } /** Parser event common to Turtle and TriG. */ -export type CompactEvent = - | { readonly kind: 'quad'; readonly quad: Quad; readonly range: CompactRange } - | { readonly kind: 'prefix'; readonly prefix: string; readonly iri: string; readonly range: CompactRange } - | { readonly kind: 'base'; readonly iri: string; readonly range: CompactRange } - | { readonly kind: 'version'; readonly version: CompactVersion; readonly range: CompactRange } - | { readonly kind: 'diagnostic'; readonly diagnostic: CompactDiagnostic } +export type CompactEventType = + | { + /** Selects the `quad` variant of CompactEventType. */ + readonly kind: 'quad' + /** RDF quad carried by this event or triplestore mutation. */ + readonly quad: Quad + /** Source range covered by this CompactEventType. */ + readonly range: CompactRangeType + } + | { + /** Selects the `prefix` variant of CompactEventType. */ + readonly kind: 'prefix' + /** Prefix label associated with this syntax or RDF name. */ + readonly prefix: string + /** IRI retained by this CompactEventType. */ + readonly iri: string + /** Source range covered by this CompactEventType. */ + readonly range: CompactRangeType + } + | { + /** Selects the `base` variant of CompactEventType. */ + readonly kind: 'base' + /** IRI retained by this CompactEventType. */ + readonly iri: string + /** Source range covered by this CompactEventType. */ + readonly range: CompactRangeType + } + | { + /** Selects the `version` variant of CompactEventType. */ + readonly kind: 'version' + /** Version marker retained by this syntax record. */ + readonly version: CompactVersionType + /** Source range covered by this CompactEventType. */ + readonly range: CompactRangeType + } + | { + /** Selects the `diagnostic` variant of CompactEventType. */ + readonly kind: 'diagnostic' + /** Structured syntax diagnostic emitted by this parser event. */ + readonly diagnostic: CompactDiagnosticType + } /** Default max token length used when the caller does not provide an override. */ const DEFAULT_MAX_TOKEN_LENGTH = 8 * 1024 * 1024 @@ -55,7 +108,7 @@ const DEFAULT_MAX_STATEMENT_EVENTS = 1_000_000 const COMPACT_THRESHOLD = 64 * 1024 /** Transient scanner token kinds. Tokens are fields on one scanner object, not allocated AST nodes. */ -const Kind = { +const KindType = { Eof: 0, Iri: 1, PName: 2, @@ -69,7 +122,7 @@ const Kind = { Prefix: 10, Base: 11, Version: 12, - Graph: 13, + GraphTermType: 13, Dot: 14, Semicolon: 15, Comma: 16, @@ -91,15 +144,17 @@ const Kind = { } as const /** Numeric compact-syntax token kind used only inside the allocation-light scanner/parser state machine. */ -type Kind = (typeof Kind)[keyof typeof Kind] +type KindType = (typeof KindType)[keyof typeof KindType] /** Position-aware parser failure used by strict mode and converted to diagnostics in tolerant mode. */ class CompactError extends SyntaxError { + /** Stable machine-readable code used to classify this diagnostic or failure. */ readonly code: string - readonly range: CompactRange + /** Source range that locates the related token, statement, feature, or diagnostic. */ + readonly range: CompactRangeType /** Creates a position-aware Turtle/TriG syntax failure that tolerant mode can convert to a diagnostic. */ - constructor(code: string, message: string, range: CompactRange) { + constructor(code: string, message: string, range: CompactRangeType) { super(message) this.name = 'RdfCompactParseError' this.code = code @@ -115,28 +170,45 @@ class CompactError extends SyntaxError { * while still giving the semantic parser one-token lookahead. */ class Scanner { + /** Caller-owned abort signal checked before expensive work and between long-running steps. */ readonly signal: AbortSignal | undefined + /** Maximum token length accepted before the scanner reports a configured limit. */ readonly maxTokenLength: number - kind: Kind = Kind.Eof + /** Current lexical token class. `Eof` means no token is currently available. */ + kind: KindType = KindType.Eof + /** Decoded token value used by parser logic; `raw` preserves the exact source spelling. */ value = '' + /** Exact source text consumed for this token before semantic decoding. */ raw = '' + /** Zero-based source offset where this record starts. */ start = 0 + /** Exclusive zero-based source offset where this record ends. */ end = 0 + /** One-based source line containing the start of this record. */ line = 1 + /** One-based source column containing the start of this record. */ column = 1 + /** Input source currently owned by this parser or scanner until it is consumed or canceled. */ #source: AsyncGenerator + /** Streaming text decoder that preserves partial UTF-8 sequences between source chunks. */ #decoder = new TextDecoder('utf-8', { fatal: true }) + /** Retained unread source text. Compaction removes consumed prefixes to keep memory bounded. */ #buffer = '' + /** Current lookup or cursor index used to avoid rescanning already consumed state. */ #index = 0 + /** Absolute source offset corresponding to the start of the retained scanner buffer. */ #absolute = 0 + /** Current one-based source line maintained as the scanner consumes characters. */ #line = 1 + /** Current one-based source column maintained as the scanner consumes characters. */ #column = 1 + /** Whether the underlying source has reached its terminal end state. */ #done = false /** Creates one incremental scanner over bounded source chunks without materializing a token array. */ - constructor(source: TextSource, options: CompactOptions) { + constructor(source: TextSourceType, options: CompactOptionsType) { this.signal = options.signal this.maxTokenLength = options.maxTokenLength ?? DEFAULT_MAX_TOKEN_LENGTH this.#source = chunks(source, options.signal) @@ -155,7 +227,7 @@ class Scanner { const first = await this.#peek() if (first === undefined) { - this.kind = Kind.Eof + this.kind = KindType.Eof this.end = this.#absolute return } @@ -163,42 +235,56 @@ class Scanner { const three = `${first}${await this.#peek(1) ?? ''}${await this.#peek(2) ?? ''}` const two = three.slice(0, 2) - if (three === '<<(') return await this.#punct(Kind.TripleStart, 3) - if (three === ')>>') return await this.#punct(Kind.TripleEnd, 3) - if (two === '<<') return await this.#punct(Kind.ReifiedStart, 2) - if (two === '>>') return await this.#punct(Kind.ReifiedEnd, 2) - if (two === '{|') return await this.#punct(Kind.AnnotationStart, 2) - if (two === '|}') return await this.#punct(Kind.AnnotationEnd, 2) - if (two === '^^') return await this.#punct(Kind.HatHat, 2) + if (three === '<<(') return await this.#punct(KindType.TripleStart, 3) + if (three === ')>>') return await this.#punct(KindType.TripleEnd, 3) + if (two === '<<') return await this.#punct(KindType.ReifiedStart, 2) + if (two === '>>') return await this.#punct(KindType.ReifiedEnd, 2) + if (two === '{|') return await this.#punct(KindType.AnnotationStart, 2) + if (two === '|}') return await this.#punct(KindType.AnnotationEnd, 2) + if (two === '^^') return await this.#punct(KindType.HatHat, 2) switch (first) { case '.': { const next = await this.#peek(1) if (next !== undefined && /[0-9]/.test(next)) return await this.#number() - return await this.#punct(Kind.Dot, 1) + return await this.#punct(KindType.Dot, 1) } - case ';': return await this.#punct(Kind.Semicolon, 1) - case ',': return await this.#punct(Kind.Comma, 1) - case '[': return await this.#punct(Kind.LBracket, 1) - case ']': return await this.#punct(Kind.RBracket, 1) - case '(': return await this.#punct(Kind.LParen, 1) - case ')': return await this.#punct(Kind.RParen, 1) - case '{': return await this.#punct(Kind.LBrace, 1) - case '}': return await this.#punct(Kind.RBrace, 1) - case '~': return await this.#punct(Kind.Tilde, 1) - case '<': return await this.#iri() + case ';': + return await this.#punct(KindType.Semicolon, 1) + case ',': + return await this.#punct(KindType.Comma, 1) + case '[': + return await this.#punct(KindType.LBracket, 1) + case ']': + return await this.#punct(KindType.RBracket, 1) + case '(': + return await this.#punct(KindType.LParen, 1) + case ')': + return await this.#punct(KindType.RParen, 1) + case '{': + return await this.#punct(KindType.LBrace, 1) + case '}': + return await this.#punct(KindType.RBrace, 1) + case '~': + return await this.#punct(KindType.Tilde, 1) + case '<': + return await this.#iri() case '"': - case "'": return await this.#string(first) - case '@': return await this.#at() - case ':': return await this.#pname() + case "'": + return await this.#string(first) + case '@': + return await this.#at() + case ':': + return await this.#pname() case '+': - case '-': return await this.#numberOrUnknown() + case '-': + return await this.#numberOrUnknown() default: if (/[0-9]/.test(first)) return await this.#number() if (two === '_:') return await this.#blank() if (isNameStart(first)) return await this.#wordOrPname() await this.#take() - this.kind = Kind.Unknown + this.kind = KindType.Unknown this.raw = first this.value = first this.end = this.#absolute @@ -206,7 +292,7 @@ class Scanner { } /** Returns a range covering the current scanner token. */ - range(): CompactRange { + range(): CompactRangeType { return { start: this.start, end: this.end, line: this.line, column: this.column } } @@ -237,7 +323,7 @@ class Scanner { } /** Punct as one isolated step of the Scanner state machine. */ - async #punct(kind: Kind, width: number): Promise { + async #punct(kind: KindType, width: number): Promise { let raw = '' for (let i = 0; i < width; i++) raw += await this.#take() ?? '' this.kind = kind @@ -258,7 +344,7 @@ class Scanner { if (char === '>') { raw += await this.#take() this.#guard(mark) - this.kind = Kind.Iri + this.kind = KindType.Iri this.value = value this.raw = raw this.end = this.#absolute @@ -291,7 +377,9 @@ class Scanner { while (true) { const char = await this.#peek() - if (char === undefined) throw this.error('turtle-string-end', 'Unterminated Turtle string literal.') + if (char === undefined) { + throw this.error('turtle-string-end', 'Unterminated Turtle string literal.') + } if (char === quote) { if (long) { if (await this.#peek(1) === quote && await this.#peek(2) === quote) { @@ -304,7 +392,10 @@ class Scanner { } } if (!long && (char === '\n' || char === '\r')) { - throw this.error('turtle-string-line', 'Short Turtle string literals cannot contain line breaks.') + throw this.error( + 'turtle-string-line', + 'Short Turtle string literals cannot contain line breaks.', + ) } if (char === '\\') { raw += await this.#take() @@ -328,7 +419,7 @@ class Scanner { } this.#guard(mark) - this.kind = Kind.String + this.kind = KindType.String this.value = value this.raw = raw this.end = this.#absolute @@ -345,10 +436,10 @@ class Scanner { this.#guard(mark) } - if (raw === '@prefix') this.kind = Kind.Prefix - else if (raw === '@base') this.kind = Kind.Base - else if (raw === '@version') this.kind = Kind.Version - else this.kind = Kind.Lang + if (raw === '@prefix') this.kind = KindType.Prefix + else if (raw === '@base') this.kind = KindType.Base + else if (raw === '@version') this.kind = KindType.Version + else this.kind = KindType.Lang this.raw = raw this.value = raw.slice(1) this.end = this.#absolute @@ -359,7 +450,9 @@ class Scanner { const mark = this.#absolute let raw = `${await this.#take() ?? ''}${await this.#take() ?? ''}` const first = await this.#peek() - if (first === undefined || !isBlankStart(first)) throw this.error('turtle-blank', 'Invalid blank-node label.') + if (first === undefined || !isBlankStart(first)) { + throw this.error('turtle-blank', 'Invalid blank-node label.') + } while (true) { const char = await this.#peek() if (char === undefined || !isBlankChar(char)) break @@ -370,7 +463,7 @@ class Scanner { this.#rewindOne('.') raw = raw.slice(0, -1) } - this.kind = Kind.Blank + this.kind = KindType.Blank this.raw = raw this.value = raw.slice(2) this.end = this.#absolute @@ -380,11 +473,14 @@ class Scanner { async #numberOrUnknown(): Promise { const next = await this.#peek(1) const after = await this.#peek(2) - if (next !== undefined && (/[0-9]/.test(next) || (next === '.' && after !== undefined && /[0-9]/.test(after)))) { + if ( + next !== undefined && + (/[0-9]/.test(next) || (next === '.' && after !== undefined && /[0-9]/.test(after))) + ) { return await this.#number() } const first = await this.#take() ?? '' - this.kind = Kind.Unknown + this.kind = KindType.Unknown this.raw = first this.value = first this.end = this.#absolute @@ -424,7 +520,9 @@ class Scanner { raw += await this.#take() char = await this.#peek() } - if (char === undefined || !/[0-9]/.test(char)) throw this.error('turtle-number', 'Exponent requires at least one digit.') + if (char === undefined || !/[0-9]/.test(char)) { + throw this.error('turtle-number', 'Exponent requires at least one digit.') + } while (char !== undefined && /[0-9]/.test(char)) { raw += await this.#take() char = await this.#peek() @@ -433,7 +531,7 @@ class Scanner { } if (!numericKind(raw)) throw this.error('turtle-number', `Invalid numeric literal '${raw}'.`) - this.kind = Kind.Number + this.kind = KindType.Number this.raw = raw this.value = raw this.end = this.#absolute @@ -458,14 +556,18 @@ class Scanner { if (char === '\\') { raw += await this.#take() const escaped = await this.#peek() - if (escaped === undefined || !isLocalEscape(escaped)) throw this.error('turtle-pname-escape', 'Invalid prefixed-name escape.') + if (escaped === undefined || !isLocalEscape(escaped)) { + throw this.error('turtle-pname-escape', 'Invalid prefixed-name escape.') + } raw += await this.#take() continue } if (char === '%') { const a = await this.#peek(1) const b = await this.#peek(2) - if (a !== undefined && b !== undefined && /[0-9A-Fa-f]/.test(a) && /[0-9A-Fa-f]/.test(b)) { + if ( + a !== undefined && b !== undefined && /[0-9A-Fa-f]/.test(a) && /[0-9A-Fa-f]/.test(b) + ) { raw += `${await this.#take()}${await this.#take()}${await this.#take()}` continue } @@ -479,7 +581,7 @@ class Scanner { this.#rewindOne('.') raw = raw.slice(0, -1) } - this.kind = Kind.PName + this.kind = KindType.PName this.raw = raw this.value = raw this.end = this.#absolute @@ -487,14 +589,14 @@ class Scanner { } const upper = raw.toUpperCase() - if (raw === 'a') this.kind = Kind.A - else if (raw === 'true') this.kind = Kind.True - else if (raw === 'false') this.kind = Kind.False - else if (upper === 'PREFIX') this.kind = Kind.Prefix - else if (upper === 'BASE') this.kind = Kind.Base - else if (upper === 'VERSION') this.kind = Kind.Version - else if (upper === 'GRAPH') this.kind = Kind.Graph - else this.kind = Kind.Unknown + if (raw === 'a') this.kind = KindType.A + else if (raw === 'true') this.kind = KindType.True + else if (raw === 'false') this.kind = KindType.False + else if (upper === 'PREFIX') this.kind = KindType.Prefix + else if (upper === 'BASE') this.kind = KindType.Base + else if (upper === 'VERSION') this.kind = KindType.Version + else if (upper === 'GRAPH') this.kind = KindType.GraphTermType + else this.kind = KindType.Unknown this.raw = raw this.value = raw this.end = this.#absolute @@ -510,7 +612,9 @@ class Scanner { if (char === '\\') { raw += await this.#take() const escaped = await this.#peek() - if (escaped === undefined || !isLocalEscape(escaped)) throw this.error('turtle-pname-escape', 'Invalid prefixed-name escape.') + if (escaped === undefined || !isLocalEscape(escaped)) { + throw this.error('turtle-pname-escape', 'Invalid prefixed-name escape.') + } raw += await this.#take() continue } @@ -531,21 +635,28 @@ class Scanner { this.#rewindOne('.') raw = raw.slice(0, -1) } - this.kind = Kind.PName + this.kind = KindType.PName this.raw = raw this.value = raw this.end = this.#absolute } /** Unicode escape as one isolated step of the Scanner state machine. */ - async #unicodeEscape(): Promise<{ readonly raw: string; readonly value: string }> { + async #unicodeEscape(): Promise<{ + /** Original escaped source spelling before decoding or normalization. */ + readonly raw: string + /** Unicode scalar decoded from the Turtle escape sequence. */ + readonly value: string + }> { const kind = await this.#take() if (kind !== 'u' && kind !== 'U') throw this.error('turtle-unicode', 'Expected Unicode escape.') const width = kind === 'u' ? 4 : 8 let hex = '' for (let i = 0; i < width; i++) { const char = await this.#take() - if (char === undefined || !/[0-9A-Fa-f]/.test(char)) throw this.error('turtle-unicode', 'Invalid Unicode escape.') + if (char === undefined || !/[0-9A-Fa-f]/.test(char)) { + throw this.error('turtle-unicode', 'Invalid Unicode escape.') + } hex += char } const point = Number.parseInt(hex, 16) @@ -588,7 +699,9 @@ class Scanner { return } const chunk = item.value - this.#buffer += typeof chunk === 'string' ? chunk : this.#decoder.decode(chunk, { stream: true }) + this.#buffer += typeof chunk === 'string' + ? chunk + : this.#decoder.decode(chunk, { stream: true }) } } @@ -618,26 +731,39 @@ class Scanner { /** Applies configured parser resource limits before accepting more input. */ #guard(start: number): void { if (this.#absolute - start > this.maxTokenLength) { - throw this.error('turtle-token-limit', `Token exceeds maxTokenLength (${this.maxTokenLength}).`) + throw this.error( + 'turtle-token-limit', + `Token exceeds maxTokenLength (${this.maxTokenLength}).`, + ) } } } /** RDF 1.2 Turtle/TriG semantic parser over the transient scanner state. */ class Parser { + /** Scanner that owns lexical buffering and source-position tracking for this parser. */ readonly scanner: Scanner - readonly options: CompactOptions + /** Validated parse options retained for the complete parser lifetime. */ + readonly options: CompactOptionsType + /** Whether the current grammar permits TriG graph blocks instead of Turtle-only statements. */ readonly allowGraphs: boolean + /** Prefix declarations available while serializing the current RDF or SPARQL document. */ readonly prefixes = new Map() + /** Maximum nested grammar depth accepted before the parser reports a configured limit. */ readonly maxDepth: number + /** Maximum statement events emitted before the parser reports a configured limit. */ readonly maxStatementEvents: number + /** Current base IRI used to resolve relative IRIs after BASE directives. */ baseIri: string | undefined - version: CompactVersion | undefined + /** RDF syntax version announced or inferred for the current document. */ + version: CompactVersionType | undefined + /** Blank-node counter used to create deterministic parser-local identifiers when the syntax requires them. */ #generated = 0 + /** Whether the parser has consumed the first significant token and therefore fixed first-statement rules. */ #started = false /** Creates semantic Turtle/TriG parser state with isolated prefixes, base IRI, graph, and statement buffers. */ - constructor(source: TextSource, options: CompactOptions, allowGraphs: boolean) { + constructor(source: TextSourceType, options: CompactOptionsType, allowGraphs: boolean) { this.scanner = new Scanner(source, options) this.options = options this.allowGraphs = allowGraphs @@ -647,63 +773,63 @@ class Parser { } /** Parses the complete document and yields semantic/directive events. */ - async *events(): AsyncGenerator { + async *events(): AsyncGenerator { try { - if (!this.#started) { - this.#started = true - await this.scanner.next() - } - - while (this.#kind() !== Kind.Eof) { - throwIfAborted(this.options.signal) - if (isDirective(this.#kind())) { - try { - yield await this.#directive() - } catch (error) { - if (!this.options.tolerant) throw error - yield { kind: 'diagnostic', diagnostic: this.#diagnostic(error) } - await this.#recoverTop() - } - continue + if (!this.#started) { + this.#started = true + await this.scanner.next() } - if (this.allowGraphs && this.#kind() === Kind.Graph) { - await this.#advance() - const graph = await this.#graphLabel(0) - if (this.#kind() !== Kind.LBrace) { - const error = this.scanner.error('trig-graph-open', "Expected '{' after GRAPH label.") - if (!this.options.tolerant) throw error - yield { kind: 'diagnostic', diagnostic: this.#diagnostic(error) } - await this.#recoverTop() + while (this.#kind() !== KindType.Eof) { + throwIfAborted(this.options.signal) + if (isDirective(this.#kind())) { + try { + yield await this.#directive() + } catch (error) { + if (!this.options.tolerant) throw error + yield { kind: 'diagnostic', diagnostic: this.#diagnostic(error) } + await this.#recoverTop() + } continue } - yield* this.#graphBlock(graph) - continue - } - if (this.allowGraphs && this.#kind() === Kind.LBrace) { - yield* this.#graphBlock(defaultGraph()) - continue - } + if (this.allowGraphs && this.#kind() === KindType.GraphTermType) { + await this.#advance() + const graph = await this.#graphLabel(0) + if (this.#kind() !== KindType.LBrace) { + const error = this.scanner.error('trig-graph-open', "Expected '{' after GRAPH label.") + if (!this.options.tolerant) throw error + yield { kind: 'diagnostic', diagnostic: this.#diagnostic(error) } + await this.#recoverTop() + continue + } + yield* this.#graphBlock(graph) + continue + } - if (this.allowGraphs && this.#kind() === Kind.LBracket) { - yield* this.#trigBracket() - continue - } + if (this.allowGraphs && this.#kind() === KindType.LBrace) { + yield* this.#graphBlock(defaultGraph()) + continue + } - if (this.allowGraphs && isGraphLabelStart(this.#kind())) { - const range = this.scanner.range() - const lead = await this.#graphLabel(0) - if (this.#kind() === Kind.LBrace) { - yield* this.#graphBlock(lead) + if (this.allowGraphs && this.#kind() === KindType.LBracket) { + yield* this.#trigBracket() continue } - yield* this.#statement(defaultGraph(), lead, range) - continue - } - yield* this.#statement(defaultGraph()) - } + if (this.allowGraphs && isGraphLabelStart(this.#kind())) { + const range = this.scanner.range() + const lead = await this.#graphLabel(0) + if (this.#kind() === KindType.LBrace) { + yield* this.#graphBlock(lead) + continue + } + yield* this.#statement(defaultGraph(), lead, range) + continue + } + + yield* this.#statement(defaultGraph()) + } } finally { await this.scanner.close() } @@ -717,14 +843,14 @@ class Parser { * distinction is only visible after the opening bracket, so it cannot be * decided by the one-token lookahead in {@link Scanner}. */ - async *#trigBracket(): AsyncGenerator { + async *#trigBracket(): AsyncGenerator { const start = this.scanner.range() await this.#advance() const node = this.#fresh() - if (this.#kind() === Kind.RBracket) { + if (this.#kind() === KindType.RBracket) { await this.#advance() - if (this.#kind() === Kind.LBrace) { + if (this.#kind() === KindType.LBrace) { yield* this.#graphBlock(node) return } @@ -737,12 +863,15 @@ class Parser { return } - const buffered: CompactEvent[] = [] + const buffered: CompactEventType[] = [] try { for await (const event of this.#trigPropertyStatement(node, start)) { buffered.push(event) if (buffered.length > this.maxStatementEvents) { - throw this.scanner.error('turtle-event-limit', `Statement exceeds maxStatementEvents (${this.maxStatementEvents}).`) + throw this.scanner.error( + 'turtle-event-limit', + `Statement exceeds maxStatementEvents (${this.maxStatementEvents}).`, + ) } } yield* buffered @@ -753,44 +882,61 @@ class Parser { } /** Parses the remainder of a non-empty top-level blank-node property-list statement. */ - async *#trigPropertyStatement(node: Subject, start: CompactRange): AsyncGenerator { - yield* this.#predicateObjectList(node, defaultGraph(), 1, start, Kind.RBracket) - if (this.#kind() !== Kind.RBracket) { - throw this.scanner.error('turtle-property-list-end', "Expected ']' to close blank-node property list.") + async *#trigPropertyStatement( + node: SubjectTermType, + start: CompactRangeType, + ): AsyncGenerator { + yield* this.#predicateObjectList(node, defaultGraph(), 1, start, KindType.RBracket) + if (this.#kind() !== KindType.RBracket) { + throw this.scanner.error( + 'turtle-property-list-end', + "Expected ']' to close blank-node property list.", + ) } await this.#advance() - if (this.#kind() !== Kind.Dot) { + if (this.#kind() !== KindType.Dot) { yield* this.#predicateObjectList(node, defaultGraph(), 0, start) } await this.#expectDot() } - /** Graph block as one isolated step of the Parser state machine. */ - async *#graphBlock(graph: Graph): AsyncGenerator { - if (this.#kind() !== Kind.LBrace) throw this.scanner.error('trig-graph-open', "Expected '{' to start graph block.") + /** GraphTermType block as one isolated step of the Parser state machine. */ + async *#graphBlock(graph: GraphTermType): AsyncGenerator { + if (this.#kind() !== KindType.LBrace) { + throw this.scanner.error('trig-graph-open', "Expected '{' to start graph block.") + } await this.#advance() - while (this.#kind() !== Kind.RBrace && this.#kind() !== Kind.Eof) { + while (this.#kind() !== KindType.RBrace && this.#kind() !== KindType.Eof) { yield* this.#statement(graph) } - if (this.#kind() !== Kind.RBrace) throw this.scanner.error('trig-graph-end', "Expected '}' to close graph block.") + if (this.#kind() !== KindType.RBrace) { + throw this.scanner.error('trig-graph-end', "Expected '}' to close graph block.") + } await this.#advance() } /** Statement as one isolated step of the Parser state machine. */ - async *#statement(graph: Graph, lead?: Subject, leadRange?: CompactRange): AsyncGenerator { + async *#statement( + graph: GraphTermType, + lead?: SubjectTermType, + leadRange?: CompactRangeType, + ): AsyncGenerator { if (!this.options.tolerant) { yield* this.#statementStrict(graph, lead, leadRange) return } - const buffered: CompactEvent[] = [] + const buffered: CompactEventType[] = [] try { for await (const event of this.#statementStrict(graph, lead, leadRange)) { buffered.push(event) if (buffered.length > this.maxStatementEvents) { - throw this.scanner.error('turtle-event-limit', `Statement exceeds maxStatementEvents (${this.maxStatementEvents}).`) + throw this.scanner.error( + 'turtle-event-limit', + `Statement exceeds maxStatementEvents (${this.maxStatementEvents}).`, + ) } } yield* buffered @@ -801,22 +947,26 @@ class Parser { } /** Statement strict as one isolated step of the Parser state machine. */ - async *#statementStrict(graph: Graph, lead?: Subject, leadRange?: CompactRange): AsyncGenerator { + async *#statementStrict( + graph: GraphTermType, + lead?: SubjectTermType, + leadRange?: CompactRangeType, + ): AsyncGenerator { const range = leadRange ?? this.scanner.range() - let subject: Subject + let subject: SubjectTermType if (lead !== undefined) { subject = lead - } else if (this.#kind() === Kind.LBracket) { + } else if (this.#kind() === KindType.LBracket) { subject = yield* this.#blankPropertyList(graph, 0) - if (this.#kind() !== Kind.Dot) { + if (this.#kind() !== KindType.Dot) { yield* this.#predicateObjectList(subject, graph, 0) } await this.#expectDot() return - } else if (this.#kind() === Kind.ReifiedStart) { + } else if (this.#kind() === KindType.ReifiedStart) { subject = yield* this.#reified(graph, 0) - if (this.#kind() !== Kind.Dot) yield* this.#predicateObjectList(subject, graph, 0) + if (this.#kind() !== KindType.Dot) yield* this.#predicateObjectList(subject, graph, 0) await this.#expectDot() return } else { @@ -827,54 +977,61 @@ class Parser { await this.#expectDot() } - /** Predicate object list as one isolated step of the Parser state machine. */ + /** PredicateTermType object list as one isolated step of the Parser state machine. */ async *#predicateObjectList( - subject: Subject, - graph: Graph, + subject: SubjectTermType, + graph: GraphTermType, depth: number, - statementRange?: CompactRange, - terminator?: Kind, - ): AsyncGenerator { + statementRange?: CompactRangeType, + terminator?: KindType, + ): AsyncGenerator { this.#depth(depth) while (true) { const predicate = await this.#verb() yield* this.#objectList(subject, predicate, graph, depth + 1, statementRange) - if (this.#kind() !== Kind.Semicolon) return + if (this.#kind() !== KindType.Semicolon) return do await this.#advance() - while (this.#kind() === Kind.Semicolon) + while (this.#kind() === KindType.Semicolon) if (terminator !== undefined && this.#kind() === terminator) return - if (this.#kind() === Kind.Dot || this.#kind() === Kind.RBracket || this.#kind() === Kind.AnnotationEnd) return + if ( + this.#kind() === KindType.Dot || this.#kind() === KindType.RBracket || + this.#kind() === KindType.AnnotationEnd + ) return } } /** Object list as one isolated step of the Parser state machine. */ async *#objectList( - subject: Subject, - predicate: Predicate, - graph: Graph, + subject: SubjectTermType, + predicate: PredicateTermType, + graph: GraphTermType, depth: number, - statementRange?: CompactRange, - ): AsyncGenerator { + statementRange?: CompactRangeType, + ): AsyncGenerator { while (true) { const start = statementRange ?? this.scanner.range() const object = yield* this.#object(graph, depth) const asserted = quad(subject, predicate, object, graph) yield { kind: 'quad', quad: asserted, range: mergeRange(start, this.scanner.range()) } yield* this.#annotations(asserted, graph, depth + 1) - if (this.#kind() !== Kind.Comma) return + if (this.#kind() !== KindType.Comma) return await this.#advance() } } /** Annotations as one isolated step of the Parser state machine. */ - async *#annotations(asserted: Quad, graph: Graph, depth: number): AsyncGenerator { + async *#annotations( + asserted: Quad, + graph: GraphTermType, + depth: number, + ): AsyncGenerator { this.#depth(depth) const tripleTerm = triple(asserted.subject, asserted.predicate, asserted.object) - let activeReifier: Subject | undefined + let activeReifier: SubjectTermType | undefined - while (this.#kind() === Kind.Tilde || this.#kind() === Kind.AnnotationStart) { - if (this.#kind() === Kind.Tilde) { + while (this.#kind() === KindType.Tilde || this.#kind() === KindType.AnnotationStart) { + if (this.#kind() === KindType.Tilde) { const start = this.scanner.range() await this.#advance() activeReifier = isIriStart(this.#kind()) || isBlankStartKind(this.#kind()) @@ -885,13 +1042,13 @@ class Parser { quad: quad(activeReifier, namedNode(RDF.reifies), tripleTerm, graph), range: mergeRange(start, this.scanner.range()), } - if (this.#kind() !== Kind.AnnotationStart) { + if (this.#kind() !== KindType.AnnotationStart) { activeReifier = undefined continue } } - if (this.#kind() === Kind.AnnotationStart) { + if (this.#kind() === KindType.AnnotationStart) { const start = this.scanner.range() const reifier = activeReifier ?? this.#fresh() if (activeReifier === undefined) { @@ -902,9 +1059,12 @@ class Parser { } } await this.#advance() - yield* this.#predicateObjectList(reifier, graph, depth + 1, start, Kind.AnnotationEnd) - if (this.#kind() !== Kind.AnnotationEnd) { - throw this.scanner.error('turtle-annotation-end', "Expected '|}' to close annotation block.") + yield* this.#predicateObjectList(reifier, graph, depth + 1, start, KindType.AnnotationEnd) + if (this.#kind() !== KindType.AnnotationEnd) { + throw this.scanner.error( + 'turtle-annotation-end', + "Expected '|}' to close annotation block.", + ) } await this.#advance() activeReifier = undefined @@ -912,49 +1072,72 @@ class Parser { } } - /** Subject as one isolated step of the Parser state machine. */ - async *#subject(graph: Graph, depth: number): AsyncGenerator { + /** SubjectTermType as one isolated step of the Parser state machine. */ + async *#subject( + graph: GraphTermType, + depth: number, + ): AsyncGenerator { this.#depth(depth) if (isIriStart(this.#kind())) return await this.#iri() - if (this.#kind() === Kind.Blank) return await this.#labelledBlank() - if (this.#kind() === Kind.LParen) return yield* this.#collection(graph, depth + 1) - throw this.scanner.error('turtle-subject', 'Expected IRI, blank node, or collection as Turtle subject.') + if (this.#kind() === KindType.Blank) return await this.#labelledBlank() + if (this.#kind() === KindType.LParen) return yield* this.#collection(graph, depth + 1) + throw this.scanner.error( + 'turtle-subject', + 'Expected IRI, blank node, or collection as Turtle subject.', + ) } /** Object as one isolated step of the Parser state machine. */ - async *#object(graph: Graph, depth: number): AsyncGenerator { + async *#object( + graph: GraphTermType, + depth: number, + ): AsyncGenerator { this.#depth(depth) if (isIriStart(this.#kind())) return await this.#iri() - if (this.#kind() === Kind.Blank) return await this.#labelledBlank() - if (this.#kind() === Kind.LBracket) return yield* this.#blankPropertyList(graph, depth + 1) - if (this.#kind() === Kind.LParen) return yield* this.#collection(graph, depth + 1) - if (this.#kind() === Kind.String || this.#kind() === Kind.Number || this.#kind() === Kind.True || this.#kind() === Kind.False) { + if (this.#kind() === KindType.Blank) return await this.#labelledBlank() + if (this.#kind() === KindType.LBracket) return yield* this.#blankPropertyList(graph, depth + 1) + if (this.#kind() === KindType.LParen) return yield* this.#collection(graph, depth + 1) + if ( + this.#kind() === KindType.String || this.#kind() === KindType.Number || + this.#kind() === KindType.True || this.#kind() === KindType.False + ) { return await this.#literal() } - if (this.#kind() === Kind.TripleStart) return yield* this.#tripleTerm(graph, depth + 1) - if (this.#kind() === Kind.ReifiedStart) return yield* this.#reified(graph, depth + 1) + if (this.#kind() === KindType.TripleStart) return yield* this.#tripleTerm(graph, depth + 1) + if (this.#kind() === KindType.ReifiedStart) return yield* this.#reified(graph, depth + 1) throw this.scanner.error('turtle-object', 'Expected Turtle RDF object.') } /** Collection as one isolated step of the Parser state machine. */ - async *#collection(graph: Graph, depth: number): AsyncGenerator { + async *#collection( + graph: GraphTermType, + depth: number, + ): AsyncGenerator { this.#depth(depth) const start = this.scanner.range() - if (this.#kind() !== Kind.LParen) throw this.scanner.error('turtle-collection', "Expected '(' to start collection.") + if (this.#kind() !== KindType.LParen) { + throw this.scanner.error('turtle-collection', "Expected '(' to start collection.") + } await this.#advance() - if (this.#kind() === Kind.RParen) { + if (this.#kind() === KindType.RParen) { await this.#advance() return namedNode(RDF.nil) } const head = this.#fresh() let current = head - while (this.#kind() !== Kind.RParen) { - if (this.#kind() === Kind.Eof) throw this.scanner.error('turtle-collection-end', "Expected ')' to close collection.") + while (this.#kind() !== KindType.RParen) { + if (this.#kind() === KindType.Eof) { + throw this.scanner.error('turtle-collection-end', "Expected ')' to close collection.") + } const object = yield* this.#object(graph, depth + 1) yield { kind: 'quad', quad: quad(current, namedNode(RDF.first), object, graph), range: start } - if (this.#kind() === Kind.RParen) { - yield { kind: 'quad', quad: quad(current, namedNode(RDF.rest), namedNode(RDF.nil), graph), range: start } + if (this.#kind() === KindType.RParen) { + yield { + kind: 'quad', + quad: quad(current, namedNode(RDF.rest), namedNode(RDF.nil), graph), + range: start, + } break } const next = this.#fresh() @@ -966,52 +1149,73 @@ class Parser { } /** Blank property list as one isolated step of the Parser state machine. */ - async *#blankPropertyList(graph: Graph, depth: number): AsyncGenerator { + async *#blankPropertyList( + graph: GraphTermType, + depth: number, + ): AsyncGenerator { this.#depth(depth) - if (this.#kind() !== Kind.LBracket) throw this.scanner.error('turtle-property-list', "Expected '['.") + if (this.#kind() !== KindType.LBracket) { + throw this.scanner.error('turtle-property-list', "Expected '['.") + } const start = this.scanner.range() await this.#advance() const node = this.#fresh() - if (this.#kind() === Kind.RBracket) { + if (this.#kind() === KindType.RBracket) { await this.#advance() return node } - yield* this.#predicateObjectList(node, graph, depth + 1, start, Kind.RBracket) - if (this.#kind() !== Kind.RBracket) throw this.scanner.error('turtle-property-list-end', "Expected ']' to close blank-node property list.") + yield* this.#predicateObjectList(node, graph, depth + 1, start, KindType.RBracket) + if (this.#kind() !== KindType.RBracket) { + throw this.scanner.error( + 'turtle-property-list-end', + "Expected ']' to close blank-node property list.", + ) + } await this.#advance() return node } /** Triple term as one isolated step of the Parser state machine. */ - async *#tripleTerm(graph: Graph, depth: number): AsyncGenerator { + async *#tripleTerm(graph: GraphTermType, depth: number): AsyncGenerator { this.#depth(depth) - if (this.#kind() !== Kind.TripleStart) throw this.scanner.error('turtle-triple-term', "Expected '<<('.") + if (this.#kind() !== KindType.TripleStart) { + throw this.scanner.error('turtle-triple-term', "Expected '<<('.") + } await this.#advance() const subject = await this.#tripleSubject() const predicate = await this.#verb() const object = yield* this.#tripleObject(graph, depth + 1) - if (this.#kind() !== Kind.TripleEnd) throw this.scanner.error('turtle-triple-term-end', "Expected ')>>' after triple term.") + if (this.#kind() !== KindType.TripleEnd) { + throw this.scanner.error('turtle-triple-term-end', "Expected ')>>' after triple term.") + } await this.#advance() return triple(subject, predicate, object) } /** Reified as one isolated step of the Parser state machine. */ - async *#reified(graph: Graph, depth: number): AsyncGenerator { + async *#reified( + graph: GraphTermType, + depth: number, + ): AsyncGenerator { this.#depth(depth) const start = this.scanner.range() - if (this.#kind() !== Kind.ReifiedStart) throw this.scanner.error('turtle-reified', "Expected '<<'.") + if (this.#kind() !== KindType.ReifiedStart) { + throw this.scanner.error('turtle-reified', "Expected '<<'.") + } await this.#advance() const subject = yield* this.#reifiedSubject(graph, depth + 1) const predicate = await this.#verb() const object = yield* this.#reifiedObject(graph, depth + 1) - let reifier: Subject | undefined - if (this.#kind() === Kind.Tilde) { + let reifier: SubjectTermType | undefined + if (this.#kind() === KindType.Tilde) { await this.#advance() reifier = isIriStart(this.#kind()) || isBlankStartKind(this.#kind()) ? await this.#reifierTerm() : this.#fresh() } - if (this.#kind() !== Kind.ReifiedEnd) throw this.scanner.error('turtle-reified-end', "Expected '>>' after reified triple.") + if (this.#kind() !== KindType.ReifiedEnd) { + throw this.scanner.error('turtle-reified-end', "Expected '>>' after reified triple.") + } await this.#advance() const value = reifier ?? this.#fresh() yield { @@ -1023,44 +1227,68 @@ class Parser { } /** Reified subject as one isolated step of the Parser state machine. */ - async *#reifiedSubject(graph: Graph, depth: number): AsyncGenerator { + async *#reifiedSubject( + graph: GraphTermType, + depth: number, + ): AsyncGenerator { if (isIriStart(this.#kind())) return await this.#iri() if (isBlankStartKind(this.#kind())) return await this.#reifierTerm() - if (this.#kind() === Kind.ReifiedStart) return yield* this.#reified(graph, depth + 1) - throw this.scanner.error('turtle-reified-subject', 'Expected IRI, blank node, or nested reified triple.') + if (this.#kind() === KindType.ReifiedStart) return yield* this.#reified(graph, depth + 1) + throw this.scanner.error( + 'turtle-reified-subject', + 'Expected IRI, blank node, or nested reified triple.', + ) } /** Reified object as one isolated step of the Parser state machine. */ - async *#reifiedObject(graph: Graph, depth: number): AsyncGenerator { + async *#reifiedObject( + graph: GraphTermType, + depth: number, + ): AsyncGenerator { if (isIriStart(this.#kind())) return await this.#iri() if (isBlankStartKind(this.#kind())) return await this.#reifierTerm() - if (this.#kind() === Kind.String || this.#kind() === Kind.Number || this.#kind() === Kind.True || this.#kind() === Kind.False) return await this.#literal() - if (this.#kind() === Kind.TripleStart) return yield* this.#tripleTerm(graph, depth + 1) - if (this.#kind() === Kind.ReifiedStart) return yield* this.#reified(graph, depth + 1) - throw this.scanner.error('turtle-reified-object', 'Expected RDF term allowed in a reified triple object.') + if ( + this.#kind() === KindType.String || this.#kind() === KindType.Number || + this.#kind() === KindType.True || this.#kind() === KindType.False + ) return await this.#literal() + if (this.#kind() === KindType.TripleStart) return yield* this.#tripleTerm(graph, depth + 1) + if (this.#kind() === KindType.ReifiedStart) return yield* this.#reified(graph, depth + 1) + throw this.scanner.error( + 'turtle-reified-object', + 'Expected RDF term allowed in a reified triple object.', + ) } /** Triple object as one isolated step of the Parser state machine. */ - async *#tripleObject(graph: Graph, depth: number): AsyncGenerator { + async *#tripleObject( + graph: GraphTermType, + depth: number, + ): AsyncGenerator { if (isIriStart(this.#kind())) return await this.#iri() - if (this.#kind() === Kind.Blank) return await this.#labelledBlank() - if (this.#kind() === Kind.LBracket) return await this.#anonymous() - if (this.#kind() === Kind.String || this.#kind() === Kind.Number || this.#kind() === Kind.True || this.#kind() === Kind.False) return await this.#literal() - if (this.#kind() === Kind.TripleStart) return yield* this.#tripleTerm(graph, depth + 1) - throw this.scanner.error('turtle-triple-object', 'Expected RDF term allowed in a triple-term object.') + if (this.#kind() === KindType.Blank) return await this.#labelledBlank() + if (this.#kind() === KindType.LBracket) return await this.#anonymous() + if ( + this.#kind() === KindType.String || this.#kind() === KindType.Number || + this.#kind() === KindType.True || this.#kind() === KindType.False + ) return await this.#literal() + if (this.#kind() === KindType.TripleStart) return yield* this.#tripleTerm(graph, depth + 1) + throw this.scanner.error( + 'turtle-triple-object', + 'Expected RDF term allowed in a triple-term object.', + ) } /** Triple subject as one isolated step of the Parser state machine. */ - async #tripleSubject(): Promise { + async #tripleSubject(): Promise { if (isIriStart(this.#kind())) return await this.#iri() - if (this.#kind() === Kind.Blank) return await this.#labelledBlank() - if (this.#kind() === Kind.LBracket) return await this.#anonymous() + if (this.#kind() === KindType.Blank) return await this.#labelledBlank() + if (this.#kind() === KindType.LBracket) return await this.#anonymous() throw this.scanner.error('turtle-triple-subject', 'Expected IRI or blank node in triple term.') } /** Verb as one isolated step of the Parser state machine. */ - async #verb(): Promise { - if (this.#kind() === Kind.A) { + async #verb(): Promise { + if (this.#kind() === KindType.A) { await this.#advance() return namedNode(RDF.type) } @@ -1069,22 +1297,24 @@ class Parser { /** Literal as one isolated step of the Parser state machine. */ async #literal(): Promise { - if (this.#kind() === Kind.True || this.#kind() === Kind.False) { + if (this.#kind() === KindType.True || this.#kind() === KindType.False) { const raw = this.scanner.raw await this.#advance() return literal(raw, namedNode(XSD.boolean)) } - if (this.#kind() === Kind.Number) { + if (this.#kind() === KindType.Number) { const raw = this.scanner.raw const datatype = numericKind(raw) if (!datatype) throw this.scanner.error('turtle-number', `Invalid numeric literal '${raw}'.`) await this.#advance() return literal(raw, namedNode(datatype)) } - if (this.#kind() !== Kind.String) throw this.scanner.error('turtle-literal', 'Expected RDF literal.') + if (this.#kind() !== KindType.String) { + throw this.scanner.error('turtle-literal', 'Expected RDF literal.') + } const value = this.scanner.value await this.#advance() - if (this.#kind() === Kind.Lang) { + if (this.#kind() === KindType.Lang) { const raw = this.scanner.value await this.#advance() const marker = raw.lastIndexOf('--') @@ -1092,13 +1322,16 @@ class Parser { const language = raw.slice(0, marker) const direction = raw.slice(marker + 2).toLowerCase() if (direction !== 'ltr' && direction !== 'rtl') { - throw this.scanner.error('turtle-direction', `Initial text direction must be ltr or rtl, got '${direction}'.`) + throw this.scanner.error( + 'turtle-direction', + `Initial text direction must be ltr or rtl, got '${direction}'.`, + ) } return literal(value, { language, direction }) } return literal(value, raw) } - if (this.#kind() === Kind.HatHat) { + if (this.#kind() === KindType.HatHat) { await this.#advance() return literal(value, await this.#iri()) } @@ -1106,20 +1339,30 @@ class Parser { } /** Directive as one isolated step of the Parser state machine. */ - async #directive(): Promise> { + async #directive(): Promise< + Exclude + > { const kind = this.#kind() const raw = this.scanner.raw const start = this.scanner.range() const oldStyle = raw.startsWith('@') await this.#advance() - if (kind === Kind.Prefix) { - if (this.#kind() !== Kind.PName || !this.scanner.raw.endsWith(':')) { - throw this.scanner.error('turtle-prefix-name', 'PREFIX requires a prefix label ending in colon.') + if (kind === KindType.Prefix) { + if (this.#kind() !== KindType.PName || !this.scanner.raw.endsWith(':')) { + throw this.scanner.error( + 'turtle-prefix-name', + 'PREFIX requires a prefix label ending in colon.', + ) } const prefix = this.scanner.raw.slice(0, -1) await this.#advance() - if (this.#kind() !== Kind.Iri) throw this.scanner.error('turtle-prefix-iri', 'PREFIX requires an IRI reference.') + if (this.#kind() !== KindType.Iri) { + throw this.scanner.error('turtle-prefix-iri', 'PREFIX requires an IRI reference.') + } const iri = this.#resolve(this.scanner.value) await this.#advance() if (oldStyle) await this.#expectDot() @@ -1127,8 +1370,10 @@ class Parser { return { kind: 'prefix', prefix, iri, range: mergeRange(start, this.scanner.range()) } } - if (kind === Kind.Base) { - if (this.#kind() !== Kind.Iri) throw this.scanner.error('turtle-base-iri', 'BASE requires an IRI reference.') + if (kind === KindType.Base) { + if (this.#kind() !== KindType.Iri) { + throw this.scanner.error('turtle-base-iri', 'BASE requires an IRI reference.') + } const iri = this.#resolve(this.scanner.value) await this.#advance() if (oldStyle) await this.#expectDot() @@ -1136,8 +1381,10 @@ class Parser { return { kind: 'base', iri, range: mergeRange(start, this.scanner.range()) } } - if (kind === Kind.Version) { - if (this.#kind() !== Kind.String) throw this.scanner.error('turtle-version', 'VERSION requires a quoted RDF version label.') + if (kind === KindType.Version) { + if (this.#kind() !== KindType.String) { + throw this.scanner.error('turtle-version', 'VERSION requires a quoted RDF version label.') + } const version = this.scanner.value if (version !== '1.1' && version !== '1.2-basic' && version !== '1.2') { throw this.scanner.error('turtle-version', `Unsupported RDF version '${version}'.`) @@ -1153,66 +1400,74 @@ class Parser { /** Iri as one isolated step of the Parser state machine. */ async #iri(): Promise { - if (this.#kind() === Kind.Iri) { + if (this.#kind() === KindType.Iri) { const value = this.#resolve(this.scanner.value) await this.#advance() return namedNode(value) } - if (this.#kind() === Kind.PName) { + if (this.#kind() === KindType.PName) { const raw = this.scanner.raw const colon = raw.indexOf(':') const prefix = raw.slice(0, colon) const local = decodeLocal(raw.slice(colon + 1)) const base = this.prefixes.get(prefix) - if (base === undefined) throw this.scanner.error('turtle-prefix', `Prefix '${prefix}' is not defined.`) + if (base === undefined) { + throw this.scanner.error('turtle-prefix', `Prefix '${prefix}' is not defined.`) + } await this.#advance() return namedNode(`${base}${local}`) } throw this.scanner.error('turtle-iri', 'Expected IRI reference or prefixed name.') } - /** Graph label as one isolated step of the Parser state machine. */ - async #graphLabel(depth: number): Promise { + /** GraphTermType label as one isolated step of the Parser state machine. */ + async #graphLabel(depth: number): Promise { this.#depth(depth) if (isIriStart(this.#kind())) return await this.#iri() - if (this.#kind() === Kind.Blank) return await this.#labelledBlank() - if (this.#kind() === Kind.LBracket) return await this.#anonymous() + if (this.#kind() === KindType.Blank) return await this.#labelledBlank() + if (this.#kind() === KindType.LBracket) return await this.#anonymous() throw this.scanner.error('trig-graph-label', 'Expected IRI or blank node as TriG graph label.') } /** Reifier term as one isolated step of the Parser state machine. */ - async #reifierTerm(): Promise { + async #reifierTerm(): Promise { if (isIriStart(this.#kind())) return await this.#iri() - if (this.#kind() === Kind.Blank) return await this.#labelledBlank() - if (this.#kind() === Kind.LBracket) return await this.#anonymous() + if (this.#kind() === KindType.Blank) return await this.#labelledBlank() + if (this.#kind() === KindType.LBracket) return await this.#anonymous() throw this.scanner.error('turtle-reifier', 'Expected IRI or blank node as reifier.') } /** Labelled blank as one isolated step of the Parser state machine. */ - async #labelledBlank(): Promise { - if (this.#kind() !== Kind.Blank) throw this.scanner.error('turtle-blank', 'Expected blank node.') + async #labelledBlank(): Promise { + if (this.#kind() !== KindType.Blank) { + throw this.scanner.error('turtle-blank', 'Expected blank node.') + } const value = `l${this.scanner.value.length}:${this.scanner.value}` await this.#advance() return blankNode(value) } /** Anonymous as one isolated step of the Parser state machine. */ - async #anonymous(): Promise { - if (this.#kind() !== Kind.LBracket) throw this.scanner.error('turtle-anon', "Expected '['.") + async #anonymous(): Promise { + if (this.#kind() !== KindType.LBracket) throw this.scanner.error('turtle-anon', "Expected '['.") await this.#advance() - if (this.#kind() !== Kind.RBracket) throw this.scanner.error('turtle-anon', "Expected ']' for anonymous blank node.") + if (this.#kind() !== KindType.RBracket) { + throw this.scanner.error('turtle-anon', "Expected ']' for anonymous blank node.") + } await this.#advance() return this.#fresh() } /** Expect dot as one isolated step of the Parser state machine. */ async #expectDot(): Promise { - if (this.#kind() !== Kind.Dot) throw this.scanner.error('turtle-period', "Expected '.' after Turtle statement.") + if (this.#kind() !== KindType.Dot) { + throw this.scanner.error('turtle-period', "Expected '.' after Turtle statement.") + } await this.#advance() } - /** Kind as one isolated step of the Parser state machine. */ - #kind(): Kind { + /** KindType as one isolated step of the Parser state machine. */ + #kind(): KindType { return this.scanner.kind } @@ -1227,23 +1482,33 @@ class Parser { if (this.baseIri !== undefined) return new URL(reference, this.baseIri).href return new URL(reference).href } catch { - throw this.scanner.error('turtle-relative-iri', `Relative IRI '${reference}' requires a base IRI.`) + throw this.scanner.error( + 'turtle-relative-iri', + `Relative IRI '${reference}' requires a base IRI.`, + ) } } /** Fresh as one isolated step of the Parser state machine. */ - #fresh(): Subject { + #fresh(): SubjectTermType { return blankNode(`g:${++this.#generated}`) } /** Depth as one isolated step of the Parser state machine. */ #depth(depth: number): void { - if (depth > this.maxDepth) throw this.scanner.error('turtle-depth', `Nested Turtle syntax exceeds maxDepth (${this.maxDepth}).`) + if (depth > this.maxDepth) { + throw this.scanner.error( + 'turtle-depth', + `Nested Turtle syntax exceeds maxDepth (${this.maxDepth}).`, + ) + } } - /** Diagnostic as one isolated step of the Parser state machine. */ - #diagnostic(error: unknown): CompactDiagnostic { - if (error instanceof CompactError) return { code: error.code, message: error.message, range: error.range } + /** DiagnosticType as one isolated step of the Parser state machine. */ + #diagnostic(error: unknown): CompactDiagnosticType { + if (error instanceof CompactError) { + return { code: error.code, message: error.message, range: error.range } + } return { code: 'turtle-syntax', message: error instanceof Error ? error.message : String(error), @@ -1256,17 +1521,20 @@ class Parser { let square = 0 let paren = 0 let annotation = 0 - while (this.#kind() !== Kind.Eof) { - if (this.#kind() === Kind.LBracket) square++ - else if (this.#kind() === Kind.RBracket) square = Math.max(0, square - 1) - else if (this.#kind() === Kind.LParen) paren++ - else if (this.#kind() === Kind.RParen) paren = Math.max(0, paren - 1) - else if (this.#kind() === Kind.AnnotationStart) annotation++ - else if (this.#kind() === Kind.AnnotationEnd) annotation = Math.max(0, annotation - 1) - else if (this.#kind() === Kind.Dot && square === 0 && paren === 0 && annotation === 0) { + while (this.#kind() !== KindType.Eof) { + if (this.#kind() === KindType.LBracket) square++ + else if (this.#kind() === KindType.RBracket) square = Math.max(0, square - 1) + else if (this.#kind() === KindType.LParen) paren++ + else if (this.#kind() === KindType.RParen) paren = Math.max(0, paren - 1) + else if (this.#kind() === KindType.AnnotationStart) annotation++ + else if (this.#kind() === KindType.AnnotationEnd) annotation = Math.max(0, annotation - 1) + else if (this.#kind() === KindType.Dot && square === 0 && paren === 0 && annotation === 0) { await this.#advance() return - } else if (this.allowGraphs && this.#kind() === Kind.RBrace && square === 0 && paren === 0 && annotation === 0) { + } else if ( + this.allowGraphs && this.#kind() === KindType.RBrace && square === 0 && paren === 0 && + annotation === 0 + ) { return } await this.#advance() @@ -1275,12 +1543,12 @@ class Parser { /** Recover top as one isolated step of the Parser state machine. */ async #recoverTop(): Promise { - while (this.#kind() !== Kind.Eof) { - if (this.#kind() === Kind.Dot) { + while (this.#kind() !== KindType.Eof) { + if (this.#kind() === KindType.Dot) { await this.#advance() return } - if (this.allowGraphs && this.#kind() === Kind.RBrace) { + if (this.allowGraphs && this.#kind() === KindType.RBrace) { await this.#advance() return } @@ -1290,7 +1558,11 @@ class Parser { } /** Parses Turtle/TriG events; `allowGraphs` selects TriG graph syntax. */ -export function parseCompact(source: TextSource, options: CompactOptions, allowGraphs: boolean): AsyncGenerator { +export function parseCompact( + source: TextSourceType, + options: CompactOptionsType, + allowGraphs: boolean, +): AsyncGenerator { return new Parser(source, options, allowGraphs).events() } @@ -1303,23 +1575,23 @@ function numericKind(raw: string): string | undefined { } /** Returns whether the supplied value satisfies the directive contract. */ -function isDirective(kind: Kind): boolean { - return kind === Kind.Prefix || kind === Kind.Base || kind === Kind.Version +function isDirective(kind: KindType): boolean { + return kind === KindType.Prefix || kind === KindType.Base || kind === KindType.Version } /** Returns whether the supplied value satisfies the iri start contract. */ -function isIriStart(kind: Kind): boolean { - return kind === Kind.Iri || kind === Kind.PName +function isIriStart(kind: KindType): boolean { + return kind === KindType.Iri || kind === KindType.PName } /** Returns whether the supplied value satisfies the blank start kind contract. */ -function isBlankStartKind(kind: Kind): boolean { - return kind === Kind.Blank || kind === Kind.LBracket +function isBlankStartKind(kind: KindType): boolean { + return kind === KindType.Blank || kind === KindType.LBracket } /** Returns whether the supplied value satisfies the graph label start contract. */ -function isGraphLabelStart(kind: Kind): boolean { - return isIriStart(kind) || kind === Kind.Blank +function isGraphLabelStart(kind: KindType): boolean { + return isIriStart(kind) || kind === KindType.Blank } /** Returns whether the supplied value satisfies the whitespace contract. */ @@ -1365,19 +1637,33 @@ function decodeLocal(value: string): string { /** Decodes Turtle backslash escapes used in local names and string-like scanner values. */ function escapeValue(value: string): string { switch (value) { - case 't': return '\t' - case 'b': return '\b' - case 'n': return '\n' - case 'r': return '\r' - case 'f': return '\f' - case '"': return '"' - case "'": return "'" - case '\\': return '\\' - default: return value + case 't': + return '\t' + case 'b': + return '\b' + case 'n': + return '\n' + case 'r': + return '\r' + case 'f': + return '\f' + case '"': + return '"' + case "'": + return "'" + case '\\': + return '\\' + default: + return value } } /** Spans two source ranges so emitted semantic events retain the full originating syntax range. */ -function mergeRange(start: CompactRange, end: CompactRange): CompactRange { - return { start: start.start, end: Math.max(start.end, end.end), line: start.line, column: start.column } +function mergeRange(start: CompactRangeType, end: CompactRangeType): CompactRangeType { + return { + start: start.start, + end: Math.max(start.end, end.end), + line: start.line, + column: start.column, + } } diff --git a/packages/rdf/dataset.ts b/packages/rdf/dataset.ts index 20b54c0..3bde4fb 100644 --- a/packages/rdf/dataset.ts +++ b/packages/rdf/dataset.ts @@ -10,24 +10,40 @@ */ import { key } from './term.ts' -import type { Graph, ObjectTerm, Predicate, Quad, Subject, Term } from './term.ts' +import type { + GraphTermType, + ObjectTermType, + PredicateTermType, + Quad, + SubjectTermType, + Term, +} from './term.ts' import { iterate } from './source.ts' /** RDF dataset match pattern. `null` and `undefined` mean wildcard. */ -export interface MatchOptions { - readonly subject?: Subject | null - readonly predicate?: Predicate | null - readonly object?: ObjectTerm | null - readonly graph?: Graph | null +export interface MatchOptionsType { + /** RDF subject term represented by this statement, pattern, or index entry. */ + readonly subject?: SubjectTermType | null + /** RDF predicate IRI represented by this statement, pattern, or index entry. */ + readonly predicate?: PredicateTermType | null + /** RDF object term represented by this statement, pattern, or index entry. */ + readonly object?: ObjectTermType | null + /** RDF graph name represented by this quad, statement, or query target. */ + readonly graph?: GraphTermType | null } /** Mutable RDF/JS-style DatasetCore implementation. */ export class Dataset implements Iterable { + /** Primary semantic-key map containing the quads currently owned by this dataset. */ readonly #quads = new Map() - readonly #subject = new Map() - readonly #predicate = new Map() - readonly #object = new Map() - readonly #graph = new Map() + /** Secondary dataset index from subject key to matching quad keys. */ + readonly #subject = new Map() + /** Secondary dataset index from predicate key to matching quad keys. */ + readonly #predicate = new Map() + /** Secondary dataset index from object key to matching quad keys. */ + readonly #object = new Map() + /** Secondary dataset index from graph key to matching quad keys. */ + readonly #graph = new Map() /** Seeds the dataset through `addAll` so initial quads and all exact-term indexes use the normal deduplication path. */ constructor(quads?: Iterable) { @@ -58,9 +74,14 @@ export class Dataset implements Iterable { } /** Imports a sync or async source with cooperative cancellation. */ - async import(source: Iterable | AsyncIterable, options: { readonly signal?: AbortSignal } = {}): Promise { + async import(source: Iterable | AsyncIterable, options: { + /** Abort signal checked before and during this operation. */ + readonly signal?: AbortSignal + } = {}): Promise { for await (const quad of iterate(source)) { - if (options.signal?.aborted) throw options.signal.reason ?? new DOMException('Aborted', 'AbortError') + if (options.signal?.aborted) { + throw options.signal.reason ?? new DOMException('Aborted', 'AbortError') + } this.add(quad) } return this @@ -101,22 +122,38 @@ export class Dataset implements Iterable { * dominant selective lookup shapes while retaining correct wildcard behavior. */ match( - subject: Subject | null = null, - predicate: Predicate | null = null, - object: ObjectTerm | null = null, - graph: Graph | null = null, + subject: SubjectTermType | null = null, + predicate: PredicateTermType | null = null, + object: ObjectTermType | null = null, + graph: GraphTermType | null = null, ): Dataset { return new Dataset(this.matchIter({ subject, predicate, object, graph })) } /** Lazily iterates matching quads without materializing another dataset. */ - *matchIter(options: MatchOptions = {}): Generator { + *matchIter(options: MatchOptionsType = {}): Generator { const { subject = null, predicate = null, object = null, graph = null } = options - const indexed: IndexBucket[] = [] - if (subject) { const value = this.#subject.get(key(subject)); if (!value) return; indexed.push(value) } - if (predicate) { const value = this.#predicate.get(key(predicate)); if (!value) return; indexed.push(value) } - if (object) { const value = this.#object.get(key(object)); if (!value) return; indexed.push(value) } - if (graph) { const value = this.#graph.get(key(graph)); if (!value) return; indexed.push(value) } + const indexed: IndexBucketType[] = [] + if (subject) { + const value = this.#subject.get(key(subject)) + if (!value) return + indexed.push(value) + } + if (predicate) { + const value = this.#predicate.get(key(predicate)) + if (!value) return + indexed.push(value) + } + if (object) { + const value = this.#object.get(key(object)) + if (!value) return + indexed.push(value) + } + if (graph) { + const value = this.#graph.get(key(graph)) + if (!value) return + indexed.push(value) + } const candidates = indexed.length === 0 ? this.#quads.keys() : bucketValues(smallest(indexed)) @@ -133,17 +170,17 @@ export class Dataset implements Iterable { /** Deletes all quads matching a pattern and returns this dataset. */ deleteMatches( - subject: Subject | null = null, - predicate: Predicate | null = null, - object: ObjectTerm | null = null, - graph: Graph | null = null, + subject: SubjectTermType | null = null, + predicate: PredicateTermType | null = null, + object: ObjectTermType | null = null, + graph: GraphTermType | null = null, ): this { for (const quad of [...this.matchIter({ subject, predicate, object, graph })]) this.delete(quad) return this } /** Returns an inexpensive exact-term cardinality estimate when an index exists. */ - estimate(options: MatchOptions = {}): number | undefined { + estimate(options: MatchOptionsType = {}): number | undefined { const counts: number[] = [] if (options.subject) counts.push(bucketSize(this.#subject.get(key(options.subject)))) if (options.predicate) counts.push(bucketSize(this.#predicate.get(key(options.predicate)))) @@ -160,14 +197,19 @@ export class Dataset implements Iterable { } /** Exact-term index bucket that stores a singleton quad key directly and promotes to a Set only after a second match. */ -type IndexBucket = string | Set +type IndexBucketType = string | Set /** Canonical semantic keys computed once per quad for deduplication plus subject/predicate/object/graph indexing. */ interface QuadKeysType { + /** Collision-safe semantic key for the indexed quad. */ readonly quad: string + /** RDF subject term represented by this statement, pattern, or index entry. */ readonly subject: string + /** RDF predicate IRI represented by this statement, pattern, or index entry. */ readonly predicate: string + /** RDF object term represented by this statement, pattern, or index entry. */ readonly object: string + /** RDF graph name represented by this quad, statement, or query target. */ readonly graph: string } @@ -177,7 +219,13 @@ function quadKeys(quad: Quad): QuadKeysType { const predicate = key(quad.predicate) const object = key(quad.object) const graph = key(quad.graph) - return { quad: `Q${part(subject)}${part(predicate)}${part(object)}${part(graph)}`, subject, predicate, object, graph } + return { + quad: `Q${part(subject)}${part(predicate)}${part(object)}${part(graph)}`, + subject, + predicate, + object, + graph, + } } /** Length-prefixes one already-serialized term key exactly like the public quad key format. */ @@ -191,7 +239,7 @@ export function dataset(quads?: Iterable): Dataset { } /** Adds one quad key, allocating a Set only after a term has multiple quads. */ -function addIndex(index: Map, termKey: string, quadKey: string): void { +function addIndex(index: Map, termKey: string, quadKey: string): void { const current = index.get(termKey) if (current === undefined) { index.set(termKey, quadKey) @@ -205,7 +253,7 @@ function addIndex(index: Map, termKey: string, quadKey: str } /** Removes one quad key and demotes two-entry Sets back to singleton strings. */ -function deleteIndex(index: Map, termKey: string, quadKey: string): void { +function deleteIndex(index: Map, termKey: string, quadKey: string): void { const current = index.get(termKey) if (current === undefined) return if (typeof current === 'string') { @@ -218,19 +266,19 @@ function deleteIndex(index: Map, termKey: string, quadKey: } /** Returns the cardinality of one optional compact index bucket. */ -function bucketSize(value: IndexBucket | undefined): number { +function bucketSize(value: IndexBucketType | undefined): number { if (value === undefined) return 0 return typeof value === 'string' ? 1 : value.size } /** Iterates singleton and multi-quad buckets through one allocation-free lookup shape. */ -function* bucketValues(value: IndexBucket): Generator { +function* bucketValues(value: IndexBucketType): Generator { if (typeof value === 'string') yield value else yield* value } /** Chooses the narrowest exact-term index before semantic verification. */ -function smallest(values: readonly IndexBucket[]): IndexBucket { +function smallest(values: readonly IndexBucketType[]): IndexBucketType { let selected = values[0]! let selectedSize = bucketSize(selected) for (let index = 1; index < values.length; index++) { diff --git a/packages/rdf/dataset_test.ts b/packages/rdf/dataset_test.ts index cc0c2fe..8f5738a 100644 --- a/packages/rdf/dataset_test.ts +++ b/packages/rdf/dataset_test.ts @@ -48,7 +48,9 @@ describe('@okikio/rdf Dataset', () => { const controller = new AbortController() controller.abort(new Error('stop-import')) - await expect(dataset().import(values(), { signal: controller.signal })).rejects.toThrow('stop-import') + await expect(dataset().import(values(), { signal: controller.signal })).rejects.toThrow( + 'stop-import', + ) }) it('compares datasets semantically and builds insertion-order-independent keys', () => { diff --git a/packages/rdf/deno.json b/packages/rdf/deno.json index 8207f3f..d76eb8b 100644 --- a/packages/rdf/deno.json +++ b/packages/rdf/deno.json @@ -8,19 +8,21 @@ "./nquads": "./nquads/mod.ts", "./turtle": "./turtle/mod.ts", "./trig": "./trig/mod.ts", + "./shape": "./shape/mod.ts", + "./ontology": "./ontology/mod.ts", + "./stream": "./stream.ts", "./jsonld": "./jsonld/mod.ts", - "./canon": "./canon/mod.ts", "./xml": "./xml/mod.ts", "./rdfa": "./rdfa/mod.ts", "./microdata": "./microdata/mod.ts", - "./shape": "./shape/mod.ts", - "./ontology": "./ontology/mod.ts" + "./canon": "./canon/mod.ts" + }, + "publish": { + "exclude": [ + "**/*_test.ts", + "**/*_bench.ts", + "**/*_property_test.ts" + ] }, - "imports": { - "jsonld": "npm:jsonld@^9.0.0", - "rdf-canonize": "npm:rdf-canonize@^5.0.0", - "rdfxml-streaming-parser": "npm:rdfxml-streaming-parser@^3.2.0", - "rdfa-streaming-parser": "npm:rdfa-streaming-parser@^3.0.2", - "microdata-rdf-streaming-parser": "npm:microdata-rdf-streaming-parser@^3.0.0" - } + "description": "Dependency-free RDF terms, datasets, native syntax processors, canonicalization, ontology models, and shape models for TypeScript runtimes." } diff --git a/packages/rdf/factory.ts b/packages/rdf/factory.ts index ca83b26..f6c6682 100644 --- a/packages/rdf/factory.ts +++ b/packages/rdf/factory.ts @@ -3,22 +3,22 @@ import { BlankNodeValue, DefaultGraphValue, - type DirectionalLanguage, - type Graph, + type DirectionalLanguageType, + type GraphTermType, + type Literal, + LiteralValue, type NamedNode, NamedNodeValue, - type ObjectTerm, - type Predicate, + type ObjectTermType, + type PredicateTermType, type Quad, QuadValue, RDF, - type Subject, + type SubjectTermType, type Term, type TermType, type Variable, VariableValue, - type Literal, - LiteralValue, XSD, } from './term.ts' @@ -62,7 +62,7 @@ export function defaultGraph(): DefaultGraphValue { */ export function literal( value: string, - languageOrDatatype?: string | NamedNode | DirectionalLanguage, + languageOrDatatype?: string | NamedNode | DirectionalLanguageType, ): Literal { if (typeof languageOrDatatype === 'string') { const language = normalizeLanguage(languageOrDatatype) @@ -70,7 +70,9 @@ export function literal( } if (languageOrDatatype !== undefined && 'termType' in languageOrDatatype) { - if (languageOrDatatype.termType !== 'NamedNode') throw new TypeError('Literal datatype must be a named node.') + if (languageOrDatatype.termType !== 'NamedNode') { + throw new TypeError('Literal datatype must be a named node.') + } return new LiteralValue(value, languageOrDatatype) } @@ -92,7 +94,12 @@ export function literal( } /** Creates a quad or, with the default graph, an RDF 1.2 triple term. */ -export function quad(subject: Subject, predicate: Predicate, object: ObjectTerm, graph: Graph = DEFAULT_GRAPH): Quad { +export function quad( + subject: SubjectTermType, + predicate: PredicateTermType, + object: ObjectTermType, + graph: GraphTermType = DEFAULT_GRAPH, +): Quad { if (object.termType === 'Quad' && object.graph.termType !== 'DefaultGraph') { throw new TypeError('An RDF 1.2 triple term cannot contain a named graph.') } @@ -100,7 +107,11 @@ export function quad(subject: Subject, predicate: Predicate, object: ObjectTerm, } /** Creates a triple term explicitly. */ -export function triple(subject: Subject, predicate: Predicate, object: ObjectTerm): Quad { +export function triple( + subject: SubjectTermType, + predicate: PredicateTermType, + object: ObjectTermType, +): Quad { return quad(subject, predicate, object) } @@ -117,7 +128,9 @@ export function fromTerm(original: Term): TermType { return defaultGraph() case 'Literal': { const value = original as Literal - if (value.direction) return literal(value.value, { language: value.language, direction: value.direction }) + if (value.direction) { + return literal(value.value, { language: value.language, direction: value.direction }) + } if (value.language) return literal(value.value, value.language) return literal(value.value, namedNode(value.datatype.value)) } @@ -129,10 +142,10 @@ export function fromTerm(original: Term): TermType { /** Copies an RDF/JS-compatible quad recursively. */ export function fromQuad(original: Quad): Quad { return quad( - fromTerm(original.subject) as Subject, - fromTerm(original.predicate) as Predicate, - fromTerm(original.object) as ObjectTerm, - fromTerm(original.graph) as Graph, + fromTerm(original.subject) as SubjectTermType, + fromTerm(original.predicate) as PredicateTermType, + fromTerm(original.object) as ObjectTermType, + fromTerm(original.graph) as GraphTermType, ) } diff --git a/packages/rdf/jsonld/compact.ts b/packages/rdf/jsonld/compact.ts new file mode 100644 index 0000000..e07449d --- /dev/null +++ b/packages/rdf/jsonld/compact.ts @@ -0,0 +1,346 @@ +/** Native JSON-LD 1.1 compaction. @module */ +import { + type ActiveContextType, + compare, + type ContextStateType, + definition, + object, + process, +} from './context.ts' +import type { JsonLdValueType } from './types.ts' +/** Options applied during compaction. */ export interface CompactOptionsType { + /** Whether compaction can replace single-value arrays with their sole value. */ + readonly compactArrays?: boolean + /** Compact absolute identifiers relative to base. */ readonly compactToRelative?: boolean + /** Deterministic key order. */ readonly ordered?: boolean +} +/** Compacts one expanded JSON-LD value using an active context. */ export async function compactValue( + active: ActiveContextType, + activeProperty: string | null, + element: JsonLdValueType, + state: ContextStateType, + options: CompactOptionsType = {}, +): Promise { + if (element === null || scalar(element)) return element + if (Array.isArray(element)) { + const result: JsonLdValueType[] = [] + for (const item of element) { + const value = await compactValue(active, activeProperty, item, state, options) + if (value !== null) result.push(value) + } + const container = definition(active, activeProperty)?.container ?? [] + return (options.compactArrays ?? true) && result.length === 1 && activeProperty !== '@graph' && + activeProperty !== '@set' && !container.includes('@list') && !container.includes('@set') + ? result[0]! + : result + } + if (valueObject(element) || (idObject(element) && Object.keys(element).length === 1)) { + return compactScalar(active, activeProperty, element, options) + } + let context = active + const node = element as Record + for ( + const type of array(node['@type'] ?? []).filter((v): v is string => typeof v === 'string').sort( + compare, + ) + ) { + const term = selectTerm(context, type), def = term ? context.terms.get(term) : undefined + if (def?.context !== undefined) { + context = await process(context, def.context, state, def.base ?? context.base, false) + } + } + const result: Record = {} + let entries = Object.entries(node) + if (options.ordered) entries = entries.sort(([a], [b]) => compare(a, b)) + for (const [property, raw] of entries) { + if (property === '@id') { + const alias = compactIri(context, '@id', undefined, true, false, options) + if (typeof raw === 'string') { + result[alias] = compactIri(context, raw, undefined, false, false, options) + } + continue + } + if (property === '@type') { + const alias = compactIri(context, '@type', undefined, true, false, options) + const values = array(raw).filter((v): v is string => typeof v === 'string').map((v) => + compactIri(context, v, undefined, true, false, options) + ) + result[alias] = collapse( + values, + options.compactArrays ?? true, + definition(context, alias)?.container ?? [], + ) + continue + } + if (property === '@reverse' && object(raw)) { + const compacted = await compactValue(context, '@reverse', raw, state, options) + if (object(compacted)) { + result[compactIri(context, '@reverse', undefined, true, false, options)] = compacted + } + continue + } + if (property.startsWith('@')) { + result[compactIri(context, property, raw, true, false, options)] = await compactValue( + context, + activeProperty, + raw, + state, + options, + ) + continue + } + for (const item of array(raw)) { + const key = compactIri(context, property, item, true, false, options), + term = context.terms.get(key), + container = term?.container ?? [] + let scoped = context + if (term?.context !== undefined) { + scoped = await process(context, term.context, state, term.base ?? context.base, true) + } + if (object(item) && Array.isArray(item['@list'])) { + const list = await compactValue(scoped, key, item['@list'], state, options) + add( + result, + key, + container.includes('@list') + ? list + : { [compactIri(context, '@list', undefined, true, false, options)]: list }, + container, + options.compactArrays ?? true, + ) + continue + } + if (container.includes('@language') && object(item) && Object.hasOwn(item, '@value')) { + const map = getMap(result, key), + lang = typeof item['@language'] === 'string' ? item['@language'] : '@none' + addMap(map, lang, compactScalar(scoped, key, item, options), options.compactArrays ?? true) + continue + } + if (container.includes('@index') && object(item)) { + const map = getMap(result, key), + index = typeof item['@index'] === 'string' ? item['@index'] : '@none', + copy = { ...item } + delete copy['@index'] + addMap( + map, + index, + await compactValue(scoped, key, copy, state, options), + options.compactArrays ?? true, + ) + continue + } + add( + result, + key, + await compactValue(scoped, key, item, state, options), + container, + options.compactArrays ?? true, + ) + } + } + return result +} +/** Compacts one expanded IRI/keyword to the best active term or compact IRI. */ export function compactIri( + active: ActiveContextType, + value: string, + item: JsonLdValueType | undefined, + vocab: boolean, + reverse = false, + options: CompactOptionsType = {}, +): string { + const direct = candidates(active, value, item, reverse) + if (vocab && direct.length) return direct[0]! + if (vocab && active.vocab && value.startsWith(active.vocab)) { + const suffix = value.slice(active.vocab.length) + if (suffix && !active.terms.has(suffix)) return suffix + } + let best: string | undefined + for (const [term, def] of active.terms) { + if (!def.prefix || !def.id || !value.startsWith(def.id) || value === def.id) continue + const candidate = `${term}:${value.slice(def.id.length)}` + if ( + !best || candidate.length < best.length || + (candidate.length === best.length && compare(candidate, best) < 0) + ) best = candidate + } + if (best) return best + if (!vocab && (options.compactToRelative ?? true) && active.base) { + return relative(value, active.base) ?? value + } + return value +} +/** Compacts identifier/value object to scalar where definition permits it. */ function compactScalar( + active: ActiveContextType, + property: string | null, + value: Readonly>, + options: CompactOptionsType, +): JsonLdValueType { + const term = definition(active, property) + if (typeof value['@id'] === 'string') { + if (term?.type === '@id') { + return compactIri(active, value['@id'], undefined, false, false, options) + } + if (term?.type === '@vocab') { + return compactIri(active, value['@id'], undefined, true, false, options) + } + return { + [compactIri(active, '@id', undefined, true, false, options)]: compactIri( + active, + value['@id'], + undefined, + false, + false, + options, + ), + } + } + const raw = value['@value'] ?? null + if (value['@type'] === '@json' && term?.type === '@json') return raw + if (typeof value['@type'] === 'string' && term?.type === value['@type']) return raw + if (!Object.hasOwn(value, '@type')) { + const language = typeof value['@language'] === 'string' + ? value['@language'].toLowerCase() + : null, + direction = value['@direction'] === 'ltr' || value['@direction'] === 'rtl' + ? value['@direction'] + : null, + termLanguage = term?.language !== undefined + ? term.language?.toLowerCase() ?? null + : active.language?.toLowerCase() ?? null, + termDirection = term?.direction !== undefined + ? term.direction ?? null + : active.direction ?? null + if (typeof raw !== 'string' || (language === termLanguage && direction === termDirection)) { + return raw + } + } + const result: Record = { + [compactIri(active, '@value', undefined, true, false, options)]: raw, + } + if (typeof value['@type'] === 'string') { + result[compactIri(active, '@type', undefined, true, false, options)] = compactIri( + active, + value['@type'], + undefined, + true, + false, + options, + ) + } + if (typeof value['@language'] === 'string') { + result[compactIri(active, '@language', undefined, true, false, options)] = value['@language'] + } + if (value['@direction'] === 'ltr' || value['@direction'] === 'rtl') { + result[compactIri(active, '@direction', undefined, true, false, options)] = value['@direction'] + } + return result +} +/** Ranks candidate terms for one expanded IRI. */ function candidates( + active: ActiveContextType, + iri: string, + item: JsonLdValueType | undefined, + reverse: boolean, +) { + const result: { + /** Candidate compact term being scored for one expanded IRI. */ + term: string + /** Ordering score used to select the preferred compact term. */ + score: number + }[] = [] + for (const [term, def] of active.terms) { + if (def.id !== iri) continue + let score = reverse === def.reverse ? 20 : 0 + if (object(item)) { + if (Object.hasOwn(item, '@list') && def.container.includes('@list')) score += 20 + if (typeof item['@id'] === 'string' && (def.type === '@id' || def.type === '@vocab')) { + score += 12 + } + if (typeof item['@type'] === 'string' && def.type === item['@type']) score += 14 + if (Object.hasOwn(item, '@value') && def.container.includes('@language')) score += 8 + } + result.push({ term, score }) + } + return result.sort((a, b) => + b.score - a.score || a.term.length - b.term.length || compare(a.term, b.term) + ).map((v) => v.term) +} +/** Selects shortest mapped term. */ function selectTerm(active: ActiveContextType, iri: string) { + return candidates(active, iri, undefined, false)[0] +} +/** Adds compacted property value respecting array containers. */ function add( + result: Record, + property: string, + value: JsonLdValueType, + container: readonly string[], + compactArrays: boolean, +) { + const force = !compactArrays || container.includes('@set'), current = result[property] + if (current === undefined) result[property] = force ? [value] : value + else if (Array.isArray(current)) current.push(value) + else result[property] = [current, value] +} +/** Gets/creates object-valued map property. */ function getMap( + result: Record, + property: string, +) { + const current = result[property] + if (object(current)) return current + const map: Record = {} + result[property] = map + return map +} +/** Adds map entry. */ function addMap( + map: Record, + key: string, + value: JsonLdValueType, + compactArrays: boolean, +) { + const current = map[key] + if (current === undefined) map[key] = compactArrays ? value : [value] + else if (Array.isArray(current)) current.push(value) + else map[key] = [current, value] +} +/** Collapses one array when allowed. */ function collapse( + values: JsonLdValueType[], + compactArrays: boolean, + container: readonly string[], +): JsonLdValueType { + return compactArrays && values.length === 1 && !container.includes('@set') ? values[0]! : values +} +/** Computes relative IRI when authority matches. */ function relative( + value: string, + base: string, +) { + try { + const target = new URL(value), source = new URL(base) + if (target.origin !== source.origin) return undefined + const from = source.pathname.split('/') + from.pop() + const to = target.pathname.split('/') + while (from[0] === to[0]) { + from.shift() + to.shift() + } + return `${'../'.repeat(from.filter(Boolean).length)}${ + to.join('/') + }${target.search}${target.hash}` || './' + } catch { + return undefined + } +} +/** Returns scalar/array as array. */ function array(value: JsonLdValueType): JsonLdValueType[] { + return Array.isArray(value) ? value : [value] +} +/** Tests scalar. */ function scalar(value: JsonLdValueType): value is string | number | boolean { + return typeof value === 'string' || typeof value === 'number' || typeof value === 'boolean' +} +/** Tests value object. */ function valueObject( + value: JsonLdValueType, +): value is Record { + return object(value) && Object.hasOwn(value, '@value') +} +/** Tests identifier object. */ function idObject( + value: JsonLdValueType, +): value is Record { + return object(value) && typeof value['@id'] === 'string' +} diff --git a/packages/rdf/jsonld/context.ts b/packages/rdf/jsonld/context.ts new file mode 100644 index 0000000..af153f2 --- /dev/null +++ b/packages/rdf/jsonld/context.ts @@ -0,0 +1,485 @@ +/** JSON-LD 1.1 active-context processing and IRI expansion. @module */ +import type { DocumentLoaderType, JsonLdValueType, ProcessingModeType } from './types.ts' +/** JSON-LD 1.1 keywords. */ export const KEYWORDS = new Set([ + '@base', + '@container', + '@context', + '@direction', + '@graph', + '@id', + '@import', + '@included', + '@index', + '@json', + '@language', + '@list', + '@nest', + '@none', + '@prefix', + '@propagate', + '@protected', + '@reverse', + '@set', + '@type', + '@value', + '@version', + '@vocab', +]) +/** Framing-only keywords. */ export const FRAME_KEYWORDS = new Set([ + '@default', + '@embed', + '@explicit', + '@omitDefault', + '@requireAll', +]) +/** One processed term definition. */ export interface TermDefinitionType { + /** Expanded IRI or keyword assigned to the term; null disables the term mapping. */ + readonly id: string | null + /** Can prefix compact IRIs. */ readonly prefix: boolean + /** Cannot be redefined by normal local contexts. */ readonly protected: boolean + /** Values are reverse properties. */ readonly reverse: boolean + /** Container mappings. */ readonly container: readonly string[] + /** Type coercion. */ readonly type?: string + /** Language mapping. */ readonly language?: string | null + /** Direction mapping. */ readonly direction?: 'ltr' | 'rtl' | null + /** Property/type-scoped context. */ readonly context?: JsonLdValueType + /** Base of scoped context. */ readonly base?: string + /** Custom @index key. */ readonly index?: string + /** Compaction nest key. */ readonly nest?: string +} +/** Active JSON-LD context. */ export interface ActiveContextType { + /** Processed JSON-LD term definitions indexed by active term name. */ + readonly terms: ReadonlyMap + /** Current base. */ readonly base?: string + /** Original operation base. */ readonly originalBase?: string + /** Default vocab. */ readonly vocab?: string + /** Default language. */ readonly language?: string | null + /** Default direction. */ readonly direction?: 'ltr' | 'rtl' | null + /** Processing mode. */ readonly mode: ProcessingModeType + /** Previous context for non-propagation. */ readonly previous?: ActiveContextType +} +/** Shared recursive context-processing state. */ export interface ContextStateType { + /** Bounded document loader used for remote JSON-LD contexts and documents. */ + readonly load: DocumentLoaderType + /** Active remote context recursion stack. */ readonly remote: Set + /** Loaded remote context cache. */ readonly cache: Map + /** Caller cancellation. */ readonly signal?: AbortSignal +} +/** JSON-LD conformance error carrying the specification code. */ export class JsonLdError + extends Error { + /** Stable JSON-LD specification error code exposed to callers. */ + readonly code: string + /** Compatibility details shape. */ readonly details: { + /** Stable JSON-LD specification error code exposed to callers. */ + readonly code: string + } + /** Creates one stable processing error. */ constructor( + code: string, + message: string, + cause?: unknown, + ) { + super(message, cause === undefined ? undefined : { cause }) + this.name = 'JsonLdError' + this.code = code + this.details = { code } + } +} +/** Creates an empty active context. */ export function initial( + base: string | undefined, + mode: ProcessingModeType = 'json-ld-1.1', +): ActiveContextType { + return { terms: new Map(), ...(base ? { base, originalBase: base } : {}), mode } +} +/** Processes local/remote contexts into an immutable active context. */ +export async function process( + active: ActiveContextType, + local: JsonLdValueType, + state: ContextStateType, + base = active.base, + propagate = true, + remote = false, +): Promise { + abort(state.signal) + let result = clone(active) + for (let context of (Array.isArray(local) ? local : [local])) { + abort(state.signal) + if (context === null) { + if ([...result.terms.values()].some((v) => v.protected)) { + fail('invalid context nullification', 'A context with protected terms cannot be nullified.') + } + const reset = initial(active.originalBase, active.mode) + result = propagate ? reset : { ...reset, previous: active } + continue + } + if (typeof context === 'string') { + const url = resolve(context, base) + if (!url) { + fail('loading remote context failed', `Remote context '${context}' cannot be resolved.`) + } + const cached = state.cache.get(url) + if (cached) { + result = await process(result, cached.context, state, cached.base, propagate, true) + continue + } + if (state.remote.has(url)) { + fail('recursive context inclusion', `Remote context recursively includes '${url}'.`) + } + state.remote.add(url) + try { + const doc = await state.load(url) + if (!object(doc.document) || !Object.hasOwn(doc.document, '@context')) { + fail('invalid remote context', `Remote context '${url}' has no @context.`) + } + const value = doc.document['@context']! + state.cache.set(url, { context: value, base: doc.documentUrl }) + result = await process(result, value, state, doc.documentUrl, propagate, true) + } finally { + state.remote.delete(url) + } + continue + } + if (!object(context)) { + fail('invalid local context', 'A local context must be null, string, object, or array.') + } + let def = { ...context } + if (Object.hasOwn(def, '@version')) { + if (def['@version'] !== 1.1) fail('invalid @version value', '@version must be 1.1.') + if (result.mode === 'json-ld-1.0') { + fail('processing mode conflict', 'JSON-LD 1.0 cannot process @version 1.1.') + } + } + if (Object.hasOwn(def, '@import')) { + if (result.mode === 'json-ld-1.0') { + fail('invalid context entry', '@import is unavailable in JSON-LD 1.0.') + } + const raw = def['@import'] + if (typeof raw !== 'string') fail('invalid @import value', '@import must be a string.') + const url = resolve(raw, base) + if (!url) { + fail('loading remote context failed', `Imported context '${raw}' cannot be resolved.`) + } + const doc = await state.load(url) + if (!object(doc.document) || !object(doc.document['@context'])) { + fail('invalid remote context', 'Imported context must contain an object @context.') + } + const imported = doc.document['@context'] as Record + if (Object.hasOwn(imported, '@import')) { + fail('invalid context entry', 'Imported context cannot contain @import.') + } + def = { ...imported, ...def } + delete def['@import'] + } + if (Object.hasOwn(def, '@propagate')) { + if (result.mode === 'json-ld-1.0') { + fail('invalid context entry', '@propagate is unavailable in JSON-LD 1.0.') + } + if (typeof def['@propagate'] !== 'boolean') { + fail('invalid @propagate value', '@propagate must be boolean.') + } + propagate = def['@propagate'] as boolean + } + if (!propagate && !result.previous) result = { ...result, previous: active } + if (Object.hasOwn(def, '@base') && !remote) { + const value = def['@base'] + if (value === null) result = omit(result, 'base') + else if (typeof value !== 'string') fail('invalid base IRI', '@base must be null or string.') + else { + const expanded = absolute(value) ? value : resolve(value, result.base) + if (!expanded) fail('invalid base IRI', `Cannot resolve @base '${value}'.`) + result = { ...result, base: expanded } + } + } + if (Object.hasOwn(def, '@vocab')) { + const value = def['@vocab'] + if (value === null) result = omit(result, 'vocab') + else if (typeof value !== 'string') { + fail('invalid vocab mapping', '@vocab must be null or string.') + } else { + const expanded = expandIri(result, value, { documentRelative: true, vocab: true }) + if (!expanded) fail('invalid vocab mapping', '@vocab does not expand to an IRI.') + result = { ...result, vocab: expanded } + } + } + if (Object.hasOwn(def, '@language')) { + const value = def['@language'] + if (value !== null && typeof value !== 'string') { + fail('invalid default language', '@language must be null or string.') + } + result = { ...result, language: typeof value === 'string' ? value.toLowerCase() : null } + } + if (Object.hasOwn(def, '@direction')) { + const value = def['@direction'] + if (value !== null && value !== 'ltr' && value !== 'rtl') { + fail('invalid base direction', '@direction must be null, ltr, or rtl.') + } + result = { ...result, direction: value as 'ltr' | 'rtl' | null } + } + const terms = new Map(result.terms), defined = new Map() + for (const key of Object.keys(def)) { + if (key.startsWith('@')) continue + create(key, def, terms, result, state, base, defined) + } + result = { ...result, terms } + } + return result +} +/** Creates/removes one term definition with recursive prefix creation. */ function create( + term: string, + local: Readonly>, + terms: Map, + active: ActiveContextType, + state: ContextStateType, + base: string | undefined, + defined: Map, +): void { + abort(state.signal) + if (defined.get(term) === true) return + if (defined.get(term) === false) { + fail('cyclic IRI mapping', `Term '${term}' has a cyclic mapping.`) + } + defined.set(term, false) + if (KEYWORDS.has(term) || term === '') { + fail('keyword redefinition', `Invalid JSON-LD term '${term}'.`) + } + let value = local[term] + const previous = terms.get(term) + if (value === null || (object(value) && value['@id'] === null)) { + if (previous?.protected) { + fail('protected term redefinition', `Protected term '${term}' cannot be removed.`) + } + terms.set(term, { id: null, prefix: false, protected: false, reverse: false, container: [] }) + defined.set(term, true) + return + } + if (typeof value === 'string') value = { '@id': value } + if (!object(value)) { + fail('invalid term definition', `Term '${term}' must map to string/null/object.`) + } + const map = value as Record + const protectedValue = map['@protected'] ?? local['@protected'] ?? false + if (typeof protectedValue !== 'boolean') { + fail('invalid @protected value', '@protected must be boolean.') + } + let id: string | null | undefined, reverse = false + if (Object.hasOwn(map, '@reverse')) { + if (typeof map['@reverse'] !== 'string' || Object.hasOwn(map, '@id')) { + fail('invalid reverse property', `Invalid reverse mapping for '${term}'.`) + } + id = expandTerm(map['@reverse'], local, terms, active, state, base, defined) + reverse = true + } else if (Object.hasOwn(map, '@id')) { + const raw = map['@id'] + if (raw !== null && typeof raw !== 'string') { + fail('invalid IRI mapping', '@id mapping must be null or string.') + } + id = raw === null + ? null + : raw === term + ? undefined + : expandTerm(raw, local, terms, active, state, base, defined) + } + if (id === undefined) { + const colon = term.indexOf(':') + if (colon > 0) { + const prefix = term.slice(0, colon) + if (Object.hasOwn(local, prefix)) create(prefix, local, terms, active, state, base, defined) + const p = terms.get(prefix) + id = p?.id && p.prefix ? `${p.id}${term.slice(colon + 1)}` : term + } else id = active.vocab ? `${active.vocab}${term}` : term + } + let type: string | undefined + if (Object.hasOwn(map, '@type')) { + const raw = map['@type'] + if (typeof raw !== 'string') fail('invalid type mapping', '@type mapping must be string.') + type = ['@id', '@vocab', '@json', '@none'].includes(raw) + ? raw + : expandTerm(raw, local, terms, active, state, base, defined) + } + const container = containers(map['@container'], active.mode, term) + let language: string | null | undefined + if (Object.hasOwn(map, '@language')) { + const raw = map['@language'] + if (raw !== null && typeof raw !== 'string') { + fail('invalid language mapping', '@language mapping must be null/string.') + } + language = typeof raw === 'string' ? raw.toLowerCase() : null + } + let direction: 'ltr' | 'rtl' | null | undefined + if (Object.hasOwn(map, '@direction')) { + const raw = map['@direction'] + if (raw !== null && raw !== 'ltr' && raw !== 'rtl') { + fail('invalid base direction', 'Invalid @direction mapping.') + } + direction = raw as 'ltr' | 'rtl' | null + } + const scoped = Object.hasOwn(map, '@context') ? map['@context'] : undefined, + index = typeof map['@index'] === 'string' ? map['@index'] : undefined, + nest = typeof map['@nest'] === 'string' ? map['@nest'] : undefined + let prefix = map['@prefix'] === true + if ( + !Object.hasOwn(map, '@prefix') && typeof id === 'string' && !term.includes(':') && + /[/#:]$/u.test(id) + ) prefix = true + const next: TermDefinitionType = { + id: id ?? null, + prefix, + protected: protectedValue, + reverse, + container, + ...(type ? { type } : {}), + ...(language !== undefined ? { language } : {}), + ...(direction !== undefined ? { direction } : {}), + ...(scoped !== undefined ? { context: scoped, ...(base ? { base } : {}) } : {}), + ...(index ? { index } : {}), + ...(nest ? { nest } : {}), + } + if (previous?.protected && !same(previous, next)) { + fail('protected term redefinition', `Protected term '${term}' cannot be redefined.`) + } + terms.set(term, next) + defined.set(term, true) +} +/** Expands a term-definition IRI, creating referenced prefixes first. */ function expandTerm( + value: string, + local: Readonly>, + terms: Map, + active: ActiveContextType, + state: ContextStateType, + base: string | undefined, + defined: Map, +): string | undefined { + if (KEYWORDS.has(value)) return value + const direct = terms.get(value) + if (direct) return direct.id ?? undefined + const colon = value.indexOf(':') + if (colon > 0) { + const prefix = value.slice(0, colon) + if (Object.hasOwn(local, prefix)) create(prefix, local, terms, active, state, base, defined) + const def = terms.get(prefix) + if (def?.id && def.prefix) return `${def.id}${value.slice(colon + 1)}` + if (absolute(value) || value.startsWith('_:')) return value + } + return active.vocab ? `${active.vocab}${value}` : value +} +/** Expands an IRI/term under an active context. */ export function expandIri( + active: ActiveContextType, + value: string, + options: { + /** Whether relative IRIs can resolve against the active document base. */ + readonly documentRelative?: boolean + /** Default vocabulary IRI used by this parser context. */ + readonly vocab?: boolean + } = {}, +): string | undefined { + if (KEYWORDS.has(value) || FRAME_KEYWORDS.has(value)) return value + if (options.vocab) { + const direct = active.terms.get(value) + if (direct) return direct.id ?? undefined + } + const colon = value.indexOf(':') + if (colon >= 0) { + const prefix = value.slice(0, colon), suffix = value.slice(colon + 1) + if (prefix === '_' || suffix.startsWith('//')) return value + const def = active.terms.get(prefix) + if (def?.id && def.prefix) return `${def.id}${suffix}` + if (absolute(value)) return value + } + if (options.vocab && active.vocab) return `${active.vocab}${value}` + if (options.documentRelative && active.base) return resolve(value, active.base) ?? value + return value +} +/** Returns aliases for one keyword. */ export function aliases( + active: ActiveContextType, + keyword: string, +) { + return [...active.terms].filter(([, v]) => v.id === keyword).map(([k]) => k).sort(compare) +} +/** Gets a definition for one active property. */ export function definition( + active: ActiveContextType, + property: string | null, +) { + return property === null ? undefined : active.terms.get(property) +} +/** Tests for non-array JSON object. */ export function object( + value: unknown, +): value is Record { + return typeof value === 'object' && value !== null && !Array.isArray(value) +} +/** Throws a stable JSON-LD error. */ export function fail( + code: string, + message: string, + cause?: unknown, +): never { + throw new JsonLdError(code, message, cause) +} +/** Throws caller cancellation. */ export function abort(signal?: AbortSignal) { + if (signal?.aborted) throw signal.reason ?? new DOMException('Aborted', 'AbortError') +} +/** Resolves one IRI reference. */ export function resolve(value: string, base?: string) { + try { + return base ? new URL(value, base).href : new URL(value).href + } catch { + return undefined + } +} +/** Tests absolute URL-like IRI. */ export function absolute(value: string) { + try { + new URL(value) + return true + } catch { + return false + } +} +/** Unicode code-point comparator. */ export function compare(left: string, right: string) { + if (left === right) return 0 + const a = [...left], b = [...right] + for (let i = 0; i < Math.min(a.length, b.length); i++) { + const av = a[i]!.codePointAt(0)!, bv = b[i]!.codePointAt(0)! + if (av !== bv) return av - bv + } + return a.length - b.length +} +/** Clones active context term state. */ function clone( + active: ActiveContextType, +): ActiveContextType { + return { ...active, terms: new Map(active.terms) } +} +/** Omits one optional context property without assigning undefined. */ function omit< + K extends keyof ActiveContextType, +>(active: ActiveContextType, key: K): ActiveContextType { + const result = { ...active } as Record + delete result[key as string] + return result as unknown as ActiveContextType +} +/** Parses one @container mapping. */ function containers( + value: JsonLdValueType | undefined, + mode: ProcessingModeType, + term: string, +): readonly string[] { + if (value === undefined) return [] + const list = Array.isArray(value) ? value : [value] + if (!list.every((v) => typeof v === 'string')) { + fail('invalid container mapping', `@container for '${term}' must contain strings.`) + } + const allowed = new Set( + mode === 'json-ld-1.0' + ? ['@list', '@set', '@index', '@language'] + : ['@graph', '@id', '@index', '@language', '@list', '@set', '@type'], + ) + if (list.some((v) => !allowed.has(v as string))) { + fail('invalid container mapping', `Invalid @container for '${term}'.`) + } + return [...new Set(list as string[])].sort(compare) +} +/** Tests protected-term structural equivalence. */ function same( + a: TermDefinitionType, + b: TermDefinitionType, +) { + return JSON.stringify({ ...a, container: [...a.container] }) === + JSON.stringify({ ...b, container: [...b.container] }) +} diff --git a/packages/rdf/jsonld/expand.ts b/packages/rdf/jsonld/expand.ts new file mode 100644 index 0000000..92121d9 --- /dev/null +++ b/packages/rdf/jsonld/expand.ts @@ -0,0 +1,451 @@ +/** Native JSON-LD 1.1 expansion. @module */ +import { + abort, + type ActiveContextType, + compare, + type ContextStateType, + definition, + expandIri, + fail, + FRAME_KEYWORDS, + KEYWORDS, + object, + process, + type TermDefinitionType, +} from './context.ts' +import type { JsonLdValueType } from './types.ts' +/** Options affecting recursive expansion. */ export interface ExpandOptionsType { + /** Whether expansion retains framing-only forms that normal expansion removes. */ + readonly frame?: boolean + /** Deterministic code-point key order. */ readonly ordered?: boolean +} +/** Expands one JSON-LD value under an active context. */ +export async function expandValue( + active: ActiveContextType, + activeProperty: string | null, + element: JsonLdValueType, + baseUrl: string | undefined, + state: ContextStateType, + options: ExpandOptionsType = {}, + fromMap = false, +): Promise { + abort(state.signal) + if (element === null) return null + const propertyDefinition = definition(active, activeProperty) + if (scalar(element)) { + if (activeProperty === null || activeProperty === '@graph') return null + let context = active + if (propertyDefinition?.context !== undefined) { + context = await process( + active, + propertyDefinition.context, + state, + propertyDefinition.base ?? baseUrl, + ) + } + return valueExpansion(context, activeProperty, element) + } + if (Array.isArray(element)) { + const result: JsonLdValueType[] = [] + for (const item of element) { + let expanded = await expandValue( + active, + activeProperty, + item, + baseUrl, + state, + options, + fromMap, + ) + if (propertyDefinition?.container.includes('@list') && Array.isArray(expanded)) { + expanded = { '@list': expanded } + } + append(result, expanded) + } + return result + } + let context = active + if ( + context.previous && !fromMap && !hasExpandedKey(element, context, '@value') && + !onlyExpandedId(element, context) + ) context = context.previous + if (propertyDefinition?.context !== undefined) { + context = await process( + context, + propertyDefinition.context, + state, + propertyDefinition.base ?? baseUrl, + ) + } + if (Object.hasOwn(element, '@context')) { + context = await process(context, element['@context']!, state, baseUrl) + } + const typeEntries = Object.entries(element).filter(([key]) => + expandIri(context, key, { vocab: true }) === '@type' + ).sort(([a], [b]) => compare(a, b)) + for (const [, raw] of typeEntries) { + for (const term of array(raw).filter((v): v is string => typeof v === 'string').sort(compare)) { + const scoped = definition(context, term) + if (scoped?.context !== undefined) { + context = await process(context, scoped.context, state, scoped.base ?? baseUrl, false) + } + } + } + const result: Record = {}, nests: string[] = [] + await entries(element, context, result, nests, activeProperty, baseUrl, state, options) + for (let i = 0; i < nests.length; i++) { + const key = nests[i]! + for (const nested of array(element[key]!)) { + if (!object(nested)) { + fail('invalid @nest value', '@nest value must contain node-object entries.') + } + const more: string[] = [] + await entries(nested, context, result, more, activeProperty, baseUrl, state, options) + nests.push(...more) + } + } + return validate(result, activeProperty, options.frame ?? false) +} +/** Expands ordinary and keyword entries into an existing node object. */ async function entries( + element: Readonly>, + active: ActiveContextType, + result: Record, + nests: string[], + activeProperty: string | null, + baseUrl: string | undefined, + state: ContextStateType, + options: ExpandOptionsType, +) { + let values = Object.entries(element) + if (options.ordered) values = values.sort(([a], [b]) => compare(a, b)) + for (const [key, value] of values) { + abort(state.signal) + if (key === '@context') continue + const property = expandIri(active, key, { vocab: true }) + if ( + !property || + (!property.includes(':') && !KEYWORDS.has(property) && !FRAME_KEYWORDS.has(property)) + ) continue + if (KEYWORDS.has(property) || (options.frame && FRAME_KEYWORDS.has(property))) { + await keyword( + key, + property, + value, + active, + result, + nests, + activeProperty, + baseUrl, + state, + options, + ) + continue + } + const term = definition(active, key), container = term?.container ?? [] + let expanded: JsonLdValueType + if (term?.type === '@json') expanded = { '@value': value, '@type': '@json' } + else if (container.includes('@language') && object(value)) { + expanded = languageMap(value, active, term, options.ordered) + } else if ( + (container.includes('@index') || container.includes('@id') || container.includes('@type')) && + object(value) + ) expanded = await mapValue(value, key, active, term, baseUrl, state, options) + else expanded = await expandValue(active, key, value, baseUrl, state, options) + if (expanded === null) continue + if (container.includes('@list') && !listObject(expanded)) { + expanded = { '@list': array(expanded) } + } + if (term?.reverse) { + const reverse = (object(result['@reverse']) ? result['@reverse'] : {}) as Record< + string, + JsonLdValueType + > + result['@reverse'] = reverse + for (const item of array(expanded)) { + if (valueObject(item) || listObject(item)) { + fail( + 'invalid reverse property value', + `Reverse property '${key}' cannot contain value/list objects.`, + ) + } + add(reverse, property, item) + } + } else for (const item of array(expanded)) add(result, property, item) + } +} +/** Expands one JSON-LD keyword/alias. */ async function keyword( + sourceKey: string, + expanded: string, + value: JsonLdValueType, + active: ActiveContextType, + result: Record, + nests: string[], + activeProperty: string | null, + baseUrl: string | undefined, + state: ContextStateType, + options: ExpandOptionsType, +) { + if (Object.hasOwn(result, expanded) && expanded !== '@type' && expanded !== '@included') { + fail('colliding keywords', `Multiple aliases for ${expanded}.`) + } + switch (expanded) { + case '@id': + if (typeof value !== 'string') fail('invalid @id value', '@id must be string.') + result['@id'] = expandIri(active, value, { documentRelative: true }) ?? value + return + case '@type': { + const output: JsonLdValueType[] = [] + for (const item of array(value)) { + if (typeof item === 'string') { + output.push(expandIri(active, item, { documentRelative: true, vocab: true }) ?? item) + } else if (options.frame && object(item) && !Object.keys(item).length) output.push({}) + else fail('invalid type value', '@type must be string or strings.') + } + result['@type'] = [...array(result['@type'] ?? []), ...output] + return + } + case '@graph': + result['@graph'] = array(await expandValue(active, '@graph', value, baseUrl, state, options)) + .filter((v) => v !== null) + return + case '@included': + if (active.mode === 'json-ld-1.0') return + result['@included'] = [ + ...array(result['@included'] ?? []), + ...array(await expandValue(active, null, value, baseUrl, state, options)).filter((v) => + v !== null + ), + ] + return + case '@value': + result['@value'] = value + return + case '@language': + if (typeof value !== 'string') { + fail('invalid language-tagged string', '@language must be string.') + } + result['@language'] = value.toLowerCase() + return + case '@direction': + if (value !== 'ltr' && value !== 'rtl') { + fail('invalid base direction', '@direction must be ltr or rtl.') + } + result['@direction'] = value + return + case '@index': + if (typeof value !== 'string') fail('invalid @index value', '@index must be string.') + result['@index'] = value + return + case '@list': + if (activeProperty !== null && activeProperty !== '@graph') { + result['@list'] = array( + await expandValue(active, activeProperty, value, baseUrl, state, options), + ) + } + return + case '@set': + result['@set'] = await expandValue(active, activeProperty, value, baseUrl, state, options) + return + case '@reverse': { + if (!object(value)) fail('invalid @reverse value', '@reverse must be object.') + const converted = await expandValue(active, '@reverse', value, baseUrl, state, options) + if (!object(converted)) return + const reverse = (object(result['@reverse']) ? result['@reverse'] : {}) as Record< + string, + JsonLdValueType + > + for (const [property, items] of Object.entries(converted)) { + if (property === '@reverse') { + if (object(items)) { + for (const [p, list] of Object.entries(items)) { + for (const item of array(list)) add(result, p, item) + } + } + } else {for (const item of array(items)) { + if (valueObject(item) || listObject(item)) { + fail('invalid reverse property value', '@reverse cannot contain value/list objects.') + } + add(reverse, property, item) + }} + } + if (Object.keys(reverse).length) result['@reverse'] = reverse + return + } + case '@nest': + nests.push(sourceKey) + return + default: + if (options.frame && FRAME_KEYWORDS.has(expanded)) result[expanded] = value + } +} +/** Expands a language map. */ function languageMap( + input: Readonly>, + active: ActiveContextType, + term: TermDefinitionType | undefined, + ordered = false, +): JsonLdValueType[] { + let values = Object.entries(input) + if (ordered) values = values.sort(([a], [b]) => compare(a, b)) + const result: JsonLdValueType[] = [] + const direction = term?.direction !== undefined ? term.direction : active.direction + for (const [language, raw] of values) { + for (const item of array(raw)) { + if (item === null) continue + if (typeof item !== 'string') { + fail('invalid language map value', 'Language map values must be strings.') + } + const value: Record = { '@value': item } + if (language !== '@none') { + value['@language'] = language.toLowerCase() + } + if (direction) value['@direction'] = direction + result.push(value) + } + } + return result +} +/** Expands index/id/type maps. */ async function mapValue( + input: Readonly>, + activeProperty: string, + active: ActiveContextType, + term: TermDefinitionType | undefined, + baseUrl: string | undefined, + state: ContextStateType, + options: ExpandOptionsType, +) { + let entries = Object.entries(input) + if (options.ordered) entries = entries.sort(([a], [b]) => compare(a, b)) + const output: JsonLdValueType[] = [], + container = term?.container ?? [], + indexKey = term?.index ?? '@index' + for (const [index, value] of entries) { + const expanded = array( + await expandValue(active, activeProperty, value, baseUrl, state, options, true), + ) + for (const candidate of expanded) { + if (!object(candidate)) { + output.push(candidate) + continue + } + const item = { ...candidate } + if (container.includes('@index') && !Object.hasOwn(item, '@index')) { + if (indexKey === '@index') { + item['@index'] = index + } else { + const property = expandIri(active, indexKey, { vocab: true }) + if (property) add(item, property, valueExpansion(active, indexKey, index)) + } + } else if (container.includes('@id') && !Object.hasOwn(item, '@id')) { + item['@id'] = expandIri(active, index, { documentRelative: true }) ?? index + } else if (container.includes('@type')) { + item['@type'] = [ + expandIri(active, index, { vocab: true }) ?? index, + ...array(item['@type'] ?? []), + ] + } + output.push(item) + } + } + return output +} +/** Expands one scalar according to the active property definition. */ export function valueExpansion( + active: ActiveContextType, + property: string, + value: JsonLdValueType, +): JsonLdValueType { + const term = active.terms.get(property) + if (term?.type === '@id' && typeof value === 'string') { + return { '@id': expandIri(active, value, { documentRelative: true }) ?? value } + } + if (term?.type === '@vocab' && typeof value === 'string') { + return { '@id': expandIri(active, value, { documentRelative: true, vocab: true }) ?? value } + } + const result: Record = { '@value': value } + if (term?.type && term.type !== '@none') result['@type'] = term.type + else if (typeof value === 'string') { + const language = term?.language !== undefined ? term.language : active.language, + direction = term?.direction !== undefined ? term.direction : active.direction + if (language) result['@language'] = language + if (direction) result['@direction'] = direction + } + return result +} +/** Validates one expanded object and removes free-floating values. */ function validate( + result: Record, + activeProperty: string | null, + frame: boolean, +): JsonLdValueType { + if (Object.hasOwn(result, '@value')) { + if (result['@value'] === null) return null + if (Array.isArray(result['@type'])) { + const types = result['@type'] + if (types.length !== 1 || typeof types[0] !== 'string') { + fail('invalid typed value', 'A value object must contain exactly one string @type.') + } + result['@type'] = types[0]! + } + if ( + Object.hasOwn(result, '@type') && + (Object.hasOwn(result, '@language') || Object.hasOwn(result, '@direction')) + ) fail('invalid value object', 'Value object cannot combine @type with @language/@direction.') + } + if (Object.hasOwn(result, '@set')) return result['@set'] ?? null + if ((activeProperty === null || activeProperty === '@graph') && !frame) { + const keys = Object.keys(result) + if (!keys.length || (keys.length === 1 && ['@id', '@value', '@list'].includes(keys[0]!))) { + return null + } + } + return result +} +/** Adds one expanded value preserving array form. */ export function add( + target: Record, + property: string, + value: JsonLdValueType, +) { + const current = target[property] + if (current === undefined) target[property] = [value] + else if (Array.isArray(current)) current.push(value) + else target[property] = [current, value] +} +/** Returns one value as array. */ export function array( + value: JsonLdValueType, +): JsonLdValueType[] { + return Array.isArray(value) ? value : [value] +} +/** Tests value object. */ export function valueObject(value: JsonLdValueType) { + return object(value) && Object.hasOwn(value, '@value') +} +/** Tests list object. */ export function listObject(value: JsonLdValueType) { + return object(value) && Object.hasOwn(value, '@list') +} +/** Tests node object. */ export function nodeObject(value: JsonLdValueType) { + return object(value) && !valueObject(value) && !listObject(value) +} +/** Appends expanded output flattening arrays/null. */ function append( + result: JsonLdValueType[], + value: JsonLdValueType, +) { + if (value === null) return + if (Array.isArray(value)) result.push(...value.filter((v) => v !== null)) + else result.push(value) +} +/** Tests scalar JSON value. */ function scalar( + value: JsonLdValueType, +): value is string | number | boolean { + return typeof value === 'string' || typeof value === 'number' || typeof value === 'boolean' +} +/** Tests if an input has a key expanding to keyword. */ function hasExpandedKey( + element: Readonly>, + active: ActiveContextType, + keyword: string, +) { + return Object.keys(element).some((key) => expandIri(active, key, { vocab: true }) === keyword) +} +/** Tests non-propagation identifier-only special case. */ function onlyExpandedId( + element: Readonly>, + active: ActiveContextType, +) { + const keys = Object.keys(element) + return keys.length === 1 && expandIri(active, keys[0]!, { vocab: true }) === '@id' +} diff --git a/packages/rdf/jsonld/frame.ts b/packages/rdf/jsonld/frame.ts new file mode 100644 index 0000000..3c9c3d9 --- /dev/null +++ b/packages/rdf/jsonld/frame.ts @@ -0,0 +1,125 @@ +/** Native JSON-LD 1.1 framing over expanded node maps. @module */ +import { compare, object } from './context.ts' +import { array } from './expand.ts' +import { flatten } from './node.ts' +import type { EmbedType, JsonLdValueType } from './types.ts' +/** Options controlling native framing. */ export interface FrameOptionsType { + /** Initial JSON-LD framing policy for embedding referenced nodes. */ + readonly embed?: EmbedType + /** Include only explicitly framed properties. */ readonly explicit?: boolean + /** Omit properties whose output would only be defaults. */ readonly omitDefault?: boolean + /** Require all declared frame properties. */ readonly requireAll?: boolean + /** Deterministic key ordering. */ readonly ordered?: boolean +} +/** Frames an expanded document against an expanded frame. */ export function frame( + document: JsonLdValueType, + frameValue: JsonLdValueType, + options: FrameOptionsType = {}, +): JsonLdValueType[] { + const nodes = flatten(document).filter(object), + byId = new Map>() + for (const node of nodes) if (typeof node['@id'] === 'string') byId.set(node['@id'], node) + const frame = Array.isArray(frameValue) ? frameValue[0] : frameValue + if (!object(frame)) return [] + const output: JsonLdValueType[] = [] + const embedded = new Set() + for (const node of nodes) { + if (matches(node, frame, options.requireAll ?? bool(frame['@requireAll'], false))) { + output.push(embed(node, frame, byId, embedded, options, new Set())) + } + } + return output +} +/** Tests whether one node matches frame id/type/property patterns. */ function matches( + node: Readonly>, + frame: Readonly>, + requireAll: boolean, +) { + if (Object.hasOwn(frame, '@id')) { + const wanted = array(frame['@id']!) + if (!wanted.some((v) => object(v) && !Object.keys(v).length || v === node['@id'])) return false + } + if (Object.hasOwn(frame, '@type')) { + const wanted = array(frame['@type']!), actual = array(node['@type'] ?? []) + if (!wanted.some((v) => object(v) && !Object.keys(v).length || actual.includes(v))) return false + } + const props = Object.keys(frame).filter((k) => !k.startsWith('@')) + if (requireAll && props.some((p) => !Object.hasOwn(node, p))) return false + if ( + !requireAll && props.length && props.every((p) => { + const value = frame[p] + return !Object.hasOwn(node, p) && (value === undefined || defaultValue(value) === undefined) + }) + ) return false + return true +} +/** Recursively embeds references while preventing cycles and respecting embed policy. */ function embed( + node: Record, + frame: Record, + byId: Map>, + embedded: Set, + options: FrameOptionsType, + path: Set, +): JsonLdValueType { + const id = typeof node['@id'] === 'string' ? node['@id'] : undefined, + mode = embedMode(frame['@embed'] ?? options.embed ?? '@once') + if (id && mode === '@never') return { '@id': id } + if (id && path.has(id)) return { '@id': id } + if (id && mode === '@once' && embedded.has(id)) return { '@id': id } + if (id) { + embedded.add(id) + path = new Set(path) + path.add(id) + } + const explicit = bool(frame['@explicit'], options.explicit ?? false), + omitDefault = bool(frame['@omitDefault'], options.omitDefault ?? false), + result: Record = {} + if (id) result['@id'] = id + if (node['@type'] !== undefined) result['@type'] = node['@type']! + let properties = explicit + ? Object.keys(frame).filter((k) => !k.startsWith('@')) + : Object.keys(node).filter((k) => !k.startsWith('@')) + properties = [...new Set(properties)].sort(compare) + for (const property of properties) { + const subRaw = frame[property], + sub = object(Array.isArray(subRaw) ? subRaw[0] : subRaw) + ? (Array.isArray(subRaw) ? subRaw[0] : subRaw) as Record + : {} + const values = array(node[property] ?? []) + if (!values.length) { + if (!omitDefault && subRaw !== undefined) { + const fallback = defaultValue(subRaw) + if (fallback !== undefined) result[property] = [{ '@preserve': fallback }] + } + continue + } + result[property] = values.map((value) => { + if (object(value) && typeof value['@id'] === 'string') { + const target = byId.get(value['@id']) + if (target) return embed(target, sub, byId, embedded, options, path) + } + return value + }) + } + if (object(node['@reverse'])) result['@reverse'] = node['@reverse']! + return result +} +/** Reads @default from one frame property. */ function defaultValue( + value: JsonLdValueType, +): JsonLdValueType | undefined { + const first = Array.isArray(value) ? value[0] : value + return object(first) && Object.hasOwn(first, '@default') ? first['@default'] : undefined +} +/** Normalizes JSON-LD framing embed spellings. */ function embedMode( + value: JsonLdValueType, +): '@always' | '@once' | '@never' { + if (value === true || value === '@always') return '@always' + if (value === false || value === '@never') return '@never' + return '@once' +} +/** Reads a boolean frame option with fallback. */ function bool( + value: JsonLdValueType | undefined, + fallback: boolean, +) { + return typeof value === 'boolean' ? value : fallback +} diff --git a/packages/rdf/jsonld/loader.ts b/packages/rdf/jsonld/loader.ts index 6041dde..50b0797 100644 --- a/packages/rdf/jsonld/loader.ts +++ b/packages/rdf/jsonld/loader.ts @@ -15,7 +15,9 @@ const JSON_LD_CONTEXT_REL = 'http://www.w3.org/ns/json-ld#context' /** Shared cache contract for caller-owned remote JSON-LD documents. */ export interface DocumentCacheType { + /** Returns the previously issued or cached value without changing ordering state. */ get(url: string): RemoteDocumentType | undefined + /** Stores one loaded remote document in the caller-provided JSON-LD cache. */ set(url: string, value: RemoteDocumentType): void } @@ -31,10 +33,15 @@ export interface LoaderOptionsType { readonly allowUrl?: (url: URL) => boolean | Promise /** Optional caller-owned cross-operation cache. */ readonly cache?: DocumentCacheType + /** Maximum number of remote JSON-LD documents admitted by one loader instance. */ readonly maxDocuments?: number + /** Maximum decoded bytes admitted by JSON-LD processing. */ readonly maxBytes?: number + /** Maximum HTTP redirects followed while loading one remote JSON-LD document. */ readonly maxRedirects?: number + /** Maximum elapsed milliseconds allowed for one remote JSON-LD fetch. */ readonly timeoutMs?: number + /** Abort signal checked before and during JSON-LD processing. */ readonly signal?: AbortSignal } @@ -60,14 +67,26 @@ export function createDocumentLoader(options: LoaderOptionsType = {}): DocumentL if (cached) return cached const pending = inflight.get(url) if (pending) return await pending - if (++documents > maxDocuments) throw new JsonLdLoadError('document-limit', `JSON-LD remote document count exceeds maxDocuments (${maxDocuments}).`, url) + if (++documents > maxDocuments) { + throw new JsonLdLoadError( + 'document-limit', + `JSON-LD remote document count exceeds maxDocuments (${maxDocuments}).`, + url, + ) + } const promise = load(url) inflight.set(url, promise) try { const document = await promise const bytes = measure(document.document) - if (bytes > maxBytes) throw new JsonLdLoadError('document-size', `JSON-LD remote document exceeds maxBytes (${maxBytes}).`, url) + if (bytes > maxBytes) { + throw new JsonLdLoadError( + 'document-size', + `JSON-LD remote document exceeds maxBytes (${maxBytes}).`, + url, + ) + } local.set(url, document) local.set(document.documentUrl, document) options.cache?.set(url, document) @@ -81,7 +100,13 @@ export function createDocumentLoader(options: LoaderOptionsType = {}): DocumentL /** Resolves one remote JSON-LD document through the bounded cache/deduplication and redirect policy. */ async function load(url: string): Promise { if (options.loadDocument) return await options.loadDocument(url) - if (!options.remote) throw new JsonLdLoadError('remote-disabled', 'Remote JSON-LD document loading is disabled.', url) + if (!options.remote) { + throw new JsonLdLoadError( + 'remote-disabled', + 'Remote JSON-LD document loading is disabled.', + url, + ) + } const fetchOptions: FetchOptionsType = { fetch: options.fetch ?? fetch, maxBytes, @@ -96,7 +121,18 @@ export function createDocumentLoader(options: LoaderOptionsType = {}): DocumentL /** Stable failure from the bounded remote-document layer. */ export class JsonLdLoadError extends Error { - readonly kind: 'remote-disabled' | 'url' | 'document-limit' | 'document-size' | 'redirect-limit' | 'http' | 'json' | 'timeout' | 'abort' + /** Discriminates the concrete JsonLdLoadError variant. */ + readonly kind: + | 'remote-disabled' + | 'url' + | 'document-limit' + | 'document-size' + | 'redirect-limit' + | 'http' + | 'json' + | 'timeout' + | 'abort' + /** Remote document URL associated with this JSON-LD load failure. */ readonly url: string /** Creates a stable JSON-LD loading failure with the requested URL and underlying cause. */ @@ -110,21 +146,34 @@ export class JsonLdLoadError extends Error { /** Fully resolved remote-fetch policy passed through every redirect so limits and URL approval cannot be bypassed mid-chain. */ interface FetchOptionsType { + /** Fetch implementation used to load remote JSON-LD documents. */ readonly fetch: typeof fetch + /** Caller policy that decides whether a resolved remote URL may be fetched. */ readonly allowUrl?: (url: URL) => boolean | Promise + /** Maximum decoded bytes admitted by JSON-LD processing. */ readonly maxBytes: number + /** Maximum HTTP redirects followed while loading one remote JSON-LD document. */ readonly maxRedirects: number + /** Maximum elapsed milliseconds allowed for one remote JSON-LD fetch. */ readonly timeoutMs: number + /** Abort signal checked before and during JSON-LD processing. */ readonly signal?: AbortSignal } /** Fetches one JSON-LD document while applying redirect, byte, timeout, and URL policy. */ -async function fetchDocument(input: string, options: FetchOptionsType): Promise { +async function fetchDocument( + input: string, + options: FetchOptionsType, +): Promise { let current = toHttpUrl(input) - for (let redirects = 0; ; redirects++) { + for (let redirects = 0;; redirects++) { abort(options.signal) if (options.allowUrl && !(await options.allowUrl(current))) { - throw new JsonLdLoadError('url', `JSON-LD remote URL is not allowed: ${current.href}`, current.href) + throw new JsonLdLoadError( + 'url', + `JSON-LD remote URL is not allowed: ${current.href}`, + current.href, + ) } const timed = timeout(options.signal, options.timeoutMs) @@ -136,38 +185,88 @@ async function fetchDocument(input: string, options: FetchOptionsType): Promise< signal: timed.signal, }) } catch (error) { - if (options.signal?.aborted) throw new JsonLdLoadError('abort', 'JSON-LD remote load was aborted.', current.href, error) - if (timed.expired()) throw new JsonLdLoadError('timeout', `JSON-LD remote load exceeded ${options.timeoutMs}ms.`, current.href, error) + if (options.signal?.aborted) { + throw new JsonLdLoadError('abort', 'JSON-LD remote load was aborted.', current.href, error) + } + if (timed.expired()) { + throw new JsonLdLoadError( + 'timeout', + `JSON-LD remote load exceeded ${options.timeoutMs}ms.`, + current.href, + error, + ) + } throw error } finally { timed.dispose() } if (response.status >= 300 && response.status < 400) { - if (redirects >= options.maxRedirects) throw new JsonLdLoadError('redirect-limit', `JSON-LD redirects exceed maxRedirects (${options.maxRedirects}).`, current.href) + if (redirects >= options.maxRedirects) { + throw new JsonLdLoadError( + 'redirect-limit', + `JSON-LD redirects exceed maxRedirects (${options.maxRedirects}).`, + current.href, + ) + } const location = response.headers.get('location') - if (!location) throw new JsonLdLoadError('http', `JSON-LD redirect ${response.status} has no Location header.`, current.href) + if (!location) { + throw new JsonLdLoadError( + 'http', + `JSON-LD redirect ${response.status} has no Location header.`, + current.href, + ) + } current = toHttpUrl(new URL(location, current).href) continue } - if (!response.ok) throw new JsonLdLoadError('http', `JSON-LD remote load returned HTTP ${response.status}.`, current.href) + if (!response.ok) { + throw new JsonLdLoadError( + 'http', + `JSON-LD remote load returned HTTP ${response.status}.`, + current.href, + ) + } const contentLength = Number(response.headers.get('content-length')) if (Number.isFinite(contentLength) && contentLength > options.maxBytes) { - throw new JsonLdLoadError('document-size', `JSON-LD remote document exceeds maxBytes (${options.maxBytes}).`, current.href) + throw new JsonLdLoadError( + 'document-size', + `JSON-LD remote document exceeds maxBytes (${options.maxBytes}).`, + current.href, + ) } const bytes = new Uint8Array(await response.arrayBuffer()) - if (bytes.byteLength > options.maxBytes) throw new JsonLdLoadError('document-size', `JSON-LD remote document exceeds maxBytes (${options.maxBytes}).`, current.href) + if (bytes.byteLength > options.maxBytes) { + throw new JsonLdLoadError( + 'document-size', + `JSON-LD remote document exceeds maxBytes (${options.maxBytes}).`, + current.href, + ) + } + const contentType = response.headers.get('content-type')?.toLowerCase() ?? '' let document: JsonLdValueType try { - document = JSON.parse(new TextDecoder('utf-8', { fatal: true }).decode(bytes)) as JsonLdValueType + const text = new TextDecoder('utf-8', { fatal: true }).decode(bytes) + document = contentType.includes('text/html') || contentType.includes('application/xhtml+xml') + ? text + : JSON.parse(text) as JsonLdValueType } catch (error) { - throw new JsonLdLoadError('json', 'Remote JSON-LD document is not valid UTF-8 JSON.', current.href, error) + throw new JsonLdLoadError( + 'json', + 'Remote JSON-LD document is not valid UTF-8 JSON or HTML text.', + current.href, + error, + ) } return { - contextUrl: contextLink(response.headers.get('link'), response.headers.get('content-type'), current), + contextUrl: contextLink( + response.headers.get('link'), + response.headers.get('content-type'), + current, + ), documentUrl: response.url || current.href, document, } @@ -185,7 +284,13 @@ function contextLink(header: string | null, contentType: string | null, base: UR const rel = /(?:^|;)\s*rel\s*=\s*(?:"([^"]*)"|([^;\s]+))/iu.exec(parameters) const relations = (rel?.[1] ?? rel?.[2] ?? '').split(/\s+/u) if (!relations.includes(JSON_LD_CONTEXT_REL)) continue - if (context !== null) throw new JsonLdLoadError('http', 'Remote document contains more than one JSON-LD context Link relation.', base.href) + if (context !== null) { + throw new JsonLdLoadError( + 'http', + 'Remote document contains more than one JSON-LD context Link relation.', + base.href, + ) + } context = new URL(match[1]!, base).href } return context @@ -220,7 +325,11 @@ function toHttpUrl(value: string): URL { throw new JsonLdLoadError('url', `Invalid JSON-LD remote URL '${value}'.`, value, error) } if (url.protocol !== 'http:' && url.protocol !== 'https:') { - throw new JsonLdLoadError('url', `Unsupported JSON-LD remote URL protocol '${url.protocol}'.`, url.href) + throw new JsonLdLoadError( + 'url', + `Unsupported JSON-LD remote URL protocol '${url.protocol}'.`, + url.href, + ) } return url } @@ -232,8 +341,11 @@ function measure(value: JsonLdValueType): number { /** Creates a disposable timeout signal without transferring ownership of the caller signal. */ function timeout(signal: AbortSignal | undefined, timeoutMs: number): { + /** Abort signal checked before and during JSON-LD processing. */ readonly signal: AbortSignal + /** Returns whether the JSON-LD fetch deadline elapsed before the request completed. */ expired(): boolean + /** Releases the timeout resource after the JSON-LD fetch settles. */ dispose(): void } { const controller = new AbortController() @@ -265,12 +377,16 @@ function abort(signal?: AbortSignal): void { /** Validates a positive finite loader limit such as bytes, redirects, or context count. */ function positive(value: number, name: string): number { - if (!Number.isSafeInteger(value) || value <= 0) throw new RangeError(`${name} must be a positive safe integer.`) + if (!Number.isSafeInteger(value) || value <= 0) { + throw new RangeError(`${name} must be a positive safe integer.`) + } return value } /** Validates a non-negative finite loader limit that may be explicitly disabled with zero. */ function nonNegative(value: number, name: string): number { - if (!Number.isSafeInteger(value) || value < 0) throw new RangeError(`${name} must be a non-negative safe integer.`) + if (!Number.isSafeInteger(value) || value < 0) { + throw new RangeError(`${name} must be a non-negative safe integer.`) + } return value } diff --git a/packages/rdf/jsonld/mod.ts b/packages/rdf/jsonld/mod.ts index 2091c9f..55fc18a 100644 --- a/packages/rdf/jsonld/mod.ts +++ b/packages/rdf/jsonld/mod.ts @@ -1,132 +1,328 @@ -/** Full JSON-LD processing facade with bounded document loading. @module */ - -import { parse as parseNQuads, write as writeNQuads } from '../nquads/mod.ts' +/** Native JSON-LD 1.1 expansion, compaction, flattening, framing, and RDF conversion. @module */ import type { Quad } from '../term.ts' -import { createDocumentLoader, type LoaderOptionsType } from './loader.ts' -import type { JsonLdValueType, ProcessorType } from './types.ts' - +import { compactValue } from './compact.ts' +import { + type ActiveContextType, + type ContextStateType, + initial, + JsonLdError, + object, + process, +} from './context.ts' +import { expandValue } from './expand.ts' +import { frame as applyFrame } from './frame.ts' +import { flatten as flattenNodes } from './node.ts' +import { fromRdf as rdfToJson, toRdf as jsonToRdf } from './rdf.ts' +import { createDocumentLoader, JsonLdLoadError, type LoaderOptionsType } from './loader.ts' +import type { + DocumentLoaderType, + EmbedType, + JsonLdValueType, + ProcessingModeType, + RdfDirectionType, + RemoteDocumentType, +} from './types.ts' export { createDocumentLoader, JsonLdLoadError } from './loader.ts' +export { JsonLdError } from './context.ts' export type { DocumentCacheType, LoaderOptionsType } from './loader.ts' -export type { DocumentLoaderType, JsonLdValueType, ProcessorType, RemoteDocumentType } from './types.ts' - -/** Default max rdf bytes used when the caller does not provide an override. */ -const DEFAULT_MAX_RDF_BYTES = 64 * 1024 * 1024 - -/** Processing options shared by JSON-LD operations. */ +export type { + DocumentLoaderType, + EmbedType, + JsonLdValueType, + ProcessingModeType, + RdfDirectionType, + RemoteDocumentType, +} from './types.ts' +/** Processing options shared by native JSON-LD operations. */ export interface OptionsType extends LoaderOptionsType { - /** Optional processor injection for tests, custom builds, or alternate conforming engines. */ - readonly processor?: ProcessorType - /** Base IRI forwarded to the JSON-LD processor. */ - readonly base?: string - /** Maximum intermediary N-Quads bytes for JSON-LD/RDF conversion. */ - readonly maxRdfBytes?: number -} - -/** JSON text serialization options. */ -export interface SerializeOptionsType extends OptionsType { + /** Base IRI for relative identifiers. */ readonly base?: string + /** JSON-LD processing mode. */ readonly processingMode?: ProcessingModeType + /** Compact single-value arrays where allowed. */ readonly compactArrays?: boolean + /** Compact absolute IRIs relative to the base. */ readonly compactToRelative?: boolean + /** Context applied before expansion. */ readonly expandContext?: JsonLdValueType + /** Extract every JSON-LD script from HTML rather than the first matching script. */ readonly extractAllScripts?: + boolean + /** Deterministic code-point ordering. */ readonly ordered?: boolean + /** Permit blank-node predicates in RDF output. */ readonly produceGeneralizedRdf?: boolean + /** Directional string RDF mapping. */ readonly rdfDirection?: RdfDirectionType + /** Convert known XSD values to JSON scalars during fromRdf. */ readonly useNativeTypes?: boolean + /** Preserve rdf:type predicate rather than @type during fromRdf. */ readonly useRdfType?: boolean + /** Initial framing embed mode. */ readonly embed?: EmbedType + /** Include only explicitly framed properties. */ readonly explicit?: boolean + /** Frame the default graph. */ readonly frameDefault?: boolean + /** Omit default-only framed properties. */ readonly omitDefault?: boolean + /** Omit unnecessary top-level @graph wrapper. */ readonly omitGraph?: boolean + /** Require every declared frame property. */ readonly requireAll?: boolean +} +/** JSON serialization options. */ export interface SerializeOptionsType extends OptionsType { + /** Number of spaces used to indent serialized JSON-LD output. */ readonly space?: number } - -/** Expands JSON-LD using the JSON-LD 1.1 processing algorithm. */ -export async function expand(input: unknown, options: OptionsType = {}): Promise { - const { processor, settings } = await prepare(options) - const result = await processor.expand(input, settings) - abort(options.signal) - return result -} - -/** Compact as a focused public package operation. */ -export async function compact(input: unknown, context: unknown, options: OptionsType = {}): Promise { - const { processor, settings } = await prepare(options) - const result = await processor.compact(input, context, settings) - abort(options.signal) - return result -} - -/** Flattens JSON-LD, optionally compacting it with a context. */ -export async function flatten(input: unknown, context?: unknown, options: OptionsType = {}): Promise { - const { processor, settings } = await prepare(options) - const result = await processor.flatten(input, context, settings) - abort(options.signal) - return result -} - -/** Frames JSON-LD with one JSON-LD frame. */ -export async function frame(input: unknown, value: unknown, options: OptionsType = {}): Promise { - const { processor, settings } = await prepare(options) - const result = await processor.frame(input, value, settings) - abort(options.signal) - return result -} - -/** Converts JSON-LD into native `@okikio/rdf` quads through standards N-Quads. */ -export async function toRdf(input: unknown, options: OptionsType = {}): Promise { - const { processor, settings } = await prepare(options) - const result = await processor.toRDF(input, { ...settings, format: 'application/n-quads' }) - if (typeof result !== 'string') throw new TypeError('JSON-LD processor did not return N-Quads for application/n-quads.') - limit(result, options.maxRdfBytes ?? DEFAULT_MAX_RDF_BYTES) - const quads: Quad[] = [] - const parseOptions = options.signal ? { signal: options.signal } : {} - for await (const value of parseNQuads(result, parseOptions)) quads.push(value) - abort(options.signal) - return quads -} - -/** Parses JSON-LD and emits native RDF quads. Processing is materialized by jsonld.js before emission. */ -export async function* parse(input: unknown, options: OptionsType = {}): AsyncGenerator { +/** Expands JSON-LD using the package-owned JSON-LD 1.1 algorithms. */ export async function expand( + input: unknown, + options: OptionsType = {}, +): Promise { + const prepared = await prepareInput(input, options), + state = stateFor(options), + active = initial(prepared.base, options.processingMode ?? 'json-ld-1.1') + let context = active + if (options.expandContext !== undefined) { + context = await process(context, options.expandContext, state, prepared.base) + } + if (prepared.contextUrl) { + context = await process(context, prepared.contextUrl, state, prepared.base) + } + const value = await expandValue(context, null, prepared.document, prepared.base, state, { + ...(options.ordered === undefined ? {} : { ordered: options.ordered }), + }) + return Array.isArray(value) ? value.filter((v) => v !== null) : value === null ? [] : [value] +} +/** Compacts JSON-LD with one caller-supplied context. */ export async function compact( + input: unknown, + contextValue: unknown, + options: OptionsType = {}, +): Promise { + const expanded = await expand(input, options), + state = stateFor(options), + contextJson = asJson(contextValue), + active = await process( + initial(options.base, options.processingMode ?? 'json-ld-1.1'), + contextJson, + state, + options.base, + ) + const compacted = await compactValue(active, null, expanded, state, { + ...(options.compactArrays === undefined ? {} : { compactArrays: options.compactArrays }), + ...(options.compactToRelative === undefined + ? {} + : { compactToRelative: options.compactToRelative }), + ...(options.ordered === undefined ? {} : { ordered: options.ordered }), + }) + if (object(compacted)) return { '@context': contextJson, ...compacted } + return { '@context': contextJson, '@graph': compacted } +} +/** Flattens JSON-LD, optionally compacting with a context. */ export async function flatten( + input: unknown, + contextValue?: unknown, + options: OptionsType = {}, +): Promise { + const values = flattenNodes(await expand(input, options)) + if (contextValue !== undefined) return compact(values, contextValue, options) + return values +} +/** Frames JSON-LD using native node-map matching and embedding. */ export async function frame( + input: unknown, + frameValue: unknown, + options: OptionsType = {}, +): Promise { + const frameJson = asJson(frameValue), + expanded = await expand(input, options), + frameExpanded = await expandFrame(frameJson, options), + framed = applyFrame(expanded, frameExpanded, { + ...(options.embed === undefined ? {} : { embed: options.embed }), + ...(options.explicit === undefined ? {} : { explicit: options.explicit }), + ...(options.omitDefault === undefined ? {} : { omitDefault: options.omitDefault }), + ...(options.requireAll === undefined ? {} : { requireAll: options.requireAll }), + ...(options.ordered === undefined ? {} : { ordered: options.ordered }), + }) + const context = object(frameJson) && Object.hasOwn(frameJson, '@context') + ? frameJson['@context'] + : undefined + if (context !== undefined) { + const compacted = await compact(framed, context, options) + if ( + options.omitGraph && object(compacted) && Array.isArray(compacted['@graph']) && + compacted['@graph'].length === 1 + ) return compacted['@graph'][0]! + return compacted + } + return options.omitGraph && framed.length === 1 ? framed[0]! : framed +} +/** Converts JSON-LD directly into native RDF quads. */ export async function toRdf( + input: unknown, + options: OptionsType = {}, +): Promise { + return jsonToRdf(await expand(input, options), { + ...(options.produceGeneralizedRdf === undefined + ? {} + : { produceGeneralizedRdf: options.produceGeneralizedRdf }), + ...(options.rdfDirection === undefined ? {} : { rdfDirection: options.rdfDirection }), + ...(options.signal === undefined ? {} : { signal: options.signal }), + }) +} +/** Parses JSON-LD and emits native RDF quads. */ export async function* parse( + input: unknown, + options: OptionsType = {}, +): AsyncGenerator { for (const value of await toRdf(input, options)) yield value } - -/** Converts native RDF quads to expanded JSON-LD. */ -export async function fromRdf(source: Iterable, options: OptionsType = {}): Promise { - abort(options.signal) - const text = writeNQuads(source) - limit(text, options.maxRdfBytes ?? DEFAULT_MAX_RDF_BYTES) - const { processor, settings } = await prepare(options) - const result = await processor.fromRDF(text, { ...settings, format: 'application/n-quads' }) - abort(options.signal) - return result -} - -/** Serializes native RDF quads as JSON-LD JSON text. */ -export async function serialize(source: Iterable, options: SerializeOptionsType = {}): Promise { +/** Converts native RDF quads into expanded JSON-LD. */ export async function fromRdf( + source: Iterable, + options: OptionsType = {}, +): Promise { + return rdfToJson(source, { + ...(options.rdfDirection === undefined ? {} : { rdfDirection: options.rdfDirection }), + ...(options.useNativeTypes === undefined ? {} : { useNativeTypes: options.useNativeTypes }), + ...(options.useRdfType === undefined ? {} : { useRdfType: options.useRdfType }), + ...(options.signal === undefined ? {} : { signal: options.signal }), + }) +} +/** Serializes native RDF quads as JSON-LD JSON text. */ export async function serialize( + source: Iterable, + options: SerializeOptionsType = {}, +): Promise { return `${JSON.stringify(await fromRdf(source, options), null, options.space)}\n` } - -/** Prepares one operation with an operation-local bounded document loader. */ -async function prepare(options: OptionsType): Promise<{ - readonly processor: ProcessorType - readonly settings: Readonly> -}> { - abort(options.signal) - const processor = options.processor ?? await defaultProcessor() - const documentLoader = createDocumentLoader(options) +/** Prepares a JSON-LD frame with framing-only keyword preservation. */ async function expandFrame( + frameValue: JsonLdValueType, + options: OptionsType, +) { + const state = stateFor(options), + active = initial(options.base, options.processingMode ?? 'json-ld-1.1'), + value = await expandValue(active, null, frameValue, options.base, state, { + frame: true, + ...(options.ordered === undefined ? {} : { ordered: options.ordered }), + }) + return Array.isArray(value) ? value : value === null ? [] : [value] +} +/** Creates operation-local remote context state. */ function stateFor( + options: OptionsType, +): ContextStateType { return { - processor, - settings: options.base === undefined ? { documentLoader } : { base: options.base, documentLoader }, + load: createDocumentLoader(options), + remote: new Set(), + cache: new Map(), + ...(options.signal ? { signal: options.signal } : {}), } } - -/** Lazily resolved JSON-LD processor shared across operations without making the RDF root import processor code. */ -let processorPromise: Promise | undefined - -/** Lazily imports jsonld.js only when this subpath actually performs processing. */ -async function defaultProcessor(): Promise { - processorPromise ??= import('jsonld').then((module) => { - const value = 'default' in module ? module.default : module - return value as unknown as ProcessorType - }) - return await processorPromise -} - -/** Increments one JSON-LD operation counter and fails before the configured work limit is exceeded. */ -function limit(value: string, maxBytes: number): void { - if (!Number.isSafeInteger(maxBytes) || maxBytes <= 0) throw new RangeError('maxRdfBytes must be a positive safe integer.') - const bytes = new TextEncoder().encode(value).byteLength - if (bytes > maxBytes) throw new RangeError(`JSON-LD RDF intermediary exceeds maxRdfBytes (${maxBytes}).`) -} - -/** Throws the caller supplied abort reason when cancellation has been requested. */ -function abort(signal?: AbortSignal): void { - if (signal?.aborted) throw signal.reason ?? new DOMException('Aborted', 'AbortError') +/** Prepared local or remote document plus effective base/context URL. */ interface PreparedType { + /** Parsed JSON-LD document prepared for the current processor operation. */ + readonly document: JsonLdValueType + /** Effective base/document URL. */ readonly base?: string + /** External context URL, when supplied by HTTP Link. */ readonly contextUrl?: string +} +/** Resolves input objects, JSON strings, remote URLs, and HTML JSON-LD script elements. */ async function prepareInput( + input: unknown, + options: OptionsType, +): Promise { + if (typeof input === 'string' && /^https?:\/\//iu.test(input)) { + const load = createDocumentLoader(options), + remote = await load(input), + base = remote.documentUrl + if (typeof remote.document === 'string') { + return { + document: html(remote.document, base, options.extractAllScripts ?? false), + base, + ...(remote.contextUrl ? { contextUrl: remote.contextUrl } : {}), + } + } + return { + document: remote.document, + base, + ...(remote.contextUrl ? { contextUrl: remote.contextUrl } : {}), + } + } + if (typeof input === 'string') { + const trimmed = input.trim() + if (trimmed.startsWith('{') || trimmed.startsWith('[')) { + try { + return { + document: JSON.parse(trimmed) as JsonLdValueType, + ...(options.base ? { base: options.base } : {}), + } + } catch (error) { + throw new JsonLdError( + 'loading document failed', + 'JSON-LD string input is not valid JSON.', + error, + ) + } + } + if (/', close) + if (type === 'application/ld+json' && (!fragment || id === fragment)) { + const text = source.slice(end + 1, close) + try { + scripts.push(JSON.parse(text) as JsonLdValueType) + } catch (error) { + throw new JsonLdError( + 'loading document failed', + 'HTML JSON-LD script is not valid JSON.', + error, + ) + } + if (fragment || !all) break + } + offset = closeEnd < 0 ? source.length : closeEnd + 1 + } + if (!scripts.length) return [] + return all ? scripts : scripts[0]! +} +/** Finds a start-tag end while respecting quoted attributes. */ function tagEnd( + text: string, + offset: number, +) { + let quote = '' + for (let i = offset; i < text.length; i++) { + const c = text[i]! + if (quote) { + if (c === quote) quote = '' + continue + } + if (c === '"' || c === "'") quote = c + else if (c === '>') return i + } + return -1 +} +/** Parses focused script attributes needed by JSON-LD extraction. */ function attributes( + value: string, +) { + const map = new Map() + const re = /([^\s=/>]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+)))?/gu + for (const match of value.matchAll(re)) { + map.set(match[1]!.toLowerCase(), match[2] ?? match[3] ?? match[4] ?? '') + } + return map +} +/** Converts unknown JSON-compatible input to the recursive JSON-LD type. */ function asJson( + value: unknown, +): JsonLdValueType { + if ( + value === null || typeof value === 'string' || typeof value === 'number' || + typeof value === 'boolean' + ) return value + if (Array.isArray(value)) return value.map(asJson) + if (typeof value === 'object') { + const result: Record = {} + for (const [key, item] of Object.entries(value)) result[key] = asJson(item) + return result + } + throw new TypeError('JSON-LD input must be JSON-compatible.') } diff --git a/packages/rdf/jsonld/mod_test.ts b/packages/rdf/jsonld/mod_test.ts index 8f4ea75..55e2c17 100644 --- a/packages/rdf/jsonld/mod_test.ts +++ b/packages/rdf/jsonld/mod_test.ts @@ -1,27 +1,20 @@ import { describe, it } from 'node:test' import { expect } from '@std/expect' -import { namedNode, quad } from '../mod.ts' -import { compact, createDocumentLoader, fromRdf, toRdf, type ProcessorType } from './mod.ts' - -const processor: ProcessorType = { - async expand(input) { return [input as never] }, - async compact(input, _context, options) { - expect(typeof options?.documentLoader).toBe('function') - return input as never - }, - async flatten(input) { return input as never }, - async frame(input) { return input as never }, - async toRDF() { return ' .\n' }, - async fromRDF(input) { - expect(typeof input).toBe('string') - return { '@id': 'https://example.com/s' } - }, -} +import { namedNode, quad, RDF } from '../mod.ts' +import { + compact, + createDocumentLoader, + expand, + flatten, + frame, + fromRdf, + serialize, + toRdf, +} from './mod.ts' describe('@okikio/rdf/jsonld', () => { - it('keeps remote document loading disabled unless the caller explicitly supplies or enables it', async () => { - const load = createDocumentLoader() - await expect(load('https://schema.org/')).rejects.toThrow('disabled') + it('keeps remote loading disabled unless the caller explicitly enables or supplies it', async () => { + await expect(createDocumentLoader()('https://example.test/context')).rejects.toThrow('disabled') }) it('deduplicates concurrent caller-owned remote document loads', async () => { @@ -33,24 +26,151 @@ describe('@okikio/rdf/jsonld', () => { return { contextUrl: null, documentUrl: url, document: { '@context': {} } } }, }) - const [left, right] = await Promise.all([load('https://example.com/context'), load('https://example.com/context')]) + const [left, right] = await Promise.all([ + load('https://example.test/context'), + load('https://example.test/context'), + ]) expect(loads).toBe(1) expect(left).toBe(right) }) - it('owns JSON-LD/RDF conversion while preserving native RDF terms', async () => { - const values = await toRdf({ '@id': 'https://example.com/s' }, { processor }) - expect(values).toHaveLength(1) - expect(values[0]?.subject.value).toBe('https://example.com/s') + it('matches the W3C basic expansion shape without an external processor', async () => { + const value = { + '@context': { + t1: 'http://example.com/t1', + t2: 'http://example.com/t2', + term1: 'http://example.com/term1', + term2: 'http://example.com/term2', + term3: 'http://example.com/term3', + term4: 'http://example.com/term4', + term5: 'http://example.com/term5', + }, + '@id': 'http://example.com/id1', + '@type': 't1', + term1: 'v1', + term2: { '@value': 'v2', '@type': 't2' }, + term3: { '@value': 'v3', '@language': 'en' }, + term4: 4, + term5: [50, 51], + } + expect(await expand(value)).toEqual([{ + '@id': 'http://example.com/id1', + '@type': ['http://example.com/t1'], + 'http://example.com/term1': [{ '@value': 'v1' }], + 'http://example.com/term2': [{ '@value': 'v2', '@type': 'http://example.com/t2' }], + 'http://example.com/term3': [{ '@value': 'v3', '@language': 'en' }], + 'http://example.com/term4': [{ '@value': 4 }], + 'http://example.com/term5': [{ '@value': 50 }, { '@value': 51 }], + }]) + }) + + it('expands transparent @nest into the containing node', async () => { + expect( + await expand({ + '@context': { '@vocab': 'http://example.org/' }, + p1: 'v1', + '@nest': { p2: 'v2' }, + }), + ).toEqual([{ + 'http://example.org/p1': [{ '@value': 'v1' }], + 'http://example.org/p2': [{ '@value': 'v2' }], + }]) + }) + + it('expands language maps and reverse relationships', async () => { + const language = await expand({ + '@context': { + vocab: 'http://example.com/vocab/', + label: { '@id': 'vocab:label', '@container': '@language' }, + }, + '@id': 'http://example.com/queen', + label: { en: 'The Queen', de: ['Die Königin', 'Ihre Majestät'] }, + }, { ordered: true }) + const languageNode = language[0] as Record + expect(languageNode['http://example.com/vocab/label']).toEqual([ + { '@value': 'Die Königin', '@language': 'de' }, + { '@value': 'Ihre Majestät', '@language': 'de' }, + { '@value': 'The Queen', '@language': 'en' }, + ]) + + const reverse = await expand({ + '@context': { name: 'http://xmlns.com/foaf/0.1/name' }, + '@id': 'http://example.com/people/markus', + name: 'Markus', + '@reverse': { + 'http://xmlns.com/foaf/0.1/knows': { + '@id': 'http://example.com/people/dave', + name: 'Dave', + }, + }, + }) + const reverseNode = reverse[0] as Record + expect(reverseNode['@reverse']).toBeDefined() + }) - const json = await fromRdf([ - quad(namedNode('https://example.com/s'), namedNode('https://example.com/p'), namedNode('https://example.com/o')), - ], { processor }) - expect(json).toEqual({ '@id': 'https://example.com/s' }) + it('compacts, flattens, and frames native expanded data', async () => { + const input = { + '@context': { + name: 'https://schema.org/name', + knows: { '@id': 'https://schema.org/knows', '@type': '@id' }, + }, + '@id': 'https://example.test/a', + name: 'A', + knows: 'https://example.test/b', + } + const compacted = await compact(input, { + name: 'https://schema.org/name', + knows: { '@id': 'https://schema.org/knows', '@type': '@id' }, + }) + expect(compacted).toBeDefined() + const flat = await flatten(input) + expect(Array.isArray(flat)).toBe(true) + const framed = await frame(input, { + '@context': { name: 'https://schema.org/name' }, + '@type': {}, + name: {}, + }) + expect(framed).toBeDefined() + }) + + it('converts RDF collections in both directions without an N-Quads intermediary', async () => { + const input = { + '@context': { items: { '@id': 'https://example.test/items', '@container': '@list' } }, + '@id': 'https://example.test/s', + items: ['a', 'b'], + } + const values = await toRdf(input) + expect(values.some((value) => value.predicate.value === RDF.first)).toBe(true) + const output = await fromRdf(values) + expect(JSON.stringify(output)).toContain('@list') + expect(await serialize(values)).toMatch(/@list/u) + }) + + it('extracts raw application/ld+json scripts and honors a document fragment target', async () => { + const html = + `` + const load = async (url: string) => ({ + contextUrl: null, + documentUrl: `${url}#second`, + document: html, + }) + const output = await expand('https://example.test/page', { loadDocument: load }) + const selected = output[0] as Record + expect(selected['@id']).toBe('urn:second') + expect(selected['https://schema.org/name']).toEqual([{ '@value': 'selected' }]) }) - it('passes the bounded loader to non-RDF JSON-LD operations too', async () => { - const value = { '@context': { name: 'https://schema.org/name' }, name: 'Widget' } - expect(await compact(value, {}, { processor })).toBe(value) + it('round-trips ordinary native RDF values', async () => { + const values = [ + quad( + namedNode('https://example.test/s'), + namedNode('https://example.test/p'), + namedNode('https://example.test/o'), + ), + ] + expect(await fromRdf(values)).toEqual([{ + '@id': 'https://example.test/s', + 'https://example.test/p': [{ '@id': 'https://example.test/o' }], + }]) }) }) diff --git a/packages/rdf/jsonld/node.ts b/packages/rdf/jsonld/node.ts new file mode 100644 index 0000000..74ba35f --- /dev/null +++ b/packages/rdf/jsonld/node.ts @@ -0,0 +1,182 @@ +/** JSON-LD node-map generation and flattening. @module */ +import { compare, object } from './context.ts' +import { array, listObject, valueObject } from './expand.ts' +import type { JsonLdValueType } from './types.ts' +/** Expanded node object stored in a graph map. */ export type NodeType = Record< + string, + JsonLdValueType +> +/** Graph keyed by expanded node identifier. */ export type GraphType = Map +/** Node map keyed by graph identifier. */ export type NodeMapType = Map +/** Deterministic blank-node issuer. */ export class Issuer { + /** Blank-node identifier map owned by this issuer. */ + readonly ids = new Map() + /** Next numeric suffix allocated by this issuer. */ + #next = 0 + /** Gets or allocates a _:b identifier. */ issue(existing?: string) { + if (existing) { + const found = this.ids.get(existing) + if (found) return found + } + const value = `_:b${this.#next++}` + if (existing) this.ids.set(existing, value) + return value + } +} +/** Creates a complete node map from expanded JSON-LD. */ export function create( + expanded: JsonLdValueType, + issuer = new Issuer(), +): NodeMapType { + const maps: NodeMapType = new Map([['@default', new Map()]]) + visit(expanded, maps, '@default', undefined, undefined, issuer) + return maps +} +/** Flattens expanded JSON-LD into a deterministic top-level node list. */ export function flatten( + expanded: JsonLdValueType, +): JsonLdValueType[] { + const maps = create(expanded), defaultGraph = maps.get('@default')! + for (const [graph, nodes] of maps) { + if (graph === '@default') continue + let owner = defaultGraph.get(graph) + if (!owner) { + owner = { '@id': graph } + defaultGraph.set(graph, owner) + } + owner['@graph'] = [...nodes.values()].filter((v) => !identifierOnly(v)).sort(byId) + } + return [...defaultGraph.values()].filter((v) => !identifierOnly(v)).sort(byId) +} +/** Recursively creates/reuses nodes and relationships. */ function visit( + value: JsonLdValueType, + maps: NodeMapType, + graph: string, + activeSubject: string | undefined, + activeProperty: string | undefined, + issuer: Issuer, + list?: JsonLdValueType[], +): string | undefined { + if (Array.isArray(value)) { + for (const item of value) visit(item, maps, graph, activeSubject, activeProperty, issuer, list) + return + } + if (!object(value)) return + if (valueObject(value)) { + if (list) list.push(value) + else if (activeSubject && activeProperty) { + addNode(maps, graph, activeSubject, activeProperty, value) + } + return + } + if (listObject(value)) { + const output: JsonLdValueType[] = [] + for (const item of array(value['@list'] ?? [])) { + visit(item, maps, graph, activeSubject, activeProperty, issuer, output) + } + const obj: { + /** Allows additional keyed values required by obj. */ + [key: string]: JsonLdValueType + } = { '@list': output } + if (typeof value['@index'] === 'string') obj['@index'] = value['@index'] + if (list) list.push(obj) + else if (activeSubject && activeProperty) { + addNode(maps, graph, activeSubject, activeProperty, obj) + } + return + } + let id = typeof value['@id'] === 'string' ? value['@id'] : issuer.issue() + if (id.startsWith('_:')) id = issuer.issue(id) + const node = getNode(maps, graph, id) + for (const [property, raw] of Object.entries(value).sort(([a], [b]) => compare(a, b))) { + if (property === '@id') continue + if (property === '@type') { + for (let type of array(raw)) { + if (typeof type !== 'string') continue + if (type.startsWith('_:')) type = issuer.issue(type) + add(node, '@type', type) + } + continue + } + if (property === '@index') { + if (!Object.hasOwn(node, '@index')) node['@index'] = raw + continue + } + if (property === '@reverse' && object(raw)) { + for (const [reverseProperty, items] of Object.entries(raw)) { + for (const item of array(items)) { + const reverseId = visit(item, maps, graph, undefined, undefined, issuer) + if (reverseId) addNode(maps, graph, reverseId, reverseProperty, { '@id': id }) + } + } + continue + } + if (property === '@graph') { + visit(raw, maps, id, undefined, undefined, issuer) + continue + } + if (property === '@included') { + visit(raw, maps, graph, undefined, undefined, issuer) + continue + } + if (property.startsWith('@')) { + node[property] = raw + continue + } + for (const item of array(raw)) { + if (valueObject(item) || listObject(item)) visit(item, maps, graph, id, property, issuer) + else { + const objectId = visit(item, maps, graph, undefined, undefined, issuer) + if (objectId) add(node, property, { '@id': objectId }) + } + } + } + if (activeSubject && activeProperty) { + addNode(maps, graph, activeSubject, activeProperty, { '@id': id }) + } + if (list) list.push({ '@id': id }) + return id +} +/** Gets/creates one graph node. */ function getNode( + maps: NodeMapType, + graph: string, + id: string, +): NodeType { + let nodes = maps.get(graph) + if (!nodes) { + nodes = new Map() + maps.set(graph, nodes) + } + let node = nodes.get(id) + if (!node) { + node = { '@id': id } + nodes.set(id, node) + } + return node +} +/** Adds one node property value preserving arrays and duplicate suppression. */ function add( + node: NodeType, + property: string, + value: JsonLdValueType, +) { + const current = node[property] + if (current === undefined) node[property] = [value] + else if ( + Array.isArray(current) && !current.some((v) => JSON.stringify(v) === JSON.stringify(value)) + ) current.push(value) +} +/** Adds one relationship to a graph node. */ function addNode( + maps: NodeMapType, + graph: string, + id: string, + property: string, + value: JsonLdValueType, +) { + add(getNode(maps, graph, id), property, value) +} +/** Tests whether a flattened node contains only its identifier. */ function identifierOnly( + node: NodeType, +) { + return Object.keys(node).length === 1 && typeof node['@id'] === 'string' +} +/** Sorts flattened nodes by identifier. */ function byId(a: NodeType, b: NodeType) { + return compare(String(a['@id'] ?? ''), String(b['@id'] ?? '')) +} diff --git a/packages/rdf/jsonld/rdf.ts b/packages/rdf/jsonld/rdf.ts new file mode 100644 index 0000000..c2c2ab2 --- /dev/null +++ b/packages/rdf/jsonld/rdf.ts @@ -0,0 +1,487 @@ +/** + * Direct JSON-LD expanded-form to RDF conversion and RDF deserialization. + * + * The converter works on native RDF terms. It does not serialize through + * N-Quads or delegate JSON-LD processing to another implementation. + * + * @module + */ + +import { blankNode, defaultGraph, literal, namedNode, quad } from '../factory.ts' +import { + type GraphTermType, + type ObjectTermType, + type Quad, + RDF, + type SubjectTermType, + XSD, +} from '../term.ts' +import { compare, object } from './context.ts' +import { listObject, valueObject } from './expand.ts' +import type { JsonLdValueType, RdfDirectionType } from './types.ts' + +/** JSON-LD i18n datatype namespace used by the `i18n-datatype` direction mapping. */ +const I18N = 'https://www.w3.org/ns/i18n#' +/** RDF namespace reconstructed from the canonical `rdf:type` IRI. */ +const RDF_NS = RDF.type.slice(0, RDF.type.lastIndexOf('#') + 1) +/** Predicate used by compound literals for their lexical value. */ +const RDF_VALUE = `${RDF_NS}value` +/** Predicate used by compound literals for their optional language. */ +const RDF_LANGUAGE = `${RDF_NS}language` +/** Predicate used by compound literals for their base direction. */ +const RDF_DIRECTION = `${RDF_NS}direction` +/** Datatype used by JSON-LD `@json` values. */ +const RDF_JSON = `${RDF_NS}JSON` + +/** Options for JSON-LD to RDF conversion. */ +export interface ToRdfOptionsType { + /** Permit blank-node predicates in generalized RDF output. */ + readonly produceGeneralizedRdf?: boolean + /** Mapping used when an expanded value has `@direction`. */ + readonly rdfDirection?: RdfDirectionType + /** Caller cancellation checked between node and list operations. */ + readonly signal?: AbortSignal +} + +/** Options for RDF to JSON-LD conversion. */ +export interface FromRdfOptionsType { + /** Directional string mapping to recognize while reading RDF literals. */ + readonly rdfDirection?: RdfDirectionType + /** Convert recognized XSD boolean and numeric lexical forms to JSON scalars. */ + readonly useNativeTypes?: boolean + /** Preserve `rdf:type` as an ordinary property instead of emitting `@type`. */ + readonly useRdfType?: boolean + /** Caller cancellation checked between quads and list reconstruction steps. */ + readonly signal?: AbortSignal +} + +/** + * Converts expanded JSON-LD directly to native RDF quads. + * + * JSON-LD lists are emitted as RDF collections. Directional strings use the + * mapping selected by `rdfDirection`; without a mapping, the direction is not + * representable in RDF 1.1 and is therefore omitted from the resulting + * literal, as required by JSON-LD's RDF conversion model. + */ +export function toRdf(expanded: JsonLdValueType, options: ToRdfOptionsType = {}): Quad[] { + const output: Quad[] = [] + const ids = new Map>() + let sequence = 0 + + /** Returns one stable native blank node for a JSON-LD blank-node identifier. */ + const getBlank = (id: string) => { + const key = id.startsWith('_:') ? id.slice(2) : id + let value = ids.get(key) + if (value === undefined) { + value = blankNode(key) + ids.set(key, value) + } + return value + } + + /** Converts an expanded identifier to a native RDF subject. */ + const getSubject = (id: string): SubjectTermType => + id.startsWith('_:') ? getBlank(id) : namedNode(id) + + /** Converts an expanded graph identifier to a native RDF graph term. */ + const getGraph = (id: string): GraphTermType => { + if (id === '@default') return defaultGraph() + return id.startsWith('_:') ? getBlank(id) : namedNode(id) + } + + /** Adds one quad to the output using the current graph. */ + const addQuad = ( + subject: SubjectTermType, + predicate: string, + value: ObjectTermType, + graph: GraphTermType, + ) => { + output.push(quad(subject, namedNode(predicate), value, graph)) + } + + /** Converts an expanded object or value into an RDF object term. */ + const getObject = (value: JsonLdValueType, graph: GraphTermType): ObjectTermType | undefined => { + if (!object(value)) return undefined + + if (typeof value['@id'] === 'string') return getSubject(value['@id']) + + if (listObject(value)) { + const values = asArray(value['@list'] ?? []) + if (values.length === 0) return namedNode(RDF.nil) + + const head = blankNode(`l${sequence++}`) + let cursor = head + + for (let index = 0; index < values.length; index++) { + abort(options.signal) + const item = getObject(values[index]!, graph) + if (item === undefined) continue + + addQuad(cursor, RDF.first, item, graph) + const last = index === values.length - 1 + const next = last ? namedNode(RDF.nil) : blankNode(`l${sequence++}`) + addQuad(cursor, RDF.rest, next, graph) + if (!last && next.termType === 'BlankNode') cursor = next + } + + return head + } + + if (!valueObject(value)) return undefined + + const raw = value['@value'] + if (value['@type'] === '@json') return literal(JSON.stringify(raw), namedNode(RDF_JSON)) + if (typeof raw !== 'string' && typeof raw !== 'number' && typeof raw !== 'boolean') { + return undefined + } + + const lexical = typeof raw === 'string' + ? raw + : typeof raw === 'boolean' + ? (raw ? 'true' : 'false') + : String(raw) + const language = typeof value['@language'] === 'string' ? value['@language'] : undefined + const direction = value['@direction'] === 'ltr' || value['@direction'] === 'rtl' + ? value['@direction'] + : undefined + + if (direction !== undefined && options.rdfDirection === 'i18n-datatype') { + return literal(lexical, namedNode(`${I18N}${language ?? ''}_${direction}`)) + } + + if (direction !== undefined && options.rdfDirection === 'compound-literal') { + const node = blankNode(`d${sequence++}`) + addQuad(node, RDF_VALUE, literal(lexical), graph) + if (language !== undefined) addQuad(node, RDF_LANGUAGE, literal(language), graph) + addQuad(node, RDF_DIRECTION, literal(direction), graph) + return node + } + + if (language !== undefined) return literal(lexical, language) + if (typeof value['@type'] === 'string') return literal(lexical, namedNode(value['@type'])) + if (typeof raw === 'boolean') return literal(lexical, namedNode(XSD.boolean)) + if (typeof raw === 'number') { + return Number.isInteger(raw) + ? literal(lexical, namedNode(XSD.integer)) + : literal(lexical, namedNode(XSD.double)) + } + return literal(lexical) + } + + /** Walks expanded node objects and emits their RDF statements. */ + const visit = (value: JsonLdValueType, graph: GraphTermType): void => { + abort(options.signal) + + if (Array.isArray(value)) { + for (const item of value) visit(item, graph) + return + } + + if (!object(value) || valueObject(value) || listObject(value)) return + + const identifier = typeof value['@id'] === 'string' ? value['@id'] : `_:n${sequence++}` + const subject = getSubject(identifier) + + for ( + const [property, raw] of Object.entries(value).sort(([left], [right]) => compare(left, right)) + ) { + if (property === '@graph') { + const target = typeof value['@id'] === 'string' ? getGraph(value['@id']) : graph + visit(raw, target) + continue + } + + if (property === '@included') { + visit(raw, graph) + continue + } + + if (property === '@type') { + for (const type of asArray(raw)) { + if (typeof type === 'string') addQuad(subject, RDF.type, getSubject(type), graph) + } + continue + } + + if (property === '@reverse' && object(raw)) { + for (const [predicate, items] of Object.entries(raw)) { + for (const item of asArray(items)) { + const reverseSubject = getObject(item, graph) + if ( + reverseSubject !== undefined && reverseSubject.termType !== 'Literal' && + reverseSubject.termType !== 'Quad' + ) { + addQuad(reverseSubject, predicate, subject, graph) + if (object(item) && !valueObject(item) && !listObject(item)) visit(item, graph) + } + } + } + continue + } + + if (property.startsWith('@')) continue + if (property.startsWith('_:') && !options.produceGeneralizedRdf) continue + + for (const item of asArray(raw)) { + const rdfObject = getObject(item, graph) + if (rdfObject !== undefined) addQuad(subject, property, rdfObject, graph) + if (object(item) && !valueObject(item) && !listObject(item)) visit(item, graph) + } + } + } + + visit(expanded, defaultGraph()) + return output +} + +/** + * Converts native RDF quads to expanded JSON-LD. + * + * Well-formed `rdf:first`/`rdf:rest` chains are reconstructed as JSON-LD + * `@list` objects. Named graphs are represented with `@graph`. RDF 1.2 triple + * terms are rejected because JSON-LD 1.1 does not define a lossless mapping + * for them. + */ +export function fromRdf( + source: Iterable, + options: FromRdfOptionsType = {}, +): JsonLdValueType[] { + const graphs = new Map>>() + + /** Returns or creates one expanded node in one graph. */ + const getNode = (graph: string, id: string) => { + let map = graphs.get(graph) + if (map === undefined) { + map = new Map() + graphs.set(graph, map) + } + + let value = map.get(id) + if (value === undefined) { + value = { '@id': id } + map.set(id, value) + } + return value + } + + /** Converts an RDF subject or graph term to an expanded identifier. */ + const getId = (term: SubjectTermType | GraphTermType) => { + if (term.termType === 'BlankNode') return `_:${term.value}` + if (term.termType === 'DefaultGraph') return '@default' + return term.value + } + + /** Converts one RDF object term to expanded JSON-LD. */ + const getValue = (term: ObjectTermType): JsonLdValueType => { + if (term.termType === 'NamedNode') return { '@id': term.value } + if (term.termType === 'BlankNode') return { '@id': `_:${term.value}` } + if (term.termType === 'Quad') { + throw new TypeError('JSON-LD 1.1 does not define an RDF triple-term mapping.') + } + + const datatype = term.datatype.value + if (options.rdfDirection === 'i18n-datatype' && datatype.startsWith(I18N)) { + const suffix = datatype.slice(I18N.length) + const split = suffix.lastIndexOf('_') + if (split >= 0) { + const language = suffix.slice(0, split) + const direction = suffix.slice(split + 1) + const result: Record = { + '@value': term.value, + '@direction': direction, + } + if (language) result['@language'] = language + return result + } + } + + if (datatype === RDF_JSON) { + try { + return { '@value': JSON.parse(term.value) as JsonLdValueType, '@type': '@json' } + } catch { + return { '@value': term.value, '@type': '@json' } + } + } + + let value: JsonLdValueType = term.value + if (options.useNativeTypes) { + if (datatype === XSD.boolean && (term.value === 'true' || term.value === 'false')) { + value = term.value === 'true' + } else if (datatype === XSD.integer && /^[+-]?\d+$/u.test(term.value)) { + value = Number(term.value) + } else if ( + (datatype === XSD.double || datatype === XSD.decimal) && Number.isFinite(Number(term.value)) + ) { + value = Number(term.value) + } + } + + const result: Record = { '@value': value } + if (term.language) result['@language'] = term.language + else if (datatype !== XSD.string) result['@type'] = datatype + return result + } + + for (const statement of source) { + abort(options.signal) + const graphId = getId(statement.graph) + const subjectId = getId(statement.subject) + const record = getNode(graphId, subjectId) + const predicate = statement.predicate.value + + if (predicate === RDF.type && !options.useRdfType) { + if (statement.object.termType === 'NamedNode') add(record, '@type', statement.object.value) + else if (statement.object.termType === 'BlankNode') { + add(record, '@type', `_:${statement.object.value}`) + } + continue + } + + add(record, predicate, getValue(statement.object)) + if (statement.object.termType === 'BlankNode') getNode(graphId, `_:${statement.object.value}`) + } + + if (options.rdfDirection === 'compound-literal') collapseDirections(graphs) + + for (const [graphId, map] of [...graphs]) { + collapseLists(map) + if (graphId === '@default') continue + + let defaultMap = graphs.get('@default') + if (defaultMap === undefined) { + defaultMap = new Map() + graphs.set('@default', defaultMap) + } + + let owner = defaultMap.get(graphId) + if (owner === undefined) { + owner = { '@id': graphId } + defaultMap.set(graphId, owner) + } + owner['@graph'] = [...map.values()].filter((value) => !isListCell(value)) + } + + return [...(graphs.get('@default')?.values() ?? [])] + .filter((value) => !isListCell(value)) + .sort((left, right) => compare(String(left['@id'] ?? ''), String(right['@id'] ?? ''))) +} + +/** Replaces well-formed RDF collection heads with JSON-LD `@list` objects. */ +function collapseLists(map: Map>) { + const heads = new Map() + + for (const [id, node] of map) { + const values = asArray(node[RDF.first] ?? []) + const rests = asArray(node[RDF.rest] ?? []) + if (values.length !== 1 || rests.length !== 1) continue + + const list: JsonLdValueType[] = [] + let current = id + const seen = new Set() + let valid = true + + while (current !== RDF.nil) { + if (seen.has(current)) { + valid = false + break + } + seen.add(current) + + const cell = map.get(current) + const first = asArray(cell?.[RDF.first] ?? []) + const rest = asArray(cell?.[RDF.rest] ?? []) + if ( + first.length !== 1 || rest.length !== 1 || !object(rest[0]) || + typeof rest[0]['@id'] !== 'string' + ) { + valid = false + break + } + + list.push(first[0]!) + current = rest[0]['@id'] + } + + if (valid) heads.set(id, { '@list': list }) + } + + if (heads.size === 0) return + + for (const node of map.values()) { + for (const [property, raw] of Object.entries(node)) { + if (property === RDF.first || property === RDF.rest) continue + node[property] = asArray(raw).map((item) => { + if (!object(item) || typeof item['@id'] !== 'string') return item + return heads.get(item['@id']) ?? item + }) + } + } +} + +/** + * Replaces JSON-LD compound-literal helper nodes with directional value + * objects before ordinary list reconstruction. + */ +function collapseDirections(graphs: Map>>) { + for (const map of graphs.values()) { + const directions = new Map() + + for (const [id, node] of map) { + const values = asArray(node[RDF_VALUE] ?? []) + const languages = asArray(node[RDF_LANGUAGE] ?? []) + const directionsRaw = asArray(node[RDF_DIRECTION] ?? []) + if (values.length !== 1 || directionsRaw.length !== 1) continue + if (!valueObject(values[0]!) || !valueObject(directionsRaw[0]!)) continue + + const valueNode = values[0] as Record + const directionNode = directionsRaw[0] as Record + const lexical = valueNode['@value'] + const direction = directionNode['@value'] + if (typeof lexical !== 'string' || (direction !== 'ltr' && direction !== 'rtl')) continue + + const result: Record = { '@value': lexical, '@direction': direction } + if (languages.length === 1 && valueObject(languages[0]!)) { + const languageNode = languages[0] as Record + const language = languageNode['@value'] + if (typeof language === 'string') result['@language'] = language + } + directions.set(id, result) + } + + if (directions.size === 0) continue + for (const node of map.values()) { + for (const [property, raw] of Object.entries(node)) { + if (property === RDF_VALUE || property === RDF_LANGUAGE || property === RDF_DIRECTION) { + continue + } + node[property] = asArray(raw).map((item) => { + if (!object(item) || typeof item['@id'] !== 'string') return item + return directions.get(item['@id']) ?? item + }) + } + } + } +} + +/** Tests whether an expanded node is an internal RDF collection cell. */ +function isListCell(node: Record) { + return Object.hasOwn(node, RDF.first) && Object.hasOwn(node, RDF.rest) +} + +/** Adds one expanded value while preserving JSON-LD's array-valued property form. */ +function add(target: Record, property: string, value: JsonLdValueType) { + const current = target[property] + if (current === undefined) target[property] = [value] + else if (Array.isArray(current)) current.push(value) + else target[property] = [current, value] +} + +/** Returns a scalar or array JSON-LD value in array form. */ +function asArray(value: JsonLdValueType): JsonLdValueType[] { + return Array.isArray(value) ? value : [value] +} + +/** Throws the caller's abort reason before additional conversion work starts. */ +function abort(signal?: AbortSignal) { + if (signal?.aborted) throw signal.reason ?? new DOMException('Aborted', 'AbortError') +} diff --git a/packages/rdf/jsonld/types.ts b/packages/rdf/jsonld/types.ts index 1f18c5c..c72be69 100644 --- a/packages/rdf/jsonld/types.ts +++ b/packages/rdf/jsonld/types.ts @@ -1,27 +1,26 @@ -/** JSON-LD processor and remote-document contracts. @module */ - -/** JSON-compatible JSON-LD input/output value. */ -export type JsonLdValueType = null | boolean | number | string | JsonLdValueType[] | { [key: string]: JsonLdValueType } - -/** Remote document shape required by the JSON-LD processing algorithms. */ -export interface RemoteDocumentType { +/** JSON-LD 1.1 native processor contracts. @module */ +/** JSON-compatible JSON-LD input/output value. */ export type JsonLdValueType = + | null + | boolean + | number + | string + | JsonLdValueType[] + | { + /** Allows arbitrary JSON object keys whose values remain JSON-LD-compatible values. */ + [key: string]: JsonLdValueType + } +/** JSON-LD processing mode. */ export type ProcessingModeType = 'json-ld-1.0' | 'json-ld-1.1' +/** RDF direction mapping accepted by RDF conversion. */ export type RdfDirectionType = + | 'i18n-datatype' + | 'compound-literal' +/** JSON-LD framing embed mode. */ export type EmbedType = '@always' | '@once' | '@never' | boolean +/** Remote document returned by a JSON-LD loader. */ export interface RemoteDocumentType { + /** External JSON-LD context URL obtained from HTTP metadata, or null when absent. */ readonly contextUrl: string | null - readonly documentUrl: string - readonly document: JsonLdValueType + /** Final document URL. */ readonly documentUrl: string + /** Parsed JSON or HTML source text. */ readonly document: JsonLdValueType } - -/** JSON-LD document loader compatible with jsonld.js. */ -export type DocumentLoaderType = ( +/** Caller-provided remote document loader. */ export type DocumentLoaderType = ( url: string, options?: Readonly>, ) => Promise - -/** Minimal processing API used by `@okikio/rdf/jsonld`. */ -export interface ProcessorType { - expand(input: unknown, options?: Readonly>): Promise - compact(input: unknown, context: unknown, options?: Readonly>): Promise - flatten(input: unknown, context?: unknown, options?: Readonly>): Promise - frame(input: unknown, frame: unknown, options?: Readonly>): Promise - toRDF(input: unknown, options?: Readonly>): Promise - fromRDF(input: unknown, options?: Readonly>): Promise -} diff --git a/packages/rdf/line.ts b/packages/rdf/line.ts index 1011d95..a1ff816 100644 --- a/packages/rdf/line.ts +++ b/packages/rdf/line.ts @@ -1,42 +1,78 @@ /** Shared streaming scanner for RDF 1.2 N-Triples and N-Quads. @module */ import { blankNode, defaultGraph, literal, namedNode, quad } from './factory.ts' -import type { Graph, ObjectTerm, Predicate, Quad, Subject } from './term.ts' -import { chunks, throwIfAborted, type TextSource } from './text.ts' +import type { + GraphTermType, + ObjectTermType, + PredicateTermType, + Quad, + SubjectTermType, +} from './term.ts' +import { chunks, type TextSourceType, throwIfAborted } from './text.ts' -export type { TextSource } from './text.ts' +export type { TextSourceType } from './text.ts' /** Source range expressed in UTF-16 code-unit offsets and one-based line/column positions. */ -export interface SourceRange { +export interface SourceRangeType { + /** Zero-based source offset where this record starts. */ readonly start: number + /** Exclusive zero-based source offset where this record ends. */ readonly end: number + /** One-based source line containing the start of this record. */ readonly line: number + /** One-based source column containing the start of this record. */ readonly column: number } /** Recoverable parser diagnostic. */ -export interface Diagnostic { +export interface DiagnosticType { + /** Stable machine-readable code used to classify this diagnostic or failure. */ readonly code: string + /** Human-readable explanation of the diagnostic or failure. */ readonly message: string - readonly range: SourceRange + /** Source range that locates the related token, statement, feature, or diagnostic. */ + readonly range: SourceRangeType } /** Parser controls for hostile input and tolerant analysis. */ -export interface ParseOptions { +export interface ParseOptionsType { + /** When true, recoverable line-syntax defects become diagnostics instead of immediate failures. */ readonly tolerant?: boolean + /** Maximum characters accepted for one logical RDF line before parsing reports a limit. */ readonly maxLineLength?: number + /** Maximum nested RDF 1.2 triple-term depth accepted before parsing reports a limit. */ readonly maxTripleDepth?: number + /** Caller-owned abort signal checked before expensive work and between long-running steps. */ readonly signal?: AbortSignal } /** RDF 1.2 version labels understood by the line syntaxes. */ -export type RdfVersion = '1.1' | '1.2-basic' | '1.2' +export type RdfVersionType = '1.1' | '1.2-basic' | '1.2' /** Event stream emitted by N-Triples/N-Quads analysis. */ -export type ParseEvent = - | { readonly kind: 'quad'; readonly quad: Quad; readonly range: SourceRange } - | { readonly kind: 'version'; readonly version: RdfVersion; readonly range: SourceRange } - | { readonly kind: 'diagnostic'; readonly diagnostic: Diagnostic } +export type ParseEventType = + | { + /** Selects the `quad` variant of ParseEventType. */ + readonly kind: 'quad' + /** RDF quad carried by this event or triplestore mutation. */ + readonly quad: Quad + /** Source range covered by this ParseEventType. */ + readonly range: SourceRangeType + } + | { + /** Selects the `version` variant of ParseEventType. */ + readonly kind: 'version' + /** Version marker retained by this syntax record. */ + readonly version: RdfVersionType + /** Source range covered by this ParseEventType. */ + readonly range: SourceRangeType + } + | { + /** Selects the `diagnostic` variant of ParseEventType. */ + readonly kind: 'diagnostic' + /** Structured syntax diagnostic emitted by this parser event. */ + readonly diagnostic: DiagnosticType + } /** Default max line length used when the caller does not provide an override. */ const DEFAULT_MAX_LINE_LENGTH = 8 * 1024 * 1024 @@ -44,20 +80,24 @@ const DEFAULT_MAX_LINE_LENGTH = 8 * 1024 * 1024 const DEFAULT_MAX_TRIPLE_DEPTH = 64 /** Internal line record retaining absolute source offsets. */ -interface LineRecord { +interface LineRecordType { /** Decoded source window that contains this logical line. */ readonly source: string /** Inclusive line start within `source`. */ readonly from: number /** Exclusive line end within `source`. */ readonly to: number + /** One-based source line containing the start of this record. */ readonly line: number /** Absolute document offset corresponding to `from`. */ readonly start: number } /** Incrementally yields logical lines and cancels a Web Stream on early return. */ -export async function* lines(source: TextSource, options: ParseOptions = {}): AsyncGenerator { +export async function* lines( + source: TextSourceType, + options: ParseOptionsType = {}, +): AsyncGenerator { const maxLineLength = options.maxLineLength ?? DEFAULT_MAX_LINE_LENGTH const decoder = new TextDecoder() let buffered = '' @@ -66,12 +106,14 @@ export async function* lines(source: TextSource, options: ParseOptions = {}): As let line = 1 /** Emits every complete line currently available without repeatedly slicing the unconsumed suffix. */ - function* drain(final: boolean): Generator { + function* drain(final: boolean): Generator { while (true) { const boundary = nextLineBreak(buffered, index, final) if (!boundary) return const length = boundary.index - index - if (length > maxLineLength) throw new SyntaxError(`RDF line exceeds maxLineLength (${maxLineLength}).`) + if (length > maxLineLength) { + throw new SyntaxError(`RDF line exceeds maxLineLength (${maxLineLength}).`) + } yield { source: buffered, from: index, to: boundary.index, line, start: base + index } index = boundary.index + boundary.length line++ @@ -96,13 +138,28 @@ export async function* lines(source: TextSource, options: ParseOptions = {}): As buffered += decoder.decode() yield* drain(true) const remaining = buffered.length - index - if (remaining > maxLineLength) throw new SyntaxError(`RDF line exceeds maxLineLength (${maxLineLength}).`) - if (remaining > 0) yield { source: buffered, from: index, to: buffered.length, line, start: base + index } + if (remaining > maxLineLength) { + throw new SyntaxError(`RDF line exceeds maxLineLength (${maxLineLength}).`) + } + if (remaining > 0) { + yield { source: buffered, from: index, to: buffered.length, line, start: base + index } + } } /** Parses one N-Triples/N-Quads line into a semantic event. */ -export function parseLine(record: LineRecord, allowGraph: boolean, options: ParseOptions = {}): ParseEvent | undefined { - const cursor = new Cursor(record.source, record.from, record.to, record.start, record.line, options.maxTripleDepth ?? DEFAULT_MAX_TRIPLE_DEPTH) +export function parseLine( + record: LineRecordType, + allowGraph: boolean, + options: ParseOptionsType = {}, +): ParseEventType | undefined { + const cursor = new Cursor( + record.source, + record.from, + record.to, + record.start, + record.line, + options.maxTripleDepth ?? DEFAULT_MAX_TRIPLE_DEPTH, + ) cursor.space() if (cursor.done || cursor.peek() === '#') return undefined @@ -124,12 +181,12 @@ export function parseLine(record: LineRecord, allowGraph: boolean, options: Pars const subject = cursor.subject() cursor.requiredSpace('Expected whitespace after RDF subject.') - const predicate = cursor.iri() as Predicate + const predicate = cursor.iri() as PredicateTermType cursor.requiredSpace('Expected whitespace after RDF predicate.') const object = cursor.object(0) cursor.space() - let graph: Graph = defaultGraph() + let graph: GraphTermType = defaultGraph() if (allowGraph && cursor.peek() !== '.') { graph = cursor.graph() cursor.space() @@ -146,22 +203,31 @@ export function parseLine(record: LineRecord, allowGraph: boolean, options: Pars } /** Converts a parser exception into a source-ranged diagnostic. */ -export function diagnostic(error: unknown, record: LineRecord): Diagnostic { - if (error instanceof ParseError) return { code: error.code, message: error.message, range: error.range } +export function diagnostic(error: unknown, record: LineRecordType): DiagnosticType { + if (error instanceof ParseError) { + return { code: error.code, message: error.message, range: error.range } + } return { code: 'rdf-syntax', message: error instanceof Error ? error.message : String(error), - range: { start: record.start, end: record.start + record.to - record.from, line: record.line, column: 1 }, + range: { + start: record.start, + end: record.start + record.to - record.from, + line: record.line, + column: 1, + }, } } /** Position-aware parser error used internally and surfaced as diagnostics in tolerant mode. */ class ParseError extends SyntaxError { + /** Stable machine-readable code used to classify this diagnostic or failure. */ readonly code: string - readonly range: SourceRange + /** Source range that locates the related token, statement, feature, or diagnostic. */ + readonly range: SourceRangeType /** Creates one source-ranged line-syntax failure for strict throwing or tolerant diagnostic conversion. */ - constructor(code: string, message: string, range: SourceRange) { + constructor(code: string, message: string, range: SourceRangeType) { super(message) this.name = 'RdfParseError' this.code = code @@ -171,16 +237,30 @@ class ParseError extends SyntaxError { /** Data-oriented cursor over one line. It emits RDF terms directly and builds no token objects. */ class Cursor { + /** Current lookup or cursor index used to avoid rescanning already consumed state. */ #index: number + /** Logical RDF line text inspected by this cursor. */ readonly source: string + /** Inclusive index of the first character in this cursor slice. */ readonly from: number + /** Exclusive index after the last character in this cursor slice. */ readonly to: number + /** Absolute source offset corresponding to the cursor slice start. */ readonly sourceStart: number + /** One-based source line containing the start of this record. */ readonly line: number + /** Maximum nested RDF 1.2 triple-term depth allowed while this cursor parses a term. */ readonly maxTripleDepth: number /** Creates a cursor over one logical RDF line with absolute source offsets and a bounded RDF 1.2 triple depth. */ - constructor(source: string, from: number, to: number, sourceStart: number, line: number, maxTripleDepth: number) { + constructor( + source: string, + from: number, + to: number, + sourceStart: number, + line: number, + maxTripleDepth: number, + ) { this.source = source this.from = from this.to = to @@ -208,7 +288,9 @@ class Cursor { /** Consumes an exact lexical token only when it fits inside this logical line. */ take(value: string): boolean { - if (this.#index + value.length > this.to || !this.source.startsWith(value, this.#index)) return false + if (this.#index + value.length > this.to || !this.source.startsWith(value, this.#index)) { + return false + } this.#index += value.length return true } @@ -243,29 +325,35 @@ class Cursor { } /** Reads a legal line-format RDF subject: IRI, blank node, or RDF 1.2 triple term where permitted. */ - subject(): Subject { + subject(): SubjectTermType { if (this.peek() === '<') return this.iri() if (this.starts('_:')) return this.blank() throw this.error('rdf-subject', 'Expected IRI or blank node as RDF subject.') } /** Reads a legal N-Quads graph label without allowing the default graph token in source syntax. */ - graph(): Graph { + graph(): GraphTermType { if (this.peek() === '<') return this.iri() if (this.starts('_:')) return this.blank() throw this.error('rdf-graph', 'Expected IRI or blank node as RDF graph label.') } /** Reads an RDF object, including bounded RDF 1.2 nested triple terms. */ - object(depth: number): ObjectTerm { + object(depth: number): ObjectTermType { if (depth > this.maxTripleDepth) { - throw this.error('rdf-depth', `Triple term nesting exceeds maxTripleDepth (${this.maxTripleDepth}).`) + throw this.error( + 'rdf-depth', + `Triple term nesting exceeds maxTripleDepth (${this.maxTripleDepth}).`, + ) } if (this.starts('<<(')) return this.triple(depth + 1) if (this.peek() === '<') return this.iri() if (this.starts('_:')) return this.blank() if (this.peek() === '"') return this.literal() - throw this.error('rdf-object', 'Expected IRI, blank node, literal, or RDF 1.2 triple term as object.') + throw this.error( + 'rdf-object', + 'Expected IRI, blank node, literal, or RDF 1.2 triple term as object.', + ) } /** Decodes one `` while rejecting forbidden characters and invalid Unicode escapes. */ @@ -284,7 +372,11 @@ class Cursor { continue } if (char <= ' ' || /[<>"{}|^`]/.test(char)) { - throw this.errorAt('rdf-iri-char', `Invalid character in IRI at column ${this.#index - this.from + 1}.`, start) + throw this.errorAt( + 'rdf-iri-char', + `Invalid character in IRI at column ${this.#index - this.from + 1}.`, + start, + ) } value += char this.#index++ @@ -350,7 +442,10 @@ class Cursor { } if (char === '\\') { const escaped = this.peek(1) - if (escaped === 't' || escaped === 'b' || escaped === 'n' || escaped === 'r' || escaped === 'f' || escaped === '"' || escaped === "'" || escaped === '\\') { + if ( + escaped === 't' || escaped === 'b' || escaped === 'n' || escaped === 'r' || + escaped === 'f' || escaped === '"' || escaped === "'" || escaped === '\\' + ) { this.#index += 2 value += escapeValue(escaped) continue @@ -358,7 +453,13 @@ class Cursor { value += this.unicodeEscape() continue } - if (char === '\n' || char === '\r') throw this.errorAt('rdf-string-line', 'Line break is not allowed in an N-Triples literal.', start) + if (char === '\n' || char === '\r') { + throw this.errorAt( + 'rdf-string-line', + 'Line break is not allowed in an N-Triples literal.', + start, + ) + } value += char this.#index++ } @@ -372,11 +473,13 @@ class Cursor { this.space() const subject = this.subject() this.requiredSpace('Expected whitespace in triple term after subject.') - const predicate = this.iri() as Predicate + const predicate = this.iri() as PredicateTermType this.requiredSpace('Expected whitespace in triple term after predicate.') const object = this.object(depth) this.space() - if (!this.take(')>>')) throw this.errorAt('rdf-triple-end', "Expected ')>>' after triple term.", start) + if (!this.take(')>>')) { + throw this.errorAt('rdf-triple-end', "Expected ')>>' after triple term.", start) + } return quad(subject, predicate, object) } @@ -385,7 +488,9 @@ class Cursor { const start = this.#index this.#index++ const kind = this.peek() - if (kind !== 'u' && kind !== 'U') throw this.errorAt('rdf-escape', 'Expected Unicode escape.', start) + if (kind !== 'u' && kind !== 'U') { + throw this.errorAt('rdf-escape', 'Expected Unicode escape.', start) + } this.#index++ const width = kind === 'u' ? 4 : 8 const text = this.source.slice(this.#index, Math.min(this.#index + width, this.to)) @@ -410,7 +515,7 @@ class Cursor { } /** Converts absolute offsets into a one-line source range with the correct one-based column. */ - range(start: number, end: number): SourceRange { + range(start: number, end: number): SourceRangeType { return { start, end, line: this.line, column: start - this.sourceStart + 1 } } @@ -435,7 +540,12 @@ function nextLineBreak( value: string, start: number, final: boolean, -): { readonly index: number; readonly length: number } | undefined { +): { + /** Offset of the next line-break sequence. */ + readonly index: number + /** Number of source code units occupied by the line-break sequence. */ + readonly length: number +} | undefined { const lf = value.indexOf('\n', start) const cr = value.indexOf('\r', start) if (lf === -1 && cr === -1) return undefined @@ -449,14 +559,23 @@ function nextLineBreak( /** Maps an N-Triples single-character escape to its decoded code point. */ function escapeValue(value: string): string { switch (value) { - case 't': return '\t' - case 'b': return '\b' - case 'n': return '\n' - case 'r': return '\r' - case 'f': return '\f' - case '"': return '"' - case "'": return "'" - case '\\': return '\\' - default: return value + case 't': + return '\t' + case 'b': + return '\b' + case 'n': + return '\n' + case 'r': + return '\r' + case 'f': + return '\f' + case '"': + return '"' + case "'": + return "'" + case '\\': + return '\\' + default: + return value } } diff --git a/packages/rdf/markup.ts b/packages/rdf/markup.ts new file mode 100644 index 0000000..cc284d9 --- /dev/null +++ b/packages/rdf/markup.ts @@ -0,0 +1,604 @@ +/** Range-first dependency-free markup scanning shared by RDF/XML, RDFa, and Microdata. @module */ + +import { chunks, type TextSourceType, throwIfAborted } from './text.ts' + +/** Default maximum decoded markup bytes accepted by one parser operation. */ +const DEFAULT_MAX_BYTES = 16 * 1024 * 1024 +/** Default maximum number of element and text nodes accepted by one parser operation. */ +const DEFAULT_MAX_NODES = 250_000 +/** Default maximum element nesting depth accepted by one parser operation. */ +const DEFAULT_MAX_DEPTH = 512 + +/** HTML elements whose end tags are forbidden. */ +const HTML_VOID = new Set([ + 'area', + 'base', + 'br', + 'col', + 'embed', + 'hr', + 'img', + 'input', + 'link', + 'meta', + 'param', + 'source', + 'track', + 'wbr', +]) +/** Optional HTML end-tag rules needed by common RDFa and Microdata documents. */ +const HTML_CLOSE: Readonly>> = { + li: new Set(['li']), + dt: new Set(['dt', 'dd']), + dd: new Set(['dt', 'dd']), + p: new Set([ + 'address', + 'article', + 'aside', + 'blockquote', + 'div', + 'dl', + 'fieldset', + 'footer', + 'form', + 'h1', + 'h2', + 'h3', + 'h4', + 'h5', + 'h6', + 'header', + 'hr', + 'menu', + 'nav', + 'ol', + 'p', + 'pre', + 'section', + 'table', + 'ul', + ]), + rt: new Set(['rt', 'rp']), + rp: new Set(['rt', 'rp']), + option: new Set(['option', 'optgroup']), + thead: new Set(['tbody', 'tfoot']), + tbody: new Set(['tbody', 'tfoot']), + tr: new Set(['tr']), + th: new Set(['th', 'td']), + td: new Set(['th', 'td']), +} + +/** One decoded markup attribute plus its source range. */ +export interface MarkupAttributeType { + /** Attribute name as exposed to host-language processing. */ readonly name: string + /** Decoded attribute value. Boolean HTML attributes use an empty string. */ readonly value: + string + /** Inclusive UTF-16 source offset where the attribute begins. */ readonly start: number + /** Exclusive UTF-16 source offset after the attribute. */ readonly end: number +} +/** Text node in the bounded markup tree. */ +export interface MarkupTextType { + /** Stable node discriminator. */ readonly kind: 'text' + /** Decoded character data. */ readonly value: string + /** Inclusive source offset. */ readonly start: number + /** Exclusive source offset. */ readonly end: number +} +/** Element node in the bounded markup tree. */ +export interface MarkupElementType { + /** Stable node discriminator. */ readonly kind: 'element' + /** Element qualified name, lower-cased only in HTML mode. */ readonly name: string + /** Decoded attributes in source order. */ readonly attributes: readonly MarkupAttributeType[] + /** Child nodes in document order. */ readonly children: readonly MarkupNodeType[] + /** Inclusive start-tag offset. */ readonly start: number + /** Exclusive matching-end-tag or recovered document offset. */ readonly end: number + /** Parent element when one exists. */ readonly parent?: MarkupElementType +} +/** Any node represented by the internal markup tree. */ +export type MarkupNodeType = MarkupElementType | MarkupTextType +/** Result of bounded markup parsing. */ +export interface MarkupDocumentType { + /** Top-level nodes in source order. */ readonly children: readonly MarkupNodeType[] + /** Decoded source retained for ranges and XML literals. */ readonly text: string +} +/** Parser limits and host-language mode. */ +export interface MarkupOptionsType { + /** Enables HTML case folding, void elements, and selected recovery. */ readonly html?: boolean + /** Maximum decoded UTF-8 bytes. */ readonly maxBytes?: number + /** Maximum total element/text node count. */ readonly maxNodes?: number + /** Maximum open element nesting depth. */ readonly maxDepth?: number + /** Caller-owned cancellation signal. */ readonly signal?: AbortSignal +} +/** Source-backed lexical token emitted by the scanner. */ +export type MarkupTokenType = + | { + /** Selects the `text` variant of MarkupTokenType. */ + readonly kind: 'text' + /** Zero-based source offset where this MarkupTokenType begins. */ + readonly start: number + /** Exclusive zero-based source offset where this MarkupTokenType ends. */ + readonly end: number + } + | { + /** Discriminates the concrete MarkupTokenType variant. */ + readonly kind: 'comment' | 'instruction' | 'declaration' + /** Zero-based source offset where this MarkupTokenType begins. */ + readonly start: number + /** Exclusive zero-based source offset where this MarkupTokenType ends. */ + readonly end: number + } + | { + /** Selects the `cdata` variant of MarkupTokenType. */ + readonly kind: 'cdata' + /** Zero-based source offset where this MarkupTokenType begins. */ + readonly start: number + /** Exclusive zero-based source offset where this MarkupTokenType ends. */ + readonly end: number + /** Zero-based source offset where CDATA character content begins. */ + readonly contentStart: number + /** Exclusive source offset where CDATA character content ends. */ + readonly contentEnd: number + } + | { + /** Selects the `start` variant of MarkupTokenType. */ + readonly kind: 'start' + /** Zero-based source offset where this MarkupTokenType begins. */ + readonly start: number + /** Exclusive zero-based source offset where this MarkupTokenType ends. */ + readonly end: number + /** Zero-based source offset where the start-tag body begins after `<`. */ + readonly bodyStart: number + /** Exclusive source offset where the start-tag body ends before `>`. */ + readonly bodyEnd: number + } + | { + /** Selects the `end` variant of MarkupTokenType. */ + readonly kind: 'end' + /** Zero-based source offset where this MarkupTokenType begins. */ + readonly start: number + /** Exclusive zero-based source offset where this MarkupTokenType ends. */ + readonly end: number + /** Zero-based source offset where the end-tag name begins. */ + readonly nameStart: number + /** Exclusive source offset where the end-tag name ends. */ + readonly nameEnd: number + } +/** Structured event consumed by the tree builder and semantic parsers. */ +export type MarkupEventType = + | { + /** Selects the `text` variant of MarkupEventType. */ + readonly kind: 'text' + /** Decoded character data carried by this markup text event. */ + readonly value: string + /** Zero-based source offset where this MarkupEventType begins. */ + readonly start: number + /** Exclusive zero-based source offset where this MarkupEventType ends. */ + readonly end: number + } + | { + /** Selects the `start` variant of MarkupEventType. */ + readonly kind: 'start' + /** Decoded MarkupEventType name used by the parser state. */ + readonly name: string + /** Decoded attributes attached to this start-tag or open-element record. */ + readonly attributes: readonly MarkupAttributeType[] + /** Whether the start tag closes itself without entering the open-element stack. */ + readonly selfClosing: boolean + /** Zero-based source offset where this MarkupEventType begins. */ + readonly start: number + /** Exclusive zero-based source offset where this MarkupEventType ends. */ + readonly end: number + } + | { + /** Selects the `end` variant of MarkupEventType. */ + readonly kind: 'end' + /** Decoded MarkupEventType name used by the parser state. */ + readonly name: string + /** Zero-based source offset where this MarkupEventType begins. */ + readonly start: number + /** Exclusive zero-based source offset where this MarkupEventType ends. */ + readonly end: number + } + +/** + * Parses XML/HTML-like input through source ranges, structured events, then a bounded tree. + * + * This follows the event-first architecture used by `@okikio/wikitext`: scanning owns + * source offsets, semantic consumers do not need a browser DOM, and the tree is only + * one consumer of the structured event stream. + */ +export async function parseMarkup( + source: TextSourceType, + options: MarkupOptionsType = {}, +): Promise { + const maxBytes = positive(options.maxBytes ?? DEFAULT_MAX_BYTES, 'maxBytes') + const maxNodes = positive(options.maxNodes ?? DEFAULT_MAX_NODES, 'maxNodes') + const maxDepth = positive(options.maxDepth ?? DEFAULT_MAX_DEPTH, 'maxDepth') + const text = await collect(source, maxBytes, options.signal) + const config = { + html: options.html ?? false, + maxNodes, + maxDepth, + ...(options.signal ? { signal: options.signal } : {}), + } + return tree(text, events(text, config), config) +} +/** Returns the first attribute value. */ +export function attr(element: MarkupElementType, name: string, html = false): string | undefined { + const target = html ? name.toLowerCase() : name + return element.attributes.find((v) => (html ? v.name.toLowerCase() : v.name) === target)?.value +} +/** Returns whether an attribute exists even when its value is empty. */ +export function hasAttr(element: MarkupElementType, name: string, html = false): boolean { + const target = html ? name.toLowerCase() : name + return element.attributes.some((v) => (html ? v.name.toLowerCase() : v.name) === target) +} +/** Concatenates descendant character data in document order. */ +export function textContent(element: MarkupElementType): string { + return element.children.map((v) => v.kind === 'text' ? v.value : textContent(v)).join('') +} +/** Yields descendant elements in source order. */ +export function* elements( + root: MarkupElementType, + includeRoot = false, +): Generator { + if (includeRoot) yield root + for (const child of root.children) { + if (child.kind === 'element') { + yield child + yield* elements(child) + } + } +} + +/** Emits source-backed lexical markup tokens in one scan. */ +export function* tokenize( + text: string, + options: Pick = {}, +): Generator { + const html = options.html ?? false + let pos = 0 + let plain = 0 + const flush = function* (end: number) { + if (end > plain) yield { kind: 'text' as const, start: plain, end } + } + while (pos < text.length) { + throwIfAborted(options.signal) + const open = text.indexOf('<', pos) + if (open < 0) break + let token: MarkupTokenType | undefined + if (text.startsWith('', open + 4) + if (end < 0) { + if (!html) throw new SyntaxError('Unterminated markup comment.') + pos = open + 1 + continue + } + token = { kind: 'comment', start: open, end: end + 3 } + } else if (text.startsWith('', open + 9) + if (end < 0) throw new SyntaxError('Unterminated CDATA section.') + token = { kind: 'cdata', start: open, end: end + 3, contentStart: open + 9, contentEnd: end } + } else if (text.startsWith('', open + 2) + if (end < 0) throw new SyntaxError('Unterminated processing instruction.') + token = { kind: 'instruction', start: open, end: end + 2 } + } else if (text.startsWith(']/u.test(text[b] ?? '')) b++ + token = { kind: 'end', start: open, end: end + 1, nameStart: a, nameEnd: b } + } else if (text.startsWith(' { + for (const token of tokenize(text, options)) { + throwIfAborted(options.signal) + if (token.kind === 'text') { + yield { + kind: 'text', + value: entities(text.slice(token.start, token.end), options.html), + start: token.start, + end: token.end, + } + } else if (token.kind === 'cdata') { + yield { + kind: 'text', + value: text.slice(token.contentStart, token.contentEnd), + start: token.contentStart, + end: token.contentEnd, + } + } else if (token.kind === 'end') { + yield { + kind: 'end', + name: name(text.slice(token.nameStart, token.nameEnd), options.html), + start: token.start, + end: token.end, + } + } else if (token.kind === 'start') { + const parsed = start(text, token.bodyStart, token.bodyEnd, options.html) + yield { kind: 'start', ...parsed, start: token.start, end: token.end } + } + } +} +/** Mutable open element used only while the tree stack is active. */ +interface OpenType { + /** Selects the `element` variant of OpenType. */ + kind: 'element' + /** Decoded OpenType name used by the parser state. */ + name: string + /** Decoded attributes attached to this start-tag or open-element record. */ + attributes: readonly MarkupAttributeType[] + /** Child markup nodes accumulated while this element remains open. */ + children: MarkupNodeType[] + /** Zero-based source offset where this OpenType begins. */ + start: number + /** Exclusive zero-based source offset where this OpenType ends. */ + end: number + /** Open parent element used to restore parser stack context. */ + parent?: OpenType +} +/** Materializes structured events into the semantic consumer tree. */ +function tree(text: string, input: Iterable, options: { + /** Whether markup rules use HTML case-folding and recovery semantics. */ + html: boolean + /** Maximum markup nodes materialized into the parser consumer tree. */ + maxNodes: number + /** Maximum nested markup element depth admitted by the parser. */ + maxDepth: number + /** Abort signal checked before and during this operation. */ + signal?: AbortSignal +}): MarkupDocumentType { + const roots: MarkupNodeType[] = [] + const stack: OpenType[] = [] + let count = 0 + const append = (node: MarkupNodeType) => { + if (++count > options.maxNodes) { + throw new RangeError(`Markup input exceeds maxNodes (${options.maxNodes}).`) + } + const p = stack.at(-1) + ;(p ? p.children : roots).push(node) + } + for (const event of input) { + throwIfAborted(options.signal) + if (event.kind === 'text') { + if (!event.value) continue + const list = stack.at(-1)?.children ?? roots + const prev = list.at(-1) + if (prev?.kind === 'text') { + ;(prev as { + /** Accumulated character data for the previous adjacent text node. */ + value: string + /** Exclusive zero-based source offset where this tree ends. */ + end: number + }).value += event.value + ;(prev as { + /** Exclusive zero-based source offset where this tree ends. */ + end: number + }).end = event.end + } else append({ kind: 'text', value: event.value, start: event.start, end: event.end }) + continue + } + if (event.kind === 'start') { + if (options.html) { + const current = stack.at(-1)?.name + if (current && HTML_CLOSE[current]?.has(event.name)) { + const closed = stack.pop() + if (closed) closed.end = event.start + } + } + if (stack.length >= options.maxDepth) { + throw new RangeError(`Markup input exceeds maxDepth (${options.maxDepth}).`) + } + const parent = stack.at(-1) + const node: OpenType = { + kind: 'element', + name: event.name, + attributes: event.attributes, + children: [], + start: event.start, + end: event.end, + ...(parent ? { parent } : {}), + } + append(node as MarkupElementType) + if (!event.selfClosing && !(options.html && HTML_VOID.has(event.name))) stack.push(node) + continue + } + if (!options.html) { + const top = stack.at(-1) + if (!top || top.name !== event.name) { + throw new SyntaxError(`Mismatched XML end tag .`) + } + top.end = event.end + stack.pop() + } else {for (let i = stack.length - 1; i >= 0; i--) { + if (stack[i]!.name !== event.name) continue + for (let j = stack.length - 1; j >= i; j--) stack[j]!.end = event.end + stack.length = i + break + }} + } + if (!options.html && stack.length) { + throw new SyntaxError(`Unclosed XML element <${stack.at(-1)!.name}>.`) + } + for (const open of stack) open.end = text.length + return { children: roots, text } +} +/** Parses one start-tag body and preserves attribute ranges. */ +function start(text: string, from: number, to: number, html: boolean): { + /** Decoded start name used by the parser state. */ + name: string + /** Decoded attributes attached to this start-tag or open-element record. */ + attributes: MarkupAttributeType[] + /** Whether the start tag closes itself without entering the open-element stack. */ + selfClosing: boolean +} { + let i = from + while (/\s/u.test(text[i] ?? '')) i++ + const ns = i + while (i < to && !/[\s/>]/u.test(text[i] ?? '')) i++ + const element = name(text.slice(ns, i), html) + const attrs: MarkupAttributeType[] = [] + let selfClosing = false + while (i < to) { + while (/\s/u.test(text[i] ?? '')) i++ + if (text[i] === '/') { + selfClosing = true + i++ + continue + } + if (i >= to) break + const s = i + while (i < to && !/[\s=/>]/u.test(text[i] ?? '')) i++ + const raw = text.slice(s, i) + if (!raw) { + i++ + continue + } + while (/\s/u.test(text[i] ?? '')) i++ + let value = '' + if (text[i] === '=') { + i++ + while (/\s/u.test(text[i] ?? '')) i++ + const q = text[i] + if (q === '"' || q === "'") { + i++ + const vs = i + while (i < to && text[i] !== q) i++ + if (i >= to) throw new SyntaxError(`Unterminated quoted attribute ${raw}.`) + value = text.slice(vs, i) + i++ + } else { + const vs = i + while (i < to && !/[\s>]/u.test(text[i] ?? '')) i++ + value = text.slice(vs, i) + } + } + attrs.push({ name: name(raw, html), value: entities(value, html), start: s, end: i }) + } + return { name: element, attributes: attrs, selfClosing } +} +/** Finds a tag terminator while ignoring quoted greater-than characters. */ +function tagEnd(text: string, offset: number, html: boolean): number { + let q = '' + for (let i = offset; i < text.length; i++) { + const c = text[i]! + if (q) { + if (c === q) q = '' + continue + } + if (c === '"' || c === "'") { + q = c + continue + } + if (c === '>') return i + } + if (html) return -1 + throw new SyntaxError('Unterminated markup tag.') +} +/** Finds a declaration terminator while respecting quotes and internal subsets. */ +function declEnd(text: string, offset: number): number { + let q = '' + let depth = 0 + for (let i = offset; i < text.length; i++) { + const c = text[i]! + if (q) { + if (c === q) q = '' + continue + } + if (c === '"' || c === "'") q = c + else if (c === '[') depth++ + else if (c === ']') depth = Math.max(0, depth - 1) + else if (c === '>' && depth === 0) return i + } + throw new SyntaxError('Unterminated markup declaration.') +} +/** Applies host-language case normalization. */ function name( + value: string, + html: boolean, +): string { + return html ? value.toLowerCase() : value +} +/** Decodes XML entities plus a small deterministic HTML named-entity set. */ +function entities(value: string, html: boolean): string { + if (!value.includes('&')) return value + const named: Record = { + amp: '&', + lt: '<', + gt: '>', + quot: '"', + apos: "'", + ...(html ? { nbsp: '\u00a0', copy: '©', reg: '®' } : {}), + } + return value.replace(/&(#x[0-9a-f]+|#[0-9]+|[a-z][a-z0-9]+);/giu, (source, body: string) => { + if (body[0] === '#') { + const n = body[1]?.toLowerCase() === 'x' + ? Number.parseInt(body.slice(2), 16) + : Number.parseInt(body.slice(1), 10) + return Number.isInteger(n) && n > 0 && n <= 0x10ffff && !(n >= 0xd800 && n <= 0xdfff) + ? String.fromCodePoint(n) + : '\ufffd' + } + return named[body.toLowerCase()] ?? source + }) +} +/** Reads and decodes source chunks under one byte limit. */ +async function collect( + source: TextSourceType, + maxBytes: number, + signal?: AbortSignal, +): Promise { + const decoder = new TextDecoder('utf-8', { fatal: true }) + const encoder = new TextEncoder() + let bytes = 0 + let out = '' + for await (const chunk of chunks(source, signal)) { + throwIfAborted(signal) + if (typeof chunk === 'string') { + bytes += encoder.encode(chunk).byteLength + out += chunk + } else { + bytes += chunk.byteLength + out += decoder.decode(chunk, { stream: true }) + } + if (bytes > maxBytes) throw new RangeError(`Markup input exceeds maxBytes (${maxBytes}).`) + } + return out + decoder.decode() +} +/** Validates a positive parser limit. */ function positive(value: number, label: string): number { + if (!Number.isSafeInteger(value) || value <= 0) { + throw new RangeError(`${label} must be a positive safe integer.`) + } + return value +} diff --git a/packages/rdf/microdata/mod.ts b/packages/rdf/microdata/mod.ts index cd7737c..adf17a0 100644 --- a/packages/rdf/microdata/mod.ts +++ b/packages/rdf/microdata/mod.ts @@ -1,64 +1,387 @@ -/** HTML Microdata-to-RDF parsing behind native RDF and Web-oriented source contracts. @module */ - -import { factory } from '../factory.ts' -import type { Graph, Quad } from '../term.ts' -import { throwIfAborted, type TextSource } from '../text.ts' -import { parseTransform } from '../transform.ts' -import type { ParserConstructorType } from './types.ts' - -export type { ParserConstructorType, ParserType } from './types.ts' - -/** Vocabulary-registry entry used by the Microdata-to-RDF conversion algorithm. */ -export interface VocabularyType { - readonly properties?: Readonly>>> - readonly [key: string]: unknown -} - -/** Microdata vocabulary registry keyed by vocabulary IRI prefix. */ -export type VocabularyRegistryType = Readonly> - -/** Options for Microdata-to-RDF parsing. */ -export interface ParseOptionsType { +/** Native HTML Microdata-to-RDF conversion. @module */ +import { blankNode, defaultGraph, literal, namedNode, quad } from '../factory.ts' +import { + attr, + elements, + hasAttr, + type MarkupElementType, + parseMarkup, + textContent, +} from '../markup.ts' +import { + type GraphTermType, + type ObjectTermType, + type Quad, + RDF, + type SubjectTermType, + XSD, +} from '../term.ts' +import { type TextSourceType, throwIfAborted } from '../text.ts' +/** Microdata registry predicate used to connect a vocabulary to additional vocabulary behavior. */ +const USES_VOCABULARY = 'http://www.w3.org/ns/rdfa#usesVocabulary', + XSD_G_YEAR = 'http://www.w3.org/2001/XMLSchema#gYear', + XSD_G_YEAR_MONTH = 'http://www.w3.org/2001/XMLSchema#gYearMonth', + XSD_TIME = 'http://www.w3.org/2001/XMLSchema#time', + XSD_DURATION = 'http://www.w3.org/2001/XMLSchema#duration' +/** Built-in Microdata vocabulary registry used when the caller does not supply a replacement. */ +const DEFAULT_VOCABULARIES: VocabularyRegistryType = { + 'http://schema.org/': { properties: { additionalType: { subPropertyOf: RDF.type } } }, + 'https://schema.org/': { properties: { additionalType: { subPropertyOf: RDF.type } } }, + 'http://microformats.org/profile/hcard': {}, +} +/** Metadata that expands one vocabulary property into extra predicates. */ export interface VocabularyPropertyType { + /** Predicate or predicates emitted in addition to the Microdata property itself. */ + readonly subPropertyOf?: string | readonly string[] + /** Equivalent predicates receiving the same value. */ readonly equivalentProperty?: + | string + | readonly string[] + /** Extension metadata retained for custom registries. */ readonly [key: string]: unknown +} +/** One Microdata vocabulary registry entry. */ export interface VocabularyType { + /** Vocabulary-specific Microdata property rules keyed by `itemprop` token. */ + readonly properties?: Readonly> + /** Extension metadata retained for custom registries. */ readonly [key: string]: unknown +} +/** Microdata vocabulary registry keyed by vocabulary IRI prefix. */ export type VocabularyRegistryType = + Readonly> +/** Options for native Microdata-to-RDF parsing. */ export interface ParseOptionsType { + /** Effective document base IRI used to resolve Microdata identifiers and URL values. */ readonly base?: string - readonly graph?: Graph - /** Parse the input as strict XML/XHTML instead of HTML. */ - readonly xml?: boolean - /** Replaces the processor's standard Microdata vocabulary registry when supplied. */ - readonly vocabularies?: VocabularyRegistryType - /** External parser injection used by tests or alternate conforming implementations. */ - readonly parser?: ParserConstructorType - readonly signal?: AbortSignal -} - + /** Target RDF graph. */ readonly graph?: GraphTermType + /** Parse as XML/XHTML when true. */ readonly xml?: boolean + /** Replaces the default vocabulary registry. */ readonly vocabularies?: VocabularyRegistryType + /** Maximum decoded source bytes. */ readonly maxBytes?: number + /** Maximum markup node count. */ readonly maxNodes?: number + /** Maximum markup depth. */ readonly maxDepth?: number + /** Caller cancellation. */ readonly signal?: AbortSignal +} +/** Shared state for one conversion. */ interface StateType { + /** Effective document base IRI shared by this Microdata conversion operation. */ + readonly base?: string + /** Document subject for vocabulary-use statements. */ readonly document?: SubjectTermType + /** Target graph. */ readonly graph: GraphTermType + /** Vocabulary registry. */ readonly vocabularies: VocabularyRegistryType + /** itemref targets by id. */ readonly ids: ReadonlyMap + /** Stable subjects for item identity/cycles. */ readonly memory: Map< + MarkupElementType, + SubjectTermType + > + /** Vocabularies already reported. */ readonly used: Set + /** Deterministic result buffer. */ readonly quads: Quad[] + /** HTML attribute rules enabled. */ readonly html: boolean + /** Caller cancellation. */ readonly signal?: AbortSignal +} +/** Effective context inherited by nested items. */ interface ItemContextType { + /** Vocabulary type IRI that controls property expansion for the current Microdata item. */ + readonly type?: string + /** Derived vocabulary. */ readonly vocabulary?: string + /** Inherited language. */ readonly language?: string +} +/** One property-bearing element plus inherited language. */ interface PropertyElementType { + /** Markup element whose Microdata property value is being evaluated. */ + readonly element: MarkupElementType + /** Effective language. */ readonly language?: string +} /** - * Parses HTML Microdata incrementally into native `@okikio/rdf` quads. - * - * Conversion follows the W3C Microdata-to-RDF algorithm implemented by the - * external parser. The external Transform and HTML parser remain subpath-only. + * Parses Microdata and yields native RDF quads without a DOM or external parser. + * Item identity is memoized so repeated `itemref` and nested-item references reuse one subject. */ -export async function* parse(source: TextSource, options: ParseOptionsType = {}): AsyncGenerator { +export async function* parse( + source: TextSourceType, + options: ParseOptionsType = {}, +): AsyncGenerator { + const html = !(options.xml ?? false) + const document = await parseMarkup(source, { + html, + ...(options.maxBytes === undefined ? {} : { maxBytes: options.maxBytes }), + ...(options.maxNodes === undefined ? {} : { maxNodes: options.maxNodes }), + ...(options.maxDepth === undefined ? {} : { maxDepth: options.maxDepth }), + ...(options.signal ? { signal: options.signal } : {}), + }) throwIfAborted(options.signal) - const Parser = options.parser ?? await defaultParser() - const parser = new Parser(parserOptions(options)) - yield* parseTransform(parser, source, { label: 'Microdata parser', ...(options.signal ? { signal: options.signal } : {}) }) -} - -/** Builds the upstream Microdata options while omitting absent optional fields. */ -function parserOptions(options: ParseOptionsType): Readonly> { - return { - dataFactory: factory, - xmlMode: options.xml ?? false, - ...(options.base === undefined ? {} : { baseIRI: options.base }), - ...(options.graph === undefined ? {} : { defaultGraph: options.graph }), - ...(options.vocabularies === undefined ? {} : { vocabRegistry: options.vocabularies }), - } -} - -/** Lazily resolved Microdata parser constructor so importing the subpath does not initialize the optional processor. */ -let parserPromise: Promise | undefined - -/** Lazily imports the Microdata implementation only when this subpath is used. */ -async function defaultParser(): Promise { - parserPromise ??= import('microdata-rdf-streaming-parser').then((module) => module.MicrodataRdfParser as unknown as ParserConstructorType) - return await parserPromise + const roots = document.children.filter((v): v is MarkupElementType => v.kind === 'element') + const all = roots.flatMap((root) => [...elements(root, true)]) + const base = documentBase(all, options.base, html) + const ids = new Map() + for (const element of all) { + const id = attr(element, 'id', html) + if (id !== undefined && !ids.has(id)) ids.set(id, element) + } + const state: StateType = { + ...(base ? { base, document: namedNode(base) } : {}), + graph: options.graph ?? defaultGraph(), + vocabularies: options.vocabularies ?? DEFAULT_VOCABULARIES, + ids, + memory: new Map(), + used: new Set(), + quads: [], + html, + ...(options.signal ? { signal: options.signal } : {}), + } + for (const element of all) { + throwIfAborted(options.signal) + if ( + !hasAttr(element, 'itemscope', html) || hasAttr(element, 'itemprop', html) || + hasAttr(element, 'itemprop-reverse', html) + ) continue + const lang = language(element, html) + item(element, state, lang ? { language: lang } : {}) + } + for (const value of state.quads) { + throwIfAborted(options.signal) + yield value + } +} +/** Converts one item and returns its stable subject. */ function item( + element: MarkupElementType, + state: StateType, + parent: ItemContextType, +): SubjectTermType { + throwIfAborted(state.signal) + const existing = state.memory.get(element) + if (existing) return existing + const identifier = attr(element, 'itemid', state.html) + const subject = identifier ? resource(identifier, state.base) ?? blankNode() : blankNode() + state.memory.set(element, subject) + const types = tokens(attr(element, 'itemtype', state.html)).map((v) => + resource(v, state.base)?.value + ).filter((v): v is string => Boolean(v)) + for (const type of types) emit(state, subject, RDF.type, namedNode(type)) + const type = types[0] ?? parent.type + const vocabulary = type ? vocabularyFor(type, state.vocabularies) : parent.vocabulary + if ( + vocabulary && state.document && registryVocabulary(vocabulary, state.vocabularies) && + !state.used.has(vocabulary) + ) { + state.used.add(vocabulary) + emit(state, state.document, USES_VOCABULARY, namedNode(vocabulary)) + } + const inheritedLanguage = language(element, state.html) ?? parent.language + const context: ItemContextType = { + ...(type ? { type } : {}), + ...(vocabulary ? { vocabulary } : {}), + ...(inheritedLanguage ? { language: inheritedLanguage } : {}), + } + for (const property of properties(element, state, context.language)) { + throwIfAborted(state.signal) + const direct = tokens(attr(property.element, 'itemprop', state.html)), + reverse = tokens(attr(property.element, 'itemprop-reverse', state.html)) + if (!direct.length && !reverse.length) continue + const object = value(property.element, state, { + ...context, + ...(property.language ? { language: property.language } : {}), + }) + for (const name of direct) { + const predicate = predicateIri(name, context, state.base) + if (!predicate) continue + emit(state, subject, predicate, object) + aliases(state, subject, name, predicate, object, context) + } + if (object.termType === 'Literal' || object.termType === 'Quad') continue + for (const name of reverse) { + const predicate = predicateIri(name, context, state.base) + if (!predicate) continue + emit(state, object, predicate, subject) + aliases(state, object, name, predicate, subject, context) + } + } + return subject +} +/** Collects property descendants without crossing nested item children. */ function properties( + element: MarkupElementType, + state: StateType, + inheritedLanguage?: string, +): PropertyElementType[] { + const output: PropertyElementType[] = [], visited = new Set() + /** Walks one property-memory root with language inheritance. */ function walk( + current: MarkupElementType, + currentLanguage?: string, + ) { + if (visited.has(current)) return + visited.add(current) + const own = language(current, state.html) ?? currentLanguage + if ( + hasAttr(current, 'itemprop', state.html) || hasAttr(current, 'itemprop-reverse', state.html) + ) output.push({ element: current, ...(own ? { language: own } : {}) }) + if (current !== element && hasAttr(current, 'itemscope', state.html)) return + for (const child of current.children) if (child.kind === 'element') walk(child, own) + } + for (const child of element.children) if (child.kind === 'element') walk(child, inheritedLanguage) + for (const id of tokens(attr(element, 'itemref', state.html))) { + const reference = state.ids.get(id) + if (reference) walk(reference, language(reference, state.html) ?? inheritedLanguage) + } + return output +} +/** Resolves the RDF value contributed by one property element. */ function value( + element: MarkupElementType, + state: StateType, + context: ItemContextType, +): ObjectTermType { + if (hasAttr(element, 'itemscope', state.html)) return item(element, state, context) + const tag = localName(element.name), urlAttr = urlProperty(tag) + if (urlAttr) { + const raw = attr(element, urlAttr, state.html), + iri = raw === undefined ? undefined : resource(raw, state.base) + if (iri) return iri + } + if (tag === 'meta') { + return languageLiteral(attr(element, 'content', state.html) ?? '', context.language) + } + if (tag === 'data' || tag === 'meter') { + return scalar(attr(element, 'value', state.html) ?? textContent(element)) + } + if (tag === 'time') { + return time(attr(element, 'datetime', state.html) ?? textContent(element), context.language) + } + return languageLiteral(textContent(element), context.language) +} +/** Emits registry aliases for one property. */ function aliases( + state: StateType, + subject: SubjectTermType, + name: string, + predicate: string, + object: ObjectTermType, + context: ItemContextType, +) { + if (!context.vocabulary) return + const entry = state.vocabularies[context.vocabulary]?.properties?.[name] + if (!entry) return + for (const alias of [...asList(entry.subPropertyOf), ...asList(entry.equivalentProperty)]) { + const resolved = predicateIri(alias, context, state.base) ?? absolute(alias) + if (resolved && resolved !== predicate) emit(state, subject, resolved, object) + } +} +/** Appends one generated statement. */ function emit( + state: StateType, + subject: SubjectTermType, + predicate: string, + object: ObjectTermType, +) { + state.quads.push(quad(subject, namedNode(predicate), object, state.graph)) +} +/** Resolves one property token with vocabulary rules. */ function predicateIri( + name: string, + context: ItemContextType, + base?: string, +) { + const direct = absolute(name) + if (direct) return direct + if (context.vocabulary) return joinVocabulary(context.vocabulary, name) + if (!base) return undefined + const url = new URL(base) + url.hash = name + return url.href +} +/** Derives the active vocabulary from an item type. */ function vocabularyFor( + type: string, + registry: VocabularyRegistryType, +) { + let match = '' + for (const candidate of Object.keys(registry)) { + if (type.startsWith(candidate) && candidate.length > match.length) match = candidate + } + if (match) return match + const hash = type.lastIndexOf('#') + if (hash >= 0) return type.slice(0, hash + 1) + const slash = type.lastIndexOf('/') + return slash >= 0 ? type.slice(0, slash + 1) : type +} +/** Tests explicit registry membership. */ function registryVocabulary( + value: string, + registry: VocabularyRegistryType, +) { + return Object.hasOwn(registry, value) +} +/** Joins a vocabulary and property token. */ function joinVocabulary(vocab: string, name: string) { + return vocab.endsWith('/') || vocab.endsWith('#') ? `${vocab}${name}` : `${vocab}#${name}` +} +/** Applies the first HTML base element. */ function documentBase( + values: readonly MarkupElementType[], + supplied: string | undefined, + html: boolean, +) { + if (!html) return supplied + const base = values.find((e) => + localName(e.name) === 'base' && attr(e, 'href', true) !== undefined + ) + const href = base ? attr(base, 'href', true) : undefined + return href ? resolve(href, supplied) : supplied +} +/** Gets directly declared language. */ function language( + element: MarkupElementType, + html: boolean, +) { + const value = attr(element, 'lang', html) ?? attr(element, 'xml:lang', html) + return value?.trim().toLowerCase() || undefined +} +/** Creates language/plain literal. */ function languageLiteral( + value: string, + language?: string, +): ObjectTermType { + return language ? literal(value, language) : literal(value) +} +/** Converts numeric data/meter lexical values. */ function scalar(value: string): ObjectTermType { + const n = value.trim() + if (/^[+-]?\d+$/u.test(n)) return literal(n, namedNode(XSD.integer)) + if (/^[+-]?(?:\d+\.\d*|\.\d+|\d+)(?:[eE][+-]?\d+)?$/u.test(n) && /[.eE]/u.test(n)) { + return literal(n, namedNode(XSD.double)) + } + return literal(value) +} +/** Converts time lexical values to matching XSD datatype. */ function time( + value: string, + language?: string, +): ObjectTermType { + const n = value.trim() + if (/^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(?::\d{2}(?:\.\d+)?)?(?:Z|[+-]\d{2}:\d{2})?$/u.test(n)) { + return literal(n, namedNode(XSD.dateTime)) + } + if (/^\d{4}-\d{2}-\d{2}$/u.test(n)) return literal(n, namedNode(XSD.date)) + if (/^\d{2}:\d{2}(?::\d{2}(?:\.\d+)?)?(?:Z|[+-]\d{2}:\d{2})?$/u.test(n)) { + return literal(n, namedNode(XSD_TIME)) + } + if (/^\d{4}-\d{2}$/u.test(n)) return literal(n, namedNode(XSD_G_YEAR_MONTH)) + if (/^\d{4}$/u.test(n)) return literal(n, namedNode(XSD_G_YEAR)) + if ( + /^-?P(?=\d|T\d)(?:\d+Y)?(?:\d+M)?(?:\d+D)?(?:T(?:\d+H)?(?:\d+M)?(?:\d+(?:\.\d+)?S)?)?$/u.test(n) + ) return literal(n, namedNode(XSD_DURATION)) + return languageLiteral(value, language) +} +/** Returns URL-valued attribute for one HTML element. */ function urlProperty(name: string) { + if (['a', 'area', 'link'].includes(name)) return 'href' + if (['audio', 'embed', 'iframe', 'img', 'source', 'track', 'video'].includes(name)) return 'src' + if (name === 'object') return 'data' + return undefined +} +/** Resolves resource IRI to named node. */ function resource(value: string, base?: string) { + const resolved = resolve(value.trim(), base) + return resolved ? namedNode(resolved) : undefined +} +/** Resolves an IRI reference. */ function resolve(value: string, base?: string) { + if (!value) return undefined + try { + return base ? new URL(value, base).href : new URL(value).href + } catch { + return undefined + } +} +/** Returns value only if absolute IRI. */ function absolute(value: string) { + try { + return new URL(value).href + } catch { + return undefined + } +} +/** Splits a space-separated token list. */ function tokens(value?: string) { + return value?.trim().split(/\s+/u).filter(Boolean) ?? [] +} +/** Returns lower-cased local element name. */ function localName(name: string) { + const colon = name.indexOf(':') + return (colon >= 0 ? name.slice(colon + 1) : name).toLowerCase() +} +/** Normalizes one-or-many registry predicate mappings. */ function asList( + value?: string | readonly string[], +): readonly string[] { + return value === undefined ? [] : typeof value === 'string' ? [value] : value } diff --git a/packages/rdf/microdata/mod_test.ts b/packages/rdf/microdata/mod_test.ts index 07e1c27..288613d 100644 --- a/packages/rdf/microdata/mod_test.ts +++ b/packages/rdf/microdata/mod_test.ts @@ -1,49 +1,58 @@ import { describe, it } from 'node:test' import { expect } from '@std/expect' -import { literal, namedNode, quad, type Quad } from '../mod.ts' -import { parse, type ParserType } from './mod.ts' +import { RDF } from '../mod.ts' +import { parse } from './mod.ts' -const fixture = quad(namedNode('https://example.test/s'), namedNode('https://example.test/p'), literal('value')) -type EventType = 'drain' | 'close' | 'error' -type ListenerType = (...args: unknown[]) => void - -class TestParser implements ParserType { - static options: Readonly> | undefined - private readonly listeners = new Map>() - private ended = false - private resolveEnd: (() => void) | undefined - constructor(options: Readonly>) { TestParser.options = options } - write(_value: string | Uint8Array): boolean { return true } - end(): void { this.ended = true; this.resolveEnd?.() } - destroy(error?: Error): void { if (error) this.emit('error', error); this.ended = true; this.resolveEnd?.(); this.emit('close') } - once(event: EventType, listener: ListenerType): this { const values = this.listeners.get(event) ?? new Set(); values.add(listener); this.listeners.set(event, values); return this } - off(event: EventType, listener: ListenerType): this { this.listeners.get(event)?.delete(listener); return this } - private emit(event: EventType, ...args: unknown[]): void { for (const listener of this.listeners.get(event) ?? []) listener(...args) } - async *[Symbol.asyncIterator](): AsyncIterator { if (!this.ended) await new Promise((resolve) => { this.resolveEnd = resolve }); yield fixture } +async function all(source: string, options = {}) { + const values = [] + for await (const value of parse(source, options)) values.push(value) + return values } describe('@okikio/rdf/microdata', () => { - it('adapts Microdata parser output back to native RDF terms', async () => { - const values: Quad[] = [] - for await (const value of parse('
', { parser: TestParser })) values.push(value) - expect(values).toHaveLength(1) - expect(values[0]?.equals(fixture)).toBe(true) + it('extracts schema vocabulary types, properties, URLs, and language natively', async () => { + const html = + `
Alice
` + const values = await all(html, { base: 'https://example.test/' }) + expect( + values.some((value) => + value.predicate.value === RDF.type && value.object.value === 'https://schema.org/Person' + ), + ).toBe(true) + expect( + values.some((value) => + value.predicate.value === 'https://schema.org/name' && + value.object.termType === 'Literal' && value.object.language === 'en' + ), + ).toBe(true) + expect( + values.some((value) => + value.predicate.value === 'https://schema.org/url' && + value.object.value === 'https://example.test/alice' + ), + ).toBe(true) }) - it('forwards Microdata base, XML mode, vocabulary registry, and native RDF factory', async () => { - const vocabularies = { 'https://schema.org/': {} } - for await (const _value of parse('
', { - parser: TestParser, - base: 'https://example.test/base/', - xml: true, - vocabularies, - })) { /* drain */ } - - expect(TestParser.options?.baseIRI).toBe('https://example.test/base/') - expect(TestParser.options?.xmlMode).toBe(true) - expect(TestParser.options?.vocabRegistry).toBe(vocabularies) - const factory = TestParser.options?.dataFactory as { namedNode(value: string): { value: string } } - expect(factory.namedNode('urn:test').value).toBe('urn:test') + it('resolves itemref properties outside the item subtree', async () => { + const html = + `
Alice
` + const values = await all(html) + expect( + values.some((value) => + value.predicate.value === 'https://schema.org/jobTitle' && value.object.value === 'Engineer' + ), + ).toBe(true) }) + it('supports reverse properties without producing reverse literal subjects', async () => { + const html = + `
` + const values = await all(html) + expect( + values.some((value) => + value.subject.value === 'https://example.test/b' && + value.object.value === 'https://example.test/a' + ), + ).toBe(true) + }) }) diff --git a/packages/rdf/microdata/types.ts b/packages/rdf/microdata/types.ts deleted file mode 100644 index 966a2d5..0000000 --- a/packages/rdf/microdata/types.ts +++ /dev/null @@ -1,11 +0,0 @@ -/** Structural Microdata parser contracts used to isolate the external implementation. @module */ - -import type { TransformParserType } from '../transform.ts' - -/** Microdata parser stream shape required by the adapter. */ -export type ParserType = TransformParserType - -/** Constructor contract for a Microdata parser implementation. */ -export interface ParserConstructorType { - new (options: Readonly>): ParserType -} diff --git a/packages/rdf/mod.ts b/packages/rdf/mod.ts index d3a66f3..d9004f7 100644 --- a/packages/rdf/mod.ts +++ b/packages/rdf/mod.ts @@ -1,5 +1,5 @@ /** - * Complete RDF programming model for TypeScript runtimes. + * Dependency-free RDF programming model and native syntax capabilities for TypeScript runtimes. * * The root keeps the core term, dataset, namespace, and source contracts light. * Concrete syntaxes live on explicit subpaths so importing `@okikio/rdf` does @@ -19,24 +19,35 @@ * @module */ -export { dataset, Dataset, datasetEquals, datasetKey } from './dataset.ts' -export type { MatchOptions } from './dataset.ts' -export { blankNode, defaultGraph, factory, fromQuad, fromTerm, literal, namedNode, quad, triple, variable } from './factory.ts' +export { Dataset, dataset, datasetEquals, datasetKey } from './dataset.ts' +export type { MatchOptionsType } from './dataset.ts' +export { + blankNode, + defaultGraph, + factory, + fromQuad, + fromTerm, + literal, + namedNode, + quad, + triple, + variable, +} from './factory.ts' export { namespace } from './namespace.ts' export type { Namespace } from './namespace.ts' export { equals, isTerm, key, RDF, XSD } from './term.ts' export type { BlankNode, DefaultGraph, - Direction, - DirectionalLanguage, - Graph, + DirectionalLanguageType, + DirectionType, + GraphTermType, Literal, NamedNode, - ObjectTerm, - Predicate, + ObjectTermType, + PredicateTermType, Quad, - Subject, + SubjectTermType, Term, TermType, Variable, diff --git a/packages/rdf/namespace.ts b/packages/rdf/namespace.ts index 9e2ad55..7b47a4e 100644 --- a/packages/rdf/namespace.ts +++ b/packages/rdf/namespace.ts @@ -5,7 +5,9 @@ import type { NamedNode } from './term.ts' /** A callable namespace that expands local names into RDF named nodes. */ export interface Namespace { + /** Creates a named node by resolving the supplied suffix against this namespace base IRI. */ (local: string): NamedNode + /** Absolute IRI represented by this record. */ readonly iri: string } diff --git a/packages/rdf/nquads/mod.ts b/packages/rdf/nquads/mod.ts index 127eae7..3cc80cb 100644 --- a/packages/rdf/nquads/mod.ts +++ b/packages/rdf/nquads/mod.ts @@ -1,11 +1,21 @@ /** RDF 1.2 N-Quads parser and serializer. @module */ -import { diagnostic, lines, parseLine, type ParseEvent, type ParseOptions, type TextSource } from '../line.ts' +import { + diagnostic, + lines, + type ParseEventType, + parseLine, + type ParseOptionsType, + type TextSourceType, +} from '../line.ts' import type { Quad } from '../term.ts' import { writeQuad } from '../write.ts' /** Emits source-ranged N-Quads semantic events. */ -export async function* analyze(source: TextSource, options: ParseOptions = {}): AsyncGenerator { +export async function* analyze( + source: TextSourceType, + options: ParseOptionsType = {}, +): AsyncGenerator { for await (const record of lines(source, options)) { try { const event = parseLine(record, true, options) @@ -18,8 +28,15 @@ export async function* analyze(source: TextSource, options: ParseOptions = {}): } /** Parses N-Quads incrementally. */ -export async function* parse(source: TextSource, options: ParseOptions = {}): AsyncGenerator { - for await (const event of analyze(source, options)) if (event.kind === 'quad') yield event.quad +export async function* parse( + source: TextSourceType, + options: ParseOptionsType = {}, +): AsyncGenerator { + for await (const event of analyze(source, options)) { + if (event.kind === 'quad') { + yield event.quad + } + } } /** Serializes RDF quads using canonical-layout-compatible line formatting. */ @@ -28,4 +45,10 @@ export function write(quads: Iterable): string { return output.length === 0 ? '' : `${output.join('\n')}\n` } -export type { Diagnostic, ParseEvent, ParseOptions, SourceRange, TextSource } from '../line.ts' +export type { + DiagnosticType, + ParseEventType, + ParseOptionsType, + SourceRangeType, + TextSourceType, +} from '../line.ts' diff --git a/packages/rdf/nquads/parse_test.ts b/packages/rdf/nquads/parse_test.ts index bcf6aee..5fec7fe 100644 --- a/packages/rdf/nquads/parse_test.ts +++ b/packages/rdf/nquads/parse_test.ts @@ -28,7 +28,8 @@ describe('@okikio/rdf/nquads', () => { }) it('emits a source-ranged diagnostic and resumes at the next record in tolerant mode', async () => { - const source = ' "ok" .\nnot rdf\n "ok" .\n' + const source = + ' "ok" .\nnot rdf\n "ok" .\n' const events = [] for await (const event of analyze(source, { tolerant: true })) events.push(event) expect(events.map((event) => event.kind)).toEqual(['quad', 'diagnostic', 'quad']) diff --git a/packages/rdf/ntriples/mod.ts b/packages/rdf/ntriples/mod.ts index dc4ca8b..adc25ba 100644 --- a/packages/rdf/ntriples/mod.ts +++ b/packages/rdf/ntriples/mod.ts @@ -1,11 +1,21 @@ /** RDF 1.2 N-Triples parser and serializer. @module */ -import { diagnostic, lines, parseLine, type ParseEvent, type ParseOptions, type TextSource } from '../line.ts' +import { + diagnostic, + lines, + type ParseEventType, + parseLine, + type ParseOptionsType, + type TextSourceType, +} from '../line.ts' import type { Quad } from '../term.ts' import { writeQuad } from '../write.ts' /** Emits source-ranged semantic events. Tolerant mode reports malformed lines and continues. */ -export async function* analyze(source: TextSource, options: ParseOptions = {}): AsyncGenerator { +export async function* analyze( + source: TextSourceType, + options: ParseOptionsType = {}, +): AsyncGenerator { for await (const record of lines(source, options)) { try { const event = parseLine(record, false, options) @@ -18,18 +28,33 @@ export async function* analyze(source: TextSource, options: ParseOptions = {}): } /** Parses N-Triples incrementally and yields RDF quads in the default graph. */ -export async function* parse(source: TextSource, options: ParseOptions = {}): AsyncGenerator { - for await (const event of analyze(source, options)) if (event.kind === 'quad') yield event.quad +export async function* parse( + source: TextSourceType, + options: ParseOptionsType = {}, +): AsyncGenerator { + for await (const event of analyze(source, options)) { + if (event.kind === 'quad') { + yield event.quad + } + } } /** Serializes RDF triples using canonical-layout-compatible line formatting. */ export function write(quads: Iterable): string { const lines: string[] = [] for (const quad of quads) { - if (quad.graph.termType !== 'DefaultGraph') throw new TypeError('N-Triples cannot serialize named graphs.') + if (quad.graph.termType !== 'DefaultGraph') { + throw new TypeError('N-Triples cannot serialize named graphs.') + } lines.push(writeQuad(quad, false)) } return lines.length === 0 ? '' : `${lines.join('\n')}\n` } -export type { Diagnostic, ParseEvent, ParseOptions, SourceRange, TextSource } from '../line.ts' +export type { + DiagnosticType, + ParseEventType, + ParseOptionsType, + SourceRangeType, + TextSourceType, +} from '../line.ts' diff --git a/packages/rdf/ntriples/parse_test.ts b/packages/rdf/ntriples/parse_test.ts index 4e8991e..2491ea9 100644 --- a/packages/rdf/ntriples/parse_test.ts +++ b/packages/rdf/ntriples/parse_test.ts @@ -1,6 +1,6 @@ import { describe, it } from 'node:test' import { expect } from '@std/expect' -import { namedNode, quad, literal, type Quad } from '../mod.ts' +import { literal, namedNode, type Quad, quad } from '../mod.ts' import { parse, write } from './mod.ts' describe('@okikio/rdf/ntriples', () => { diff --git a/packages/rdf/ontology/index.ts b/packages/rdf/ontology/index.ts index 56738a3..e07cfbc 100644 --- a/packages/rdf/ontology/index.ts +++ b/packages/rdf/ontology/index.ts @@ -4,7 +4,9 @@ import type { ClassType, ModelType, PropertyType } from './model.ts' /** Read-only indexes over one ontology model. */ export class OntologyIndex { + /** Ontology class lookup indexed by absolute class IRI. */ readonly #classes: ReadonlyMap + /** Ontology property lookup indexed by absolute property IRI. */ readonly #properties: ReadonlyMap /** Builds immutable lookup maps over a parsed ontology model without performing entailment. */ diff --git a/packages/rdf/ontology/read.ts b/packages/rdf/ontology/inspect.ts similarity index 77% rename from packages/rdf/ontology/read.ts rename to packages/rdf/ontology/inspect.ts index f3b677b..6aecc82 100644 --- a/packages/rdf/ontology/read.ts +++ b/packages/rdf/ontology/inspect.ts @@ -1,4 +1,4 @@ -/** RDFS and directly interpretable OWL ontology reader. @module */ +/** RDFS and directly interpretable OWL ontology inspector. @module */ import { iterate } from '../source.ts' import { key, XSD } from '../term.ts' @@ -93,78 +93,122 @@ const PROPERTY_CHARACTERISTICS = new Map([ /** One RDF source contributing to an ontology model. */ export interface OntologySourceType extends SourceType { + /** RDF quads supplied by or retained for this source. */ readonly quads: Iterable | AsyncIterable } /** Extension predicates and resource limits for ontology ingestion. */ -export interface ReadOptions { +export interface InspectOptionsType { /** Additional predicates that behave like vocabulary-domain declarations. */ readonly domainPredicates?: readonly string[] /** Additional predicates that behave like vocabulary-range declarations. */ readonly rangePredicates?: readonly string[] /** Maximum quads across all sources. Default is 5,000,000. */ readonly maxQuads?: number + /** Caller-owned abort signal checked before expensive work and between long-running steps. */ readonly signal?: AbortSignal } /** Mutable accumulation record for one named class before deterministic sets/metadata are frozen into the public ontology model. */ -interface MutableClass { +interface MutableClassType { + /** Absolute IRI represented by this record. */ readonly iri: string + /** Localized labels retained from the RDF source. */ readonly labels: TextType[] + /** Localized descriptive comments retained from the RDF source. */ readonly comments: TextType[] + /** Direct superclass IRIs declared for this class. */ readonly superClasses: Set + /** Class IRIs declared equivalent to this class. */ readonly equivalentClasses: Set + /** Class IRIs declared disjoint with this class. */ readonly disjointClasses: Set + /** Whether the source explicitly marks this ontology term as deprecated. */ deprecated: boolean } /** Mutable accumulation record for one named property before relationship sets and characteristics are normalized. */ -interface MutableProperty { +interface MutablePropertyType { + /** Absolute IRI represented by this record. */ readonly iri: string + /** RDF/OWL property kinds observed for this property. */ readonly kinds: Set + /** Localized labels retained from the RDF source. */ readonly labels: TextType[] + /** Localized descriptive comments retained from the RDF source. */ readonly comments: TextType[] + /** Class IRIs declared as domains of this property. */ readonly domains: Set + /** Class or datatype IRIs declared as ranges of this property. */ readonly ranges: Set + /** Direct super-property IRIs declared for this property. */ readonly superProperties: Set + /** Property IRIs declared equivalent to this property. */ readonly equivalentProperties: Set + /** Property IRIs declared as inverses of this property. */ readonly inverseOf: Set + /** Property IRIs declared disjoint with this property. */ readonly disjointProperties: Set + /** OWL property characteristics, such as functional or transitive, retained as normalized identifiers. */ readonly characteristics: Set + /** Whether the source explicitly marks this ontology term as deprecated. */ deprecated: boolean } /** Label/comment assertion deferred until its subject is known to be a modeled class, property, or datatype. */ -interface PendingText { +interface PendingTextType { + /** Stable identifier of the source document that contributed this record. */ readonly sourceId: string + /** Semantic model field that receives the deferred ontology value after all source declarations are known. */ readonly field: 'labels' | 'comments' + /** Source quad retained until the referenced ontology term is available. */ readonly quad: Quad } /** `owl:deprecated` assertion deferred until the subject kind is known, avoiding premature lossy classification. */ -interface PendingDeprecated { +interface PendingDeprecatedType { + /** Stable identifier of the source document that contributed this record. */ readonly sourceId: string + /** Source quad retained until deprecation metadata can be attached to its ontology term. */ readonly quad: Quad } /** - * Reads named RDFS/OWL declarations into a deterministic ontology model. + * Inspects named RDFS/OWL declarations into a deterministic ontology model. * * This operation does not perform RDFS or OWL entailment. It records declared * named relationships and preserves every unsupported assertion so reasoning - * engines or future readers can interpret richer class expressions later. + * engines or future consumers can interpret richer class expressions later. + * + * Input sources remain caller-owned. `maxQuads` limits materialized work across + * all sources, and `signal` can stop ingestion between quads. + * + * @example + * ```ts + * import { namedNode, quad } from '@okikio/rdf' + * import * as ontology from '@okikio/rdf/ontology' + * + * const model = await ontology.inspect([{ + * id: 'example', + * quads: [quad( + * namedNode('https://example.test/Person'), + * namedNode('http://www.w3.org/1999/02/22-rdf-syntax-ns#type'), + * namedNode('http://www.w3.org/2002/07/owl#Class'), + * )], + * }]) + * ``` */ -export async function read( +export async function inspect( sources: readonly OntologySourceType[], - options: ReadOptions = {}, + options: InspectOptionsType = {}, ): Promise { - const classes = new Map() - const properties = new Map() + const classes = new Map() + const properties = new Map() const datatypes = new Set() const assertions: AssertionType[] = [] const diagnostics: DiagnosticType[] = [] - const pendingText: PendingText[] = [] - const pendingDeprecated: PendingDeprecated[] = [] + const pendingText: PendingTextType[] = [] + const pendingDeprecated: PendingDeprecatedType[] = [] const domainPredicates = new Set([RDFS_DOMAIN, ...(options.domainPredicates ?? [])]) const rangePredicates = new Set([RDFS_RANGE, ...(options.rangePredicates ?? [])]) const maxQuads = options.maxQuads ?? 5_000_000 @@ -172,8 +216,12 @@ export async function read( for (const source of sources) { for await (const quad of iterate(source.quads)) { - if (options.signal?.aborted) throw options.signal.reason ?? new DOMException('Aborted', 'AbortError') - if (++count > maxQuads) throw new RangeError(`Ontology input exceeds the configured ${maxQuads} quad limit.`) + if (options.signal?.aborted) { + throw options.signal.reason ?? new DOMException('Aborted', 'AbortError') + } + if (++count > maxQuads) { + throw new RangeError(`Ontology input exceeds the configured ${maxQuads} quad limit.`) + } if (quad.subject.termType !== 'NamedNode') { assertions.push(toAssertion(quad, source.id)) continue @@ -198,7 +246,10 @@ export async function read( getProperty(properties, subject).characteristics.add(characteristic) continue } - if (objectIri === RDFS_DATATYPE || objectIri.startsWith(`${XSD.string.slice(0, XSD.string.lastIndexOf('#') + 1)}`)) { + if ( + objectIri === RDFS_DATATYPE || + objectIri.startsWith(`${XSD.string.slice(0, XSD.string.lastIndexOf('#') + 1)}`) + ) { datatypes.add(subject) continue } @@ -272,9 +323,9 @@ export async function read( /** Attaches labels/comments only to classified named ontology resources and records otherwise-unclassified text. */ function attachText( - classes: ReadonlyMap, - properties: ReadonlyMap, - pending: readonly PendingText[], + classes: ReadonlyMap, + properties: ReadonlyMap, + pending: readonly PendingTextType[], assertions: AssertionType[], diagnostics: DiagnosticType[], ): void { @@ -299,9 +350,9 @@ function attachText( /** Interprets supported deprecation literals while retaining malformed or unclassified declarations diagnostically. */ function attachDeprecation( - classes: ReadonlyMap, - properties: ReadonlyMap, - pending: readonly PendingDeprecated[], + classes: ReadonlyMap, + properties: ReadonlyMap, + pending: readonly PendingDeprecatedType[], assertions: AssertionType[], diagnostics: DiagnosticType[], ): void { @@ -339,7 +390,7 @@ function attachDeprecation( } /** Returns or creates the mutable accumulator for one named ontology class. */ -function getClass(values: Map, iri: string): MutableClass { +function getClass(values: Map, iri: string): MutableClassType { let value = values.get(iri) if (!value) { value = { @@ -357,7 +408,7 @@ function getClass(values: Map, iri: string): MutableClass } /** Returns or creates the mutable accumulator for one named ontology property. */ -function getProperty(values: Map, iri: string): MutableProperty { +function getProperty(values: Map, iri: string): MutablePropertyType { let value = values.get(iri) if (!value) { value = { @@ -380,7 +431,7 @@ function getProperty(values: Map, iri: string): Mutable } /** Converts a mutable class accumulator into stable sorted serializable ontology output. */ -function freezeClass(value: MutableClass): ClassType { +function freezeClass(value: MutableClassType): ClassType { return { iri: value.iri, labels: sortText(value.labels), @@ -393,7 +444,7 @@ function freezeClass(value: MutableClass): ClassType { } /** Converts a mutable property accumulator into stable sorted serializable ontology output. */ -function freezeProperty(value: MutableProperty): PropertyType { +function freezeProperty(value: MutablePropertyType): PropertyType { return { iri: value.iri, kinds: [...value.kinds].sort(), @@ -412,7 +463,14 @@ function freezeProperty(value: MutableProperty): PropertyType { /** Converts the supplied value to text without changing semantic identity. */ function toText(value: Literal): TextType { - const text: { value: string; language?: string; direction?: 'ltr' | 'rtl' } = { value: value.value } + const text: { + /** Literal lexical form preserved for localized ontology text. */ + value: string + /** BCP 47 language tag retained for this localized RDF value. */ + language?: string + /** RDF 1.2 base text direction retained for this localized RDF value. */ + direction?: 'ltr' | 'rtl' + } = { value: value.value } if (value.language) text.language = value.language if (value.direction) text.direction = value.direction return text @@ -438,7 +496,16 @@ function toAssertion(value: Quad, sourceId: string): AssertionType { /** Converts retained RDF assertions to stable string records without carrying runtime term objects. */ function stripQuads(source: OntologySourceType): SourceType { - const value: { id: string; iri?: string; version?: string; hash?: string } = { id: source.id } + const value: { + /** Stable source identifier retained for this stripped ontology source record. */ + id: string + /** IRI retained by this value. */ + iri?: string + /** Version marker retained by this syntax record. */ + version?: string + /** N-degree hash selected for this canonicalization candidate. */ + hash?: string + } = { id: source.id } if (source.iri !== undefined) value.iri = source.iri if (source.version !== undefined) value.version = source.version if (source.hash !== undefined) value.hash = source.hash @@ -456,9 +523,10 @@ function sortText(values: readonly TextType[]): TextType[] { /** Compare assertion using deterministic semantic ordering. */ function compareAssertion(left: AssertionType, right: AssertionType): number { - return `${left.sourceId}\u0000${left.subject}\u0000${left.predicate}\u0000${left.object}`.localeCompare( - `${right.sourceId}\u0000${right.subject}\u0000${right.predicate}\u0000${right.object}`, - ) + return `${left.sourceId}\u0000${left.subject}\u0000${left.predicate}\u0000${left.object}` + .localeCompare( + `${right.sourceId}\u0000${right.subject}\u0000${right.predicate}\u0000${right.object}`, + ) } /** Compare diagnostic using deterministic semantic ordering. */ @@ -469,6 +537,11 @@ function compareDiagnostic(left: DiagnosticType, right: DiagnosticType): number } /** Orders named ontology resources by IRI so source file order does not affect the model. */ -function byIri(left: T, right: T): number { +function byIri< + T extends { + /** IRI retained by this byIri. */ + readonly iri: string + }, +>(left: T, right: T): number { return left.iri.localeCompare(right.iri) } diff --git a/packages/rdf/ontology/read_test.ts b/packages/rdf/ontology/inspect_test.ts similarity index 85% rename from packages/rdf/ontology/read_test.ts rename to packages/rdf/ontology/inspect_test.ts index 3888c15..380d0cc 100644 --- a/packages/rdf/ontology/read_test.ts +++ b/packages/rdf/ontology/inspect_test.ts @@ -1,13 +1,13 @@ import { describe, it } from 'node:test' import { expect } from '@std/expect' import { parse } from '../turtle/mod.ts' -import { index, read } from './mod.ts' +import { index, inspect } from './mod.ts' const SCHEMA_DOMAIN = 'https://schema.org/domainIncludes' const SCHEMA_RANGE = 'https://schema.org/rangeIncludes' describe('@okikio/rdf/ontology', () => { - it('reads named RDFS/OWL relationships and configurable vocabulary aliases', async () => { + it('inspects named RDFS/OWL relationships and configurable vocabulary aliases', async () => { const source = ` @prefix rdf: . @prefix rdfs: . @@ -20,7 +20,7 @@ describe('@okikio/rdf/ontology', () => { ex:name a rdf:Property, owl:FunctionalProperty ; schema:domainIncludes ex:Thing ; schema:rangeIncludes schema:Text . ` - const model = await read([{ id: 'test', quads: parse(source) }], { + const model = await inspect([{ id: 'test', quads: parse(source) }], { domainPredicates: [SCHEMA_DOMAIN], rangePredicates: [SCHEMA_RANGE], }) @@ -37,7 +37,7 @@ describe('@okikio/rdf/ontology', () => { @prefix ex: . ex:Product a owl:Class ; rdfs:subClassOf [ a owl:Restriction ; owl:onProperty ex:name ] . ` - const model = await read([{ id: 'test', quads: parse(source) }]) + const model = await inspect([{ id: 'test', quads: parse(source) }]) expect(model.classes).toHaveLength(1) expect(model.assertions.length > 0).toBe(true) }) diff --git a/packages/rdf/ontology/mod.ts b/packages/rdf/ontology/mod.ts index e0d879d..998e192 100644 --- a/packages/rdf/ontology/mod.ts +++ b/packages/rdf/ontology/mod.ts @@ -1,16 +1,16 @@ /** * Generic RDFS and OWL ontology interpretation helpers. * - * The reader records named declarations and relationships without performing + * The inspector records named declarations and relationships without performing * entailment. Anonymous OWL expressions and unsupported axioms are retained as - * assertions so a reasoner or future parser can interpret them later. + * assertions so a reasoner or future processor can interpret them later. * * @module */ export { index, OntologyIndex } from './index.ts' -export { read } from './read.ts' -export type { OntologySourceType, ReadOptions } from './read.ts' +export { inspect } from './inspect.ts' +export type { InspectOptionsType, OntologySourceType } from './inspect.ts' export type { AssertionType, ClassType, diff --git a/packages/rdf/ontology/model.ts b/packages/rdf/ontology/model.ts index 0bd65f0..9531dec 100644 --- a/packages/rdf/ontology/model.ts +++ b/packages/rdf/ontology/model.ts @@ -11,27 +11,41 @@ /** Localized ontology text. */ export interface TextType { + /** Human-readable ontology text without its optional language metadata. */ readonly value: string + /** BCP 47 language tag associated with this localized RDF value. */ readonly language?: string + /** RDF 1.2 base text direction associated with this language value. */ readonly direction?: 'ltr' | 'rtl' } /** Source metadata attached to ontology assertions. */ export interface SourceType { + /** Caller-supplied source identifier retained on derived ontology assertions for provenance. */ readonly id: string + /** Absolute IRI represented by this record. */ readonly iri?: string + /** Optional source vocabulary or release version retained as provenance; the inspector does not interpret it. */ readonly version?: string + /** Stable source-content hash when the caller provides one for provenance or cache identity. */ readonly hash?: string } /** Named ontology class and directly interpretable RDFS/OWL relationships. */ export interface ClassType { + /** Absolute IRI represented by this record. */ readonly iri: string + /** Localized labels retained from the RDF source. */ readonly labels: readonly TextType[] + /** Localized descriptive comments retained from the RDF source. */ readonly comments: readonly TextType[] + /** Direct superclass IRIs declared for this class. */ readonly superClasses: readonly string[] + /** Class IRIs declared equivalent to this class. */ readonly equivalentClasses: readonly string[] + /** Class IRIs declared disjoint with this class. */ readonly disjointClasses: readonly string[] + /** Whether the source explicitly marks this ontology term as deprecated. */ readonly deprecated: boolean } @@ -50,43 +64,70 @@ export type PropertyCharacteristicType = /** Named RDF/OWL property. */ export interface PropertyType { + /** Absolute IRI represented by this record. */ readonly iri: string + /** RDF/OWL property kinds observed for this property. */ readonly kinds: readonly PropertyKindType[] + /** Localized labels retained from the RDF source. */ readonly labels: readonly TextType[] + /** Localized descriptive comments retained from the RDF source. */ readonly comments: readonly TextType[] + /** Class IRIs declared as domains of this property. */ readonly domains: readonly string[] + /** Class or datatype IRIs declared as ranges of this property. */ readonly ranges: readonly string[] + /** Direct super-property IRIs declared for this property. */ readonly superProperties: readonly string[] + /** Property IRIs declared equivalent to this property. */ readonly equivalentProperties: readonly string[] + /** Property IRIs declared as inverses of this property. */ readonly inverseOf: readonly string[] + /** Property IRIs declared disjoint with this property. */ readonly disjointProperties: readonly string[] + /** OWL property characteristics, such as functional or transitive, retained as normalized identifiers. */ readonly characteristics: readonly PropertyCharacteristicType[] + /** Whether the source explicitly marks this ontology term as deprecated. */ readonly deprecated: boolean } -/** RDF assertion retained when this reader does not interpret it. */ +/** RDF assertion retained when the ontology inspector does not interpret it. */ export interface AssertionType { + /** Stable identifier of the source document that contributed this record. */ readonly sourceId: string + /** RDF subject term represented by this statement, pattern, or index entry. */ readonly subject: string + /** RDF predicate IRI represented by this statement, pattern, or index entry. */ readonly predicate: string + /** RDF object term represented by this statement, pattern, or index entry. */ readonly object: string + /** RDF graph name represented by this quad, statement, or query target. */ readonly graph: string } -/** Ontology-reader problem that does not require discarding the source graph. */ +/** Ontology-inspection problem that does not require discarding the source graph. */ export interface DiagnosticType { + /** Stable machine-readable code used to classify this diagnostic or failure. */ readonly code: 'unclassified-text' | 'unclassified-deprecation' | 'invalid-deprecation' + /** Human-readable explanation of the diagnostic or failure. */ readonly message: string + /** Stable identifier of the source document that contributed this record. */ readonly sourceId: string + /** RDF term associated with this diagnostic. */ readonly term?: string } /** Stable ontology intermediate representation. */ export interface ModelType { + /** Source provenance records retained by the normalized ontology or vocabulary model. */ readonly sources: readonly SourceType[] + /** Normalized RDF/OWL class records discovered across the inspected sources. */ readonly classes: readonly ClassType[] + /** Property records or property definitions owned by this model. */ readonly properties: readonly PropertyType[] + /** Datatype IRIs discovered or referenced by the inspected ontology sources. */ readonly datatypes: readonly string[] + /** RDF assertions retained because the current semantic layer does not interpret them further. */ readonly assertions: readonly AssertionType[] + /** Structured diagnostics retained so recoverable source information is not silently discarded. */ readonly diagnostics: readonly DiagnosticType[] } diff --git a/packages/rdf/package.json b/packages/rdf/package.json index 074c32d..9d27217 100644 --- a/packages/rdf/package.json +++ b/packages/rdf/package.json @@ -9,22 +9,16 @@ "./nquads": "./nquads/mod.ts", "./turtle": "./turtle/mod.ts", "./trig": "./trig/mod.ts", + "./shape": "./shape/mod.ts", + "./ontology": "./ontology/mod.ts", + "./stream": "./stream.ts", "./jsonld": "./jsonld/mod.ts", - "./canon": "./canon/mod.ts", "./xml": "./xml/mod.ts", "./rdfa": "./rdfa/mod.ts", "./microdata": "./microdata/mod.ts", - "./shape": "./shape/mod.ts", - "./ontology": "./ontology/mod.ts" - }, - "dependencies": { - "jsonld": "^9.0.0", - "rdf-canonize": "^5.0.0", - "rdfxml-streaming-parser": "^3.2.0", - "rdfa-streaming-parser": "^3.0.2", - "microdata-rdf-streaming-parser": "^3.0.0" + "./canon": "./canon/mod.ts" }, - "description": "Complete RDF programming model for TypeScript runtimes.", + "description": "Dependency-free RDF terms, datasets, native syntax processors, canonicalization, ontology models, and shape models for TypeScript runtimes.", "license": "MIT", "repository": { "type": "git", diff --git a/packages/rdf/rdfa/mod.ts b/packages/rdf/rdfa/mod.ts index 4741e13..7ed7786 100644 --- a/packages/rdf/rdfa/mod.ts +++ b/packages/rdf/rdfa/mod.ts @@ -1,76 +1,518 @@ -/** RDFa 1.1 parsing behind native RDF and Web-oriented source contracts. @module */ - -import { factory } from '../factory.ts' -import type { Graph, Quad } from '../term.ts' -import { throwIfAborted, type TextSource } from '../text.ts' -import { parseTransform } from '../transform.ts' -import type { ParserConstructorType } from './types.ts' - -export type { ParserConstructorType, ParserType } from './types.ts' - -/** RDFa profiles implemented by the external RDFa 1.1 processor. */ -export type ProfileType = '' | 'core' | 'html' | 'xhtml' | 'svg' | 'xml' - -/** Known content types that select an RDFa host-language profile. */ -export type ContentType = +/** Native RDFa 1.1 extraction over the package-owned markup parser. @module */ +import { blankNode, defaultGraph, literal, namedNode, quad } from '../factory.ts' +import { + attr, + elements, + hasAttr, + type MarkupElementType, + parseMarkup, + textContent, +} from '../markup.ts' +import { + type GraphTermType, + type ObjectTermType, + type Quad, + RDF, + type SubjectTermType, +} from '../term.ts' +import { type TextSourceType, throwIfAborted } from '../text.ts' +/** RDFa vocabulary-use predicate emitted for documents that declare a default vocabulary. */ +const USES_VOCABULARY = 'http://www.w3.org/ns/rdfa#usesVocabulary', + XML_LITERAL = 'http://www.w3.org/1999/02/22-rdf-syntax-ns#XMLLiteral', + XHTML = 'http://www.w3.org/1999/xhtml/vocab#' +/** Initial RDFa prefix mappings defined by the processor profile. */ +const PREFIXES: Readonly> = { + cc: 'http://creativecommons.org/ns#', + dc: 'http://purl.org/dc/terms/', + dcterms: 'http://purl.org/dc/terms/', + foaf: 'http://xmlns.com/foaf/0.1/', + og: 'http://ogp.me/ns#', + owl: 'http://www.w3.org/2002/07/owl#', + rdf: 'http://www.w3.org/1999/02/22-rdf-syntax-ns#', + rdfa: 'http://www.w3.org/ns/rdfa#', + rdfs: 'http://www.w3.org/2000/01/rdf-schema#', + schema: 'https://schema.org/', + skos: 'http://www.w3.org/2004/02/skos/core#', + xhv: XHTML, + xsd: 'http://www.w3.org/2001/XMLSchema#', +} +/** Initial XHTML RDFa terms available without an explicit prefix declaration. */ +const TERMS = new Set([ + 'alternate', + 'appendix', + 'bookmark', + 'chapter', + 'contents', + 'copyright', + 'first', + 'glossary', + 'help', + 'icon', + 'index', + 'last', + 'license', + 'meta', + 'next', + 'p3pv1', + 'prev', + 'role', + 'section', + 'stylesheet', + 'subsection', + 'start', + 'top', + 'up', +]) +/** RDFa host-language profile. */ export type ProfileType = + | '' + | 'core' + | 'html' + | 'xhtml' + | 'svg' + | 'xml' +/** Known host-language media types. */ export type ContentTypeType = | 'text/html' | 'application/xhtml+xml' | 'application/xml' | 'text/xml' | 'image/svg+xml' - -/** RDFa feature switches exposed by the upstream processor. */ -export type FeatureType = Readonly> - -/** Options for RDFa 1.1 parsing. */ -export interface ParseOptionsType { +/** Optional host-language feature flags. */ export type FeatureType = Readonly< + Record +> +/** Native RDFa parser options. */ export interface ParseOptionsType { + /** Effective base IRI used while resolving relative identifiers during RDFa parsing. */ readonly base?: string - readonly graph?: Graph - readonly language?: string - readonly vocab?: string - /** Host-language content type. Prefer this over a manual profile when known. */ - readonly contentType?: ContentType - /** Explicit RDFa host-language profile when a content type is unavailable. */ - readonly profile?: ProfileType - /** Fine-grained processor features for specialist integrations. */ - readonly features?: FeatureType - /** External parser injection used by tests or alternate conforming implementations. */ - readonly parser?: ParserConstructorType - readonly signal?: AbortSignal -} - + /** Target graph. */ readonly graph?: GraphTermType + /** Initial language. */ readonly language?: string + /** Initial default vocabulary. */ readonly vocab?: string + /** Host media type. */ readonly contentType?: ContentTypeType + /** Explicit profile. */ readonly profile?: ProfileType + /** Reserved feature switches. */ readonly features?: FeatureType + /** Max decoded bytes. */ readonly maxBytes?: number + /** Max nodes. */ readonly maxNodes?: number + /** Max depth. */ readonly maxDepth?: number + /** Caller cancellation. */ readonly signal?: AbortSignal +} +/** Direction used when materializing `rel` and `rev` relationships. */ +type DirectionType = 'forward' | 'reverse' +/** Relationship waiting for a descendant resource. */ interface IncompleteType { + /** Expanded predicate IRI represented by this pending RDFa relationship. */ + readonly predicate: string + /** Relationship direction. */ readonly direction: DirectionType +} +/** Accumulated RDFa list. */ interface ListType { + /** RDF subject that owns this accumulated RDFa list. */ + readonly subject: SubjectTermType + /** Predicate IRI. */ readonly predicate: string + /** Values in source order. */ readonly values: ObjectTermType[] +} +/** Recursive RDFa evaluation context. */ interface ContextType { + /** Effective base IRI inherited by this RDFa evaluation context. */ + readonly base?: string + /** Parent subject. */ readonly parentSubject?: SubjectTermType + /** Parent object used for chaining. */ readonly parentObject?: SubjectTermType + /** Pending parent relations. */ readonly incomplete: readonly IncompleteType[] + /** Open list mappings. */ readonly lists: Map + /** Prefix mappings. */ readonly prefixes: ReadonlyMap + /** Initial term mappings. */ readonly terms: ReadonlyMap + /** Default vocabulary. */ readonly vocab?: string + /** Inherited language. */ readonly language?: string +} +/** State owned by one parse operation. */ interface StateType { + /** RDF graph that receives quads produced by this RDFa parse operation. */ + readonly graph: GraphTermType + /** Deterministic output. */ readonly quads: Quad[] + /** HTML host rules. */ readonly html: boolean + /** Original source for XML literals. */ readonly text: string + /** Caller cancellation. */ readonly signal?: AbortSignal +} /** - * Parses RDFa 1.1 incrementally into native `@okikio/rdf` quads. - * - * The RDFa implementation stays behind this subpath because its HTML parser and - * Node-style stream dependencies should not enter the root RDF module graph. + * Parses RDFa 1.1 to native quads using package-owned markup events and state. + * Prefix/CURIE expansion, vocabularies, relation chaining, reverse relations, + * `typeof`, typed values, and `inlist` collections are handled directly. */ -export async function* parse(source: TextSource, options: ParseOptionsType = {}): AsyncGenerator { - throwIfAborted(options.signal) - const Parser = options.parser ?? await defaultParser() - const parser = new Parser(parserOptions(options)) - yield* parseTransform(parser, source, { label: 'RDFa parser', ...(options.signal ? { signal: options.signal } : {}) }) -} - -/** Builds the upstream RDFa options while omitting absent optional fields. */ -function parserOptions(options: ParseOptionsType): Readonly> { - return { - dataFactory: factory, - ...(options.base === undefined ? {} : { baseIRI: options.base }), - ...(options.graph === undefined ? {} : { defaultGraph: options.graph }), - ...(options.language === undefined ? {} : { language: options.language }), - ...(options.vocab === undefined ? {} : { vocab: options.vocab }), - ...(options.contentType === undefined ? {} : { contentType: options.contentType }), - ...(options.profile === undefined ? {} : { profile: options.profile }), - ...(options.features === undefined ? {} : { features: options.features }), - } -} - -/** Lazily resolved RDFa parser constructor so importing the subpath does not initialize the optional processor. */ -let parserPromise: Promise | undefined - -/** Lazily imports the RDFa implementation only when this subpath is used. */ -async function defaultParser(): Promise { - parserPromise ??= import('rdfa-streaming-parser').then((module) => module.RdfaParser as unknown as ParserConstructorType) - return await parserPromise +export async function* parse( + source: TextSourceType, + options: ParseOptionsType = {}, +): AsyncGenerator { + const html = profile(options) === 'html' + const document = await parseMarkup(source, { + html, + ...(options.maxBytes === undefined ? {} : { maxBytes: options.maxBytes }), + ...(options.maxNodes === undefined ? {} : { maxNodes: options.maxNodes }), + ...(options.maxDepth === undefined ? {} : { maxDepth: options.maxDepth }), + ...(options.signal ? { signal: options.signal } : {}), + }) + const roots = document.children.filter((v): v is MarkupElementType => v.kind === 'element') + const base = hostBase(roots, options.base, html) + const prefixes = new Map(Object.entries(PREFIXES)), terms = new Map() + if (html) { for (const term of TERMS) terms.set(term, `${XHTML}${term}`) } + const parentSubject = base ? namedNode(base) : undefined + const context: ContextType = { + ...(base ? { base } : {}), + ...(parentSubject ? { parentSubject } : {}), + incomplete: [], + lists: new Map(), + prefixes, + terms, + ...(options.vocab ? { vocab: options.vocab } : {}), + ...(options.language ? { language: options.language.toLowerCase() } : {}), + } + const state: StateType = { + graph: options.graph ?? defaultGraph(), + quads: [], + html, + text: document.text, + ...(options.signal ? { signal: options.signal } : {}), + } + for (const root of roots) visit(root, context, state, true) + finishLists(context.lists, state) + for (const q of state.quads) { + throwIfAborted(options.signal) + yield q + } +} +/** Processes one element through the recursive RDFa evaluation sequence. */ function visit( + element: MarkupElementType, + inherited: ContextType, + state: StateType, + root = false, +): void { + throwIfAborted(state.signal) + const prefixes = mappings(element, inherited.prefixes, state.html), + base = xmlBase(element, inherited.base, state.html), + language = localLanguage(element, state.html) ?? inherited.language, + vocabRaw = attr(element, 'vocab', state.html), + vocab = vocabRaw === undefined + ? inherited.vocab + : vocabRaw === '' + ? undefined + : iri(vocabRaw, base) + if (vocabRaw !== undefined && vocab && base) { + emit(state, namedNode(base), USES_VOCABULARY, namedNode(vocab)) + } + const local: ContextType = { + ...(base ? { base } : {}), + ...(inherited.parentSubject ? { parentSubject: inherited.parentSubject } : {}), + ...(inherited.parentObject ? { parentObject: inherited.parentObject } : {}), + incomplete: inherited.incomplete, + lists: inherited.lists, + prefixes, + terms: inherited.terms, + ...(vocab ? { vocab } : {}), + ...(language ? { language } : {}), + } + const rel = predicates(attr(element, 'rel', state.html), local), + rev = predicates(attr(element, 'rev', state.html), local), + property = predicates(attr(element, 'property', state.html), local), + types = predicates(attr(element, 'typeof', state.html), local), + hasRelation = rel.length > 0 || rev.length > 0, + about = resource(attr(element, 'about', state.html), local, false), + resourceValue = resource(attr(element, 'resource', state.html), local, false), + href = iriTerm(attr(element, 'href', state.html), base), + src = iriTerm(attr(element, 'src', state.html), base) + let subject: SubjectTermType | undefined, + object: SubjectTermType | undefined, + typed: SubjectTermType | undefined, + skip = false + if (!hasRelation) { + if ( + property.length && !hasAttr(element, 'content', state.html) && + !hasAttr(element, 'datatype', state.html) + ) { + subject = about ?? (root ? documentTerm(base) : inherited.parentObject) + if (types.length) { + typed = about ?? (root ? documentTerm(base) : resourceValue ?? href ?? src ?? blankNode()) + object = typed + } + } else { + subject = about ?? resourceValue ?? href ?? src + if (!subject) { + if (root) subject = documentTerm(base) + else if (types.length) subject = blankNode() + else if (inherited.parentObject) { + subject = inherited.parentObject + if (!property.length) skip = true + } + } + if (types.length) typed = subject + } + } else { + subject = about ?? (root ? documentTerm(base) : inherited.parentObject) + if (types.length && about) typed = subject + object = resourceValue ?? href ?? src + if (!object && types.length && !about) object = blankNode() + if (types.length && !about) typed = object + } + if (typed) { for (const type of types) emit(state, typed, RDF.type, namedNode(type)) } + let lists = inherited.lists + if ( + subject && + ((inherited.parentObject && !subject.equals(inherited.parentObject)) || + (!inherited.parentObject && inherited.parentSubject && + !subject.equals(inherited.parentSubject))) + ) lists = new Map() + const pending: IncompleteType[] = [] + if (subject && object) { + if (hasAttr(element, 'inlist', state.html)) { + for (const p of rel) addList(lists, subject, p, object) + } else for (const p of rel) emit(state, subject, p, object) + for (const p of rev) emit(state, object, p, subject) + } else if (subject && hasRelation) { + for (const p of rel) pending.push({ predicate: p, direction: 'forward' }) + for (const p of rev) pending.push({ predicate: p, direction: 'reverse' }) + object = blankNode() + } + if (property.length && subject) { + const value = propertyObject(element, local, state, typed) + if (value) { + if (hasAttr(element, 'inlist', state.html)) { + for (const p of property) addList(lists, subject, p, value) + } else for (const p of property) emit(state, subject, p, value) + } + } + if (!skip && subject) complete(inherited, subject, state) + const child: ContextType = skip + ? { ...inherited, prefixes, ...(language ? { language } : {}), ...(vocab ? { vocab } : {}) } + : { + ...(base ? { base } : {}), + ...(subject ?? inherited.parentSubject + ? { parentSubject: subject ?? inherited.parentSubject } + : {}), + ...(object ?? subject ?? inherited.parentSubject + ? { parentObject: object ?? subject ?? inherited.parentSubject } + : {}), + incomplete: pending, + lists, + prefixes, + terms: inherited.terms, + ...(vocab ? { vocab } : {}), + ...(language ? { language } : {}), + } + for (const c of element.children) if (c.kind === 'element') visit(c, child, state) + if (lists !== inherited.lists) finishLists(lists, state) +} +/** Completes parent relationships once a descendant subject appears. */ function complete( + context: ContextType, + subject: SubjectTermType, + state: StateType, +) { + if (!context.parentSubject) return + for (const pending of context.incomplete) { + pending.direction === 'forward' + ? emit(state, context.parentSubject, pending.predicate, subject) + : emit(state, subject, pending.predicate, context.parentSubject) + } +} +/** Resolves a property value through RDFa literal/resource precedence. */ function propertyObject( + element: MarkupElementType, + context: ContextType, + state: StateType, + typed?: SubjectTermType, +): ObjectTermType | undefined { + const content = attr(element, 'content', state.html), + raw = attr(element, 'datatype', state.html), + datatype = raw === undefined || raw === '' ? undefined : term(raw, context) + if (raw !== undefined && raw !== '' && datatype && datatype !== XML_LITERAL) { + return literal(content ?? textContent(element), namedNode(datatype)) + } + if (raw === '') { + return context.language + ? literal(content ?? textContent(element), context.language) + : literal(content ?? textContent(element)) + } + if (datatype === XML_LITERAL) return literal(xmlChildren(element, state), namedNode(XML_LITERAL)) + if (content !== undefined) { + return context.language ? literal(content, context.language) : literal(content) + } + if (!hasAttr(element, 'rel', state.html) && !hasAttr(element, 'rev', state.html)) { + const resourceValue = resource(attr(element, 'resource', state.html), context, false) ?? + iriTerm(attr(element, 'href', state.html), context.base) ?? + iriTerm(attr(element, 'src', state.html), context.base) + if (resourceValue) return resourceValue + } + if (hasAttr(element, 'typeof', state.html) && !hasAttr(element, 'about', state.html) && typed) { + return typed + } + const value = textContent(element) + return context.language ? literal(value, context.language) : literal(value) +} +/** Adds a list value. */ function addList( + lists: Map, + subject: SubjectTermType, + predicate: string, + object: ObjectTermType, +) { + const key = `${subject.termType}:${subject.value}\0${predicate}` + let list = lists.get(key) + if (!list) { + list = { subject, predicate, values: [] } + lists.set(key, list) + } + list.values.push(object) +} +/** Emits accumulated RDF collections. */ function finishLists( + lists: Map, + state: StateType, +) { + for (const list of lists.values()) { + if (!list.values.length) { + emit(state, list.subject, list.predicate, namedNode(RDF.nil)) + continue + } + const head = blankNode() + emit(state, list.subject, list.predicate, head) + let cursor = head + for (let i = 0; i < list.values.length; i++) { + emit(state, cursor, RDF.first, list.values[i]!) + const last = i === list.values.length - 1, next = last ? namedNode(RDF.nil) : blankNode() + emit(state, cursor, RDF.rest, next) + if (!last) cursor = next as ReturnType + } + } + lists.clear() +} +/** Appends one statement. */ function emit( + state: StateType, + subject: SubjectTermType, + predicate: string, + object: ObjectTermType, +) { + state.quads.push(quad(subject, namedNode(predicate), object, state.graph)) +} +/** Builds local prefix mappings. */ function mappings( + element: MarkupElementType, + inherited: ReadonlyMap, + html: boolean, +) { + const values = new Map(inherited) + for (const a of element.attributes) { + const lower = a.name.toLowerCase() + if (lower.startsWith('xmlns:')) values.set(lower.slice(6), a.value) + } + const prefix = attr(element, 'prefix', html) + if (prefix) { + for (const m of prefix.matchAll(/([^\s:]+):\s+([^\s]+)/gu)) { + values.set(m[1]!.toLowerCase(), m[2]!) + } + } + return values +} +/** Expands a predicate/token list. */ function predicates( + value: string | undefined, + context: ContextType, +) { + return value?.trim().split(/\s+/u).filter(Boolean).map((v) => term(v, context)).filter(( + v, + ): v is string => Boolean(v)) ?? [] +} +/** Expands one term, CURIE, or absolute IRI. */ function term( + value: string, + context: ContextType, +): string | undefined { + const input = safe(value) + if (!input) return undefined + if (!input.includes(':') && /^[A-Za-z_][A-Za-z0-9._\-/]*$/u.test(input)) { + if (context.vocab) return `${context.vocab}${input}` + return context.terms.get(input) ?? context.terms.get(input.toLowerCase()) + } + const curie = expandCurie(input, context) + if (curie?.termType === 'NamedNode') return curie.value + return absolute(input) +} +/** Resolves SafeCURIE/CURIE/blank node/IRI resources. */ function resource( + value: string | undefined, + context: ContextType, + termAllowed: boolean, +): SubjectTermType | undefined { + if (value === undefined) return undefined + const input = safe(value) + if (!input) return undefined + if (termAllowed && !input.includes(':')) { + const expanded = term(input, context) + return expanded ? namedNode(expanded) : undefined + } + return expandCurie(input, context) ?? iriTerm(input, context.base) +} +/** Expands one CURIE. */ function expandCurie( + value: string, + context: ContextType, +): SubjectTermType | undefined { + const colon = value.indexOf(':') + if (colon < 0) return undefined + const prefix = value.slice(0, colon).toLowerCase(), ref = value.slice(colon + 1) + if (prefix === '_') return blankNode(ref) + const base = context.prefixes.get(prefix) + return base === undefined ? undefined : namedNode(`${base}${ref}`) +} +/** Resolves ordinary IRI reference. */ function iriTerm(value: string | undefined, base?: string) { + if (value === undefined) return undefined + const resolved = iri(value, base) + return resolved ? namedNode(resolved) : undefined +} +/** Resolves an IRI reference. */ function iri(value: string, base?: string) { + try { + return base ? new URL(value, base).href : new URL(value).href + } catch { + return undefined + } +} +/** Returns only absolute IRI values. */ function absolute(value: string) { + try { + return new URL(value).href + } catch { + return undefined + } +} +/** Removes SafeCURIE brackets. */ function safe(value: string) { + const input = value.trim() + if (!input.startsWith('[')) return input + if (!input.endsWith(']')) return undefined + return input.slice(1, -1).trim() || undefined +} +/** Applies xml:base outside HTML. */ function xmlBase( + element: MarkupElementType, + inherited: string | undefined, + html: boolean, +) { + if (html) return inherited + const value = attr(element, 'xml:base', false) + return value === undefined ? inherited : iri(value, inherited) +} +/** Gets local language. */ function localLanguage(element: MarkupElementType, html: boolean) { + const value = attr(element, 'xml:lang', html) ?? attr(element, 'lang', html) + return value?.trim().toLowerCase() || undefined +} +/** Selects host profile. */ function profile(options: ParseOptionsType): ProfileType { + if (options.profile) return options.profile + if (options.contentType === 'application/xhtml+xml') return 'xhtml' + if (options.contentType === 'image/svg+xml') return 'svg' + if (options.contentType === 'application/xml' || options.contentType === 'text/xml') return 'xml' + return 'html' +} +/** Applies first HTML base[href]. */ function hostBase( + roots: readonly MarkupElementType[], + supplied: string | undefined, + html: boolean, +) { + if (!html) return supplied + for (const root of roots) { + for (const e of elements(root, true)) { + if (e.name === 'base') { + const href = attr(e, 'href', true) + if (href !== undefined) return iri(href, supplied) + } + } + } + return supplied +} +/** Returns implicit document subject. */ function documentTerm(base?: string) { + return base ? namedNode(base) : undefined +} +/** Returns original descendant markup for XMLLiteral. */ function xmlChildren( + element: MarkupElementType, + state: StateType, +) { + if (!element.children.length) return '' + return state.text.slice(element.children[0]!.start, element.children.at(-1)!.end) } diff --git a/packages/rdf/rdfa/mod_test.ts b/packages/rdf/rdfa/mod_test.ts index b701c50..c42fae7 100644 --- a/packages/rdf/rdfa/mod_test.ts +++ b/packages/rdf/rdfa/mod_test.ts @@ -1,50 +1,58 @@ import { describe, it } from 'node:test' import { expect } from '@std/expect' -import { literal, namedNode, quad, type Quad } from '../mod.ts' -import { parse, type ParserType } from './mod.ts' +import { RDF } from '../mod.ts' +import { parse } from './mod.ts' -const fixture = quad(namedNode('https://example.test/s'), namedNode('https://example.test/p'), literal('value')) -type EventType = 'drain' | 'close' | 'error' -type ListenerType = (...args: unknown[]) => void - -class TestParser implements ParserType { - static options: Readonly> | undefined - private readonly listeners = new Map>() - private ended = false - private resolveEnd: (() => void) | undefined - constructor(options: Readonly>) { TestParser.options = options } - write(_value: string | Uint8Array): boolean { return true } - end(): void { this.ended = true; this.resolveEnd?.() } - destroy(error?: Error): void { if (error) this.emit('error', error); this.ended = true; this.resolveEnd?.(); this.emit('close') } - once(event: EventType, listener: ListenerType): this { const values = this.listeners.get(event) ?? new Set(); values.add(listener); this.listeners.set(event, values); return this } - off(event: EventType, listener: ListenerType): this { this.listeners.get(event)?.delete(listener); return this } - private emit(event: EventType, ...args: unknown[]): void { for (const listener of this.listeners.get(event) ?? []) listener(...args) } - async *[Symbol.asyncIterator](): AsyncIterator { if (!this.ended) await new Promise((resolve) => { this.resolveEnd = resolve }); yield fixture } +async function all(source: string, options = {}) { + const values = [] + for await (const value of parse(source, options)) values.push(value) + return values } describe('@okikio/rdf/rdfa', () => { - it('adapts RDFa parser output back to native RDF terms', async () => { - const values: Quad[] = [] - for await (const value of parse('
', { parser: TestParser, contentType: 'text/html' })) values.push(value) - expect(values).toHaveLength(1) - expect(values[0]?.equals(fixture)).toBe(true) + it('extracts vocab properties, typeof, and resource relationships natively', async () => { + const html = + `
AliceB
` + const values = await all(html, { contentType: 'text/html' }) + expect( + values.some((value) => + value.predicate.value === RDF.type && value.object.value === 'https://schema.org/Person' + ), + ).toBe(true) + expect( + values.some((value) => + value.predicate.value === 'https://schema.org/name' && value.object.value === 'Alice' + ), + ).toBe(true) + expect( + values.some((value) => + value.predicate.value === 'https://schema.org/knows' && + value.object.value === 'https://example.test/b' + ), + ).toBe(true) }) - it('forwards RDFa host-language options and the native RDF factory', async () => { - for await (const _value of parse('
', { - parser: TestParser, - base: 'https://example.test/base/', - contentType: 'text/html', - language: 'en', - vocab: 'https://schema.org/', - })) { /* drain */ } - - expect(TestParser.options?.baseIRI).toBe('https://example.test/base/') - expect(TestParser.options?.contentType).toBe('text/html') - expect(TestParser.options?.language).toBe('en') - expect(TestParser.options?.vocab).toBe('https://schema.org/') - const factory = TestParser.options?.dataFactory as { namedNode(value: string): { value: string } } - expect(factory.namedNode('urn:test').value).toBe('urn:test') + it('expands declared prefixes and reverse relations', async () => { + const html = + `
AliceB
` + const values = await all(html) + expect(values.some((value) => value.predicate.value === 'http://xmlns.com/foaf/0.1/name')).toBe( + true, + ) + expect( + values.some((value) => + value.subject.value === 'https://example.test/b' && + value.predicate.value === 'http://xmlns.com/foaf/0.1/knows' && + value.object.value === 'https://example.test/a' + ), + ).toBe(true) }) + it('emits @inlist relationships as RDF collections', async () => { + const html = + `
` + const values = await all(html) + expect(values.some((value) => value.predicate.value === RDF.first)).toBe(true) + expect(values.some((value) => value.predicate.value === RDF.rest)).toBe(true) + }) }) diff --git a/packages/rdf/rdfa/types.ts b/packages/rdf/rdfa/types.ts deleted file mode 100644 index 5188939..0000000 --- a/packages/rdf/rdfa/types.ts +++ /dev/null @@ -1,11 +0,0 @@ -/** Structural RDFa parser contracts used to isolate the external implementation. @module */ - -import type { TransformParserType } from '../transform.ts' - -/** RDFa parser stream shape required by the adapter. */ -export type ParserType = TransformParserType - -/** Constructor contract for an RDFa parser implementation. */ -export interface ParserConstructorType { - new (options: Readonly>): ParserType -} diff --git a/packages/rdf/shape/index.ts b/packages/rdf/shape/index.ts index e5eafc4..da3a52f 100644 --- a/packages/rdf/shape/index.ts +++ b/packages/rdf/shape/index.ts @@ -1,12 +1,15 @@ /** Indexed view over one materialized SHACL shapes graph. @module */ import { key } from '../term.ts' -import type { ObjectTerm, Quad, Subject, Term } from '../term.ts' +import type { ObjectTermType, Quad, SubjectTermType, Term } from '../term.ts' -/** Subject/predicate index used by the SHACL reader and property-path parser. */ +/** Subject/predicate index used by the SHACL inspector and property-path parser. */ export class ShapeIndex { - readonly #subjects = new Map() - readonly #values = new Map>() + /** SHACL index of materialized subject terms discovered in the shapes graph. */ + readonly #subjects = new Map() + /** SHACL predicate/value index used for repeated direct lookups during shape inspection. */ + readonly #values = new Map>() + /** Primary semantic-key map containing the quads currently owned by this dataset. */ readonly #quads = new Map() /** Adds one quad to the index. */ @@ -35,24 +38,24 @@ export class ShapeIndex { } /** Returns each indexed subject. */ - subjects(): Iterable { + subjects(): Iterable { return this.#subjects.values() } /** Returns predicate values for one RDF subject. */ - get(subject: Subject, predicate: string): readonly ObjectTerm[] { + get(subject: SubjectTermType, predicate: string): readonly ObjectTermType[] { return this.#values.get(key(subject))?.get(predicate) ?? [] } /** Returns all quads for one RDF subject. */ - quads(subject: Subject): readonly Quad[] { + quads(subject: SubjectTermType): readonly Quad[] { return this.#quads.get(key(subject)) ?? [] } /** Returns whether a node is an RDF list cell. */ isList(term: Term): boolean { if (term.termType !== 'NamedNode' && term.termType !== 'BlankNode') return false - const subject = term as Subject + const subject = term as SubjectTermType return this.get(subject, RDF_FIRST).length > 0 || this.get(subject, RDF_REST).length > 0 } } diff --git a/packages/rdf/shape/read.ts b/packages/rdf/shape/inspect.ts similarity index 61% rename from packages/rdf/shape/read.ts rename to packages/rdf/shape/inspect.ts index f230e05..271dca0 100644 --- a/packages/rdf/shape/read.ts +++ b/packages/rdf/shape/inspect.ts @@ -1,7 +1,7 @@ /** - * Loss-preserving SHACL Core shapes-graph reader. + * Loss-preserving SHACL Core shapes-graph inspector. * - * The reader materializes the shapes graph because SHACL lists and property + * The inspector materializes the shapes graph because SHACL lists and property * paths require random access. It does not validate data graphs and it does not * evaluate SHACL 1.2 Node Expressions. Known Core statements are normalized * into the semantic model; unsupported or malformed statements remain in the @@ -12,9 +12,9 @@ import { iterate } from '../source.ts' import { key, RDF, XSD } from '../term.ts' -import type { Literal, ObjectTerm, Quad, Subject } from '../term.ts' +import type { Literal, ObjectTermType, Quad, SubjectTermType } from '../term.ts' import { ShapeIndex } from './index.ts' -import { readList } from './list.ts' +import { getList } from './list.ts' import type { AssertionType, ConstraintType, @@ -30,12 +30,12 @@ import type { TextType, VersionType, } from './model.ts' -import { readPath } from './path.ts' +import { getPath } from './path.ts' import { assertion, id, literal, term, text } from './value.ts' /** SHACL namespace used to identify Core shapes, targets, constraints, and metadata. */ const SH = 'http://www.w3.org/ns/shacl#' -/** RDFS namespace used for label/comment metadata understood by the shape reader. */ +/** RDFS namespace used for label/comment metadata understood by the shape inspector. */ const RDFS = 'http://www.w3.org/2000/01/rdf-schema#' /** IRI identifying explicit SHACL node shapes. */ @@ -49,31 +49,96 @@ const BY_TYPES = `${SH}ByTypes` /** Known SHACL predicates that are sufficient evidence to discover an implicit shape resource. */ const SHAPE_PREDICATES = new Set([ - `${SH}path`, `${SH}targetNode`, `${SH}targetClass`, `${SH}targetSubjectsOf`, `${SH}targetObjectsOf`, - `${SH}targetWhere`, `${SH}shape`, `${SH}severity`, `${SH}message`, `${SH}deactivated`, - `${SH}class`, `${SH}datatype`, `${SH}nodeKind`, `${SH}minCount`, `${SH}maxCount`, - `${SH}minExclusive`, `${SH}minInclusive`, `${SH}maxExclusive`, `${SH}maxInclusive`, - `${SH}minLength`, `${SH}maxLength`, `${SH}pattern`, `${SH}flags`, `${SH}singleLine`, - `${SH}languageIn`, `${SH}uniqueLang`, `${SH}memberShape`, `${SH}minListLength`, `${SH}maxListLength`, - `${SH}uniqueMembers`, `${SH}equals`, `${SH}disjoint`, `${SH}subsetOf`, `${SH}lessThan`, - `${SH}lessThanOrEquals`, `${SH}not`, `${SH}and`, `${SH}or`, `${SH}xone`, `${SH}node`, `${SH}property`, - `${SH}someValue`, `${SH}qualifiedValueShape`, `${SH}qualifiedMinCount`, `${SH}qualifiedMaxCount`, - `${SH}qualifiedValueShapesDisjoint`, `${SH}reifierShape`, `${SH}reificationRequired`, `${SH}closed`, - `${SH}ignoredProperties`, `${SH}hasValue`, `${SH}in`, `${SH}rootClass`, `${SH}uniqueValuesFor`, - `${SH}name`, `${SH}description`, `${SH}intent`, `${SH}agentInstruction`, `${SH}codeIdentifier`, - `${SH}unit`, `${SH}order`, `${SH}group`, `${SH}values`, `${SH}defaultValue`, + `${SH}path`, + `${SH}targetNode`, + `${SH}targetClass`, + `${SH}targetSubjectsOf`, + `${SH}targetObjectsOf`, + `${SH}targetWhere`, + `${SH}shape`, + `${SH}severity`, + `${SH}message`, + `${SH}deactivated`, + `${SH}class`, + `${SH}datatype`, + `${SH}nodeKind`, + `${SH}minCount`, + `${SH}maxCount`, + `${SH}minExclusive`, + `${SH}minInclusive`, + `${SH}maxExclusive`, + `${SH}maxInclusive`, + `${SH}minLength`, + `${SH}maxLength`, + `${SH}pattern`, + `${SH}flags`, + `${SH}singleLine`, + `${SH}languageIn`, + `${SH}uniqueLang`, + `${SH}memberShape`, + `${SH}minListLength`, + `${SH}maxListLength`, + `${SH}uniqueMembers`, + `${SH}equals`, + `${SH}disjoint`, + `${SH}subsetOf`, + `${SH}lessThan`, + `${SH}lessThanOrEquals`, + `${SH}not`, + `${SH}and`, + `${SH}or`, + `${SH}xone`, + `${SH}node`, + `${SH}property`, + `${SH}someValue`, + `${SH}qualifiedValueShape`, + `${SH}qualifiedMinCount`, + `${SH}qualifiedMaxCount`, + `${SH}qualifiedValueShapesDisjoint`, + `${SH}reifierShape`, + `${SH}reificationRequired`, + `${SH}closed`, + `${SH}ignoredProperties`, + `${SH}hasValue`, + `${SH}in`, + `${SH}rootClass`, + `${SH}uniqueValuesFor`, + `${SH}name`, + `${SH}description`, + `${SH}intent`, + `${SH}agentInstruction`, + `${SH}codeIdentifier`, + `${SH}unit`, + `${SH}order`, + `${SH}group`, + `${SH}values`, + `${SH}defaultValue`, ]) -/** Core predicates that require an `unsupported-version` diagnostic when read in SHACL 1.0 mode. */ +/** Core predicates that require an `unsupported-version` diagnostic when inspected in SHACL 1.0 mode. */ const SHACL_12_PREDICATES = new Set([ - `${SH}targetWhere`, `${SH}shape`, `${SH}singleLine`, `${SH}memberShape`, `${SH}minListLength`, - `${SH}maxListLength`, `${SH}uniqueMembers`, `${SH}subsetOf`, `${SH}someValue`, `${SH}reifierShape`, - `${SH}reificationRequired`, `${SH}rootClass`, `${SH}uniqueValuesFor`, `${SH}intent`, `${SH}agentInstruction`, - `${SH}codeIdentifier`, `${SH}values`, `${SH}defaultValue`, + `${SH}targetWhere`, + `${SH}shape`, + `${SH}singleLine`, + `${SH}memberShape`, + `${SH}minListLength`, + `${SH}maxListLength`, + `${SH}uniqueMembers`, + `${SH}subsetOf`, + `${SH}someValue`, + `${SH}reifierShape`, + `${SH}reificationRequired`, + `${SH}rootClass`, + `${SH}uniqueValuesFor`, + `${SH}intent`, + `${SH}agentInstruction`, + `${SH}codeIdentifier`, + `${SH}values`, + `${SH}defaultValue`, ]) -/** Reader resource limits and draft-version interpretation. */ -export interface ReadOptions { +/** Inspector resource limits and draft-version interpretation. */ +export interface InspectOptionsType { /** Core vocabulary generation to interpret. Default is the current 1.2 draft. */ readonly version?: VersionType /** Maximum number of quads to materialize. Default is 1,000,000. */ @@ -82,32 +147,58 @@ export interface ReadOptions { readonly maxListItems?: number /** Maximum nested property-path depth. Default is 256. */ readonly maxPathDepth?: number + /** Caller-owned abort signal checked before expensive work and between long-running steps. */ readonly signal?: AbortSignal } -/** Shared materialized-shape read state carrying version policy, random-access index, diagnostics, and recursion/list limits. */ +/** Shared materialized-shape inspection state carrying version policy, random-access index, diagnostics, and recursion/list limits. */ interface StateType { + /** SHACL Core version selected for this inspection. */ readonly version: VersionType + /** Random-access semantic index used by the inspector instead of repeatedly scanning the complete graph. */ readonly index: ShapeIndex + /** Structured diagnostics retained so recoverable source information is not silently discarded. */ readonly diagnostics: DiagnosticType[] + /** Maximum RDF-list members followed from one list before inspection reports a limit. */ readonly maxListItems: number + /** Maximum recursive SHACL path depth followed before inspection reports a limit. */ readonly maxPathDepth: number } -/** Read from the supplied source while preserving caller ownership. */ -export async function read( +/** + * Inspects the supplied shapes graph without performing SHACL validation. + * + * The inspector materializes at most `maxQuads`, follows bounded RDF lists and + * property paths, and preserves unsupported assertions for later evaluators. + * The caller retains ownership of the source iterable and abort signal. + * + * @example + * ```ts + * import * as shape from '@okikio/rdf/shape' + * + * const graph = await shape.inspect(quads, { version: '1.0' }) + * for (const value of graph.shapes) console.log(value.id) + * ``` + */ +export async function inspect( source: Iterable | AsyncIterable, - options: ReadOptions = {}, + options: InspectOptionsType = {}, ): Promise { const version = options.version ?? '1.2' - if (version !== '1.0' && version !== '1.2') throw new TypeError(`Unsupported SHACL version '${String(version)}'.`) + if (version !== '1.0' && version !== '1.2') { + throw new TypeError(`Unsupported SHACL version '${String(version)}'.`) + } const maxQuads = options.maxQuads ?? 1_000_000 const index = new ShapeIndex() const quads: Quad[] = [] for await (const quad of iterate(source)) { - if (options.signal?.aborted) throw options.signal.reason ?? new DOMException('Aborted', 'AbortError') - if (quads.length >= maxQuads) throw new RangeError(`SHACL shapes graph exceeds the configured ${maxQuads} quad limit.`) + if (options.signal?.aborted) { + throw options.signal.reason ?? new DOMException('Aborted', 'AbortError') + } + if (quads.length >= maxQuads) { + throw new RangeError(`SHACL shapes graph exceeds the configured ${maxQuads} quad limit.`) + } quads.push(quad) index.add(quad) } @@ -125,7 +216,7 @@ export async function read( const shapeKeys = new Set() for (const subject of candidates) { - const shape = readShape(state, subject) + const shape = createShape(state, subject) if (!shape) continue shapes.push(shape) shapeKeys.add(key(subject)) @@ -146,25 +237,31 @@ export async function read( } /** Discovers resources with SHACL type, target, path, or constraint evidence without treating arbitrary labelled ontology resources as shapes. */ -function discoverShapes(index: ShapeIndex): Subject[] { - const values = new Map() +function discoverShapes(index: ShapeIndex): SubjectTermType[] { + const values = new Map() for (const subject of index.subjects()) { const types = index.get(subject, RDF.type) const explicit = types.some((value) => value.termType === 'NamedNode' && (value.value === NODE_SHAPE || value.value === PROPERTY_SHAPE || value.value === SHAPE_CLASS) ) - const hasShapePredicate = index.quads(subject).some((quad) => SHAPE_PREDICATES.has(quad.predicate.value)) + const hasShapePredicate = index.quads(subject).some((quad) => + SHAPE_PREDICATES.has(quad.predicate.value) + ) if (explicit || hasShapePredicate) values.set(key(subject), subject) } return [...values.values()].sort((left, right) => key(left).localeCompare(key(right))) } /** Read shape from the supplied source while preserving caller ownership. */ -function readShape(state: StateType, subject: Subject): ShapeType | undefined { +function createShape(state: StateType, subject: SubjectTermType): ShapeType | undefined { const shapeId = id(subject) if (!shapeId) { - state.diagnostics.push({ code: 'invalid-shape-id', severity: 'error', message: 'SHACL shape identifier must be an IRI or blank node.' }) + state.diagnostics.push({ + code: 'invalid-shape-id', + severity: 'error', + message: 'SHACL shape identifier must be an IRI or blank node.', + }) return undefined } @@ -175,45 +272,75 @@ function readShape(state: StateType, subject: Subject): ShapeType | undefined { .map((value) => value.value) .sort() const pathValues = state.index.get(subject, `${SH}path`) - const explicitProperty = state.index.get(subject, RDF.type).some((value) => value.termType === 'NamedNode' && value.value === PROPERTY_SHAPE) + const explicitProperty = state.index.get(subject, RDF.type).some((value) => + value.termType === 'NamedNode' && value.value === PROPERTY_SHAPE + ) const kind = explicitProperty || pathValues.length > 0 ? 'property' : 'node' let path: PathType | undefined if (pathValues.length > 0) { consumed.add(`${SH}path`) - if (pathValues.length !== 1) addCardinality(state, shapeId, `${SH}path`, 'A property shape must have at most one sh:path value.') - path = readPath(state.index, pathValues[0]!, pathOptions(state, shapeId, `${SH}path`)) + if (pathValues.length !== 1) { + addCardinality( + state, + shapeId, + `${SH}path`, + 'A property shape must have at most one sh:path value.', + ) + } + path = getPath(state.index, pathValues[0]!, pathOptions(state, shapeId, `${SH}path`)) } - const targets = readTargets(state, subject, shapeId, consumed) - const severity = readIriSingleton(state, subject, shapeId, `${SH}severity`, consumed) - const messages = readTextValues(state, subject, shapeId, `${SH}message`, consumed) - const deactivated = readTermValues(state, subject, `${SH}deactivated`, consumed) - const constraints = readConstraints(state, subject, shapeId, consumed) - const metadata = readMetadata(state, subject, shapeId, consumed) - const assertions = readAssertions(state, subject, consumed) + const targets = getTargets(state, subject, shapeId, consumed) + const severity = getIri(state, subject, shapeId, `${SH}severity`, consumed) + const messages = getTextValues(state, subject, shapeId, `${SH}message`, consumed) + const deactivated = getTermValues(state, subject, `${SH}deactivated`, consumed) + const constraints = getConstraints(state, subject, shapeId, consumed) + const metadata = getMetadata(state, subject, shapeId, consumed) + const assertions = getAssertions(state, subject, consumed) const value: { + /** RDF node that identifies the shape currently being normalized. */ id: IdType + /** Discriminates the concrete value variant. */ kind: 'node' | 'property' + /** Named rdf:type IRIs retained from the source shape. */ types: readonly string[] + /** Normalized SHACL property path when the source shape declares one. */ path?: PathType + /** Explicit target declarations collected for the shape. */ targets: readonly TargetType[] + /** Diagnostic severity used to decide whether inspection can continue. */ severity?: string + /** Localized validation messages attached to the shape. */ messages: readonly TextType[] + /** Loss-preserving deactivation expressions retained from the source graph. */ deactivated: readonly TermType[] + /** Normalized SHACL Core constraints attached to the shape. */ constraints: readonly ConstraintType[] + /** Non-validating metadata collected for the normalized shape. */ metadata: MetadataType + /** Source assertions not consumed by the current semantic inspector. */ assertions: readonly AssertionType[] - } = { id: shapeId, kind, types, targets, messages, deactivated, constraints, metadata, assertions } + } = { + id: shapeId, + kind, + types, + targets, + messages, + deactivated, + constraints, + metadata, + assertions, + } if (path) value.path = path if (severity) value.severity = severity return value } /** Read targets from the supplied source while preserving caller ownership. */ -function readTargets( +function getTargets( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, ): TargetType[] { @@ -221,7 +348,12 @@ function readTargets( for (const value of state.index.get(subject, `${SH}targetNode`)) { const record = term(value) if (record) targets.push({ kind: 'node', value: record }) - else addInvalid(state, shape, `${SH}targetNode`, 'sh:targetNode value cannot be represented as a shape term.') + else {addInvalid( + state, + shape, + `${SH}targetNode`, + 'sh:targetNode value cannot be represented as a shape term.', + )} } consumeWhenPresent(state, subject, consumed, `${SH}targetNode`) pushIriTargets(state, subject, shape, consumed, `${SH}targetClass`, 'class', targets) @@ -231,14 +363,24 @@ function readTargets( for (const value of state.index.get(subject, `${SH}targetWhere`)) { const record = term(value) if (record) targets.push({ kind: 'where', expression: record }) - else addInvalid(state, shape, `${SH}targetWhere`, 'sh:targetWhere value cannot be represented as an RDF term.') + else {addInvalid( + state, + shape, + `${SH}targetWhere`, + 'sh:targetWhere value cannot be represented as an RDF term.', + )} } consumeWhenPresent(state, subject, consumed, `${SH}targetWhere`) for (const value of state.index.get(subject, `${SH}shape`)) { const shapeValue = id(value) if (shapeValue) targets.push({ kind: 'shape', shape: shapeValue }) - else addInvalid(state, shape, `${SH}shape`, 'sh:shape target must reference an IRI or blank node.') + else {addInvalid( + state, + shape, + `${SH}shape`, + 'sh:shape target must reference an IRI or blank node.', + )} } consumeWhenPresent(state, subject, consumed, `${SH}shape`) return targets @@ -247,7 +389,7 @@ function readTargets( /** Reads every IRI-valued target assertion, retaining invalid values through diagnostics instead of silently coercing them. */ function pushIriTargets( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, predicate: string, @@ -262,9 +404,9 @@ function pushIriTargets( } /** Read constraints from the supplied source while preserving caller ownership. */ -function readConstraints( +function getConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, ): ConstraintType[] { @@ -273,27 +415,150 @@ function readConstraints( pushChoiceConstraints(state, subject, shape, consumed, `${SH}class`, 'class', constraints) pushChoiceConstraints(state, subject, shape, consumed, `${SH}datatype`, 'datatype', constraints) pushChoiceConstraints(state, subject, shape, consumed, `${SH}nodeKind`, 'nodeKind', constraints) - pushIntegerConstraints(state, subject, shape, consumed, `${SH}minCount`, 'minCount', 'count', constraints) - pushIntegerConstraints(state, subject, shape, consumed, `${SH}maxCount`, 'maxCount', 'count', constraints) - pushLiteralConstraints(state, subject, shape, consumed, `${SH}minExclusive`, 'minExclusive', constraints) - pushLiteralConstraints(state, subject, shape, consumed, `${SH}minInclusive`, 'minInclusive', constraints) - pushLiteralConstraints(state, subject, shape, consumed, `${SH}maxExclusive`, 'maxExclusive', constraints) - pushLiteralConstraints(state, subject, shape, consumed, `${SH}maxInclusive`, 'maxInclusive', constraints) - pushIntegerConstraints(state, subject, shape, consumed, `${SH}minLength`, 'minLength', 'length', constraints) - pushIntegerConstraints(state, subject, shape, consumed, `${SH}maxLength`, 'maxLength', 'length', constraints) + pushIntegerConstraints( + state, + subject, + shape, + consumed, + `${SH}minCount`, + 'minCount', + 'count', + constraints, + ) + pushIntegerConstraints( + state, + subject, + shape, + consumed, + `${SH}maxCount`, + 'maxCount', + 'count', + constraints, + ) + pushLiteralConstraints( + state, + subject, + shape, + consumed, + `${SH}minExclusive`, + 'minExclusive', + constraints, + ) + pushLiteralConstraints( + state, + subject, + shape, + consumed, + `${SH}minInclusive`, + 'minInclusive', + constraints, + ) + pushLiteralConstraints( + state, + subject, + shape, + consumed, + `${SH}maxExclusive`, + 'maxExclusive', + constraints, + ) + pushLiteralConstraints( + state, + subject, + shape, + consumed, + `${SH}maxInclusive`, + 'maxInclusive', + constraints, + ) + pushIntegerConstraints( + state, + subject, + shape, + consumed, + `${SH}minLength`, + 'minLength', + 'length', + constraints, + ) + pushIntegerConstraints( + state, + subject, + shape, + consumed, + `${SH}maxLength`, + 'maxLength', + 'length', + constraints, + ) pushPatternConstraints(state, subject, shape, consumed, constraints) - pushBooleanConstraints(state, subject, shape, consumed, `${SH}singleLine`, 'singleLine', constraints) + pushBooleanConstraints( + state, + subject, + shape, + consumed, + `${SH}singleLine`, + 'singleLine', + constraints, + ) pushLanguageConstraints(state, subject, shape, consumed, constraints) - pushBooleanConstraints(state, subject, shape, consumed, `${SH}uniqueLang`, 'uniqueLang', constraints) - pushShapeConstraints(state, subject, shape, consumed, `${SH}memberShape`, 'memberShape', constraints) - pushIntegerConstraints(state, subject, shape, consumed, `${SH}minListLength`, 'minListLength', 'length', constraints) - pushIntegerConstraints(state, subject, shape, consumed, `${SH}maxListLength`, 'maxListLength', 'length', constraints) - pushBooleanConstraints(state, subject, shape, consumed, `${SH}uniqueMembers`, 'uniqueMembers', constraints) + pushBooleanConstraints( + state, + subject, + shape, + consumed, + `${SH}uniqueLang`, + 'uniqueLang', + constraints, + ) + pushShapeConstraints( + state, + subject, + shape, + consumed, + `${SH}memberShape`, + 'memberShape', + constraints, + ) + pushIntegerConstraints( + state, + subject, + shape, + consumed, + `${SH}minListLength`, + 'minListLength', + 'length', + constraints, + ) + pushIntegerConstraints( + state, + subject, + shape, + consumed, + `${SH}maxListLength`, + 'maxListLength', + 'length', + constraints, + ) + pushBooleanConstraints( + state, + subject, + shape, + consumed, + `${SH}uniqueMembers`, + 'uniqueMembers', + constraints, + ) - for (const [predicate, kind] of [ - [`${SH}equals`, 'equals'], [`${SH}disjoint`, 'disjoint'], [`${SH}subsetOf`, 'subsetOf'], - [`${SH}lessThan`, 'lessThan'], [`${SH}lessThanOrEquals`, 'lessThanOrEquals'], - ] as const) pushPathConstraints(state, subject, shape, consumed, predicate, kind, constraints) + for ( + const [predicate, kind] of [ + [`${SH}equals`, 'equals'], + [`${SH}disjoint`, 'disjoint'], + [`${SH}subsetOf`, 'subsetOf'], + [`${SH}lessThan`, 'lessThan'], + [`${SH}lessThanOrEquals`, 'lessThanOrEquals'], + ] as const + ) pushPathConstraints(state, subject, shape, consumed, predicate, kind, constraints) pushShapeConstraints(state, subject, shape, consumed, `${SH}not`, 'not', constraints) pushShapeListConstraints(state, subject, shape, consumed, `${SH}and`, 'and', constraints) @@ -303,14 +568,35 @@ function readConstraints( pushShapeConstraints(state, subject, shape, consumed, `${SH}property`, 'property', constraints) pushShapeConstraints(state, subject, shape, consumed, `${SH}someValue`, 'someValue', constraints) pushQualifiedConstraints(state, subject, shape, consumed, constraints) - pushShapeConstraints(state, subject, shape, consumed, `${SH}reifierShape`, 'reifierShape', constraints) - pushBooleanConstraints(state, subject, shape, consumed, `${SH}reificationRequired`, 'reificationRequired', constraints) + pushShapeConstraints( + state, + subject, + shape, + consumed, + `${SH}reifierShape`, + 'reifierShape', + constraints, + ) + pushBooleanConstraints( + state, + subject, + shape, + consumed, + `${SH}reificationRequired`, + 'reificationRequired', + constraints, + ) pushClosedConstraints(state, subject, shape, consumed, constraints) for (const value of state.index.get(subject, `${SH}hasValue`)) { const record = term(value) if (record) constraints.push({ kind: 'hasValue', value: record }) - else addInvalid(state, shape, `${SH}hasValue`, 'sh:hasValue cannot be represented as a shape term.') + else {addInvalid( + state, + shape, + `${SH}hasValue`, + 'sh:hasValue cannot be represented as a shape term.', + )} } consumeWhenPresent(state, subject, consumed, `${SH}hasValue`) pushTermListConstraints(state, subject, shape, consumed, `${SH}in`, constraints) @@ -323,7 +609,7 @@ function readConstraints( /** Reads class/datatype/node-kind constraints as either one IRI or a SHACL 1.2 IRI choice list. */ function pushChoiceConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, predicate: string, @@ -331,21 +617,23 @@ function pushChoiceConstraints( constraints: ConstraintType[], ): void { for (const value of state.index.get(subject, predicate)) { - const choices = readIriChoices(state, value, shape, predicate) + const choices = getIriChoices(state, value, shape, predicate) if (choices) constraints.push({ kind, choices }) } consumeWhenPresent(state, subject, consumed, predicate) } /** Read iri choices from the supplied source while preserving caller ownership. */ -function readIriChoices( +function getIriChoices( state: StateType, - value: ObjectTerm, + value: ObjectTermType, shape: IdType, predicate: string, ): readonly string[] | undefined { - if ((value.termType === 'NamedNode' || value.termType === 'BlankNode') && state.index.isList(value)) { - const members = readList(state.index, value, listOptions(state, shape, predicate)) + if ( + (value.termType === 'NamedNode' || value.termType === 'BlankNode') && state.index.isList(value) + ) { + const members = getList(state.index, value, listOptions(state, shape, predicate)) if (!members) return undefined const choices: string[] = [] for (const member of members) { @@ -365,7 +653,7 @@ function readIriChoices( /** Reads non-negative integer constraints and records malformed/cardinality violations without dropping their source assertions. */ function pushIntegerConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, predicate: string, @@ -376,10 +664,20 @@ function pushIntegerConstraints( for (const value of state.index.get(subject, predicate)) { const integer = integerValue(value) if (integer === undefined || integer < 0) { - addInvalid(state, shape, predicate, `${local(predicate)} must be a non-negative xsd:integer literal.`) + addInvalid( + state, + shape, + predicate, + `${local(predicate)} must be a non-negative xsd:integer literal.`, + ) continue } - constraints.push(field === 'count' ? { kind: kind as 'minCount' | 'maxCount', count: integer } : { kind: kind as 'minLength' | 'maxLength' | 'minListLength' | 'maxListLength', length: integer }) + constraints.push( + field === 'count' ? { kind: kind as 'minCount' | 'maxCount', count: integer } : { + kind: kind as 'minLength' | 'maxLength' | 'minListLength' | 'maxListLength', + length: integer, + }, + ) } consumeWhenPresent(state, subject, consumed, predicate) } @@ -387,7 +685,7 @@ function pushIntegerConstraints( /** Preserves literal-valued comparison constraints as RDF terms so datatype ordering remains a validator concern. */ function pushLiteralConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, predicate: string, @@ -404,14 +702,16 @@ function pushLiteralConstraints( /** Combines `sh:pattern` with its optional `sh:flags` value while diagnosing duplicate or non-string values. */ function pushPatternConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, constraints: ConstraintType[], ): void { const flags = state.index.get(subject, `${SH}flags`) let flag: string | undefined - if (flags.length > 1) addCardinality(state, shape, `${SH}flags`, 'A shape must have at most one sh:flags value.') + if (flags.length > 1) { + addCardinality(state, shape, `${SH}flags`, 'A shape must have at most one sh:flags value.') + } if (flags[0]) { if (isStringLiteral(flags[0])) flag = flags[0].value else addInvalid(state, shape, `${SH}flags`, 'sh:flags must be an xsd:string literal.') @@ -423,7 +723,11 @@ function pushPatternConstraints( addInvalid(state, shape, `${SH}pattern`, 'sh:pattern must be an xsd:string literal.') continue } - constraints.push(flag === undefined ? { kind: 'pattern', pattern: value.value } : { kind: 'pattern', pattern: value.value, flags: flag }) + constraints.push( + flag === undefined + ? { kind: 'pattern', pattern: value.value } + : { kind: 'pattern', pattern: value.value, flags: flag }, + ) } consumeWhenPresent(state, subject, consumed, `${SH}pattern`) } @@ -431,19 +735,24 @@ function pushPatternConstraints( /** Reads language-list constraints from RDF lists and preserves their declared member order. */ function pushLanguageConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, constraints: ConstraintType[], ): void { for (const value of state.index.get(subject, `${SH}languageIn`)) { - const members = readList(state.index, value, listOptions(state, shape, `${SH}languageIn`)) + const members = getList(state.index, value, listOptions(state, shape, `${SH}languageIn`)) if (!members) continue const languages: string[] = [] let valid = true for (const member of members) { if (!isStringLiteral(member)) { - addInvalid(state, shape, `${SH}languageIn`, 'sh:languageIn list members must be xsd:string literals.') + addInvalid( + state, + shape, + `${SH}languageIn`, + 'sh:languageIn list members must be xsd:string literals.', + ) valid = false break } @@ -457,7 +766,7 @@ function pushLanguageConstraints( /** Reads singleton `xsd:boolean` constraints using RDF lexical boolean rules. */ function pushBooleanConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, predicate: string, @@ -478,7 +787,7 @@ function pushBooleanConstraints( /** Reads constraints that reference exactly one named or blank-node shape and diagnoses non-shape terms. */ function pushShapeConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, predicate: string, @@ -488,7 +797,12 @@ function pushShapeConstraints( for (const value of state.index.get(subject, predicate)) { const reference = id(value) if (reference) constraints.push({ kind, shape: reference }) - else addInvalid(state, shape, predicate, `${local(predicate)} must reference a shape IRI or blank node.`) + else {addInvalid( + state, + shape, + predicate, + `${local(predicate)} must reference a shape IRI or blank node.`, + )} } consumeWhenPresent(state, subject, consumed, predicate) } @@ -496,7 +810,7 @@ function pushShapeConstraints( /** Resolves SHACL shape lists such as `and`, `or`, and `xone` under the configured list limits. */ function pushShapeListConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, predicate: string, @@ -504,14 +818,19 @@ function pushShapeListConstraints( constraints: ConstraintType[], ): void { for (const value of state.index.get(subject, predicate)) { - const members = readList(state.index, value, listOptions(state, shape, predicate)) + const members = getList(state.index, value, listOptions(state, shape, predicate)) if (!members) continue const shapes: IdType[] = [] let valid = true for (const member of members) { const reference = id(member) if (!reference) { - addInvalid(state, shape, predicate, `${local(predicate)} list members must reference shapes.`) + addInvalid( + state, + shape, + predicate, + `${local(predicate)} list members must reference shapes.`, + ) valid = false break } @@ -522,10 +841,10 @@ function pushShapeListConstraints( consumeWhenPresent(state, subject, consumed, predicate) } -/** Reads property-pair constraints using the full SHACL path reader rather than restricting them to predicate IRIs. */ +/** Inspects property-pair constraints through the full SHACL path model instead of restricting them to predicate IRIs. */ function pushPathConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, predicate: string, @@ -533,7 +852,7 @@ function pushPathConstraints( constraints: ConstraintType[], ): void { for (const value of state.index.get(subject, predicate)) { - const path = readPath(state.index, value, pathOptions(state, shape, predicate)) + const path = getPath(state.index, value, pathOptions(state, shape, predicate)) constraints.push({ kind, path }) } consumeWhenPresent(state, subject, consumed, predicate) @@ -542,7 +861,7 @@ function pushPathConstraints( /** Aggregates qualified shape, min/max count, and disjointness assertions into one qualified-value constraint. */ function pushQualifiedConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, constraints: ConstraintType[], @@ -551,31 +870,58 @@ function pushQualifiedConstraints( const mins = state.index.get(subject, `${SH}qualifiedMinCount`) const maxes = state.index.get(subject, `${SH}qualifiedMaxCount`) const disjoints = state.index.get(subject, `${SH}qualifiedValueShapesDisjoint`) - for (const predicate of [`${SH}qualifiedValueShape`, `${SH}qualifiedMinCount`, `${SH}qualifiedMaxCount`, `${SH}qualifiedValueShapesDisjoint`]) { + for ( + const predicate of [ + `${SH}qualifiedValueShape`, + `${SH}qualifiedMinCount`, + `${SH}qualifiedMaxCount`, + `${SH}qualifiedValueShapesDisjoint`, + ] + ) { consumeWhenPresent(state, subject, consumed, predicate) } if (!shapes.length && !mins.length && !maxes.length && !disjoints.length) return if (shapes.length !== 1) { - addInvalid(state, shape, `${SH}qualifiedValueShape`, 'Qualified cardinality requires exactly one sh:qualifiedValueShape.') + addInvalid( + state, + shape, + `${SH}qualifiedValueShape`, + 'Qualified cardinality requires exactly one sh:qualifiedValueShape.', + ) return } const reference = id(shapes[0]!) if (!reference) { - addInvalid(state, shape, `${SH}qualifiedValueShape`, 'sh:qualifiedValueShape must reference a shape.') + addInvalid( + state, + shape, + `${SH}qualifiedValueShape`, + 'sh:qualifiedValueShape must reference a shape.', + ) return } const minCount = optionalInteger(state, mins, shape, `${SH}qualifiedMinCount`) const maxCount = optionalInteger(state, maxes, shape, `${SH}qualifiedMaxCount`) const disjoint = optionalBoolean(state, disjoints, shape, `${SH}qualifiedValueShapesDisjoint`) if (minCount === undefined && maxCount === undefined) { - addInvalid(state, shape, `${SH}qualifiedValueShape`, 'Qualified cardinality requires sh:qualifiedMinCount or sh:qualifiedMaxCount.') + addInvalid( + state, + shape, + `${SH}qualifiedValueShape`, + 'Qualified cardinality requires sh:qualifiedMinCount or sh:qualifiedMaxCount.', + ) return } const value: { + /** Selects the `qualified` variant of value. */ kind: 'qualified' + /** SHACL shape identifier associated with this constraint or diagnostic. */ shape: IdType + /** Minimum number of values that must satisfy the qualified value shape. */ minCount?: number + /** Maximum number of values that may satisfy the qualified value shape. */ maxCount?: number + /** Whether sibling qualified value shapes are required to be disjoint. */ disjoint?: boolean } = { kind: 'qualified', shape: reference } if (minCount !== undefined) value.minCount = minCount @@ -587,12 +933,17 @@ function pushQualifiedConstraints( /** Reads `sh:closed` as boolean or SHACL 1.2 `sh:ByTypes` and resolves the optional ignored-properties list. */ function pushClosedConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, constraints: ConstraintType[], ): void { - const ignored = readIriList(state, state.index.get(subject, `${SH}ignoredProperties`)[0], shape, `${SH}ignoredProperties`) ?? [] + const ignored = getIriList( + state, + state.index.get(subject, `${SH}ignoredProperties`)[0], + shape, + `${SH}ignoredProperties`, + ) ?? [] consumeWhenPresent(state, subject, consumed, `${SH}ignoredProperties`) for (const value of state.index.get(subject, `${SH}closed`)) { const boolean = booleanValue(value) @@ -604,7 +955,12 @@ function pushClosedConstraints( constraints.push({ kind: 'closed', mode: 'byTypes', ignoredProperties: ignored }) continue } - addInvalid(state, shape, `${SH}closed`, 'sh:closed must be xsd:boolean or sh:ByTypes in SHACL 1.2.') + addInvalid( + state, + shape, + `${SH}closed`, + 'sh:closed must be xsd:boolean or sh:ByTypes in SHACL 1.2.', + ) } consumeWhenPresent(state, subject, consumed, `${SH}closed`) } @@ -612,14 +968,14 @@ function pushClosedConstraints( /** Resolves RDF-list term constraints such as `sh:in` while retaining each RDF term without JSON coercion. */ function pushTermListConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, predicate: string, constraints: ConstraintType[], ): void { for (const value of state.index.get(subject, predicate)) { - const members = readList(state.index, value, listOptions(state, shape, predicate)) + const members = getList(state.index, value, listOptions(state, shape, predicate)) if (!members) continue const values = members.map(term) if (values.some((entry) => entry === undefined)) { @@ -634,7 +990,7 @@ function pushTermListConstraints( /** Reads singleton IRI-valued constraints such as `sh:rootClass` and diagnoses non-IRI values. */ function pushIriConstraints( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, predicate: string, @@ -651,52 +1007,73 @@ function pushIriConstraints( /** Reads SHACL 1.2 `sh:uniqueValuesFor` as a non-empty list of property paths. */ function pushUniqueValuesFor( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, constraints: ConstraintType[], ): void { for (const value of state.index.get(subject, `${SH}uniqueValuesFor`)) { - const members = state.index.isList(value) ? readList(state.index, value, listOptions(state, shape, `${SH}uniqueValuesFor`)) : [value] + const members = state.index.isList(value) + ? getList(state.index, value, listOptions(state, shape, `${SH}uniqueValuesFor`)) + : [value] if (!members) continue - const paths = members.map((member) => readPath(state.index, member, pathOptions(state, shape, `${SH}uniqueValuesFor`))) + const paths = members.map((member) => + getPath(state.index, member, pathOptions(state, shape, `${SH}uniqueValuesFor`)) + ) constraints.push({ kind: 'uniqueValuesFor', paths }) } consumeWhenPresent(state, subject, consumed, `${SH}uniqueValuesFor`) } /** Read metadata from the supplied source while preserving caller ownership. */ -function readMetadata( +function getMetadata( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, consumed: Set, ): MetadataType { const names = [ - ...readTextValues(state, subject, shape, `${SH}name`, consumed), - ...readTextValues(state, subject, shape, `${RDFS}label`, consumed), + ...getTextValues(state, subject, shape, `${SH}name`, consumed), + ...getTextValues(state, subject, shape, `${RDFS}label`, consumed), ] const descriptions = [ - ...readTextValues(state, subject, shape, `${SH}description`, consumed), - ...readTextValues(state, subject, shape, `${RDFS}comment`, consumed), + ...getTextValues(state, subject, shape, `${SH}description`, consumed), + ...getTextValues(state, subject, shape, `${RDFS}comment`, consumed), ] - const intents = readTextValues(state, subject, shape, `${SH}intent`, consumed) - const agentInstructions = readTextValues(state, subject, shape, `${SH}agentInstruction`, consumed) - const codeIdentifiers = readStringValues(state, subject, shape, `${SH}codeIdentifier`, consumed) - const units = readTermValues(state, subject, `${SH}unit`, consumed) - const order = readLiteralValues(state, subject, shape, `${SH}order`, consumed) - const groups = readIdValues(state, subject, shape, `${SH}group`, consumed) - const values = readTermValues(state, subject, `${SH}values`, consumed) - const defaultValues = readTermValues(state, subject, `${SH}defaultValue`, consumed) - return { names, descriptions, intents, agentInstructions, codeIdentifiers, units, order, groups, values, defaultValues } + const intents = getTextValues(state, subject, shape, `${SH}intent`, consumed) + const agentInstructions = getTextValues(state, subject, shape, `${SH}agentInstruction`, consumed) + const codeIdentifiers = getStringValues(state, subject, shape, `${SH}codeIdentifier`, consumed) + const units = getTermValues(state, subject, `${SH}unit`, consumed) + const order = getLiteralValues(state, subject, shape, `${SH}order`, consumed) + const groups = getIdValues(state, subject, shape, `${SH}group`, consumed) + const values = getTermValues(state, subject, `${SH}values`, consumed) + const defaultValues = getTermValues(state, subject, `${SH}defaultValue`, consumed) + return { + names, + descriptions, + intents, + agentInstructions, + codeIdentifiers, + units, + order, + groups, + values, + defaultValues, + } } /** Read assertions from the supplied source while preserving caller ownership. */ -function readAssertions(state: StateType, subject: Subject, consumed: ReadonlySet): AssertionType[] { +function getAssertions( + state: StateType, + subject: SubjectTermType, + consumed: ReadonlySet, +): AssertionType[] { const values: AssertionType[] = [] const invalidPredicates = new Set( state.diagnostics - .filter((diagnostic) => diagnostic.shape && idKey(diagnostic.shape) === idKey(id(subject)!) && diagnostic.predicate) + .filter((diagnostic) => + diagnostic.shape && idKey(diagnostic.shape) === idKey(id(subject)!) && diagnostic.predicate + ) .map((diagnostic) => diagnostic.predicate!), ) for (const quad of state.index.quads(subject)) { @@ -709,9 +1086,9 @@ function readAssertions(state: StateType, subject: Subject, consumed: ReadonlySe } /** Read text values from the supplied source while preserving caller ownership. */ -function readTextValues( +function getTextValues( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, predicate: string, consumed: Set, @@ -726,9 +1103,9 @@ function readTextValues( } /** Read string values from the supplied source while preserving caller ownership. */ -function readStringValues( +function getStringValues( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, predicate: string, consumed: Set, @@ -743,7 +1120,12 @@ function readStringValues( } /** Read term values from the supplied source while preserving caller ownership. */ -function readTermValues(state: StateType, subject: Subject, predicate: string, consumed: Set): TermType[] { +function getTermValues( + state: StateType, + subject: SubjectTermType, + predicate: string, + consumed: Set, +): TermType[] { const values: TermType[] = [] for (const value of state.index.get(subject, predicate)) { const record = term(value) @@ -754,9 +1136,9 @@ function readTermValues(state: StateType, subject: Subject, predicate: string, c } /** Read literal values from the supplied source while preserving caller ownership. */ -function readLiteralValues( +function getLiteralValues( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, predicate: string, consumed: Set, @@ -771,9 +1153,9 @@ function readLiteralValues( } /** Read id values from the supplied source while preserving caller ownership. */ -function readIdValues( +function getIdValues( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, predicate: string, consumed: Set, @@ -782,23 +1164,30 @@ function readIdValues( for (const value of state.index.get(subject, predicate)) { const reference = id(value) if (reference) values.push(reference) - else addInvalid(state, shape, predicate, `${local(predicate)} must reference an IRI or blank node.`) + else {addInvalid( + state, + shape, + predicate, + `${local(predicate)} must reference an IRI or blank node.`, + )} } consumeWhenPresent(state, subject, consumed, predicate) return values } /** Read iri singleton from the supplied source while preserving caller ownership. */ -function readIriSingleton( +function getIri( state: StateType, - subject: Subject, + subject: SubjectTermType, shape: IdType, predicate: string, consumed: Set, ): string | undefined { const values = state.index.get(subject, predicate) consumeWhenPresent(state, subject, consumed, predicate) - if (values.length > 1) addCardinality(state, shape, predicate, `${local(predicate)} must have at most one value.`) + if (values.length > 1) { + addCardinality(state, shape, predicate, `${local(predicate)} must have at most one value.`) + } const value = values[0] if (!value) return undefined if (value.termType === 'NamedNode') return value.value @@ -807,14 +1196,14 @@ function readIriSingleton( } /** Read iri list from the supplied source while preserving caller ownership. */ -function readIriList( +function getIriList( state: StateType, - value: ObjectTerm | undefined, + value: ObjectTermType | undefined, shape: IdType, predicate: string, ): readonly string[] | undefined { if (!value) return undefined - const members = readList(state.index, value, listOptions(state, shape, predicate)) + const members = getList(state.index, value, listOptions(state, shape, predicate)) if (!members) return undefined const values: string[] = [] for (const member of members) { @@ -830,40 +1219,51 @@ function readIriList( /** Reads an optional singleton non-negative `xsd:integer`, diagnosing duplicate or invalid lexical values. */ function optionalInteger( state: StateType, - values: readonly ObjectTerm[], + values: readonly ObjectTermType[], shape: IdType, predicate: string, ): number | undefined { - if (values.length > 1) addCardinality(state, shape, predicate, `${local(predicate)} must have at most one value.`) + if (values.length > 1) { + addCardinality(state, shape, predicate, `${local(predicate)} must have at most one value.`) + } if (!values[0]) return undefined const value = integerValue(values[0]) - if (value === undefined || value < 0) addInvalid(state, shape, predicate, `${local(predicate)} must be a non-negative xsd:integer.`) + if (value === undefined || value < 0) { + addInvalid(state, shape, predicate, `${local(predicate)} must be a non-negative xsd:integer.`) + } return value !== undefined && value >= 0 ? value : undefined } /** Reads an optional singleton `xsd:boolean`, accepting canonical and numeric RDF boolean lexical forms. */ function optionalBoolean( state: StateType, - values: readonly ObjectTerm[], + values: readonly ObjectTermType[], shape: IdType, predicate: string, ): boolean | undefined { - if (values.length > 1) addCardinality(state, shape, predicate, `${local(predicate)} must have at most one value.`) + if (values.length > 1) { + addCardinality(state, shape, predicate, `${local(predicate)} must have at most one value.`) + } if (!values[0]) return undefined const value = booleanValue(values[0]) - if (value === undefined) addInvalid(state, shape, predicate, `${local(predicate)} must be xsd:boolean.`) + if (value === undefined) { + addInvalid(state, shape, predicate, `${local(predicate)} must be xsd:boolean.`) + } return value } /** Decodes a safe JavaScript integer only from an `xsd:integer` lexical form. */ -function integerValue(value: ObjectTerm): number | undefined { - if (value.termType !== 'Literal' || value.datatype.value !== XSD.integer || !/^[+-]?\d+$/.test(value.value)) return undefined +function integerValue(value: ObjectTermType): number | undefined { + if ( + value.termType !== 'Literal' || value.datatype.value !== XSD.integer || + !/^[+-]?\d+$/.test(value.value) + ) return undefined const integer = Number(value.value) return Number.isSafeInteger(integer) ? integer : undefined } /** Decodes RDF `xsd:boolean` lexical forms without accepting JavaScript truthiness. */ -function booleanValue(value: ObjectTerm): boolean | undefined { +function booleanValue(value: ObjectTermType): boolean | undefined { if (value.termType !== 'Literal' || value.datatype.value !== XSD.boolean) return undefined if (value.value === 'true' || value.value === '1') return true if (value.value === 'false' || value.value === '0') return false @@ -871,21 +1271,26 @@ function booleanValue(value: ObjectTerm): boolean | undefined { } /** Returns whether the supplied value satisfies the string literal contract. */ -function isStringLiteral(value: ObjectTerm): value is Literal { +function isStringLiteral(value: ObjectTermType): value is Literal { return value.termType === 'Literal' && value.datatype.value === XSD.string } /** Marks a predicate consumed only when the source graph actually contains at least one value for it. */ -function consumeWhenPresent(state: StateType, subject: Subject, consumed: Set, predicate: string): void { +function consumeWhenPresent( + state: StateType, + subject: SubjectTermType, + consumed: Set, + predicate: string, +): void { if (state.index.get(subject, predicate).length) consumed.add(predicate) } -/** Creates bounded RDF-list read options that attach diagnostics to the owning shape and predicate. */ +/** Creates bounded RDF-list traversal options that attach diagnostics to the owning shape and predicate. */ function listOptions(state: StateType, shape: IdType, predicate: string) { return { maxItems: state.maxListItems, diagnostics: state.diagnostics, shape, predicate } } -/** Creates bounded SHACL-path read options that attach recursion diagnostics to the owning assertion. */ +/** Creates bounded SHACL-path inspection options that attach recursion diagnostics to the owning assertion. */ function pathOptions(state: StateType, shape: IdType, predicate: string) { return { maxDepth: state.maxPathDepth, @@ -896,15 +1301,15 @@ function pathOptions(state: StateType, shape: IdType, predicate: string) { } } -/** Records SHACL 1.2-only terms encountered in 1.0 mode while leaving those assertions available for loss-preserving reads. */ -function recordVersionDiagnostics(state: StateType, subject: Subject, shape: IdType): void { +/** Records SHACL 1.2-only terms encountered in 1.0 mode while retaining those assertions for loss-preserving inspection. */ +function recordVersionDiagnostics(state: StateType, subject: SubjectTermType, shape: IdType): void { if (state.version !== '1.0') return const types = state.index.get(subject, RDF.type) if (types.some((value) => value.termType === 'NamedNode' && value.value === SHAPE_CLASS)) { state.diagnostics.push({ code: 'unsupported-version', severity: 'warning', - message: 'sh:ShapeClass is a SHACL 1.2 feature and is retained while reading in 1.0 mode.', + message: 'sh:ShapeClass is a SHACL 1.2 feature and is retained while inspecting in 1.0 mode.', shape, predicate: RDF.type, }) @@ -914,7 +1319,9 @@ function recordVersionDiagnostics(state: StateType, subject: Subject, shape: IdT state.diagnostics.push({ code: 'unsupported-version', severity: 'warning', - message: `${local(predicate)} is a SHACL 1.2 Core feature and is retained while reading in 1.0 mode.`, + message: `${ + local(predicate) + } is a SHACL 1.2 Core feature and is retained while inspecting in 1.0 mode.`, shape, predicate, }) @@ -923,7 +1330,13 @@ function recordVersionDiagnostics(state: StateType, subject: Subject, shape: IdT /** Records a SHACL source-cardinality error without mutating or discarding the original RDF assertion. */ function addCardinality(state: StateType, shape: IdType, predicate: string, message: string): void { - state.diagnostics.push({ code: 'invalid-cardinality', severity: 'error', message, shape, predicate }) + state.diagnostics.push({ + code: 'invalid-cardinality', + severity: 'error', + message, + shape, + predicate, + }) } /** Records an invalid known SHACL value; raw assertions remain available to newer or external interpreters. */ @@ -943,10 +1356,15 @@ function idKey(value: IdType): string { /** Compare assertion using deterministic semantic ordering. */ function compareAssertion(left: AssertionType, right: AssertionType): number { - return `${idKey(left.subject)}\u0000${left.predicate}\u0000${JSON.stringify(left.object)}`.localeCompare(`${idKey(right.subject)}\u0000${right.predicate}\u0000${JSON.stringify(right.object)}`) + return `${idKey(left.subject)}\u0000${left.predicate}\u0000${JSON.stringify(left.object)}` + .localeCompare( + `${idKey(right.subject)}\u0000${right.predicate}\u0000${JSON.stringify(right.object)}`, + ) } /** Compare diagnostic using deterministic semantic ordering. */ function compareDiagnostic(left: DiagnosticType, right: DiagnosticType): number { - return `${left.code}\u0000${left.predicate ?? ''}\u0000${left.message}`.localeCompare(`${right.code}\u0000${right.predicate ?? ''}\u0000${right.message}`) + return `${left.code}\u0000${left.predicate ?? ''}\u0000${left.message}`.localeCompare( + `${right.code}\u0000${right.predicate ?? ''}\u0000${right.message}`, + ) } diff --git a/packages/rdf/shape/read_test.ts b/packages/rdf/shape/inspect_test.ts similarity index 73% rename from packages/rdf/shape/read_test.ts rename to packages/rdf/shape/inspect_test.ts index 0abccf5..60e88e5 100644 --- a/packages/rdf/shape/read_test.ts +++ b/packages/rdf/shape/inspect_test.ts @@ -2,16 +2,16 @@ import { describe, it } from 'node:test' import { expect } from '@std/expect' import { blankNode, literal, namedNode, quad, RDF, XSD } from '../mod.ts' import { parse } from '../turtle/mod.ts' -import { read } from './mod.ts' +import { inspect } from './mod.ts' const SH = 'http://www.w3.org/ns/shacl#' -function getShape(graph: Awaited>, suffix: string) { +function getShape(graph: Awaited>, suffix: string) { return graph.shapes.find((shape) => shape.id.value.endsWith(suffix)) } describe('@okikio/rdf/shape', () => { - it('reads SHACL 1.2 Core paths, constraints, metadata, and extension assertions', async () => { + it('inspects SHACL 1.2 Core paths, constraints, metadata, and extension assertions', async () => { const source = ` @prefix sh: . @prefix ex: . @@ -35,23 +35,36 @@ describe('@okikio/rdf/shape', () => { ] . ` - const graph = await read(parse(source), { version: '1.2' }) + const graph = await inspect(parse(source), { version: '1.2' }) expect(graph.diagnostics).toHaveLength(0) const person = getShape(graph, 'PersonShape') expect(person?.types.includes('https://example.com/Profile')).toBe(true) - expect(person?.constraints.some((value) => value.kind === 'closed' && value.mode === 'byTypes')).toBe(true) - expect(person?.constraints.some((value) => value.kind === 'uniqueValuesFor' && value.paths.length === 2)).toBe(true) + expect(person?.constraints.some((value) => value.kind === 'closed' && value.mode === 'byTypes')) + .toBe(true) + expect( + person?.constraints.some((value) => + value.kind === 'uniqueValuesFor' && value.paths.length === 2 + ), + ).toBe(true) - const property = graph.shapes.find((shape) => shape.kind === 'property' && shape.id.kind === 'blank') + const property = graph.shapes.find((shape) => + shape.kind === 'property' && shape.id.kind === 'blank' + ) expect(property?.path?.kind).toBe('alternative') if (property?.path?.kind === 'alternative') { expect(property.path.items).toHaveLength(2) expect(property.path.items[1]?.kind).toBe('inverse') } - expect(property?.constraints.some((value) => value.kind === 'class' && value.choices.length === 2)).toBe(true) + expect( + property?.constraints.some((value) => value.kind === 'class' && value.choices.length === 2), + ).toBe(true) expect(property?.metadata.names[0]?.value).toBe('Display name') - expect(property?.assertions.some((value) => value.predicate === 'https://example.com/customConstraint')).toBe(true) + expect( + property?.assertions.some((value) => + value.predicate === 'https://example.com/customConstraint' + ), + ).toBe(true) }) it('reports cyclic property paths without recursive overflow', async () => { @@ -63,7 +76,7 @@ describe('@okikio/rdf/shape', () => { quad(path, namedNode(`${SH}inversePath`), path), ] - const graph = await read(values) + const graph = await inspect(values) expect(graph.diagnostics.some((value) => value.code === 'path-cycle')).toBe(true) expect(graph.shapes[0]?.path?.kind).toBe('inverse') }) @@ -76,16 +89,20 @@ describe('@okikio/rdf/shape', () => { quad(shape, namedNode(`${SH}closed`), literal('not-a-boolean', namedNode(XSD.boolean))), ] - const graph = await read(values) + const graph = await inspect(values) expect(graph.shapes[0]?.constraints).toHaveLength(0) expect(graph.diagnostics.filter((value) => value.code === 'invalid-value')).toHaveLength(2) - expect(graph.shapes[0]?.assertions.some((value) => value.predicate === `${SH}minCount`)).toBe(true) - expect(graph.shapes[0]?.assertions.some((value) => value.predicate === `${SH}closed`)).toBe(true) + expect(graph.shapes[0]?.assertions.some((value) => value.predicate === `${SH}minCount`)).toBe( + true, + ) + expect(graph.shapes[0]?.assertions.some((value) => value.predicate === `${SH}closed`)).toBe( + true, + ) }) - it('warns when 1.2-only Core terms are read through the 1.0 interpretation mode', async () => { + it('warns when 1.2-only Core terms are inspected through the 1.0 interpretation mode', async () => { const shape = namedNode('https://example.com/Shape') - const graph = await read([ + const graph = await inspect([ quad(shape, namedNode(RDF.type), namedNode(`${SH}NodeShape`)), quad(shape, namedNode(`${SH}singleLine`), literal('true', namedNode(XSD.boolean))), ], { version: '1.0' }) diff --git a/packages/rdf/shape/list.ts b/packages/rdf/shape/list.ts index b7b5e7f..b9bd6c4 100644 --- a/packages/rdf/shape/list.ts +++ b/packages/rdf/shape/list.ts @@ -1,38 +1,44 @@ -/** SHACL list reader with cycle and cardinality diagnostics. @module */ +/** SHACL list inspector with cycle and cardinality diagnostics. @module */ import { key } from '../term.ts' -import type { ObjectTerm, Subject } from '../term.ts' +import type { ObjectTermType, SubjectTermType } from '../term.ts' import { RDF_FIRST, RDF_NIL, RDF_REST, ShapeIndex } from './index.ts' import type { DiagnosticType, IdType } from './model.ts' /** Limits for one SHACL list traversal. */ -export interface ListOptions { +export interface ListOptionsType { + /** Maximum RDF-list members followed before the list helper reports a limit. */ readonly maxItems: number + /** Structured diagnostics retained so recoverable source information is not silently discarded. */ readonly diagnostics: DiagnosticType[] + /** Owning shape identifier attached to list diagnostics. */ readonly shape?: IdType + /** RDF predicate IRI represented by this statement, pattern, or index entry. */ readonly predicate?: string } /** Read list from the supplied source while preserving caller ownership. */ -export function readList( +export function getList( index: ShapeIndex, - start: ObjectTerm, - options: ListOptions, -): readonly ObjectTerm[] | undefined { + start: ObjectTermType, + options: ListOptionsType, +): readonly ObjectTermType[] | undefined { if (start.termType === 'NamedNode' && start.value === RDF_NIL) return [] if (start.termType !== 'NamedNode' && start.termType !== 'BlankNode') { addInvalid(options, 'SHACL list head must be an IRI or blank node.') return undefined } - const values: ObjectTerm[] = [] + const values: ObjectTermType[] = [] const seen = new Set() - let cursor: Subject = start + let cursor: SubjectTermType = start while (!(cursor.termType === 'NamedNode' && cursor.value === RDF_NIL)) { const cursorKey = key(cursor) if (seen.has(cursorKey)) { - options.diagnostics.push(diagnostic('list-cycle', 'SHACL list contains an rdf:rest cycle.', options)) + options.diagnostics.push( + diagnostic('list-cycle', 'SHACL list contains an rdf:rest cycle.', options), + ) return undefined } seen.add(cursorKey) @@ -45,7 +51,10 @@ export function readList( const first = index.get(cursor, RDF_FIRST) const rest = index.get(cursor, RDF_REST) if (first.length !== 1 || rest.length !== 1) { - addInvalid(options, 'Each SHACL list cell must have exactly one rdf:first and one rdf:rest value.') + addInvalid( + options, + 'Each SHACL list cell must have exactly one rdf:first and one rdf:rest value.', + ) return undefined } @@ -62,7 +71,7 @@ export function readList( } /** Records a malformed RDF-list node so shape parsing can continue without silently accepting it. */ -function addInvalid(options: ListOptions, message: string): void { +function addInvalid(options: ListOptionsType, message: string): void { options.diagnostics.push(diagnostic('invalid-list', message, options)) } @@ -70,13 +79,18 @@ function addInvalid(options: ListOptions, message: string): void { function diagnostic( code: DiagnosticType['code'], message: string, - options: ListOptions, + options: ListOptionsType, ): DiagnosticType { const value: { + /** Stable machine-readable code for this diagnostic or failure. */ code: DiagnosticType['code'] + /** Diagnostic severity used to decide whether inspection can continue. */ severity: 'error' + /** Human-readable explanation of this SHACL inspection diagnostic. */ message: string + /** SHACL shape identifier associated with this constraint or diagnostic. */ shape?: IdType + /** RDF predicate IRI represented by this statement or operation filter. */ predicate?: string } = { code, severity: 'error', message } if (options.shape) value.shape = options.shape diff --git a/packages/rdf/shape/mod.ts b/packages/rdf/shape/mod.ts index 975ad6f..e05458e 100644 --- a/packages/rdf/shape/mod.ts +++ b/packages/rdf/shape/mod.ts @@ -1,7 +1,7 @@ /** * SHACL shape and property-path infrastructure. * - * This subpath reads shapes graphs into a loss-preserving, versioned IR. It is + * This subpath inspects shapes graphs into a loss-preserving, versioned IR. It is * intentionally separate from ontology interpretation and does not claim full * SHACL validation. SHACL 1.2 extension specifications can add evaluators over * the retained RDF term records without changing the Core model. @@ -11,17 +11,17 @@ * import * as turtle from '@okikio/rdf/turtle' * import * as shape from '@okikio/rdf/shape' * - * const graph = await shape.read(turtle.parse(source), { version: '1.2' }) + * const graph = await shape.inspect(turtle.parse(source), { version: '1.2' }) * const person = graph.shapes.find((value) => value.id.value.endsWith('PersonShape')) * ``` * * @module */ -export { read } from './read.ts' -export type { ReadOptions } from './read.ts' -export { readPath } from './path.ts' -export type { PathOptions } from './path.ts' +export { inspect } from './inspect.ts' +export type { InspectOptionsType } from './inspect.ts' +export { getPath } from './path.ts' +export type { PathOptionsType } from './path.ts' export type { AssertionType, BlankType, diff --git a/packages/rdf/shape/model.ts b/packages/rdf/shape/model.ts index 044e15a..5c78bb1 100644 --- a/packages/rdf/shape/model.ts +++ b/packages/rdf/shape/model.ts @@ -9,20 +9,24 @@ * @module */ -import type { Direction } from '../term.ts' +import type { DirectionType } from '../term.ts' -/** SHACL language family understood by the reader. */ +/** SHACL language family understood by the inspector. */ export type VersionType = '1.0' | '1.2' /** Stable serializable reference to an RDF IRI. */ export interface IriType { + /** Identifies this shape term as an RDF IRI reference. */ readonly kind: 'iri' + /** Absolute RDF IRI preserved for this shape term. */ readonly value: string } /** Stable serializable reference to an RDF blank node. */ export interface BlankType { + /** Identifies this shape term as an RDF blank-node reference. */ readonly kind: 'blank' + /** Blank-node identifier preserved from the source shapes graph. */ readonly value: string } @@ -31,18 +35,27 @@ export type IdType = IriType | BlankType /** Serializable RDF literal retained by the shape model. */ export interface LiteralType { + /** Identifies this shape term as an RDF literal. */ readonly kind: 'literal' + /** RDF literal lexical form preserved by the shape model. */ readonly value: string + /** Datatype IRI that defines how the RDF literal lexical form is interpreted. */ readonly datatype: string + /** BCP 47 language tag associated with this localized RDF value. */ readonly language?: string - readonly direction?: Direction + /** RDF 1.2 base text direction associated with this language value. */ + readonly direction?: DirectionType } /** Serializable RDF 1.2 triple term retained by the shape model. */ export interface TripleType { + /** Identifies this shape term as an RDF 1.2 triple term. */ readonly kind: 'triple' + /** RDF subject term represented by this statement, pattern, or index entry. */ readonly subject: IdType + /** RDF predicate IRI represented by this statement, pattern, or index entry. */ readonly predicate: string + /** RDF object term represented by this statement, pattern, or index entry. */ readonly object: TermType } @@ -51,108 +64,392 @@ export type TermType = IriType | BlankType | LiteralType | TripleType /** Localized human-facing SHACL text. */ export interface TextType { + /** Human-readable SHACL text without its optional language metadata. */ readonly value: string + /** BCP 47 language tag associated with this localized RDF value. */ readonly language?: string - readonly direction?: Direction + /** RDF 1.2 base text direction associated with this language value. */ + readonly direction?: DirectionType + /** Datatype IRI that defines how the RDF literal lexical form is interpreted. */ readonly datatype: string } /** SHACL property path with the complete Core path constructor family. */ export type PathType = - | { readonly kind: 'predicate'; readonly iri: string } - | { readonly kind: 'sequence'; readonly items: readonly PathType[] } - | { readonly kind: 'alternative'; readonly items: readonly PathType[] } - | { readonly kind: 'inverse'; readonly path: PathType } - | { readonly kind: 'zeroOrMore'; readonly path: PathType } - | { readonly kind: 'oneOrMore'; readonly path: PathType } - | { readonly kind: 'zeroOrOne'; readonly path: PathType } - | { readonly kind: 'unknown'; readonly value: TermType } + | { + /** Selects a direct predicate path. */ + readonly kind: 'predicate' + /** Predicate IRI traversed by the path. */ + readonly iri: string + } + | { + /** Selects an ordered SHACL sequence path. */ + readonly kind: 'sequence' + /** Path components evaluated from left to right. */ + readonly items: readonly PathType[] + } + | { + /** Selects a SHACL alternative path. */ + readonly kind: 'alternative' + /** Alternative path branches accepted by the expression. */ + readonly items: readonly PathType[] + } + | { + /** Selects an inverse SHACL path. */ + readonly kind: 'inverse' + /** Path whose traversal direction is reversed. */ + readonly path: PathType + } + | { + /** Selects a zero-or-more SHACL path. */ + readonly kind: 'zeroOrMore' + /** Path that can be traversed zero or more times. */ + readonly path: PathType + } + | { + /** Selects a one-or-more SHACL path. */ + readonly kind: 'oneOrMore' + /** Path that must be traversed at least once. */ + readonly path: PathType + } + | { + /** Selects a zero-or-one SHACL path. */ + readonly kind: 'zeroOrOne' + /** Optional path that can be traversed at most once. */ + readonly path: PathType + } + | { + /** Retains a path expression the current inspector cannot normalize. */ + readonly kind: 'unknown' + /** Loss-preserving RDF term that represented the unsupported path expression. */ + readonly value: TermType + } /** Explicit target attached to a shape. */ export type TargetType = - | { readonly kind: 'node'; readonly value: TermType } - | { readonly kind: 'class'; readonly iri: string } - | { readonly kind: 'subjectsOf'; readonly iri: string } - | { readonly kind: 'objectsOf'; readonly iri: string } - | { readonly kind: 'where'; readonly expression: TermType } - | { readonly kind: 'shape'; readonly shape: IdType } + | { + /** Selects an explicit focus-node target. */ + readonly kind: 'node' + /** RDF node selected as the focus node. */ + readonly value: TermType + } + | { + /** Selects nodes that are instances of one class. */ + readonly kind: 'class' + /** Class IRI used by the target. */ + readonly iri: string + } + | { + /** Selects subjects of one predicate. */ + readonly kind: 'subjectsOf' + /** Predicate whose subjects become focus nodes. */ + readonly iri: string + } + | { + /** Selects objects of one predicate. */ + readonly kind: 'objectsOf' + /** Predicate whose objects become focus nodes. */ + readonly iri: string + } + | { + /** Retains a SHACL 1.2 `sh:targetWhere` expression. */ + readonly kind: 'where' + /** RDF expression used to calculate target nodes. */ + readonly expression: TermType + } + | { + /** Selects nodes through another shape target. */ + readonly kind: 'shape' + /** Referenced shape that defines the target. */ + readonly shape: IdType + } /** Known SHACL Core constraint retained in semantic form. */ export type ConstraintType = - | { readonly kind: 'class'; readonly choices: readonly string[] } - | { readonly kind: 'datatype'; readonly choices: readonly string[] } - | { readonly kind: 'nodeKind'; readonly choices: readonly string[] } - | { readonly kind: 'minCount'; readonly count: number } - | { readonly kind: 'maxCount'; readonly count: number } - | { readonly kind: 'minExclusive'; readonly value: LiteralType } - | { readonly kind: 'minInclusive'; readonly value: LiteralType } - | { readonly kind: 'maxExclusive'; readonly value: LiteralType } - | { readonly kind: 'maxInclusive'; readonly value: LiteralType } - | { readonly kind: 'minLength'; readonly length: number } - | { readonly kind: 'maxLength'; readonly length: number } - | { readonly kind: 'pattern'; readonly pattern: string; readonly flags?: string } - | { readonly kind: 'singleLine'; readonly value: boolean } - | { readonly kind: 'languageIn'; readonly languages: readonly string[] } - | { readonly kind: 'uniqueLang'; readonly value: boolean } - | { readonly kind: 'memberShape'; readonly shape: IdType } - | { readonly kind: 'minListLength'; readonly length: number } - | { readonly kind: 'maxListLength'; readonly length: number } - | { readonly kind: 'uniqueMembers'; readonly value: boolean } - | { readonly kind: 'equals'; readonly path: PathType } - | { readonly kind: 'disjoint'; readonly path: PathType } - | { readonly kind: 'subsetOf'; readonly path: PathType } - | { readonly kind: 'lessThan'; readonly path: PathType } - | { readonly kind: 'lessThanOrEquals'; readonly path: PathType } - | { readonly kind: 'not'; readonly shape: IdType } - | { readonly kind: 'and'; readonly shapes: readonly IdType[] } - | { readonly kind: 'or'; readonly shapes: readonly IdType[] } - | { readonly kind: 'xone'; readonly shapes: readonly IdType[] } - | { readonly kind: 'node'; readonly shape: IdType } - | { readonly kind: 'property'; readonly shape: IdType } - | { readonly kind: 'someValue'; readonly shape: IdType } - | { - readonly kind: 'qualified' - readonly shape: IdType - readonly minCount?: number - readonly maxCount?: number - readonly disjoint?: boolean - } - | { readonly kind: 'reifierShape'; readonly shape: IdType } - | { readonly kind: 'reificationRequired'; readonly value: boolean } - | { - readonly kind: 'closed' - readonly mode: boolean | 'byTypes' - readonly ignoredProperties: readonly string[] - } - | { readonly kind: 'hasValue'; readonly value: TermType } - | { readonly kind: 'in'; readonly values: readonly TermType[] } - | { readonly kind: 'rootClass'; readonly iri: string } - | { readonly kind: 'uniqueValuesFor'; readonly paths: readonly PathType[] } + | { + /** Selects the `sh:class` constraint component. */ + readonly kind: 'class' + /** Class IRIs accepted for the value node. */ + readonly choices: readonly string[] + } + | { + /** Selects the `sh:datatype` constraint component. */ + readonly kind: 'datatype' + /** Datatype IRIs accepted for literal value nodes. */ + readonly choices: readonly string[] + } + | { + /** Selects the `sh:nodeKind` constraint component. */ + readonly kind: 'nodeKind' + /** Node-kind IRIs accepted for the value node. */ + readonly choices: readonly string[] + } + | { + /** Selects a minimum value-count constraint. */ + readonly kind: 'minCount' + /** Minimum number of values required by the shape. */ + readonly count: number + } + | { + /** Selects a maximum value-count constraint. */ + readonly kind: 'maxCount' + /** Maximum number of values permitted by the shape. */ + readonly count: number + } + | { + /** Selects an exclusive lower-bound constraint. */ + readonly kind: 'minExclusive' + /** RDF literal used as the exclusive lower bound. */ + readonly value: LiteralType + } + | { + /** Selects an inclusive lower-bound constraint. */ + readonly kind: 'minInclusive' + /** RDF literal used as the inclusive lower bound. */ + readonly value: LiteralType + } + | { + /** Selects an exclusive upper-bound constraint. */ + readonly kind: 'maxExclusive' + /** RDF literal used as the exclusive upper bound. */ + readonly value: LiteralType + } + | { + /** Selects an inclusive upper-bound constraint. */ + readonly kind: 'maxInclusive' + /** RDF literal used as the inclusive upper bound. */ + readonly value: LiteralType + } + | { + /** Selects a minimum string-length constraint. */ + readonly kind: 'minLength' + /** Minimum Unicode string length accepted by the constraint. */ + readonly length: number + } + | { + /** Selects a maximum string-length constraint. */ + readonly kind: 'maxLength' + /** Maximum Unicode string length accepted by the constraint. */ + readonly length: number + } + | { + /** Selects a regular-expression pattern constraint. */ + readonly kind: 'pattern' + /** Regular-expression pattern supplied by `sh:pattern`. */ + readonly pattern: string + /** Optional regular-expression flags supplied by `sh:flags`. */ + readonly flags?: string + } + | { + /** Selects the SHACL 1.2 single-line string constraint. */ + readonly kind: 'singleLine' + /** Whether line-break characters are disallowed. */ + readonly value: boolean + } + | { + /** Selects an allowed-language constraint. */ + readonly kind: 'languageIn' + /** Language ranges accepted by the constraint. */ + readonly languages: readonly string[] + } + | { + /** Selects the unique-language constraint. */ + readonly kind: 'uniqueLang' + /** Whether at most one value per language is required. */ + readonly value: boolean + } + | { + /** Selects the SHACL 1.2 member-shape constraint. */ + readonly kind: 'memberShape' + /** Shape that every list member must satisfy. */ + readonly shape: IdType + } + | { + /** Selects a minimum RDF-list length constraint. */ + readonly kind: 'minListLength' + /** Minimum number of list members required. */ + readonly length: number + } + | { + /** Selects a maximum RDF-list length constraint. */ + readonly kind: 'maxListLength' + /** Maximum number of list members permitted. */ + readonly length: number + } + | { + /** Selects the unique-list-members constraint. */ + readonly kind: 'uniqueMembers' + /** Whether duplicate RDF list members are disallowed. */ + readonly value: boolean + } + | { + /** Selects an equality constraint against another property path. */ + readonly kind: 'equals' + /** Path whose values must equal the current value set. */ + readonly path: PathType + } + | { + /** Selects a disjoint-value constraint against another property path. */ + readonly kind: 'disjoint' + /** Path whose values must not overlap the current value set. */ + readonly path: PathType + } + | { + /** Selects a subset constraint against another property path. */ + readonly kind: 'subsetOf' + /** Path whose values must contain the current value set. */ + readonly path: PathType + } + | { + /** Selects a strict ordering constraint against another property path. */ + readonly kind: 'lessThan' + /** Path whose values provide the strict comparison target. */ + readonly path: PathType + } + | { + /** Selects a non-strict ordering constraint against another property path. */ + readonly kind: 'lessThanOrEquals' + /** Path whose values provide the comparison target. */ + readonly path: PathType + } + | { + /** Selects logical negation of another shape. */ + readonly kind: 'not' + /** Shape that the focus/value node must not satisfy. */ + readonly shape: IdType + } + | { + /** Selects logical conjunction of several shapes. */ + readonly kind: 'and' + /** Shapes that must all be satisfied. */ + readonly shapes: readonly IdType[] + } + | { + /** Selects logical disjunction of several shapes. */ + readonly kind: 'or' + /** Shapes of which at least one must be satisfied. */ + readonly shapes: readonly IdType[] + } + | { + /** Selects exclusive disjunction of several shapes. */ + readonly kind: 'xone' + /** Shapes of which exactly one must be satisfied. */ + readonly shapes: readonly IdType[] + } + | { + /** Selects the `sh:node` constraint component. */ + readonly kind: 'node' + /** Shape that each value node must satisfy. */ + readonly shape: IdType + } + | { + /** Selects the `sh:property` constraint component. */ + readonly kind: 'property' + /** Property shape applied to the current focus node. */ + readonly shape: IdType + } + | { + /** Selects the SHACL 1.2 some-value constraint. */ + readonly kind: 'someValue' + /** Shape that at least one value node must satisfy. */ + readonly shape: IdType + } + | { + /** Selects qualified value-shape cardinality. */ + readonly kind: 'qualified' + /** Shape used to classify qualified values. */ + readonly shape: IdType + /** Minimum qualified value count, when declared. */ + readonly minCount?: number + /** Maximum qualified value count, when declared. */ + readonly maxCount?: number + /** Whether sibling qualified shapes must be disjoint. */ + readonly disjoint?: boolean + } + | { + /** Selects the RDF 1.2 reifier-shape constraint. */ + readonly kind: 'reifierShape' + /** Shape applied to reifiers of the value statement. */ + readonly shape: IdType + } + | { + /** Selects the RDF 1.2 reification-required constraint. */ + readonly kind: 'reificationRequired' + /** Whether a matching reifier is required. */ + readonly value: boolean + } + | { + /** Selects a closed-shape constraint. */ + readonly kind: 'closed' + /** Closed-mode flag or SHACL 1.2 type-derived mode. */ + readonly mode: boolean | 'byTypes' + /** Predicate IRIs ignored when checking undeclared properties. */ + readonly ignoredProperties: readonly string[] + } + | { + /** Selects a required-value constraint. */ + readonly kind: 'hasValue' + /** RDF value that must be present. */ + readonly value: TermType + } + | { + /** Selects an enumeration constraint. */ + readonly kind: 'in' + /** RDF values accepted by the enumeration. */ + readonly values: readonly TermType[] + } + | { + /** Selects the SHACL 1.2 root-class constraint. */ + readonly kind: 'rootClass' + /** Root class IRI used for hierarchy membership. */ + readonly iri: string + } + | { + /** Selects the SHACL 1.2 unique-values-for constraint. */ + readonly kind: 'uniqueValuesFor' + /** Property paths whose combined values must remain unique. */ + readonly paths: readonly PathType[] + } /** Non-validating and extension-facing metadata retained on one shape. */ export interface MetadataType { + /** Localized names retained for display, documentation, or generated symbols. */ readonly names: readonly TextType[] + /** Localized descriptive text retained from SHACL metadata. */ readonly descriptions: readonly TextType[] + /** Human-facing SHACL intent annotations retained as metadata. */ readonly intents: readonly TextType[] + /** Agent instruction annotations retained as data. The library never executes them. */ readonly agentInstructions: readonly TextType[] + /** Suggested code identifiers retained from source metadata. */ readonly codeIdentifiers: readonly string[] + /** Unit RDF terms associated with the shape metadata. */ readonly units: readonly TermType[] + /** Ordering literals retained without coercing provider-specific numeric semantics. */ readonly order: readonly LiteralType[] + /** SHACL group identifiers associated with this shape. */ readonly groups: readonly IdType[] + /** Value expressions retained from source metadata. */ readonly values: readonly TermType[] + /** Default-value expressions retained from source metadata. */ readonly defaultValues: readonly TermType[] } -/** One assertion the current Core reader intentionally does not interpret. */ +/** One assertion the current Core inspector intentionally does not interpret. */ export interface AssertionType { + /** RDF subject term represented by this statement, pattern, or index entry. */ readonly subject: IdType + /** RDF predicate IRI represented by this statement, pattern, or index entry. */ readonly predicate: string + /** RDF object term represented by this statement, pattern, or index entry. */ readonly object: TermType + /** RDF graph name represented by this quad, statement, or query target. */ readonly graph?: IdType } -/** Structured shape-reader diagnostic. */ +/** Structured shape-inspection diagnostic. */ export interface DiagnosticType { + /** Stable machine-readable code used to classify this diagnostic or failure. */ readonly code: | 'invalid-shape-id' | 'invalid-value' @@ -162,39 +459,56 @@ export interface DiagnosticType { | 'path-cycle' | 'invalid-cardinality' | 'unsupported-version' + /** Whether this inspection diagnostic is recoverable (`warning`) or invalidates the affected construct (`error`). */ readonly severity: 'warning' | 'error' + /** Human-readable explanation of the diagnostic or failure. */ readonly message: string + /** Shape identifier associated with this diagnostic when one is known. */ readonly shape?: IdType + /** RDF predicate IRI represented by this statement, pattern, or index entry. */ readonly predicate?: string } /** One node or property shape. */ export interface ShapeType { + /** RDF node that identifies this SHACL shape in the source graph. */ readonly id: IdType + /** Selects node-shape or property-shape semantics for this normalized record. */ readonly kind: 'node' | 'property' /** Every named rdf:type declared for this shape, including extension types. */ readonly types: readonly string[] + /** SHACL property path evaluated by this property shape. Node shapes normally omit it. */ readonly path?: PathType + /** Explicit SHACL targets that select focus nodes for this shape. */ readonly targets: readonly TargetType[] + /** Explicit `sh:severity` IRI attached to this shape, when present. */ readonly severity?: string + /** Localized SHACL validation messages attached to this shape. */ readonly messages: readonly TextType[] /** * Raw deactivation expression values. * * SHACL 1.0 normally uses one boolean. SHACL 1.2 generalizes this field to - * node-expression machinery, so the reader preserves the RDF values instead + * node-expression machinery, so the inspector preserves the RDF values instead * of pretending they are all booleans. */ readonly deactivated: readonly TermType[] + /** SHACL Core constraints normalized for this shape. */ readonly constraints: readonly ConstraintType[] + /** Non-validating SHACL metadata retained for this shape. */ readonly metadata: MetadataType + /** RDF assertions retained because the current semantic layer does not interpret them further. */ readonly assertions: readonly AssertionType[] } /** Loss-preserving SHACL shapes graph intermediate representation. */ export interface GraphType { + /** SHACL Core version used while interpreting the source shapes graph. */ readonly version: VersionType + /** Shapes discovered and normalized from the materialized SHACL graph. */ readonly shapes: readonly ShapeType[] + /** Structured diagnostics retained so recoverable source information is not silently discarded. */ readonly diagnostics: readonly DiagnosticType[] + /** RDF assertions retained because the current semantic layer does not interpret them further. */ readonly assertions: readonly AssertionType[] } diff --git a/packages/rdf/shape/path.ts b/packages/rdf/shape/path.ts index c8ef2da..39f7a54 100644 --- a/packages/rdf/shape/path.ts +++ b/packages/rdf/shape/path.ts @@ -1,9 +1,9 @@ -/** SHACL Core property-path reader. @module */ +/** SHACL Core property-path inspector. @module */ import { key } from '../term.ts' -import type { ObjectTerm, Subject } from '../term.ts' +import type { ObjectTermType, SubjectTermType } from '../term.ts' import { ShapeIndex } from './index.ts' -import { readList } from './list.ts' +import { getList } from './list.ts' import type { DiagnosticType, IdType, PathType } from './model.ts' import { term } from './value.ts' @@ -19,24 +19,33 @@ const PATH_PREDICATES = [ ] as const /** Limits and diagnostics shared by recursive path parsing. */ -export interface PathOptions { +export interface PathOptionsType { + /** Maximum recursive SHACL property-path depth followed before inspection reports a limit. */ readonly maxDepth: number + /** Maximum RDF-list members followed from one list before inspection reports a limit. */ readonly maxListItems: number + /** Structured diagnostics retained so recoverable source information is not silently discarded. */ readonly diagnostics: DiagnosticType[] + /** Owning shape identifier attached to path diagnostics. */ readonly shape?: IdType + /** RDF predicate IRI represented by this statement, pattern, or index entry. */ readonly predicate?: string } /** Parses one RDF term as a SHACL Core property path. */ -export function readPath(index: ShapeIndex, value: ObjectTerm, options: PathOptions): PathType { - return readPathAt(index, value, options, new Set(), 0) +export function getPath( + index: ShapeIndex, + value: ObjectTermType, + options: PathOptionsType, +): PathType { + return getPathAt(index, value, options, new Set(), 0) } /** Read path at from the supplied source while preserving caller ownership. */ -function readPathAt( +function getPathAt( index: ShapeIndex, - value: ObjectTerm, - options: PathOptions, + value: ObjectTermType, + options: PathOptionsType, active: Set, depth: number, ): PathType { @@ -44,35 +53,71 @@ function readPathAt( return { kind: 'predicate', iri: value.value } } - if (value.termType !== 'NamedNode' && value.termType !== 'BlankNode') return unknown(value, options, 'SHACL path must be an IRI or blank node.') - if (depth >= options.maxDepth) return unknown(value, options, `SHACL path exceeds the configured depth limit of ${options.maxDepth}.`) + if (value.termType !== 'NamedNode' && value.termType !== 'BlankNode') { + return unknown(value, options, 'SHACL path must be an IRI or blank node.') + } + if (depth >= options.maxDepth) { + return unknown( + value, + options, + `SHACL path exceeds the configured depth limit of ${options.maxDepth}.`, + ) + } - const subject = value as Subject + const subject = value as SubjectTermType const subjectKey = key(subject) if (active.has(subjectKey)) { - options.diagnostics.push(pathDiagnostic('path-cycle', 'SHACL property path contains a cycle.', options)) + options.diagnostics.push( + pathDiagnostic('path-cycle', 'SHACL property path contains a cycle.', options), + ) return unknownValue(value) } const nextActive = new Set(active) nextActive.add(subjectKey) if (index.isList(subject)) { - const values = readList(index, subject, listOptions(options)) - if (!values || values.length < 2) return unknown(value, options, 'A SHACL sequence path must contain at least two path members.') - return { kind: 'sequence', items: values.map((item) => readPathAt(index, item, options, nextActive, depth + 1)) } + const values = getList(index, subject, listOptions(options)) + if (!values || values.length < 2) { + return unknown( + value, + options, + 'A SHACL sequence path must contain at least two path members.', + ) + } + return { + kind: 'sequence', + items: values.map((item) => getPathAt(index, item, options, nextActive, depth + 1)), + } } - const constructors = PATH_PREDICATES.flatMap((predicate) => index.get(subject, predicate).map((object) => ({ predicate, object }))) - if (constructors.length !== 1) return unknown(value, options, 'A blank-node SHACL path must have exactly one Core path constructor.') + const constructors = PATH_PREDICATES.flatMap((predicate) => + index.get(subject, predicate).map((object) => ({ predicate, object })) + ) + if (constructors.length !== 1) { + return unknown( + value, + options, + 'A blank-node SHACL path must have exactly one Core path constructor.', + ) + } const constructor = constructors[0]! if (constructor.predicate === `${SH}alternativePath`) { - const values = readList(index, constructor.object, listOptions(options)) - if (!values || values.length < 2) return unknown(value, options, 'sh:alternativePath must reference a list with at least two members.') - return { kind: 'alternative', items: values.map((item) => readPathAt(index, item, options, nextActive, depth + 1)) } + const values = getList(index, constructor.object, listOptions(options)) + if (!values || values.length < 2) { + return unknown( + value, + options, + 'sh:alternativePath must reference a list with at least two members.', + ) + } + return { + kind: 'alternative', + items: values.map((item) => getPathAt(index, item, options, nextActive, depth + 1)), + } } - const child = readPathAt(index, constructor.object, options, nextActive, depth + 1) + const child = getPathAt(index, constructor.object, options, nextActive, depth + 1) if (constructor.predicate === `${SH}inversePath`) return { kind: 'inverse', path: child } if (constructor.predicate === `${SH}zeroOrMorePath`) return { kind: 'zeroOrMore', path: child } if (constructor.predicate === `${SH}oneOrMorePath`) return { kind: 'oneOrMore', path: child } @@ -80,30 +125,33 @@ function readPathAt( } /** Detects whether a blank-node path uses one of the SHACL path constructor predicates. */ -function hasConstructor(index: ShapeIndex, subject: Subject): boolean { +function hasConstructor(index: ShapeIndex, subject: SubjectTermType): boolean { return PATH_PREDICATES.some((predicate) => index.get(subject, predicate).length > 0) } /** Retains an unrecognized path node as a loss-preserving unknown path record. */ -function unknown(value: ObjectTerm, options: PathOptions, message: string): PathType { +function unknown(value: ObjectTermType, options: PathOptionsType, message: string): PathType { options.diagnostics.push(pathDiagnostic('invalid-path', message, options)) return unknownValue(value) } /** Retains a path assertion whose value cannot be normalized by the current SHACL profile. */ -function unknownValue(value: ObjectTerm): PathType { +function unknownValue(value: ObjectTermType): PathType { const record = term(value) if (!record) throw new TypeError('SHACL path term could not be represented.') return { kind: 'unknown', value: record } } - -/** Projects path-reader state into the shared RDF-list traversal options. */ -function listOptions(options: PathOptions) { +/** Projects property-path inspection state into the shared RDF-list traversal options. */ +function listOptions(options: PathOptionsType) { const value: { + /** Maximum RDF-list members traversed while decoding this SHACL property path. */ maxItems: number + /** Shared diagnostic sink that receives invalid-list and path-cycle reports. */ diagnostics: DiagnosticType[] + /** SHACL shape identifier associated with this constraint or diagnostic. */ shape?: IdType + /** RDF predicate IRI represented by this statement or operation filter. */ predicate?: string } = { maxItems: options.maxListItems, diagnostics: options.diagnostics } if (options.shape) value.shape = options.shape @@ -115,13 +163,18 @@ function listOptions(options: PathOptions) { function pathDiagnostic( code: 'invalid-path' | 'path-cycle', message: string, - options: PathOptions, + options: PathOptionsType, ): DiagnosticType { const value: { + /** Stable machine-readable code for this diagnostic or failure. */ code: 'invalid-path' | 'path-cycle' + /** Diagnostic severity used to decide whether inspection can continue. */ severity: 'error' + /** Human-readable explanation of this SHACL inspection diagnostic. */ message: string + /** SHACL shape identifier associated with this constraint or diagnostic. */ shape?: IdType + /** RDF predicate IRI represented by this statement or operation filter. */ predicate?: string } = { code, severity: 'error', message } if (options.shape) value.shape = options.shape diff --git a/packages/rdf/shape/value.ts b/packages/rdf/shape/value.ts index 3c66dde..be30182 100644 --- a/packages/rdf/shape/value.ts +++ b/packages/rdf/shape/value.ts @@ -1,6 +1,6 @@ /** Serializable RDF term conversion for the SHACL model. @module */ -import type { Graph, Literal, Quad, Term } from '../term.ts' +import type { GraphTermType, Literal, Quad, Term } from '../term.ts' import type { AssertionType, IdType, LiteralType, TermType, TextType } from './model.ts' /** Converts an RDF graph node into a serializable SHACL identifier. */ @@ -35,10 +35,15 @@ export function term(value: Term): TermType | undefined { /** Converts an RDF literal into a serializable SHACL literal. */ export function literal(value: Literal): LiteralType { const record: { + /** Selects the `literal` variant of record. */ kind: 'literal' + /** RDF literal lexical form preserved in the temporary serializable record. */ value: string + /** Datatype IRI associated with this RDF literal value. */ datatype: string + /** BCP 47 language tag retained for this localized RDF value. */ language?: string + /** RDF 1.2 base text direction retained for this localized RDF value. */ direction?: Literal['direction'] extends '' ? never : 'ltr' | 'rtl' } = { kind: 'literal', @@ -53,9 +58,13 @@ export function literal(value: Literal): LiteralType { /** Converts an RDF literal into localized human-facing text. */ export function text(value: Literal): TextType { const record: { + /** Literal lexical form preserved as human-facing SHACL text. */ value: string + /** Datatype IRI associated with this RDF literal value. */ datatype: string + /** BCP 47 language tag retained for this localized RDF value. */ language?: string + /** RDF 1.2 base text direction retained for this localized RDF value. */ direction?: 'ltr' | 'rtl' } = { value: value.value, datatype: value.datatype.value } if (value.language) record.language = value.language @@ -69,7 +78,16 @@ export function assertion(quad: Quad): AssertionType | undefined { if (!object) return undefined const subject = id(quad.subject) if (!subject) return undefined - const record: { subject: IdType; predicate: string; object: TermType; graph?: IdType } = { + const record: { + /** RDF subject term represented by this statement or operation filter. */ + subject: IdType + /** RDF predicate IRI represented by this statement or operation filter. */ + predicate: string + /** RDF object term represented by this statement or operation filter. */ + object: TermType + /** RDF graph that receives quads produced by SHACL inspection. */ + graph?: IdType + } = { subject, predicate: quad.predicate.value, object, @@ -80,6 +98,6 @@ export function assertion(quad: Quad): AssertionType | undefined { } /** Returns a serializable graph identifier when a quad belongs to a named graph. */ -function graphId(graph: Graph): IdType | undefined { +function graphId(graph: GraphTermType): IdType | undefined { return graph.termType === 'DefaultGraph' ? undefined : id(graph) } diff --git a/packages/rdf/source.ts b/packages/rdf/source.ts index 851537a..4933b34 100644 --- a/packages/rdf/source.ts +++ b/packages/rdf/source.ts @@ -10,7 +10,11 @@ export interface AsyncSource extends AsyncIterable {} /** A sink that consumes an RDF quad sequence without taking source ownership. */ export interface Sink { - import(source: Iterable | AsyncIterable, options?: { readonly signal?: AbortSignal }): Promise + /** Consumes quads from the supplied source and returns the sink-specific terminal result. */ + import(source: Iterable | AsyncIterable, options?: { + /** Abort signal checked before and during this operation. */ + readonly signal?: AbortSignal + }): Promise } /** Converts sync or async quad input into one async iteration contract. */ diff --git a/packages/rdf/stream.ts b/packages/rdf/stream.ts new file mode 100644 index 0000000..c7e2532 --- /dev/null +++ b/packages/rdf/stream.ts @@ -0,0 +1,150 @@ +/** Streaming adapter primitives shared by external RDF parser packages. @module */ + +import { fromQuad } from './factory.ts' +import type { Quad } from './term.ts' +import { chunks, type TextSourceType, throwIfAborted } from './text.ts' + +export type { TextSourceType } from './text.ts' + +/** + * Minimal writable async parser used by external streaming RDF adapters. + * + * The interface models only the operations required to connect a Node-style + * parser to the repository's Web-oriented source contract. Implementations + * remain owned by the adapter operation that creates them. + */ +export interface Parser extends AsyncIterable { + /** Writes one text or byte chunk and reports whether more input can be accepted immediately. */ + write(chunk: string | Uint8Array): boolean + /** Signals that no more source chunks will arrive. */ + end(): void + /** Adds a one-shot lifecycle listener used for backpressure and failure propagation. */ + once(event: 'drain' | 'close' | 'error', listener: (...args: unknown[]) => void): this + /** Removes a lifecycle listener after the wait settles or cancellation wins. */ + off(event: 'drain' | 'close' | 'error', listener: (...args: unknown[]) => void): this + /** Releases parser resources and optionally forwards the terminal source failure. */ + destroy(error?: Error): void +} + +/** Controls one external parser stream adaptation. */ +export interface ParseOptionsType { + /** Human-readable parser name included in lifecycle errors. */ + readonly label: string + /** Caller-owned cancellation signal. Cancellation destroys unfinished parser work. */ + readonly signal?: AbortSignal +} + +/** + * Adapts an external writable parser to the repository's async RDF stream. + * + * The parser is owned by this operation. The input source is borrowed. + * Returning from iteration early cancels the source pump and destroys the + * parser so upstream network, file, or Web Stream work cannot outlive its + * consumer. + * + * @example + * ```ts + * import * as stream from '@okikio/rdf/stream' + * + * for await (const quad of stream.parse(parser, source, { label: 'RDF/XML' })) { + * console.log(quad) + * } + * ``` + */ +export async function* parse( + parser: Parser, + source: TextSourceType, + options: ParseOptionsType, +): AsyncGenerator { + throwIfAborted(options.signal) + const lifecycle = new AbortController() + const signal = options.signal + ? AbortSignal.any([options.signal, lifecycle.signal]) + : lifecycle.signal + let complete = false + const pumping = pump(parser, source, signal, options.label) + + try { + for await (const value of parser) { + throwIfAborted(signal) + yield fromQuad(value) + } + await pumping + complete = true + } finally { + if (!complete) { + lifecycle.abort( + new DOMException( + `${options.label} consumer stopped before source completion`, + 'AbortError', + ), + ) + parser.destroy() + } + await pumping.catch(() => undefined) + } +} + +/** Feeds source chunks into one parser while preserving writable backpressure. */ +async function pump( + parser: Parser, + source: TextSourceType, + signal: AbortSignal, + label: string, +): Promise { + try { + for await (const value of chunks(source, signal)) { + throwIfAborted(signal) + if (!parser.write(value)) await drain(parser, signal, label) + } + throwIfAborted(signal) + parser.end() + } catch (error) { + parser.destroy(toError(error)) + throw error + } +} + +/** Waits for parser write capacity while cancellation and parser failure remain observable. */ +function drain(parser: Parser, signal: AbortSignal, label: string): Promise { + return new Promise((resolve, reject) => { + /** Removes every listener installed by this single backpressure wait. */ + const cleanup = () => { + signal.removeEventListener('abort', onAbort) + parser.off('drain', onDrain) + parser.off('close', onClose) + parser.off('error', onError) + } + /** Completes the wait when the parser can accept another source chunk. */ + const onDrain = () => { + cleanup() + resolve() + } + /** Fails the wait when the parser closes before writable capacity returns. */ + const onClose = () => { + cleanup() + reject(new Error(`${label} closed while waiting for writable capacity.`)) + } + /** Preserves the parser failure as the terminal backpressure error. */ + const onError = (value: unknown) => { + cleanup() + reject(toError(value)) + } + /** Stops the wait when the caller or enclosing parser lifecycle is cancelled. */ + const onAbort = () => { + cleanup() + reject(signal.reason ?? new DOMException('Aborted', 'AbortError')) + } + + if (signal.aborted) return onAbort() + signal.addEventListener('abort', onAbort, { once: true }) + parser.once('drain', onDrain) + parser.once('close', onClose) + parser.once('error', onError) + }) +} + +/** Converts a non-Error thrown value without replacing an existing Error identity. */ +function toError(value: unknown): Error { + return value instanceof Error ? value : new Error(String(value)) +} diff --git a/packages/rdf/term.ts b/packages/rdf/term.ts index 7ea659b..13b356f 100644 --- a/packages/rdf/term.ts +++ b/packages/rdf/term.ts @@ -9,55 +9,67 @@ */ /** Initial text direction attached to an RDF 1.2 directional language string. */ -export type Direction = 'ltr' | 'rtl' +export type DirectionType = 'ltr' | 'rtl' /** RDF/JS-compatible base term. */ export interface Term { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ readonly termType: 'NamedNode' | 'BlankNode' | 'Literal' | 'Variable' | 'DefaultGraph' | 'Quad' + /** RDF/JS lexical value. DefaultGraph and Quad use the required empty string. */ readonly value: string + /** Returns whether the supplied RDF term is term-equal to this term. */ equals(other?: Term | null): boolean } /** An RDF IRI term. */ export interface NamedNode extends Term { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ readonly termType: 'NamedNode' } /** An RDF blank node. */ export interface BlankNode extends Term { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ readonly termType: 'BlankNode' } /** An RDF query variable used by RDF/JS-compatible query surfaces. */ export interface Variable extends Term { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ readonly termType: 'Variable' } /** The default graph name. */ export interface DefaultGraph extends Term { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ readonly termType: 'DefaultGraph' + /** RDF/JS requires the default graph value to be the empty string. */ readonly value: '' } /** RDF 1.2 literal, including optional language direction. */ export interface Literal extends Term { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ readonly termType: 'Literal' + /** BCP 47 language tag associated with this localized RDF value. */ readonly language: string - readonly direction: Direction | '' + /** RDF 1.2 base text direction associated with this language value. */ + readonly direction: DirectionType | '' + /** Datatype IRI that defines how the RDF literal lexical form is interpreted. */ readonly datatype: NamedNode } /** RDF triple subject. RDF 1.2 triple terms are object terms, not graph subjects. */ -export type Subject = NamedNode | BlankNode +export type SubjectTermType = NamedNode | BlankNode /** RDF triple predicate. */ -export type Predicate = NamedNode +export type PredicateTermType = NamedNode /** RDF triple object, including RDF 1.2 triple terms. */ -export type ObjectTerm = NamedNode | BlankNode | Literal | Quad +export type ObjectTermType = NamedNode | BlankNode | Literal | Quad /** RDF dataset graph name. */ -export type Graph = DefaultGraph | NamedNode | BlankNode +export type GraphTermType = DefaultGraph | NamedNode | BlankNode /** * RDF/JS-compatible quad. @@ -67,21 +79,29 @@ export type Graph = DefaultGraph | NamedNode | BlankNode * triple's object so the native model remains consistent with RDF 1.2. */ export interface Quad extends Term { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ readonly termType: 'Quad' + /** RDF/JS requires a Quad term value to be the empty string. */ readonly value: '' - readonly subject: Subject - readonly predicate: Predicate - readonly object: ObjectTerm - readonly graph: Graph + /** RDF subject term represented by this statement, pattern, or index entry. */ + readonly subject: SubjectTermType + /** RDF predicate IRI represented by this statement, pattern, or index entry. */ + readonly predicate: PredicateTermType + /** RDF object term represented by this statement, pattern, or index entry. */ + readonly object: ObjectTermType + /** RDF graph name represented by this quad, statement, or query target. */ + readonly graph: GraphTermType } /** Any public RDF term. */ export type TermType = NamedNode | BlankNode | Literal | Variable | DefaultGraph | Quad /** RDF/JS-compatible directional-language factory input. */ -export interface DirectionalLanguage { +export interface DirectionalLanguageType { + /** BCP 47 language tag associated with this localized RDF value. */ readonly language: string - readonly direction?: Direction | null + /** RDF 1.2 base text direction associated with this language value. */ + readonly direction?: DirectionType | null } /** RDF namespace constants used by the core term model. */ @@ -112,8 +132,10 @@ export const XSD = { /** Base immutable term implementation. */ abstract class BaseTerm implements Term { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ abstract readonly termType: Term['termType'] + /** Immutable lexical value used by simple RDF-term equality. */ readonly value: string /** Stores the immutable lexical value shared by concrete RDF term implementations. */ @@ -123,29 +145,35 @@ abstract class BaseTerm implements Term { /** Compares simple RDF terms by term kind and lexical value. */ equals(other?: Term | null): boolean { - return other !== null && other !== undefined && this.termType === other.termType && this.value === other.value + return other !== null && other !== undefined && this.termType === other.termType && + this.value === other.value } } /** Immutable named-node implementation. */ export class NamedNodeValue extends BaseTerm implements NamedNode { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ readonly termType = 'NamedNode' as const } /** Immutable blank-node implementation. */ export class BlankNodeValue extends BaseTerm implements BlankNode { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ readonly termType = 'BlankNode' as const } /** Immutable variable implementation. */ export class VariableValue extends BaseTerm implements Variable { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ readonly termType = 'Variable' as const } /** Immutable default-graph singleton implementation. */ export class DefaultGraphValue extends BaseTerm implements DefaultGraph { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ readonly termType = 'DefaultGraph' as const - readonly value = '' as const + /** Required empty RDF/JS value for the default graph singleton. */ + override readonly value = '' as const /** Creates the RDF default graph singleton value with the required empty lexical form. */ constructor() { @@ -160,13 +188,22 @@ export class DefaultGraphValue extends BaseTerm implements DefaultGraph { /** Immutable RDF literal implementation. */ export class LiteralValue extends BaseTerm implements Literal { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ readonly termType = 'Literal' as const + /** Datatype IRI that defines how the RDF literal lexical form is interpreted. */ readonly datatype: NamedNode + /** BCP 47 language tag associated with this localized RDF value. */ readonly language: string - readonly direction: Direction | '' + /** RDF 1.2 base text direction associated with this language value. */ + readonly direction: DirectionType | '' /** Stores literal lexical form, datatype, language, and RDF 1.2 base direction without coercion. */ - constructor(value: string, datatype: NamedNode, language = '', direction: Direction | '' = '') { + constructor( + value: string, + datatype: NamedNode, + language = '', + direction: DirectionType | '' = '', + ) { super(value) this.datatype = datatype this.language = language @@ -186,15 +223,26 @@ export class LiteralValue extends BaseTerm implements Literal { /** Immutable quad and triple-term implementation. */ export class QuadValue extends BaseTerm implements Quad { + /** RDF/JS term-kind discriminator used for standards-compatible term interoperability. */ readonly termType = 'Quad' as const - readonly value = '' as const - readonly subject: Subject - readonly predicate: Predicate - readonly object: ObjectTerm - readonly graph: Graph + /** Required empty RDF/JS value for quad and triple-term objects. */ + override readonly value = '' as const + /** RDF subject term represented by this statement, pattern, or index entry. */ + readonly subject: SubjectTermType + /** RDF predicate IRI represented by this statement, pattern, or index entry. */ + readonly predicate: PredicateTermType + /** RDF object term represented by this statement, pattern, or index entry. */ + readonly object: ObjectTermType + /** RDF graph name represented by this quad, statement, or query target. */ + readonly graph: GraphTermType /** Stores one immutable quad; default-graph quads can also represent RDF 1.2 triple terms. */ - constructor(subject: Subject, predicate: Predicate, object: ObjectTerm, graph: Graph) { + constructor( + subject: SubjectTermType, + predicate: PredicateTermType, + object: ObjectTermType, + graph: GraphTermType, + ) { super('') this.subject = subject this.predicate = predicate @@ -244,11 +292,15 @@ export function key(term: Term): string { return 'D' case 'Literal': { const literal = term as Literal - return `L${atom('', literal.value)}${atom('', literal.datatype.value)}${atom('', literal.language.toLowerCase())}${atom('', literal.direction)}` + return `L${atom('', literal.value)}${atom('', literal.datatype.value)}${ + atom('', literal.language.toLowerCase()) + }${atom('', literal.direction)}` } case 'Quad': { const quad = term as Quad - return `Q${atom('', key(quad.subject))}${atom('', key(quad.predicate))}${atom('', key(quad.object))}${atom('', key(quad.graph))}` + return `Q${atom('', key(quad.subject))}${atom('', key(quad.predicate))}${ + atom('', key(quad.object)) + }${atom('', key(quad.graph))}` } } } diff --git a/packages/rdf/term_test.ts b/packages/rdf/term_test.ts index 1dfb2f2..33d4c3c 100644 --- a/packages/rdf/term_test.ts +++ b/packages/rdf/term_test.ts @@ -1,8 +1,6 @@ import { describe, it } from 'node:test' import { expect } from '@std/expect' import { - RDF, - XSD, blankNode, defaultGraph, equals, @@ -11,8 +9,10 @@ import { literal, namedNode, quad, + RDF, triple, variable, + XSD, } from './mod.ts' describe('@okikio/rdf terms and factories', () => { diff --git a/packages/rdf/text.ts b/packages/rdf/text.ts index a9b77cd..45e8edb 100644 --- a/packages/rdf/text.ts +++ b/packages/rdf/text.ts @@ -3,7 +3,7 @@ const DIRECT_CHUNK_SIZE = 16 * 1024 /** Byte/text source accepted by streaming RDF parsers. */ -export type TextSource = +export type TextSourceType = | string | Uint8Array | Iterable @@ -17,7 +17,10 @@ export type TextSource = * source completion. This prevents an upstream producer from continuing work * after a parser or its caller has stopped reading. */ -export async function* chunks(source: TextSource, signal?: AbortSignal): AsyncGenerator { +export async function* chunks( + source: TextSourceType, + signal?: AbortSignal, +): AsyncGenerator { if (typeof source === 'string') { for (let offset = 0; offset < source.length; offset += DIRECT_CHUNK_SIZE) { throwIfAborted(signal) @@ -48,7 +51,11 @@ export async function* chunks(source: TextSource, signal?: AbortSignal): AsyncGe yield item.value } } finally { - if (!complete) await reader.cancel('RDF parser consumer stopped before source completion').catch(() => undefined) + if (!complete) { + await reader.cancel('RDF parser consumer stopped before source completion').catch(() => + undefined + ) + } reader.releaseLock() } } @@ -67,7 +74,6 @@ export async function* chunks(source: TextSource, signal?: AbortSignal): AsyncGe } } - /** Read from the supplied source while preserving caller ownership. */ function read( reader: ReadableStreamDefaultReader, diff --git a/packages/rdf/text_test.ts b/packages/rdf/text_test.ts index d51e02c..07b3a02 100644 --- a/packages/rdf/text_test.ts +++ b/packages/rdf/text_test.ts @@ -3,10 +3,15 @@ import { expect } from '@std/expect' import { chunks, throwIfAborted } from './text.ts' /** Collects parser source chunks as decoded strings for source-contract tests. */ -async function collect(source: Parameters[0], signal?: AbortSignal): Promise { +async function collect( + source: Parameters[0], + signal?: AbortSignal, +): Promise { const decoder = new TextDecoder() const values: string[] = [] - for await (const value of chunks(source, signal)) values.push(typeof value === 'string' ? value : decoder.decode(value)) + for await (const value of chunks(source, signal)) { + values.push(typeof value === 'string' ? value : decoder.decode(value)) + } return values } diff --git a/packages/rdf/transform.ts b/packages/rdf/transform.ts deleted file mode 100644 index 21b3602..0000000 --- a/packages/rdf/transform.ts +++ /dev/null @@ -1,95 +0,0 @@ -/** Internal bridge from Node-style streaming RDF parsers to Web-oriented RDF sources. @module */ - -import { fromQuad } from './factory.ts' -import type { Quad } from './term.ts' -import { chunks, throwIfAborted, type TextSource } from './text.ts' - -/** Minimal stream surface required from an external streaming RDF parser. */ -export interface TransformParserType extends AsyncIterable { - write(chunk: string | Uint8Array): boolean - end(): void - once(event: 'drain' | 'close' | 'error', listener: (...args: unknown[]) => void): this - off(event: 'drain' | 'close' | 'error', listener: (...args: unknown[]) => void): this - destroy(error?: Error): void -} - -/** - * Reads an external Transform parser through the project's async-iterable RDF contract. - * - * The parser resource is owned by this operation. The source is borrowed. Returning - * from iteration early aborts the source pump and destroys the parser so upstream - * network, file, or Web Stream work cannot continue without a consumer. - */ -export async function* parseTransform( - parser: TransformParserType, - source: TextSource, - options: { readonly label: string; readonly signal?: AbortSignal }, -): AsyncGenerator { - throwIfAborted(options.signal) - const lifecycle = new AbortController() - const signal = options.signal ? AbortSignal.any([options.signal, lifecycle.signal]) : lifecycle.signal - let complete = false - const pumping = pump(parser, source, signal, options.label) - - try { - for await (const value of parser) { - throwIfAborted(signal) - yield fromQuad(value) - } - await pumping - complete = true - } finally { - if (!complete) { - lifecycle.abort(new DOMException(`${options.label} consumer stopped before source completion`, 'AbortError')) - parser.destroy() - } - await pumping.catch(() => undefined) - } -} - -/** Feeds source chunks into the parser while respecting writable backpressure. */ -async function pump( - parser: TransformParserType, - source: TextSource, - signal: AbortSignal, - label: string, -): Promise { - try { - for await (const value of chunks(source, signal)) { - throwIfAborted(signal) - if (!parser.write(value)) await drain(parser, signal, label) - } - throwIfAborted(signal) - parser.end() - } catch (error) { - parser.destroy(toError(error)) - throw error - } -} - -/** Waits for parser capacity while remaining interruptible by caller cancellation. */ -function drain(parser: TransformParserType, signal: AbortSignal, label: string): Promise { - return new Promise((resolve, reject) => { - const cleanup = () => { - signal.removeEventListener('abort', onAbort) - parser.off('drain', onDrain) - parser.off('close', onClose) - parser.off('error', onError) - } - const onDrain = () => { cleanup(); resolve() } - const onClose = () => { cleanup(); reject(new Error(`${label} closed while waiting for writable capacity.`)) } - const onError = (value: unknown) => { cleanup(); reject(toError(value)) } - const onAbort = () => { cleanup(); reject(signal.reason ?? new DOMException('Aborted', 'AbortError')) } - - if (signal.aborted) return onAbort() - signal.addEventListener('abort', onAbort, { once: true }) - parser.once('drain', onDrain) - parser.once('close', onClose) - parser.once('error', onError) - }) -} - -/** Converts the supplied value to error without changing semantic identity. */ -function toError(value: unknown): Error { - return value instanceof Error ? value : new Error(String(value)) -} diff --git a/packages/rdf/trig/mod.ts b/packages/rdf/trig/mod.ts index 9bc122b..a1178b8 100644 --- a/packages/rdf/trig/mod.ts +++ b/packages/rdf/trig/mod.ts @@ -1,20 +1,31 @@ /** RDF 1.2 TriG parser and conservative streaming-friendly serializer. @module */ -import { parseCompact, type CompactEvent, type CompactOptions } from '../compact.ts' -import type { TextSource } from '../text.ts' +import { type CompactEventType, type CompactOptionsType, parseCompact } from '../compact.ts' +import type { TextSourceType } from '../text.ts' import type { Quad } from '../term.ts' import { writeQuad, writeTerm } from '../write.ts' -export type { CompactDiagnostic as Diagnostic, CompactEvent as ParseEvent, CompactOptions as ParseOptions, CompactRange as SourceRange } from '../compact.ts' -export type { TextSource } from '../text.ts' +export type { + CompactDiagnosticType as DiagnosticType, + CompactEventType as ParseEventType, + CompactOptionsType as ParseOptionsType, + CompactRangeType as SourceRangeType, +} from '../compact.ts' +export type { TextSourceType } from '../text.ts' /** Emits TriG directives, semantic quads, and optional tolerant diagnostics. */ -export function events(source: TextSource, options: CompactOptions = {}): AsyncGenerator { +export function events( + source: TextSourceType, + options: CompactOptionsType = {}, +): AsyncGenerator { return parseCompact(source, options, true) } /** Parses TriG incrementally and emits semantic RDF dataset quads. */ -export async function* parse(source: TextSource, options: CompactOptions = {}): AsyncGenerator { +export async function* parse( + source: TextSourceType, + options: CompactOptionsType = {}, +): AsyncGenerator { for await (const event of events(source, options)) { if (event.kind === 'quad') yield event.quad } @@ -26,7 +37,10 @@ export async function* parse(source: TextSource, options: CompactOptions = {}): * Repeated graph blocks are legal TriG and let the serializer retain an O(1) * working set for an arbitrary input iteration order. */ -export function serialize(source: Iterable, options: { readonly version?: boolean } = {}): string { +export function serialize(source: Iterable, options: { + /** Version marker retained by this syntax record. */ + readonly version?: boolean +} = {}): string { const lines: string[] = [] if (options.version) lines.push('VERSION "1.2"') for (const value of source) { diff --git a/packages/rdf/trig/parse_test.ts b/packages/rdf/trig/parse_test.ts index c775a65..fe39d8d 100644 --- a/packages/rdf/trig/parse_test.ts +++ b/packages/rdf/trig/parse_test.ts @@ -10,11 +10,15 @@ async function collect(source: AsyncIterable): Promise { describe('@okikio/rdf/trig', () => { it('retains default and named graph identity', async () => { - const values = await collect(parse('PREFIX : \n:a :p :d .\n:g { :a :p :n . }')) + const values = await collect( + parse('PREFIX : \n:a :p :d .\n:g { :a :p :n . }'), + ) expect(values.map((value) => value.graph.value)).toEqual(['', 'https://example.com/g']) }) it('distinguishes a top-level property list from the empty anonymous graph label', async () => { - const property = await collect(parse('@prefix : . [ :inside :value ] :outside :tail .')) + const property = await collect( + parse('@prefix : . [ :inside :value ] :outside :tail .'), + ) expect(property).toHaveLength(2) expect(property.every((item) => item.graph.termType === 'DefaultGraph')).toBe(true) expect(property[0]?.subject.value).toBe(property[1]?.subject.value) @@ -23,5 +27,4 @@ describe('@okikio/rdf/trig', () => { expect(graph).toHaveLength(1) expect(graph[0]?.graph.termType).toBe('BlankNode') }) - }) diff --git a/packages/rdf/turtle/mod.ts b/packages/rdf/turtle/mod.ts index da82f66..7b814ec 100644 --- a/packages/rdf/turtle/mod.ts +++ b/packages/rdf/turtle/mod.ts @@ -1,20 +1,31 @@ /** RDF 1.2 Turtle parser and conservative serializer. @module */ -import { parseCompact, type CompactEvent, type CompactOptions } from '../compact.ts' -import type { TextSource } from '../text.ts' +import { type CompactEventType, type CompactOptionsType, parseCompact } from '../compact.ts' +import type { TextSourceType } from '../text.ts' import type { Quad } from '../term.ts' import { writeQuad } from '../write.ts' -export type { CompactDiagnostic as Diagnostic, CompactEvent as ParseEvent, CompactOptions as ParseOptions, CompactRange as SourceRange } from '../compact.ts' -export type { TextSource } from '../text.ts' +export type { + CompactDiagnosticType as DiagnosticType, + CompactEventType as ParseEventType, + CompactOptionsType as ParseOptionsType, + CompactRangeType as SourceRangeType, +} from '../compact.ts' +export type { TextSourceType } from '../text.ts' /** Emits Turtle directives, semantic quads, and optional tolerant diagnostics. */ -export function events(source: TextSource, options: CompactOptions = {}): AsyncGenerator { +export function events( + source: TextSourceType, + options: CompactOptionsType = {}, +): AsyncGenerator { return parseCompact(source, options, false) } /** Parses Turtle incrementally and emits semantic RDF quads. */ -export async function* parse(source: TextSource, options: CompactOptions = {}): AsyncGenerator { +export async function* parse( + source: TextSourceType, + options: CompactOptionsType = {}, +): AsyncGenerator { for await (const event of events(source, options)) { if (event.kind === 'quad') yield event.quad } @@ -27,11 +38,16 @@ export async function* parse(source: TextSource, options: CompactOptions = {}): * prefix compaction. Named-graph quads are rejected because Turtle represents a * graph, while TriG represents a dataset. */ -export function serialize(source: Iterable, options: { readonly version?: boolean } = {}): string { +export function serialize(source: Iterable, options: { + /** Version marker retained by this syntax record. */ + readonly version?: boolean +} = {}): string { const lines: string[] = [] if (options.version) lines.push('VERSION "1.2"') for (const value of source) { - if (value.graph.termType !== 'DefaultGraph') throw new TypeError('Turtle serialization cannot contain named-graph quads.') + if (value.graph.termType !== 'DefaultGraph') { + throw new TypeError('Turtle serialization cannot contain named-graph quads.') + } lines.push(writeQuad(value, false)) } return lines.length === 0 ? '' : `${lines.join('\n')}\n` diff --git a/packages/rdf/turtle/parse_test.ts b/packages/rdf/turtle/parse_test.ts index 3b475fe..b7e6f86 100644 --- a/packages/rdf/turtle/parse_test.ts +++ b/packages/rdf/turtle/parse_test.ts @@ -25,7 +25,11 @@ describe('@okikio/rdf/turtle', () => { }) it('buffers a malformed statement before tolerant diagnostics are released', async () => { - const values = await collect(events('PREFIX : \n:a :p [ :q :r ; BROKEN ] .\n:b :p :c .', { tolerant: true })) + const values = await collect( + events('PREFIX : \n:a :p [ :q :r ; BROKEN ] .\n:b :p :c .', { + tolerant: true, + }), + ) expect(values.filter((value) => value.kind === 'quad')).toHaveLength(1) expect(values.filter((value) => value.kind === 'diagnostic')).toHaveLength(1) }) diff --git a/packages/rdf/write.ts b/packages/rdf/write.ts index 86ae913..9ac9cfd 100644 --- a/packages/rdf/write.ts +++ b/packages/rdf/write.ts @@ -13,15 +13,24 @@ export function writeQuad(quad: Quad, includeGraph: boolean): string { /** Serializes one RDF term in RDF 1.2 line syntax. */ export function writeTerm(term: Term): string { switch (term.termType) { - case 'NamedNode': return `<${escapeIri(term.value)}>` - case 'BlankNode': return `_:${term.value}` - case 'DefaultGraph': return '' - case 'Variable': return `?${term.value}` - case 'Literal': return writeLiteral(term as Literal) + case 'NamedNode': + return `<${escapeIri(term.value)}>` + case 'BlankNode': + return `_:${term.value}` + case 'DefaultGraph': + return '' + case 'Variable': + return `?${term.value}` + case 'Literal': + return writeLiteral(term as Literal) case 'Quad': { const quad = term as Quad - if (quad.graph.termType !== 'DefaultGraph') throw new TypeError('Embedded RDF triple term must use the default graph.') - return `<<( ${writeTerm(quad.subject)} ${writeTerm(quad.predicate)} ${writeTerm(quad.object)} )>>` + if (quad.graph.termType !== 'DefaultGraph') { + throw new TypeError('Embedded RDF triple term must use the default graph.') + } + return `<<( ${writeTerm(quad.subject)} ${writeTerm(quad.predicate)} ${ + writeTerm(quad.object) + } )>>` } } } @@ -29,20 +38,27 @@ export function writeTerm(term: Term): string { /** Write literal deterministically to the caller-owned output. */ function writeLiteral(literal: Literal): string { const lexical = `"${escapeString(literal.value)}"` - if (literal.language) return `${lexical}@${literal.language}${literal.direction ? `--${literal.direction}` : ''}` + if (literal.language) { + return `${lexical}@${literal.language}${literal.direction ? `--${literal.direction}` : ''}` + } if (literal.datatype.value === XSD.string) return lexical return `${lexical}^^${writeTerm(literal.datatype)}` } /** Escapes the control characters required by N-Triples/N-Quads string literal syntax. */ function escapeString(value: string): string { - return value.replace(/\\/g, '\\\\').replace(/"/g, '\\"').replace(/\t/g, '\\t').replace(/\n/g, '\\n').replace(/\r/g, '\\r') + return value.replace(/\\/g, '\\\\').replace(/"/g, '\\"').replace(/\t/g, '\\t').replace( + /\n/g, + '\\n', + ).replace(/\r/g, '\\r') } /** Escapes characters forbidden directly inside N-Triples/N-Quads IRI references. */ function escapeIri(value: string): string { return value.replace(/[<>"{}|^`\\\u0000-\u0020]/g, (char) => { const point = char.codePointAt(0)! - return point <= 0xffff ? `\\u${point.toString(16).padStart(4, '0').toUpperCase()}` : `\\U${point.toString(16).padStart(8, '0').toUpperCase()}` + return point <= 0xffff + ? `\\u${point.toString(16).padStart(4, '0').toUpperCase()}` + : `\\U${point.toString(16).padStart(8, '0').toUpperCase()}` }) } diff --git a/packages/rdf/write_test.ts b/packages/rdf/write_test.ts index 321d7c2..102b703 100644 --- a/packages/rdf/write_test.ts +++ b/packages/rdf/write_test.ts @@ -10,8 +10,12 @@ describe('@okikio/rdf line serializer primitives', () => { }) it('serializes directional literals and RDF 1.2 triple terms', () => { - expect(writeTerm(literal('bonjour', { language: 'fr', direction: 'ltr' }))).toBe('"bonjour"@fr--ltr') - expect(writeTerm(triple(namedNode('urn:s'), namedNode('urn:p'), literal('o')))).toBe('<<( "o" )>>') + expect(writeTerm(literal('bonjour', { language: 'fr', direction: 'ltr' }))).toBe( + '"bonjour"@fr--ltr', + ) + expect(writeTerm(triple(namedNode('urn:s'), namedNode('urn:p'), literal('o')))).toBe( + '<<( "o" )>>', + ) }) it('includes named graphs only for N-Quads output', () => { diff --git a/packages/rdf/xml/mod.ts b/packages/rdf/xml/mod.ts index cf60c44..7c43772 100644 --- a/packages/rdf/xml/mod.ts +++ b/packages/rdf/xml/mod.ts @@ -1,87 +1,382 @@ -/** Streaming RDF 1.1/1.2 XML parsing behind Web-oriented source contracts. @module */ - -import { blankNode, defaultGraph, literal, namedNode, quad, variable } from '../factory.ts' -import type { Graph, NamedNode, Quad } from '../term.ts' -import { throwIfAborted, type TextSource } from '../text.ts' -import { parseTransform } from '../transform.ts' -import type { ParserConstructorType } from './types.ts' - -export type { ParserConstructorType, ParserType } from './types.ts' - -/** Options for RDF/XML parsing. */ -export interface ParseOptionsType { - /** Initial base IRI used before xml:base declarations are encountered. */ +/** Native RDF 1.1/1.2 XML parsing over the package-owned range-first markup parser. @module */ +import { blankNode, defaultGraph, literal, namedNode, quad, triple } from '../factory.ts' +import { attr, hasAttr, type MarkupElementType, parseMarkup, textContent } from '../markup.ts' +import { + type GraphTermType, + type ObjectTermType, + type Quad, + RDF, + type SubjectTermType, +} from '../term.ts' +import { type TextSourceType, throwIfAborted } from '../text.ts' +/** RDF namespace IRI used to expand RDF/XML syntax names. */ +const RDF_NS = 'http://www.w3.org/1999/02/22-rdf-syntax-ns#', + XML_NS = 'http://www.w3.org/XML/1998/namespace', + ITS_NS = 'http://www.w3.org/2005/11/its', + XML_LITERAL = `${RDF_NS}XMLLiteral` +/** Options for native RDF/XML parsing. */ export interface ParseOptionsType { + /** Initial base IRI used to resolve relative RDF/XML identifiers. */ readonly base?: string - /** Graph assigned to parsed RDF/XML triples. RDF/XML itself serializes one graph. */ - readonly graph?: Graph - /** Reject malformed XML instead of accepting the parser's lenient compatibility mode. Defaults to true. */ - readonly strict?: boolean - /** Include line/column information in parser errors where supported. */ - readonly trackPosition?: boolean - /** Permit repeated rdf:ID values. Defaults to false. */ - readonly allowDuplicateRdfIds?: boolean - /** Validate RDF IRIs. Defaults to true. */ - readonly validateIri?: boolean - /** Parse unsupported rdf:version values instead of rejecting them. Defaults to false. */ - readonly parseUnsupportedVersions?: boolean - /** Version provided by an application/rdf+xml media-type parameter. */ - readonly version?: '1.1' | '1.2-basic' | '1.2' - /** External parser injection used by tests or alternate conforming implementations. */ - readonly parser?: ParserConstructorType - readonly signal?: AbortSignal -} - + /** Target graph. */ readonly graph?: GraphTermType + /** Reject malformed XML. Retained for API clarity; native parsing is strict. */ readonly strict?: + boolean + /** Track positions in future diagnostics. */ readonly trackPosition?: boolean + /** Permit repeated rdf:ID/base pairs. */ readonly allowDuplicateRdfIds?: boolean + /** Validate generated IRIs. */ readonly validateIri?: boolean + /** Accept unknown rdf:version values. */ readonly parseUnsupportedVersions?: boolean + /** Media-type RDF/XML version. */ readonly version?: '1.1' | '1.2-basic' | '1.2' + /** Maximum decoded source bytes. */ readonly maxBytes?: number + /** Maximum markup nodes. */ readonly maxNodes?: number + /** Maximum nesting depth. */ readonly maxDepth?: number + /** Caller cancellation. */ readonly signal?: AbortSignal +} +/** Namespace/base/language state inherited by XML descendants. */ interface ContextType { + /** In-scope XML namespace prefix mappings inherited by this element. */ + readonly ns: ReadonlyMap + /** In-scope base IRI. */ readonly base?: string + /** In-scope language. */ readonly language?: string + /** In-scope direction. */ readonly direction?: 'ltr' | 'rtl' + /** Effective RDF version. */ readonly version: '1.1' | '1.2-basic' | '1.2' +} +/** Per-operation parser state. */ interface StateType { + /** RDF graph that receives quads produced by this RDF/XML parse operation. */ + readonly graph: GraphTermType + /** Original source for XML literal slices. */ readonly text: string + /** Result quads. */ readonly quads: Quad[] + /** Stable rdf:nodeID blank nodes. */ readonly nodes: Map> + /** Used rdf:ID/base keys. */ readonly ids: Set + /** Whether repeated rdf:ID is allowed. */ readonly allowDuplicate: boolean + /** IRI validation switch. */ readonly validateIri: boolean + /** Caller cancellation. */ readonly signal?: AbortSignal +} /** - * Parses RDF/XML incrementally into native `@okikio/rdf` quads. + * Parses RDF/XML directly into native RDF terms. * - * The public source contract stays on strings, byte chunks, async iterables, - * and Web `ReadableStream`s. The external parser's Node-style Transform is an - * implementation detail and is destroyed when the caller stops early. + * Namespace expansion, `xml:base`, language, typed nodes, property attributes, + * collections, XML literals, classic `rdf:ID` reification, RDF 1.2 triple terms, + * and RDF 1.2 annotation reifiers are handled without an external SAX/RDF parser. */ -export async function* parse(source: TextSource, options: ParseOptionsType = {}): AsyncGenerator { - throwIfAborted(options.signal) - const Parser = options.parser ?? await defaultParser() - const parser = new Parser(parserOptions(options)) - yield* parseTransform(parser, source, { label: 'RDF/XML parser', ...(options.signal ? { signal: options.signal } : {}) }) -} - -/** Converts project RDF factory calls into the RDF/JS DataFactory shape expected by the parser. */ -const dataFactory = { - namedNode, - blankNode, - defaultGraph, - variable, - quad, - /** Adapts the RDF/JS literal factory signature while preserving RDF 1.2 directional language literals when an upstream parser supplies direction. */ - literal(value: string, languageOrDatatype?: string | NamedNode, direction?: 'ltr' | 'rtl'): ReturnType { - if (direction !== undefined) { - if (typeof languageOrDatatype !== 'string') throw new TypeError('Directional RDF/XML literal requires a language tag.') - return literal(value, { language: languageOrDatatype, direction }) +export async function* parse( + source: TextSourceType, + options: ParseOptionsType = {}, +): AsyncGenerator { + const document = await parseMarkup(source, { + html: false, + ...(options.maxBytes === undefined ? {} : { maxBytes: options.maxBytes }), + ...(options.maxNodes === undefined ? {} : { maxNodes: options.maxNodes }), + ...(options.maxDepth === undefined ? {} : { maxDepth: options.maxDepth }), + ...(options.signal ? { signal: options.signal } : {}), + }) + const roots = document.children.filter((v): v is MarkupElementType => v.kind === 'element') + if (roots.length !== 1) { + throw new SyntaxError('RDF/XML document must contain one document element.') + } + const root = roots[0]! + const rootContext = context(root, { + ns: new Map([['rdf', RDF_NS], ['xml', XML_NS], ['its', ITS_NS]]), + ...(options.base ? { base: options.base } : {}), + version: options.version ?? '1.2', + }) + const rootName = expandName(root.name, rootContext) + const state: StateType = { + graph: options.graph ?? defaultGraph(), + text: document.text, + quads: [], + nodes: new Map(), + ids: new Set(), + allowDuplicate: options.allowDuplicateRdfIds ?? false, + validateIri: options.validateIri ?? true, + ...(options.signal ? { signal: options.signal } : {}), + } + if (rootName === `${RDF_NS}RDF`) { + const version = attr(root, 'rdf:version') + if ( + version && !['1.1', '1.2-basic', '1.2'].includes(version) && + !(options.parseUnsupportedVersions ?? false) + ) throw new SyntaxError(`Unsupported rdf:version '${version}'.`) + for (const child of children(root)) node(child, rootContext, state) + } else node(root, rootContext, state) + for (const value of state.quads) { + throwIfAborted(options.signal) + yield value + } +} +/** Parses one RDF node element and returns its subject. */ function node( + element: MarkupElementType, + parent: ContextType, + state: StateType, +): SubjectTermType { + throwIfAborted(state.signal) + const ctx = context(element, parent) + const about = attr(element, 'rdf:about'), + id = attr(element, 'rdf:ID'), + nodeId = attr(element, 'rdf:nodeID') + if ([about, id, nodeId].filter((v) => v !== undefined).length > 1) { + throw new SyntaxError('RDF/XML node element cannot combine rdf:about, rdf:ID, and rdf:nodeID.') + } + let subject: SubjectTermType + if (about !== undefined) subject = namedNode(resolveIri(about, ctx.base, state)) + else if (id !== undefined) { + const iri = idIri(id, ctx, state) + subject = namedNode(iri) + } else if (nodeId !== undefined) subject = blank(state, nodeId) + else subject = blankNode() + const type = expandName(element.name, ctx) + if (type !== `${RDF_NS}Description` && type !== `${RDF_NS}RDF`) { + emit(state, subject, RDF.type, namedNode(type)) + } + for (const a of element.attributes) { + const iri = attributeIri(a.name, ctx) + if ( + !iri || control(iri) || iri === `${XML_NS}lang` || iri === `${XML_NS}base` || + iri === `${ITS_NS}dir` + ) continue + emit(state, subject, iri, langLiteral(a.value, ctx)) + } + let li = 1 + for (const child of children(element)) { + let predicate = expandName(child.name, context(child, ctx)) + if (predicate === `${RDF_NS}li`) predicate = `${RDF_NS}_${li++}` + property(child, subject, predicate, ctx, state) + } + return subject +} +/** Parses one RDF property element and emits its statement plus reification/annotation statements. */ function property( + element: MarkupElementType, + subject: SubjectTermType, + predicate: string, + parent: ContextType, + state: StateType, +): void { + const ctx = context(element, parent), + parseType = attr(element, 'rdf:parseType'), + resourceAttr = attr(element, 'rdf:resource'), + nodeId = attr(element, 'rdf:nodeID'), + datatype = attr(element, 'rdf:datatype'), + kids = children(element) + let object: ObjectTermType + if (parseType === 'Resource') { + const b = blankNode() + object = b + for (const a of element.attributes) { + const iri = attributeIri(a.name, ctx) + if (iri && !control(iri)) emit(state, b, iri, langLiteral(a.value, ctx)) + } + let li = 1 + for (const child of kids) { + let p = expandName(child.name, context(child, ctx)) + if (p === `${RDF_NS}li`) p = `${RDF_NS}_${li++}` + property(child, b, p, ctx, state) + } + } else if (parseType === 'Collection') { + const values = kids.map((child) => node(child, ctx, state)) + object = list(values, state) + } else if (parseType === 'Literal') { + object = literal(xmlChildren(element, state), namedNode(XML_LITERAL)) + } else if (parseType === 'Triple') { + if (kids.length !== 1) { + throw new SyntaxError('rdf:parseType="Triple" requires exactly one node element.') + } + const temp: StateType = { ...state, quads: [] } + node(kids[0]!, ctx, temp) + if (temp.quads.length !== 1) { + throw new SyntaxError('RDF/XML triple term must describe exactly one triple.') + } + const q = temp.quads[0]! + object = triple(q.subject, q.predicate, q.object) + } else if (resourceAttr !== undefined) { + object = namedNode(resolveIri(resourceAttr, ctx.base, state)) + } else if (nodeId !== undefined) object = blank(state, nodeId) + else if (kids.length === 1) object = node(kids[0]!, ctx, state) + else if (kids.length > 1) { + throw new SyntaxError('RDF/XML resource property element contains more than one node element.') + } else { + const value = textContent(element) + object = datatype !== undefined + ? literal(value, namedNode(resolveIri(datatype, ctx.base, state))) + : langLiteral(value, ctx) + for (const a of element.attributes) { + const iri = attributeIri(a.name, ctx) + if (iri && !control(iri)) { + const b = blankNode() + object = b + emit(state, b, iri, langLiteral(a.value, ctx)) + } } - return literal(value, languageOrDatatype) - }, -} as const - -/** Builds the external parser options without serializing absent optional fields as undefined. */ -function parserOptions(options: ParseOptionsType): Readonly> { + } + emit(state, subject, predicate, object) + const statement = quad(subject, namedNode(predicate), object) + const id = attr(element, 'rdf:ID') + if (id !== undefined) { + const reifier = namedNode(idIri(id, ctx, state)) + emit(state, reifier, RDF.type, namedNode(RDF.statement)) + emit(state, reifier, RDF.subject, subject) + emit(state, reifier, RDF.predicate, namedNode(predicate)) + emit(state, reifier, RDF.object, object) + } + const annotation = attr(element, 'rdf:annotation'), + annotationNode = attr(element, 'rdf:annotationNodeID') + if (annotation !== undefined || annotationNode !== undefined) { + const reifier = annotation !== undefined + ? namedNode(resolveIri(annotation, ctx.base, state)) + : blank(state, annotationNode!) + emit(state, reifier, RDF.reifies, statement) + } +} +/** Builds an RDF collection and returns its head. */ function list( + values: readonly SubjectTermType[], + state: StateType, +): ObjectTermType { + if (!values.length) return namedNode(RDF.nil) + const head = blankNode() + let cursor = head + for (let i = 0; i < values.length; i++) { + emit(state, cursor, RDF.first, values[i]!) + const last = i === values.length - 1, next = last ? namedNode(RDF.nil) : blankNode() + emit(state, cursor, RDF.rest, next) + if (!last) cursor = next as ReturnType + } + return head +} +/** Applies namespace, base, language, direction, and version declarations. */ function context( + element: MarkupElementType, + parent: ContextType, +): ContextType { + const ns = new Map(parent.ns) + for (const a of element.attributes) { + if (a.name === 'xmlns') ns.set('', a.value) + else if (a.name.startsWith('xmlns:')) ns.set(a.name.slice(6), a.value) + } + const baseRaw = attr(element, 'xml:base'), + base = baseRaw === undefined ? parent.base : resolve(baseRaw, parent.base), + langRaw = attr(element, 'xml:lang'), + dirRaw = attr(element, 'its:dir'), + version = attr(element, 'rdf:version') as ContextType['version'] | undefined return { - dataFactory, - strict: options.strict ?? true, - trackPosition: options.trackPosition ?? true, - allowDuplicateRdfIds: options.allowDuplicateRdfIds ?? false, - validateUri: options.validateIri ?? true, - parseUnsupportedVersions: options.parseUnsupportedVersions ?? false, - ...(options.base === undefined ? {} : { baseIRI: options.base }), - ...(options.graph === undefined ? {} : { defaultGraph: options.graph }), - ...(options.version === undefined ? {} : { version: options.version }), - } -} - -/** Lazily resolved RDF/XML parser constructor so importing the subpath does not initialize the optional processor. */ -let parserPromise: Promise | undefined - -/** Lazily imports the RDF/XML implementation only when the subpath is used. */ -async function defaultParser(): Promise { - parserPromise ??= import('rdfxml-streaming-parser').then((module) => module.RdfXmlParser as unknown as ParserConstructorType) - return await parserPromise + ns, + ...(base ? { base } : {}), + ...(langRaw !== undefined + ? (langRaw ? { language: langRaw.toLowerCase() } : {}) + : parent.language + ? { language: parent.language } + : {}), + ...(dirRaw === 'ltr' || dirRaw === 'rtl' + ? { direction: dirRaw } + : parent.direction + ? { direction: parent.direction } + : {}), + version: version ?? parent.version, + } +} +/** Expands an element/attribute QName through the in-scope namespace map. */ function expandName( + name: string, + ctx: ContextType, +): string { + const colon = name.indexOf(':') + if (colon < 0) { + const base = ctx.ns.get('') + if (!base) throw new SyntaxError(`Unqualified RDF/XML name '${name}' has no default namespace.`) + return `${base}${name}` + } + const base = ctx.ns.get(name.slice(0, colon)) + if (!base) throw new SyntaxError(`Unknown XML namespace prefix '${name.slice(0, colon)}'.`) + return `${base}${name.slice(colon + 1)}` +} +/** Expands a non-xmlns attribute QName. */ function attributeIri( + name: string, + ctx: ContextType, +): string | undefined { + if (name === 'xmlns' || name.startsWith('xmlns:')) return undefined + const colon = name.indexOf(':') + if (colon < 0) return undefined + const base = ctx.ns.get(name.slice(0, colon)) + return base ? `${base}${name.slice(colon + 1)}` : undefined +} +/** Tests RDF/XML structural/control attribute IRIs. */ function control(iri: string) { + return new Set([ + `${RDF_NS}ID`, + `${RDF_NS}about`, + `${RDF_NS}annotation`, + `${RDF_NS}annotationNodeID`, + `${RDF_NS}parseType`, + `${RDF_NS}resource`, + `${RDF_NS}nodeID`, + `${RDF_NS}datatype`, + `${RDF_NS}version`, + ]).has(iri) +} +/** Returns child elements only. */ function children(element: MarkupElementType) { + return element.children.filter((v): v is MarkupElementType => v.kind === 'element') +} +/** Resolves and validates one IRI. */ function resolveIri( + value: string, + base: string | undefined, + state: StateType, +) { + const iri = resolve(value, base) + if (!iri) throw new SyntaxError(`Invalid RDF/XML IRI '${value}'.`) + if (state.validateIri) { + try { + new URL(iri) + } catch { + throw new SyntaxError(`Invalid RDF/XML IRI '${iri}'.`) + } + } + return iri +} +/** Resolves an IRI reference. */ function resolve(value: string, base?: string) { + try { + return base ? new URL(value, base).href : new URL(value).href + } catch { + return undefined + } +} +/** Resolves rdf:ID and enforces document uniqueness. */ function idIri( + value: string, + ctx: ContextType, + state: StateType, +) { + if (!/^[A-Za-z_][A-Za-z0-9._-]*$/u.test(value)) { + throw new SyntaxError(`Invalid rdf:ID '${value}'.`) + } + const iri = resolveIri(`#${value}`, ctx.base, state), key = `${ctx.base ?? ''}\0${value}` + if (!state.allowDuplicate && state.ids.has(key)) { + throw new SyntaxError(`Duplicate rdf:ID '${value}'.`) + } + state.ids.add(key) + return iri +} +/** Gets a stable blank node for rdf:nodeID. */ function blank(state: StateType, id: string) { + let value = state.nodes.get(id) + if (!value) { + value = blankNode(id) + state.nodes.set(id, value) + } + return value +} +/** Creates plain/language/directional literal from inherited context. */ function langLiteral( + value: string, + ctx: ContextType, +): ObjectTermType { + return ctx.language + ? literal(value, { + language: ctx.language, + ...(ctx.direction ? { direction: ctx.direction } : {}), + }) + : literal(value) +} +/** Emits one quad to operation buffer. */ function emit( + state: StateType, + subject: SubjectTermType, + predicate: string, + object: ObjectTermType, +) { + state.quads.push(quad(subject, namedNode(predicate), object, state.graph)) +} +/** Returns original child markup for rdf:XMLLiteral. */ function xmlChildren( + element: MarkupElementType, + state: StateType, +) { + if (!element.children.length) return '' + return state.text.slice(element.children[0]!.start, element.children.at(-1)!.end) } diff --git a/packages/rdf/xml/mod_test.ts b/packages/rdf/xml/mod_test.ts index 9979ddf..0eba265 100644 --- a/packages/rdf/xml/mod_test.ts +++ b/packages/rdf/xml/mod_test.ts @@ -1,138 +1,67 @@ import { describe, it } from 'node:test' import { expect } from '@std/expect' -import { literal, namedNode, quad, type Quad } from '../mod.ts' -import { parse, type ParserType } from './mod.ts' +import { RDF } from '../mod.ts' +import { parse } from './mod.ts' -const fixture = quad(namedNode('https://example.test/s'), namedNode('https://example.test/p'), literal('value')) -type EventType = 'drain' | 'close' | 'error' -type ListenerType = (...args: unknown[]) => void - -/** Minimal event surface for testing the Node Transform bridge without owning a host emitter dependency. */ -class Emitter { - private readonly listeners = new Map>() - - once(event: EventType, listener: ListenerType): this { - const wrapper: ListenerType = (...args) => { this.off(event, wrapper); listener(...args) } - const values = this.listeners.get(event) ?? new Set() - values.add(wrapper) - this.listeners.set(event, values) - return this - } - - off(event: EventType, listener: ListenerType): this { - this.listeners.get(event)?.delete(listener) - return this - } - - protected emit(event: EventType, ...args: unknown[]): void { - for (const listener of [...(this.listeners.get(event) ?? [])]) listener(...args) - } -} - -/** Parser fixture that records constructor options and exercises writable backpressure. */ -class TestParser extends Emitter implements ParserType { - static options: Readonly> | undefined - private ended = false - private resolveEnd: (() => void) | undefined - private first = true - - constructor(options: Readonly>) { - super() - TestParser.options = options - } - - write(_value: string | Uint8Array): boolean { - if (this.first) { - this.first = false - queueMicrotask(() => this.emit('drain')) - return false - } - return true - } - - end(): void { this.ended = true; this.resolveEnd?.() } - destroy(error?: Error): void { - if (error) this.emit('error', error) - this.ended = true - this.resolveEnd?.() - this.emit('close') - } - - async *[Symbol.asyncIterator](): AsyncIterator { - if (!this.ended) await new Promise((resolve) => { this.resolveEnd = resolve }) - yield fixture - } +async function all(source: string) { + const values = [] + for await (const value of parse(source)) values.push(value) + return values } -/** Parser fixture that leaves output open so early iterator return must destroy it. */ -class EarlyParser extends Emitter implements ParserType { - static destroyed = false - private available = false - private resolveValue: (() => void) | undefined - - constructor(_options: Readonly>) { - super() - EarlyParser.destroyed = false - } - - write(_value: string | Uint8Array): boolean { - this.available = true - this.resolveValue?.() - return true - } - - end(): void {} - destroy(error?: Error): void { - EarlyParser.destroyed = true - if (error) this.emit('error', error) - this.emit('close') - } - - async *[Symbol.asyncIterator](): AsyncIterator { - if (!this.available) await new Promise((resolve) => { this.resolveValue = resolve }) - yield fixture - await new Promise(() => {}) - } -} +const head = + `` describe('@okikio/rdf/xml', () => { - it('forwards RDF/XML options and supplies the native RDF 1.2 factory', async () => { - const values: Quad[] = [] - for await (const value of parse('', { - parser: TestParser, - base: 'https://example.test/base/', - version: '1.2', - })) values.push(value) + it('parses typed nodes, language literals, and resource properties natively', async () => { + const values = await all( + `${head}Widget`, + ) + expect( + values.some((value) => + value.predicate.value === RDF.type && value.object.value === 'http://example.org/Thing' + ), + ).toBe(true) + expect( + values.some((value) => + value.predicate.value === 'http://example.org/name' && + value.object.termType === 'Literal' && value.object.language === 'en' + ), + ).toBe(true) + expect( + values.some((value) => + value.predicate.value === 'http://example.org/link' && + value.object.value === 'http://example.test/other' + ), + ).toBe(true) + }) - expect(values).toHaveLength(1) - expect(values[0]?.equals(fixture)).toBe(true) - expect(TestParser.options?.strict).toBe(true) - expect(TestParser.options?.trackPosition).toBe(true) - expect(TestParser.options?.baseIRI).toBe('https://example.test/base/') - expect(TestParser.options?.version).toBe('1.2') - const factory = TestParser.options?.dataFactory as { - literal(value: string, language?: string, direction?: 'ltr' | 'rtl'): ReturnType - } - const directional = factory.literal('مرحبا', 'ar', 'rtl') - expect(directional.language).toBe('ar') - expect(directional.direction).toBe('rtl') + it('parses resource and collection parse types', async () => { + const values = await all( + `${head}x`, + ) + expect(values.some((value) => value.predicate.value === RDF.first)).toBe(true) + expect(values.some((value) => value.predicate.value === RDF.rest)).toBe(true) + expect(values.some((value) => value.predicate.value === 'http://example.org/name')).toBe(true) }) - it('honors writable backpressure before ending the external parser', async () => { - const values: Quad[] = [] - for await (const value of parse([''], { parser: TestParser })) values.push(value) + it('creates an RDF 1.2 triple term without asserting the quoted triple', async () => { + const values = await all( + `${head}`, + ) expect(values).toHaveLength(1) + expect(values[0]?.object.termType).toBe('Quad') + if (values[0]?.object.termType === 'Quad') { + expect(values[0].object.predicate.value).toBe('http://example.org/p') + } }) - it('cancels a pending Web Stream read and destroys the parser on early return', async () => { - let cancelled = false - const stream = new ReadableStream({ - start(controller) { controller.enqueue(new TextEncoder().encode('')) }, - cancel() { cancelled = true }, - }) - - for await (const _value of parse(stream, { parser: EarlyParser })) break - expect(EarlyParser.destroyed).toBe(true) - expect(cancelled).toBe(true) + it('emits classic rdf:ID reification and RDF 1.2 annotation reifiers', async () => { + const values = await all( + `${head}value`, + ) + expect(values.some((value) => value.predicate.value === RDF.subject)).toBe(true) + const annotation = values.find((value) => value.predicate.value === RDF.reifies) + expect(annotation?.object.termType).toBe('Quad') }) }) diff --git a/packages/rdf/xml/types.ts b/packages/rdf/xml/types.ts deleted file mode 100644 index fea5137..0000000 --- a/packages/rdf/xml/types.ts +++ /dev/null @@ -1,11 +0,0 @@ -/** Structural RDF/XML parser contracts used to hide the external stream implementation. @module */ - -import type { TransformParserType } from '../transform.ts' - -/** RDF/XML parser stream shape required by the adapter. */ -export type ParserType = TransformParserType - -/** Constructor contract for an RDF/XML parser implementation. */ -export interface ParserConstructorType { - new (options: Readonly>): ParserType -} -- 2.51.2