diff --git a/playground/src/routes/+page.svelte b/playground/src/routes/+page.svelte index 68f5b82..c9cf0de 100644 --- a/playground/src/routes/+page.svelte +++ b/playground/src/routes/+page.svelte @@ -65,7 +65,9 @@ let record = $derived(parseResult.record); let jsonError = $derived(parseResult.error); - let paths: PathInfo[] = $derived(record ? enumerate(record) : []); + let paths: PathInfo[] = $derived( + record ? Array.from(enumerate(record), ([p]) => p) : [] + ); let pathResult = $derived.by(() => { const trimmed = pathInput.trim(); diff --git a/ref-impl-js/src/index.ts b/ref-impl-js/src/index.ts index 8fe87d4..bdd96a7 100644 --- a/ref-impl-js/src/index.ts +++ b/ref-impl-js/src/index.ts @@ -221,48 +221,65 @@ function applyQualifiers( } // Enumerate all RecordPaths reachable from a record. -export function enumerate(record: Record): PathInfo[] { - const paths = new Map(); - enumObject(record, '', false, paths); - return Array.from(paths.values()); +// Returns a generator yielding [PathInfo, value] pairs, deduplicated by path. +export function* enumerate( + record: Record +): Generator<[PathInfo, unknown]> { + const seen = new Set(); + yield* enumObject(seen, record, '', false); } -function enumObject( +function* enumObject( + seen: Set, obj: Record, prefix: string, - isVector: boolean, - paths: Map -) { + isVector: boolean +): Generator<[PathInfo, unknown]> { + const vtype = isVector ? 'vector' : 'scalar'; + for (const key of Object.keys(obj)) { const child = obj[key]; const escaped = escapeFieldName(key); const keyPath = prefix ? prefix + '.' + escaped : escaped; - const vtype = isVector ? 'vector' : 'scalar'; if (child === null || child === undefined || typeof child !== 'object') { - paths.set(keyPath, { path: keyPath, type: vtype }); + if (!seen.has(keyPath)) { + seen.add(keyPath); + yield [{ path: keyPath, type: vtype }, child]; + } } else if (Array.isArray(child)) { - paths.set(keyPath, { path: keyPath, type: vtype }); - enumArray(child, keyPath, isVector, paths); + if (!seen.has(keyPath)) { + seen.add(keyPath); + yield [{ path: keyPath, type: vtype }, child]; + } + yield* enumArray(seen, child, keyPath); } else if ((child as Record).$type) { - paths.set(keyPath, { path: keyPath, type: vtype }); + if (!seen.has(keyPath)) { + seen.add(keyPath); + yield [{ path: keyPath, type: vtype }, child]; + } const nsid = (child as Record).$type as string; const qualified = keyPath + '{' + nsid + '}'; - paths.set(qualified, { path: qualified, type: vtype }); - enumObject(child as Record, qualified, isVector, paths); + if (!seen.has(qualified)) { + seen.add(qualified); + yield [{ path: qualified, type: vtype }, child]; + } + yield* enumObject(seen, child as Record, qualified, isVector); } else { - paths.set(keyPath, { path: keyPath, type: vtype }); - enumObject(child as Record, keyPath, isVector, paths); + if (!seen.has(keyPath)) { + seen.add(keyPath); + yield [{ path: keyPath, type: vtype }, child]; + } + yield* enumObject(seen, child as Record, keyPath, isVector); } } } -function enumArray( +function* enumArray( + seen: Set, arr: unknown[], - prefix: string, - _parentIsVector: boolean, - paths: Map -) { + prefix: string +): Generator<[PathInfo, unknown]> { const hasUnion = arr.some( (el) => typeof el === 'object' && @@ -272,53 +289,60 @@ function enumArray( ); if (hasUnion) { - const byType: Record[]> = {}; - const plain: unknown[] = []; + let hasPlain = false; for (const el of arr) { - if ( - typeof el === 'object' && - el !== null && - !Array.isArray(el) && - (el as Record).$type - ) { - const nsid = (el as Record).$type as string; - (byType[nsid] || (byType[nsid] = [])).push(el as Record); + const nsid = + typeof el === 'object' && el !== null && !Array.isArray(el) + ? ((el as Record).$type as string | undefined) + : undefined; + if (nsid) { + const qp = prefix + '[' + nsid + ']'; + if (!seen.has(qp)) { + seen.add(qp); + yield [{ path: qp, type: 'vector' }, el]; + } + if (typeof el === 'object' && el !== null && !Array.isArray(el)) { + yield* enumObject(seen, el as Record, qp, true); + } } else { - plain.push(el); + hasPlain = true; + yield* enumValue(seen, el, prefix + '[]'); } } - for (const [nsid, elements] of Object.entries(byType)) { - const qp = prefix + '[' + nsid + ']'; - paths.set(qp, { path: qp, type: 'vector' }); - for (const el of elements) { - enumObject(el, qp, true, paths); + if (hasPlain) { + const bare = prefix + '[]'; + if (!seen.has(bare)) { + seen.add(bare); + yield [{ path: bare, type: 'vector' }, arr]; } } - if (plain.length > 0) { - enumPlainArray(plain, prefix + '[]', paths); - } } else { - enumPlainArray(arr, prefix + '[]', paths); + const bare = prefix + '[]'; + if (!seen.has(bare)) { + seen.add(bare); + yield [{ path: bare, type: 'vector' }, arr]; + } + for (const el of arr) { + yield* enumValue(seen, el, bare); + } } } -function enumPlainArray(arr: unknown[], prefix: string, paths: Map) { - paths.set(prefix, { path: prefix, type: 'vector' }); - for (const el of arr) { - if (el === null || el === undefined || typeof el !== 'object') { - // scalar elements — path is the array prefix itself - } else if (Array.isArray(el)) { - enumArray(el, prefix, true, paths); - } else { - enumObject(el as Record, prefix, true, paths); - } +function* enumValue( + seen: Set, + value: unknown, + prefix: string +): Generator<[PathInfo, unknown]> { + if (value === null || value === undefined || typeof value !== 'object') { + return; + } + if (Array.isArray(value)) { + yield* enumArray(seen, value, prefix); + } else { + yield* enumObject(seen, value as Record, prefix, true); } } -// TODO: enumerateMatching? pass a test fn that accepts (path, value) pairs and returns bool -// could just filter the output of enumerate, but this avoids collecting lots of stuff we don't need -// (or enumerate could be a generator?? that might be even nicer) - export function isVector(pathStr: string): boolean { for (let i = 0; i < pathStr.length; i++) { if (pathStr[i] === '!' && i + 1 < pathStr.length) { diff --git a/ref-impl-js/test/interop.test.ts b/ref-impl-js/test/interop.test.ts index 9f89317..9cffec1 100644 --- a/ref-impl-js/test/interop.test.ts +++ b/ref-impl-js/test/interop.test.ts @@ -31,10 +31,10 @@ const enumFixture = loadFixture('enumerate.json'); describe('enumerate', () => { for (const t of enumFixture.tests) { it(t.description, () => { - const result = enumerate(t.record); - // Compare as sets of {path, type} - const resultSet = new Set(result.map((p: PathInfo) => `${p.path}:${p.type}`)); + const resultSet = new Set( + Array.from(enumerate(t.record), ([p]) => `${p.path}:${p.type}`) + ); const expectedSet = new Set( t.expected.map((p: { path: string; type: string }) => `${p.path}:${p.type}`) ); diff --git a/ref-impl-rust/src/lib.rs b/ref-impl-rust/src/lib.rs index d27f7e7..65aebe1 100644 --- a/ref-impl-rust/src/lib.rs +++ b/ref-impl-rust/src/lib.rs @@ -330,125 +330,277 @@ fn apply_qualifiers( // -- Enumerator -- -struct PathCollector { +const DEFAULT_MAX_DEPTH: usize = 64; + +/// Returns a lazy iterator over all `(PathInfo, &Value)` pairs reachable from +/// a record. Paths are deduplicated; each unique path is yielded once. +pub fn enumerate(record: &Value) -> Paths<'_> { + Paths::new(record, DEFAULT_MAX_DEPTH) +} + +/// Work items for the stack-based tree walk. +enum Work<'a> { + /// Yield this path+value if not yet seen. + Emit { + path: String, + path_type: PathType, + value: &'a Value, + }, + /// Expand an object's entries onto the stack. + Object { + obj: &'a serde_json::Map, + prefix: String, + is_vector: bool, + depth: usize, + }, + /// Expand an array's elements onto the stack. + Array { + arr: &'a [Value], + arr_value: &'a Value, + prefix: String, + depth: usize, + }, +} + +pub struct Paths<'a> { + stack: Vec>, seen: HashSet, - paths: Vec, + max_depth: usize, } -impl PathCollector { - fn new() -> Self { - Self { +impl<'a> Paths<'a> { + fn new(record: &'a Value, max_depth: usize) -> Self { + let mut paths = Self { + stack: Vec::new(), seen: HashSet::new(), - paths: Vec::new(), - } - } - - fn insert(&mut self, path: &str, path_type: PathType) { - if self.seen.insert(path.to_string()) { - self.paths.push(PathInfo { - path: path.to_string(), - path_type, + max_depth, + }; + if let Some(obj) = record.as_object() { + paths.stack.push(Work::Object { + obj, + prefix: String::new(), + is_vector: false, + depth: 0, }); } + paths } -} -pub fn enumerate(record: &Value) -> Vec { - let mut collector = PathCollector::new(); - if let Some(obj) = record.as_object() { - enum_object(obj, "", false, &mut collector); + pub fn with_max_depth(mut self, max_depth: usize) -> Self { + self.max_depth = max_depth; + self } - collector.paths -} - -fn enum_object( - obj: &serde_json::Map, - prefix: &str, - is_vector: bool, - out: &mut PathCollector, -) { - let vtype = if is_vector { - PathType::Vector - } else { - PathType::Scalar - }; - for (key, child) in obj { - let escaped = escape_field_name(key); - let key_path = if prefix.is_empty() { - escaped + fn expand_object( + &mut self, + obj: &'a serde_json::Map, + prefix: &str, + is_vector: bool, + depth: usize, + ) { + let vtype = if is_vector { + PathType::Vector } else { - format!("{prefix}.{escaped}") + PathType::Scalar }; - out.insert(&key_path, vtype); + // Push in reverse so the first key is at the top of the stack. + let entries: Vec<_> = obj.iter().collect(); + for (key, child) in entries.into_iter().rev() { + let escaped = escape_field_name(key); + let key_path = if prefix.is_empty() { + escaped + } else { + format!("{prefix}.{escaped}") + }; - match child { - Value::Array(arr) => enum_array(arr, &key_path, is_vector, out), - Value::Object(child_obj) => { - match child_obj.get("$type").and_then(|t| t.as_str()) { - Some(nsid) => { - let qualified = format!("{key_path}{{{nsid}}}"); - out.insert(&qualified, vtype); - enum_object(child_obj, &qualified, is_vector, out); + // Push children first (deeper in stack), then the emit (top). + match child { + Value::Array(arr) => { + self.stack.push(Work::Array { + arr, + arr_value: child, + prefix: key_path.clone(), + depth: depth + 1, + }); + self.stack.push(Work::Emit { + path: key_path, + path_type: vtype, + value: child, + }); + } + Value::Object(child_obj) => { + match child_obj.get("$type").and_then(|t| t.as_str()) { + Some(nsid) => { + let qualified = format!("{key_path}{{{nsid}}}"); + self.stack.push(Work::Object { + obj: child_obj, + prefix: qualified.clone(), + is_vector, + depth: depth + 1, + }); + self.stack.push(Work::Emit { + path: qualified, + path_type: vtype, + value: child, + }); + self.stack.push(Work::Emit { + path: key_path, + path_type: vtype, + value: child, + }); + } + None => { + self.stack.push(Work::Object { + obj: child_obj, + prefix: key_path.clone(), + is_vector, + depth: depth + 1, + }); + self.stack.push(Work::Emit { + path: key_path, + path_type: vtype, + value: child, + }); + } } - None => enum_object(child_obj, &key_path, is_vector, out), + } + _ => { + self.stack.push(Work::Emit { + path: key_path, + path_type: vtype, + value: child, + }); } } - _ => {} // scalars already inserted } } -} -fn enum_array(arr: &[Value], prefix: &str, _is_vector: bool, out: &mut PathCollector) { - let has_union = arr - .iter() - .any(|el| el.as_object().is_some_and(|o| o.contains_key("$type"))); - - if has_union { - // Partition into typed (union) and plain elements. - // Use a Vec of pairs to preserve encounter order across types. - let mut seen_types = HashSet::new(); - let mut has_plain = false; - - for el in arr { - match el.as_object().and_then(|o| o.get("$type")).and_then(|t| t.as_str()) { - Some(nsid) => { - let qp = format!("{prefix}[{nsid}]"); - if seen_types.insert(nsid.to_string()) { - out.insert(&qp, PathType::Vector); + fn expand_array( + &mut self, + arr: &'a [Value], + arr_value: &'a Value, + prefix: &str, + depth: usize, + ) { + let has_union = arr + .iter() + .any(|el| el.as_object().is_some_and(|o| o.contains_key("$type"))); + + if has_union { + let mut has_plain = false; + + for el in arr.iter().rev() { + match el + .as_object() + .and_then(|o| o.get("$type")) + .and_then(|t| t.as_str()) + { + Some(nsid) => { + let qp = format!("{prefix}[{nsid}]"); + if let Some(obj) = el.as_object() { + self.stack.push(Work::Object { + obj, + prefix: qp.clone(), + is_vector: true, + depth: depth + 1, + }); + } + self.stack.push(Work::Emit { + path: qp, + path_type: PathType::Vector, + value: el, + }); } - if let Some(obj) = el.as_object() { - enum_object(obj, &qp, true, out); + None => { + has_plain = true; + self.expand_child_value(el, &format!("{prefix}[]"), depth + 1); } } - None => has_plain = true, } - } - if has_plain { + + if has_plain { + self.stack.push(Work::Emit { + path: format!("{prefix}[]"), + path_type: PathType::Vector, + value: arr_value, + }); + } + } else { let bare = format!("{prefix}[]"); - out.insert(&bare, PathType::Vector); - for el in arr.iter().filter(|el| { - !el.as_object() - .is_some_and(|o| o.contains_key("$type")) - }) { - enum_value(el, &bare, out); + + for el in arr.iter().rev() { + self.expand_child_value(el, &bare, depth + 1); } + + self.stack.push(Work::Emit { + path: bare, + path_type: PathType::Vector, + value: arr_value, + }); } - } else { - let bare = format!("{prefix}[]"); - out.insert(&bare, PathType::Vector); - for el in arr { - enum_value(el, &bare, out); + } + + fn expand_child_value(&mut self, value: &'a Value, prefix: &str, depth: usize) { + match value { + Value::Object(obj) => { + self.stack.push(Work::Object { + obj, + prefix: prefix.to_string(), + is_vector: true, + depth, + }); + } + Value::Array(arr) => { + self.stack.push(Work::Array { + arr, + arr_value: value, + prefix: prefix.to_string(), + depth, + }); + } + _ => {} } } } -fn enum_value(value: &Value, prefix: &str, out: &mut PathCollector) { - match value { - Value::Array(inner) => enum_array(inner, prefix, true, out), - Value::Object(obj) => enum_object(obj, prefix, true, out), - _ => {} +impl<'a> Iterator for Paths<'a> { + type Item = (PathInfo, &'a Value); + + fn next(&mut self) -> Option { + loop { + match self.stack.pop()? { + Work::Emit { + path, + path_type, + value, + } => { + if self.seen.insert(path.clone()) { + return Some((PathInfo { path, path_type }, value)); + } + } + Work::Object { + obj, + prefix, + is_vector, + depth, + } => { + if depth <= self.max_depth { + self.expand_object(obj, &prefix, is_vector, depth); + } + } + Work::Array { + arr, + arr_value, + prefix, + depth, + } => { + if depth <= self.max_depth { + self.expand_array(arr, arr_value, &prefix, depth); + } + } + } + } } } diff --git a/ref-impl-rust/tests/interop.rs b/ref-impl-rust/tests/interop.rs index 90100d4..1266386 100644 --- a/ref-impl-rust/tests/interop.rs +++ b/ref-impl-rust/tests/interop.rs @@ -55,10 +55,8 @@ struct EnumExpected { fn enumerate_tests() { let f: EnumFixture = serde_json::from_str(ENUMERATE_JSON).unwrap(); for t in &f.tests { - let result = enumerate(&t.record); - let result_set: HashSet = result - .iter() - .map(|p| { + let result_set: HashSet = enumerate(&t.record) + .map(|(p, _value)| { let ty = match p.path_type { PathType::Scalar => "scalar", PathType::Vector => "vector", diff --git a/spec.md b/spec.md index f4171de..5f40029 100644 --- a/spec.md +++ b/spec.md @@ -223,7 +223,7 @@ RecordPath depends on several properties of the atproto [data model](https://atp Lexicon evolution rules forbid **type changes** across schema revisions, so a plain object ref cannot be converted to a union in a lexicon-forward-compatibility-compliant revision under the same NSID, so RecordPath canonicalization is hopefully safe from forward-compatible lexicon changes. -todo: +#### todo - api recommendations for scalar vs vector queries @@ -235,3 +235,9 @@ todo: - backlinks example: match links on non-link fields if they happen to parse - bring back the expected order of matches: depth-first-search order (probably as a "should") + +- include cbor/drisl -- keep json for examples, but all this should be applicable + +#### questions + +- should an empty RecordPath be legal? would match the entire record.