diff --git a/.zed/settings.json b/.zed/settings.json new file mode 100644 index 0000000..16f470c --- /dev/null +++ b/.zed/settings.json @@ -0,0 +1,11 @@ +// Folder-specific settings +// +// For a full list of overridable settings, and general information on folder-specific settings, +// see the documentation: https://zed.dev/docs/configuring-zed#settings-files +{ + "languages": { + "Erlang": { + "format_on_save": "off" + } + } +} diff --git a/TODO b/TODO index a20a785..384c2a5 100644 --- a/TODO +++ b/TODO @@ -1,6 +1,6 @@ [x] `swap_delete` -[ ] `n_chunks` -[ ] `sized_chunks` +[x] `n_chunks` / split_n +[x] `sized_chunks` [x] deprecate `length` for `size` [-] `reduce` [x] deprecate first/last rest/leading diff --git a/create_concat_plan.py b/create_concat_plan.py new file mode 100644 index 0000000..4d8c69c --- /dev/null +++ b/create_concat_plan.py @@ -0,0 +1,144 @@ +#!/usr/bin/env python3 +""" +Implementation of CREATE-CONCAT-PLAN algorithm from RRB-Vector paper. +This shows how nodes should be rebalanced during concatenation. +""" + +def create_concat_plan(T, M=16, emax=2): + """ + Create a rebalancing plan for concatenating nodes. + + Args: + T: List of node sizes (number of elements/children in each node) + M: Branching factor (default 16 for Erlang) + emax: Maximum extra search steps (default 2) + + Returns: + (c, n) where: + c: List of rebalanced node sizes + n: Number of nodes after rebalancing + """ + # Line 2: Create array to hold node sizes + c = [0] * len(T) + + # Line 3-7: Initialize c with input sizes and calculate total + S = 0 + for i in range(len(T)): + c[i] = T[i] + S += T[i] + + # Line 8: Calculate optimal number of nodes + T_opt = (S + M - 1) // M # Ceiling division + + # Line 9: Current number of nodes + n = len(T) + + # Line 10: Start from first node + i = 0 + + # Line 11: While we have too many nodes compared to optimal + print(f"Initial: T_opt={T_opt}, emax={emax}, n={n}") + print(f"Condition: T_opt + emax < n → {T_opt} + {emax} < {n} → {T_opt + emax < n}") + print(f"Initial sizes: {c[:n]}") + print() + + iteration = 0 + while T_opt + emax < n: + iteration += 1 + print(f"--- Iteration {iteration} ---") + print(f"Starting at i={i}, n={n}") + + # Line 12-14: Find first node that's too small + while c[i] >= M - emax // 2: + print(f" c[{i}]={c[i]} >= {M - emax//2}, skipping") + i += 1 + + print(f" Found small node at i={i}: c[{i}]={c[i]}") + + # Line 15: Remainder to distribute + r = c[i] + print(f" r={r} (will distribute this over remaining nodes)") + + # Line 16-21: Redistribute r over following nodes + dist_start = i + while r > 0: + print(f" i={i}, r={r}, c[{i+1}]={c[i+1] if i+1 < n else 'N/A'}") + MIN_SIZE = min(r + c[i + 1], M) + c[i] = MIN_SIZE + r = r + c[i + 1] - c[i] + print(f" → c[{i}]={c[i]}, new r={r}") + i += 1 + + print(f" After redistribution from {dist_start}: {c[dist_start:i]}") + + # Line 22-25: Shift remaining nodes left + print(f" Shifting nodes {i}..{n-1} left by 1") + for j in range(i, n - 1): + c[j] = c[j + 1] + + # Line 26-27: Adjust position and count + i -= 1 + n -= 1 + + print(f" After shift: n={n}, i={i}") + print(f" Current sizes: {c[:n]}") + print(f" Check: T_opt + emax < n → {T_opt} + {emax} < {n} → {T_opt + emax < n}") + print() + + # Line 29: Return rebalanced plan + return c[:n], n + + +if __name__ == "__main__": + print("=" * 60) + print("Testing CREATE-CONCAT-PLAN with [14, 3, 14]") + print("=" * 60) + print() + + # Test case from the issue + T = [14, 3, 14] + M = 16 + emax = 2 + + print(f"Input: {T}") + print(f"M (branch factor): {M}") + print(f"emax: {emax}") + print() + + result, n = create_concat_plan(T, M, emax) + + print("=" * 60) + print("RESULT") + print("=" * 60) + print(f"Original: {T} (sum={sum(T)})") + print(f"Rebalanced: {result} (sum={sum(result)})") + print(f"Number of nodes: {len(T)} → {n}") + print() + + # Verify + assert sum(result) == sum(T), "Total elements changed!" + print("✓ Total elements preserved") + + # Check if all nodes satisfy minimum size + min_size = M - emax + print(f"\nChecking minimum size constraint (>= {min_size}):") + for i, size in enumerate(result): + status = "✓" if size >= min_size else "✗" + print(f" Node {i}: {size} {status}") + + print("\n" + "=" * 60) + print("Testing with other cases") + print("=" * 60) + + test_cases = [ + [14, 14], + [14, 14, 14, 14], + [10, 10, 10], + [16, 3], + ] + + for T in test_cases: + print(f"\nInput: {T}") + result, n = create_concat_plan(T, M, emax) + print(f"Result: {result}") + print() diff --git a/src/iv.gleam b/src/iv.gleam index 8db9d2a..4cd1fe2 100644 --- a/src/iv.gleam +++ b/src/iv.gleam @@ -108,12 +108,14 @@ //// [concat_list](#concat_list "Concatenate many arrays"), //// [flatten](#flatten "Concatenate nested arrays"), //// [split](#split "Split an array at an index"), +//// [split_n](#split_n "N-way split") //// [slice](#slice "Get a slice of the array"), //// [slice_clamped](#slice_clamped "Get a slice of the array"), //// [drop_first](#drop_first "Remove the first elements"), //// [drop_last](#drop_last "Remove the last elements"), //// [take_first](#take_first "Take the first elements"), //// [take_last](#take_last "Take the last elements"), +//// [sized_chunk](#sized_chunk "Split an array into chunks") //// //// #### Transform //// [reverse](#reverse "Reverse the order"), @@ -136,6 +138,7 @@ //// [fold_right](#fold_right "Loop end to start, with state"), //// [index_fold_right](#index_fold_right "Loop end to start, with index and state") +import gleam/int import gleam/list import gleam/yielder.{type Yielder} import iv/internal/builder @@ -1741,6 +1744,75 @@ pub fn take_last(from array: Array(item), up_to n: Int) -> Array(item) { drop_first(array, size(array) - n) } +/// Returns an array of chunks containing `count` elements each. +/// +/// If the last chunk does not have count elements, it is instead a partial +/// chunk, with less than count elements. +/// +/// For any count less than 1 this function behaves as if it was set to 1. +/// +///
+/// +/// Back to top ↑ +/// +///
+pub fn sized_chunk(in array: Array(item), into count: Int) -> Array(Array(item)) { + sized_chunk_loop(array, int.min(1, count), []) +} + +fn sized_chunk_loop( + array: Array(item), + count: Int, + chunks: List(Array(item)), +) -> Array(Array(item)) { + let size = size(array) + case size <= count { + True -> + case size { + 0 -> from_reverse_list(chunks) + _ -> from_reverse_list([array, ..chunks]) + } + False -> { + let #(chunk, rest) = split(array, count) + sized_chunk_loop(rest, count, [chunk, ..chunks]) + } + } +} + +/// Returns an array distributing its elements evenly into `n` chunks. +/// +/// If there are less than `n` elements in the array, less chunks may be +/// returned. +/// +/// For any count less than 1 this function behaves as if it was set to 1. +/// +///
+/// +/// Back to top ↑ +/// +///
+pub fn split_n(in array: Array(item), into n_chunks: Int) -> Array(Array(item)) { + split_n_loop(array, size(array) / n_chunks, size(array) % n_chunks, []) +} + +fn split_n_loop(array, count, rest, chunks) { + let size = size(array) + case size <= count { + True -> + case size { + 0 -> from_reverse_list(chunks) + _ -> from_reverse_list([array, ..chunks]) + } + False -> { + let #(chunk, array) = case rest > 0 { + False -> split(array, count) + True -> split(array, count + 1) + } + split_n_loop(array, count, rest - 1, [chunk, ..chunks]) + } + } +} + /// Return the array without the first element. If the array is empty, /// `Error(Nil)` is returned. ///