diff --git a/.changeset/git-guard-reads-the-index.md b/.changeset/git-guard-reads-the-index.md new file mode 100644 index 0000000..9eda507 --- /dev/null +++ b/.changeset/git-guard-reads-the-index.md @@ -0,0 +1,18 @@ +--- +'@pdsjs/git': patch +'@pdsjs/node': patch +'@pdsjs/cloudflare': patch +--- + +The protected-branch check reads the commits out of the chain instead of +downloading it. The guard answers one question, whether the old tip is an +ancestor of the new one, and every object in the repository was materialized +to answer it. On pds.js's own repository that costs 199 MB, which no +Cloudflare isolate has. + +The lazy object store moves to `store.js`, where the write guard reads it as +the repository browser does: the bundle indexes, then ranged reads for the +commits alone. Both platform adapters give the guard a `getBlobRange` +wherever the blob store reads ranges. The same check costs 43 MB and 33 ms, +against 199 MB and 428 ms. A chain with no index, or a store with no ranges, +reads whole as before. diff --git a/packages/cloudflare/src/index.js b/packages/cloudflare/src/index.js index 40ccdde..e6018b7 100644 --- a/packages/cloudflare/src/index.js +++ b/packages/cloudflare/src/index.js @@ -1187,6 +1187,20 @@ export class PDSDurableObject { ? result.data : new Uint8Array(/** @type {ArrayBuffer} */ (result.data)); }, + // A store that reads ranges lets the ancestry check read the commits + // out of the chain. getStream takes inclusive positions. + getBlobRange: blobs.getStream + ? async (cid, start, end) => { + const did = await actorStorage.getDid(); + if (!did) return null; + const found = await blobs.getStream?.(did, cid, { + start, + end: end - 1, + }); + if (!found) return null; + return new Uint8Array(await new Response(found.body).arrayBuffer()); + } + : undefined, }); if (env.PDS_ENABLE_SPACES === 'true') { diff --git a/packages/git/src/rules.js b/packages/git/src/rules.js index f13d6d0..25dbf25 100644 --- a/packages/git/src/rules.js +++ b/packages/git/src/rules.js @@ -18,7 +18,8 @@ import { parseBundle } from './bundle.js'; import { indexPacks, parseCommit, resolveToCommit } from './objects.js'; -import { concatChunks, parseRepoRecord } from './record.js'; +import { concatChunks, indexPartsOf, parseRepoRecord } from './record.js'; +import { createLazyStore } from './store.js'; export const GIT_CONFIG_COLLECTION = 'dev.pdsjs.git.config'; @@ -114,13 +115,64 @@ function refMap(value) { } } +/** + * The commit graph, read through the object index so the server holds the + * commits alone. A chain whose bundles all carry an index, on a store that + * reads ranges, costs the index blobs and the commits; every other chain + * downloads whole. + * @param {import('./record.js').GitRepoRecord} record + * @param {(cid: string) => Promise} getBlob + * @param {(cid: string, start: number, end: number) => Promise} getBlobRange + * @returns {Promise|null>} the + * commits and tags by id, or null if the chain has no index to read + */ +async function loadCommitsLazily(record, getBlob, getBlobRange) { + if (record.bundles.length === 0) return null; + if (record.bundles.some((bundle) => indexPartsOf(bundle).length === 0)) { + return null; + } + /** @param {string} cid @param {number} [start] @param {number} [end] */ + const read = async (cid, start, end) => { + const bytes = + start === undefined || end === undefined + ? await getBlob(cid) + : await getBlobRange(cid, start, end); + if (!bytes) { + throw new GitRuleError( + `bundle chunk ${cid} is not in blob storage`, + 'InvalidRecord', + ); + } + return bytes; + }; + const store = await createLazyStore( + record, + (cid) => read(cid), + (cid, start, end) => read(cid, start, end), + ); + const shas = await store.prefetchCommits(); + /** @type {Map} */ + const objects = new Map(); + for (const sha of shas) { + const object = store.peek(sha); + if (object) objects.set(sha, object); + } + return objects; +} + /** * @param {unknown} value - the new repo record, whose chain holds the graph * @param {(cid: string) => Promise} getBlob - * @returns {Promise>} + * @param {((cid: string, start: number, end: number) => Promise)} [getBlobRange] + * @returns {Promise>} the + * commits and tags the chain holds, by id */ -async function loadObjects(value, getBlob) { +async function loadObjects(value, getBlob, getBlobRange) { const record = parseRepoRecord(value); + if (getBlobRange) { + const lazy = await loadCommitsLazily(record, getBlob, getBlobRange); + if (lazy) return lazy; + } /** @type {Uint8Array[]} */ const packs = []; for (const bundle of record.bundles) { @@ -138,7 +190,9 @@ async function loadObjects(value, getBlob) { } packs.push(parseBundle(concatChunks(chunks)).pack); } - return await indexPacks(packs); + // The guard walks commit parents, so trees and blobs are needed only as + // delta bases while the pack is read. + return await indexPacks(packs, undefined, undefined, { retain: 'commits' }); } /** @@ -237,9 +291,13 @@ function assertValidConfig(value) { * @param {(space: string|null, collection: string, rkey: string) => Promise} ctx.getRecord - * a decoded record value from the public repo (space null) or a space * @param {(cid: string) => Promise} ctx.getBlob + * @param {(cid: string, start: number, end: number) => Promise} [ctx.getBlobRange] - + * Optional: one byte range of a blob, end exclusive. A store that reads + * ranges lets the ancestry check read commits out of the chain rather + * than downloading it whole. * @returns {{collections: string[], check: (write: GuardedWrite) => Promise}} */ -export function createGitWriteGuard({ getRecord, getBlob }) { +export function createGitWriteGuard({ getRecord, getBlob, getBlobRange }) { return { collections: ['dev.pdsjs.git.repo', GIT_CONFIG_COLLECTION], @@ -283,7 +341,7 @@ export function createGitWriteGuard({ getRecord, getBlob }) { ); } if (newSha === oldSha) continue; - objects ??= await loadObjects(nextValue, getBlob); + objects ??= await loadObjects(nextValue, getBlob, getBlobRange); if (!isAncestorOf(objects, oldSha, newSha)) { throw new GitRuleError( `${branch} is protected; only fast-forward pushes are accepted`, diff --git a/packages/git/test/rules.test.js b/packages/git/test/rules.test.js index 594b233..dd0d343 100644 --- a/packages/git/test/rules.test.js +++ b/packages/git/test/rules.test.js @@ -1,7 +1,16 @@ -// The write guard's rule logic against synthetic records. Fast-forward -// ancestry over real packs is exercised end to end in space-remote.test.js; -// everything decidable from the records alone is here. -import { describe, expect, it } from 'vitest'; +// The write guard's rule logic. Everything decidable from the records alone +// runs against synthetic records; the ancestry check runs against a real +// pack, both through the object index and through a whole-chain read. The +// push path around it is exercised end to end in space-remote.test.js. +import { execFile } from 'node:child_process'; +import { randomBytes } from 'node:crypto'; +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { promisify } from 'node:util'; +import { afterAll, beforeAll, describe, expect, it } from 'vitest'; +import { buildBundleIndex, encodePackIndex } from '../src/pack-index.js'; +import { splitIntoChunks } from '../src/record.js'; import { createGitWriteGuard, GIT_CONFIG_COLLECTION, @@ -209,3 +218,141 @@ describe('the write guard on a config record', () => { ).rejects.toThrow(/{registry, name}/); }); }); + +describe('ancestry over the object index', () => { + /** @type {string} */ + let dir = ''; + /** @type {string[]} */ + const commits = []; + /** @type {Record} */ + let nextValue = {}; + /** @type {Map} */ + const blobs = new Map(); + /** How many bytes each read handed back, by kind. */ + let read = { whole: 0, ranged: 0 }; + let bundleSize = 0; + + const env = { + .../** @type {Record} */ (process.env), + GIT_AUTHOR_NAME: 'Test', + GIT_AUTHOR_EMAIL: 'test@example.com', + GIT_COMMITTER_NAME: 'Test', + GIT_COMMITTER_EMAIL: 'test@example.com', + }; + /** @param {string} cwd @param {string[]} args */ + const runGit = async (cwd, args) => + (await promisify(execFile)('git', args, { cwd, env })).stdout.trim(); + + beforeAll(async () => { + dir = mkdtempSync(join(tmpdir(), 'pdsjs-git-rules-')); + const repo = join(dir, 'repo'); + await runGit(dir, ['init', '-b', 'main', repo]); + // A file body far larger than the commits, so a walk that reads the + // whole chain is plain to see in the byte counts. + writeFileSync(join(repo, 'big.bin'), randomBytes(512 * 1024)); + for (const message of ['one', 'two', 'three']) { + writeFileSync(join(repo, 'log.txt'), `${message}\n`); + await runGit(repo, ['add', '.']); + await runGit(repo, ['commit', '-m', message]); + commits.push(await runGit(repo, ['rev-parse', 'HEAD'])); + } + await runGit(repo, ['bundle', 'create', join(dir, 'b.bundle'), 'main']); + const bytes = new Uint8Array(readFileSync(join(dir, 'b.bundle'))); + bundleSize = bytes.length; + + /** @param {Uint8Array} data @param {string} cid */ + const put = (data, cid) => { + blobs.set(cid, data); + return { + $type: 'blob', + ref: { $link: cid }, + mimeType: 'application/octet-stream', + size: data.length, + }; + }; + const parts = splitIntoChunks(bytes, 64 * 1024).map((chunk, i) => + put(chunk, `bafkreipart${i}`), + ); + const encoded = encodePackIndex(await buildBundleIndex(bytes)); + const indexParts = splitIntoChunks(encoded, 4096).map((chunk, i) => + put(chunk, `bafkreiindex${i}`), + ); + nextValue = { + $type: 'dev.pdsjs.git.repo', + name: 'repo', + refs: [{ name: 'refs/heads/main', sha: commits[2] }], + bundles: [{ heads: [commits[2]], parts, indexParts }], + createdAt: '2026-08-24T00:00:00.000Z', + }; + }, 60_000); + + afterAll(() => rmSync(dir, { recursive: true, force: true })); + + /** @param {{ranges: boolean}} opts */ + function guard(opts) { + read = { whole: 0, ranged: 0 }; + return createGitWriteGuard({ + getRecord: async (_space, collection) => + collection === GIT_CONFIG_COLLECTION + ? { protectedBranches: ['main'] } + : null, + getBlob: async (cid) => { + const blob = blobs.get(cid) ?? null; + if (blob) read.whole += blob.length; + return blob; + }, + getBlobRange: opts.ranges + ? async (cid, start, end) => { + const blob = blobs.get(cid); + if (!blob) return null; + const slice = blob.subarray(start, end); + read.ranged += slice.length; + return slice; + } + : undefined, + }); + } + + /** @param {string} prevSha @param {{ranges: boolean}} opts */ + const check = (prevSha, opts) => + guard(opts).check({ + space: null, + collection: 'dev.pdsjs.git.repo', + rkey: 'repo', + prevValue: { + $type: 'dev.pdsjs.git.repo', + name: 'repo', + refs: [{ name: 'refs/heads/main', sha: prevSha }], + bundles: [], + createdAt: '2026-08-24T00:00:00.000Z', + }, + nextValue, + }); + + it('takes a fast-forward without reading the file bodies', async () => { + await expect(check(commits[0], { ranges: true })).resolves.toBe(undefined); + // The index blobs are read whole; the pack is read by range, and the + // commits are a small part of it. + expect(read.ranged).toBeGreaterThan(0); + expect(read.ranged).toBeLessThan(bundleSize / 8); + expect(read.whole + read.ranged).toBeLessThan(bundleSize / 2); + }); + + it('refuses a tip the protected branch does not reach', async () => { + const orphan = 'f'.repeat(40); + await expect(check(orphan, { ranges: true })).rejects.toThrow( + /only fast-forward/, + ); + }); + + it('agrees with the whole-chain read when no store reads ranges', async () => { + await expect(check(commits[0], { ranges: false })).resolves.toBe(undefined); + // Without ranges every chunk is downloaded, which is the cost the index + // path avoids. + expect(read.whole).toBeGreaterThanOrEqual(bundleSize); + const orphan = 'f'.repeat(40); + await expect(check(orphan, { ranges: false })).rejects.toThrow( + /only fast-forward/, + ); + }); +}); diff --git a/packages/node/src/index.js b/packages/node/src/index.js index 9f478bd..a91237c 100644 --- a/packages/node/src/index.js +++ b/packages/node/src/index.js @@ -445,6 +445,20 @@ export async function createServer({ ? result.data : new Uint8Array(/** @type {ArrayBuffer} */ (result.data)); }, + // A store that reads ranges lets the ancestry check read the commits + // out of the chain. getStream takes inclusive positions. + getBlobRange: blobs.getStream + ? async (cid, start, end) => { + const did = await actorStorage.getDid(); + if (!did) return null; + const found = await blobs.getStream?.(did, cid, { + start, + end: end - 1, + }); + if (!found) return null; + return new Uint8Array(await new Response(found.body).arrayBuffer()); + } + : undefined, }); } catch { // Optional peer not installed