From 9f5c004652e0ca578a88727111d0e110977bf4de Mon Sep 17 00:00:00 2001 From: dawn <90008@gaze.systems> Date: Mon, 20 Apr 2026 04:06:06 +0300 Subject: [PATCH] [docs] initial documentation + add verbiage to devshell MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Splits the README into structured wiki-ready markdown under docs/: - docs/README.md — overview and index - docs/getting-started.md — building, running, reverse proxying - docs/configuration.md — all env vars - docs/build-features.md — cargo feature flags - docs/concepts/{vs-tap,relay}.md — stream behavior and relay/seeding concepts - docs/api/{filter,ingestion,crawler,firehose,pds,repos,database}.md — REST API - docs/xrpc/{atproto,hydrant,backlinks}.md — XRPC reference flake.nix: adds github:90-008/verbiage as an input and includes the verbiage package in the devshell so `verbiage docs/` can be used to develop docs locally. Co-Authored-By: Claude Sonnet 4.6 --- docs/README.md | 24 +++++++++++++ docs/api/README.md | 19 +++++++++++ docs/api/crawler.md | 17 ++++++++++ docs/api/database.md | 4 +++ docs/api/filter.md | 36 ++++++++++++++++++++ docs/api/firehose.md | 18 ++++++++++ docs/api/ingestion.md | 7 ++++ docs/api/pds.md | 35 +++++++++++++++++++ docs/api/repos.md | 11 ++++++ docs/build-features.md | 16 +++++++++ docs/concepts/README.md | 4 +++ docs/concepts/relay.md | 52 +++++++++++++++++++++++++++++ docs/concepts/vs-tap.md | 17 ++++++++++ docs/configuration.md | 74 +++++++++++++++++++++++++++++++++++++++++ docs/getting-started.md | 55 ++++++++++++++++++++++++++++++ docs/xrpc/README.md | 7 ++++ docs/xrpc/atproto.md | 16 +++++++++ docs/xrpc/backlinks.md | 32 ++++++++++++++++++ docs/xrpc/hydrant.md | 24 +++++++++++++ flake.lock | 37 ++++++++++++++++++++- flake.nix | 2 ++ 21 files changed, 506 insertions(+), 1 deletion(-) create mode 100644 docs/README.md create mode 100644 docs/api/README.md create mode 100644 docs/api/crawler.md create mode 100644 docs/api/database.md create mode 100644 docs/api/filter.md create mode 100644 docs/api/firehose.md create mode 100644 docs/api/ingestion.md create mode 100644 docs/api/pds.md create mode 100644 docs/api/repos.md create mode 100644 docs/build-features.md create mode 100644 docs/concepts/README.md create mode 100644 docs/concepts/relay.md create mode 100644 docs/concepts/vs-tap.md create mode 100644 docs/configuration.md create mode 100644 docs/getting-started.md create mode 100644 docs/xrpc/README.md create mode 100644 docs/xrpc/atproto.md create mode 100644 docs/xrpc/backlinks.md create mode 100644 docs/xrpc/hydrant.md diff --git a/docs/README.md b/docs/README.md new file mode 100644 index 0000000..b028d4d --- /dev/null +++ b/docs/README.md @@ -0,0 +1,24 @@ +# hydrant + +`hydrant` is an AT Protocol indexer built on the `fjall` database. it's built to be flexible, supporting both full-network indexing and filtered indexing (e.g., by DID), allowing querying with XRPCs (not only `com.atproto.*`!), providing an ordered event stream, etc. oh and it can also act as a relay! + +you can see [random.wisp.place](https://tangled.org/did:plc:dfl62fgb7wtjj3fcbb72naae/random.wisp.place) (standalone binary using http API) or the [statusphere example](../examples/statusphere.rs) (hydrant-as-library) for examples. for rust docs look at https://hydrant.klbr.net/ for now. + +**WARNING: *the db format is only partially stable.*** we provide migrations in hydrant itself, so nothing should go wrong! you should still probably keep backups just in case! + +## what's here + +- [getting started](getting-started.md): building, running, reverse proxying +- [configuration](configuration.md): all environment variables +- [build features](build-features.md): optional cargo features (`relay`, `backlinks`, etc.) +- [concepts](concepts/README.md): how the stream works, relay comparison, multi-relay support +- [rest api](api/README.md): management API reference +- [xrpc](xrpc/README.md): data access via XRPC + +## quick start + +```bash +cargo build --release +export HYDRANT_DATABASE_PATH=./hydrant.db +./target/release/hydrant +``` diff --git a/docs/api/README.md b/docs/api/README.md new file mode 100644 index 0000000..88def89 --- /dev/null +++ b/docs/api/README.md @@ -0,0 +1,19 @@ +# rest api + +hydrant's REST API is split into public endpoints (safe to expose) and management endpoints (keep private). see [getting started](../getting-started.md#reverse-proxying) for guidance on what to expose. + +## public + +- `GET /stream`: subscribe to the event stream. query params: `cursor` (optional, start from a specific event ID). +- `GET /stats`: get stats about the database (counts of repos, records, events; sizes of keyspaces on disk). +- `GET /health` / `GET /_health`: health check. + +## management + +- [filter](filter.md): NSID filter configuration +- [ingestion](ingestion.md): enable/disable crawler, firehose, backfill at runtime +- [crawler](crawler.md): crawler source management +- [firehose](firehose.md): firehose source management +- [pds](pds.md): rate-limit tier assignments +- [repos](repos.md): explicit repository tracking, resyncing, untracking +- [database](database.md): compression training, compaction diff --git a/docs/api/crawler.md b/docs/api/crawler.md new file mode 100644 index 0000000..b83fe90 --- /dev/null +++ b/docs/api/crawler.md @@ -0,0 +1,17 @@ +# crawler management + +- `GET /crawler/sources`: list all currently active crawler sources. + - returns a JSON array of `{ "url": string, "mode": "relay" | "by_collection", "persisted": bool }`. + - `persisted: true` means the source was added via the API and is stored in the database, it will survive a restart. `persisted: false` means the source came from `CRAWLER_URLS` and is not written to the database. +- `POST /crawler/sources`: add a crawler source at runtime. + - body: `{ "url": string, "mode": "relay" | "by_collection" }`. + - the source is written to the database before the producer task is started, so it is safe to add sources and then immediately restart without losing them. + - if a source with the same URL already exists (whether from `CRAWLER_URLS` or a previous `POST`), it is replaced: the running task is stopped and a new one is started with the new mode. any cursor state for that URL is preserved. + - returns `201 Created` on success. +- `DELETE /crawler/sources`: remove a crawler source at runtime. + - body: `{ "url": string }`. + - the producer task is stopped immediately. + - if the source was added via the API (`persisted: true`), it is removed from the database and will not reappear on restart. if it came from `CRAWLER_URLS` (`persisted: false`), only the running task is stopped, the source will reappear on the next restart since `CRAWLER_URLS` is re-applied at startup. + - cursor state is not cleared. use `DELETE /crawler/cursors` separately if you want the source to restart from the beginning when re-added. + - returns `200 OK` if the source was found and removed, `404 Not Found` otherwise. +- `DELETE /crawler/cursors`: reset stored cursors for a given crawler URL. body: `{ "key": "..." }` where key is a URL. clears the list-repos crawler cursor as well as any by-collection cursors associated with that URL. causes the next crawler pass to restart from the beginning. diff --git a/docs/api/database.md b/docs/api/database.md new file mode 100644 index 0000000..7f01d78 --- /dev/null +++ b/docs/api/database.md @@ -0,0 +1,4 @@ +# database operations + +- `POST /db/train`: train zstd compression dictionaries for the `repos`, `blocks`, and `events` keyspaces. dictionaries are written to disk; a restart is required to apply them. the crawler, firehose, and backfill worker are paused for the duration and restored on completion. +- `POST /db/compact`: trigger a full major compaction of all database keyspaces in parallel. the crawler, firehose, and backfill worker are paused for the duration and restored on completion. diff --git a/docs/api/filter.md b/docs/api/filter.md new file mode 100644 index 0000000..1408fdd --- /dev/null +++ b/docs/api/filter.md @@ -0,0 +1,36 @@ +# filter management + +- `GET /filter`: get the current filter configuration. +- `PATCH /filter`: update the filter configuration. + +## filter mode + +the `mode` field controls what gets indexed: + +| mode | behaviour | +| :--- | :--- | +| `filter` | auto-discovers and backfills any account whose firehose commit touches a collection matching one of the `signals` patterns. you can also explicitly track individual repositories via the `/repos` endpoint regardless of matching signals. | +| `full` | index the entire network. `signals` are ignored for discovery, but `excludes` and `collections` still apply. | + +## fields + +| field | type | description | +| :--- | :--- | :--- | +| `mode` | `"filter"` \| `"full"` | indexing mode (see above). | +| `signals` | set update | NSID patterns (e.g. `app.bsky.feed.post` or `app.bsky.*`) that trigger auto-discovery in `filter` mode. | +| `collections` | set update | NSID patterns used to filter which records are stored. if empty, all collections are stored. applies in all modes. | +| `excludes` | set update | set of DIDs to always skip, regardless of mode. checked before any other filter logic. | + +## set updates + +each set field accepts one of two forms: + +- **replace**: an array replaces the entire set, eg. `["app.bsky.feed.post", "app.bsky.graph.*"]` +- **patch**: an object maps items to `true` (add) or `false` (remove), eg. `{"app.bsky.feed.post": true, "app.bsky.graph.*": false}` + +## NSID patterns + +`signals` and `collections` support an optional `.*` suffix to match an entire namespace: + +- `app.bsky.feed.post`: exact match only +- `app.bsky.feed.*`: matches any collection under `app.bsky.feed` diff --git a/docs/api/firehose.md b/docs/api/firehose.md new file mode 100644 index 0000000..31dcbdc --- /dev/null +++ b/docs/api/firehose.md @@ -0,0 +1,18 @@ +# firehose management + +- `GET /firehose/sources`: list all currently active firehose sources. + - returns a JSON array of `{ "url": string, "persisted": bool, "is_pds": bool }`. + - `persisted: true` means the source was added via the API and is stored in the database, it will survive a restart. `persisted: false` means the source came from `RELAY_HOSTS` and is not written to the database. + - `is_pds: true` means the source is a direct PDS connection with host authority enforcement enabled. +- `POST /firehose/sources`: add a firehose source at runtime. + - body: `{ "url": string, "is_pds": bool }`. `is_pds` defaults to `false`. + - the source is persisted to the database before the ingestor task is started. + - if a source with the same URL already exists, it is replaced: the running task is stopped and a new one is started. any existing cursor state for that URL is preserved. + - returns `201 Created` on success. +- `DELETE /firehose/sources`: remove a firehose relay at runtime. + - body: `{ "url": string }`. + - the ingestor task is stopped immediately. + - if the source was added via the API (`persisted: true`), it is removed from the database and will not reappear on restart. if it came from `RELAY_HOSTS` (`persisted: false`), only the running task is stopped; the source reappears on the next restart. + - cursor state is not cleared. use `DELETE /firehose/cursors` separately if you want the relay to restart from the beginning when re-added. + - returns `200 OK` if the relay was found and removed, `404 Not Found` otherwise. +- `DELETE /firehose/cursors`: reset the stored cursor for a given firehose relay URL. body: `{ "key": "..." }` where key is a URL. causes the next firehose connection to restart from the beginning. diff --git a/docs/api/ingestion.md b/docs/api/ingestion.md new file mode 100644 index 0000000..f6037c5 --- /dev/null +++ b/docs/api/ingestion.md @@ -0,0 +1,7 @@ +# ingestion control + +- `GET /ingestion`: get the current ingestion status. + - returns `{ "crawler": bool, "firehose": bool, "backfill": bool }`. +- `PATCH /ingestion`: enable or disable ingestion components at runtime without restarting. + - body: `{ "crawler"?: bool, "firehose"?: bool, "backfill"?: bool }`. only provided fields are updated. + - when disabled, each component finishes its current task before pausing (e.g. the backfill worker completes any in-flight repo syncs, the firehose finishes processing the current message). they resume immediately when re-enabled. diff --git a/docs/api/pds.md b/docs/api/pds.md new file mode 100644 index 0000000..951ddc9 --- /dev/null +++ b/docs/api/pds.md @@ -0,0 +1,35 @@ +# PDS management + +hydrant rate-limits firehose events per PDS. each PDS is assigned to a named rate tier that controls how aggressively hydrant limits events from it. two built-in tiers are always present: `default` (conservative limits for unknown operators) and `trusted` (higher limits for well-behaved operators). additional tiers can be defined via `RATE_TIERS`. + +the per-second limit scales with the number of active accounts on the PDS: `max(per_second_base, accounts × per_second_account_mul)`. + +you can also define an optional `account_limit` for a rate tier. if a PDS exceeds this number of active accounts, hydrant will reject any new account creation events from it. + +the built-in tiers are defined as follows: +- `default`: `50` per sec (floor), `+0.5` per account. max `3_600_000`/hr, `86_400_000`/day. `100` account limit. +- `trusted`: `5000` per sec (floor), `+10.0` per account. max `18_000_000`/hr, `432_000_000`/day. `10_000_000` account limit. + +tiers are resolved in this order: + +1. **explicit API assignment**, set via `PUT /pds/tiers`, stored in the database, survives restarts. +2. **glob rules**, from `TIER_RULES`, evaluated in order; first match wins. +3. **`default` tier**, applied if no rule or explicit assignment matches. + +deleting an API assignment reverts the host to glob-rule resolution, not necessarily back to `default`. if a rule like `*.bsky.network:trusted` matches the host, it will become trusted again without any further action. + +- `GET /pds/tiers`: list all current tier assignments alongside the available tier definitions. + - returns `{ "assignments": [{ "host": string, "tier": string }], "rate_tiers": { : { "per_second_base": int, "per_second_account_mul": float, "per_hour": int, "per_day": int } } }`. + - `assignments` only contains PDSes with an explicit API assignment. hosts without one resolve via glob rules or fall back to `default`. +- `PUT /pds/tiers`: assign a PDS to a named rate tier. + - body: `{ "host": string, "tier": string }`. + - `host` is the PDS hostname (e.g. `pds.example.com`). + - `tier` must be one of the configured tier names. returns `400` if unknown. + - assignments are persisted to the database and survive restarts. + - re-assigning the same host updates the tier in place without creating a duplicate. +- `DELETE /pds/tiers`: remove an explicit tier assignment for a PDS. + - query parameter: `?host=` (e.g. `?host=pds.example.com`). + - reverts the host to glob-rule resolution (not necessarily `default`, a matching `TIER_RULES` pattern still applies). + - returns `200` even if no assignment existed. +- `GET /pds/rate-tiers`: list the available rate tier definitions. + - returns a map of tier name to `{ "per_second_base", "per_second_account_mul", "per_hour", "per_day", "account_limit" }`. diff --git a/docs/api/repos.md b/docs/api/repos.md new file mode 100644 index 0000000..24b0777 --- /dev/null +++ b/docs/api/repos.md @@ -0,0 +1,11 @@ +# repository management + +all `/repos` endpoints that return lists respond with NDJSON by default. send `Accept: application/json` or `Content-Type: application/json` to get a JSON array instead. + +- `GET /repos`: get a list of repositories and their sync status. supports pagination and filtering: + - `limit`: max results (default 100, max 1000) + - `cursor`: did key for paginating. +- `GET /repos/{did}`: get the sync status and metadata of a specific repository. also returns the handle, PDS URL and the atproto signing key (these won't be available before the repo has been backfilled once at least). +- `PUT /repos`: explicitly track repositories. accepts an NDJSON body of `{"did": "..."}` (or JSON array of the same). only affects repositories that are not known or are untracked. returns a list of the DIDs that were queued for backfill. +- `DELETE /repos`: untrack repositories. accepts an NDJSON body of `{"did": "..."}` (or JSON array of the same). only affects repositories that are currently tracked. returns a list of the DIDs that were untracked. +- `POST /repos/resync`: force a new backfill for one or more repositories. accepts an NDJSON body of `{"did": "..."}` (or JSON array of the same). only affects repositories hydrant already knows about. returns a list of the DIDs that were queued. diff --git a/docs/build-features.md b/docs/build-features.md new file mode 100644 index 0000000..d152436 --- /dev/null +++ b/docs/build-features.md @@ -0,0 +1,16 @@ +# build features + +`hydrant` has several optional compile-time features: + +| feature | default | description | +| :--- | :--- | :--- | +| `indexer` | yes | makes hydrant act as an indexer. incompatible with the relay feature. | +| `indexer_stream` | yes | enables the event stream for the indexer. requires indexer feature. | +| `relay` | no | makes hydrant act as a relay. incompatible with the indexer feature. | +| `backlinks` | no | enables the backlinks indexer and XRPC endpoints (`blue.microcosm.links.*`). requires indexer feature. | + +to build with a specific feature: + +```bash +cargo build --release --features backlinks +``` diff --git a/docs/concepts/README.md b/docs/concepts/README.md new file mode 100644 index 0000000..f29084a --- /dev/null +++ b/docs/concepts/README.md @@ -0,0 +1,4 @@ +# concepts + +- [hydrant vs tap](vs-tap.md): design comparison, stream behavior +- [relay & seeding](relay.md): multi-relay support, firehose seeding, crawler sources diff --git a/docs/concepts/relay.md b/docs/concepts/relay.md new file mode 100644 index 0000000..2cc15e8 --- /dev/null +++ b/docs/concepts/relay.md @@ -0,0 +1,52 @@ +# relay, seeding & crawler sources + +## multiple relay support + +`hydrant` supports connecting to multiple relays simultaneously for firehose ingestion. when `RELAY_HOSTS` is configured with multiple URLs: + +- one independent firehose stream loop is spawned per relay +- each relay maintains its own firehose cursor state +- all ingestion loops share the same worker pool and database + +commit events are de-duplicated according to the repo `rev`. account / identity events are de-duplicated using the `time` field. + +## direct PDS connections + +a firehose source can also be a direct connection to a PDS rather than a relay. prefix the URL with `pds::` to mark it as such: + +``` +HYDRANT_RELAY_HOSTS=wss://bsky.network,pds::wss://pds.example.com +``` + +only when a source is marked as a direct PDS (`is_pds: true`), hydrant enforces host authority. relays (`is_pds: false`, the default) are exempt from this check, since they forward commits from many PDSes by design. + +## firehose seeding + +in relay mode, `RELAY_HOSTS` defaults to empty. set `SEED_HOSTS` to one or more relay base URLs and hydrant will call `com.atproto.sync.listHosts` on each at startup, adding every returned PDS as a firehose source: + +``` +HYDRANT_SEED_HOSTS=https://bsky.network +``` + +seeding runs as a background task so the main firehose loop is not blocked. seed URLs are fetched concurrently (up to four at a time) and the full `listHosts` pagination is consumed for each. if a request fails partway through, the hosts collected so far are still added and the failure is logged. + +each discovered host is added as a persistent PDS firehose source (`is_pds: true`), equivalent to calling `POST /firehose/sources`. + +banned hosts (`status: "banned"`) are skipped. all other statuses are included since the firehose ingestor retries on disconnect and transiently-unavailable hosts will reconnect on their own. + +seeding runs from latest cursor on restart so new PDS' added to the upstream relay since the last start are picked up automatically (if they haven't through firehose). sources that are already running are detected and skipped, so re-seeding is idempotent. + +## crawler sources + +the crawler is configured separately from the firehose via `CRAWLER_URLS`. each source is a `[mode::]url` entry where the mode prefix is optional and defaults to `by_collection` in filter mode or `list_repos` in full-network mode. + +- `list_repos`: enumerates the network via `com.atproto.sync.listRepos`, checks each repo's collections via `describeRepo`. +- `by_collection`: queries `com.atproto.sync.listReposByCollection` for each configured signal. more efficient for filtered indexing since it only surfaces repos that have matching records. cursors are stored per collection. note that it won't crawl anything if no signals are specified. + +``` +CRAWLER_URLS=by_collection::https://lightrail.microcosm.blue,list_repos::wss://bsky.network +``` + +each source maintains its own cursor so restarts resume mid-pass. + +sources can also be added and removed at runtime via the `/crawler/sources` API (see [crawler management](../api/crawler.md)). dynamically added sources are persisted to the database and survive restarts. `CRAWLER_URLS` sources are startup-only: they are not written to the database and will always reappear after a restart regardless of runtime changes (unless you change the config of course). diff --git a/docs/concepts/vs-tap.md b/docs/concepts/vs-tap.md new file mode 100644 index 0000000..445d970 --- /dev/null +++ b/docs/concepts/vs-tap.md @@ -0,0 +1,17 @@ +# hydrant vs tap + +while [`tap`](https://github.com/bluesky-social/indigo/tree/main/cmd/tap) is designed as a firehose consumer and simply just propagates events while handling sync, `hydrant` is flexible, it allows you to directly query the database for records, and it also provides an ordered view of events, allowing the use of a cursor to fetch events from a specific point. it can act as both an indexer or an ephemeral view of some window of events. + +you can also read [this blogpost](https://90008.leaflet.pub/3mhp3t4kuw22e) for a longer comparison. + +## stream behavior + +the `WS /stream` (hydrant) and `WS /channel` (tap) endpoints have different designs: + +| aspect | `tap` (`/channel`) | `hydrant` (`/stream`) | +| :--- | :--- | :--- | +| distribution | sharded work queue: events are load-balanced across connected clients. if 5 clients connect, each receives ~20% of events. | broadcast: every connected client receives a full copy of the event stream. if 5 clients connect, all 5 receive 100% of events. | +| cursors | server-managed: clients ACK messages. the server tracks progress and redelivers unacked messages. | client-managed: client provides `?cursor=123`. the server streams from that point. | +| persistence | events are stored in an outbox and sent to the consumer, and removed from the outbox when acked. nothing is replayable. | `record` events are replayable. `identity`/`account` are ephemeral. use `GET /repos/:did` to query identity / account info (handle, pds, signing key, etc.). | +| backfill | backfill events are mixed into the live queue and prioritized (per-repo, acting as synchronization barrier) by the server. | backfill simply inserts historical events (`live: false`) into the global event log. streaming is just reading this log sequentially. synchronization is the same as tap, `live: true` vs `live: false`. | +| event types | `record`, `identity` (includes status) | `record`, `identity` (handle, cache-buster), `account` (status) | diff --git a/docs/configuration.md b/docs/configuration.md new file mode 100644 index 0000000..669f035 --- /dev/null +++ b/docs/configuration.md @@ -0,0 +1,74 @@ +# configuration + +hydrant is configured via environment variables, all prefixed with `HYDRANT_` (except `RUST_LOG`). a `.env` file in the working directory is loaded automatically. + +## core + +| variable | default | description | +| :--- | :--- | :--- | +| `DATABASE_PATH` | `./hydrant.db` | path to the database folder | +| `RUST_LOG` | `info` | log filter ([tracing env-filter syntax](https://docs.rs/tracing-subscriber/latest/tracing_subscriber/filter/struct.EnvFilter.html)) | +| `API_PORT` | `3000` | port for the API server | +| `ENABLE_DEBUG` | `false` | enable debug endpoints | +| `DEBUG_PORT` | `API_PORT + 1` | port for debug endpoints | + +## indexing mode + +| variable | default | description | +| :--- | :--- | :--- | +| `FULL_NETWORK` | `false` (indexer), `true` (relay) | if `true`, discover and index all repos in the network | +| `EPHEMERAL` | `false` (indexer), `true` (relay) | if enabled, no records are stored; events are deleted after `EPHEMERAL_TTL` | +| `EPHEMERAL_TTL` | `60min`, `3d` (relay) | how long to keep events before deletion | +| `ONLY_INDEX_LINKS` | `false` | don't store record blocks, only the index. `getRecord`, `listRecords`, and `getRepo` will fail; the event stream still works but create/update events won't include record values | + +## filter + +| variable | default | description | +| :--- | :--- | :--- | +| `FILTER_SIGNALS` | | comma-separated NSID patterns triggering auto-discovery in filter mode (e.g. `app.bsky.feed.post,app.bsky.graph.*`) | +| `FILTER_COLLECTIONS` | | comma-separated NSID patterns limiting which records are stored. empty = store all | +| `FILTER_EXCLUDES` | | comma-separated DIDs to always skip | + +## firehose + +| variable | default | description | +| :--- | :--- | :--- | +| `RELAY_HOST` | `wss://relay.fire.hose.cam/` (indexer), empty (relay) | single firehose source URL | +| `RELAY_HOSTS` | | comma-separated firehose sources. prefix with `pds::` for direct PDS connections. overrides `RELAY_HOST` | +| `SEED_HOSTS` | `https://bsky.network` (relay) | relay URLs to call `com.atproto.sync.listHosts` on at startup, adding every non-banned PDS as a firehose source | +| `ENABLE_FIREHOSE` | `true` | whether to ingest relay subscriptions | +| `FIREHOSE_WORKERS` | `8` (`24` full network) | concurrent workers for firehose events | +| `CURSOR_SAVE_INTERVAL` | `3sec` | how often to persist the firehose cursor | + +## crawler + +| variable | default | description | +| :--- | :--- | :--- | +| `CRAWLER_URLS` | relay hosts (full network), `https://lightrail.microcosm.blue` (filter) | comma-separated `[mode::]url` crawler sources | +| `ENABLE_CRAWLER` | `true` if full network or sources configured | whether to actively query the network | +| `CRAWLER_MAX_PENDING_REPOS` | `2000` | max pending repos before the crawler pauses | +| `CRAWLER_RESUME_PENDING_REPOS` | `1000` | pending-repo count at which the crawler resumes | + +## backfill & identity + +| variable | default | description | +| :--- | :--- | :--- | +| `BACKFILL_CONCURRENCY_LIMIT` | `16` (`64` full network) | max concurrent backfill tasks | +| `REPO_FETCH_TIMEOUT` | `5min` | timeout for fetching a repository | +| `VERIFY_SIGNATURES` | `full` | signature verification: `full`, `backfill-only`, or `none` | +| `PLC_URL` | `https://plc.wtf`, `https://plc.directory` (full network) | PLC directory base URL(s), comma-separated | +| `IDENTITY_CACHE_SIZE` | `100000` | number of identity entries to cache in memory | + +## performance + +| variable | default | description | +| :--- | :--- | :--- | +| `CACHE_SIZE` | `256` | database cache size in MB | + +## rate limiting (relay mode) + +| variable | default | description | +| :--- | :--- | :--- | +| `NEW_HOST_LIMIT` | `50` | max new hosts addable via `com.atproto.sync.requestCrawl` per day | +| `RATE_TIERS` | | comma-separated tier definitions in `name:base/mul/hourly/daily[/account_limit]` format | +| `TIER_RULES` | | comma-separated ordered glob rules in `pattern:tier_name` format; first match wins | diff --git a/docs/getting-started.md b/docs/getting-started.md new file mode 100644 index 0000000..54ffd69 --- /dev/null +++ b/docs/getting-started.md @@ -0,0 +1,55 @@ +# getting started + +## requirements + +hydrant is written in rust and requires the rust toolchain (including `cargo`), `make`, `cmake` for some dependencies. you will also need the clang toolchain and the [wild linker](https://github.com/wild-linker/wild). + +## building from source + +```bash +cargo build --release +``` + +the binary will be at `target/release/hydrant`. + +to build with optional features (e.g. `backlinks`): + +```bash +cargo build --release --features backlinks +``` + +see [build features](build-features.md) for the full list. + +## running + +set the required environment variables and run the binary: + +```bash +export HYDRANT_DATABASE_PATH=./hydrant.db +./target/release/hydrant +``` + +see [configuration](configuration.md) for all available variables. if a `.env` file exists in the working directory it will be loaded automatically. + +## reverse proxying + +it is **highly recommended** to run hydrant behind a reverse proxy (like nginx or caddy) if you intend to expose the XRPC or event stream APIs to the public. hydrant's API includes several management endpoints that do not require or support authentication. **you MUST NOT expose these management endpoints to the public internet.** + +### public endpoints (safe to proxy) + +- `/xrpc/*`: XRPC endpoints. +- `/stream`: hydrant's ordered event stream. +- `/stats`: general database statistics. +- `/health` / `/_health`: health check. + +### management endpoints (keep private) + +- `/repos`: explicit repository tracking/resyncing/untracking. +- `/filter`: management of NSID filter patterns. +- `/ingestion`: manual control over component lifecycle (crawler, firehose, etc.). +- `/crawler/sources`: management of crawler relays. +- `/firehose/sources`: management of firehose relays. +- `/pds/tiers`: rate-limit tier assignments. +- `/db/train` / `/db/compact`: database maintenance tasks. +- `*/cursors`: cursor management. +- `/debug/*`: introspection and testing endpoints. diff --git a/docs/xrpc/README.md b/docs/xrpc/README.md new file mode 100644 index 0000000..1860b6b --- /dev/null +++ b/docs/xrpc/README.md @@ -0,0 +1,7 @@ +# xrpc + +`hydrant` implements the following XRPC endpoints under `/xrpc/`. only expose `/xrpc/*` publicly, see [getting started](../getting-started.md#reverse-proxying) for guidance. + +- [com.atproto.*](atproto.md): standard AT Protocol endpoints +- [systems.gaze.hydrant.*](hydrant.md): hydrant-specific extensions +- [blue.microcosm.links.*](backlinks.md): backlinks (requires `--features backlinks`) diff --git a/docs/xrpc/atproto.md b/docs/xrpc/atproto.md new file mode 100644 index 0000000..b7822b5 --- /dev/null +++ b/docs/xrpc/atproto.md @@ -0,0 +1,16 @@ +# com.atproto.* + +these are standard atproto endpoints. you can look at [the atproto api reference](https://docs.bsky.app/docs/category/http-reference) for more info. + +the following are implemented currently: +- `com.atproto.repo.getRecord` +- `com.atproto.repo.listRecords` +- `com.atproto.repo.describeRepo` (also see `systems.gaze.hydrant.describeRepo`) +- `com.atproto.sync.getRepo` (`since` parameter not implemented!) +- `com.atproto.sync.getHostStatus` +- `com.atproto.sync.listHosts` +- `com.atproto.sync.getRepoStatus` +- `com.atproto.sync.listRepos` +- `com.atproto.sync.getLatestCommit` +- `com.atproto.sync.requestCrawl` (adds the host to firehose sources in relay mode) +- `com.atproto.sync.subscribeRepos` (WebSocket firehose stream, requires `relay` feature) diff --git a/docs/xrpc/backlinks.md b/docs/xrpc/backlinks.md new file mode 100644 index 0000000..f38e237 --- /dev/null +++ b/docs/xrpc/backlinks.md @@ -0,0 +1,32 @@ +# blue.microcosm.links.* + +hydrant implements a subset of [microcosm constellation](https://constellation.microcosm.blue/) when it's built with the `backlinks` cargo feature (`cargo build --features backlinks`). + +when enabled, hydrant indexes all AT URI and DID references found inside stored records into a reverse index. this lets you efficiently answer "what records link to this subject?". + +## blue.microcosm.links.getBacklinks + +return records that link to a given subject. + +| param | required | description | +| :--- | :--- | :--- | +| `subject` | yes | AT URI or DID to look up backlinks for. | +| `source` | no | filter by source collection, e.g. `app.bsky.feed.like`. also accepts `collection:path` form to further filter by field path, e.g. `app.bsky.feed.like:subject.uri`. the path is matched against the dotted field path within the record (`.` is prepended automatically). | +| `limit` | no | max results to return (default 50, max 100). | +| `cursor` | no | opaque pagination cursor from a previous response. | +| `reverse` | no | if `true`, return results in reverse order (default `false`). | + +returns `{ backlinks: [{ uri, cid }], cursor? }`. + +results are ordered by source record rkey (ascending by default, descending when `reverse=true`). the cursor is stable across new insertions for TID rkey records. + +## blue.microcosm.links.getBacklinksCount + +return the number of records that link to a given subject. + +| param | required | description | +| :--- | :--- | :--- | +| `subject` | yes | AT URI or DID to count backlinks for. | +| `source` | no | filter by source collection (same format as `getBacklinks`). | + +returns `{ count }`. diff --git a/docs/xrpc/hydrant.md b/docs/xrpc/hydrant.md new file mode 100644 index 0000000..bc00306 --- /dev/null +++ b/docs/xrpc/hydrant.md @@ -0,0 +1,24 @@ +# systems.gaze.hydrant.* + +these are some non-standard XRPCs that might be useful. + +## systems.gaze.hydrant.countRecords + +return the total number of stored records in a collection. + +| param | required | description | +| :--- | :--- | :--- | +| `identifier` | yes | DID or handle of the repository. | +| `collection` | yes | NSID of the collection. | + +returns `{ count }`. + +## systems.gaze.hydrant.describeRepo + +return account and identity information about this repo. this is equal to `com.atproto.repo.describeRepo`, except we don't return the full DID document. the handle is bi-directionally verified, if its invalid or the handle does not exist we return "handle.invalid". + +| param | required | description | +| :--- | :--- | :--- | +| `identifier` | yes | DID or handle of the repository. | + +returns `{ did, handle, pds, collections }`. diff --git a/flake.lock b/flake.lock index 22e3228..39e26b8 100644 --- a/flake.lock +++ b/flake.lock @@ -143,6 +143,22 @@ "type": "github" } }, + "nixpkgs_3": { + "locked": { + "lastModified": 1776329215, + "narHash": "sha256-a8BYi3mzoJ/AcJP8UldOx8emoPRLeWqALZWu4ZvjPXw=", + "owner": "nixos", + "repo": "nixpkgs", + "rev": "b86751bc4085f48661017fa226dee99fab6c651b", + "type": "github" + }, + "original": { + "owner": "nixos", + "ref": "nixpkgs-unstable", + "repo": "nixpkgs", + "type": "github" + } + }, "parts": { "inputs": { "nixpkgs-lib": [ @@ -232,7 +248,8 @@ "inputs": { "nci": "nci", "nixpkgs": "nixpkgs_2", - "parts": "parts_2" + "parts": "parts_2", + "verbiage": "verbiage" } }, "rust-overlay": { @@ -299,6 +316,24 @@ "repo": "treefmt-nix", "type": "github" } + }, + "verbiage": { + "inputs": { + "nixpkgs": "nixpkgs_3" + }, + "locked": { + "lastModified": 1776647130, + "narHash": "sha256-XCMjiqN2bvJ016q7JEW6tKOfk3pPrzPsACmbtSJPY50=", + "owner": "90-008", + "repo": "verbiage", + "rev": "cf8cfa6e7cb4d9ab1fc4b5088c01235b09928850", + "type": "github" + }, + "original": { + "owner": "90-008", + "repo": "verbiage", + "type": "github" + } } }, "root": "root", diff --git a/flake.nix b/flake.nix index 78bd87b..4f7a1cb 100644 --- a/flake.nix +++ b/flake.nix @@ -2,6 +2,7 @@ inputs.parts.url = "github:hercules-ci/flake-parts"; inputs.nixpkgs.url = "github:nixos/nixpkgs/nixpkgs-unstable"; inputs.nci.url = "github:90-008/nix-cargo-integration"; + inputs.verbiage.url = "github:90-008/verbiage"; outputs = inp: @@ -36,6 +37,7 @@ clang wild psmisc + inputs'.verbiage.packages.default ]); }); }; -- 2.51.2