diff --git a/Cargo.lock b/Cargo.lock index b6a90e0..6334ff6 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -423,6 +423,23 @@ version = "0.3.31" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "05f29059c0c2090612e8d742178b0580d2dc940c837851ad723096f87af6663e" +[[package]] +name = "futures-io" +version = "0.3.31" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9e5c1b78ca4aae1ac06c48a526a655760685149f0d465d21f37abfe57ce075c6" + +[[package]] +name = "futures-macro" +version = "0.3.31" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "162ee34ebcb7c64a8abebc059ce0fee27c2262618d7b60ed8faf72fef13c3650" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + [[package]] name = "futures-sink" version = "0.3.31" @@ -442,9 +459,14 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9fa08315bb612088cc391249efdc3bc77536f16c91f6cf495e6fbe85b20a4a81" dependencies = [ "futures-core", + "futures-io", + "futures-macro", + "futures-sink", "futures-task", + "memchr", "pin-project-lite", "pin-utils", + "slab", ] [[package]] @@ -1032,6 +1054,15 @@ dependencies = [ [[package]] name = "pai-worker" version = "0.1.0" +dependencies = [ + "chrono", + "pai-core", + "rss", + "serde", + "serde_json", + "serde_urlencoded", + "worker", +] [[package]] name = "percent-encoding" @@ -1039,6 +1070,26 @@ version = "2.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9b4f627cb1b25917193a259e49bdad08f671f8d9708acfd5fe0a8c1455d87220" +[[package]] +name = "pin-project" +version = "1.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "677f1add503faace112b9f1373e43e9e054bfdd22ff1a63c1bc485eaec6a6a8a" +dependencies = [ + "pin-project-internal", +] + +[[package]] +name = "pin-project-internal" +version = "1.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e918e4ff8c4549eb882f14b3a4bc8c8bc93de829416eacf579f1207a8fbf861" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + [[package]] name = "pin-project-lite" version = "0.2.16" @@ -1200,6 +1251,15 @@ version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "88f8660c1ff60292143c98d08fc6e2f654d722db50410e3f3797d40baaf9d8f3" +[[package]] +name = "rss" +version = "2.0.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b2107738f003660f0a91f56fd3e3bd3ab5d918b2ddaf1e1ec2136fb1c46f71bf" +dependencies = [ + "quick-xml", +] + [[package]] name = "rusqlite" version = "0.37.0" @@ -1314,6 +1374,17 @@ dependencies = [ "serde_derive", ] +[[package]] +name = "serde-wasm-bindgen" +version = "0.6.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8302e169f0eddcc139c70f139d19d6467353af16f9fce27e8c30158036a1e16b" +dependencies = [ + "js-sys", + "serde", + "wasm-bindgen", +] + [[package]] name = "serde_core" version = "1.0.228" @@ -1841,6 +1912,19 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "wasm-streams" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "15053d8d85c7eccdbefef60f06769760a563c7f0a9d6902a13d35c7800b0ad65" +dependencies = [ + "futures-util", + "js-sys", + "wasm-bindgen", + "wasm-bindgen-futures", + "web-sys", +] + [[package]] name = "web-sys" version = "0.3.82" @@ -2089,6 +2173,64 @@ version = "0.46.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f17a85883d4e6d00e8a97c586de764dabcc06133f7f1d55dce5cdc070ad7fe59" +[[package]] +name = "worker" +version = "0.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d9320293035d2074f1fb84baf7d79d7932c183dd04a7e0c143dc75db0d0037ac" +dependencies = [ + "async-trait", + "bytes", + "chrono", + "futures-channel", + "futures-util", + "http", + "http-body", + "js-sys", + "matchit", + "pin-project", + "serde", + "serde-wasm-bindgen", + "serde_json", + "serde_urlencoded", + "tokio", + "url", + "wasm-bindgen", + "wasm-bindgen-futures", + "wasm-streams", + "web-sys", + "worker-macros", + "worker-sys", +] + +[[package]] +name = "worker-macros" +version = "0.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "eb37d4f9d99921836a1e4dc21e6041df9b0c2c5fe3c230edddd172a8ef9e251e" +dependencies = [ + "async-trait", + "proc-macro2", + "quote", + "syn", + "wasm-bindgen", + "wasm-bindgen-futures", + "wasm-bindgen-macro-support", + "worker-sys", +] + +[[package]] +name = "worker-sys" +version = "0.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "07b4e2ca5d405247a986d533bba78c396c941835747977631168b8b05304f1b6" +dependencies = [ + "cfg-if", + "js-sys", + "wasm-bindgen", + "web-sys", +] + [[package]] name = "writeable" version = "0.6.2" diff --git a/DEPLOYMENT.md b/DEPLOYMENT.md index 5c24b0c..a73da6e 100644 --- a/DEPLOYMENT.md +++ b/DEPLOYMENT.md @@ -1,10 +1,31 @@ # Personal Activity Index – Deployment Guide -This guide walks through two common reverse proxy setups for `pai serve`: **nginx** and **Caddy**. Both sections include native (host binary) instructions and optional Docker paths if you prefer containerized deployments. +This guide walks through two common reverse proxy setups for `pai serve`: **nginx** and **Caddy**. +Both sections include native (host binary) instructions and optional Docker paths if you prefer containerized deployments. + +## Table of Contents + +- [Prerequisites](#prerequisites) +- [nginx Deployment](#nginx-deployment) + - [Host Setup](#host-setup) + - [nginx Config](#nginx-config) + - [Optional: nginx via Docker](#optional-nginx-via-docker) +- [Caddy Deployment](#caddy-deployment) + - [Host Setup](#host-setup-1) + - [Caddyfile Example](#caddyfile-example) + - [Optional: Caddy + Docker Compose](#optional-caddy--docker-compose) +- [Health Checks & Monitoring](#health-checks--monitoring) +- [Cloudflare Worker Deployment](#cloudflare-worker-deployment) + - [Prerequisites](#prerequisites-1) + - [Quick Start](#quick-start) + - [Cron Triggers](#cron-triggers) + - [API Endpoints](#api-endpoints) + - [Local Development](#local-development) + - [Monitoring](#monitoring) ## Prerequisites -1. Build the CLI binary: +1. Build binary: ```sh cargo build --release -p pai @@ -13,7 +34,6 @@ This guide walks through two common reverse proxy setups for `pai serve`: **ngin The binary will live at `target/release/pai`. 2. Prepare a configuration + database location. The default locations follow the XDG spec, but you can override them with `-C` (config dir) and `-d` (database path). - 3. Run a sync at least once so the database has data: ```sh @@ -160,13 +180,121 @@ Use the same `Caddyfile` contents as above, but point `reverse_proxy` to `pai:80 ## Health Checks & Monitoring -- `GET /status` – lightweight JSON (`status`, total items, counts per `source_kind`). Ideal for load balancer health probes. +- `GET /status` – lightweight JSON (`status`, version, uptime, total items, counts per `source_kind`). Ideal for load balancer health probes. - `GET /api/feed?limit=1` ensures the server can read from SQLite and return real data. - `GET /api/item/{id}` is handy for debugging a specific record. - Consider wiring `/status` into nginx/Caddy health checks (`/healthz`) or your platform’s monitoring agents. -## Security Tips +## Cloudflare Worker Deployment + +The Personal Activity Index can also be deployed as a Cloudflare Worker with D1 database, providing a serverless alternative to self-hosting. + +### Prerequisites + +1. Cloudflare account with Workers enabled +2. [Wrangler CLI](https://developers.cloudflare.com/workers/wrangler/install-and-update/) installed +3. Rust toolchain with `wasm32-unknown-unknown` target + +### Quick Start + +#### 1. Generate Scaffolding + +Use the `pai cf-init` command to generate Cloudflare Worker configuration: + +```sh +# Dry run to preview files +pai cf-init --dry-run -o cloudflare-deployment + +# Create scaffolding +pai cf-init -o cloudflare-deployment +cd cloudflare-deployment +``` + +This creates: + +- `wrangler.example.toml` - Worker configuration template +- `schema.sql` - D1 database schema +- `README.md` - Deployment instructions + +#### 2. Create D1 Database + +```sh +wrangler d1 create personal-activity-db +``` + +Copy the database ID from the output and update `wrangler.example.toml`: + +```toml +[[d1_databases]] +binding = "DB" +database_name = "personal-activity-db" +database_id = "your-database-id-here" # Replace with actual ID +``` + +Then copy to the active config: + +```sh +cp wrangler.example.toml wrangler.toml +``` + +#### 3. Initialize Database Schema + +```sh +wrangler d1 execute personal-activity-db --file=schema.sql +``` + +#### 4. Build and Deploy + +```sh +# Build the worker +cd .. +cargo install worker-build +worker-build --release -p pai-worker + +# Deploy +cd cloudflare-deployment +wrangler deploy +``` + +### Cron Triggers + +The worker includes a scheduled event handler for automatic syncing. Configure the schedule in `wrangler.toml`: + +```toml +[triggers] +crons = ["0 * * * *"] # Every hour at minute 0 +``` + +Common schedules: + +- `*/30 * * * *` - Every 30 minutes +- `0 */6 * * *` - Every 6 hours +- `0 0 * * *` - Daily at midnight + +### API Endpoints + +The Worker exposes the same API as the self-hosted server: + +- `GET /api/feed?source_kind=bluesky&limit=20` - List items +- `GET /api/item/{id}` - Get single item +- `GET /status` - Health check + +### Local Development + +Test the worker locally before deploying: + +```sh +wrangler dev +``` + +This starts a local server at `http://localhost:8787` with live reload. + +### Monitoring + +View logs in real-time: + +```sh +wrangler tail +``` -- Bind the `pai serve` process to `127.0.0.1` and let the proxy handle TLS. -- Run the binary as an unprivileged user with read/write access only to the DB path. -- Regularly rotate TLS certificates (Caddy does this automatically; for nginx use certbot or similar). +Or check logs in the [Cloudflare Dashboard](https://dash.cloudflare.com) under Workers & Pages. diff --git a/README.md b/README.md index 42040d5..29c6328 100644 --- a/README.md +++ b/README.md @@ -35,6 +35,12 @@ pai list -n 10 # Check database pai db-check + +# Install the manpage so `man pai` works +pai man --install + +# Generate manpage to a file +pai man -o pai.1 ```
@@ -65,6 +71,7 @@ See [config.example.toml](./config.example.toml) for a complete example with all ## Documentation - CLI synopsis: `pai -h`, `pai -h`, or `pai man` for the generated `pai(1)` page. +- `pai man --install [--install-dir DIR]` copies `pai.1` into a MANPATH directory (defaults to `~/.local/share/man/man1`) so `man pai` works like any other UNIX tool. - Database schema and config reference: [config.example.toml](./config.example.toml). - Deployment topologies: [DEPLOYMENT.md](./DEPLOYMENT.md). diff --git a/TODO.md b/TODO.md new file mode 100644 index 0000000..a1e20c5 --- /dev/null +++ b/TODO.md @@ -0,0 +1,466 @@ + +# Personal Activity Index CLI – Roadmap & Tasks + +Objective: +Build a POSIX-style Rust CLI that ingests content from Substack, Bluesky, and Leaflet into SQLite, with an optional Cloudflare Worker + D1 deployment path. + +Targets: + +- Self-host: single binary + SQLite. +- Cloudflare: Rust Worker + D1 + Cron triggers. + +## Workspace & Architecture + +**Goal:** Shared core library, CLI frontend, and Worker frontend, with clear separation of concerns. + +- [x] Create Cargo workspace layout: + - [x] `core/` – shared types, fetchers, and storage traits. + - [x] `cli/` – POSIX-style binary (`pai`). + - [x] `worker/` – Cloudflare Worker using `workers-rs`. +- [x] In `core/`: + - [x] Define `SourceKind` enum: `substack`, `bluesky`, `leaflet`. + - [x] Define `Item` struct with fields: + - [x] `id`, `source_kind`, `source_id`, `author`, `title`, `summary`, + `url`, `content_html`, `published_at`, `created_at`. + - [x] Define `Storage` trait with at minimum: + - [x] `insert_or_replace_item(&self, item: &Item) -> Result<()>` + - [x] `list_items(&self, filter: &ListFilter) -> Result>` + - [x] Define `SourceFetcher` trait: + - [x] `fn sync(&self, storage: &dyn Storage) -> Result<()>` +- [x] In `cli/`: + - [x] Add argument parsing that follows POSIX conventions: + - Options of the form `-h`, `-V`, `-C dir`, `-d path`, etc. + - Options come before operands/subcommands where possible. + - [x] Define subcommands (as operands) with their own POSIX-style options: + - [x] `sync` + - [x] `list` + - [x] `export` + - [x] `serve` +- [x] In `core/`: + - [x] Implement `sync_all_sources(config, storage)` that calls each fetcher. + +## Milestone 1 – Local SQLite Storage (Self-host Base) + +**Goal:** `pai` can sync data into a local SQLite file. + +- [x] Choose SQLite crate (native mode): + - [x] e.g. `rusqlite` +- [x] Define SQL schema and migrations: + - [x] `items` table: + + ```sql + CREATE TABLE IF NOT EXISTS items ( + id TEXT PRIMARY KEY, + source_kind TEXT NOT NULL, + source_id TEXT NOT NULL, + author TEXT, + title TEXT, + summary TEXT, + url TEXT NOT NULL, + content_html TEXT, + published_at TEXT NOT NULL, + created_at TEXT NOT NULL DEFAULT CURRENT_TIMESTAMP + ); + + CREATE INDEX IF NOT EXISTS idx_items_source_date ON items (source_kind, source_id, published_at DESC); + ``` + + - [x] Embed migrations or provide `schema.sql` + `pai db-migrate` command. +- [x] Implement `SqliteStorage` in `cli/`: + - [x] Opens/creates DB at `-d path` or `$XDG_DATA_HOME/pai/pai.db` fallback. + - [x] Implements `Storage` trait. +- [x] Implement `pai sync` path: + - [x] `pai sync` → load config → open SQLite → call `sync_all_sources`. + - [x] Exit codes: + - [x] `0` on success, non-zero on failure. +- [x] Add `pai db-check`: + - [x] Verifies schema and prints basic stats (item count per source). + +## Milestone 2 – Source Integrations ✅ + +**Goal:** All three sources can be ingested via the CLI. + +**Status:** COMPLETE - All three source integrations (Substack RSS, Bluesky AT Protocol, Leaflet RSS) are implemented and tested with real data. + +### 2.1 Substack (Pattern Matched) + +- [x] Add config support: + + ```toml + [sources.substack] + enabled = true + base_url = "https://patternmatched.substack.com" + ``` + +- [x] Implement `SubstackFetcher` in `core/`: + + - [x] Fetch `{base_url}/feed`. + - [x] Parse RSS using `feed-rs`. + - [x] Map ``: + + - [x] `id` = GUID if present, otherwise `link`. + - [x] `source_kind = "substack"`. + - [x] `source_id = "patternmatched.substack.com"`. + - [x] `title`, `summary` from RSS `title`/`description`. + - [x] `url` from `link`. + - [x] `published_at` from `pubDate` (normalized to ISO 8601). +- [x] Wire into `sync_all_sources` when enabled. + +### 2.2 Bluesky (desertthunder.dev) + +- [x] Add config support: + + ```toml + [sources.bluesky] + enabled = true + handle = "desertthunder.dev" + ``` + +- [x] Implement `BlueskyFetcher` in `core/`: + + - [x] Fetch: + + - [x] `https://public.api.bsky.app/xrpc/app.bsky.feed.getAuthorFeed?actor=desertthunder.dev&limit=N` + - [x] Filter out reposts/quotes (only original posts). + - [x] Map `post` record: + + - [x] `id` = `uri` (AT URI). + - [x] `source_kind = "bluesky"`. + - [x] `source_id = "desertthunder.dev"`. + - [x] `title` = truncated text up to N chars. + - [x] `summary` = full text (or truncated). + - [x] `url` = canonical `https://bsky.app/profile/…/post/…` derived from URI. + - [x] `published_at` = `record.createdAt` (ISO 8601 already). + - [ ] Optional: + + - [ ] Support pagination via `cursor` until a configured max number of posts. + +### 2.3 Leaflet (desertthunder / stormlightlabs) + +- [x] Add config support: + + ```toml + [[sources.leaflet]] + enabled = true + id = "desertthunder" + base_url = "https://desertthunder.leaflet.pub" + + [[sources.leaflet]] + enabled = true + id = "stormlightlabs" + base_url = "https://stormlightlabs.leaflet.pub" + ``` + +- [x] Use AT Protocol instead of HTML parsing: + + - [x] Use `com.atproto.repo.listRecords` with collection `pub.leaflet.post`. + +- [x] Implement `LeafletFetcher` in `core/`: + + - [x] For each configured pub: + + - [x] Fetch records using AT Protocol. + - [x] Parse `pub.leaflet.post` records. + - [x] For each post: + + - [x] Extract `title` from record. + - [x] Extract `publishedAt` or `createdAt`. + - [x] Derive summary from `summary` or `content` field. + - [x] Generate URL using `slug` or record ID. + - [x] Normalize date to ISO 8601 for `published_at`. + - [x] Insert or replace items in storage. + +- [x] Wire into `sync_all_sources`. + +## Milestone 3 – Query, Filter, and Export (CLI Only) + +**Goal:** Make local data usable even without HTTP. + +- [x] Implement `pai list`: + - [x] Syntax: `pai list [options]` (options before operands). + - [x] Options: + - [x] `-k kind` filter by `source_kind` (`substack`, `bluesky`, `leaflet`). + - [x] `-S id` filter by `source_id` (host/handle). + - [x] `-n N` limit number of results (default 20). + - [x] `-s time` “since time” (e.g. ISO 8601, or “7d” shorthand if desired). + - [x] `-q pattern` simple substring filter on title/summary. + - [x] Render as ASCII table or simple text. +- [x] Implement `pai export`: + - [x] Syntax: `pai export -f format [-o file]`. + - [x] Supported formats: + - [x] `json` (default). + - [x] `ndjson` (optional). + - [x] `rss` (optional aggregate). + - [x] Options: + - [x] `-f format` (`json`, `rss`, …). + - [x] `-o path` output file (default stdout). +- [x] Implement exit statuses for typical cases: + - [x] `0` on success. + - [x] `>0` on error (bad args, DB error, network failure, etc.). + +## Milestone 4 – Self-hosted HTTP Server Mode + +**Goal:** Provide a small HTTP API backed by SQLite for self-hosted deployments. + +- [x] Add `serve` subcommand in `cli/`: + - [x] Syntax: `pai serve [options]`. + - [x] Options: + - [x] `-d path` database path. + - [x] `-a addr` listen address (default `127.0.0.1:8080`). + - [x] Follows POSIX conventions: all options before operands. +- [x] Implement HTTP server (`axum`): + - [x] `GET /api/feed` – list all items, newest first. + - [x] Query params: + - [x] `source_kind`, `source_id`, `limit`, `since`, `q`. + - [x] Optional: + - [x] `GET /api/item/{id}` for a single item. +- [x] Ensure graceful shutdown and clean error handling. +- [x] Document reverse-proxy examples (Caddy, nginx). + +## Milestone 5 – Cloudflare Worker + D1 Frontend + +**Goal:** Provide an alternative deployment path using Cloudflare Workers with D1 and Cron triggers. + +- [ ] In `worker/`: + - [ ] Depend on `worker` crate with `d1` feature enabled. + - [ ] Reuse `core::Item` and parsing code (ensure crates are WASM-friendly). +- [ ] Configure D1: + - [ ] Provide `schema.sql` compatible with D1 (same `items` table). + - [ ] Example `wrangler.toml` with `[[d1_databases]]` binding. +- [ ] Implement Worker routes: + - [ ] `GET /api/feed` with similar semantics as CLI server. +- [ ] Implement `scheduled` handler for Cron: + - [ ] On each scheduled run, call per-source syncers writing to D1. + - [ ] Document cron configuration in `wrangler.toml`. +- [ ] Add `pai cf-init` in `cli/`: + - [ ] Generates a starter `wrangler.toml`. + - [ ] Prints instructions to create D1 DB and bind it. + +## Milestone 6 – POSIX Polish, Packaging, and Docs + +**Goal:** Make the CLI feel like a “real UNIX utility” and easy to adopt. + +- [ ] Verify POSIX-style argument handling: + - [ ] Short options only in usage syntax; long options are optional extensions. + - [ ] Options before operands/subcommands in docs and examples. + - [ ] Support grouped short options where meaningful (e.g. `-hv`). +- [ ] Implement: + - [ ] `-h` – usage synopsis and options (per POSIX convention). + - [ ] `-V` – version info. +- [ ] Add manpage-style documentation using clap_mangen () in build.rs: + - [ ] `man/pai.1` with SYNOPSIS, DESCRIPTION, OPTIONS, OPERANDS, EXIT STATUS, ENVIRONMENT, FILES, EXAMPLES. +- [ ] Publish `pai` crate to crates.io. +- [ ] Write README with: + - [ ] Self-hosted quickstart. + - [ ] Cloudflare Worker quickstart. + - [ ] Config reference (`config.toml`). + +## 2. CLI & Config Spec (POSIX-style) + +### 2.1 POSIX argument conventions you’re aligning with + +Key constraints you want to follow: + +- Options are introduced by a single `-` followed by a single letter (`-h`, `-V`, `-d path`). :contentReference[oaicite:0]{index=0} +- Options that require arguments use a separate token: `-d path` rather than `-dpath`. :contentReference[oaicite:1]{index=1} +- Options appear before operands (here, subcommands and file paths) in the recommended syntax: + `utility_name [-a] [-b arg] operand1 operand2 …`. :contentReference[oaicite:2]{index=2} +- `-h` for help, `-V` for version are widely conventional. :contentReference[oaicite:3]{index=3} + +You *can* still offer `--long-option` aliases as a GNU-style extension; just document the POSIX short forms as canonical. :contentReference[oaicite:4]{index=4} + +### 2.2 CLI synopsis + +**Utility name:** `pai` (single binary). + +#### Global synopsis + +```text +pai [-hV] [-C config_dir] [-d db_path] command [command-options] [command-operands] +``` + +- `-h` + Print usage and exit. + +- `-V` + Print version and exit. + +- `-C config_dir` + Set configuration directory. Default: `$XDG_CONFIG_HOME/pai` or `$HOME/.config/pai`. + +- `-d db_path` + Path to SQLite database file. Default: `$XDG_DATA_HOME/pai/pai.db` or `$HOME/.local/share/pai/pai.db`. + +Subcommands are treated as **operands** in POSIX terms; each subcommand then has its own POSIX-style options. + +### 2.3 Subcommands and their options + +#### 1. `sync` – fetch and store content + +```text +pai [-C config_dir] [-d db_path] sync [-a] [-k kind] [-S source_id] +``` + +Options: + +- `-a` + Sync all configured sources (default if `-k` not specified). + +- `-k kind` + Sync only a particular source kind: + + - `substack` + - `bluesky` + - `leaflet` + +- `-S source_id` + Sync only a specific source instance (e.g. `patternmatched.substack.com`, `desertthunder.dev`, `desertthunder.leaflet.pub`, `stormlightlabs.leaflet.pub`). + +Examples: + +```sh +pai sync -a +pai sync -k substack +pai sync -k leaflet -S desertthunder.leaflet.pub +``` + +#### 2. `list` – inspect stored items + +```text +pai [-C config_dir] [-d db_path] list [-k kind] [-S source_id] [-n number] [-s since] [-q pattern] +``` + +Options: + +- `-k kind` + Filter by source kind (`substack`, `bluesky`, `leaflet`). + +- `-S source_id` + Filter by specific source id (host or handle). + +- `-n number` + Maximum number of items to display (default 20). + +- `-s since` + Only show items published at or after this time. The CLI can accept ISO 8601 (`2025-11-23T00:00:00Z`) and, as a convenience, relative strings like `7d`, `24h` if you want. + +- `-q pattern` + Filter items whose title/summary contains the given substring. + +#### 3. `export` – produce feeds/files + +```text +pai [-C config_dir] [-d db_path] export [-k kind] [-S source_id] [-n number] [-s since] [-q pattern] [-f format] [-o file] +``` + +Options (in addition to `list` filters): + +- `-f format` + Output format: + + - `json` (default) + - `ndjson` + - `rss` (optional) + +- `-o file` + Output file. Default is standard output. + +Examples: + +```sh +pai export -f json -o activity.json +pai export -k bluesky -n 50 -f ndjson +``` + +#### 4. `serve` – self-host HTTP API + +```text +pai [-C config_dir] [-d db_path] serve [-a address] +``` + +Options: + +- `-a address` + Address to bind HTTP server to. Default: `127.0.0.1:8080`. + +The HTTP API mirrors the query semantics of `list` and `export`: + +- `GET /api/feed?source_kind=bluesky&limit=50&since=...&q=...` + +#### 5. `cf-init` – scaffold Cloudflare deployment + +```text +pai cf-init [-o dir] +``` + +Options: + +- `-o dir` + Directory into which to write `wrangler.toml`, `schema.sql`, and a sample `worker` entry point. Default: current directory. + +This command doesn’t need DB access; it just writes templates and prints next steps (create D1 DB, bind it, set up Cron). + +### 2.4 `config.toml` spec + +**Default location:** + +- `$XDG_CONFIG_HOME/pai/config.toml` or +- `$HOME/.config/pai/config.toml` if `XDG_CONFIG_HOME` is unset. + +**Top-level layout:** + +```toml +[database] +# Path to SQLite database for self-host mode. +# Ignored by the Worker; used only by `pai` binary. +path = "/home/owais/.local/share/pai/pai.db" + +[deployment] +# Which deploy targets are configured. +# "sqlite" is always available; "cloudflare" is optional. +mode = "sqlite" # or "cloudflare" + +[deployment.cloudflare] +# Optional metadata for generating wrangler.toml, etc. +worker_name = "personal-activity-index" +d1_binding = "DB" +database_name = "personal_activity_db" + +[sources.substack] +enabled = true +base_url = "https://patternmatched.substack.com" + +[sources.bluesky] +enabled = true +handle = "desertthunder.dev" + +[[sources.leaflet]] +enabled = true +id = "desertthunder" +base_url = "https://desertthunder.leaflet.pub" + +[[sources.leaflet]] +enabled = true +id = "stormlightlabs" +base_url = "https://stormlightlabs.leaflet.pub" +``` + +**Notes:** + +- The CLI should **not** require the Cloudflare section unless a user explicitly wants to generate Worker scaffolding. +- The Worker itself will get its D1 binding and Cron schedule from `wrangler.toml` and the Cloudflare dashboard, not from this config file; you just reuse the same schema and `Item` type. + +### 2.5 POSIX compliance checklist + +When you implement the CLI parsing, you can sanity-check against POSIX & GNU guidance: + +- Short options are single letters with a single leading `-`. ([The Open Group][1]) +- Options precede non-option arguments (your commands and operands) in the usage examples. ([The Open Group][1]) +- Options that take arguments are formatted as `-x arg` rather than `-xarg` in documentation. ([gnu.org][2]) +- You provide `-h` / `-V` and consistent help text. ([Baeldung on Kotlin][3]) +- Long options (`--help`, `--version`, `--config-dir`, etc.) can be supported as extensions but are not required for conformance. ([Software Engineering Stack Exchange][4]) + +[1]: https://pubs.opengroup.org/onlinepubs/9699919799/basedefs/V1_chap12.html "12. Utility Conventions" +[2]: https://www.gnu.org/s/libc/manual/html_node/Argument-Syntax.html "Argument Syntax (The GNU C Library)" +[3]: https://www.baeldung.com/linux/posix "A Guide to POSIX | Baeldung on Linux" +[4]: https://softwareengineering.stackexchange.com/questions/70357/command-line-options-style-posix-or-what "Command line options style - POSIX or what?" diff --git a/cli/src/app.rs b/cli/src/app.rs index dff2549..a1316ef 100644 --- a/cli/src/app.rs +++ b/cli/src/app.rs @@ -109,4 +109,31 @@ pub enum Commands { #[arg(short = 'f')] force: bool, }, + + /// Generate or install the pai(1) manpage + Man { + /// Output file (default: stdout) + #[arg(short = 'o', value_name = "FILE")] + output: Option, + + /// Install into a manpath directory (defaults to ~/.local/share/man if unset) + #[arg(long)] + install: bool, + + /// Custom directory for --install (e.g., /usr/local/share/man) + #[arg(long, value_name = "DIR")] + install_dir: Option, + }, + + /// Initialize Cloudflare Worker deployment scaffolding + #[command(name = "cf-init")] + CfInit { + /// Output directory for scaffolding (default: current directory) + #[arg(short = 'o', value_name = "DIR")] + output_dir: Option, + + /// Dry run - show what would be created without writing files + #[arg(long)] + dry_run: bool, + }, } diff --git a/cli/src/main.rs b/cli/src/main.rs index 69bb1ee..a2bf97d 100644 --- a/cli/src/main.rs +++ b/cli/src/main.rs @@ -8,9 +8,9 @@ use chrono::{DateTime, Duration, Utc}; use clap::Parser; use owo_colors::OwoColorize; use pai_core::{Config, Item, ListFilter, PaiError, SourceKind}; -use std::fs::File; +use std::fs; use std::io::{self, Write}; -use std::path::PathBuf; +use std::path::{Path, PathBuf}; use std::str::FromStr; use storage::SqliteStorage; @@ -18,6 +18,7 @@ const PUBLISHED_WIDTH: usize = 19; const KIND_WIDTH: usize = 9; const SOURCE_WIDTH: usize = 24; const TITLE_WIDTH: usize = 60; +const MAN_PAGE: &str = include_str!(env!("PAI_MAN_PAGE")); fn main() { let cli = Cli::parse(); @@ -31,6 +32,8 @@ fn main() { Commands::Serve { address } => handle_serve(cli.db_path, address), Commands::DbCheck => handle_db_check(cli.db_path), Commands::Init { force } => handle_init(cli.config_dir, force), + Commands::Man { output, install, install_dir } => handle_man(output, install, install_dir), + Commands::CfInit { output_dir, dry_run } => handle_cf_init(output_dir, dry_run), }; if let Err(e) = result { @@ -170,11 +173,10 @@ fn handle_init(config_dir: Option, force: bool) -> Result<(), PaiError> return Err(PaiError::Config("Config file already exists".to_string())); } - std::fs::create_dir_all(&config_dir) - .map_err(|e| PaiError::Config(format!("Failed to create config directory: {e}")))?; + fs::create_dir_all(&config_dir).map_err(|e| PaiError::Config(format!("Failed to create config directory: {e}")))?; let default_config = include_str!("../../config.example.toml"); - std::fs::write(&config_path, default_config) + fs::write(&config_path, default_config) .map_err(|e| PaiError::Config(format!("Failed to write config file: {e}")))?; println!("{} Created configuration file", "Success:".green().bold()); @@ -195,6 +197,203 @@ fn handle_init(config_dir: Option, force: bool) -> Result<(), PaiError> Ok(()) } +fn handle_man(output: Option, install: bool, install_dir: Option) -> Result<(), PaiError> { + if install && output.is_some() { + return Err(PaiError::InvalidArgument( + "Use either --install or -o/--output when generating manpages".to_string(), + )); + } + + let target = if install { Some(resolve_man_install_path(install_dir)?) } else { output }; + + let mut writer = create_output_writer(target.as_ref())?; + writer.write_all(MAN_PAGE.as_bytes()).map_err(PaiError::Io)?; + writer.flush().map_err(PaiError::Io)?; + + if let Some(path) = target { + if install { + println!("{} Installed manpage to {}", "Success:".green(), path.display()); + if let Some(root) = man_root_for(&path) { + println!( + "{} Ensure {} is on your MANPATH, then run {}", + "Hint:".yellow(), + root.display(), + "man pai".bright_black() + ); + } else { + println!( + "{} Run man pai after adding the install dir to MANPATH.", + "Hint:".yellow() + ); + } + } else { + println!("{} Wrote manpage to {}", "Success:".green(), path.display()); + } + } + + Ok(()) +} + +fn resolve_man_install_path(custom_dir: Option) -> Result { + let base = if let Some(dir) = custom_dir { dir } else { find_writable_man_dir()? }; + + let install_dir = if base.file_name().map(|os| os == "man1").unwrap_or(false) { base } else { base.join("man1") }; + + fs::create_dir_all(&install_dir).map_err(|e| { + PaiError::Io(io::Error::new( + e.kind(), + format!("Failed to create man directory {}: {}", install_dir.display(), e), + )) + })?; + + Ok(install_dir.join("pai.1")) +} + +fn find_writable_man_dir() -> Result { + let candidates = [ + dirs::data_local_dir().map(|d| d.join("man")), + dirs::home_dir().map(|d| d.join(".local/share/man")), + Some(PathBuf::from("/usr/local/share/man")), + Some(PathBuf::from("/opt/homebrew/share/man")), + Some(PathBuf::from("/usr/local/Homebrew/share/man")), + ]; + + for candidate in candidates.iter().flatten() { + if candidate.exists() { + let test_file = candidate.join(".pai-write-test"); + if fs::write(&test_file, b"test").is_ok() { + let _ = fs::remove_file(&test_file); + return Ok(candidate.clone()); + } + } else if let Some(parent) = candidate.parent() { + if parent.exists() { + let test_dir = candidate.join("man1"); + if fs::create_dir_all(&test_dir).is_ok() { + let _ = fs::remove_dir_all(&test_dir); + return Ok(candidate.clone()); + } + } + } + } + + if let Some(data_dir) = dirs::data_local_dir() { + return Ok(data_dir.join("man")); + } + + Err(PaiError::Config( + "Unable to find a writable man page directory. Use --install-dir to specify one.".to_string(), + )) +} + +fn man_root_for(path: &Path) -> Option<&Path> { + path.parent()?.parent() +} + +fn handle_cf_init(output_dir: Option, dry_run: bool) -> Result<(), PaiError> { + let target_dir = output_dir.unwrap_or_else(|| PathBuf::from(".")); + + let wrangler_template = include_str!("../../worker/wrangler.example.toml"); + let schema_sql = include_str!("../../worker/schema.sql"); + + let readme_content = r#"# Cloudflare Worker Deployment + +## Quick Start + +1. **Create D1 Database:** + ```sh + wrangler d1 create personal-activity-db + ``` + +2. **Copy the configuration:** + ```sh + cp wrangler.example.toml wrangler.toml + ``` + +3. **Update `wrangler.toml`:** + - Replace `{DATABASE_ID}` with the ID from step 1 + - Adjust `name` and `database_name` if desired + +4. **Initialize the database schema:** + ```sh + wrangler d1 execute personal-activity-db --file=schema.sql + ``` + +5. **Build the worker:** + ```sh + cd .. + cargo install worker-build + worker-build --release -p pai-worker + ``` + +6. **Deploy:** + ```sh + cd worker + wrangler deploy + ``` + +## Testing Locally + +Run the worker locally with: +```sh +wrangler dev +``` + +## Scheduled Syncs + +The worker is configured with a cron trigger (see `wrangler.toml`). The default schedule runs every hour. +To modify the schedule, edit the `crons` array in `wrangler.toml`. + +## API Endpoints + +- `GET /api/feed` - List items with optional filters +- `GET /api/item/:id` - Get a single item by ID +- `GET /status` - Health check + +## Environment Variables + +Configure in `wrangler.toml` under `[vars]`: +- `LOG_LEVEL` - Set logging verbosity (optional) +"#; + + let files = vec![ + ("wrangler.example.toml", wrangler_template), + ("schema.sql", schema_sql), + ("README.md", readme_content), + ]; + + if dry_run { + println!("{} Dry run - showing files that would be created:\n", "Info:".cyan()); + for (filename, content) in &files { + let path = target_dir.join(filename); + println!(" {} {}", "Would create:".bright_black(), path.display()); + println!(" {} bytes", content.len()); + } + println!("\n{} Run without --dry-run to create these files", "Hint:".yellow()); + return Ok(()); + } + + fs::create_dir_all(&target_dir)?; + + for (filename, content) in &files { + let path = target_dir.join(filename); + if path.exists() { + println!("{} {} already exists, skipping", "Warning:".yellow(), filename); + continue; + } + fs::write(&path, content)?; + println!("{} Created {}", "Success:".green(), path.display()); + } + + println!("\n{} Cloudflare Worker scaffolding created!", "Success:".green().bold()); + println!("\n{} Next steps:", "Info:".cyan()); + println!(" 1. cd {}", target_dir.display()); + println!(" 2. Read README.md for deployment instructions"); + println!(" 3. wrangler d1 create personal-activity-db"); + println!(" 4. Update wrangler.example.toml with your database ID"); + + Ok(()) +} + fn normalize_since_input(since: Option) -> Result, PaiError> { normalize_since_with_now(since, Utc::now()) } @@ -224,9 +423,10 @@ fn normalize_since_with_now(since: Option, now: DateTime) -> Result return Ok(Some(dt.with_timezone(&Utc).to_rfc3339())); } - Err(PaiError::InvalidArgument(format!( + let msg = format!( "Invalid since value '{value}'. Use ISO 8601 (e.g. 2024-01-01T00:00:00Z) or relative forms like 7d/24h/60m." - ))) + ); + Err(PaiError::InvalidArgument(msg)) } fn parse_relative_duration(input: &str) -> Option { @@ -296,10 +496,10 @@ fn create_output_writer(path: Option<&PathBuf>) -> Result, PaiErr if let Some(path) = path { if let Some(parent) = path.parent() { if !parent.as_os_str().is_empty() { - std::fs::create_dir_all(parent)?; + fs::create_dir_all(parent)?; } } - let file = File::create(path)?; + let file = fs::File::create(path)?; Ok(Box::new(file)) } else { Ok(Box::new(io::stdout())) @@ -548,4 +748,10 @@ mod tests { let truncated = truncate_for_column("abcdefghijklmnopqrstuvwxyz", 8); assert_eq!(truncated, "abcde..."); } + + #[test] + fn manpage_contains_name_section() { + assert!(MAN_PAGE.contains("NAME")); + assert!(MAN_PAGE.contains("pai")); + } } diff --git a/cli/src/server.rs b/cli/src/server.rs index 90f7343..ef2fbcf 100644 --- a/cli/src/server.rs +++ b/cli/src/server.rs @@ -10,11 +10,11 @@ use axum::{ use owo_colors::OwoColorize; use pai_core::{Item, ListFilter, PaiError, SourceKind}; use serde::{Deserialize, Serialize}; -use std::io; -use std::{net::SocketAddr, path::PathBuf, sync::Arc}; +use std::{io, net::SocketAddr, path::PathBuf, sync::Arc, time::Instant}; use tokio::net::TcpListener; const DEFAULT_LIMIT: usize = 20; +const VERSION: &str = env!("CARGO_PKG_VERSION"); /// Launches the HTTP server using the provided SQLite database path and address. pub(crate) fn serve(db_path: PathBuf, address: String) -> Result<(), PaiError> { @@ -35,7 +35,7 @@ async fn run_server(db_path: PathBuf, addr: SocketAddr) -> Result<(), PaiError> storage.verify_schema()?; drop(storage); - let state = AppState { db_path: Arc::new(db_path) }; + let state = AppState { db_path: Arc::new(db_path), start_time: Instant::now() }; let app = Router::new() .route("/api/feed", get(feed_handler)) @@ -56,6 +56,7 @@ async fn run_server(db_path: PathBuf, addr: SocketAddr) -> Result<(), PaiError> #[derive(Clone)] struct AppState { db_path: Arc, + start_time: Instant, } impl AppState { @@ -72,7 +73,14 @@ impl AppState { .map(|(kind, count)| SourceStat { kind, count }) .collect(); - Ok(StatusResponse { status: "ok", database_path: self.db_path.display().to_string(), total_items, sources }) + Ok(StatusResponse { + status: "ok", + version: VERSION, + uptime_seconds: self.start_time.elapsed().as_secs(), + database_path: self.db_path.display().to_string(), + total_items, + sources, + }) } } @@ -111,6 +119,8 @@ struct FeedResponse { #[derive(Serialize)] struct StatusResponse { status: &'static str, + version: &'static str, + uptime_seconds: u64, database_path: String, total_items: usize, sources: Vec, @@ -240,7 +250,7 @@ mod tests { fn status_snapshot_reports_counts() { let dir = tempdir().unwrap(); let db_path = dir.path().join("status.db"); - let state = AppState { db_path: Arc::new(db_path.clone()) }; + let state = AppState { db_path: Arc::new(db_path.clone()), start_time: Instant::now() }; let storage = state.open_storage().unwrap(); let now = Utc::now().to_rfc3339(); @@ -260,6 +270,8 @@ mod tests { let snapshot = state.status_snapshot().unwrap(); assert_eq!(snapshot.status, "ok"); + assert_eq!(snapshot.version, VERSION); + assert!(snapshot.uptime_seconds < 5); assert_eq!(snapshot.total_items, 1); assert_eq!(snapshot.sources.len(), 1); assert_eq!(snapshot.sources[0].kind, "substack"); diff --git a/conf/.gitignore b/conf/.gitignore new file mode 100644 index 0000000..e6cabe6 --- /dev/null +++ b/conf/.gitignore @@ -0,0 +1,4 @@ +# Ignore actual config files (keep only examples in git) +/nginx.local.conf +/Caddyfile.local +*.local.* diff --git a/conf/Caddyfile b/conf/Caddyfile new file mode 100644 index 0000000..21f9e4f --- /dev/null +++ b/conf/Caddyfile @@ -0,0 +1,31 @@ +# Caddyfile for Personal Activity Index +# Caddy automatically handles HTTPS with Let's Encrypt + +# Basic configuration for localhost (HTTP only) +localhost { + reverse_proxy 127.0.0.1:8080 + encode gzip zstd +} + +# Configuration with custom domain +# Uncomment and replace example.com with your domain: +# +# pai.example.com { +# reverse_proxy 127.0.0.1:8080 +# encode gzip zstd +# +# header { +# Referrer-Policy "no-referrer-when-downgrade" +# X-Content-Type-Options "nosniff" +# X-Frame-Options "SAMEORIGIN" +# } +# +# # Optional: Rate limiting +# # rate_limit { +# # zone static { +# # key {remote_host} +# # events 100 +# # window 1m +# # } +# # } +# } diff --git a/conf/README.md b/conf/README.md new file mode 100644 index 0000000..d1b72f4 --- /dev/null +++ b/conf/README.md @@ -0,0 +1,307 @@ +# Personal Activity Index - Reverse Proxy Configurations + +This directory contains example reverse proxy configurations for deploying the Personal Activity Index HTTP server behind nginx or Caddy. + +## Quick Start + +### Option 1: nginx + +#### macOS + +1. Install nginx: + + ```sh + brew install nginx + ``` + +2. Copy the configuration: + + ```sh + # For localhost testing + cp nginx.conf /opt/homebrew/etc/nginx/servers/pai.conf + + # Or symlink to keep it in sync + ln -s $(pwd)/nginx.conf /opt/homebrew/etc/nginx/servers/pai.conf + ``` + +3. Start the pai server: + + ```sh + pai serve -a 127.0.0.1:8080 + ``` + +4. Start nginx: + + ```sh + brew services start nginx + ``` + +5. Access at + +#### Linux + +1. Install nginx: + + ```sh + # Debian/Ubuntu + sudo apt install nginx + + # RHEL/Fedora + sudo dnf install nginx + ``` + +2. Copy the configuration: + + ```sh + sudo cp nginx.conf /etc/nginx/sites-available/pai + sudo ln -s /etc/nginx/sites-available/pai /etc/nginx/sites-enabled/ + ``` + +3. Start the pai server: + + ```sh + pai serve -a 127.0.0.1:8080 + ``` + +4. Test and reload nginx: + + ```sh + sudo nginx -t + sudo systemctl reload nginx + ``` + +5. Access at + +### Option 2: Caddy + +#### macOS + +1. Install Caddy: + + ```sh + brew install caddy + ``` + +2. Copy the Caddyfile: + + ```sh + cp Caddyfile /opt/homebrew/etc/Caddyfile + ``` + +3. Start the pai server: + + ```sh + pai serve -a 127.0.0.1:8080 + ``` + +4. Start Caddy: + + ```sh + brew services start caddy + ``` + +5. Access at + +#### Linux + +1. Install Caddy: + + ```sh + # See https://caddyserver.com/docs/install + + # Debian/Ubuntu + sudo apt install -y debian-keyring debian-archive-keyring apt-transport-https + curl -1sLf 'https://dl.cloudsmith.io/public/caddy/stable/gpg.key' | sudo gpg --dearmor -o /usr/share/keyrings/caddy-stable-archive-keyring.gpg + curl -1sLf 'https://dl.cloudsmith.io/public/caddy/stable/debian.deb.txt' | sudo tee /etc/apt/sources.list.d/caddy-stable.list + sudo apt update + sudo apt install caddy + ``` + +2. Copy the Caddyfile: + + ```sh + sudo cp Caddyfile /etc/caddy/Caddyfile + ``` + +3. Start the pai server: + + ```sh + pai serve -a 127.0.0.1:8080 + ``` + +4. Reload Caddy: + + ```sh + sudo systemctl reload caddy + ``` + +5. Access at + +## Production Deployment with Custom Domain + +### nginx with SSL + +1. Edit `nginx.conf` and replace `localhost` with your domain (e.g., `pai.example.com`) + +2. Obtain SSL certificates using certbot: + + ```sh + # macOS + brew install certbot + + # Linux + sudo apt install certbot python3-certbot-nginx # Debian/Ubuntu + sudo dnf install certbot python3-certbot-nginx # RHEL/Fedora + ``` + +3. Get certificates: + + ```sh + sudo certbot --nginx -d pai.example.com + ``` + +4. Certbot will automatically update your nginx configuration with SSL settings + +5. Set up auto-renewal: + + ```sh + # Test renewal + sudo certbot renew --dry-run + + # On Linux, certbot sets up a systemd timer automatically + # On macOS, add to crontab: + sudo crontab -e + # Add: 0 0 * * * certbot renew --quiet + ``` + +### Caddy with Custom Domain + +1. Edit `Caddyfile` and uncomment the production section + +2. Replace `pai.example.com` with your actual domain + +3. Ensure DNS A/AAAA records point to your server + +4. Reload Caddy: + + ```sh + sudo systemctl reload caddy # Linux + brew services restart caddy # macOS + ``` + +Caddy automatically obtains and renews SSL certificates from Let's Encrypt - no additional configuration needed! + +## Running pai as a System Service + +### macOS (launchd) + +Create `/Library/LaunchDaemons/com.pai.server.plist`: + +```xml + + + + + Label + com.pai.server + ProgramArguments + + /usr/local/bin/pai + serve + -a + 127.0.0.1:8080 + -d + /var/lib/pai/pai.db + + RunAtLoad + + KeepAlive + + StandardOutPath + /var/log/pai/stdout.log + StandardErrorPath + /var/log/pai/stderr.log + WorkingDirectory + /var/lib/pai + + +``` + +Load the service: + +```sh +sudo launchctl load /Library/LaunchDaemons/com.pai.server.plist +``` + +### Linux (systemd) + +Create `/etc/systemd/system/pai.service`: + +```ini +[Unit] +Description=Personal Activity Index +After=network.target + +[Service] +Type=simple +ExecStart=/usr/local/bin/pai serve -a 127.0.0.1:8080 -d /var/lib/pai/pai.db +Restart=on-failure +RestartSec=5 +User=pai +Group=pai +WorkingDirectory=/var/lib/pai + +[Install] +WantedBy=multi-user.target +``` + +Create the pai user and directories: + +```sh +sudo useradd -r -s /bin/false pai +sudo mkdir -p /var/lib/pai +sudo chown pai:pai /var/lib/pai +``` + +Enable and start the service: + +```sh +sudo systemctl daemon-reload +sudo systemctl enable pai +sudo systemctl start pai +``` + +Check status: + +```sh +sudo systemctl status pai +``` + +View logs: + +```sh +sudo journalctl -u pai -f +``` + +## Testing + +Verify the proxy is working: + +```sh +# Health check +curl http://localhost/status + +# API endpoint +curl http://localhost/api/feed?limit=5 + +# Specific item +curl http://localhost/api/item/some-item-id +``` + +## Additional Resources + +- [nginx documentation](https://nginx.org/en/docs/) +- [Caddy documentation](https://caddyserver.com/docs/) +- [Let's Encrypt](https://letsencrypt.org/) +- [Personal Activity Index main documentation](../README.md) +- [Deployment guide](../DEPLOYMENT.md) diff --git a/conf/nginx.conf b/conf/nginx.conf new file mode 100644 index 0000000..71349ef --- /dev/null +++ b/conf/nginx.conf @@ -0,0 +1,30 @@ +# nginx configuration for Personal Activity Index +# This file provides a basic reverse proxy setup for pai serve + +server { + listen 80; + server_name localhost; + + # For a custom domain, replace localhost with your domain and add SSL configuration: + # listen 443 ssl http2; + # ssl_certificate /path/to/cert.pem; + # ssl_certificate_key /path/to/key.pem; + + location / { + proxy_pass http://127.0.0.1:8080; + proxy_set_header Host $host; + proxy_set_header X-Real-IP $remote_addr; + proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; + proxy_set_header X-Forwarded-Proto $scheme; + + # Optional: Add response headers + add_header X-Frame-Options "SAMEORIGIN" always; + add_header X-Content-Type-Options "nosniff" always; + } + + # Optional: Health check endpoint for load balancers + location /healthz { + proxy_pass http://127.0.0.1:8080/status; + access_log off; + } +} diff --git a/worker/Cargo.toml b/worker/Cargo.toml index 9613f59..47ea11a 100644 --- a/worker/Cargo.toml +++ b/worker/Cargo.toml @@ -1,6 +1,19 @@ [package] name = "pai-worker" version = "0.1.0" -edition = "2024" +edition = "2021" + +[lib] +crate-type = ["cdylib"] [dependencies] +pai-core = { path = "../core" } +worker = { version = "0.6", features = ["d1"] } +serde = { version = "1.0", features = ["derive"] } +serde_json = "1.0" +serde_urlencoded = "0.7" +chrono = { version = "0.4", default-features = false, features = [ + "serde", + "wasmbind", +] } +rss = { version = "2.0", default-features = false } diff --git a/worker/schema.sql b/worker/schema.sql new file mode 100644 index 0000000..933f947 --- /dev/null +++ b/worker/schema.sql @@ -0,0 +1,18 @@ +-- Personal Activity Index D1 Schema +-- This schema is compatible with both SQLite (CLI) and D1 (Worker) + +CREATE TABLE IF NOT EXISTS items ( + id TEXT PRIMARY KEY, + source_kind TEXT NOT NULL, + source_id TEXT NOT NULL, + author TEXT, + title TEXT, + summary TEXT, + url TEXT NOT NULL, + content_html TEXT, + published_at TEXT NOT NULL, + created_at TEXT NOT NULL DEFAULT CURRENT_TIMESTAMP +); + +CREATE INDEX IF NOT EXISTS idx_items_source_date ON items (source_kind, source_id, published_at DESC); +CREATE INDEX IF NOT EXISTS idx_items_published ON items (published_at DESC); diff --git a/worker/src/lib.rs b/worker/src/lib.rs index b93cf3f..4691297 100644 --- a/worker/src/lib.rs +++ b/worker/src/lib.rs @@ -1,5 +1,425 @@ -pub fn add(left: u64, right: u64) -> u64 { - left + right +use pai_core::{Item, ListFilter, SourceKind}; +use serde::{Deserialize, Serialize}; +use wasm_bindgen::JsValue; +use worker::*; + +#[derive(Deserialize)] +struct SyncConfig { + substack: Option, + bluesky: Option, + leaflet: Vec, +} + +#[derive(Deserialize)] +struct SubstackConfig { + base_url: String, +} + +#[derive(Deserialize)] +struct BlueskyConfig { + handle: String, +} + +#[derive(Deserialize)] +struct LeafletConfig { + id: String, + base_url: String, +} + +#[derive(Deserialize)] +struct FeedParams { + source_kind: Option, + source_id: Option, + limit: Option, + since: Option, + q: Option, +} + +#[derive(Serialize)] +struct FeedResponse { + items: Vec, +} + +#[derive(Serialize)] +struct StatusResponse { + status: &'static str, + version: &'static str, +} + +#[event(fetch)] +async fn fetch(req: Request, env: Env, _ctx: Context) -> Result { + let router = Router::new(); + router + .get_async("/api/feed", |req, ctx| async move { handle_feed(req, ctx).await }) + .get_async("/api/item/:id", |_req, ctx| async move { + let id = ctx + .param("id") + .ok_or_else(|| Error::RustError("Missing id parameter".into()))?; + handle_item(id, &ctx).await + }) + .get("/status", |_req, _ctx| { + let version = env!("CARGO_PKG_VERSION"); + let status = StatusResponse { status: "ok", version }; + Response::from_json(&status) + }) + .run(req, env) + .await +} + +#[event(scheduled)] +async fn scheduled(_event: ScheduledEvent, env: Env, _ctx: ScheduleContext) { + if let Err(e) = run_sync(&env).await { + console_error!("Scheduled sync failed: {}", e); + } +} + +async fn handle_feed(req: Request, ctx: RouteContext<()>) -> Result { + let url = req.url()?; + let params: FeedParams = serde_urlencoded::from_str(url.query().unwrap_or("")) + .map_err(|e| Error::RustError(format!("Invalid query parameters: {e}")))?; + + let filter = ListFilter { + source_kind: params.source_kind, + source_id: params.source_id, + limit: Some(params.limit.unwrap_or(20)), + since: params.since, + query: params.q, + }; + + let db = ctx.env.d1("DB")?; + let items = query_items(&db, &filter).await?; + + let response = FeedResponse { items }; + Response::from_json(&response) +} + +async fn handle_item(id: &str, ctx: &RouteContext<()>) -> Result { + let db = ctx.env.d1("DB")?; + let stmt = db.prepare("SELECT * FROM items WHERE id = ?1").bind(&[id.into()])?; + + let result = stmt.first::(None).await?; + + match result { + Some(item) => Response::from_json(&item), + None => Response::error("Item not found", 404), + } +} + +async fn query_items(db: &D1Database, filter: &ListFilter) -> Result> { + let mut query = String::from( + "SELECT id, source_kind, source_id, author, title, summary, url, content_html, published_at, created_at FROM items WHERE 1=1" + ); + let mut bindings = vec![]; + + if let Some(kind) = filter.source_kind { + query.push_str(" AND source_kind = ?"); + bindings.push(kind.to_string().into()); + } + + if let Some(ref source_id) = filter.source_id { + query.push_str(" AND source_id = ?"); + bindings.push(source_id.clone().into()); + } + + if let Some(ref since) = filter.since { + query.push_str(" AND published_at >= ?"); + bindings.push(since.clone().into()); + } + + if let Some(ref q) = filter.query { + query.push_str(" AND (title LIKE ? OR summary LIKE ?)"); + let pattern = format!("%{q}%"); + bindings.push(pattern.clone().into()); + bindings.push(pattern.into()); + } + + query.push_str(" ORDER BY published_at DESC"); + + if let Some(limit) = filter.limit { + query.push_str(" LIMIT ?"); + bindings.push((limit as i64).into()); + } + + let mut stmt = db.prepare(&query); + for binding in bindings { + stmt = stmt.bind(&[binding])?; + } + + let results = stmt.all().await?; + let items: Vec = results.results()?; + + Ok(items) +} + +async fn run_sync(env: &Env) -> Result<()> { + let config = load_sync_config(env)?; + + let db = env.d1("DB")?; + let mut synced = 0; + + if let Some(substack_config) = config.substack { + match sync_substack(&substack_config, &db).await { + Ok(count) => { + console_log!("Synced {} items from Substack", count); + synced += count; + } + Err(e) => console_error!("Substack sync failed: {}", e), + } + } + + if let Some(bluesky_config) = config.bluesky { + match sync_bluesky(&bluesky_config, &db).await { + Ok(count) => { + console_log!("Synced {} items from Bluesky", count); + synced += count; + } + Err(e) => console_error!("Bluesky sync failed: {}", e), + } + } + + for leaflet_config in config.leaflet { + match sync_leaflet(&leaflet_config, &db).await { + Ok(count) => { + console_log!("Synced {} items from Leaflet ({})", count, leaflet_config.id); + synced += count; + } + Err(e) => console_error!("Leaflet sync failed for {}: {}", leaflet_config.id, e), + } + } + + console_log!("Sync completed: {} total items", synced); + Ok(()) +} + +fn load_sync_config(env: &Env) -> Result { + let substack = env + .var("SUBSTACK_URL") + .ok() + .map(|url| SubstackConfig { base_url: url.to_string() }); + + let bluesky = env + .var("BLUESKY_HANDLE") + .ok() + .map(|handle| BlueskyConfig { handle: handle.to_string() }); + + let leaflet = if let Ok(urls) = env.var("LEAFLET_URLS") { + urls.to_string() + .split(',') + .filter_map(|entry| { + let parts: Vec<&str> = entry.trim().splitn(2, ':').collect(); + if parts.len() == 2 { + Some(LeafletConfig { id: parts[0].to_string(), base_url: parts[1].to_string() }) + } else { + None + } + }) + .collect() + } else { + Vec::new() + }; + + Ok(SyncConfig { substack, bluesky, leaflet }) +} + +async fn sync_substack(config: &SubstackConfig, db: &D1Database) -> Result { + let feed_url = format!("{}/feed", config.base_url); + + let mut req = Request::new(&feed_url, Method::Get)?; + req.headers_mut()?.set("User-Agent", "pai-worker/0.1.0")?; + + let mut resp = Fetch::Request(req).send().await?; + let body = resp.text().await?; + + let channel = + rss::Channel::read_from(body.as_bytes()).map_err(|e| Error::RustError(format!("Failed to parse RSS: {e}")))?; + + let source_id = normalize_source_id(&config.base_url); + let mut count = 0; + + for item in channel.items() { + let id = item.guid().map(|g| g.value()).unwrap_or(item.link().unwrap_or("")); + let url = item.link().unwrap_or(id); + let title = item.title(); + let summary = item.description(); + let author = item.author(); + let content_html = item.content(); + + let published_at = item + .pub_date() + .and_then(|s| chrono::DateTime::parse_from_rfc2822(s).ok()) + .map(|dt| dt.to_rfc3339()) + .unwrap_or_else(|| chrono::Utc::now().to_rfc3339()); + + let created_at = chrono::Utc::now().to_rfc3339(); + + let stmt = db.prepare( + "INSERT OR REPLACE INTO items (id, source_kind, source_id, author, title, summary, url, content_html, published_at, created_at) + VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10)" + ); + + stmt.bind(&[ + id.into(), + "substack".into(), + source_id.clone().into(), + author.map(|s| s.into()).unwrap_or(JsValue::NULL), + title.map(|s| s.into()).unwrap_or(JsValue::NULL), + summary.map(|s| s.into()).unwrap_or(JsValue::NULL), + url.into(), + content_html.map(|s| s.into()).unwrap_or(JsValue::NULL), + published_at.into(), + created_at.into(), + ])? + .run() + .await?; + + count += 1; + } + + Ok(count) +} + +async fn sync_bluesky(config: &BlueskyConfig, db: &D1Database) -> Result { + let api_url = format!( + "https://public.api.bsky.app/xrpc/app.bsky.feed.getAuthorFeed?actor={}&limit=50", + config.handle + ); + + let mut req = Request::new(&api_url, Method::Get)?; + req.headers_mut()?.set("User-Agent", "pai-worker/0.1.0")?; + + let mut resp = Fetch::Request(req).send().await?; + let json: serde_json::Value = resp.json().await?; + + let feed = json["feed"] + .as_array() + .ok_or_else(|| Error::RustError("Invalid Bluesky response".into()))?; + + let mut count = 0; + + for item in feed { + let post = &item["post"]; + + if item.get("reason").is_some() { + continue; + } + + let uri = post["uri"] + .as_str() + .ok_or_else(|| Error::RustError("Missing URI".into()))?; + let record = &post["record"]; + let text = record["text"].as_str().unwrap_or(""); + + let post_id = uri.split('/').next_back().unwrap_or(""); + let url = format!("https://bsky.app/profile/{}/post/{}", config.handle, post_id); + + let title = if text.len() > 100 { format!("{}...", &text[..97]) } else { text.to_string() }; + + let published_at = record["createdAt"].as_str().unwrap_or("").to_string(); + let created_at = chrono::Utc::now().to_rfc3339(); + + let stmt = db.prepare( + "INSERT OR REPLACE INTO items (id, source_kind, source_id, author, title, summary, url, content_html, published_at, created_at) + VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10)" + ); + + stmt.bind(&[ + uri.into(), + "bluesky".into(), + config.handle.clone().into(), + config.handle.clone().into(), + title.into(), + text.into(), + url.into(), + JsValue::NULL, + published_at.into(), + created_at.into(), + ])? + .run() + .await?; + + count += 1; + } + + Ok(count) +} + +async fn sync_leaflet(config: &LeafletConfig, db: &D1Database) -> Result { + let host = normalize_source_id(&config.base_url); + let subdomain = host.split('.').next().unwrap_or(&host); + let did = format!("{subdomain}.bsky.social"); + + let api_url = format!( + "https://public.api.bsky.app/xrpc/com.atproto.repo.listRecords?repo={did}&collection=pub.leaflet.post&limit=50" + ); + + let mut req = Request::new(&api_url, Method::Get)?; + req.headers_mut()?.set("User-Agent", "pai-worker/0.1.0")?; + + let mut resp = Fetch::Request(req).send().await?; + let json: serde_json::Value = resp.json().await?; + + let records = json["records"] + .as_array() + .ok_or_else(|| Error::RustError("Invalid Leaflet response".into()))?; + + let mut count = 0; + + for record in records { + let uri = record["uri"] + .as_str() + .ok_or_else(|| Error::RustError("Missing URI".into()))?; + let value = &record["value"]; + + let title = value["title"].as_str().unwrap_or("Untitled"); + let summary = value["summary"].as_str().or(value["content"].as_str()).unwrap_or(""); + let slug = value["slug"].as_str().unwrap_or(""); + + let url = if !slug.is_empty() { + format!("{}/{}", config.base_url, slug) + } else { + format!("{}/post/{}", config.base_url, uri.split('/').next_back().unwrap_or("")) + }; + + let published_at = value["publishedAt"] + .as_str() + .or(value["createdAt"].as_str()) + .unwrap_or("") + .to_string(); + + let created_at = chrono::Utc::now().to_rfc3339(); + + let stmt = db.prepare( + "INSERT OR REPLACE INTO items (id, source_kind, source_id, author, title, summary, url, content_html, published_at, created_at) + VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10)" + ); + + stmt.bind(&[ + uri.into(), + "leaflet".into(), + config.id.clone().into(), + JsValue::NULL, + title.into(), + summary.into(), + url.into(), + JsValue::NULL, + published_at.into(), + created_at.into(), + ])? + .run() + .await?; + + count += 1; + } + + Ok(count) +} + +fn normalize_source_id(base_url: &str) -> String { + base_url + .trim_start_matches("https://") + .trim_start_matches("http://") + .trim_end_matches('/') + .to_string() } #[cfg(test)] @@ -7,8 +427,158 @@ mod tests { use super::*; #[test] - fn it_works() { - let result = add(2, 2); - assert_eq!(result, 4); + fn test_normalize_source_id_https() { + assert_eq!( + normalize_source_id("https://patternmatched.substack.com"), + "patternmatched.substack.com" + ); + } + + #[test] + fn test_normalize_source_id_http() { + assert_eq!(normalize_source_id("http://example.com/"), "example.com"); + } + + #[test] + fn test_normalize_source_id_trailing_slash() { + assert_eq!(normalize_source_id("https://test.leaflet.pub/"), "test.leaflet.pub"); + } + + #[test] + fn test_normalize_source_id_no_protocol() { + assert_eq!(normalize_source_id("example.com"), "example.com"); + } + + #[test] + fn test_bluesky_title_truncation_short() { + let text = "Short post"; + let title = if text.len() > 100 { format!("{}...", &text[..97]) } else { text.to_string() }; + assert_eq!(title, "Short post"); + } + + #[test] + fn test_bluesky_title_truncation_long() { + let text = "a".repeat(150); + let title = if text.len() > 100 { format!("{}...", &text[..97]) } else { text.to_string() }; + assert_eq!(title.len(), 100); + assert!(title.ends_with("...")); + } + + #[test] + fn test_bluesky_title_truncation_boundary() { + let text = "a".repeat(100); + let title = if text.len() > 100 { format!("{}...", &text[..97]) } else { text.to_string() }; + assert_eq!(title, text); + } + + #[test] + fn test_leaflet_url_with_slug() { + let base_url = "https://test.leaflet.pub"; + let slug = "my-post"; + let url = if !slug.is_empty() { + format!("{base_url}/{slug}") + } else { + format!("{}/post/{}", base_url, "fallback") + }; + assert_eq!(url, "https://test.leaflet.pub/my-post"); + } + + #[test] + fn test_leaflet_url_without_slug() { + let base_url = "https://test.leaflet.pub"; + let slug = ""; + let uri = "at://did:plc:abc123/pub.leaflet.post/xyz789"; + let post_id = uri.split('/').next_back().unwrap_or(""); + let url = if !slug.is_empty() { format!("{base_url}/{slug}") } else { format!("{base_url}/post/{post_id}") }; + assert_eq!(url, "https://test.leaflet.pub/post/xyz789"); + } + + #[test] + fn test_bluesky_post_id_extraction() { + let uri = "at://did:plc:abc123/app.bsky.feed.post/3ld7xyqnvqk2a"; + let post_id = uri.split('/').next_back().unwrap_or(""); + assert_eq!(post_id, "3ld7xyqnvqk2a"); + } + + #[test] + fn test_bluesky_url_construction() { + let handle = "desertthunder.dev"; + let post_id = "3ld7xyqnvqk2a"; + let url = format!("https://bsky.app/profile/{handle}/post/{post_id}"); + assert_eq!(url, "https://bsky.app/profile/desertthunder.dev/post/3ld7xyqnvqk2a"); + } + + #[test] + fn test_leaflet_config_parsing() { + let entry = "desertthunder:https://desertthunder.leaflet.pub"; + let parts: Vec<&str> = entry.trim().splitn(2, ':').collect(); + assert_eq!(parts.len(), 2); + assert_eq!(parts[0], "desertthunder"); + assert_eq!(parts[1], "https://desertthunder.leaflet.pub"); + } + + #[test] + fn test_leaflet_config_parsing_invalid() { + let entry = "invalid-entry-no-colon"; + let parts: Vec<&str> = entry.trim().splitn(2, ':').collect(); + assert_ne!(parts.len(), 2); + } + + #[test] + fn test_leaflet_config_parsing_multiple() { + let urls = "id1:https://pub1.leaflet.pub,id2:https://pub2.leaflet.pub"; + let configs: Vec<_> = urls + .split(',') + .filter_map(|entry| { + let parts: Vec<&str> = entry.trim().splitn(2, ':').collect(); + if parts.len() == 2 { + Some((parts[0].to_string(), parts[1].to_string())) + } else { + None + } + }) + .collect(); + + assert_eq!(configs.len(), 2); + assert_eq!(configs[0].0, "id1"); + assert_eq!(configs[0].1, "https://pub1.leaflet.pub"); + assert_eq!(configs[1].0, "id2"); + assert_eq!(configs[1].1, "https://pub2.leaflet.pub"); + } + + #[test] + fn test_substack_feed_url_construction() { + let base_url = "https://patternmatched.substack.com"; + let feed_url = format!("{base_url}/feed"); + assert_eq!(feed_url, "https://patternmatched.substack.com/feed"); + } + + #[test] + fn test_bluesky_api_url_construction() { + let handle = "desertthunder.dev"; + let api_url = format!("https://public.api.bsky.app/xrpc/app.bsky.feed.getAuthorFeed?actor={handle}&limit=50"); + assert_eq!( + api_url, + "https://public.api.bsky.app/xrpc/app.bsky.feed.getAuthorFeed?actor=desertthunder.dev&limit=50" + ); + } + + #[test] + fn test_leaflet_did_construction() { + let subdomain = "desertthunder"; + let did = format!("{subdomain}.bsky.social"); + assert_eq!(did, "desertthunder.bsky.social"); + } + + #[test] + fn test_leaflet_api_url_construction() { + let did = "desertthunder.bsky.social"; + let api_url = format!( + "https://public.api.bsky.app/xrpc/com.atproto.repo.listRecords?repo={did}&collection=pub.leaflet.post&limit=50" + ); + assert_eq!( + api_url, + "https://public.api.bsky.app/xrpc/com.atproto.repo.listRecords?repo=desertthunder.bsky.social&collection=pub.leaflet.post&limit=50" + ); } } diff --git a/worker/wrangler.example.toml b/worker/wrangler.example.toml new file mode 100644 index 0000000..a433ddc --- /dev/null +++ b/worker/wrangler.example.toml @@ -0,0 +1,36 @@ +# Cloudflare Workers configuration for Personal Activity Index +# Copy this file to wrangler.toml and update with your values + +name = "personal-activity-index" +main = "build/worker/index.js" +compatibility_date = "2025-01-15" + +# D1 Database Binding +# Create your D1 database with: +# wrangler d1 create personal-activity-db +# Then replace {DATABASE_ID} below with the ID from the output +[[d1_databases]] +binding = "DB" +database_name = "personal-activity-db" +database_id = "{DATABASE_ID}" + +# Cron Triggers for scheduled syncs +# Runs every hour at minute 0 +[triggers] +crons = ["0 * * * *"] + +# Environment variables for source configuration +# Configure the sources you want to sync +[vars] +# Substack RSS feed URL +SUBSTACK_URL = "https://patternmatched.substack.com" + +# Bluesky handle +BLUESKY_HANDLE = "desertthunder.dev" + +# Leaflet publications (comma-separated id:url pairs) +# Format: "id1:https://pub1.leaflet.pub,id2:https://pub2.leaflet.pub" +LEAFLET_URLS = "desertthunder:https://desertthunder.leaflet.pub,stormlightlabs:https://stormlightlabs.leaflet.pub" + +# Optional: Logging level +# LOG_LEVEL = "info"