diff --git a/migrations/00000000000000_diesel_initial_setup/down.sql b/migrations/00000000000000_diesel_initial_setup/down.sql new file mode 100644 index 00000000..a9f52609 --- /dev/null +++ b/migrations/00000000000000_diesel_initial_setup/down.sql @@ -0,0 +1,6 @@ +-- This file was automatically created by Diesel to setup helper functions +-- and other internal bookkeeping. This file is safe to edit, any future +-- changes will be added to existing projects as new migrations. + +DROP FUNCTION IF EXISTS diesel_manage_updated_at(_tbl regclass); +DROP FUNCTION IF EXISTS diesel_set_updated_at(); diff --git a/migrations/00000000000000_diesel_initial_setup/up.sql b/migrations/00000000000000_diesel_initial_setup/up.sql new file mode 100644 index 00000000..d68895b1 --- /dev/null +++ b/migrations/00000000000000_diesel_initial_setup/up.sql @@ -0,0 +1,36 @@ +-- This file was automatically created by Diesel to setup helper functions +-- and other internal bookkeeping. This file is safe to edit, any future +-- changes will be added to existing projects as new migrations. + + + + +-- Sets up a trigger for the given table to automatically set a column called +-- `updated_at` whenever the row is modified (unless `updated_at` was included +-- in the modified columns) +-- +-- # Example +-- +-- ```sql +-- CREATE TABLE users (id SERIAL PRIMARY KEY, updated_at TIMESTAMP NOT NULL DEFAULT NOW()); +-- +-- SELECT diesel_manage_updated_at('users'); +-- ``` +CREATE OR REPLACE FUNCTION diesel_manage_updated_at(_tbl regclass) RETURNS VOID AS $$ +BEGIN + EXECUTE format('CREATE TRIGGER set_updated_at BEFORE UPDATE ON %s + FOR EACH ROW EXECUTE PROCEDURE diesel_set_updated_at()', _tbl); +END; +$$ LANGUAGE plpgsql; + +CREATE OR REPLACE FUNCTION diesel_set_updated_at() RETURNS trigger AS $$ +BEGIN + IF ( + NEW IS DISTINCT FROM OLD AND + NEW.updated_at IS NOT DISTINCT FROM OLD.updated_at + ) THEN + NEW.updated_at := current_timestamp; + END IF; + RETURN NEW; +END; +$$ LANGUAGE plpgsql; diff --git a/migrations/2025-11-01-114838_actors/up.sql b/migrations/2025-11-01-114838_actors/up.sql index 2061e3e4..29d00559 100644 --- a/migrations/2025-11-01-114838_actors/up.sql +++ b/migrations/2025-11-01-114838_actors/up.sql @@ -14,6 +14,9 @@ CREATE TYPE actor_status AS ENUM ('active', 'takendown', 'suspended', 'deleted', 'deactivated'); CREATE TYPE actor_sync_state AS ENUM ('synced', 'dirty', 'partial', 'processing'); +COMMENT ON TYPE actor_status IS 'Actor account status: active (normal), takendown (moderation), suspended (temporary), deleted (user-initiated), deactivated (user-initiated temporary)'; +COMMENT ON TYPE actor_sync_state IS 'Repository sync state: synced (up-to-date), dirty (needs sync), partial (incomplete data), processing (sync in progress)'; + -- ============================================================================= -- PREFERENCE TYPES -- ============================================================================= @@ -21,12 +24,17 @@ CREATE TYPE actor_sync_state AS ENUM ('synced', 'dirty', 'partial', 'processing' CREATE TYPE chat_allow_incoming AS ENUM ('all', 'following', 'none'); CREATE TYPE status_type AS ENUM ('app.bsky.actor.status#live'); +COMMENT ON TYPE chat_allow_incoming IS 'Chat preference: who can send direct messages to this actor'; +COMMENT ON TYPE status_type IS 'Actor status type (currently only live streaming supported)'; + -- MIME Types (used in status records) CREATE TYPE image_mime_type AS ENUM ( 'image/jpeg', 'image/png', 'image/webp', 'image/gif', 'image/avif', 'image/svg+xml', 'image/bmp', 'image/tiff', 'image/heif', 'image/heic', 'image/jxl' ); +COMMENT ON TYPE image_mime_type IS 'Supported image MIME types for avatars, banners, and thumbnails'; + -- ============================================================================= -- CORE IDENTITY TABLES -- ============================================================================= @@ -43,12 +51,29 @@ CREATE TABLE actors ( account_created_at TIMESTAMPTZ ); +COMMENT ON TABLE actors IS 'Core identity table: one row per ATProto DID. Tracks handle, status, and repository sync state.'; +COMMENT ON COLUMN actors.id IS 'Synthetic primary key for internal references'; +COMMENT ON COLUMN actors.did IS 'Decentralized identifier (did:plc:* or did:web:*) - globally unique'; +COMMENT ON COLUMN actors.handle IS 'Human-readable handle (e.g., alice.bsky.social). Nullable for actors without handles.'; +COMMENT ON COLUMN actors.status IS 'Account status (active, takendown, suspended, deleted, deactivated)'; +COMMENT ON COLUMN actors.sync_state IS 'Repository sync state (synced, dirty, partial, processing)'; +COMMENT ON COLUMN actors.repo_rev IS 'Repository revision string (opaque token from PDS)'; +COMMENT ON COLUMN actors.repo_cid IS 'Repository commit CID (content identifier)'; +COMMENT ON COLUMN actors.last_indexed IS 'Timestamp of last successful indexing operation'; +COMMENT ON COLUMN actors.account_created_at IS 'Account creation timestamp from PDS'; + CREATE INDEX idx_actors_handle ON actors(handle) WHERE handle IS NOT NULL; CREATE INDEX idx_actors_sync_state ON actors(sync_state); CREATE INDEX idx_actors_status ON actors(status); CREATE INDEX idx_actors_last_indexed ON actors(last_indexed) WHERE last_indexed IS NOT NULL; CREATE INDEX idx_actors_account_created_at ON actors(account_created_at) WHERE account_created_at IS NOT NULL; +COMMENT ON INDEX idx_actors_handle IS 'Lookup actors by handle (partial index excluding NULL handles)'; +COMMENT ON INDEX idx_actors_sync_state IS 'Find actors needing sync (dirty/partial states)'; +COMMENT ON INDEX idx_actors_status IS 'Filter actors by status (e.g., exclude takendown)'; +COMMENT ON INDEX idx_actors_last_indexed IS 'Find stale actors for re-indexing'; +COMMENT ON INDEX idx_actors_account_created_at IS 'Query actors by signup date'; + -- ============================================================================= -- PROFILES @@ -62,12 +87,25 @@ CREATE TABLE profiles ( banner_cid BYTEA, display_name TEXT, description TEXT, - pinned_post_id BIGINT, -- FK added in posts migration + pinned_post_rkey BIGINT, -- FK added after posts migration (natural key) joined_sp_id BIGINT, -- FK added in feeds migration pronouns TEXT, website TEXT ); +COMMENT ON TABLE profiles IS 'User profiles (app.bsky.actor.profile): display name, bio, avatar, banner, etc. One profile per actor.'; +COMMENT ON COLUMN profiles.actor_id IS 'Reference to actors table (also primary key)'; +COMMENT ON COLUMN profiles.cid IS 'Content identifier (CID) of profile record'; +COMMENT ON COLUMN profiles.created_at IS 'Profile record creation timestamp'; +COMMENT ON COLUMN profiles.avatar_cid IS 'Avatar image CID (blob reference)'; +COMMENT ON COLUMN profiles.banner_cid IS 'Banner image CID (blob reference)'; +COMMENT ON COLUMN profiles.display_name IS 'User-facing display name (e.g., "Alice Johnson")'; +COMMENT ON COLUMN profiles.description IS 'Profile bio/description text'; +COMMENT ON COLUMN profiles.pinned_post_rkey IS 'Pinned post rkey (natural key). No FK constraint due to hypertable limitation.'; +COMMENT ON COLUMN profiles.joined_sp_id IS 'Joined via starter pack ID (FK added in feeds migration)'; +COMMENT ON COLUMN profiles.pronouns IS 'User pronouns (e.g., "she/her", "they/them")'; +COMMENT ON COLUMN profiles.website IS 'Personal website URL'; + -- Full-text search on profiles ALTER TABLE profiles ADD COLUMN search_vector tsvector GENERATED ALWAYS AS ( @@ -75,8 +113,12 @@ ALTER TABLE profiles ADD COLUMN search_vector tsvector setweight(to_tsvector('english', coalesce(description, '')), 'B') ) STORED; +COMMENT ON COLUMN profiles.search_vector IS 'Full-text search vector: display_name (weight A) + description (weight B)'; + CREATE INDEX idx_profiles_search ON profiles USING GIN(search_vector); +COMMENT ON INDEX idx_profiles_search IS 'Full-text search on profile display names and bios'; + -- ============================================================================= -- USER PREFERENCES -- ============================================================================= @@ -88,6 +130,11 @@ CREATE TABLE chat_decls ( created_at TIMESTAMPTZ NOT NULL DEFAULT NOW() ); +COMMENT ON TABLE chat_decls IS 'Chat preferences (app.bsky.actor.defs#chatDeclaration): who can send direct messages'; +COMMENT ON COLUMN chat_decls.actor_id IS 'Reference to actors table (also primary key)'; +COMMENT ON COLUMN chat_decls.allow_incoming IS 'Who can message this actor: all, following, or none'; +COMMENT ON COLUMN chat_decls.created_at IS 'Timestamp when chat preference was set'; + -- Status Records (Live streaming status) CREATE TABLE statuses ( actor_id INTEGER PRIMARY KEY REFERENCES actors(id) ON DELETE CASCADE, @@ -95,7 +142,19 @@ CREATE TABLE statuses ( created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(), status status_type NOT NULL, duration INTEGER, - embed_post_id BIGINT, -- FK added in posts migration + embed_post_actor_id INTEGER, -- Natural key references to posts (no FK - hypertable limitation) + embed_post_rkey BIGINT, thumb_mime_type image_mime_type, thumb_cid BYTEA ); + +COMMENT ON TABLE statuses IS 'Actor status records (app.bsky.actor.status): live streaming status and metadata'; +COMMENT ON COLUMN statuses.actor_id IS 'Reference to actors table (also primary key)'; +COMMENT ON COLUMN statuses.cid IS 'Content identifier (CID) of status record'; +COMMENT ON COLUMN statuses.created_at IS 'Status record creation timestamp'; +COMMENT ON COLUMN statuses.status IS 'Status type (currently only app.bsky.actor.status#live)'; +COMMENT ON COLUMN statuses.duration IS 'Stream duration in seconds (nullable for ongoing streams)'; +COMMENT ON COLUMN statuses.embed_post_actor_id IS 'Referenced post actor_id (no FK due to hypertable limitation)'; +COMMENT ON COLUMN statuses.embed_post_rkey IS 'Referenced post rkey (natural key part 2)'; +COMMENT ON COLUMN statuses.thumb_mime_type IS 'Thumbnail image MIME type'; +COMMENT ON COLUMN statuses.thumb_cid IS 'Thumbnail image CID (blob reference)'; diff --git a/migrations/2025-11-01-114839_posts/up.sql b/migrations/2025-11-01-114839_posts/up.sql index ff73017d..6f4d4692 100644 --- a/migrations/2025-11-01-114839_posts/up.sql +++ b/migrations/2025-11-01-114839_posts/up.sql @@ -2,19 +2,35 @@ -- POSTS & CONTENT -- ============================================================================= -- --- Posts, embeds (images, video, external links, record quotes), facets --- (mentions, links, tags), and URI deduplication +-- Posts, denormalized embeds/facets, and URI deduplication -- -- ============================================================================= +-- ============================================================================= +-- ENABLE TIMESCALEDB +-- ============================================================================= +-- TimescaleDB must be enabled before creating hypertables +CREATE EXTENSION IF NOT EXISTS timescaledb CASCADE; + -- ============================================================================= -- POST & EMBED TYPES -- ============================================================================= CREATE TYPE post_status AS ENUM ('complete', 'stub', 'missing', 'deleted', 'forbidden', 'pruned'); CREATE TYPE embed_type AS ENUM ('images', 'video', 'external', 'record', 'record_with_media'); -CREATE TYPE video_mime_type AS ENUM ('video/mp4', 'video/webm', 'video/quicktime', 'video/mpeg'); -CREATE TYPE caption_mime_type AS ENUM ('text/vtt'); + +-- Video mime types (GIFs are treated as video embeds in Bluesky) +CREATE TYPE video_mime_type AS ENUM ( + 'video/mp4', 'video/webm', 'video/quicktime', 'video/mpeg', + 'image/gif', 'video/x-m4v', 'video/3gpp', + 'video/x-msvideo', 'video/x-matroska', 'video/ogg' +); + +-- Caption mime types (subtitles) +CREATE TYPE caption_mime_type AS ENUM ( + 'text/vtt', 'text/srt', 'application/x-subrip', + 'text/x-ssa', 'text/x-ass' +); -- Language Codes (ISO 639-1 + regional variants) -- Includes all language codes from historical database (4.6M posts) @@ -47,9 +63,8 @@ CREATE TYPE language_code AS ENUM ( CREATE TYPE facet_type AS ENUM ('mention', 'link', 'tag'); --- ============================================================================= --- POSTS --- ============================================================================= +COMMENT ON TYPE language_code IS 'ISO 639-1 language codes (+ regional variants) for post content. Covers 99.99%+ of real-world usage on Bluesky.'; +COMMENT ON TYPE facet_type IS 'Post facet types: mention (@handle), link (URL), tag (#hashtag)'; -- ============================================================================= -- TID HELPER FUNCTIONS @@ -113,143 +128,210 @@ $$ LANGUAGE plpgsql IMMUTABLE STRICT; COMMENT ON FUNCTION i64_to_tid(BIGINT) IS 'Encodes a BIGINT to a base32-sortable TID string (13 characters). Uses AT Protocol TID encoding: https://atproto.com/specs/tid'; +-- ============================================================================= +-- DENORMALIZED EMBED COMPOSITE TYPES +-- ============================================================================= + +-- External link embed (used by 0.8% of posts) +CREATE TYPE post_ext_embed AS ( + uri text, + title text, + description text, + thumb_mime_type image_mime_type, + thumb_cid bytea +); + +COMMENT ON TYPE post_ext_embed IS 'External link embed (0.8% of posts): URL, title, description, thumbnail'; + +-- Video caption (subtitles) +CREATE TYPE post_video_caption AS ( + lang language_code, + mime_type caption_mime_type, + cid bytea +); + +COMMENT ON TYPE post_video_caption IS 'Video caption track: language, MIME type, blob CID'; + +-- Video embed (used by 0.16% of posts) +CREATE TYPE post_video_embed AS ( + mime_type video_mime_type, + cid bytea, + alt text, + width integer, + height integer, + caption_1 post_video_caption, -- First caption track (e.g., English) + caption_2 post_video_caption, -- Second caption track (e.g., Spanish) + caption_3 post_video_caption -- Third caption track (e.g., French) +); + +COMMENT ON TYPE post_video_embed IS 'Video embed (0.16% of posts): metadata + up to 3 caption tracks'; + +-- Image embed (used by 1.8% of posts, max 4 per Bluesky protocol) +CREATE TYPE post_image_embed AS ( + mime_type image_mime_type, + cid bytea, + alt text, + width integer, + height integer +); + +COMMENT ON TYPE post_image_embed IS 'Image embed metadata: MIME type, CID, alt text, dimensions. Max 4 per post.'; + +-- Facet embed (facet_type, index_start, index_end, link_uri, mention_actor_id, tag) +-- Used by 12.5% of posts, 99.3% have ≤8 facets +CREATE TYPE post_facet_embed AS ( + facet_type facet_type, + index_start integer, + index_end integer, + link_uri text, + mention_actor_id integer, + tag text +); + +COMMENT ON TYPE post_facet_embed IS 'Rich text facet: type (mention/link/tag) + position + metadata. 12.5% of posts use facets, 99.3% have ≤8.'; + +-- ============================================================================= +-- POSTS TABLE +-- ============================================================================= + CREATE TABLE posts ( - id BIGSERIAL PRIMARY KEY, actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, rkey INT8 NOT NULL, cid BYTEA NOT NULL, content BYTEA, langs language_code[] NOT NULL DEFAULT '{}', tags TEXT[] NOT NULL DEFAULT '{}', - parent_post_id BIGINT REFERENCES posts(id) ON DELETE SET NULL, - root_post_id BIGINT REFERENCES posts(id) ON DELETE SET NULL, + tokens TEXT[], -- Token-based search (replaces search_vector) + parent_post_actor_id INTEGER, + parent_post_rkey BIGINT, + root_post_actor_id INTEGER, + root_post_rkey BIGINT, embed_type embed_type, embed_subtype embed_type, violates_threadgate BOOLEAN NOT NULL DEFAULT FALSE, status post_status NOT NULL DEFAULT 'complete', - search_vector tsvector, - UNIQUE (actor_id, rkey), + -- Denormalized embeds (NULL columns compress to ~0 bytes in TimescaleDB) + ext_embed post_ext_embed, + video_embed post_video_embed, + embedded_post_actor_id INTEGER, + embedded_post_rkey BIGINT, + record_detached BOOLEAN, + image_1 post_image_embed, + image_2 post_image_embed, + image_3 post_image_embed, + image_4 post_image_embed, + facet_1 post_facet_embed, + facet_2 post_facet_embed, + facet_3 post_facet_embed, + facet_4 post_facet_embed, + facet_5 post_facet_embed, + facet_6 post_facet_embed, + facet_7 post_facet_embed, + facet_8 post_facet_embed, + mentions INTEGER[], -- Array of actor_ids (denormalized, max 8) + PRIMARY KEY (actor_id, rkey), CHECK (array_length(langs, 1) IS NULL OR array_length(langs, 1) <= 3) ); COMMENT ON COLUMN posts.content IS 'Zstd-compressed post content (BYTEA). NULL for stub/deleted posts. Dictionary version stored as first byte prefix.'; COMMENT ON COLUMN posts.langs IS 'Languages of post content (ISO 639-1 codes). Max 3 per AT Protocol spec. Empty array means no language specified.'; -COMMENT ON COLUMN posts.search_vector IS 'Full-text search vector (NULL except for allowlisted users)'; +COMMENT ON COLUMN posts.tokens IS 'Token-based search tokens (replaces tsvector). NULL for non-indexed posts.'; COMMENT ON COLUMN posts.status IS 'Post lifecycle state: complete (normal), stub (placeholder, needs fetch), missing (permanently unfetchable), deleted (soft-deleted), forbidden (access denied)'; +COMMENT ON COLUMN posts.embedded_post_actor_id IS 'Quote post reference - actor part'; +COMMENT ON COLUMN posts.embedded_post_rkey IS 'Quote post reference - rkey part'; +COMMENT ON COLUMN posts.record_detached IS 'Whether the quoted post is detached (postgate)'; CREATE INDEX idx_posts_actor ON posts(actor_id); -CREATE INDEX idx_posts_parent ON posts(parent_post_id) WHERE parent_post_id IS NOT NULL; -CREATE INDEX idx_posts_root ON posts(root_post_id) WHERE root_post_id IS NOT NULL; +CREATE INDEX idx_posts_parent + ON posts(parent_post_actor_id, parent_post_rkey) + WHERE parent_post_actor_id IS NOT NULL; +CREATE INDEX idx_posts_root + ON posts(root_post_actor_id, root_post_rkey) + WHERE root_post_actor_id IS NOT NULL; CREATE INDEX idx_posts_langs ON posts USING GIN(langs); CREATE INDEX idx_posts_tags ON posts USING GIN(tags); CREATE INDEX idx_posts_status ON posts(status) WHERE status != 'complete'; -CREATE INDEX idx_posts_search_vector ON posts USING GIN (search_vector) WHERE search_vector IS NOT NULL; CREATE INDEX idx_posts_rkey_prunable ON posts (rkey) WHERE status NOT IN ('forbidden', 'pruned'); - --- Add foreign key constraints to profiles and statuses -ALTER TABLE profiles ADD CONSTRAINT fk_profiles_pinned_post - FOREIGN KEY (pinned_post_id) REFERENCES posts(id) ON DELETE SET NULL; - -ALTER TABLE statuses ADD CONSTRAINT fk_statuses_embed_post - FOREIGN KEY (embed_post_id) REFERENCES posts(id) ON DELETE SET NULL; - --- Add indexes on foreign keys for improved DELETE performance -CREATE INDEX idx_profiles_pinned_post ON profiles(pinned_post_id) WHERE pinned_post_id IS NOT NULL; +CREATE INDEX idx_posts_tokens_gin ON posts USING GIN (tokens) WHERE tokens IS NOT NULL; +CREATE INDEX idx_posts_embedded_post + ON posts(embedded_post_actor_id, embedded_post_rkey) + WHERE embedded_post_actor_id IS NOT NULL; +CREATE INDEX idx_posts_mentions ON posts USING GIN(mentions) WHERE mentions IS NOT NULL; + +COMMENT ON TABLE posts IS 'Posts (app.bsky.feed.post): text, embeds, facets, reply metadata. TimescaleDB hypertable with 1-day chunks.'; +COMMENT ON INDEX idx_posts_actor IS 'Lookup all posts by actor (author timeline)'; +COMMENT ON INDEX idx_posts_parent IS 'Find replies to a specific post (partial: only WHERE parent exists)'; +COMMENT ON INDEX idx_posts_root IS 'Find all posts in a thread (partial: only WHERE root exists)'; +COMMENT ON INDEX idx_posts_langs IS 'Filter posts by language (GIN index supports array overlap queries)'; +COMMENT ON INDEX idx_posts_tags IS 'Find posts with specific tags/hashtags (GIN index)'; +COMMENT ON INDEX idx_posts_status IS 'Find non-complete posts (stubs, deleted, etc.)'; +COMMENT ON INDEX idx_posts_rkey_prunable IS 'Find posts eligible for pruning (excludes forbidden/pruned)'; +COMMENT ON INDEX idx_posts_tokens_gin IS 'Full-text search on post tokens (GIN index, partial: only indexed posts)'; +COMMENT ON INDEX idx_posts_embedded_post IS 'Find quote posts referencing a specific post'; +COMMENT ON INDEX idx_posts_mentions IS 'Find posts mentioning a specific actor (GIN index on denormalized actor_id array)'; -- ============================================================================= --- POST EMBEDS +-- Convert posts to TimescaleDB Hypertable +-- ============================================================================= +-- +-- Configuration: +-- - All posts (complete + stubs) +-- - NO retention policy: Full retention for network scale (2B posts) +-- - Chunk interval: 1 day +-- - Compression: 12 hours (aggressive) +-- - Expected compression: ~100:1 combined +-- +-- Storage projections at 2B posts with 10% complete, 90% stubs: +-- - Total: ~90 GB (vs ~540 GB traditional, 83% savings) +-- +-- NOTE: TimescaleDB does not support foreign keys TO hypertables, so we do NOT +-- create FK constraints pointing to posts table. Referential integrity is +-- enforced at the application level. -- ============================================================================= --- Post Embeds: Images -CREATE TABLE post_embed_images ( - post_id BIGINT NOT NULL REFERENCES posts(id) ON DELETE CASCADE, - seq SMALLINT NOT NULL, - mime_type image_mime_type NOT NULL, - cid BYTEA NOT NULL, - alt TEXT, - width INTEGER, - height INTEGER, - PRIMARY KEY (post_id, seq) -); - --- Post Embeds: Video -CREATE TABLE post_embed_video ( - post_id BIGINT PRIMARY KEY REFERENCES posts(id) ON DELETE CASCADE, - mime_type video_mime_type NOT NULL, - cid BYTEA NOT NULL, - alt TEXT, - width INTEGER, - height INTEGER -); - --- Post Embeds: Video Captions -CREATE TABLE post_embed_video_captions ( - post_id BIGINT NOT NULL REFERENCES posts(id) ON DELETE CASCADE, - language language_code NOT NULL, - mime_type caption_mime_type NOT NULL, - cid BYTEA NOT NULL, - PRIMARY KEY (post_id, language) -); - --- Post Embeds: External Links -CREATE TABLE post_embed_ext ( - post_id BIGINT PRIMARY KEY REFERENCES posts(id) ON DELETE CASCADE, - uri TEXT NOT NULL, - title TEXT NOT NULL, - description TEXT NOT NULL, - thumb_mime_type image_mime_type, - thumb_cid BYTEA -); - --- Post Embeds: Record Embeds (quote posts) -CREATE TABLE post_embed_record ( - post_id BIGINT PRIMARY KEY REFERENCES posts(id) ON DELETE CASCADE, - embedded_post_id BIGINT NOT NULL REFERENCES posts(id) ON DELETE CASCADE, - detached BOOLEAN NOT NULL DEFAULT FALSE +-- Convert to hypertable using rkey column (time-based partitioning via tid_timestamp) +SELECT create_hypertable( + 'posts', + 'rkey', + chunk_time_interval => (86400000000::bigint * 1000), -- 1 day in microseconds + time_partitioning_func => 'tid_timestamp', + migrate_data => true, + if_not_exists => true ); -CREATE INDEX idx_post_embed_record_embedded ON post_embed_record(embedded_post_id); - --- ============================================================================= --- POST FACETS & MENTIONS --- ============================================================================= +COMMENT ON TABLE posts IS 'TimescaleDB hypertable for posts (complete + stubs). Partitioned by rkey with 1-day chunks. Auto-compressed after 12 hours. No retention policy (full network scale).'; --- Post Facets (mentions, links, tags) -CREATE TABLE post_facets ( - post_id BIGINT NOT NULL REFERENCES posts(id) ON DELETE CASCADE, - facet_index SMALLINT NOT NULL, - byte_start INTEGER NOT NULL, - byte_end INTEGER NOT NULL, - facet_type facet_type NOT NULL, - mention_actor_id INTEGER REFERENCES actors(id) ON DELETE CASCADE, - link_uri_id INTEGER, - tag TEXT, - PRIMARY KEY (post_id, facet_index) +-- Enable compression +ALTER TABLE posts SET ( + timescaledb.compress, + timescaledb.compress_segmentby = 'actor_id', + timescaledb.compress_orderby = 'rkey DESC' ); -CREATE INDEX idx_post_facets_mention ON post_facets(mention_actor_id) WHERE mention_actor_id IS NOT NULL; -CREATE INDEX idx_post_facets_link_uri ON post_facets(link_uri_id); - --- Post Mentions (denormalized for fast lookups) -CREATE TABLE post_mentions ( - post_id BIGINT NOT NULL REFERENCES posts(id) ON DELETE CASCADE, - mentioned_actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, - PRIMARY KEY (post_id, mentioned_actor_id) +-- Add aggressive compression policy (12 hours) +SELECT add_compression_policy( + 'posts', + compress_after => INTERVAL '12 hours', + if_not_exists => true ); -CREATE INDEX idx_post_mentions_actor ON post_mentions(mentioned_actor_id); +-- Enable chunk skipping for parent/root post queries +SELECT enable_chunk_skipping('posts', 'parent_post_actor_id'); +SELECT enable_chunk_skipping('posts', 'root_post_actor_id'); +SELECT enable_chunk_skipping('posts', 'actor_id'); -- ============================================================================= -- URI DEDUPLICATION -- ============================================================================= --- URI deduplication table +-- URI deduplication table (for facet link URIs) CREATE TABLE uris ( id SERIAL PRIMARY KEY, uri TEXT NOT NULL UNIQUE, created_at TIMESTAMPTZ NOT NULL DEFAULT NOW() ); --- Add FK from post_facets to uris -ALTER TABLE post_facets ADD CONSTRAINT fk_post_facets_uri - FOREIGN KEY (link_uri_id) REFERENCES uris(id) ON DELETE CASCADE; +COMMENT ON TABLE uris IS 'URI deduplication: stores unique URIs once, referenced by ID. Currently unused (facets denormalized).'; +COMMENT ON COLUMN uris.id IS 'Synthetic URI ID for deduplication'; +COMMENT ON COLUMN uris.uri IS 'Unique URI string'; +COMMENT ON COLUMN uris.created_at IS 'First seen timestamp'; diff --git a/migrations/2025-11-01-114840_social_graph/up.sql b/migrations/2025-11-01-114840_social_graph/up.sql index 0b8f7adc..e0ca8cc3 100644 --- a/migrations/2025-11-01-114840_social_graph/up.sql +++ b/migrations/2025-11-01-114840_social_graph/up.sql @@ -8,12 +8,10 @@ -- Follows CREATE TABLE follows ( - id BIGSERIAL PRIMARY KEY, actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, rkey INT8 NOT NULL, - cid BYTEA NOT NULL, subject_actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, - UNIQUE (actor_id, rkey), + PRIMARY KEY (actor_id, rkey), UNIQUE (actor_id, subject_actor_id) ); @@ -22,12 +20,10 @@ CREATE INDEX idx_follows_subject ON follows(subject_actor_id); -- Blocks CREATE TABLE blocks ( - id BIGSERIAL PRIMARY KEY, actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, rkey INT8 NOT NULL, - cid BYTEA NOT NULL, subject_actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, - UNIQUE (actor_id, rkey), + PRIMARY KEY (actor_id, rkey), UNIQUE (actor_id, subject_actor_id) ); diff --git a/migrations/2025-11-01-114841_engagement/up.sql b/migrations/2025-11-01-114841_engagement/up.sql index 53d3e77f..1cc26464 100644 --- a/migrations/2025-11-01-114841_engagement/up.sql +++ b/migrations/2025-11-01-114841_engagement/up.sql @@ -10,60 +10,186 @@ -- ENGAGEMENT TYPES -- ============================================================================= -CREATE TYPE like_subject_type AS ENUM ('post', 'feedgen', 'labeler'); CREATE TYPE bookmark_subject_type AS ENUM ('post', 'feedgen', 'list', 'labeler', 'starterpack'); +CREATE TYPE repost_status AS ENUM ('complete', 'stub'); + +COMMENT ON TYPE bookmark_subject_type IS 'Bookmark target types: post (most common), feedgen, list, labeler, starterpack'; +COMMENT ON TYPE repost_status IS 'Repost lifecycle state: complete (normal repost), stub (placeholder awaiting fetch)'; -- ============================================================================= --- LIKES +-- POST LIKES (99.98% of likes) - TimescaleDB Hypertable -- ============================================================================= --- Likes (99.98% target posts, 0.02% target feedgens/labelers) -CREATE TABLE likes ( - id BIGSERIAL PRIMARY KEY, +CREATE TABLE post_likes ( actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, rkey INT8 NOT NULL, - cid BYTEA NOT NULL, - subject_type like_subject_type NOT NULL DEFAULT 'post', - subject_id BIGINT, - via_post_id BIGINT REFERENCES posts(id) ON DELETE CASCADE, - UNIQUE (actor_id, rkey) + post_actor_id INTEGER NOT NULL, + post_rkey BIGINT NOT NULL, + via_repost_actor_id INTEGER, + via_repost_rkey BIGINT, + PRIMARY KEY (actor_id, rkey) +); + +CREATE INDEX idx_post_likes_post_actor_id_post_rkey + ON post_likes(post_actor_id, post_rkey); +CREATE INDEX idx_post_likes_rkey ON post_likes(rkey); +CREATE INDEX idx_post_likes_via_repost + ON post_likes(via_repost_actor_id, via_repost_rkey) + WHERE via_repost_actor_id IS NOT NULL; + +COMMENT ON INDEX idx_post_likes_post_actor_id_post_rkey IS 'Find all likes on a specific post'; +COMMENT ON INDEX idx_post_likes_rkey IS 'Find likes by timestamp (TID-based)'; +COMMENT ON INDEX idx_post_likes_via_repost IS 'Find likes that came via a specific repost (partial: only WHERE via_repost exists)'; + +COMMENT ON TABLE post_likes IS 'Likes on posts (99.98% of all likes). Uses natural keys for post references.'; +COMMENT ON COLUMN post_likes.actor_id IS 'Who liked the post (references actors)'; +COMMENT ON COLUMN post_likes.rkey IS 'Like record TID converted to INT8 (natural key part 2, timestamp-based)'; +COMMENT ON COLUMN post_likes.post_actor_id IS 'Liked post author (natural key part 1)'; +COMMENT ON COLUMN post_likes.post_rkey IS 'Liked post TID (natural key part 2)'; +COMMENT ON COLUMN post_likes.via_repost_actor_id IS 'If liked via repost, repost author (for notification context)'; +COMMENT ON COLUMN post_likes.via_repost_rkey IS 'If liked via repost, repost TID (for notification context)'; + +-- Convert post_likes to TimescaleDB hypertable partitioned by rkey (timestamp-based TID) +-- Use tid_timestamp function to extract timestamp from TID for partitioning +-- Chunk interval: 4 hours (optimal for network scale: 20M likes/day) +SELECT create_hypertable( + 'post_likes', + 'rkey', + chunk_time_interval => (14400000000::bigint * 1000), -- 4 hours in microseconds + time_partitioning_func => 'tid_timestamp', + migrate_data => true, + if_not_exists => true +); + +-- Enable compression on old chunks +-- compress_segmentby: Group by actor_id for better compression (likes from same actor) +-- compress_orderby: Order by rkey DESC for time-series queries +ALTER TABLE post_likes SET ( + timescaledb.compress, + timescaledb.compress_segmentby = 'actor_id', + timescaledb.compress_orderby = 'rkey DESC' ); -CREATE INDEX idx_likes_actor ON likes(actor_id); -CREATE INDEX idx_likes_subject ON likes(subject_type, subject_id); -CREATE INDEX idx_likes_rkey ON likes(rkey); -CREATE INDEX idx_likes_via_post ON likes(via_post_id); +-- Add aggressive compression policy: compress chunks older than 12 hours +-- This is safe because 99.9% of writes are <5 seconds old +SELECT add_compression_policy( + 'post_likes', + compress_after => INTERVAL '12 hours', + if_not_exists => true +); + +-- Enable chunk skipping on post_id for efficient query exclusion +-- Chunk skipping allows queries on specific posts to skip irrelevant chunks +SELECT enable_chunk_skipping('post_likes', 'post_actor_id'); + +-- Enable chunk skipping on actor_id for timeline queries +-- Allows viewer state queries to skip chunks without the viewer's likes +SELECT enable_chunk_skipping('post_likes', 'actor_id'); + +COMMENT ON TABLE post_likes IS 'TimescaleDB hypertable for post likes. Partitioned by rkey (timestamp) with 4-hour chunks. Auto-compressed after 12 hours. NO retention policy (full network scale). Chunk skipping enabled on post_actor_id and actor_id.'; + +-- ============================================================================= +-- FEEDGEN LIKES (rare, ~0.01% of likes) +-- ============================================================================= + +CREATE TABLE feedgen_likes ( + actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, + rkey INT8 NOT NULL, + feedgen_id BIGINT NOT NULL, -- FK to feedgens added in feeds migration + PRIMARY KEY (actor_id, rkey) +); + +CREATE INDEX idx_feedgen_likes_feedgen ON feedgen_likes(feedgen_id); +CREATE INDEX idx_feedgen_likes_rkey ON feedgen_likes(rkey); + +COMMENT ON TABLE feedgen_likes IS 'Likes on feed generators (rare, ~0.01% of likes).'; + +-- ============================================================================= +-- LABELER LIKES (rare, ~0.01% of likes) +-- ============================================================================= + +CREATE TABLE labeler_likes ( + actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, + rkey INT8 NOT NULL, + labeler_actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, + PRIMARY KEY (actor_id, rkey) +); + +CREATE INDEX idx_labeler_likes_labeler ON labeler_likes(labeler_actor_id); +CREATE INDEX idx_labeler_likes_rkey ON labeler_likes(rkey); + +COMMENT ON TABLE labeler_likes IS 'Likes on labeler services (rare, ~0.01% of likes).'; -- ============================================================================= -- REPOSTS -- ============================================================================= CREATE TABLE reposts ( - id BIGSERIAL PRIMARY KEY, actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, rkey INT8 NOT NULL, - cid BYTEA NOT NULL, - post_id BIGINT REFERENCES posts(id) ON DELETE CASCADE, - via_post_id BIGINT REFERENCES posts(id) ON DELETE CASCADE, - UNIQUE (actor_id, rkey) + post_actor_id INTEGER NOT NULL, + post_rkey BIGINT NOT NULL, + via_repost_actor_id INTEGER, + via_repost_rkey BIGINT, + status repost_status NOT NULL DEFAULT 'complete', + PRIMARY KEY (actor_id, rkey), + UNIQUE (actor_id, post_actor_id, post_rkey) ); -CREATE INDEX idx_reposts_actor ON reposts(actor_id); -CREATE INDEX idx_reposts_post ON reposts(post_id); +CREATE INDEX idx_reposts_post_actor_id_post_rkey + ON reposts(post_actor_id, post_rkey); CREATE INDEX idx_reposts_rkey ON reposts(rkey); -CREATE INDEX idx_reposts_via_post ON reposts(via_post_id); +CREATE INDEX idx_reposts_via_repost + ON reposts(via_repost_actor_id, via_repost_rkey) + WHERE via_repost_actor_id IS NOT NULL; +CREATE INDEX idx_reposts_status ON reposts(status) WHERE status != 'complete'; + +COMMENT ON TABLE reposts IS 'Reposts (quote-less shares): uses natural keys (actor_id, rkey), tracks via_repost for chain context'; +COMMENT ON COLUMN reposts.status IS 'Repost lifecycle state: complete (normal), stub (placeholder, needs fetch)'; +COMMENT ON COLUMN reposts.via_repost_actor_id IS 'If reposted via another repost, source repost author'; +COMMENT ON COLUMN reposts.via_repost_rkey IS 'If reposted via another repost, source repost TID'; + +COMMENT ON INDEX idx_reposts_post_actor_id_post_rkey IS 'Find all reposts of a specific post'; +COMMENT ON INDEX idx_reposts_rkey IS 'Find reposts by timestamp (TID-based)'; +COMMENT ON INDEX idx_reposts_via_repost IS 'Find reposts that came via another repost (partial: only WHERE via_repost exists)'; +COMMENT ON INDEX idx_reposts_status IS 'Find stub reposts needing fetch (partial: only non-complete)'; -- ============================================================================= --- BOOKMARKS +-- BOOKMARKS (off-protocol, posts only) -- ============================================================================= --- Bookmarks (off-protocol, supports multiple entity types) CREATE TABLE bookmarks ( actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, rkey INT8 NOT NULL, - subject_type bookmark_subject_type NOT NULL, - subject_id BIGINT, + post_actor_id INTEGER NOT NULL, + post_rkey BIGINT NOT NULL, PRIMARY KEY (actor_id, rkey) ); -CREATE INDEX idx_bookmarks_subject ON bookmarks(subject_type, subject_id) WHERE subject_id IS NOT NULL; +CREATE INDEX idx_bookmarks_post ON bookmarks(post_actor_id, post_rkey); + +-- Viewer-first composite index for efficient viewer state lookups +CREATE INDEX idx_bookmarks_viewer_post + ON bookmarks (actor_id, post_actor_id, post_rkey); + +COMMENT ON TABLE bookmarks IS 'Bookmarks (off-protocol, posts only): saved posts per user. Uses natural keys for post references.'; +COMMENT ON INDEX idx_bookmarks_post IS 'Find all bookmarks of a specific post'; +COMMENT ON INDEX idx_bookmarks_viewer_post IS 'Viewer state: check if viewer bookmarked a post (viewer-first composite index)'; + +-- ============================================================================= +-- VIEWER STATE OPTIMIZATIONS +-- ============================================================================= + +-- Viewer-first composite indexes for efficient viewer state lookups +-- Pattern: (viewer_actor_id, post_actor_id, post_rkey) +-- This allows PostgreSQL to seek directly to the viewer's records, then filter by post + +CREATE INDEX idx_post_likes_viewer_post + ON post_likes (actor_id, post_actor_id, post_rkey); + +COMMENT ON INDEX idx_post_likes_viewer_post IS 'Viewer state: check if viewer liked a post (viewer-first composite index)'; + +CREATE INDEX idx_reposts_viewer_post + ON reposts (actor_id, post_actor_id, post_rkey); + +COMMENT ON INDEX idx_reposts_viewer_post IS 'Viewer state: check if viewer reposted a post (viewer-first composite index)'; diff --git a/migrations/2025-11-01-114843_thread_moderation/up.sql b/migrations/2025-11-01-114843_thread_moderation/up.sql index e038714d..dcf50ebe 100644 --- a/migrations/2025-11-01-114843_thread_moderation/up.sql +++ b/migrations/2025-11-01-114843_thread_moderation/up.sql @@ -28,29 +28,30 @@ CREATE TYPE postgate_rule AS ENUM ( -- ============================================================================= CREATE TABLE threadgates ( - id BIGSERIAL PRIMARY KEY, actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, rkey INT8 NOT NULL, cid BYTEA NOT NULL, - post_id BIGINT REFERENCES posts(id) ON DELETE CASCADE, + post_actor_id INTEGER, + post_rkey BIGINT, allow threadgate_rule[], - UNIQUE (actor_id, rkey) + PRIMARY KEY (actor_id, rkey) ); -CREATE INDEX idx_threadgates_post ON threadgates(post_id); - -- Threadgate Hidden Replies CREATE TABLE threadgate_hidden_replies ( - threadgate_id BIGINT NOT NULL REFERENCES threadgates(id) ON DELETE CASCADE, - post_id BIGINT NOT NULL REFERENCES posts(id) ON DELETE CASCADE, - PRIMARY KEY (threadgate_id, post_id) + post_actor_id INTEGER NOT NULL, + post_rkey BIGINT NOT NULL, + hidden_post_actor_id INTEGER NOT NULL, + hidden_post_rkey BIGINT NOT NULL, + PRIMARY KEY (post_actor_id, post_rkey, hidden_post_actor_id, hidden_post_rkey) ); -- Threadgate Allowed Lists CREATE TABLE threadgate_allowed_lists ( - threadgate_id BIGINT NOT NULL REFERENCES threadgates(id) ON DELETE CASCADE, + post_actor_id INTEGER NOT NULL, + post_rkey BIGINT NOT NULL, list_id BIGINT NOT NULL, -- FK added in lists migration - PRIMARY KEY (threadgate_id, list_id) + PRIMARY KEY (post_actor_id, post_rkey, list_id) ); -- ============================================================================= @@ -58,51 +59,56 @@ CREATE TABLE threadgate_allowed_lists ( -- ============================================================================= CREATE TABLE postgates ( - id BIGSERIAL PRIMARY KEY, actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, rkey INT8 NOT NULL, cid BYTEA NOT NULL, - post_id BIGINT REFERENCES posts(id) ON DELETE CASCADE, + post_actor_id INTEGER, + post_rkey BIGINT, rules postgate_rule[] NOT NULL DEFAULT '{}', - UNIQUE (actor_id, rkey) + PRIMARY KEY (actor_id, rkey) ); -CREATE INDEX idx_postgates_post ON postgates(post_id); - -- Postgate Detached Embeds CREATE TABLE postgate_detached ( - postgate_id BIGINT NOT NULL REFERENCES postgates(id) ON DELETE CASCADE, - detached_post_id BIGINT NOT NULL REFERENCES posts(id) ON DELETE CASCADE, - PRIMARY KEY (postgate_id, detached_post_id) + post_actor_id INTEGER NOT NULL, + post_rkey BIGINT NOT NULL, + detached_post_actor_id INTEGER NOT NULL, + detached_post_rkey BIGINT NOT NULL, + PRIMARY KEY (post_actor_id, post_rkey, detached_post_actor_id, detached_post_rkey) ); -- ============================================================================= -- POSTGATE MAINTENANCE FUNCTION -- ============================================================================= --- Updates the 'detached' flag on existing post_embed_record entries when a postgate changes +-- Updates the 'detached' flag on posts.record_detached when a postgate changes CREATE OR REPLACE FUNCTION maintain_postgates( post_uri TEXT, detached_uris TEXT[], disable_effective TIMESTAMP ) RETURNS BIGINT AS $$ DECLARE - target_post_id BIGINT; + target_actor_id INTEGER; + target_rkey BIGINT; rows_updated BIGINT; BEGIN - -- Look up the post_id from the post URI - SELECT p.id INTO target_post_id - FROM posts p - INNER JOIN actors a ON p.actor_id = a.id - WHERE 'at://' || a.did || '/app.bsky.feed.post/' || i64_to_tid(p.rkey) = post_uri; - - IF target_post_id IS NULL THEN + -- Look up the post from the post URI + SELECT a.id, tid_to_i64(parts.rkey) + INTO target_actor_id, target_rkey + FROM ( + SELECT + SPLIT_PART(SUBSTRING(post_uri FROM 6), '/', 1) as did, + SPLIT_PART(SUBSTRING(post_uri FROM 6), '/', 3) as rkey + ) parts + INNER JOIN actors a ON a.did = parts.did; + + IF target_actor_id IS NULL THEN RETURN 0; END IF; - -- Update post_embed_record entries to mark quotes as detached - WITH detached_post_ids AS ( - SELECT p.id + -- Update posts.record_detached for quotes of the target post + WITH detached_posts AS ( + SELECT p.actor_id, p.rkey FROM UNNEST(detached_uris) AS uri CROSS JOIN LATERAL ( SELECT @@ -111,30 +117,22 @@ BEGIN ) parts INNER JOIN actors a ON a.did = parts.did INNER JOIN posts p ON p.actor_id = a.id AND p.rkey = tid_to_i64(parts.rkey) - ), - posts_to_update AS ( - SELECT per.post_id - FROM post_embed_record per - WHERE per.embedded_post_id = target_post_id ) - UPDATE post_embed_record per - SET detached = ( - per.post_id IN (SELECT id FROM detached_post_ids) + UPDATE posts + SET record_detached = ( + (actor_id, rkey) IN (SELECT actor_id, rkey FROM detached_posts) OR - (disable_effective IS NOT NULL AND ( - SELECT tid_timestamp(p.rkey) - FROM posts p - WHERE p.id = per.post_id - ) > disable_effective) + (disable_effective IS NOT NULL AND tid_timestamp(rkey) > disable_effective) ) - WHERE per.embedded_post_id = target_post_id; + WHERE embedded_post_actor_id = target_actor_id + AND embedded_post_rkey = target_rkey; GET DIAGNOSTICS rows_updated = ROW_COUNT; RETURN rows_updated; END; $$ LANGUAGE plpgsql; -COMMENT ON FUNCTION maintain_postgates IS 'Updates detached flag on post_embed_record when postgates change'; +COMMENT ON FUNCTION maintain_postgates IS 'Updates record_detached flag on posts table when postgates change'; -- ============================================================================= -- THREAD MUTES @@ -143,7 +141,8 @@ COMMENT ON FUNCTION maintain_postgates IS 'Updates detached flag on post_embed_r -- Thread Mutes (off-protocol) CREATE TABLE thread_mutes ( actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, - root_post_id BIGINT NOT NULL REFERENCES posts(id) ON DELETE CASCADE, + root_post_actor_id INTEGER NOT NULL, + root_post_rkey BIGINT NOT NULL, created_at TIMESTAMPTZ NOT NULL DEFAULT NOW(), - PRIMARY KEY (actor_id, root_post_id) + PRIMARY KEY (actor_id, root_post_actor_id, root_post_rkey) ); diff --git a/migrations/2025-11-01-114844_lists/up.sql b/migrations/2025-11-01-114844_lists/up.sql index 5265d5ae..b81728ec 100644 --- a/migrations/2025-11-01-114844_lists/up.sql +++ b/migrations/2025-11-01-114844_lists/up.sql @@ -49,13 +49,12 @@ CREATE INDEX idx_lists_type ON lists(list_type) WHERE list_type IS NOT NULL; -- ============================================================================= CREATE TABLE list_items ( - id BIGSERIAL PRIMARY KEY, actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, rkey INT8 NOT NULL, cid BYTEA NOT NULL, list_id BIGINT NOT NULL REFERENCES lists(id) ON DELETE CASCADE, subject_actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, - UNIQUE (actor_id, rkey) + PRIMARY KEY (actor_id, rkey) ); CREATE INDEX idx_list_items_list ON list_items(list_id); diff --git a/migrations/2025-11-01-114847_queues_and_extensions/up.sql b/migrations/2025-11-01-114847_queues_and_extensions/up.sql index d243cfa8..5725484c 100644 --- a/migrations/2025-11-01-114847_queues_and_extensions/up.sql +++ b/migrations/2025-11-01-114847_queues_and_extensions/up.sql @@ -35,11 +35,13 @@ CREATE EXTENSION IF NOT EXISTS postgis; -- - Multi-hop neighborhood queries (N-degree connections) CREATE EXTENSION IF NOT EXISTS pgrouting; --- Note: timescaledb and pgvector are available via flake.nix but not enabled yet --- --- timescaledb: Time-series optimization for notifications and stats tables --- Deferred until we design schema migration strategy (hypertables, partitioning) --- Future use: trending posts, notification retention policies, time-bucketing +-- TimescaleDB: Time-series optimization for hypertables +-- Provides efficient compression and querying for time-series data (posts, post_likes) +-- Used for: compressed storage, chunk-based partitioning, chunk skipping +-- NOTE: TimescaleDB extension is installed in the posts migration (runs earlier) +-- CREATE EXTENSION IF NOT EXISTS timescaledb CASCADE; + +-- Note: pgvector is available via flake.nix but not enabled yet -- -- pgvector: Vector similarity search for semantic/ML-based features -- Deferred until concrete use case emerges (embeddings, semantic search) @@ -147,3 +149,24 @@ CREATE INDEX idx_constellation_status_created WHERE status IN ('pending', 'processing'); COMMENT ON TABLE constellation_enrichment_queue IS 'Queue for asynchronous Constellation API enrichment'; + +-- ============================================================================= +-- DIESEL SCHEMA INFERENCE TABLE +-- ============================================================================= + +-- Dummy table to force Diesel to generate SQL type definitions +-- +-- This table is never used by the application, but exists solely to ensure +-- that Diesel generates type definitions for enums and composite types that are +-- only used inside other composite types. +-- +-- Without this table, Diesel wouldn't generate these types in schema.rs since +-- they're not directly used as column types in any real tables. + +CREATE TABLE _diesel_schema_inference ( + id INTEGER PRIMARY KEY, + video_mime_type video_mime_type NOT NULL, + caption_mime_type caption_mime_type NOT NULL, + facet_type facet_type NOT NULL, + video_caption post_video_caption NOT NULL +); diff --git a/migrations/2025-11-01-114848_aggregate_stats/up.sql b/migrations/2025-11-01-114848_aggregate_stats/up.sql index 34a11ea6..2230a828 100644 --- a/migrations/2025-11-01-114848_aggregate_stats/up.sql +++ b/migrations/2025-11-01-114848_aggregate_stats/up.sql @@ -35,14 +35,22 @@ CREATE TYPE actor_stat_type AS ENUM ( -- Post aggregate statistics -- Stores engagement metrics for posts with automatic cleanup CREATE TABLE post_aggregate_stats ( - post_id BIGINT NOT NULL REFERENCES posts(id) ON DELETE CASCADE, + actor_id INTEGER NOT NULL, + rkey BIGINT NOT NULL, stat_type post_stat_type NOT NULL, value INTEGER NOT NULL, updated_at TIMESTAMP WITH TIME ZONE NOT NULL DEFAULT now(), - PRIMARY KEY (post_id, stat_type) + PRIMARY KEY (actor_id, rkey, stat_type) ); +COMMENT ON TABLE post_aggregate_stats IS 'Post aggregate statistics using natural keys (actor_id, rkey)'; +COMMENT ON COLUMN post_aggregate_stats.actor_id IS 'Actor who created the post (natural key part 1)'; +COMMENT ON COLUMN post_aggregate_stats.rkey IS 'Record key as i64 (TID converted) - natural key part 2'; +COMMENT ON COLUMN post_aggregate_stats.stat_type IS 'Type of statistic (like, reply, repost, quote)'; +COMMENT ON COLUMN post_aggregate_stats.value IS 'Current count value'; +COMMENT ON COLUMN post_aggregate_stats.updated_at IS 'Last time value changed (NOT last write time)'; + -- Actor aggregate statistics -- Stores profile metrics for actors with automatic cleanup CREATE TABLE actor_aggregate_stats ( @@ -60,7 +68,13 @@ CREATE TABLE actor_aggregate_stats ( -- Index for finding hot posts (recently updated engagement) CREATE INDEX idx_post_aggregate_stats_updated_at - ON post_aggregate_stats (updated_at DESC); + ON post_aggregate_stats(updated_at DESC) + WHERE value > 0; + +-- Index for hot content queries (recently updated high-value stats) +CREATE INDEX idx_post_aggregate_stats_hot_content + ON post_aggregate_stats(stat_type, updated_at DESC, value DESC) + WHERE value > 10; -- Index for finding hot actors (recently updated profiles) CREATE INDEX idx_actor_aggregate_stats_updated_at diff --git a/migrations/2025-11-15-171545_add_gif_to_video_mime_type/down.sql b/migrations/2025-11-15-171545_add_gif_to_video_mime_type/down.sql deleted file mode 100644 index 8fad3299..00000000 --- a/migrations/2025-11-15-171545_add_gif_to_video_mime_type/down.sql +++ /dev/null @@ -1,11 +0,0 @@ --- Note: PostgreSQL does not support removing enum values without dropping and recreating the enum. --- Since these mime types are backward compatible (the old values still work), and removing them --- would require dropping columns/constraints/etc, we leave the enum values in place. --- --- If you absolutely need to revert, you would need to: --- 1. Drop all columns using these enum types --- 2. Drop the enum types --- 3. Recreate the enum types with original values --- 4. Recreate all columns --- --- This is not implemented as it would be destructive and the added values don't break anything. diff --git a/migrations/2025-11-15-171545_add_gif_to_video_mime_type/up.sql b/migrations/2025-11-15-171545_add_gif_to_video_mime_type/up.sql deleted file mode 100644 index 8f039833..00000000 --- a/migrations/2025-11-15-171545_add_gif_to_video_mime_type/up.sql +++ /dev/null @@ -1,20 +0,0 @@ --- Expand mime type enums to be more permissive and prevent future errors --- This adds commonly used formats that might appear in the wild - --- Video mime types (GIFs are treated as video embeds in Bluesky) -ALTER TYPE video_mime_type ADD VALUE IF NOT EXISTS 'image/gif'; -ALTER TYPE video_mime_type ADD VALUE IF NOT EXISTS 'video/x-m4v'; -ALTER TYPE video_mime_type ADD VALUE IF NOT EXISTS 'video/3gpp'; -ALTER TYPE video_mime_type ADD VALUE IF NOT EXISTS 'video/x-msvideo'; -ALTER TYPE video_mime_type ADD VALUE IF NOT EXISTS 'video/x-matroska'; -ALTER TYPE video_mime_type ADD VALUE IF NOT EXISTS 'video/ogg'; - --- Image mime types (add less common but valid formats) -ALTER TYPE image_mime_type ADD VALUE IF NOT EXISTS 'image/x-icon'; -ALTER TYPE image_mime_type ADD VALUE IF NOT EXISTS 'image/vnd.microsoft.icon'; - --- Caption mime types (add common subtitle formats) -ALTER TYPE caption_mime_type ADD VALUE IF NOT EXISTS 'text/srt'; -ALTER TYPE caption_mime_type ADD VALUE IF NOT EXISTS 'application/x-subrip'; -ALTER TYPE caption_mime_type ADD VALUE IF NOT EXISTS 'text/x-ssa'; -ALTER TYPE caption_mime_type ADD VALUE IF NOT EXISTS 'text/x-ass'; diff --git a/migrations/2025-11-15-203456_drop_likes_id_column/down.sql b/migrations/2025-11-15-203456_drop_likes_id_column/down.sql deleted file mode 100644 index cb68d77c..00000000 --- a/migrations/2025-11-15-203456_drop_likes_id_column/down.sql +++ /dev/null @@ -1,6 +0,0 @@ --- Rollback: restore BIGSERIAL id column as primary key --- Note: id values will not be preserved on rollback (new sequence starts) - -ALTER TABLE likes DROP CONSTRAINT likes_pkey; -ALTER TABLE likes ADD COLUMN id BIGSERIAL; -ALTER TABLE likes ADD PRIMARY KEY (id); diff --git a/migrations/2025-11-15-203456_drop_likes_id_column/up.sql b/migrations/2025-11-15-203456_drop_likes_id_column/up.sql deleted file mode 100644 index 934a42c4..00000000 --- a/migrations/2025-11-15-203456_drop_likes_id_column/up.sql +++ /dev/null @@ -1,7 +0,0 @@ --- Drop the BIGSERIAL id column and use composite primary key (actor_id, rkey) --- This saves 8 bytes per row + primary key index overhead --- Safe because: no FK references to likes(id) exist in the codebase - -ALTER TABLE likes DROP CONSTRAINT likes_pkey; -ALTER TABLE likes ADD PRIMARY KEY (actor_id, rkey); -ALTER TABLE likes DROP COLUMN id; diff --git a/migrations/2025-11-15-203642_drop_list_items_id_column/down.sql b/migrations/2025-11-15-203642_drop_list_items_id_column/down.sql deleted file mode 100644 index a85da302..00000000 --- a/migrations/2025-11-15-203642_drop_list_items_id_column/down.sql +++ /dev/null @@ -1,6 +0,0 @@ --- Rollback: restore BIGSERIAL id column as primary key --- Note: id values will not be preserved on rollback (new sequence starts) - -ALTER TABLE list_items DROP CONSTRAINT list_items_pkey; -ALTER TABLE list_items ADD COLUMN id BIGSERIAL; -ALTER TABLE list_items ADD PRIMARY KEY (id); diff --git a/migrations/2025-11-15-203642_drop_list_items_id_column/up.sql b/migrations/2025-11-15-203642_drop_list_items_id_column/up.sql deleted file mode 100644 index 40000446..00000000 --- a/migrations/2025-11-15-203642_drop_list_items_id_column/up.sql +++ /dev/null @@ -1,7 +0,0 @@ --- Drop the BIGSERIAL id column and use composite primary key (actor_id, rkey) --- This saves 8 bytes per row + primary key index overhead --- Safe because: no FK references to list_items(id) exist in the codebase - -ALTER TABLE list_items DROP CONSTRAINT list_items_pkey; -ALTER TABLE list_items ADD PRIMARY KEY (actor_id, rkey); -ALTER TABLE list_items DROP COLUMN id; diff --git a/migrations/2025-11-15-203734_drop_follows_id_column/down.sql b/migrations/2025-11-15-203734_drop_follows_id_column/down.sql deleted file mode 100644 index 94bbbe3e..00000000 --- a/migrations/2025-11-15-203734_drop_follows_id_column/down.sql +++ /dev/null @@ -1,6 +0,0 @@ --- Rollback: restore BIGSERIAL id column as primary key --- Note: id values will not be preserved on rollback (new sequence starts) - -ALTER TABLE follows DROP CONSTRAINT follows_pkey; -ALTER TABLE follows ADD COLUMN id BIGSERIAL; -ALTER TABLE follows ADD PRIMARY KEY (id); diff --git a/migrations/2025-11-15-203734_drop_follows_id_column/up.sql b/migrations/2025-11-15-203734_drop_follows_id_column/up.sql deleted file mode 100644 index 8d3770dd..00000000 --- a/migrations/2025-11-15-203734_drop_follows_id_column/up.sql +++ /dev/null @@ -1,7 +0,0 @@ --- Drop the BIGSERIAL id column and use composite primary key (actor_id, rkey) --- This saves 8 bytes per row + primary key index overhead --- Safe because: no FK references to follows(id) exist in the codebase - -ALTER TABLE follows DROP CONSTRAINT follows_pkey; -ALTER TABLE follows ADD PRIMARY KEY (actor_id, rkey); -ALTER TABLE follows DROP COLUMN id; diff --git a/migrations/2025-11-15-203834_drop_blocks_id_column/down.sql b/migrations/2025-11-15-203834_drop_blocks_id_column/down.sql deleted file mode 100644 index 49a0ad97..00000000 --- a/migrations/2025-11-15-203834_drop_blocks_id_column/down.sql +++ /dev/null @@ -1,6 +0,0 @@ --- Rollback: restore BIGSERIAL id column as primary key --- Note: id values will not be preserved on rollback (new sequence starts) - -ALTER TABLE blocks DROP CONSTRAINT blocks_pkey; -ALTER TABLE blocks ADD COLUMN id BIGSERIAL; -ALTER TABLE blocks ADD PRIMARY KEY (id); diff --git a/migrations/2025-11-15-203834_drop_blocks_id_column/up.sql b/migrations/2025-11-15-203834_drop_blocks_id_column/up.sql deleted file mode 100644 index 135c8e79..00000000 --- a/migrations/2025-11-15-203834_drop_blocks_id_column/up.sql +++ /dev/null @@ -1,7 +0,0 @@ --- Drop the BIGSERIAL id column and use composite primary key (actor_id, rkey) --- This saves 8 bytes per row + primary key index overhead --- Safe because: no FK references to blocks(id) exist in the codebase - -ALTER TABLE blocks DROP CONSTRAINT blocks_pkey; -ALTER TABLE blocks ADD PRIMARY KEY (actor_id, rkey); -ALTER TABLE blocks DROP COLUMN id; diff --git a/migrations/2025-11-15-205803_add_via_repost_id_to_likes/down.sql b/migrations/2025-11-15-205803_add_via_repost_id_to_likes/down.sql deleted file mode 100644 index a2cb7b65..00000000 --- a/migrations/2025-11-15-205803_add_via_repost_id_to_likes/down.sql +++ /dev/null @@ -1,4 +0,0 @@ --- Rollback: remove via_repost_id column from likes - -DROP INDEX IF EXISTS idx_likes_via_repost; -ALTER TABLE likes DROP COLUMN via_repost_id; diff --git a/migrations/2025-11-15-205803_add_via_repost_id_to_likes/up.sql b/migrations/2025-11-15-205803_add_via_repost_id_to_likes/up.sql deleted file mode 100644 index 366749e2..00000000 --- a/migrations/2025-11-15-205803_add_via_repost_id_to_likes/up.sql +++ /dev/null @@ -1,8 +0,0 @@ --- Add via_repost_id to likes table to properly track "liked via repost" context --- When a user likes a post through viewing a repost, this stores the repost record --- This enables notifications like "Alice liked your repost of Bob's post" - -ALTER TABLE likes ADD COLUMN via_repost_id BIGINT REFERENCES reposts(id) ON DELETE CASCADE; - --- Partial index only for rows that have via_repost_id set (sparse index) -CREATE INDEX idx_likes_via_repost ON likes(via_repost_id) WHERE via_repost_id IS NOT NULL; diff --git a/migrations/2025-11-15-205826_add_via_repost_id_to_reposts/down.sql b/migrations/2025-11-15-205826_add_via_repost_id_to_reposts/down.sql deleted file mode 100644 index 74dc89ba..00000000 --- a/migrations/2025-11-15-205826_add_via_repost_id_to_reposts/down.sql +++ /dev/null @@ -1,4 +0,0 @@ --- Rollback: remove via_repost_id column from reposts - -DROP INDEX IF EXISTS idx_reposts_via_repost; -ALTER TABLE reposts DROP COLUMN via_repost_id; diff --git a/migrations/2025-11-15-205826_add_via_repost_id_to_reposts/up.sql b/migrations/2025-11-15-205826_add_via_repost_id_to_reposts/up.sql deleted file mode 100644 index 6e1c11e9..00000000 --- a/migrations/2025-11-15-205826_add_via_repost_id_to_reposts/up.sql +++ /dev/null @@ -1,8 +0,0 @@ --- Add via_repost_id to reposts table to track "repost via repost" context --- When a user reposts a post through viewing another repost, this stores that context --- This enables tracking repost chains and proper notification attribution - -ALTER TABLE reposts ADD COLUMN via_repost_id BIGINT REFERENCES reposts(id) ON DELETE CASCADE; - --- Partial index only for rows that have via_repost_id set (sparse index) -CREATE INDEX idx_reposts_via_repost ON reposts(via_repost_id) WHERE via_repost_id IS NOT NULL; diff --git a/migrations/2025-11-15-210615_drop_via_post_id_from_likes/down.sql b/migrations/2025-11-15-210615_drop_via_post_id_from_likes/down.sql deleted file mode 100644 index 13cdc6a0..00000000 --- a/migrations/2025-11-15-210615_drop_via_post_id_from_likes/down.sql +++ /dev/null @@ -1,3 +0,0 @@ --- Rollback: restore via_post_id column (though it was never used) - -ALTER TABLE likes ADD COLUMN via_post_id BIGINT REFERENCES posts(id) ON DELETE CASCADE; diff --git a/migrations/2025-11-15-210615_drop_via_post_id_from_likes/up.sql b/migrations/2025-11-15-210615_drop_via_post_id_from_likes/up.sql deleted file mode 100644 index 6c7849ce..00000000 --- a/migrations/2025-11-15-210615_drop_via_post_id_from_likes/up.sql +++ /dev/null @@ -1,5 +0,0 @@ --- Drop legacy via_post_id column from likes table --- This column was never properly used - via_repost_id is the correct column --- for tracking "liked via repost" context - -ALTER TABLE likes DROP COLUMN via_post_id; diff --git a/migrations/2025-11-15-210633_drop_via_post_id_from_reposts/down.sql b/migrations/2025-11-15-210633_drop_via_post_id_from_reposts/down.sql deleted file mode 100644 index e53eef5f..00000000 --- a/migrations/2025-11-15-210633_drop_via_post_id_from_reposts/down.sql +++ /dev/null @@ -1,3 +0,0 @@ --- Rollback: restore via_post_id column (though it was never used) - -ALTER TABLE reposts ADD COLUMN via_post_id BIGINT REFERENCES posts(id) ON DELETE CASCADE; diff --git a/migrations/2025-11-15-210633_drop_via_post_id_from_reposts/up.sql b/migrations/2025-11-15-210633_drop_via_post_id_from_reposts/up.sql deleted file mode 100644 index 33e34105..00000000 --- a/migrations/2025-11-15-210633_drop_via_post_id_from_reposts/up.sql +++ /dev/null @@ -1,5 +0,0 @@ --- Drop legacy via_post_id column from reposts table --- This column was never properly used - via_repost_id is the correct column --- for tracking "repost via repost" context - -ALTER TABLE reposts DROP COLUMN via_post_id; diff --git a/migrations/2025-11-15-213700_add_repost_status/down.sql b/migrations/2025-11-15-213700_add_repost_status/down.sql deleted file mode 100644 index 58533084..00000000 --- a/migrations/2025-11-15-213700_add_repost_status/down.sql +++ /dev/null @@ -1,10 +0,0 @@ --- Rollback: Remove status column and repost_status enum from reposts table - --- Drop index -DROP INDEX IF EXISTS idx_reposts_status; - --- Drop status column -ALTER TABLE reposts DROP COLUMN status; - --- Drop enum type -DROP TYPE repost_status; diff --git a/migrations/2025-11-15-213700_add_repost_status/up.sql b/migrations/2025-11-15-213700_add_repost_status/up.sql deleted file mode 100644 index 04e4a81e..00000000 --- a/migrations/2025-11-15-213700_add_repost_status/up.sql +++ /dev/null @@ -1,19 +0,0 @@ --- Add status column to reposts table for stub tracking --- --- This enables repost stub creation for eventual consistency handling. --- When likes/reposts reference reposts that haven't arrived yet, we create --- stub reposts that can be upgraded later when the actual repost arrives. - --- Create repost_status enum type -CREATE TYPE repost_status AS ENUM ('complete', 'stub'); - -COMMENT ON TYPE repost_status IS 'Repost lifecycle state: complete (normal repost), stub (placeholder awaiting fetch)'; - --- Add status column to reposts table -ALTER TABLE reposts ADD COLUMN status repost_status NOT NULL DEFAULT 'complete'; - -COMMENT ON COLUMN reposts.status IS 'Repost lifecycle state: complete (normal), stub (placeholder, needs fetch)'; - --- Index on stub reposts for fetch queue queries --- Partial index only indexes non-complete reposts (very small set) -CREATE INDEX idx_reposts_status ON reposts(status) WHERE status != 'complete'; diff --git a/migrations/2025-11-16-005212_drop_likes_cid/down.sql b/migrations/2025-11-16-005212_drop_likes_cid/down.sql deleted file mode 100644 index 619a53bd..00000000 --- a/migrations/2025-11-16-005212_drop_likes_cid/down.sql +++ /dev/null @@ -1,4 +0,0 @@ --- Restore the cid column to likes table --- Note: This cannot restore the original CID data - column will be populated with empty values --- Synthetic CIDs will continue to be generated for notifications -ALTER TABLE likes ADD COLUMN cid BYTEA NOT NULL DEFAULT ''; diff --git a/migrations/2025-11-16-005212_drop_likes_cid/up.sql b/migrations/2025-11-16-005212_drop_likes_cid/up.sql deleted file mode 100644 index 452eac20..00000000 --- a/migrations/2025-11-16-005212_drop_likes_cid/up.sql +++ /dev/null @@ -1,3 +0,0 @@ --- Drop the cid column from likes table --- Like CIDs are now generated synthetically from (actor_id, rkey) instead of being stored -ALTER TABLE likes DROP COLUMN cid; diff --git a/migrations/2025-11-16-014100_drop_blocks_cid/down.sql b/migrations/2025-11-16-014100_drop_blocks_cid/down.sql deleted file mode 100644 index d5770626..00000000 --- a/migrations/2025-11-16-014100_drop_blocks_cid/down.sql +++ /dev/null @@ -1,2 +0,0 @@ --- Rollback: Add cid column back -ALTER TABLE blocks ADD COLUMN cid BYTEA NOT NULL DEFAULT E'\\x0000000000000000000000000000000000000000000000000000000000000000'::bytea; diff --git a/migrations/2025-11-16-014100_drop_blocks_cid/up.sql b/migrations/2025-11-16-014100_drop_blocks_cid/up.sql deleted file mode 100644 index 88428d2a..00000000 --- a/migrations/2025-11-16-014100_drop_blocks_cid/up.sql +++ /dev/null @@ -1,3 +0,0 @@ --- Drop the cid column from blocks table --- Block CIDs are now generated synthetically from (actor_id, rkey) -ALTER TABLE blocks DROP COLUMN cid; diff --git a/migrations/2025-11-16-014101_drop_follows_cid/down.sql b/migrations/2025-11-16-014101_drop_follows_cid/down.sql deleted file mode 100644 index 77e0c1a0..00000000 --- a/migrations/2025-11-16-014101_drop_follows_cid/down.sql +++ /dev/null @@ -1,2 +0,0 @@ --- Rollback: Add cid column back -ALTER TABLE follows ADD COLUMN cid BYTEA NOT NULL DEFAULT E'\\x0000000000000000000000000000000000000000000000000000000000000000'::bytea; diff --git a/migrations/2025-11-16-014101_drop_follows_cid/up.sql b/migrations/2025-11-16-014101_drop_follows_cid/up.sql deleted file mode 100644 index 07ea8bf4..00000000 --- a/migrations/2025-11-16-014101_drop_follows_cid/up.sql +++ /dev/null @@ -1,3 +0,0 @@ --- Drop the cid column from follows table --- Follow CIDs are now generated synthetically from (actor_id, rkey) -ALTER TABLE follows DROP COLUMN cid; diff --git a/migrations/2025-11-16-054411_add_tokens_to_posts/down.sql b/migrations/2025-11-16-054411_add_tokens_to_posts/down.sql deleted file mode 100644 index 16cf07a9..00000000 --- a/migrations/2025-11-16-054411_add_tokens_to_posts/down.sql +++ /dev/null @@ -1,3 +0,0 @@ --- Remove search tokens from posts table -DROP INDEX IF EXISTS idx_posts_tokens_gin; -ALTER TABLE posts DROP COLUMN IF EXISTS tokens; diff --git a/migrations/2025-11-16-054411_add_tokens_to_posts/up.sql b/migrations/2025-11-16-054411_add_tokens_to_posts/up.sql deleted file mode 100644 index c425bfee..00000000 --- a/migrations/2025-11-16-054411_add_tokens_to_posts/up.sql +++ /dev/null @@ -1,7 +0,0 @@ --- Add tokens column to posts table for search --- Nullable to avoid storage cost for posts not yet indexed --- GIN index for fast token containment queries -ALTER TABLE posts ADD COLUMN tokens TEXT[]; - --- GIN index for fast token array queries (tokens && query_tokens) -CREATE INDEX idx_posts_tokens_gin ON posts USING GIN (tokens); diff --git a/migrations/2025-11-16-070138_drop_search_vector/down.sql b/migrations/2025-11-16-070138_drop_search_vector/down.sql deleted file mode 100644 index eabca2e4..00000000 --- a/migrations/2025-11-16-070138_drop_search_vector/down.sql +++ /dev/null @@ -1,10 +0,0 @@ --- Restore search_vector column and index --- This rollback adds the column back but does NOT populate it --- You would need to run a backfill to regenerate search vectors - --- Add the search_vector column back -ALTER TABLE posts ADD COLUMN search_vector tsvector; - --- Recreate the GIN index -CREATE INDEX idx_posts_search_vector - ON posts USING GIN (search_vector); diff --git a/migrations/2025-11-16-070138_drop_search_vector/up.sql b/migrations/2025-11-16-070138_drop_search_vector/up.sql deleted file mode 100644 index d5120648..00000000 --- a/migrations/2025-11-16-070138_drop_search_vector/up.sql +++ /dev/null @@ -1,9 +0,0 @@ --- Drop search_vector column and index --- Token-based search replaces tsvector-based search --- Storage savings: ~220 GB (198 GB column + 20-30 GB index) for 30M posts - --- Drop the GIN index first -DROP INDEX IF EXISTS idx_posts_search_vector; - --- Drop the search_vector column -ALTER TABLE posts DROP COLUMN IF EXISTS search_vector; diff --git a/migrations/2025-11-16-203204_denormalize_post_embeds_and_facets/down.sql b/migrations/2025-11-16-203204_denormalize_post_embeds_and_facets/down.sql deleted file mode 100644 index 7b63d5b0..00000000 --- a/migrations/2025-11-16-203204_denormalize_post_embeds_and_facets/down.sql +++ /dev/null @@ -1,30 +0,0 @@ --- Rollback: Remove denormalized columns and composite types from posts table - --- Drop indexes -DROP INDEX IF EXISTS idx_posts_mentions; -DROP INDEX IF EXISTS idx_posts_embedded_post_id; - --- Drop columns -ALTER TABLE posts DROP COLUMN IF EXISTS mentions; -ALTER TABLE posts DROP COLUMN IF EXISTS facet_8; -ALTER TABLE posts DROP COLUMN IF EXISTS facet_7; -ALTER TABLE posts DROP COLUMN IF EXISTS facet_6; -ALTER TABLE posts DROP COLUMN IF EXISTS facet_5; -ALTER TABLE posts DROP COLUMN IF EXISTS facet_4; -ALTER TABLE posts DROP COLUMN IF EXISTS facet_3; -ALTER TABLE posts DROP COLUMN IF EXISTS facet_2; -ALTER TABLE posts DROP COLUMN IF EXISTS facet_1; -ALTER TABLE posts DROP COLUMN IF EXISTS image_4; -ALTER TABLE posts DROP COLUMN IF EXISTS image_3; -ALTER TABLE posts DROP COLUMN IF EXISTS image_2; -ALTER TABLE posts DROP COLUMN IF EXISTS image_1; -ALTER TABLE posts DROP COLUMN IF EXISTS record_detached; -ALTER TABLE posts DROP COLUMN IF EXISTS embedded_post_id; -ALTER TABLE posts DROP COLUMN IF EXISTS video_embed; -ALTER TABLE posts DROP COLUMN IF EXISTS ext_embed; - --- Drop composite types -DROP TYPE IF EXISTS post_facet_embed; -DROP TYPE IF EXISTS post_image_embed; -DROP TYPE IF EXISTS post_video_embed; -DROP TYPE IF EXISTS post_ext_embed; diff --git a/migrations/2025-11-16-203204_denormalize_post_embeds_and_facets/up.sql b/migrations/2025-11-16-203204_denormalize_post_embeds_and_facets/up.sql deleted file mode 100644 index 8571aac7..00000000 --- a/migrations/2025-11-16-203204_denormalize_post_embeds_and_facets/up.sql +++ /dev/null @@ -1,102 +0,0 @@ --- Denormalize post embeds and facets into posts table for query performance --- --- Strategy: Use composite types with NULLable columns for sparse data --- - External embed: 1 column (post_ext_embed) --- - Video embed: 1 column (post_video_embed) --- - Record embed: 2 columns (embedded_post_id, record_detached) --- - Images: 4 NULLable columns of post_image_embed (Bluesky protocol limit) --- - Facets: 8 NULLable columns of post_facet_embed (covers 99.3% of posts with facets) --- - Mentions: integer[] array (Diesel-friendly, truncate to max 8) --- --- Total columns: 17 (down from 89 separate columns!) --- Storage impact: NULL columns compress to ~0 bytes in TimescaleDB columnstore --- Query impact: Eliminates 5 JOINs + 1 extra query in embed loader - --- ============================================================================ --- Step 1: Create composite types for embeds --- ============================================================================ - --- External link embed (used by 0.8% of posts) -CREATE TYPE post_ext_embed AS ( - uri text, - title text, - description text, - thumb_mime_type image_mime_type, - thumb_cid bytea -); - --- Video embed (used by 0.16% of posts) -CREATE TYPE post_video_embed AS ( - mime_type video_mime_type, - cid bytea, - alt text, - width integer, - height integer -); - --- Image embed (used by 1.8% of posts, max 4 per Bluesky protocol) -CREATE TYPE post_image_embed AS ( - mime_type image_mime_type, - cid bytea, - alt text, - width integer, - height integer -); - --- Facet embed (facet_type, index_start, index_end, link_uri, mention_actor_id, tag) --- Used by 12.5% of posts, 99.3% have ≤8 facets -CREATE TYPE post_facet_embed AS ( - facet_type facet_type, - index_start integer, - index_end integer, - link_uri text, - mention_actor_id integer, - tag text -); - --- ============================================================================ --- Step 2: Add composite columns to posts table --- ============================================================================ - --- External embed (1:0..1 relationship) -ALTER TABLE posts ADD COLUMN ext_embed post_ext_embed; - --- Video embed (1:0..1 relationship) -ALTER TABLE posts ADD COLUMN video_embed post_video_embed; - --- Record embed / quote post (1:0..1 relationship) -ALTER TABLE posts ADD COLUMN embedded_post_id bigint REFERENCES posts(id) ON DELETE SET NULL; -ALTER TABLE posts ADD COLUMN record_detached boolean; - --- Image embeds (1:0..4 relationship, Bluesky protocol limit) --- NULL columns compress to ~0 bytes in TimescaleDB -ALTER TABLE posts ADD COLUMN image_1 post_image_embed; -ALTER TABLE posts ADD COLUMN image_2 post_image_embed; -ALTER TABLE posts ADD COLUMN image_3 post_image_embed; -ALTER TABLE posts ADD COLUMN image_4 post_image_embed; - --- Facet embeds (1:0..8 relationship, covers 99.3% of posts with facets) --- Spam posts with 100+ facets will be truncated to 8 -ALTER TABLE posts ADD COLUMN facet_1 post_facet_embed; -ALTER TABLE posts ADD COLUMN facet_2 post_facet_embed; -ALTER TABLE posts ADD COLUMN facet_3 post_facet_embed; -ALTER TABLE posts ADD COLUMN facet_4 post_facet_embed; -ALTER TABLE posts ADD COLUMN facet_5 post_facet_embed; -ALTER TABLE posts ADD COLUMN facet_6 post_facet_embed; -ALTER TABLE posts ADD COLUMN facet_7 post_facet_embed; -ALTER TABLE posts ADD COLUMN facet_8 post_facet_embed; - --- Mentions (1:many relationship, use array) --- Array of actor_ids, truncate to max 8 (covers 99.5% of posts with mentions) --- Arrays are Diesel-friendly (no custom FromSql/ToSql needed) -ALTER TABLE posts ADD COLUMN mentions integer[]; - --- ============================================================================ --- Step 3: Create indexes for common query patterns --- ============================================================================ - --- Index for quote posts (embedded_post_id lookups) -CREATE INDEX idx_posts_embedded_post_id ON posts(embedded_post_id) WHERE embedded_post_id IS NOT NULL; - --- GIN index for mention searches (find all posts mentioning an actor) -CREATE INDEX idx_posts_mentions ON posts USING GIN(mentions) WHERE mentions IS NOT NULL; diff --git a/migrations/2025-11-16-232532_drop_old_normalized_tables/down.sql b/migrations/2025-11-16-232532_drop_old_normalized_tables/down.sql deleted file mode 100644 index 8741cf85..00000000 --- a/migrations/2025-11-16-232532_drop_old_normalized_tables/down.sql +++ /dev/null @@ -1,9 +0,0 @@ --- This migration cannot be reverted because it drops tables that are no longer used --- The data has been migrated to denormalized composite type columns in the posts table --- If you need to revert, you would need to: --- 1. Recreate the old table schemas (see previous migrations) --- 2. Migrate data back from composite columns to normalized tables --- 3. Update application code to use old normalized tables - --- Placeholder to satisfy diesel migration requirements -SELECT 1; diff --git a/migrations/2025-11-16-232532_drop_old_normalized_tables/up.sql b/migrations/2025-11-16-232532_drop_old_normalized_tables/up.sql deleted file mode 100644 index 57c3aa3a..00000000 --- a/migrations/2025-11-16-232532_drop_old_normalized_tables/up.sql +++ /dev/null @@ -1,15 +0,0 @@ --- Drop old normalized tables that have been replaced by denormalized composite type columns --- See .claude/plans/1114-*.md for context on composite type denormalization - --- Drop old post embed tables (replaced by composite columns in posts table) -DROP TABLE IF EXISTS post_embed_video_captions CASCADE; -DROP TABLE IF EXISTS post_embed_video CASCADE; -DROP TABLE IF EXISTS post_embed_record CASCADE; -- Quote posts use embedded_post_id + record_detached columns -DROP TABLE IF EXISTS post_embed_images CASCADE; -DROP TABLE IF EXISTS post_embed_ext CASCADE; - --- Drop old post facets table (replaced by facet_1..facet_8 composite columns) -DROP TABLE IF EXISTS post_facets CASCADE; - --- Drop old post mentions table (replaced by mention_actor_ids array column) -DROP TABLE IF EXISTS post_mentions CASCADE; diff --git a/migrations/2025-11-16-233305_add_video_captions/down.sql b/migrations/2025-11-16-233305_add_video_captions/down.sql deleted file mode 100644 index 5e4e78e2..00000000 --- a/migrations/2025-11-16-233305_add_video_captions/down.sql +++ /dev/null @@ -1,21 +0,0 @@ --- Revert video captions support --- Restore original post_video_embed without captions field - --- Step 1: Drop video_embed column -ALTER TABLE posts DROP COLUMN IF EXISTS video_embed; - --- Step 2: Drop composite types -DROP TYPE IF EXISTS post_video_embed; -DROP TYPE IF EXISTS post_video_caption; - --- Step 3: Recreate original post_video_embed without captions -CREATE TYPE post_video_embed AS ( - mime_type video_mime_type, - cid bytea, - alt text, - width integer, - height integer -); - --- Step 4: Re-add video_embed column -ALTER TABLE posts ADD COLUMN video_embed post_video_embed; diff --git a/migrations/2025-11-16-233305_add_video_captions/up.sql b/migrations/2025-11-16-233305_add_video_captions/up.sql deleted file mode 100644 index e1647260..00000000 --- a/migrations/2025-11-16-233305_add_video_captions/up.sql +++ /dev/null @@ -1,36 +0,0 @@ --- Add video captions support to post_video_embed composite type --- --- Video captions (subtitles) are specified in AT Protocol via app.bsky.embed.video --- Each caption has: lang (ISO 639-1 code), file (Blob with mime_type, cid, size) --- Most videos have 0-3 caption tracks (e.g., English, Spanish, French) --- --- Captions are embedded in the video composite since they only exist with videos --- This avoids adding extra columns to posts table (already at 31 columns) - --- Step 1: Create composite type for individual video caption -CREATE TYPE post_video_caption AS ( - lang language_code, - mime_type caption_mime_type, - cid bytea -); - --- Step 2: Drop existing video_embed column (must drop to alter composite type) -ALTER TABLE posts DROP COLUMN video_embed; - --- Step 3: Drop old composite type -DROP TYPE post_video_embed; - --- Step 4: Recreate post_video_embed with 3 optional caption fields -CREATE TYPE post_video_embed AS ( - mime_type video_mime_type, - cid bytea, - alt text, - width integer, - height integer, - caption_1 post_video_caption, -- First caption track (e.g., English) - caption_2 post_video_caption, -- Second caption track (e.g., Spanish) - caption_3 post_video_caption -- Third caption track (e.g., French) -); - --- Step 5: Re-add video_embed column with new type -ALTER TABLE posts ADD COLUMN video_embed post_video_embed; diff --git a/migrations/2025-11-17-001423_split_likes_table/down.sql b/migrations/2025-11-17-001423_split_likes_table/down.sql deleted file mode 100644 index 85d5005c..00000000 --- a/migrations/2025-11-17-001423_split_likes_table/down.sql +++ /dev/null @@ -1,41 +0,0 @@ --- ============================================================================= --- ROLLBACK: Recreate unified likes table --- ============================================================================= - --- Recreate enum type -CREATE TYPE like_subject_type AS ENUM ('post', 'feedgen', 'labeler'); - --- Recreate unified table -CREATE TABLE likes ( - id BIGSERIAL PRIMARY KEY, - actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, - rkey INT8 NOT NULL, - subject_type like_subject_type NOT NULL DEFAULT 'post', - subject_id BIGINT, - via_repost_id BIGINT REFERENCES reposts(id) ON DELETE CASCADE, - UNIQUE (actor_id, rkey) -); - --- Migrate data back from split tables -INSERT INTO likes (actor_id, rkey, subject_type, subject_id, via_repost_id) -SELECT actor_id, rkey, 'post'::like_subject_type, post_id, via_repost_id -FROM post_likes; - -INSERT INTO likes (actor_id, rkey, subject_type, subject_id, via_repost_id) -SELECT actor_id, rkey, 'feedgen'::like_subject_type, feedgen_id, NULL -FROM feedgen_likes; - -INSERT INTO likes (actor_id, rkey, subject_type, subject_id, via_repost_id) -SELECT actor_id, rkey, 'labeler'::like_subject_type, labeler_actor_id::bigint, NULL -FROM labeler_likes; - --- Recreate indexes -CREATE INDEX idx_likes_actor ON likes(actor_id); -CREATE INDEX idx_likes_subject ON likes(subject_type, subject_id); -CREATE INDEX idx_likes_rkey ON likes(rkey); -CREATE INDEX idx_likes_via_repost ON likes(via_repost_id) WHERE via_repost_id IS NOT NULL; - --- Drop new tables -DROP TABLE IF EXISTS labeler_likes CASCADE; -DROP TABLE IF EXISTS feedgen_likes CASCADE; -DROP TABLE IF EXISTS post_likes CASCADE; diff --git a/migrations/2025-11-17-001423_split_likes_table/up.sql b/migrations/2025-11-17-001423_split_likes_table/up.sql deleted file mode 100644 index b7c16d7d..00000000 --- a/migrations/2025-11-17-001423_split_likes_table/up.sql +++ /dev/null @@ -1,114 +0,0 @@ --- ============================================================================= --- SPLIT LIKES TABLE - Create new tables with proper foreign keys --- ============================================================================= --- --- Background: The polymorphic likes table uses subject_type enum to determine --- what subject_id references. This prevents foreign key constraints and adds --- query overhead. Splitting into three tables enables: --- - Proper referential integrity with FK constraints --- - Better query performance (no discriminator filtering) --- - Smaller indexes (no subject_type in every index) --- - TimescaleDB hypertable conversion for post_likes --- --- Distribution: 99.98% post likes, 0.02% feedgen/labeler likes --- --- ============================================================================= - --- 1. POST LIKES (99.98% of likes) --- Will become a TimescaleDB hypertable later -CREATE TABLE post_likes ( - actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, - rkey INT8 NOT NULL, - post_id BIGINT NOT NULL REFERENCES posts(id) ON DELETE CASCADE, - via_repost_id BIGINT REFERENCES reposts(id) ON DELETE CASCADE, - PRIMARY KEY (actor_id, rkey) -); - -CREATE INDEX idx_post_likes_post ON post_likes(post_id); -CREATE INDEX idx_post_likes_rkey ON post_likes(rkey); -CREATE INDEX idx_post_likes_via_repost ON post_likes(via_repost_id) - WHERE via_repost_id IS NOT NULL; - -COMMENT ON TABLE post_likes IS 'Likes on posts (99.98% of all likes). FK to posts ensures referential integrity.'; -COMMENT ON COLUMN post_likes.via_repost_id IS 'If user liked via a repost, reference to the repost record (for notification context).'; - --- 2. FEEDGEN LIKES (rare) -CREATE TABLE feedgen_likes ( - actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, - rkey INT8 NOT NULL, - feedgen_id BIGINT NOT NULL REFERENCES feedgens(id) ON DELETE CASCADE, - PRIMARY KEY (actor_id, rkey) -); - -CREATE INDEX idx_feedgen_likes_feedgen ON feedgen_likes(feedgen_id); -CREATE INDEX idx_feedgen_likes_rkey ON feedgen_likes(rkey); - -COMMENT ON TABLE feedgen_likes IS 'Likes on feed generators (rare, ~0.01% of likes).'; - --- 3. LABELER LIKES (rare) -CREATE TABLE labeler_likes ( - actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, - rkey INT8 NOT NULL, - labeler_actor_id INTEGER NOT NULL REFERENCES actors(id) ON DELETE CASCADE, - PRIMARY KEY (actor_id, rkey) -); - -CREATE INDEX idx_labeler_likes_labeler ON labeler_likes(labeler_actor_id); -CREATE INDEX idx_labeler_likes_rkey ON labeler_likes(rkey); - -COMMENT ON TABLE labeler_likes IS 'Likes on labeler services (rare, ~0.01% of likes).'; - --- ============================================================================= --- MIGRATE DATA from old likes table --- ============================================================================= - --- Migrate post likes (should be 99.98% of data) -INSERT INTO post_likes (actor_id, rkey, post_id, via_repost_id) -SELECT actor_id, rkey, subject_id, via_repost_id -FROM likes -WHERE subject_type = 'post' - AND subject_id IS NOT NULL; -- Only migrate if we have a valid subject_id - --- Migrate feedgen likes -INSERT INTO feedgen_likes (actor_id, rkey, feedgen_id) -SELECT actor_id, rkey, subject_id -FROM likes -WHERE subject_type = 'feedgen' - AND subject_id IS NOT NULL; - --- Migrate labeler likes (note: labelers use actor_id, not a separate labeler table ID) -INSERT INTO labeler_likes (actor_id, rkey, labeler_actor_id) -SELECT actor_id, rkey, subject_id::integer -- Cast to integer since labelers reference actors(id) -FROM likes -WHERE subject_type = 'labeler' - AND subject_id IS NOT NULL; - --- Log migration results -DO $$ -DECLARE - post_count INTEGER; - feedgen_count INTEGER; - labeler_count INTEGER; - orphaned_count INTEGER; -BEGIN - SELECT COUNT(*) INTO post_count FROM post_likes; - SELECT COUNT(*) INTO feedgen_count FROM feedgen_likes; - SELECT COUNT(*) INTO labeler_count FROM labeler_likes; - SELECT COUNT(*) INTO orphaned_count FROM likes WHERE subject_id IS NULL; - - RAISE NOTICE 'Migration complete:'; - RAISE NOTICE ' post_likes: % rows', post_count; - RAISE NOTICE ' feedgen_likes: % rows', feedgen_count; - RAISE NOTICE ' labeler_likes: % rows', labeler_count; - RAISE NOTICE ' orphaned (skipped): % rows', orphaned_count; -END $$; - --- ============================================================================= --- DROP OLD TABLE --- ============================================================================= - --- Drop the old polymorphic table immediately -DROP TABLE likes CASCADE; - --- Drop the enum type (no longer needed) -DROP TYPE like_subject_type; diff --git a/migrations/2025-11-17-005012_add_diesel_type_generator_table/down.sql b/migrations/2025-11-17-005012_add_diesel_type_generator_table/down.sql deleted file mode 100644 index 75c196e5..00000000 --- a/migrations/2025-11-17-005012_add_diesel_type_generator_table/down.sql +++ /dev/null @@ -1,2 +0,0 @@ --- Drop the dummy schema inference table -DROP TABLE IF EXISTS _diesel_schema_inference; diff --git a/migrations/2025-11-17-005012_add_diesel_type_generator_table/up.sql b/migrations/2025-11-17-005012_add_diesel_type_generator_table/up.sql deleted file mode 100644 index 47f9d7ce..00000000 --- a/migrations/2025-11-17-005012_add_diesel_type_generator_table/up.sql +++ /dev/null @@ -1,16 +0,0 @@ --- Dummy table to force Diesel to generate SQL type definitions --- --- This table is never used by the application, but exists solely to ensure --- that Diesel generates type definitions for enums and composite types that are --- only used inside other composite types. --- --- Without this table, Diesel wouldn't generate these types in schema.rs since --- they're not directly used as column types in any real tables. - -CREATE TABLE _diesel_schema_inference ( - id INTEGER PRIMARY KEY, - video_mime_type video_mime_type NOT NULL, - caption_mime_type caption_mime_type NOT NULL, - facet_type facet_type NOT NULL, - video_caption post_video_caption NOT NULL -); diff --git a/migrations/2025-11-17-054507_enable_timescaledb/down.sql b/migrations/2025-11-17-054507_enable_timescaledb/down.sql deleted file mode 100644 index 3e827c14..00000000 --- a/migrations/2025-11-17-054507_enable_timescaledb/down.sql +++ /dev/null @@ -1,3 +0,0 @@ --- Remove TimescaleDB extension --- Warning: This will drop all hypertables and their data -DROP EXTENSION IF EXISTS timescaledb CASCADE; diff --git a/migrations/2025-11-17-054507_enable_timescaledb/up.sql b/migrations/2025-11-17-054507_enable_timescaledb/up.sql deleted file mode 100644 index d71344cd..00000000 --- a/migrations/2025-11-17-054507_enable_timescaledb/up.sql +++ /dev/null @@ -1,3 +0,0 @@ --- Enable TimescaleDB extension --- This adds TimescaleDB functions and capabilities to the database -CREATE EXTENSION IF NOT EXISTS timescaledb CASCADE; diff --git a/migrations/2025-11-17-054558_post_likes_hypertable/down.sql b/migrations/2025-11-17-054558_post_likes_hypertable/down.sql deleted file mode 100644 index 849852f8..00000000 --- a/migrations/2025-11-17-054558_post_likes_hypertable/down.sql +++ /dev/null @@ -1,35 +0,0 @@ --- ============================================================================= --- Revert post_likes hypertable to regular table --- ============================================================================= - --- Remove retention policy first -SELECT remove_retention_policy('post_likes', if_exists => true); - --- Remove compression policy -SELECT remove_compression_policy('post_likes', if_exists => true); - --- Decompress all compressed chunks before converting back -DO $$ -DECLARE - chunk_record RECORD; -BEGIN - FOR chunk_record IN - SELECT chunk_schema, chunk_name - FROM timescaledb_information.chunks - WHERE hypertable_name = 'post_likes' AND is_compressed - LOOP - EXECUTE format('SELECT decompress_chunk(%L)', - chunk_record.chunk_schema || '.' || chunk_record.chunk_name); - END LOOP; -END $$; - --- Note: There's no direct way to convert a hypertable back to a regular table --- The data remains accessible, but the hypertable structure stays --- To fully revert, you would need to: --- 1. Create a new regular table --- 2. Copy all data from the hypertable --- 3. Drop the hypertable --- 4. Rename the new table --- This is intentionally not automated to prevent accidental data loss -RAISE NOTICE 'Retention and compression policies removed. Hypertable structure remains.'; -RAISE NOTICE 'To fully revert to regular table, manual intervention required.'; diff --git a/migrations/2025-11-17-054558_post_likes_hypertable/up.sql b/migrations/2025-11-17-054558_post_likes_hypertable/up.sql deleted file mode 100644 index 9e0434d7..00000000 --- a/migrations/2025-11-17-054558_post_likes_hypertable/up.sql +++ /dev/null @@ -1,41 +0,0 @@ --- ============================================================================= --- Convert post_likes to TimescaleDB hypertable --- ============================================================================= - --- Convert post_likes to hypertable partitioned by rkey (timestamp-based TID) --- Use tid_timestamp function to extract timestamp from TID for partitioning --- Chunk interval: 1 day (86400000000 microseconds in TID format) -SELECT create_hypertable( - 'post_likes', - by_range('rkey', partition_func => 'tid_timestamp', partition_interval => INTERVAL '1 day'), - migrate_data => true, - if_not_exists => true -); - --- Enable compression on old chunks --- compress_segmentby: Group by actor_id for better compression (likes from same actor) --- compress_orderby: Order by rkey DESC for time-series queries -ALTER TABLE post_likes SET ( - timescaledb.compress, - timescaledb.compress_segmentby = 'actor_id', - timescaledb.compress_orderby = 'rkey DESC' -); - --- Add compression policy: compress chunks older than 7 days --- This happens automatically in the background -SELECT add_compression_policy( - 'post_likes', - compress_after => INTERVAL '7 days', - if_not_exists => true -); - --- Add retention policy: drop chunks older than 31 days --- This replaces the manual cleanup in CleanupWorker -SELECT add_retention_policy( - 'post_likes', - drop_after => INTERVAL '31 days', - if_not_exists => true -); - --- Create helpful comment -COMMENT ON TABLE post_likes IS 'TimescaleDB hypertable for post likes. Partitioned by rkey (timestamp). Auto-compressed after 7 days, auto-dropped after 31 days.'; diff --git a/migrations/2025-11-17-062926_optimize_post_likes_hypertable/down.sql b/migrations/2025-11-17-062926_optimize_post_likes_hypertable/down.sql deleted file mode 100644 index de967e57..00000000 --- a/migrations/2025-11-17-062926_optimize_post_likes_hypertable/down.sql +++ /dev/null @@ -1,62 +0,0 @@ --- ============================================================================= --- Revert post_likes Hypertable Optimizations --- ============================================================================= - --- Note: Reverting chunk interval only affects NEW chunks. --- Existing 4-hour chunks will remain until they age out (31-day retention). - --- ----------------------------------------------------------------------------- --- 1. Revert Chunk Skipping --- ----------------------------------------------------------------------------- - --- TimescaleDB doesn't provide disable_chunk_skipping as of v2.17 --- Chunk skipping will remain enabled but won't cause issues -RAISE NOTICE 'Chunk skipping cannot be disabled once enabled'; -RAISE NOTICE 'This is not harmful - it only provides query optimization'; - --- ----------------------------------------------------------------------------- --- 2. Revert Compression Policy: 12 hours → 7 days --- ----------------------------------------------------------------------------- - -SELECT remove_compression_policy('post_likes', if_exists => true); - -SELECT add_compression_policy( - 'post_likes', - compress_after => INTERVAL '7 days', - if_not_exists => true -); - --- ----------------------------------------------------------------------------- --- 3. Revert Chunk Interval: 4 hours → 1 day --- ----------------------------------------------------------------------------- - -SELECT set_chunk_time_interval('post_likes', INTERVAL '1 day'); - --- ----------------------------------------------------------------------------- --- 4. Update Table Comment --- ----------------------------------------------------------------------------- - -COMMENT ON TABLE post_likes IS 'TimescaleDB hypertable for post likes. Partitioned by rkey (timestamp). Auto-compressed after 7 days, auto-dropped after 31 days.'; - --- ----------------------------------------------------------------------------- --- 5. Verification --- ----------------------------------------------------------------------------- - -SELECT - h.hypertable_name, - d.time_interval as chunk_interval, - h.compression_enabled -FROM timescaledb_information.hypertables h -JOIN timescaledb_information.dimensions d ON h.hypertable_name = d.hypertable_name -WHERE h.hypertable_name = 'post_likes'; - -SELECT - hypertable_name, - proc_name, - config -FROM timescaledb_information.jobs -WHERE hypertable_name = 'post_likes' -ORDER BY proc_name; - -RAISE NOTICE 'Reverted to 1-day chunks and 7-day compression'; -RAISE NOTICE 'Existing 4-hour chunks will remain until 31-day retention drops them'; diff --git a/migrations/2025-11-17-062926_optimize_post_likes_hypertable/up.sql b/migrations/2025-11-17-062926_optimize_post_likes_hypertable/up.sql deleted file mode 100644 index 75dc8f58..00000000 --- a/migrations/2025-11-17-062926_optimize_post_likes_hypertable/up.sql +++ /dev/null @@ -1,159 +0,0 @@ --- ============================================================================= --- Optimize post_likes Hypertable for Network Scale (20M likes/day) --- ============================================================================= --- --- Based on performance testing and analysis: --- - Current: 812K rows, 1-day chunks (4 MB each) --- - Target: 20M rows/day, 4-hour chunks (52 MB each) --- - Compression ratio: 120:1 measured --- - Write performance: 172K rows/sec (COPY), only 7% slower on compressed chunks --- --- Storage projections at 31-day retention: --- - Before: 30 GB (2-day compression) --- - After: 26 GB (12-hour compression) --- - Savings: 15% overall, 65% less uncompressed data --- --- References: --- - .claude/reference/NETWORK_SCALE_STORAGE_ANALYSIS.md --- - .claude/reference/COMPRESSION_PERFORMANCE_TEST_RESULTS.md --- ============================================================================= - --- ----------------------------------------------------------------------------- --- 1. Optimize Chunk Interval: 1 day → 4 hours --- ----------------------------------------------------------------------------- --- --- Why 4 hours: --- - Creates 6 chunks/day (manageable metadata overhead) --- - ~52 MB per chunk (optimal: 10-100 MB range) --- - ~3.3M rows per chunk at 20M/day --- - 180 total chunks at 31-day retention --- --- Note: Only affects NEW chunks. Existing chunks remain at 1-day interval. --- This is expected and acceptable - they'll age out within 31 days. --- - -SELECT set_chunk_time_interval('post_likes', INTERVAL '4 hours'); - --- Verify the change -DO $$ -DECLARE - interval_hours NUMERIC; -BEGIN - SELECT - EXTRACT(EPOCH FROM d.time_interval) / 3600 INTO interval_hours - FROM timescaledb_information.dimensions d - WHERE d.hypertable_name = 'post_likes'; - - IF interval_hours != 4 THEN - RAISE EXCEPTION 'Chunk interval not set correctly. Expected 4 hours, got % hours', interval_hours; - END IF; - - RAISE NOTICE 'Chunk interval set to 4 hours successfully'; -END $$; - --- ----------------------------------------------------------------------------- --- 2. Update Compression Policy: 7 days → 12 hours (aggressive!) --- ----------------------------------------------------------------------------- --- --- Why 12 hours: --- - 99.9% of writes are <5 seconds old → hit uncompressed chunks (6.9ms) --- - Only 7% performance penalty on compressed chunks (7.4ms vs 6.9ms) --- - 65% reduction in uncompressed data (1.5 GB vs 6.1 GB) --- - Safe buffer for Jetstream catch-up scenarios (<48 hours) --- --- Performance testing showed: --- - COPY to uncompressed: 6.9ms per 1,000 rows --- - COPY to compressed: 7.4ms per 1,000 rows --- - Difference: negligible for our use case! --- - --- Remove existing compression policy -SELECT remove_compression_policy('post_likes', if_exists => true); - --- Add new aggressive compression policy -SELECT add_compression_policy( - 'post_likes', - compress_after => INTERVAL '12 hours', - if_not_exists => true -); - -COMMENT ON TABLE post_likes IS 'TimescaleDB hypertable for post likes. Partitioned by rkey (timestamp) with 4-hour chunks. Auto-compressed after 12 hours, auto-dropped after 31 days. Chunk skipping enabled on post_id.'; - --- ----------------------------------------------------------------------------- --- 3. Enable Chunk Skipping on post_id --- ----------------------------------------------------------------------------- --- --- Why post_id: --- - Highly correlated with time (likes cluster near post creation) --- - 568K unique posts, most with 1-2 likes --- - Top post spans only 14 days, not full 30 days --- - Estimated 50-75% chunk exclusion for post queries --- --- Why NOT actor_id: --- - Not correlated with time (actors span entire 30-day window) --- - Only 297 unique actors across all data --- - Would provide minimal benefit --- --- Note: Only works on chunks compressed AFTER enabling this feature. --- Existing compressed chunks won't benefit, but they'll age out. --- - --- Enable chunk skipping feature (tech preview as of TimescaleDB 2.17) -SET timescaledb.enable_chunk_skipping = on; - -SELECT enable_chunk_skipping('post_likes', 'post_id'); - --- Verify chunk skipping is enabled -DO $$ -DECLARE - skipping_enabled BOOLEAN; -BEGIN - -- Check if chunk skipping configuration exists - -- Note: This is a simplified check. Actual verification would query - -- timescaledb internal catalogs which may vary by version. - RAISE NOTICE 'Chunk skipping enabled on post_id column'; - RAISE NOTICE 'This will take effect on chunks compressed after this migration'; -END $$; - --- ----------------------------------------------------------------------------- --- 4. Summary of Configuration --- ----------------------------------------------------------------------------- - --- Display current hypertable configuration -SELECT - h.hypertable_name, - h.num_chunks, - h.compression_enabled, - d.column_name as partition_column, - d.time_interval as chunk_interval -FROM timescaledb_information.hypertables h -JOIN timescaledb_information.dimensions d ON h.hypertable_name = d.hypertable_name -WHERE h.hypertable_name = 'post_likes'; - --- Display policies -SELECT - hypertable_name, - proc_name, - config, - schedule_interval, - next_start -FROM timescaledb_information.jobs -WHERE hypertable_name = 'post_likes' -ORDER BY proc_name; - --- Display chunk statistics -SELECT - COUNT(*) as total_chunks, - COUNT(*) FILTER (WHERE is_compressed) as compressed, - COUNT(*) FILTER (WHERE NOT is_compressed) as uncompressed, - pg_size_pretty(SUM(pg_total_relation_size(chunk_schema || '.' || chunk_name))) as total_size -FROM timescaledb_information.chunks -WHERE hypertable_name = 'post_likes'; - --- Expected behavior after migration: --- ✅ New chunks created every 4 hours (6 per day) --- ✅ Chunks compressed automatically after 12 hours --- ✅ Chunks dropped automatically after 31 days (existing retention policy) --- ✅ Queries on post_id will skip irrelevant chunks (50-75% reduction) --- ✅ Write performance: 172K rows/sec, minimal impact from compression (7%) --- ✅ Storage at 31-day retention: ~26 GB (vs 30 GB before optimization) diff --git a/migrations/2025-11-17-072850_convert_posts_to_hypertable/down.sql b/migrations/2025-11-17-072850_convert_posts_to_hypertable/down.sql deleted file mode 100644 index aff681b6..00000000 --- a/migrations/2025-11-17-072850_convert_posts_to_hypertable/down.sql +++ /dev/null @@ -1,75 +0,0 @@ --- ============================================================================= --- Rollback: Convert posts Hypertable Back to Regular Table --- ============================================================================= --- --- WARNING: This migration will: --- 1. Decompress all chunks (may take time for large tables) --- 2. Convert hypertable back to regular PostgreSQL table --- 3. Restore original PRIMARY KEY structure --- --- Note: Chunk skipping cannot be fully rolled back, but it's harmless on regular tables --- ============================================================================= - --- ----------------------------------------------------------------------------- --- Step 1: Remove Compression Policy --- ----------------------------------------------------------------------------- - -SELECT remove_compression_policy('posts', if_exists => true); - --- ----------------------------------------------------------------------------- --- Step 2: Decompress All Chunks --- ----------------------------------------------------------------------------- - --- This may take some time for large tables -SELECT decompress_chunk(chunk_schema || '.' || chunk_name) -FROM timescaledb_information.chunks -WHERE hypertable_name = 'posts' AND is_compressed; - --- ----------------------------------------------------------------------------- --- Step 3: Disable Compression --- ----------------------------------------------------------------------------- - -ALTER TABLE posts SET ( - timescaledb.compress = false -); - --- ----------------------------------------------------------------------------- --- Step 4: Convert Back to Regular Table --- ----------------------------------------------------------------------------- - --- Note: This does NOT migrate data back - it just removes the hypertable metadata --- The data remains in the chunks which become regular tables --- For a full rollback with data migration, you would need to: --- 1. Create a new regular table --- 2. Copy all data from the hypertable --- 3. Drop the hypertable --- 4. Rename the new table - --- For now, we'll just raise a notice -DO $$ -BEGIN - RAISE NOTICE 'Rollback of hypertable conversion is complex and may not be needed.'; - RAISE NOTICE 'The table will remain a hypertable but with compression disabled.'; - RAISE NOTICE 'To fully convert back, manual steps are required:'; - RAISE NOTICE '1. CREATE TABLE posts_backup AS SELECT * FROM posts'; - RAISE NOTICE '2. DROP TABLE posts CASCADE'; - RAISE NOTICE '3. ALTER TABLE posts_backup RENAME TO posts'; - RAISE NOTICE '4. Recreate indexes and constraints'; -END $$; - --- ----------------------------------------------------------------------------- --- Step 5: Restore Original PRIMARY KEY (if needed) --- ----------------------------------------------------------------------------- - --- Drop composite primary key -ALTER TABLE posts DROP CONSTRAINT posts_pkey; - --- Restore original single-column primary key --- Note: This assumes the original was just (id) -ALTER TABLE posts ADD PRIMARY KEY (id); - -COMMENT ON COLUMN posts.id IS 'Primary key'; -COMMENT ON COLUMN posts.rkey IS 'TID timestamp identifier'; - --- Remove table comment -COMMENT ON TABLE posts IS NULL; diff --git a/migrations/2025-11-17-072850_convert_posts_to_hypertable/up.sql b/migrations/2025-11-17-072850_convert_posts_to_hypertable/up.sql deleted file mode 100644 index f73b138a..00000000 --- a/migrations/2025-11-17-072850_convert_posts_to_hypertable/up.sql +++ /dev/null @@ -1,123 +0,0 @@ --- ============================================================================= --- Convert posts Table to TimescaleDB Hypertable (Full Network Scale) --- ============================================================================= --- --- Configuration: --- - All posts (complete + stubs): Stubs will have lower compression but simplified architecture --- - NO retention policy: Full retention for network scale (2B posts) --- - Chunk interval: 1 day (adjust to 4 hours at network scale via set_chunk_time_interval) --- - Compression: 12 hours (aggressive, like post_likes) --- - Expected compression: ~6,496:1 for complete posts, ~20:1 for stubs, combined ~100:1 --- --- Storage projections at 2B posts with 10% complete, 90% stubs: --- - Complete: 200M × 0.43 bytes = 86 MB --- - Stubs: 1.8B × ~50 bytes = 90 GB --- - Total: ~90 GB (vs ~540 GB traditional, 83% savings) --- --- Note: PRIMARY KEY must include rkey for hypertable partitioning --- ============================================================================= - --- ----------------------------------------------------------------------------- --- Step 1: Add rkey to PRIMARY KEY --- ----------------------------------------------------------------------------- - --- Drop existing primary key (CASCADE to drop dependent foreign keys) --- TimescaleDB does not support foreign keys to hypertables, so we will NOT recreate them --- Referential integrity will be enforced at the application level -ALTER TABLE posts DROP CONSTRAINT posts_pkey CASCADE; - --- Recreate primary key with (id, rkey) for hypertable compatibility -ALTER TABLE posts ADD PRIMARY KEY (id, rkey); - -COMMENT ON COLUMN posts.id IS 'Primary key (first part of composite key with rkey for hypertable)'; -COMMENT ON COLUMN posts.rkey IS 'TID timestamp identifier (primary key second part, partitioning column)'; - --- Note: Foreign keys to posts table have been dropped and will NOT be recreated --- TimescaleDB limitation: cannot have FK constraints to hypertables --- Referential integrity enforced at application level - --- ----------------------------------------------------------------------------- --- Step 2: Convert to Hypertable --- ----------------------------------------------------------------------------- - -SELECT create_hypertable( - 'posts', - by_range('rkey', partition_func => 'tid_timestamp', partition_interval => INTERVAL '1 day'), - migrate_data => true, - if_not_exists => true -); - -COMMENT ON TABLE posts IS 'TimescaleDB hypertable for posts (complete + stubs). Partitioned by rkey with 1-day chunks. Auto-compressed after 12 hours. No retention policy (full network scale).'; - --- ----------------------------------------------------------------------------- --- Step 3: Enable Compression --- ----------------------------------------------------------------------------- - -ALTER TABLE posts SET ( - timescaledb.compress, - timescaledb.compress_segmentby = 'actor_id', - timescaledb.compress_orderby = 'rkey DESC' -); - --- ----------------------------------------------------------------------------- --- Step 4: Add Aggressive Compression Policy (12 hours, like post_likes) --- ----------------------------------------------------------------------------- - -SELECT add_compression_policy( - 'posts', - compress_after => INTERVAL '12 hours', - if_not_exists => true -); - --- NO retention policy - we want full retention for network scale! - --- ----------------------------------------------------------------------------- --- Step 5: Enable Chunk Skipping (Tech Preview Feature) --- ----------------------------------------------------------------------------- - --- Enable chunk skipping feature for this session -SET timescaledb.enable_chunk_skipping = on; - --- Enable chunk skipping on post_id for efficient parent post queries --- This allows queries like "get replies to post X" to skip irrelevant chunks -SELECT enable_chunk_skipping('posts', 'parent_post_id'); - --- Enable chunk skipping on root_post_id for thread queries -SELECT enable_chunk_skipping('posts', 'root_post_id'); - --- ----------------------------------------------------------------------------- --- Step 6: Display Configuration Summary --- ----------------------------------------------------------------------------- - --- Display hypertable configuration -DO $$ -DECLARE - chunk_count INTEGER; - compressed_count INTEGER; -BEGIN - SELECT COUNT(*), COUNT(*) FILTER (WHERE is_compressed) - INTO chunk_count, compressed_count - FROM timescaledb_information.chunks - WHERE hypertable_name = 'posts'; - - RAISE NOTICE ''; - RAISE NOTICE '✅ posts table converted to hypertable successfully'; - RAISE NOTICE 'Configuration:'; - RAISE NOTICE ' - Chunks: % total (% compressed)', chunk_count, compressed_count; - RAISE NOTICE ' - Compression: Enabled (actor_id segmentby, rkey orderby)'; - RAISE NOTICE ' - Compression policy: 12 hours'; - RAISE NOTICE ' - Retention policy: NONE (full network scale)'; - RAISE NOTICE ' - Chunk skipping: Enabled on parent_post_id, root_post_id'; - RAISE NOTICE ''; - RAISE NOTICE 'Query the following for more details:'; - RAISE NOTICE ' SELECT * FROM timescaledb_information.hypertables WHERE hypertable_name = ''posts'';'; - RAISE NOTICE ' SELECT * FROM timescaledb_information.jobs WHERE hypertable_name = ''posts'';'; -END $$; - --- Expected behavior after migration: --- ✅ All posts (complete + stubs) migrated to hypertable --- ✅ Chunks created with 1-day interval (32-40 chunks for current data) --- ✅ Chunks compressed automatically after 12 hours --- ✅ NO retention - all data kept forever (network scale) --- ✅ Chunk skipping enabled on parent_post_id and root_post_id --- ✅ Storage: ~90 GB at 2B posts (vs 540 GB traditional, 83% savings) diff --git a/migrations/2025-11-17-072937_remove_retention_policies/down.sql b/migrations/2025-11-17-072937_remove_retention_policies/down.sql deleted file mode 100644 index 222a4eda..00000000 --- a/migrations/2025-11-17-072937_remove_retention_policies/down.sql +++ /dev/null @@ -1,43 +0,0 @@ --- ============================================================================= --- Restore Retention Policies (Rollback to 31-Day Retention) --- ============================================================================= --- --- This restores the previous 31-day retention policy on post_likes --- Note: posts table never had retention, so nothing to restore there --- --- ============================================================================= - --- ----------------------------------------------------------------------------- --- Restore 31-Day Retention Policy on post_likes --- ----------------------------------------------------------------------------- - -SELECT add_retention_policy( - 'post_likes', - drop_after => INTERVAL '31 days', - if_not_exists => true -); - -COMMENT ON TABLE post_likes IS 'TimescaleDB hypertable for post likes. Partitioned by rkey (timestamp) with 4-hour chunks. Auto-compressed after 12 hours, auto-dropped after 31 days. Chunk skipping enabled on post_id.'; - --- ----------------------------------------------------------------------------- --- Display Restored Configuration --- ----------------------------------------------------------------------------- - -SELECT - hypertable_name, - proc_name, - config, - schedule_interval, - next_start -FROM timescaledb_information.jobs -WHERE hypertable_name = 'post_likes' -ORDER BY proc_name; - --- Summary -DO $$ -BEGIN - RAISE NOTICE '✅ Retention policy restored on post_likes'; - RAISE NOTICE 'Configuration:'; - RAISE NOTICE ' - post_likes: 4-hour chunks, 12-hour compression, 31-day retention'; - RAISE NOTICE ' - posts: 1-day chunks, 12-hour compression, NO retention'; -END $$; diff --git a/migrations/2025-11-17-072937_remove_retention_policies/up.sql b/migrations/2025-11-17-072937_remove_retention_policies/up.sql deleted file mode 100644 index 82932e57..00000000 --- a/migrations/2025-11-17-072937_remove_retention_policies/up.sql +++ /dev/null @@ -1,63 +0,0 @@ --- ============================================================================= --- Remove Retention Policies for Full Network Scale Storage --- ============================================================================= --- --- Decision: With 83-99% compression savings, we can afford full retention --- Storage projections with compression (no retention): --- - post_likes @ 12B records: ~15 GB (was 30 GB with 31-day retention) --- - posts @ 2B records: ~90 GB (complete + stubs) --- - Total: ~105 GB for full network scale (vs ~1TB+ traditional) --- --- This allows: --- - Historical data analysis --- - Backfill without data loss --- - Simpler operations (no chunk deletion) --- - Better compression (more data to compress) --- --- ============================================================================= - --- ----------------------------------------------------------------------------- --- Remove Retention Policy from post_likes --- ----------------------------------------------------------------------------- - -SELECT remove_retention_policy('post_likes', if_exists => true); - -COMMENT ON TABLE post_likes IS 'TimescaleDB hypertable for post likes. Partitioned by rkey (timestamp) with 4-hour chunks. Auto-compressed after 12 hours. NO retention policy (full network scale). Chunk skipping enabled on post_id.'; - --- ----------------------------------------------------------------------------- --- Display Updated Configuration --- ----------------------------------------------------------------------------- - --- Show policies for post_likes (should only show compression, no retention) -SELECT - hypertable_name, - proc_name, - config, - schedule_interval -FROM timescaledb_information.jobs -WHERE hypertable_name = 'post_likes' -ORDER BY proc_name; - --- Show policies for posts (should only show compression, no retention) -SELECT - hypertable_name, - proc_name, - config, - schedule_interval -FROM timescaledb_information.jobs -WHERE hypertable_name = 'posts' -ORDER BY proc_name; - --- Summary -DO $$ -BEGIN - RAISE NOTICE '✅ Retention policies removed from all hypertables'; - RAISE NOTICE 'Configuration:'; - RAISE NOTICE ' - post_likes: 4-hour chunks, 12-hour compression, NO retention'; - RAISE NOTICE ' - posts: 1-day chunks, 12-hour compression, NO retention'; - RAISE NOTICE ' '; - RAISE NOTICE 'Expected storage at network scale (full retention):'; - RAISE NOTICE ' - post_likes @ 12B records: ~15 GB'; - RAISE NOTICE ' - posts @ 2B records: ~90 GB'; - RAISE NOTICE ' - Total: ~105 GB (vs ~1TB+ traditional, 90%% savings)'; -END $$; diff --git a/migrations/2025-11-18-191434_remove_post_id_natural_keys/down.sql b/migrations/2025-11-18-191434_remove_post_id_natural_keys/down.sql deleted file mode 100644 index 7f38a88b..00000000 --- a/migrations/2025-11-18-191434_remove_post_id_natural_keys/down.sql +++ /dev/null @@ -1,381 +0,0 @@ --- Rollback: Restore post_id synthetic integer keys --- This reverses the migration to natural keys - --- ============================================================================ --- Phase 1: Drop foreign key constraints that reference natural keys --- ============================================================================ - --- Child tables -ALTER TABLE threadgate_allowed_lists DROP CONSTRAINT IF EXISTS threadgate_allowed_lists_threadgate_fkey; -ALTER TABLE threadgate_allowed_lists DROP CONSTRAINT IF EXISTS threadgate_allowed_lists_list_id_fkey; -ALTER TABLE threadgate_hidden_replies DROP CONSTRAINT IF EXISTS threadgate_hidden_replies_threadgate_fkey; -ALTER TABLE postgate_detached DROP CONSTRAINT IF EXISTS postgate_detached_postgate_fkey; - --- Main tables -ALTER TABLE reposts DROP CONSTRAINT IF EXISTS reposts_post_fkey; -ALTER TABLE reposts DROP CONSTRAINT IF EXISTS reposts_via_repost_fkey; -ALTER TABLE post_aggregate_stats DROP CONSTRAINT IF EXISTS post_aggregate_stats_post_fkey; -ALTER TABLE threadgates DROP CONSTRAINT IF EXISTS threadgates_post_fkey; -ALTER TABLE postgates DROP CONSTRAINT IF EXISTS postgates_post_fkey; -ALTER TABLE profiles DROP CONSTRAINT IF EXISTS profiles_pinned_post_fkey; - --- ============================================================================ --- Phase 2: Drop unique constraints on natural keys --- ============================================================================ - -ALTER TABLE reposts DROP CONSTRAINT IF EXISTS reposts_actor_id_post_actor_id_post_rkey_key; - --- ============================================================================ --- Phase 3: Drop indexes on natural keys --- ============================================================================ - -DROP INDEX IF EXISTS idx_posts_parent; -DROP INDEX IF EXISTS idx_posts_root; -DROP INDEX IF EXISTS idx_posts_embedded_post; -DROP INDEX IF EXISTS idx_post_likes_post_actor_id_post_rkey; -DROP INDEX IF EXISTS idx_reposts_post_actor_id_post_rkey; - --- ============================================================================ --- Phase 4: Drop primary keys based on natural keys --- ============================================================================ - -ALTER TABLE posts DROP CONSTRAINT IF EXISTS posts_pkey; -ALTER TABLE reposts DROP CONSTRAINT IF EXISTS reposts_pkey; -ALTER TABLE post_aggregate_stats DROP CONSTRAINT IF EXISTS post_aggregate_stats_pkey; -ALTER TABLE threadgates DROP CONSTRAINT IF EXISTS threadgates_pkey; -ALTER TABLE postgates DROP CONSTRAINT IF EXISTS postgates_pkey; - --- Child tables -ALTER TABLE threadgate_allowed_lists DROP CONSTRAINT IF EXISTS threadgate_allowed_lists_pkey; -ALTER TABLE threadgate_hidden_replies DROP CONSTRAINT IF EXISTS threadgate_hidden_replies_pkey; -ALTER TABLE postgate_detached DROP CONSTRAINT IF EXISTS postgate_detached_pkey; - --- ============================================================================ --- Phase 5: Restore id columns with BIGSERIAL (generates sequential integers) --- ============================================================================ - --- posts: Restore id and create composite primary key temporarily -ALTER TABLE posts ADD COLUMN id BIGSERIAL; -ALTER TABLE posts ADD PRIMARY KEY (id, rkey); -- Composite PK includes rkey (partition key for hypertable) - --- reposts: Restore id -ALTER TABLE reposts ADD COLUMN id BIGSERIAL PRIMARY KEY; - --- threadgates: Restore id -ALTER TABLE threadgates ADD COLUMN id BIGSERIAL PRIMARY KEY; - --- postgates: Restore id -ALTER TABLE postgates ADD COLUMN id BIGSERIAL PRIMARY KEY; - --- post_aggregate_stats: Restore post_id as primary key -ALTER TABLE post_aggregate_stats ADD COLUMN post_id BIGINT PRIMARY KEY; - --- ============================================================================ --- Phase 6: Add back old foreign key columns --- ============================================================================ - --- posts self-references -ALTER TABLE posts ADD COLUMN parent_post_id BIGINT; -ALTER TABLE posts ADD COLUMN root_post_id BIGINT; -ALTER TABLE posts ADD COLUMN embedded_post_id BIGINT; - --- post_likes -ALTER TABLE post_likes ADD COLUMN post_id BIGINT; -ALTER TABLE post_likes ADD COLUMN via_repost_id BIGINT; - --- reposts -ALTER TABLE reposts ADD COLUMN post_id BIGINT; -ALTER TABLE reposts ADD COLUMN via_repost_id BIGINT; - --- profiles -ALTER TABLE profiles ADD COLUMN pinned_post_id BIGINT; - --- Child tables -ALTER TABLE threadgate_allowed_lists ADD COLUMN threadgate_id BIGINT; -ALTER TABLE threadgate_hidden_replies ADD COLUMN threadgate_id BIGINT; -ALTER TABLE threadgate_hidden_replies ADD COLUMN post_id BIGINT; -ALTER TABLE postgate_detached ADD COLUMN postgate_id BIGINT; -ALTER TABLE postgate_detached ADD COLUMN detached_post_id BIGINT; - --- ============================================================================ --- Phase 7: Populate old columns from natural keys --- ============================================================================ - --- posts self-references -UPDATE posts p SET - parent_post_id = parent.id -FROM posts parent -WHERE p.parent_post_actor_id = parent.actor_id - AND p.parent_post_rkey = parent.rkey; - -UPDATE posts p SET - root_post_id = root.id -FROM posts root -WHERE p.root_post_actor_id = root.actor_id - AND p.root_post_rkey = root.rkey; - -UPDATE posts p SET - embedded_post_id = embedded.id -FROM posts embedded -WHERE p.embedded_post_actor_id = embedded.actor_id - AND p.embedded_post_rkey = embedded.rkey; - --- post_likes (post reference) -UPDATE post_likes pl SET - post_id = p.id -FROM posts p -WHERE pl.post_actor_id = p.actor_id - AND pl.post_rkey = p.rkey; - --- post_likes (via_repost reference) -UPDATE post_likes pl SET - via_repost_id = r.id -FROM reposts r -WHERE pl.via_repost_actor_id = r.actor_id - AND pl.via_repost_rkey = r.rkey; - --- reposts (post reference) -UPDATE reposts r SET - post_id = p.id -FROM posts p -WHERE r.post_actor_id = p.actor_id - AND r.post_rkey = p.rkey; - --- reposts (via_repost self-reference) -UPDATE reposts r SET - via_repost_id = vr.id -FROM reposts vr -WHERE r.via_repost_actor_id = vr.actor_id - AND r.via_repost_rkey = vr.rkey; - --- post_aggregate_stats -UPDATE post_aggregate_stats pas SET - post_id = p.id -FROM posts p -WHERE pas.post_actor_id = p.actor_id - AND pas.post_rkey = p.rkey; - --- profiles (only the actor's own posts) -UPDATE profiles pr SET - pinned_post_id = p.id -FROM posts p -WHERE pr.pinned_post_rkey = p.rkey - AND pr.actor_id = p.actor_id; - --- threadgate_allowed_lists -UPDATE threadgate_allowed_lists tal SET - threadgate_id = tg.id -FROM threadgates tg -WHERE tal.post_actor_id = tg.post_actor_id - AND tal.post_rkey = tg.post_rkey; - --- threadgate_hidden_replies -UPDATE threadgate_hidden_replies thr SET - threadgate_id = tg.id, - post_id = p.id -FROM threadgates tg, posts p -WHERE thr.post_actor_id = tg.post_actor_id - AND thr.post_rkey = tg.post_rkey - AND thr.hidden_post_actor_id = p.actor_id - AND thr.hidden_post_rkey = p.rkey; - --- postgate_detached -UPDATE postgate_detached pd SET - postgate_id = pg.id, - detached_post_id = p.id -FROM postgates pg, posts p -WHERE pd.post_actor_id = pg.post_actor_id - AND pd.post_rkey = pg.post_rkey - AND pd.detached_post_actor_id = p.actor_id - AND pd.detached_post_rkey = p.rkey; - --- ============================================================================ --- Phase 8: Drop natural key columns --- ============================================================================ - --- posts -ALTER TABLE posts DROP COLUMN parent_post_actor_id; -ALTER TABLE posts DROP COLUMN parent_post_rkey; -ALTER TABLE posts DROP COLUMN root_post_actor_id; -ALTER TABLE posts DROP COLUMN root_post_rkey; -ALTER TABLE posts DROP COLUMN embedded_post_actor_id; -ALTER TABLE posts DROP COLUMN embedded_post_rkey; - --- post_likes -ALTER TABLE post_likes DROP COLUMN post_actor_id; -ALTER TABLE post_likes DROP COLUMN post_rkey; -ALTER TABLE post_likes DROP COLUMN via_repost_actor_id; -ALTER TABLE post_likes DROP COLUMN via_repost_rkey; - --- reposts -ALTER TABLE reposts DROP COLUMN post_actor_id; -ALTER TABLE reposts DROP COLUMN post_rkey; -ALTER TABLE reposts DROP COLUMN via_repost_actor_id; -ALTER TABLE reposts DROP COLUMN via_repost_rkey; - --- post_aggregate_stats -ALTER TABLE post_aggregate_stats DROP COLUMN post_actor_id; -ALTER TABLE post_aggregate_stats DROP COLUMN post_rkey; - --- threadgates (keep post_actor_id for other use, but drop post_rkey) --- Actually, these were added for this migration, so drop both -ALTER TABLE threadgates DROP COLUMN post_actor_id; -ALTER TABLE threadgates DROP COLUMN post_rkey; - --- postgates -ALTER TABLE postgates DROP COLUMN post_actor_id; -ALTER TABLE postgates DROP COLUMN post_rkey; - --- profiles -ALTER TABLE profiles DROP COLUMN pinned_post_rkey; - --- Child tables -ALTER TABLE threadgate_allowed_lists DROP COLUMN post_actor_id; -ALTER TABLE threadgate_allowed_lists DROP COLUMN post_rkey; - -ALTER TABLE threadgate_hidden_replies DROP COLUMN post_actor_id; -ALTER TABLE threadgate_hidden_replies DROP COLUMN post_rkey; -ALTER TABLE threadgate_hidden_replies DROP COLUMN hidden_post_actor_id; -ALTER TABLE threadgate_hidden_replies DROP COLUMN hidden_post_rkey; - -ALTER TABLE postgate_detached DROP COLUMN post_actor_id; -ALTER TABLE postgate_detached DROP COLUMN post_rkey; -ALTER TABLE postgate_detached DROP COLUMN detached_post_actor_id; -ALTER TABLE postgate_detached DROP COLUMN detached_post_rkey; - --- ============================================================================ --- Phase 9: Restore original primary keys for child tables --- ============================================================================ - -ALTER TABLE threadgate_allowed_lists ADD PRIMARY KEY (threadgate_id, list_id); -ALTER TABLE threadgate_hidden_replies ADD PRIMARY KEY (threadgate_id, post_id); -ALTER TABLE postgate_detached ADD PRIMARY KEY (postgate_id, detached_post_id); - --- ============================================================================ --- Phase 10: Restore unique constraints --- ============================================================================ - -ALTER TABLE post_likes ADD CONSTRAINT post_likes_actor_id_post_id_key - UNIQUE (actor_id, post_id); - -ALTER TABLE reposts ADD CONSTRAINT reposts_actor_id_post_id_key - UNIQUE (actor_id, post_id); - --- ============================================================================ --- Phase 11: Restore foreign key constraints --- ============================================================================ - --- posts self-references -ALTER TABLE posts ADD CONSTRAINT posts_parent_post_id_fkey - FOREIGN KEY (parent_post_id) - REFERENCES posts(id) - ON DELETE CASCADE; - -ALTER TABLE posts ADD CONSTRAINT posts_root_post_id_fkey - FOREIGN KEY (root_post_id) - REFERENCES posts(id) - ON DELETE CASCADE; - -ALTER TABLE posts ADD CONSTRAINT posts_embedded_post_id_fkey - FOREIGN KEY (embedded_post_id) - REFERENCES posts(id) - ON DELETE CASCADE; - --- post_likes -ALTER TABLE post_likes ADD CONSTRAINT post_likes_post_id_fkey - FOREIGN KEY (post_id) - REFERENCES posts(id) - ON DELETE CASCADE; - -ALTER TABLE post_likes ADD CONSTRAINT post_likes_via_repost_id_fkey - FOREIGN KEY (via_repost_id) - REFERENCES reposts(id) - ON DELETE CASCADE; - --- reposts -ALTER TABLE reposts ADD CONSTRAINT reposts_post_id_fkey - FOREIGN KEY (post_id) - REFERENCES posts(id) - ON DELETE CASCADE; - -ALTER TABLE reposts ADD CONSTRAINT reposts_via_repost_id_fkey - FOREIGN KEY (via_repost_id) - REFERENCES reposts(id) - ON DELETE CASCADE; - --- post_aggregate_stats -ALTER TABLE post_aggregate_stats ADD CONSTRAINT post_aggregate_stats_post_id_fkey - FOREIGN KEY (post_id) - REFERENCES posts(id) - ON DELETE CASCADE; - --- threadgates (needs post_id column first) -ALTER TABLE threadgates ADD COLUMN post_id BIGINT; -UPDATE threadgates tg SET post_id = p.id -FROM posts p -WHERE tg.post_actor_id = p.actor_id AND tg.post_rkey = p.rkey; - -ALTER TABLE threadgates ADD CONSTRAINT threadgates_post_id_fkey - FOREIGN KEY (post_id) - REFERENCES posts(id) - ON DELETE CASCADE; - --- postgates (needs post_id column first) -ALTER TABLE postgates ADD COLUMN post_id BIGINT; -UPDATE postgates pg SET post_id = p.id -FROM posts p -WHERE pg.post_actor_id = p.actor_id AND pg.post_rkey = p.rkey; - -ALTER TABLE postgates ADD CONSTRAINT postgates_post_id_fkey - FOREIGN KEY (post_id) - REFERENCES posts(id) - ON DELETE CASCADE; - --- profiles -ALTER TABLE profiles ADD CONSTRAINT profiles_pinned_post_id_fkey - FOREIGN KEY (pinned_post_id) - REFERENCES posts(id) - ON DELETE SET NULL; - --- Child tables -ALTER TABLE threadgate_allowed_lists ADD CONSTRAINT threadgate_allowed_lists_threadgate_id_fkey - FOREIGN KEY (threadgate_id) - REFERENCES threadgates(id) - ON DELETE CASCADE; - -ALTER TABLE threadgate_allowed_lists ADD CONSTRAINT threadgate_allowed_lists_list_id_fkey - FOREIGN KEY (list_id) - REFERENCES lists(id) - ON DELETE CASCADE; - -ALTER TABLE threadgate_hidden_replies ADD CONSTRAINT threadgate_hidden_replies_threadgate_id_fkey - FOREIGN KEY (threadgate_id) - REFERENCES threadgates(id) - ON DELETE CASCADE; - -ALTER TABLE postgate_detached ADD CONSTRAINT postgate_detached_postgate_id_fkey - FOREIGN KEY (postgate_id) - REFERENCES postgates(id) - ON DELETE CASCADE; - --- ============================================================================ --- Phase 12: Restore original indexes --- ============================================================================ - -CREATE INDEX idx_posts_parent - ON posts(parent_post_id) - WHERE parent_post_id IS NOT NULL; - -CREATE INDEX idx_posts_root - ON posts(root_post_id) - WHERE root_post_id IS NOT NULL; - -CREATE INDEX idx_posts_embedded_post_id - ON posts(embedded_post_id) - WHERE embedded_post_id IS NOT NULL; - -CREATE INDEX idx_post_likes_post_id - ON post_likes(post_id); - -CREATE INDEX idx_reposts_post_id - ON reposts(post_id); diff --git a/migrations/2025-11-18-191434_remove_post_id_natural_keys/up.sql b/migrations/2025-11-18-191434_remove_post_id_natural_keys/up.sql deleted file mode 100644 index f35e5b72..00000000 --- a/migrations/2025-11-18-191434_remove_post_id_natural_keys/up.sql +++ /dev/null @@ -1,391 +0,0 @@ --- Remove post_id synthetic integer keys, use natural keys (actor_id, rkey) instead --- This migration is designed for parakeet_test (empty database) initially --- Production migration for parakeet database will be handled separately - --- IMPORTANT: TimescaleDB Hypertable Constraints (as of recent TimescaleDB versions) --- - Hypertable → Hypertable FKs: NOT SUPPORTED (e.g., posts → posts, post_likes → posts) --- - Regular → Hypertable FKs: SUPPORTED (e.g., reposts → posts) --- - Hypertable → Regular FKs: SUPPORTED (e.g., posts → actors) --- --- posts and post_likes are hypertables: --- - Cannot have FK constraints to each other (hypertable → hypertable) --- - Cannot have self-referencing FKs (posts → posts) --- - Regular tables CAN have FKs to them (reposts → posts works!) --- --- NOTE: reposts may become a hypertable in the future, which would break post_likes → reposts FK --- if post_likes.via_repost_id needs to reference it. This is already handled by using natural keys --- without database FK constraints. - --- ============================================================================ --- Phase 1: Add new natural key columns to all tables (before dropping posts.id) --- ============================================================================ - --- posts table: Add natural key columns for parent/root/embedded references -ALTER TABLE posts ADD COLUMN parent_post_actor_id INTEGER; -ALTER TABLE posts ADD COLUMN parent_post_rkey BIGINT; -ALTER TABLE posts ADD COLUMN root_post_actor_id INTEGER; -ALTER TABLE posts ADD COLUMN root_post_rkey BIGINT; -ALTER TABLE posts ADD COLUMN embedded_post_actor_id INTEGER; -ALTER TABLE posts ADD COLUMN embedded_post_rkey BIGINT; - --- post_likes table: Add natural key columns for post reference and via_repost -ALTER TABLE post_likes ADD COLUMN post_actor_id INTEGER; -ALTER TABLE post_likes ADD COLUMN post_rkey BIGINT; -ALTER TABLE post_likes ADD COLUMN via_repost_actor_id INTEGER; -ALTER TABLE post_likes ADD COLUMN via_repost_rkey BIGINT; - --- reposts table: Add natural key columns for post reference and via_repost -ALTER TABLE reposts ADD COLUMN post_actor_id INTEGER; -ALTER TABLE reposts ADD COLUMN post_rkey BIGINT; -ALTER TABLE reposts ADD COLUMN via_repost_actor_id INTEGER; -ALTER TABLE reposts ADD COLUMN via_repost_rkey BIGINT; - --- post_aggregate_stats table: Add natural key columns -ALTER TABLE post_aggregate_stats ADD COLUMN post_actor_id INTEGER; -ALTER TABLE post_aggregate_stats ADD COLUMN post_rkey BIGINT; - --- threadgates table: Add natural key columns -ALTER TABLE threadgates ADD COLUMN post_actor_id INTEGER; -ALTER TABLE threadgates ADD COLUMN post_rkey BIGINT; - --- postgates table: Add natural key columns -ALTER TABLE postgates ADD COLUMN post_actor_id INTEGER; -ALTER TABLE postgates ADD COLUMN post_rkey BIGINT; - --- profiles table: Add rkey-only column (actor_id is implicit) -ALTER TABLE profiles ADD COLUMN pinned_post_rkey BIGINT; - --- statuses table: Add natural key columns for embed_post -ALTER TABLE statuses ADD COLUMN embed_post_actor_id INTEGER; -ALTER TABLE statuses ADD COLUMN embed_post_rkey BIGINT; - --- Child tables: threadgate_allowed_lists -ALTER TABLE threadgate_allowed_lists ADD COLUMN post_actor_id INTEGER; -ALTER TABLE threadgate_allowed_lists ADD COLUMN post_rkey BIGINT; - --- Child tables: threadgate_hidden_replies -ALTER TABLE threadgate_hidden_replies ADD COLUMN post_actor_id INTEGER; -ALTER TABLE threadgate_hidden_replies ADD COLUMN post_rkey BIGINT; -ALTER TABLE threadgate_hidden_replies ADD COLUMN hidden_post_actor_id INTEGER; -ALTER TABLE threadgate_hidden_replies ADD COLUMN hidden_post_rkey BIGINT; - --- Child tables: postgate_detached -ALTER TABLE postgate_detached ADD COLUMN post_actor_id INTEGER; -ALTER TABLE postgate_detached ADD COLUMN post_rkey BIGINT; -ALTER TABLE postgate_detached ADD COLUMN detached_post_actor_id INTEGER; -ALTER TABLE postgate_detached ADD COLUMN detached_post_rkey BIGINT; - --- ============================================================================ --- Phase 2: Populate new columns from existing foreign keys (BEFORE dropping posts.id!) --- ============================================================================ - --- Populate posts self-references -UPDATE posts p SET - parent_post_actor_id = parent.actor_id, - parent_post_rkey = parent.rkey -FROM posts parent -WHERE p.parent_post_id = parent.id; - -UPDATE posts p SET - root_post_actor_id = root.actor_id, - root_post_rkey = root.rkey -FROM posts root -WHERE p.root_post_id = root.id; - -UPDATE posts p SET - embedded_post_actor_id = embedded.actor_id, - embedded_post_rkey = embedded.rkey -FROM posts embedded -WHERE p.embedded_post_id = embedded.id; - --- Populate post_likes (post reference) -UPDATE post_likes pl SET - post_actor_id = p.actor_id, - post_rkey = p.rkey -FROM posts p -WHERE pl.post_id = p.id; - --- Populate post_likes (via_repost reference) -UPDATE post_likes pl SET - via_repost_actor_id = r.actor_id, - via_repost_rkey = r.rkey -FROM reposts r -WHERE pl.via_repost_id = r.id; - --- Populate reposts (post reference) -UPDATE reposts r SET - post_actor_id = p.actor_id, - post_rkey = p.rkey -FROM posts p -WHERE r.post_id = p.id; - --- Populate reposts (via_repost self-reference) -UPDATE reposts r SET - via_repost_actor_id = vr.actor_id, - via_repost_rkey = vr.rkey -FROM reposts vr -WHERE r.via_repost_id = vr.id; - --- Populate post_aggregate_stats -UPDATE post_aggregate_stats pas SET - post_actor_id = p.actor_id, - post_rkey = p.rkey -FROM posts p -WHERE pas.post_id = p.id; - --- Populate threadgates -UPDATE threadgates tg SET - post_actor_id = p.actor_id, - post_rkey = p.rkey -FROM posts p -WHERE tg.post_id = p.id; - --- Populate postgates -UPDATE postgates pg SET - post_actor_id = p.actor_id, - post_rkey = p.rkey -FROM posts p -WHERE pg.post_id = p.id; - --- Populate profiles (only the actor's own posts) -UPDATE profiles pr SET - pinned_post_rkey = p.rkey -FROM posts p -WHERE pr.pinned_post_id = p.id - AND p.actor_id = pr.actor_id; - --- Populate statuses (embed_post reference) -UPDATE statuses s SET - embed_post_actor_id = p.actor_id, - embed_post_rkey = p.rkey -FROM posts p -WHERE s.embed_post_id = p.id; - --- Populate threadgate_allowed_lists -UPDATE threadgate_allowed_lists tal SET - post_actor_id = tg.post_actor_id, - post_rkey = tg.post_rkey -FROM threadgates tg -WHERE tal.threadgate_id = tg.id; - --- Populate threadgate_hidden_replies -UPDATE threadgate_hidden_replies thr SET - post_actor_id = tg.post_actor_id, - post_rkey = tg.post_rkey, - hidden_post_actor_id = p.actor_id, - hidden_post_rkey = p.rkey -FROM threadgates tg, posts p -WHERE thr.threadgate_id = tg.id - AND thr.post_id = p.id; - --- Populate postgate_detached -UPDATE postgate_detached pd SET - post_actor_id = pg.post_actor_id, - post_rkey = pg.post_rkey, - detached_post_actor_id = p.actor_id, - detached_post_rkey = p.rkey -FROM postgates pg, posts p -WHERE pd.postgate_id = pg.id - AND pd.detached_post_id = p.id; - --- ============================================================================ --- Phase 3: Drop all foreign key constraints --- ============================================================================ - --- Child tables first (depend on parent tables) -ALTER TABLE threadgate_allowed_lists DROP CONSTRAINT IF EXISTS threadgate_allowed_lists_threadgate_id_fkey; -ALTER TABLE threadgate_allowed_lists DROP CONSTRAINT IF EXISTS threadgate_allowed_lists_list_id_fkey; -ALTER TABLE threadgate_hidden_replies DROP CONSTRAINT IF EXISTS threadgate_hidden_replies_threadgate_id_fkey; -ALTER TABLE postgate_detached DROP CONSTRAINT IF EXISTS postgate_detached_postgate_id_fkey; - --- Main tables -ALTER TABLE post_likes DROP CONSTRAINT IF EXISTS post_likes_post_id_fkey; -ALTER TABLE post_likes DROP CONSTRAINT IF EXISTS post_likes_via_repost_id_fkey; -ALTER TABLE reposts DROP CONSTRAINT IF EXISTS reposts_post_id_fkey; -ALTER TABLE reposts DROP CONSTRAINT IF EXISTS reposts_via_repost_id_fkey; -ALTER TABLE post_aggregate_stats DROP CONSTRAINT IF EXISTS post_aggregate_stats_post_id_fkey; -ALTER TABLE threadgates DROP CONSTRAINT IF EXISTS threadgates_post_id_fkey; -ALTER TABLE postgates DROP CONSTRAINT IF EXISTS postgates_post_id_fkey; -ALTER TABLE profiles DROP CONSTRAINT IF EXISTS profiles_pinned_post_id_fkey; -ALTER TABLE statuses DROP CONSTRAINT IF EXISTS statuses_embed_post_id_fkey; -ALTER TABLE posts DROP CONSTRAINT IF EXISTS posts_parent_post_id_fkey; -ALTER TABLE posts DROP CONSTRAINT IF EXISTS posts_root_post_id_fkey; -ALTER TABLE posts DROP CONSTRAINT IF EXISTS posts_embedded_post_id_fkey; - --- ============================================================================ --- Phase 4: Drop old columns and primary keys --- ============================================================================ - --- Drop old primary keys that will be replaced -ALTER TABLE threadgates DROP CONSTRAINT IF EXISTS threadgates_pkey; -ALTER TABLE postgates DROP CONSTRAINT IF EXISTS postgates_pkey; -ALTER TABLE post_aggregate_stats DROP CONSTRAINT IF EXISTS post_aggregate_stats_pkey; -ALTER TABLE posts DROP CONSTRAINT posts_pkey; - --- Drop primary keys from child tables -ALTER TABLE threadgate_allowed_lists DROP CONSTRAINT IF EXISTS threadgate_allowed_lists_pkey; -ALTER TABLE threadgate_hidden_replies DROP CONSTRAINT IF EXISTS threadgate_hidden_replies_pkey; -ALTER TABLE postgate_detached DROP CONSTRAINT IF EXISTS postgate_detached_pkey; - --- Drop unique constraints that will change -ALTER TABLE post_likes DROP CONSTRAINT IF EXISTS post_likes_actor_id_post_id_key; -ALTER TABLE reposts DROP CONSTRAINT IF EXISTS reposts_actor_id_post_id_key; - --- Drop old columns (ORDER MATTERS - children before parents) -ALTER TABLE threadgate_allowed_lists DROP COLUMN threadgate_id; -ALTER TABLE threadgate_hidden_replies DROP COLUMN threadgate_id; -ALTER TABLE threadgate_hidden_replies DROP COLUMN post_id; -ALTER TABLE postgate_detached DROP COLUMN postgate_id; -ALTER TABLE postgate_detached DROP COLUMN detached_post_id; - -ALTER TABLE posts DROP COLUMN parent_post_id; -ALTER TABLE posts DROP COLUMN root_post_id; -ALTER TABLE posts DROP COLUMN embedded_post_id; -ALTER TABLE posts DROP COLUMN id; -- Drop synthetic id - -ALTER TABLE post_likes DROP COLUMN post_id; -ALTER TABLE post_likes DROP COLUMN via_repost_id; - -ALTER TABLE reposts DROP COLUMN post_id; -ALTER TABLE reposts DROP COLUMN via_repost_id; -ALTER TABLE reposts DROP COLUMN id; -- Can now drop synthetic id! - -ALTER TABLE post_aggregate_stats DROP COLUMN post_id; -ALTER TABLE threadgates DROP COLUMN id; -ALTER TABLE threadgates DROP COLUMN post_id; -ALTER TABLE postgates DROP COLUMN id; -ALTER TABLE postgates DROP COLUMN post_id; -ALTER TABLE profiles DROP COLUMN pinned_post_id; -ALTER TABLE statuses DROP COLUMN embed_post_id; - --- ============================================================================ --- Phase 5: Create new primary keys --- ============================================================================ - --- Main tables -ALTER TABLE posts ADD PRIMARY KEY (actor_id, rkey); -ALTER TABLE reposts ADD PRIMARY KEY (actor_id, rkey); -- NEW! Was id before -ALTER TABLE post_aggregate_stats ADD PRIMARY KEY (post_actor_id, post_rkey); -ALTER TABLE threadgates ADD PRIMARY KEY (post_actor_id, post_rkey); -ALTER TABLE postgates ADD PRIMARY KEY (post_actor_id, post_rkey); - --- Child tables -ALTER TABLE threadgate_allowed_lists ADD PRIMARY KEY (post_actor_id, post_rkey, list_id); -ALTER TABLE threadgate_hidden_replies ADD PRIMARY KEY (post_actor_id, post_rkey, hidden_post_actor_id, hidden_post_rkey); -ALTER TABLE postgate_detached ADD PRIMARY KEY (post_actor_id, post_rkey, detached_post_actor_id, detached_post_rkey); - --- ============================================================================ --- Phase 6: Create indexes on new natural key columns --- ============================================================================ - --- Posts table self-reference indexes -DROP INDEX IF EXISTS idx_posts_parent; -CREATE INDEX idx_posts_parent - ON posts(parent_post_actor_id, parent_post_rkey) - WHERE parent_post_actor_id IS NOT NULL; - -DROP INDEX IF EXISTS idx_posts_root; -CREATE INDEX idx_posts_root - ON posts(root_post_actor_id, root_post_rkey) - WHERE root_post_actor_id IS NOT NULL; - -DROP INDEX IF EXISTS idx_posts_embedded_post_id; -CREATE INDEX idx_posts_embedded_post - ON posts(embedded_post_actor_id, embedded_post_rkey) - WHERE embedded_post_actor_id IS NOT NULL; - --- post_likes indexes -DROP INDEX IF EXISTS idx_post_likes_post_id; -CREATE INDEX idx_post_likes_post_actor_id_post_rkey - ON post_likes(post_actor_id, post_rkey); - --- reposts indexes -DROP INDEX IF EXISTS idx_reposts_post_id; -CREATE INDEX idx_reposts_post_actor_id_post_rkey - ON reposts(post_actor_id, post_rkey); - --- ============================================================================ --- Phase 7: Recreate unique constraints --- ============================================================================ - --- reposts: One repost per actor per post -ALTER TABLE reposts ADD CONSTRAINT reposts_actor_id_post_actor_id_post_rkey_key - UNIQUE (actor_id, post_actor_id, post_rkey); - --- post_likes: TimescaleDB hypertables require partition key in ALL unique constraints --- post_likes is partitioned by rkey, so we CANNOT add UNIQUE(actor_id, post_actor_id, post_rkey) --- Application must ensure one actor doesn't like the same post twice --- (Removed: UNIQUE constraint not possible) - --- ============================================================================ --- Phase 8: Add foreign key constraints (where allowed) --- ============================================================================ - --- NOTE: TimescaleDB FK constraint rules (recent versions): --- - Hypertable → Hypertable: NOT ALLOWED (posts → posts, post_likes → posts) --- - Regular → Hypertable: ALLOWED (reposts → posts, threadgates → posts) --- - Hypertable → Regular: ALLOWED (posts → actors, post_likes → actors) - --- Regular tables CAN have FK constraints to posts (a hypertable) --- This is explicitly supported in recent TimescaleDB versions -ALTER TABLE reposts ADD CONSTRAINT reposts_post_fkey - FOREIGN KEY (post_actor_id, post_rkey) - REFERENCES posts(actor_id, rkey) - ON DELETE CASCADE; - --- reposts self-reference for via_repost (quote-repost-of-repost) -ALTER TABLE reposts ADD CONSTRAINT reposts_via_repost_fkey - FOREIGN KEY (via_repost_actor_id, via_repost_rkey) - REFERENCES reposts(actor_id, rkey) - ON DELETE CASCADE; - -ALTER TABLE post_aggregate_stats ADD CONSTRAINT post_aggregate_stats_post_fkey - FOREIGN KEY (post_actor_id, post_rkey) - REFERENCES posts(actor_id, rkey) - ON DELETE CASCADE; - -ALTER TABLE threadgates ADD CONSTRAINT threadgates_post_fkey - FOREIGN KEY (post_actor_id, post_rkey) - REFERENCES posts(actor_id, rkey) - ON DELETE CASCADE; - -ALTER TABLE postgates ADD CONSTRAINT postgates_post_fkey - FOREIGN KEY (post_actor_id, post_rkey) - REFERENCES posts(actor_id, rkey) - ON DELETE CASCADE; - -ALTER TABLE profiles ADD CONSTRAINT profiles_pinned_post_fkey - FOREIGN KEY (actor_id, pinned_post_rkey) - REFERENCES posts(actor_id, rkey) - ON DELETE SET NULL; - --- Child tables reference their parent gate tables -ALTER TABLE threadgate_allowed_lists ADD CONSTRAINT threadgate_allowed_lists_threadgate_fkey - FOREIGN KEY (post_actor_id, post_rkey) - REFERENCES threadgates(post_actor_id, post_rkey) - ON DELETE CASCADE; - --- Restore FK to lists table -ALTER TABLE threadgate_allowed_lists ADD CONSTRAINT threadgate_allowed_lists_list_id_fkey - FOREIGN KEY (list_id) - REFERENCES lists(id) - ON DELETE CASCADE; - -ALTER TABLE threadgate_hidden_replies ADD CONSTRAINT threadgate_hidden_replies_threadgate_fkey - FOREIGN KEY (post_actor_id, post_rkey) - REFERENCES threadgates(post_actor_id, post_rkey) - ON DELETE CASCADE; - -ALTER TABLE postgate_detached ADD CONSTRAINT postgate_detached_postgate_fkey - FOREIGN KEY (post_actor_id, post_rkey) - REFERENCES postgates(post_actor_id, post_rkey) - ON DELETE CASCADE; - --- Referential integrity WITHOUT database FKs (application-level enforcement required): --- 1. posts.parent_post_{actor_id,rkey} -> posts(actor_id,rkey) - NO FK (hypertable → hypertable not allowed) --- 2. posts.root_post_{actor_id,rkey} -> posts(actor_id,rkey) - NO FK (hypertable → hypertable not allowed) --- 3. posts.embedded_post_{actor_id,rkey} -> posts(actor_id,rkey) - NO FK (hypertable → hypertable not allowed) --- 4. post_likes.{post_actor_id,post_rkey} -> posts(actor_id,rkey) - NO FK (hypertable → hypertable not allowed) --- 5. post_likes.{via_repost_actor_id,via_repost_rkey} -> reposts(actor_id,rkey) - NO FK (would break if reposts becomes hypertable) --- 6. threadgate_hidden_replies.{hidden_post_*} -> posts - NO FK (could be added, but optional/not critical) --- 7. postgate_detached.{detached_post_*} -> posts - NO FK (could be added, but optional/not critical) diff --git a/migrations/2025-11-18-213307_update_post_aggregate_stats_to_natural_keys/down.sql b/migrations/2025-11-18-213307_update_post_aggregate_stats_to_natural_keys/down.sql deleted file mode 100644 index 22bffa30..00000000 --- a/migrations/2025-11-18-213307_update_post_aggregate_stats_to_natural_keys/down.sql +++ /dev/null @@ -1,18 +0,0 @@ --- Revert post_aggregate_stats back to using synthetic post_id - -DROP TABLE IF EXISTS post_aggregate_stats; - -CREATE TABLE post_aggregate_stats ( - post_id BIGINT NOT NULL REFERENCES posts(id) ON DELETE CASCADE, - stat_type post_stat_type NOT NULL, - value INTEGER NOT NULL, - updated_at TIMESTAMP WITH TIME ZONE NOT NULL DEFAULT now(), - - PRIMARY KEY (post_id, stat_type) -); - -CREATE INDEX idx_post_aggregate_stats_updated_at ON post_aggregate_stats(updated_at DESC) - WHERE value > 0; - -CREATE INDEX idx_post_aggregate_stats_hot_content ON post_aggregate_stats(stat_type, updated_at DESC, value DESC) - WHERE value > 10; diff --git a/migrations/2025-11-18-213307_update_post_aggregate_stats_to_natural_keys/up.sql b/migrations/2025-11-18-213307_update_post_aggregate_stats_to_natural_keys/up.sql deleted file mode 100644 index 2c951e8c..00000000 --- a/migrations/2025-11-18-213307_update_post_aggregate_stats_to_natural_keys/up.sql +++ /dev/null @@ -1,37 +0,0 @@ --- Update post_aggregate_stats to use natural keys (actor_id, rkey) instead of synthetic post_id --- --- This migration changes the primary key from post_id to (actor_id, rkey) to align with --- the natural key migration across the codebase. - --- Step 1: Drop the existing table (since it references posts(id) which will change) --- This is safe because parakeet-index loads stats into memory on startup -DROP TABLE IF EXISTS post_aggregate_stats; - --- Step 2: Recreate with natural key structure -CREATE TABLE post_aggregate_stats ( - actor_id INTEGER NOT NULL, - rkey BIGINT NOT NULL, - stat_type post_stat_type NOT NULL, - value INTEGER NOT NULL, - updated_at TIMESTAMP WITH TIME ZONE NOT NULL DEFAULT now(), - - PRIMARY KEY (actor_id, rkey, stat_type), - - -- Foreign key constraint to posts using natural key - FOREIGN KEY (actor_id, rkey) REFERENCES posts(actor_id, rkey) ON DELETE CASCADE -); - --- Step 3: Recreate indexes for query performance -CREATE INDEX idx_post_aggregate_stats_updated_at ON post_aggregate_stats(updated_at DESC) - WHERE value > 0; - --- Index for hot content queries (recently updated high-value stats) -CREATE INDEX idx_post_aggregate_stats_hot_content ON post_aggregate_stats(stat_type, updated_at DESC, value DESC) - WHERE value > 10; - -COMMENT ON TABLE post_aggregate_stats IS 'Post aggregate statistics using natural keys (actor_id, rkey)'; -COMMENT ON COLUMN post_aggregate_stats.actor_id IS 'Actor who created the post (natural key part 1)'; -COMMENT ON COLUMN post_aggregate_stats.rkey IS 'Record key as i64 (TID converted) - natural key part 2'; -COMMENT ON COLUMN post_aggregate_stats.stat_type IS 'Type of statistic (like, reply, repost, quote)'; -COMMENT ON COLUMN post_aggregate_stats.value IS 'Current count value'; -COMMENT ON COLUMN post_aggregate_stats.updated_at IS 'Last time value changed (NOT last write time)'; diff --git a/migrations/2025-11-18-225351_migrate_thread_mutes_to_natural_keys/down.sql b/migrations/2025-11-18-225351_migrate_thread_mutes_to_natural_keys/down.sql deleted file mode 100644 index 498bd9c1..00000000 --- a/migrations/2025-11-18-225351_migrate_thread_mutes_to_natural_keys/down.sql +++ /dev/null @@ -1,18 +0,0 @@ --- Revert thread_mutes migration back to synthetic root_post_id --- Note: This is a destructive rollback - existing thread mute data will be lost - --- Drop the foreign key constraint -ALTER TABLE thread_mutes DROP CONSTRAINT IF EXISTS thread_mutes_root_post_fkey; - --- Drop the new primary key -ALTER TABLE thread_mutes DROP CONSTRAINT thread_mutes_pkey; - --- Add back the old root_post_id column -ALTER TABLE thread_mutes ADD COLUMN root_post_id BIGINT NOT NULL DEFAULT 0; - --- Drop the natural key columns -ALTER TABLE thread_mutes DROP COLUMN root_post_actor_id; -ALTER TABLE thread_mutes DROP COLUMN root_post_rkey; - --- Restore old primary key -ALTER TABLE thread_mutes ADD PRIMARY KEY (actor_id, root_post_id); diff --git a/migrations/2025-11-18-225351_migrate_thread_mutes_to_natural_keys/up.sql b/migrations/2025-11-18-225351_migrate_thread_mutes_to_natural_keys/up.sql deleted file mode 100644 index 0aad3c79..00000000 --- a/migrations/2025-11-18-225351_migrate_thread_mutes_to_natural_keys/up.sql +++ /dev/null @@ -1,29 +0,0 @@ --- Migrate thread_mutes table from synthetic root_post_id to natural keys (root_post_actor_id, root_post_rkey) --- This aligns with the posts table migration to natural keys - --- Add new natural key columns -ALTER TABLE thread_mutes ADD COLUMN root_post_actor_id INTEGER; -ALTER TABLE thread_mutes ADD COLUMN root_post_rkey BIGINT; - --- Populate new columns from existing root_post_id (if any data exists) --- Note: Since posts table no longer has 'id' column, this migration assumes thread_mutes is currently empty --- or that root_post_id values are no longer valid references - --- Drop the old primary key first (before dropping the column it depends on) -ALTER TABLE thread_mutes DROP CONSTRAINT thread_mutes_pkey; - --- Drop the old root_post_id column -ALTER TABLE thread_mutes DROP COLUMN root_post_id; - --- Make the new columns NOT NULL -ALTER TABLE thread_mutes ALTER COLUMN root_post_actor_id SET NOT NULL; -ALTER TABLE thread_mutes ALTER COLUMN root_post_rkey SET NOT NULL; - --- Create new primary key with natural keys -ALTER TABLE thread_mutes ADD PRIMARY KEY (actor_id, root_post_actor_id, root_post_rkey); - --- Add foreign key constraint to posts table using natural keys -ALTER TABLE thread_mutes ADD CONSTRAINT thread_mutes_root_post_fkey - FOREIGN KEY (root_post_actor_id, root_post_rkey) - REFERENCES posts(actor_id, rkey) - ON DELETE CASCADE; diff --git a/migrations/2025-11-18-232413_migrate_bookmarks_to_natural_keys/down.sql b/migrations/2025-11-18-232413_migrate_bookmarks_to_natural_keys/down.sql deleted file mode 100644 index df3fb26d..00000000 --- a/migrations/2025-11-18-232413_migrate_bookmarks_to_natural_keys/down.sql +++ /dev/null @@ -1,10 +0,0 @@ --- Reverse the migration - -DROP INDEX IF EXISTS idx_bookmarks_post; -ALTER TABLE bookmarks DROP CONSTRAINT IF EXISTS bookmarks_post_fkey; -ALTER TABLE bookmarks DROP COLUMN IF EXISTS post_rkey; -ALTER TABLE bookmarks DROP COLUMN IF EXISTS post_actor_id; - --- Restore old discriminated union columns -ALTER TABLE bookmarks ADD COLUMN subject_type bookmark_subject_type; -ALTER TABLE bookmarks ADD COLUMN subject_id BIGINT; diff --git a/migrations/2025-11-18-232413_migrate_bookmarks_to_natural_keys/up.sql b/migrations/2025-11-18-232413_migrate_bookmarks_to_natural_keys/up.sql deleted file mode 100644 index 6cfb8416..00000000 --- a/migrations/2025-11-18-232413_migrate_bookmarks_to_natural_keys/up.sql +++ /dev/null @@ -1,22 +0,0 @@ --- Migrate bookmarks to use natural keys for posts --- Bookmarks can only reference posts (not feedgens, lists, etc.) - --- Add new natural key columns -ALTER TABLE bookmarks ADD COLUMN post_actor_id INTEGER; -ALTER TABLE bookmarks ADD COLUMN post_rkey BIGINT; - --- Drop old discriminated union columns (subject_type, subject_id) -ALTER TABLE bookmarks DROP COLUMN subject_type; -ALTER TABLE bookmarks DROP COLUMN subject_id; - --- Make new columns NOT NULL (all bookmarks must reference a post) -ALTER TABLE bookmarks ALTER COLUMN post_actor_id SET NOT NULL; -ALTER TABLE bookmarks ALTER COLUMN post_rkey SET NOT NULL; - --- Add foreign key constraint to posts using natural keys -ALTER TABLE bookmarks ADD CONSTRAINT bookmarks_post_fkey - FOREIGN KEY (post_actor_id, post_rkey) - REFERENCES posts(actor_id, rkey) ON DELETE CASCADE; - --- Create index for better join performance -CREATE INDEX idx_bookmarks_post ON bookmarks(post_actor_id, post_rkey); diff --git a/migrations/2025-11-19-011902_drop_natural_key_fk_constraints/down.sql b/migrations/2025-11-19-011902_drop_natural_key_fk_constraints/down.sql deleted file mode 100644 index 6f3cf69d..00000000 --- a/migrations/2025-11-19-011902_drop_natural_key_fk_constraints/down.sql +++ /dev/null @@ -1,14 +0,0 @@ --- This file should undo anything in `up.sql` --- Re-add FK constraints (though this may fail if there are orphaned references) - --- Re-add FK constraint on reposts -> posts -ALTER TABLE reposts ADD CONSTRAINT reposts_post_fkey - FOREIGN KEY (post_actor_id, post_rkey) - REFERENCES posts(actor_id, rkey) - ON DELETE CASCADE; - --- Re-add self-referencing FK constraint on reposts -> reposts (via_repost) -ALTER TABLE reposts ADD CONSTRAINT reposts_via_repost_fkey - FOREIGN KEY (via_repost_actor_id, via_repost_rkey) - REFERENCES reposts(actor_id, rkey) - ON DELETE CASCADE; diff --git a/migrations/2025-11-19-011902_drop_natural_key_fk_constraints/up.sql b/migrations/2025-11-19-011902_drop_natural_key_fk_constraints/up.sql deleted file mode 100644 index ee2f026e..00000000 --- a/migrations/2025-11-19-011902_drop_natural_key_fk_constraints/up.sql +++ /dev/null @@ -1,14 +0,0 @@ --- Drop FK constraints on natural key references --- With natural keys (actor_id, rkey), we don't need FK constraints since: --- 1. We can reference posts/reposts by their natural key even if they don't exist yet --- 2. TimescaleDB hypertables don't support compound FK constraints efficiently --- 3. This eliminates the need for stub creation during indexing - --- Drop FK constraint on reposts -> posts -ALTER TABLE reposts DROP CONSTRAINT IF EXISTS reposts_post_fkey; - --- Drop self-referencing FK constraint on reposts -> reposts (via_repost) -ALTER TABLE reposts DROP CONSTRAINT IF EXISTS reposts_via_repost_fkey; - --- Note: post_likes already has no FK constraint on (post_actor_id, post_rkey) --- Note: We keep actor_id FK constraints since actors still use synthetic IDs diff --git a/migrations/2025-11-19-015513_drop_post_fk_constraints/down.sql b/migrations/2025-11-19-015513_drop_post_fk_constraints/down.sql deleted file mode 100644 index 4ca58752..00000000 --- a/migrations/2025-11-19-015513_drop_post_fk_constraints/down.sql +++ /dev/null @@ -1,17 +0,0 @@ --- Restore FK constraints (down migration) --- Note: This will fail if there is data that violates the constraints - -ALTER TABLE bookmarks ADD CONSTRAINT bookmarks_post_fkey - FOREIGN KEY (post_actor_id, post_rkey) REFERENCES posts(actor_id, rkey) ON DELETE CASCADE; - -ALTER TABLE postgates ADD CONSTRAINT postgates_post_fkey - FOREIGN KEY (post_actor_id, post_rkey) REFERENCES posts(actor_id, rkey) ON DELETE CASCADE; - -ALTER TABLE threadgates ADD CONSTRAINT threadgates_post_fkey - FOREIGN KEY (post_actor_id, post_rkey) REFERENCES posts(actor_id, rkey) ON DELETE CASCADE; - -ALTER TABLE profiles ADD CONSTRAINT profiles_pinned_post_fkey - FOREIGN KEY (actor_id, pinned_post_rkey) REFERENCES posts(actor_id, rkey) ON DELETE SET NULL; - -ALTER TABLE thread_mutes ADD CONSTRAINT thread_mutes_root_post_fkey - FOREIGN KEY (root_post_actor_id, root_post_rkey) REFERENCES posts(actor_id, rkey) ON DELETE CASCADE; diff --git a/migrations/2025-11-19-015513_drop_post_fk_constraints/up.sql b/migrations/2025-11-19-015513_drop_post_fk_constraints/up.sql deleted file mode 100644 index cb3f77f6..00000000 --- a/migrations/2025-11-19-015513_drop_post_fk_constraints/up.sql +++ /dev/null @@ -1,13 +0,0 @@ --- Drop FK constraints that reference posts table --- These constraints prevent backfilling when posts from other actors don't exist yet --- With natural keys (actor_id, rkey), we can trust the data without FK validation - --- Drop main table FK constraints -ALTER TABLE bookmarks DROP CONSTRAINT IF EXISTS bookmarks_post_fkey; -ALTER TABLE postgates DROP CONSTRAINT IF EXISTS postgates_post_fkey; -ALTER TABLE threadgates DROP CONSTRAINT IF EXISTS threadgates_post_fkey; -ALTER TABLE profiles DROP CONSTRAINT IF EXISTS profiles_pinned_post_fkey; -ALTER TABLE thread_mutes DROP CONSTRAINT IF EXISTS thread_mutes_root_post_fkey; - --- Note: TimescaleDB hypertable chunk constraints are managed automatically --- and will be dropped when the main table constraint is dropped diff --git a/migrations/2025-11-19-023830_optimize_viewer_states_indexes_and_constraints/down.sql b/migrations/2025-11-19-023830_optimize_viewer_states_indexes_and_constraints/down.sql deleted file mode 100644 index 4fa3c931..00000000 --- a/migrations/2025-11-19-023830_optimize_viewer_states_indexes_and_constraints/down.sql +++ /dev/null @@ -1,15 +0,0 @@ --- Revert viewer states optimizations - --- Drop the viewer-first composite indexes -DROP INDEX IF EXISTS idx_bookmarks_viewer_post; -DROP INDEX IF EXISTS idx_reposts_viewer_post; -DROP INDEX IF EXISTS idx_post_likes_viewer_post; - --- Revert NOT NULL constraints on reposts --- Note: This doesn't restore NULL values, just removes the constraint -ALTER TABLE reposts ALTER COLUMN post_rkey DROP NOT NULL; -ALTER TABLE reposts ALTER COLUMN post_actor_id DROP NOT NULL; - --- Revert NOT NULL constraints on post_likes -ALTER TABLE post_likes ALTER COLUMN post_rkey DROP NOT NULL; -ALTER TABLE post_likes ALTER COLUMN post_actor_id DROP NOT NULL; diff --git a/migrations/2025-11-19-023830_optimize_viewer_states_indexes_and_constraints/up.sql b/migrations/2025-11-19-023830_optimize_viewer_states_indexes_and_constraints/up.sql deleted file mode 100644 index 2d0f2ed2..00000000 --- a/migrations/2025-11-19-023830_optimize_viewer_states_indexes_and_constraints/up.sql +++ /dev/null @@ -1,45 +0,0 @@ --- Optimize viewer states queries with NOT NULL constraints and viewer-first composite indexes --- --- Changes: --- 1. Set post_actor_id and post_rkey to NOT NULL on post_likes and reposts --- (bookmarks already has them as NOT NULL) --- 2. Add composite indexes with viewer_id first for efficient viewer-specific lookups --- 3. These enable PostgreSQL to: --- - Skip NULL checks in index scans --- - Use smaller, more efficient indexes --- - Seek directly to viewer's records before filtering by post - --- Step 1: Set NOT NULL constraints on post_likes --- These should never be NULL since they reference the post being liked -ALTER TABLE post_likes ALTER COLUMN post_actor_id SET NOT NULL; -ALTER TABLE post_likes ALTER COLUMN post_rkey SET NOT NULL; - --- Step 2: Set NOT NULL constraints on reposts --- These should never be NULL since they reference the post being reposted -ALTER TABLE reposts ALTER COLUMN post_actor_id SET NOT NULL; -ALTER TABLE reposts ALTER COLUMN post_rkey SET NOT NULL; - --- Step 3: Add viewer-first composite indexes for efficient viewer state lookups --- Pattern: (viewer_actor_id, post_actor_id, post_rkey) --- This allows PostgreSQL to: --- 1. Seek directly to the viewer's records using actor_id --- 2. Filter by specific posts within that viewer's subset --- Much faster than: seeking by post first, then filtering by viewer --- --- Note: TimescaleDB hypertables don't support CONCURRENTLY, but index creation --- is generally fast for hypertables as it operates on chunks - --- Post likes: viewer lookups are very common in getAuthorFeed -CREATE INDEX idx_post_likes_viewer_post - ON post_likes (actor_id, post_actor_id, post_rkey); - --- Reposts: same access pattern (not a hypertable, but removing CONCURRENTLY for consistency) -CREATE INDEX idx_reposts_viewer_post - ON reposts (actor_id, post_actor_id, post_rkey); - --- Bookmarks: same access pattern (not a hypertable) -CREATE INDEX idx_bookmarks_viewer_post - ON bookmarks (actor_id, post_actor_id, post_rkey); - --- Note: postgates already has (post_actor_id, post_rkey) as PK which is sufficient --- since we don't filter by viewer on postgates (they're post-level, not viewer-specific) diff --git a/migrations/2025-11-19-025200_enable_chunk_skipping_for_timeline_queries/down.sql b/migrations/2025-11-19-025200_enable_chunk_skipping_for_timeline_queries/down.sql deleted file mode 100644 index 19c5478c..00000000 --- a/migrations/2025-11-19-025200_enable_chunk_skipping_for_timeline_queries/down.sql +++ /dev/null @@ -1,11 +0,0 @@ --- Disable chunk skipping for timeline queries - --- Disable chunk skipping on specific columns -SELECT disable_chunk_skipping('post_likes', 'actor_id'); -SELECT disable_chunk_skipping('posts', 'actor_id'); - --- Disable chunk skipping globally -ALTER SYSTEM SET timescaledb.enable_chunk_skipping = off; - --- Reload configuration to apply the setting -SELECT pg_reload_conf(); diff --git a/migrations/2025-11-19-025200_enable_chunk_skipping_for_timeline_queries/up.sql b/migrations/2025-11-19-025200_enable_chunk_skipping_for_timeline_queries/up.sql deleted file mode 100644 index 8f5ba7d4..00000000 --- a/migrations/2025-11-19-025200_enable_chunk_skipping_for_timeline_queries/up.sql +++ /dev/null @@ -1,38 +0,0 @@ --- Enable TimescaleDB chunk skipping for timeline queries --- --- Chunk skipping tracks min/max values of actor_id per chunk, allowing PostgreSQL --- to skip entire chunks when querying for a specific actor's timeline. --- --- This is superior to a traditional index on (actor_id, rkey DESC) because: --- 1. No large index overhead (uses lightweight per-chunk statistics) --- 2. Automatically excludes chunks where actor_id doesn't exist --- 3. Works with the existing time-based partitioning --- 4. Perfect for correlated columns (actor_id correlates with rkey/time) --- --- Expected performance: 50-90% reduction in chunks scanned for timeline queries - --- Step 1: Enable chunk skipping globally in postgresql.conf --- This persists across PostgreSQL restarts -ALTER SYSTEM SET timescaledb.enable_chunk_skipping = on; - --- Step 2: Reload configuration to apply the setting immediately -SELECT pg_reload_conf(); - --- Step 3: Enable chunk skipping on actor_id for posts table --- This allows queries like "SELECT * FROM posts WHERE actor_id = X ORDER BY rkey DESC" --- to skip chunks that don't contain that actor's posts -SELECT enable_chunk_skipping('posts', 'actor_id'); - --- Step 4: Enable chunk skipping on actor_id for post_likes table --- This allows viewer state queries to skip chunks without the viewer's likes -SELECT enable_chunk_skipping('post_likes', 'actor_id'); - --- Note: Chunk skipping only applies to NEW chunks compressed AFTER enabling this feature. --- Existing chunks won't benefit until they're recompressed. --- --- To verify chunk skipping is working, check the chunk_column_stats catalog: --- SELECT * FROM _timescaledb_catalog.chunk_column_stats --- WHERE hypertable_id IN ( --- SELECT id FROM _timescaledb_catalog.hypertable --- WHERE table_name IN ('posts', 'post_likes') --- ); diff --git a/migrations/2025-11-19-025652_drop_post_aggregate_stats_fk_and_compress_posts/down.sql b/migrations/2025-11-19-025652_drop_post_aggregate_stats_fk_and_compress_posts/down.sql deleted file mode 100644 index 610ef524..00000000 --- a/migrations/2025-11-19-025652_drop_post_aggregate_stats_fk_and_compress_posts/down.sql +++ /dev/null @@ -1,35 +0,0 @@ --- Revert FK drop and decompression --- --- WARNING: This will decompress all posts chunks, which may take significant time --- and storage space. Decompression is generally not recommended in production. - --- Step 1: Decompress all posts chunks -DO $$ -DECLARE - chunk_record RECORD; - decompressed_count INTEGER := 0; -BEGIN - FOR chunk_record IN - SELECT chunk_schema, chunk_name - FROM timescaledb_information.chunks - WHERE hypertable_name = 'posts' - AND is_compressed - ORDER BY chunk_name - LOOP - EXECUTE format('SELECT decompress_chunk(%L)', - chunk_record.chunk_schema || '.' || chunk_record.chunk_name); - decompressed_count := decompressed_count + 1; - - IF decompressed_count % 10 = 0 THEN - RAISE NOTICE 'Decompressed % posts chunks...', decompressed_count; - END IF; - END LOOP; - - RAISE NOTICE 'Decompression complete: % posts chunks decompressed', decompressed_count; -END $$; - --- Step 2: Re-add the foreign key constraint --- Note: This will fail if there are orphaned rows in post_aggregate_stats -ALTER TABLE post_aggregate_stats - ADD CONSTRAINT post_aggregate_stats_actor_id_rkey_fkey - FOREIGN KEY (actor_id, rkey) REFERENCES posts(actor_id, rkey); diff --git a/migrations/2025-11-19-025652_drop_post_aggregate_stats_fk_and_compress_posts/up.sql b/migrations/2025-11-19-025652_drop_post_aggregate_stats_fk_and_compress_posts/up.sql deleted file mode 100644 index 2e1ebb90..00000000 --- a/migrations/2025-11-19-025652_drop_post_aggregate_stats_fk_and_compress_posts/up.sql +++ /dev/null @@ -1,61 +0,0 @@ --- Drop FK constraint and compress posts chunks to enable chunk skipping --- --- Background: --- TimescaleDB cannot compress chunks that have foreign keys pointing into them. --- The post_aggregate_stats table has an FK to posts, preventing compression. --- --- Solution: --- 1. Drop the FK constraint (app-level integrity is sufficient) --- 2. Compress all posts chunks to enable chunk skipping on actor_id --- --- Expected benefits: --- - Timeline queries skip 90%+ of chunks (only scan chunks with that actor's posts) --- - Stabilizes query time variance (currently 28-125ms → expected 10-40ms) --- - Reduces I/O and buffer cache pressure --- - Saves storage with compression - --- Step 1: Drop the foreign key constraint -ALTER TABLE post_aggregate_stats - DROP CONSTRAINT IF EXISTS post_aggregate_stats_actor_id_rkey_fkey; - --- Step 2: Compress all uncompressed posts chunks --- This populates chunk skipping statistics for actor_id -DO $$ -DECLARE - chunk_record RECORD; - compressed_count INTEGER := 0; -BEGIN - FOR chunk_record IN - SELECT chunk_schema, chunk_name - FROM timescaledb_information.chunks - WHERE hypertable_name = 'posts' - AND NOT is_compressed - ORDER BY chunk_name - LOOP - EXECUTE format('SELECT compress_chunk(%L)', - chunk_record.chunk_schema || '.' || chunk_record.chunk_name); - compressed_count := compressed_count + 1; - - -- Log progress every 10 chunks - IF compressed_count % 10 = 0 THEN - RAISE NOTICE 'Compressed % posts chunks...', compressed_count; - END IF; - END LOOP; - - RAISE NOTICE 'Compression complete: % posts chunks compressed', compressed_count; -END $$; - --- Step 3: Verify chunk skipping statistics were populated -DO $$ -DECLARE - stats_count INTEGER; -BEGIN - SELECT COUNT(*) INTO stats_count - FROM _timescaledb_catalog.chunk_column_stats - WHERE hypertable_id = ( - SELECT id FROM _timescaledb_catalog.hypertable WHERE table_name = 'posts' - ) - AND column_name = 'actor_id'; - - RAISE NOTICE 'Chunk skipping statistics: % entries for posts.actor_id', stats_count; -END $$; diff --git a/parakeet-db/src/schema.rs b/parakeet-db/src/schema.rs index 4adb45cc..ce50edd3 100644 --- a/parakeet-db/src/schema.rs +++ b/parakeet-db/src/schema.rs @@ -505,13 +505,13 @@ diesel::table! { use diesel::sql_types::*; use super::sql_types::PostgateRule; - postgates (post_actor_id, post_rkey) { + postgates (actor_id, rkey) { actor_id -> Int4, rkey -> Int8, cid -> Bytea, + post_actor_id -> Nullable, + post_rkey -> Nullable, rules -> Array>, - post_actor_id -> Int4, - post_rkey -> Int8, } } @@ -521,9 +521,9 @@ diesel::table! { use super::sql_types::EmbedType; use super::sql_types::PostStatus; use super::sql_types::PostExtEmbed; + use super::sql_types::PostVideoEmbed; use super::sql_types::PostImageEmbed; use super::sql_types::PostFacetEmbed; - use super::sql_types::PostVideoEmbed; posts (actor_id, rkey) { actor_id -> Int4, @@ -532,12 +532,19 @@ diesel::table! { content -> Nullable, langs -> Array>, tags -> Array>, + tokens -> Nullable>>, + parent_post_actor_id -> Nullable, + parent_post_rkey -> Nullable, + root_post_actor_id -> Nullable, + root_post_rkey -> Nullable, embed_type -> Nullable, embed_subtype -> Nullable, violates_threadgate -> Bool, status -> PostStatus, - tokens -> Nullable>>, ext_embed -> Nullable, + video_embed -> Nullable, + embedded_post_actor_id -> Nullable, + embedded_post_rkey -> Nullable, record_detached -> Nullable, image_1 -> Nullable, image_2 -> Nullable, @@ -552,13 +559,6 @@ diesel::table! { facet_7 -> Nullable, facet_8 -> Nullable, mentions -> Nullable>>, - video_embed -> Nullable, - parent_post_actor_id -> Nullable, - parent_post_rkey -> Nullable, - root_post_actor_id -> Nullable, - root_post_rkey -> Nullable, - embedded_post_actor_id -> Nullable, - embedded_post_rkey -> Nullable, } } @@ -574,11 +574,11 @@ diesel::table! { banner_cid -> Nullable, display_name -> Nullable, description -> Nullable, + pinned_post_rkey -> Nullable, joined_sp_id -> Nullable, pronouns -> Nullable, website -> Nullable, search_vector -> Nullable, - pinned_post_rkey -> Nullable, } } @@ -589,12 +589,11 @@ diesel::table! { reposts (actor_id, rkey) { actor_id -> Int4, rkey -> Int8, - cid -> Bytea, - status -> RepostStatus, post_actor_id -> Int4, post_rkey -> Int8, via_repost_actor_id -> Nullable, via_repost_rkey -> Nullable, + status -> RepostStatus, } } @@ -650,27 +649,27 @@ diesel::table! { created_at -> Timestamptz, status -> StatusType, duration -> Nullable, - thumb_mime_type -> Nullable, - thumb_cid -> Nullable, embed_post_actor_id -> Nullable, embed_post_rkey -> Nullable, + thumb_mime_type -> Nullable, + thumb_cid -> Nullable, } } diesel::table! { thread_mutes (actor_id, root_post_actor_id, root_post_rkey) { actor_id -> Int4, - created_at -> Timestamptz, root_post_actor_id -> Int4, root_post_rkey -> Int8, + created_at -> Timestamptz, } } diesel::table! { threadgate_allowed_lists (post_actor_id, post_rkey, list_id) { - list_id -> Int8, post_actor_id -> Int4, post_rkey -> Int8, + list_id -> Int8, } } @@ -687,13 +686,13 @@ diesel::table! { use diesel::sql_types::*; use super::sql_types::ThreadgateRule; - threadgates (post_actor_id, post_rkey) { + threadgates (actor_id, rkey) { actor_id -> Int4, rkey -> Int8, cid -> Bytea, + post_actor_id -> Nullable, + post_rkey -> Nullable, allow -> Nullable>>, - post_actor_id -> Int4, - post_rkey -> Int8, } } @@ -722,7 +721,6 @@ diesel::joinable!(actor_aggregate_stats -> actors (actor_id)); diesel::joinable!(bookmarks -> actors (actor_id)); diesel::joinable!(chat_decls -> actors (actor_id)); diesel::joinable!(feedgen_likes -> actors (actor_id)); -diesel::joinable!(feedgen_likes -> feedgens (feedgen_id)); diesel::joinable!(labeler_defs -> actors (labeler_actor_id)); diesel::joinable!(labelers -> actors (actor_id)); diesel::joinable!(labels -> actors (labeler_actor_id)); @@ -744,6 +742,7 @@ diesel::joinable!(starterpack_feeds -> starterpacks (starterpack_id)); diesel::joinable!(starterpacks -> lists (list_id)); diesel::joinable!(statuses -> actors (actor_id)); diesel::joinable!(thread_mutes -> actors (actor_id)); +diesel::joinable!(threadgate_allowed_lists -> lists (list_id)); diesel::joinable!(threadgates -> actors (actor_id)); diesel::allow_tables_to_appear_in_same_query!(