From 5a4b6716406a3eeb94b0cf5de74152aa8d92a147 Mon Sep 17 00:00:00 2001 From: Thomas Rademaker Date: Fri, 15 May 2026 17:01:55 -0400 Subject: [PATCH] comments --- appview/account_status.go | 30 + appview/apns/apns.go | 303 ++++++++++ appview/catalog/catalog.go | 432 ++++++++++++++ appview/catalog/pi_resolver.go | 290 +++++++++ appview/config.go | 18 + .../migrations/0005_phase1_observability.sql | 51 ++ .../0006_phase2_catalog_subjects.sql | 150 +++++ .../migrations/0007_phase3_notifications.sql | 67 +++ .../migrations/0008_phase4_moderation.sql | 37 ++ appview/database/models.go | 358 +++++++++--- appview/firehose.go | 30 +- appview/handlers/admin.go | 115 +++- appview/handlers/bookmark.go | 47 +- appview/handlers/catalog.go | 202 +++++++ appview/handlers/comment.go | 553 +++++++++++++++--- appview/handlers/devices.go | 93 +++ appview/handlers/episode_state.go | 14 +- appview/handlers/episodes.go | 7 +- appview/handlers/handlers.go | 51 +- appview/handlers/inbox.go | 38 +- appview/handlers/labels.go | 463 +++++++++++++++ appview/handlers/notifications.go | 284 +++++++++ appview/handlers/profile.go | 36 +- appview/handlers/recommendation.go | 50 +- appview/handlers/social.go | 79 ++- appview/handlers/subscribe.go | 118 ++++ appview/handlers/subscription.go | 41 +- appview/indexer/backfill.go | 206 +++++++ appview/indexer/bookmark.go | 93 ++- appview/indexer/comment.go | 224 +++++-- appview/indexer/episode_state.go | 38 +- appview/indexer/errors.go | 64 ++ appview/indexer/events.go | 93 +++ appview/indexer/identity.go | 161 +++++ appview/indexer/indexer.go | 132 ++++- appview/indexer/notifications.go | 187 ++++++ appview/indexer/recommendation.go | 37 +- appview/indexer/stats.go | 99 ++-- appview/indexer/strongref.go | 51 ++ appview/indexer/subscription.go | 27 +- appview/indexer/threadgate.go | 109 ++++ appview/indexer/watchdog.go | 156 +++++ appview/server.go | 82 ++- cmd/effem-appview/main.go | 53 ++ go.mod | 2 + go.sum | 13 + lexicons/lexicons.go | 9 + lexicons/xyz/effem/actor/profile.json | 12 + lexicons/xyz/effem/admin/applyLabel.json | 44 ++ lexicons/xyz/effem/admin/importLabels.json | 54 ++ lexicons/xyz/effem/admin/listLabels.json | 58 ++ lexicons/xyz/effem/admin/removeLabel.json | 21 + lexicons/xyz/effem/feed/bookmark.json | 9 +- lexicons/xyz/effem/feed/comment.json | 57 +- lexicons/xyz/effem/feed/commentLike.json | 26 + lexicons/xyz/effem/feed/defs.json | 52 -- lexicons/xyz/effem/feed/episode.json | 32 + lexicons/xyz/effem/feed/episodeState.json | 9 +- lexicons/xyz/effem/feed/list.json | 3 +- lexicons/xyz/effem/feed/podcast.json | 30 + lexicons/xyz/effem/feed/recommendation.json | 6 +- .../xyz/effem/feed/subscribeComments.json | 59 ++ lexicons/xyz/effem/feed/subscription.json | 9 +- lexicons/xyz/effem/feed/threadgate.json | 37 ++ .../effem/notification/getPreferences.json | 31 + .../effem/notification/listNotifications.json | 69 +++ .../effem/notification/registerDevice.json | 22 + .../effem/notification/setPreferences.json | 23 + .../effem/notification/unregisterDevice.json | 20 + .../xyz/effem/notification/updateSeen.json | 20 + 70 files changed, 5921 insertions(+), 575 deletions(-) create mode 100644 appview/account_status.go create mode 100644 appview/apns/apns.go create mode 100644 appview/catalog/catalog.go create mode 100644 appview/catalog/pi_resolver.go create mode 100644 appview/database/migrations/0005_phase1_observability.sql create mode 100644 appview/database/migrations/0006_phase2_catalog_subjects.sql create mode 100644 appview/database/migrations/0007_phase3_notifications.sql create mode 100644 appview/database/migrations/0008_phase4_moderation.sql create mode 100644 appview/handlers/catalog.go create mode 100644 appview/handlers/devices.go create mode 100644 appview/handlers/labels.go create mode 100644 appview/handlers/notifications.go create mode 100644 appview/handlers/subscribe.go create mode 100644 appview/indexer/backfill.go create mode 100644 appview/indexer/errors.go create mode 100644 appview/indexer/events.go create mode 100644 appview/indexer/identity.go create mode 100644 appview/indexer/notifications.go create mode 100644 appview/indexer/strongref.go create mode 100644 appview/indexer/threadgate.go create mode 100644 appview/indexer/watchdog.go create mode 100644 lexicons/lexicons.go create mode 100644 lexicons/xyz/effem/admin/applyLabel.json create mode 100644 lexicons/xyz/effem/admin/importLabels.json create mode 100644 lexicons/xyz/effem/admin/listLabels.json create mode 100644 lexicons/xyz/effem/admin/removeLabel.json create mode 100644 lexicons/xyz/effem/feed/commentLike.json delete mode 100644 lexicons/xyz/effem/feed/defs.json create mode 100644 lexicons/xyz/effem/feed/episode.json create mode 100644 lexicons/xyz/effem/feed/podcast.json create mode 100644 lexicons/xyz/effem/feed/subscribeComments.json create mode 100644 lexicons/xyz/effem/feed/threadgate.json create mode 100644 lexicons/xyz/effem/notification/getPreferences.json create mode 100644 lexicons/xyz/effem/notification/listNotifications.json create mode 100644 lexicons/xyz/effem/notification/registerDevice.json create mode 100644 lexicons/xyz/effem/notification/setPreferences.json create mode 100644 lexicons/xyz/effem/notification/unregisterDevice.json create mode 100644 lexicons/xyz/effem/notification/updateSeen.json diff --git a/appview/account_status.go b/appview/account_status.go new file mode 100644 index 0000000..a514e50 --- /dev/null +++ b/appview/account_status.go @@ -0,0 +1,30 @@ +package appview + +import ( + "context" + "time" + + "gorm.io/gorm/clause" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" +) + +// upsertAccountStatus records the latest activation state for a DID from the +// firehose #account event. Read handlers filter on `active = false` to hide +// content from takendown / deactivated / suspended accounts. +func (srv *Server) upsertAccountStatus(ctx context.Context, did string, active bool, status string) error { + if did == "" { + return nil + } + row := database.AccountStatus{ + DID: did, + Active: active, + Status: status, + UpdatedAt: time.Now(), + } + return srv.db.WithContext(ctx).Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "did"}}, + DoUpdates: clause.AssignmentColumns([]string{ + "active", "status", "updated_at", + }), + }).Create(&row).Error +} diff --git a/appview/apns/apns.go b/appview/apns/apns.go new file mode 100644 index 0000000..e39eaf6 --- /dev/null +++ b/appview/apns/apns.go @@ -0,0 +1,303 @@ +// Package apns hosts the AppView's APNs push dispatcher. +// +// The dispatcher consumes the `push_outbox` table that the indexer populates +// when notifications fire. For each pending row it fans out a push to every +// device the recipient has registered. Failed deliveries either retry (with +// attempt counter + exponential backoff) or quietly soft-delete the device +// token when APNs feedback indicates it's permanently invalid. +package apns + +import ( + "context" + "crypto/ecdsa" + "encoding/json" + "errors" + "fmt" + "log/slog" + "time" + + "github.com/sideshow/apns2" + "github.com/sideshow/apns2/payload" + "github.com/sideshow/apns2/token" + "gorm.io/gorm" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" +) + +// Config captures everything we need to authenticate with APNs. All fields +// are required when push is enabled; if any are missing we disable the +// dispatcher and log the absence — the app still works without push. +type Config struct { + // KeyPath is the on-disk location of the .p8 private key downloaded + // from the Apple Developer portal. + KeyPath string + // KeyID identifies the .p8 key in Apple's directory. + KeyID string + // TeamID is the Apple Developer team that owns the bundle ID. + TeamID string + // BundleID is the iOS app's bundle identifier; sent as the APNs topic. + BundleID string + // Environment is "sandbox" for dev builds and "production" for App Store + // / TestFlight builds. The dispatcher picks the right APNs host + // per-token via device_tokens.environment so a single AppView can drive + // both worlds. +} + +// Dispatcher pulls pending push_outbox rows and ships them to APNs. +type Dispatcher struct { + db *gorm.DB + logger *slog.Logger + cfg Config + keyPath string + authKey *ecdsa.PrivateKey + prodClient *apns2.Client + devClient *apns2.Client + enabled bool +} + +const ( + dispatcherInterval = 5 * time.Second + dispatcherBatchSize = 64 + dispatcherMaxAttempts = 5 +) + +// New returns a Dispatcher. If the config is incomplete (no key path) the +// dispatcher returns with Enabled() == false; pushes are silently dropped +// instead of erroring. This lets dev runs work without a real APNs key. +func New(db *gorm.DB, logger *slog.Logger, cfg Config) (*Dispatcher, error) { + d := &Dispatcher{db: db, logger: logger.With("component", "apns"), cfg: cfg} + if cfg.KeyPath == "" || cfg.KeyID == "" || cfg.TeamID == "" || cfg.BundleID == "" { + d.logger.Info("apns disabled — incomplete config") + return d, nil + } + key, err := token.AuthKeyFromFile(cfg.KeyPath) + if err != nil { + return nil, fmt.Errorf("loading apns key: %w", err) + } + d.authKey = key + d.prodClient = apns2.NewTokenClient(&token.Token{ + AuthKey: key, KeyID: cfg.KeyID, TeamID: cfg.TeamID, + }).Production() + d.devClient = apns2.NewTokenClient(&token.Token{ + AuthKey: key, KeyID: cfg.KeyID, TeamID: cfg.TeamID, + }).Development() + d.enabled = true + d.logger.Info("apns enabled", "key_id", cfg.KeyID, "team_id", cfg.TeamID, "bundle", cfg.BundleID) + return d, nil +} + +// Enabled reports whether the dispatcher will actually send pushes. When +// false, the goroutine still runs but marks rows as sent without contacting +// APNs — useful for dev runs that don't have a key. +func (d *Dispatcher) Enabled() bool { return d.enabled } + +// Run is the goroutine that drains push_outbox. Returns only when ctx is +// cancelled. +func (d *Dispatcher) Run(ctx context.Context) { + if d == nil { + return + } + ticker := time.NewTicker(dispatcherInterval) + defer ticker.Stop() + for { + select { + case <-ctx.Done(): + return + case <-ticker.C: + d.drainOnce(ctx) + } + } +} + +func (d *Dispatcher) drainOnce(ctx context.Context) { + var rows []database.PushOutbox + err := d.db.WithContext(ctx). + Where("sent_at IS NULL AND attempts < ?", dispatcherMaxAttempts). + Order("queued_at ASC"). + Limit(dispatcherBatchSize). + Find(&rows).Error + if err != nil { + d.logger.Warn("push_outbox scan failed", "err", err) + return + } + for _, row := range rows { + d.dispatch(ctx, row) + } +} + +func (d *Dispatcher) dispatch(ctx context.Context, row database.PushOutbox) { + if !d.enabled { + // Mark as sent without doing anything so the row doesn't pile up + // in dev environments without an APNs key. + now := time.Now().UTC() + d.db.WithContext(ctx).Model(&database.PushOutbox{}). + Where("id = ?", row.ID). + Updates(map[string]any{"sent_at": &now, "attempts": row.Attempts + 1}) + return + } + + var tokens []database.DeviceToken + if err := d.db.WithContext(ctx). + Where("did = ? AND invalid_at IS NULL", row.RecipientDID). + Find(&tokens).Error; err != nil { + d.markFailed(ctx, row, fmt.Sprintf("token lookup: %v", err)) + return + } + if len(tokens) == 0 { + // No devices registered; consider the push delivered. A future + // registerDevice call will pick up subsequent notifications. + d.markSent(ctx, row) + return + } + + for _, dev := range tokens { + client := d.clientFor(dev.Environment) + if client == nil { + continue + } + note := d.buildNotification(dev, row) + resp, err := client.PushWithContext(ctx, note) + if err != nil { + d.markFailed(ctx, row, fmt.Sprintf("apns push: %v", err)) + return + } + if !resp.Sent() { + if isFatalAPNsReason(resp.Reason) { + d.invalidateToken(ctx, dev.ID, resp.Reason) + continue + } + d.markFailed(ctx, row, fmt.Sprintf("apns %d %s", resp.StatusCode, resp.Reason)) + return + } + } + d.markSent(ctx, row) +} + +func (d *Dispatcher) clientFor(env string) *apns2.Client { + switch env { + case "production": + return d.prodClient + case "sandbox": + return d.devClient + default: + return nil + } +} + +func (d *Dispatcher) buildNotification(dev database.DeviceToken, row database.PushOutbox) *apns2.Notification { + var payloadJSON map[string]any + _ = json.Unmarshal(row.Payload, &payloadJSON) + + reason, _ := payloadJSON["reason"].(string) + subjectURI, _ := payloadJSON["subject_uri"].(string) + sourceURI, _ := payloadJSON["source_uri"].(string) + actorDID, _ := payloadJSON["actor_did"].(string) + + body := payload.NewPayload(). + AlertTitle(titleForReason(reason)). + AlertBody(bodyForReason(reason)). + Category(categoryForReason(reason)). + ThreadID(subjectURI). + Custom("reason", reason). + Custom("subject_uri", subjectURI). + Custom("source_uri", sourceURI). + Custom("actor_did", actorDID). + Sound("default"). + MutableContent() + + topic := dev.BundleID + if topic == "" { + topic = d.cfg.BundleID + } + + return &apns2.Notification{ + DeviceToken: dev.Token, + Topic: topic, + Payload: body, + Expiration: time.Now().Add(24 * time.Hour), + } +} + +func titleForReason(reason string) string { + switch reason { + case "reply": + return "New reply" + case "mention": + return "You were mentioned" + case "like": + return "Someone liked your comment" + default: + return "New activity" + } +} + +func bodyForReason(reason string) string { + switch reason { + case "reply": + return "Tap to view the reply." + case "mention": + return "Tap to view the comment." + case "like": + return "Tap to view the comment." + default: + return "" + } +} + +func categoryForReason(reason string) string { + switch reason { + case "reply": + return "COMMENT_REPLY" + case "mention": + return "MENTION" + case "like": + return "COMMENT_LIKE" + default: + return "DEFAULT" + } +} + +// isFatalAPNsReason returns true for response reasons that mean the device +// token is permanently bad. The dispatcher soft-deletes those tokens so we +// stop sending to them. +func isFatalAPNsReason(reason string) bool { + switch reason { + case "BadDeviceToken", "Unregistered", "DeviceTokenNotForTopic": + return true + } + return false +} + +func (d *Dispatcher) markSent(ctx context.Context, row database.PushOutbox) { + now := time.Now().UTC() + if err := d.db.WithContext(ctx).Model(&database.PushOutbox{}). + Where("id = ?", row.ID). + Updates(map[string]any{"sent_at": &now, "attempts": row.Attempts + 1}).Error; err != nil { + d.logger.Warn("push_outbox mark sent failed", "err", err, "id", row.ID) + } +} + +func (d *Dispatcher) markFailed(ctx context.Context, row database.PushOutbox, reason string) { + if err := d.db.WithContext(ctx).Model(&database.PushOutbox{}). + Where("id = ?", row.ID). + Updates(map[string]any{ + "attempts": row.Attempts + 1, + "last_error": reason, + }).Error; err != nil { + d.logger.Warn("push_outbox mark failed", "err", err, "id", row.ID) + } +} + +func (d *Dispatcher) invalidateToken(ctx context.Context, id uint, reason string) { + now := time.Now().UTC() + if err := d.db.WithContext(ctx).Model(&database.DeviceToken{}). + Where("id = ?", id). + Updates(map[string]any{"invalid_at": &now}).Error; err != nil { + d.logger.Warn("device token invalidate failed", "err", err, "id", id, "reason", reason) + return + } + d.logger.Info("device token invalidated by apns", "id", id, "reason", reason) +} + +// ErrPushDisabled is returned by helpers that need to noop when no APNs +// config is present. Kept here so callers don't have to type-assert. +var ErrPushDisabled = errors.New("apns push disabled") diff --git a/appview/catalog/catalog.go b/appview/catalog/catalog.go new file mode 100644 index 0000000..ec4f9a7 --- /dev/null +++ b/appview/catalog/catalog.go @@ -0,0 +1,432 @@ +// Package catalog manages the canonical episode and podcast records that +// every subject-bearing collection references via com.atproto.repo.strongRef. +// +// Records are deterministic: the same (podcastGuid, episodeGuid) always +// produces the same rkey and the same CID, so two clients independently +// resolving the same episode arrive at the same strongRef without +// coordination. The AppView is the canonical publisher today; in the +// future other implementations can either trust our catalog or run their +// own and reconcile. +package catalog + +import ( + "context" + "crypto/sha256" + "encoding/base32" + "encoding/json" + "errors" + "fmt" + "strings" + "time" + + "github.com/bluesky-social/indigo/atproto/atdata" + "github.com/ipfs/go-cid" + "github.com/multiformats/go-multihash" + "gorm.io/gorm" + "gorm.io/gorm/clause" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" +) + +const ( + // dagCBORCodec is the multicodec identifier for dag-cbor, the canonical + // CID codec used by AT Protocol records. + dagCBORCodec = 0x71 + + // rkeyEncoding is RFC 4648 base32 lowercase without padding. AT Proto + // rkeys allow alphanumerics, dot, dash, underscore, tilde, and colon — + // base32 alphanumerics fit cleanly inside that without escaping. + rkeyHashBytes = 16 // 16 bytes → 26-char rkey, plenty unique per scheme. +) + +// Identity is the natural key the catalog dedupes on. +type EpisodeIdentity struct { + PodcastGuid string + EpisodeGuid string +} + +type PodcastIdentity struct { + PodcastGuid string +} + +// StrongRef is the value clients embed in records. +type StrongRef struct { + URI string `json:"uri"` + CID string `json:"cid"` +} + +// EpisodeFields is everything an upstream resolver needs to publish a new +// catalog row. PodcastGuid + EpisodeGuid are required because they are the +// natural key; the others are best-effort metadata used by consumers. +type EpisodeFields struct { + PodcastGuid string + EpisodeGuid string + Title string + PublishedAt string + FeedURL string + EnclosureURL string + DurationS *int + PodcastIndexFeedID *int64 + PodcastIndexEpisodeID *int64 +} + +// PodcastFields mirrors EpisodeFields at the show level. +type PodcastFields struct { + PodcastGuid string + Title string + FeedURL string + Author string + ArtworkURL string + Language string + PodcastIndexFeedID *int64 +} + +// Resolver fetches episode / podcast metadata when the catalog doesn't have +// it yet. The AppView wires this to a Podcast Index lookup; tests use a +// stub that returns canned data. +type Resolver interface { + ResolveEpisode(ctx context.Context, id EpisodeIdentity) (*EpisodeFields, error) + ResolvePodcast(ctx context.Context, id PodcastIdentity) (*PodcastFields, error) +} + +// Service is the public surface. Concurrent calls for the same key are +// safe — the unique index on (podcast_guid, episode_guid) (or podcast_guid +// alone for podcasts) lets the database arbitrate. +type Service struct { + db *gorm.DB + resolver Resolver + catalogDID string +} + +// Config controls the AT-URI namespace. +type Config struct { + // CatalogDID is the DID that owns the catalog repo. Used to build + // AT-URIs of the form at:////. did:web rooted at + // the AppView's catalog domain is a natural choice. + CatalogDID string +} + +func New(db *gorm.DB, resolver Resolver, cfg Config) *Service { + if cfg.CatalogDID == "" { + cfg.CatalogDID = "did:web:catalog.effem.app" + } + return &Service{db: db, resolver: resolver, catalogDID: cfg.CatalogDID} +} + +// ErrUnknownEpisode is returned when a caller asks for a strongRef but no +// row exists and the resolver couldn't fill it in. Callers should treat +// this as a transient missing-record signal and retry once they have more +// information (e.g. the source repo's actual record). +var ErrUnknownEpisode = errors.New("episode not in catalog and resolver cannot find it") +var ErrUnknownPodcast = errors.New("podcast not in catalog and resolver cannot find it") + +// ResolveEpisode returns the strongRef + cached row for (podcastGuid, episodeGuid), +// creating the catalog entry on first request. The episode_catalog row +// becomes the single source of truth for "what is this episode?" — every +// subject-bearing record references its AT-URI. +func (s *Service) ResolveEpisode(ctx context.Context, id EpisodeIdentity) (*database.EpisodeCatalog, error) { + if id.PodcastGuid == "" || id.EpisodeGuid == "" { + return nil, fmt.Errorf("episode identity requires non-empty podcastGuid and episodeGuid") + } + + var row database.EpisodeCatalog + err := s.db.WithContext(ctx). + Where("podcast_guid = ? AND episode_guid = ?", id.PodcastGuid, id.EpisodeGuid). + First(&row).Error + if err == nil { + return &row, nil + } + if !errors.Is(err, gorm.ErrRecordNotFound) { + return nil, fmt.Errorf("episode_catalog lookup: %w", err) + } + + fields, err := s.resolver.ResolveEpisode(ctx, id) + if err != nil { + return nil, fmt.Errorf("resolver: %w", err) + } + if fields == nil { + return nil, ErrUnknownEpisode + } + + return s.upsertEpisode(ctx, *fields) +} + +// EnsureEpisodeFromRecord registers a catalog row sourced from a strongRef +// that already exists in the wild (someone commented on an episode before +// we knew about it). Callers pass in the exact record fields they received, +// so the resulting row has the same content as the upstream publisher +// expected — and therefore the same CID. +func (s *Service) EnsureEpisodeFromRecord(ctx context.Context, fields EpisodeFields) (*database.EpisodeCatalog, error) { + if fields.PodcastGuid == "" || fields.EpisodeGuid == "" { + return nil, fmt.Errorf("episode fields require non-empty podcastGuid and episodeGuid") + } + return s.upsertEpisode(ctx, fields) +} + +// LookupEpisodeByURI returns the catalog row for a previously-published +// strongRef. Used by the indexer to denormalise subjects on read. +func (s *Service) LookupEpisodeByURI(ctx context.Context, uri string) (*database.EpisodeCatalog, error) { + var row database.EpisodeCatalog + if err := s.db.WithContext(ctx).Where("at_uri = ?", uri).First(&row).Error; err != nil { + return nil, err + } + return &row, nil +} + +// LookupEpisodeByPI returns the catalog row for the given Podcast Index IDs, +// if any. Convenience for the social-overlay path that reads PI IDs from +// the PI proxy responses. +func (s *Service) LookupEpisodeByPI(ctx context.Context, feedID, episodeID int64) (*database.EpisodeCatalog, error) { + var row database.EpisodeCatalog + if err := s.db.WithContext(ctx). + Where("podcast_index_feed_id = ? AND podcast_index_episode_id = ?", feedID, episodeID). + First(&row).Error; err != nil { + return nil, err + } + return &row, nil +} + +func (s *Service) upsertEpisode(ctx context.Context, f EpisodeFields) (*database.EpisodeCatalog, error) { + rkey := deriveRkey("episode", f.PodcastGuid, f.EpisodeGuid) + atURI := fmt.Sprintf("at://%s/xyz.effem.feed.episode/%s", s.catalogDID, rkey) + + record := episodeRecord(f) + c, err := canonicalCID(record) + if err != nil { + return nil, fmt.Errorf("compute episode cid: %w", err) + } + + row := database.EpisodeCatalog{ + PodcastGuid: f.PodcastGuid, + EpisodeGuid: f.EpisodeGuid, + ATURI: atURI, + CID: c, + Rkey: rkey, + Title: f.Title, + PublishedAt: f.PublishedAt, + FeedURL: f.FeedURL, + EnclosureURL: f.EnclosureURL, + DurationS: f.DurationS, + PodcastIndexFeedID: f.PodcastIndexFeedID, + PodcastIndexEpisodeID: f.PodcastIndexEpisodeID, + } + if err := s.db.WithContext(ctx).Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "podcast_guid"}, {Name: "episode_guid"}}, + // at_uri, cid, rkey are derived from the natural key + record + // content; they should never change for the same identity but it's + // cheap to refresh them on conflict in case our hashing changes. + DoUpdates: clause.AssignmentColumns([]string{ + "at_uri", "cid", "rkey", "title", "published_at", "feed_url", + "enclosure_url", "duration_s", "podcast_index_feed_id", + "podcast_index_episode_id", + }), + }).Create(&row).Error; err != nil { + return nil, err + } + return &row, nil +} + +// ResolvePodcast mirrors ResolveEpisode for podcasts. The natural key is +// podcastGuid alone. +func (s *Service) ResolvePodcast(ctx context.Context, id PodcastIdentity) (*database.PodcastCatalog, error) { + if id.PodcastGuid == "" { + return nil, fmt.Errorf("podcast identity requires non-empty podcastGuid") + } + + var row database.PodcastCatalog + err := s.db.WithContext(ctx).Where("podcast_guid = ?", id.PodcastGuid).First(&row).Error + if err == nil { + return &row, nil + } + if !errors.Is(err, gorm.ErrRecordNotFound) { + return nil, fmt.Errorf("podcast_catalog lookup: %w", err) + } + + fields, err := s.resolver.ResolvePodcast(ctx, id) + if err != nil { + return nil, fmt.Errorf("resolver: %w", err) + } + if fields == nil { + return nil, ErrUnknownPodcast + } + + return s.upsertPodcast(ctx, *fields) +} + +func (s *Service) EnsurePodcastFromRecord(ctx context.Context, fields PodcastFields) (*database.PodcastCatalog, error) { + if fields.PodcastGuid == "" { + return nil, fmt.Errorf("podcast fields require non-empty podcastGuid") + } + return s.upsertPodcast(ctx, fields) +} + +func (s *Service) LookupPodcastByURI(ctx context.Context, uri string) (*database.PodcastCatalog, error) { + var row database.PodcastCatalog + if err := s.db.WithContext(ctx).Where("at_uri = ?", uri).First(&row).Error; err != nil { + return nil, err + } + return &row, nil +} + +func (s *Service) LookupPodcastByPI(ctx context.Context, feedID int64) (*database.PodcastCatalog, error) { + var row database.PodcastCatalog + if err := s.db.WithContext(ctx).Where("podcast_index_feed_id = ?", feedID).First(&row).Error; err != nil { + return nil, err + } + return &row, nil +} + +func (s *Service) upsertPodcast(ctx context.Context, f PodcastFields) (*database.PodcastCatalog, error) { + rkey := deriveRkey("podcast", f.PodcastGuid, "") + atURI := fmt.Sprintf("at://%s/xyz.effem.feed.podcast/%s", s.catalogDID, rkey) + + record := podcastRecord(f) + c, err := canonicalCID(record) + if err != nil { + return nil, fmt.Errorf("compute podcast cid: %w", err) + } + + row := database.PodcastCatalog{ + PodcastGuid: f.PodcastGuid, + ATURI: atURI, + CID: c, + Rkey: rkey, + Title: f.Title, + FeedURL: f.FeedURL, + Author: f.Author, + ArtworkURL: f.ArtworkURL, + Language: f.Language, + PodcastIndexFeedID: f.PodcastIndexFeedID, + } + if err := s.db.WithContext(ctx).Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "podcast_guid"}}, + DoUpdates: clause.AssignmentColumns([]string{ + "at_uri", "cid", "rkey", "title", "feed_url", "author", + "artwork_url", "language", "podcast_index_feed_id", + }), + }).Create(&row).Error; err != nil { + return nil, err + } + return &row, nil +} + +// Record serialises a catalog row into the AT Proto record shape clients +// will see when they dereference the AT-URI. Same shape that goes through +// canonicalCID to produce the cached CID, so callers can verify. +func (s *Service) EpisodeRecord(row *database.EpisodeCatalog) map[string]any { + return episodeRecord(EpisodeFields{ + PodcastGuid: row.PodcastGuid, + EpisodeGuid: row.EpisodeGuid, + Title: row.Title, + PublishedAt: row.PublishedAt, + FeedURL: row.FeedURL, + EnclosureURL: row.EnclosureURL, + DurationS: row.DurationS, + PodcastIndexFeedID: row.PodcastIndexFeedID, + PodcastIndexEpisodeID: row.PodcastIndexEpisodeID, + }) +} + +func (s *Service) PodcastRecord(row *database.PodcastCatalog) map[string]any { + return podcastRecord(PodcastFields{ + PodcastGuid: row.PodcastGuid, + Title: row.Title, + FeedURL: row.FeedURL, + Author: row.Author, + ArtworkURL: row.ArtworkURL, + Language: row.Language, + PodcastIndexFeedID: row.PodcastIndexFeedID, + }) +} + +// EpisodeStrongRef pulls the canonical strongRef out of a row. +func EpisodeStrongRef(row *database.EpisodeCatalog) StrongRef { + return StrongRef{URI: row.ATURI, CID: row.CID} +} + +func PodcastStrongRef(row *database.PodcastCatalog) StrongRef { + return StrongRef{URI: row.ATURI, CID: row.CID} +} + +// --- internal: deterministic rkey + CID -------------------------------- + +func deriveRkey(scope, k1, k2 string) string { + h := sha256.Sum256([]byte(scope + "\x00" + k1 + "\x00" + k2)) + return strings.ToLower(base32.StdEncoding.WithPadding(base32.NoPadding).EncodeToString(h[:rkeyHashBytes])) +} + +func canonicalCID(record map[string]any) (string, error) { + encoded, err := atdata.MarshalCBOR(record) + if err != nil { + return "", err + } + mh, err := multihash.Sum(encoded, multihash.SHA2_256, -1) + if err != nil { + return "", err + } + return cid.NewCidV1(dagCBORCodec, mh).String(), nil +} + +func episodeRecord(f EpisodeFields) map[string]any { + rec := map[string]any{ + "$type": "xyz.effem.feed.episode", + "podcastGuid": f.PodcastGuid, + "episodeGuid": f.EpisodeGuid, + "title": f.Title, + "publishedAt": f.PublishedAt, + } + addOptional(rec, "feedUrl", f.FeedURL) + addOptional(rec, "enclosureUrl", f.EnclosureURL) + if f.DurationS != nil { + rec["durationS"] = int64(*f.DurationS) + } + if f.PodcastIndexFeedID != nil || f.PodcastIndexEpisodeID != nil { + pi := map[string]any{} + if f.PodcastIndexFeedID != nil { + pi["feedId"] = *f.PodcastIndexFeedID + } + if f.PodcastIndexEpisodeID != nil { + pi["episodeId"] = *f.PodcastIndexEpisodeID + } + rec["podcastIndex"] = pi + } + return rec +} + +func podcastRecord(f PodcastFields) map[string]any { + rec := map[string]any{ + "$type": "xyz.effem.feed.podcast", + "podcastGuid": f.PodcastGuid, + "title": f.Title, + } + addOptional(rec, "feedUrl", f.FeedURL) + addOptional(rec, "author", f.Author) + addOptional(rec, "artworkUrl", f.ArtworkURL) + addOptional(rec, "language", f.Language) + if f.PodcastIndexFeedID != nil { + rec["podcastIndex"] = map[string]any{"feedId": *f.PodcastIndexFeedID} + } + return rec +} + +func addOptional(m map[string]any, key, value string) { + if value != "" { + m[key] = value + } +} + +// timeToATString is a convenience for resolvers that have a time.Time and +// need to emit an AT Proto datetime string. Kept here so multiple callers +// don't drift on formatting. +func TimeToATString(t time.Time) string { + if t.IsZero() { + return "" + } + return t.UTC().Format("2006-01-02T15:04:05.000Z") +} + +// AsJSON returns the canonical record encoded as JSON. Convenient for +// XRPC endpoints that need to serve the catalog record alongside the +// strongRef. +func (s *Service) AsJSON(record map[string]any) (json.RawMessage, error) { + return json.Marshal(record) +} diff --git a/appview/catalog/pi_resolver.go b/appview/catalog/pi_resolver.go new file mode 100644 index 0000000..60b70df --- /dev/null +++ b/appview/catalog/pi_resolver.go @@ -0,0 +1,290 @@ +package catalog + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "strconv" + "strings" + "time" + + "tangled.org/sparrowtek.com/effem-AppView/appview/podcastindex" +) + +// PIResolver bridges the catalog to the Podcast Index API. It accepts +// identities keyed on (podcastGuid, episodeGuid) and uses PI's by-id and +// by-guid endpoints to fetch the metadata that goes into a catalog row. +// +// Episodes: PI's /episodes/byguid takes a podcast feed id plus an episode +// guid. We accept either the PI episode id (fastest) via PodcastIndexEpisodeID +// passed through the identity hints, or we walk podcastGuid → feed → episode +// for clients that only know guids. +type PIResolver struct { + pi *podcastindex.CachedClient +} + +func NewPIResolver(pi *podcastindex.CachedClient) *PIResolver { + return &PIResolver{pi: pi} +} + +// PIHints carries optional Podcast Index IDs alongside the natural key. +// Callers that have them (e.g. the iOS app reading from PodcastIndexKit) +// pass them through so the resolver can take a fast path. +// +// Stored on a `pi_hints` ctx value so PIResolver picks them up without +// muddying the Identity struct that gets persisted to the catalog row. +type PIHints struct { + PodcastIndexFeedID *int64 + PodcastIndexEpisodeID *int64 +} + +type piHintsKey struct{} + +// WithPIHints attaches feed/episode ID hints to ctx. The resolver consults +// them when it needs to walk PI for a catalog miss. +func WithPIHints(ctx context.Context, hints PIHints) context.Context { + return context.WithValue(ctx, piHintsKey{}, hints) +} + +func piHintsFromContext(ctx context.Context) PIHints { + h, _ := ctx.Value(piHintsKey{}).(PIHints) + return h +} + +func (r *PIResolver) ResolveEpisode(ctx context.Context, id EpisodeIdentity) (*EpisodeFields, error) { + hints := piHintsFromContext(ctx) + + var raw json.RawMessage + var err error + + if hints.PodcastIndexEpisodeID != nil && *hints.PodcastIndexEpisodeID > 0 { + raw, err = r.pi.GetEpisodeByID(*hints.PodcastIndexEpisodeID) + if err != nil { + return nil, fmt.Errorf("pi GetEpisodeByID: %w", err) + } + } else { + // Without PI IDs we can't reach PI's by-guid path safely (it requires + // the feed id we don't have). Tell the caller; once Phase 3+ wires + // guid-only lookups via a different source, this returns properly. + return nil, fmt.Errorf("pi resolver needs at least PodcastIndexEpisodeID hint to find %q/%q", + id.PodcastGuid, id.EpisodeGuid) + } + + episode, ok := extractEpisode(raw) + if !ok { + return nil, errors.New("pi response did not contain an episode object") + } + + fields, err := piEpisodeFields(episode) + if err != nil { + return nil, err + } + // Confirm guids match — otherwise we'd be writing a row keyed on guids + // that don't actually identify this episode. + if fields.PodcastGuid != id.PodcastGuid || fields.EpisodeGuid != id.EpisodeGuid { + return nil, fmt.Errorf("pi episode guids %q/%q do not match requested %q/%q", + fields.PodcastGuid, fields.EpisodeGuid, id.PodcastGuid, id.EpisodeGuid) + } + return fields, nil +} + +func (r *PIResolver) ResolvePodcast(ctx context.Context, id PodcastIdentity) (*PodcastFields, error) { + hints := piHintsFromContext(ctx) + if hints.PodcastIndexFeedID == nil || *hints.PodcastIndexFeedID <= 0 { + return nil, fmt.Errorf("pi resolver needs PodcastIndexFeedID hint to find podcast %q", id.PodcastGuid) + } + raw, err := r.pi.GetPodcastByFeedID(*hints.PodcastIndexFeedID) + if err != nil { + return nil, fmt.Errorf("pi GetPodcastByFeedID: %w", err) + } + feed, ok := extractFeed(raw) + if !ok { + return nil, errors.New("pi response did not contain a feed object") + } + fields, err := piPodcastFields(feed) + if err != nil { + return nil, err + } + if fields.PodcastGuid != id.PodcastGuid { + return nil, fmt.Errorf("pi feed guid %q does not match requested %q", + fields.PodcastGuid, id.PodcastGuid) + } + return fields, nil +} + +// ResolveEpisodeFromPIID is a convenience used by the +// xyz.effem.feed.resolveEpisode XRPC, where the caller has PI IDs and wants +// the strongRef plus the catalog row without separately resolving guids. +// Returns the EpisodeFields directly so callers can hand them to +// catalog.EnsureEpisodeFromRecord. +func (r *PIResolver) ResolveEpisodeFromPIID(ctx context.Context, episodeID int64) (*EpisodeFields, error) { + raw, err := r.pi.GetEpisodeByID(episodeID) + if err != nil { + return nil, fmt.Errorf("pi GetEpisodeByID: %w", err) + } + episode, ok := extractEpisode(raw) + if !ok { + return nil, errors.New("pi response did not contain an episode object") + } + return piEpisodeFields(episode) +} + +func (r *PIResolver) ResolvePodcastFromPIID(ctx context.Context, feedID int64) (*PodcastFields, error) { + raw, err := r.pi.GetPodcastByFeedID(feedID) + if err != nil { + return nil, fmt.Errorf("pi GetPodcastByFeedID: %w", err) + } + feed, ok := extractFeed(raw) + if !ok { + return nil, errors.New("pi response did not contain a feed object") + } + return piPodcastFields(feed) +} + +// ---- PI payload extraction --------------------------------------------- + +func extractEpisode(raw json.RawMessage) (map[string]any, bool) { + var envelope map[string]any + if err := json.Unmarshal(raw, &envelope); err != nil { + return nil, false + } + if ep, ok := envelope["episode"].(map[string]any); ok { + return ep, true + } + // PI sometimes returns episodes as a single-item array under "items". + if items, ok := envelope["items"].([]any); ok && len(items) > 0 { + if ep, ok := items[0].(map[string]any); ok { + return ep, true + } + } + return nil, false +} + +func extractFeed(raw json.RawMessage) (map[string]any, bool) { + var envelope map[string]any + if err := json.Unmarshal(raw, &envelope); err != nil { + return nil, false + } + if feed, ok := envelope["feed"].(map[string]any); ok { + return feed, true + } + return nil, false +} + +func piEpisodeFields(ep map[string]any) (*EpisodeFields, error) { + podcastGuid := asString(ep["podcastGuid"]) + episodeGuid := asString(ep["guid"]) + if podcastGuid == "" || episodeGuid == "" { + return nil, fmt.Errorf("pi episode missing required guids: podcastGuid=%q episodeGuid=%q", + podcastGuid, episodeGuid) + } + title := asString(ep["title"]) + if title == "" { + title = "Untitled Episode" + } + + publishedAt := piTimeToATString(ep["datePublished"]) + if publishedAt == "" { + publishedAt = TimeToATString(time.Now()) + } + + fields := &EpisodeFields{ + PodcastGuid: podcastGuid, + EpisodeGuid: episodeGuid, + Title: title, + PublishedAt: publishedAt, + FeedURL: asString(ep["feedUrl"]), + EnclosureURL: asString(ep["enclosureUrl"]), + } + if d := asInt64(ep["duration"]); d != nil { + dur := int(*d) + fields.DurationS = &dur + } + if id := asInt64(ep["feedId"]); id != nil { + fields.PodcastIndexFeedID = id + } + if id := asInt64(ep["id"]); id != nil { + fields.PodcastIndexEpisodeID = id + } + return fields, nil +} + +func piPodcastFields(feed map[string]any) (*PodcastFields, error) { + podcastGuid := asString(feed["podcastGuid"]) + if podcastGuid == "" { + return nil, fmt.Errorf("pi feed missing podcastGuid") + } + title := asString(feed["title"]) + if title == "" { + title = "Untitled Podcast" + } + fields := &PodcastFields{ + PodcastGuid: podcastGuid, + Title: title, + FeedURL: asString(feed["url"]), + Author: asString(feed["author"]), + ArtworkURL: firstNonEmpty(asString(feed["artwork"]), asString(feed["image"])), + Language: asString(feed["language"]), + } + if id := asInt64(feed["id"]); id != nil { + fields.PodcastIndexFeedID = id + } + return fields, nil +} + +// piTimeToATString accepts PI's "datePublished" which is a Unix epoch in +// seconds (sometimes as a JSON number, sometimes as a string) and renders +// it as an AT Proto datetime string. Returns empty for missing/invalid. +func piTimeToATString(v any) string { + if v == nil { + return "" + } + if i := asInt64(v); i != nil && *i > 0 { + return TimeToATString(time.Unix(*i, 0).UTC()) + } + return "" +} + +func asString(v any) string { + s, _ := v.(string) + return strings.TrimSpace(s) +} + +func asInt64(v any) *int64 { + switch x := v.(type) { + case int: + i := int64(x) + return &i + case int32: + i := int64(x) + return &i + case int64: + return &x + case float64: + i := int64(x) + return &i + case string: + if s := strings.TrimSpace(x); s != "" { + i, err := strconv.ParseInt(s, 10, 64) + if err == nil { + return &i + } + } + case json.Number: + i, err := x.Int64() + if err == nil { + return &i + } + } + return nil +} + +func firstNonEmpty(values ...string) string { + for _, v := range values { + if v != "" { + return v + } + } + return "" +} diff --git a/appview/config.go b/appview/config.go index d4cd902..a1595b2 100644 --- a/appview/config.go +++ b/appview/config.go @@ -31,8 +31,26 @@ type Config struct { RelayHost string PLCHost string + // CatalogDID owns the episode/podcast catalog repo. Defaults to + // "did:web:catalog.effem.app" when empty. The catalog is hosted by the + // AppView itself; the DID just gives strongRefs a stable AT-URI prefix. + CatalogDID string + // LabelerDID is the source DID stamped on every moderation label + // emitted by xyz.effem.admin.applyLabel. Defaults to + // "did:web:labeler.effem.app" when empty. Read paths always include + // labels with this DID; labels from any other src_did only appear + // when the viewer's labelerSubscriptions list contains that DID. + LabelerDID string PIKey string PISecret string + + // APNs (push notifications). All four fields are required to enable + // the dispatcher; any missing field leaves push disabled so dev runs + // without a key still work end-to-end. + APNsKeyPath string + APNsKeyID string + APNsTeamID string + APNsBundleID string FirehoseParallel int FirehoseQueueLimit int AuthRequired bool diff --git a/appview/database/migrations/0005_phase1_observability.sql b/appview/database/migrations/0005_phase1_observability.sql new file mode 100644 index 0000000..103d4ba --- /dev/null +++ b/appview/database/migrations/0005_phase1_observability.sql @@ -0,0 +1,51 @@ +-- Phase 1: observability + recovery + identity hydration. +-- +-- indexer_errors — every record the indexer rejected, retryable +-- account_status — PDS/Relay account activation state (takedown enforcement) +-- did_pds — cached PDS endpoint per DID (for blob URLs + listRecords) +-- profiles additions — handle + avatar_cid + handle_resolved_at + +CREATE TABLE "indexer_errors" ( + "id" BIGSERIAL PRIMARY KEY, + "did" VARCHAR(255) NOT NULL, + "collection" VARCHAR(255) NOT NULL, + "rkey" VARCHAR(512) NOT NULL, + "cid" VARCHAR(256), + "error_message" TEXT NOT NULL, + "attempts" INTEGER NOT NULL DEFAULT 1, + "last_attempt_at" TIMESTAMPTZ NOT NULL DEFAULT NOW(), + "resolved_at" TIMESTAMPTZ, + "created_at" TIMESTAMPTZ NOT NULL DEFAULT NOW() +); +-- One row per (did, collection, rkey). Re-encountering the same record on +-- retry should update attempts in place, not create a duplicate. +CREATE UNIQUE INDEX "idx_indexer_errors_record" + ON "indexer_errors" ("did", "collection", "rkey"); +-- Watchdog scans for unresolved rows ordered by stalest first. +CREATE INDEX "idx_indexer_errors_retry" + ON "indexer_errors" ("last_attempt_at") + WHERE "resolved_at" IS NULL; + +CREATE TABLE "account_status" ( + "did" VARCHAR(255) PRIMARY KEY, + "active" BOOLEAN NOT NULL DEFAULT TRUE, + "status" VARCHAR(64), + "updated_at" TIMESTAMPTZ NOT NULL DEFAULT NOW() +); +-- Read-side filter scans active = FALSE; small set, but cheap to index. +CREATE INDEX "idx_account_status_active" + ON "account_status" ("active"); + +CREATE TABLE "did_pds" ( + "did" VARCHAR(255) PRIMARY KEY, + "pds_endpoint" VARCHAR(2048) NOT NULL, + "resolved_at" TIMESTAMPTZ NOT NULL DEFAULT NOW() +); + +-- Profile enrichment: handle resolved out-of-band via PLC; avatar CID +-- captured from the indexer when an xyz.effem.actor.profile record carries +-- an avatar blob. +ALTER TABLE "profiles" + ADD COLUMN "handle" VARCHAR(255), + ADD COLUMN "handle_resolved_at" TIMESTAMPTZ, + ADD COLUMN "avatar_cid" VARCHAR(256); diff --git a/appview/database/migrations/0006_phase2_catalog_subjects.sql b/appview/database/migrations/0006_phase2_catalog_subjects.sql new file mode 100644 index 0000000..2424635 --- /dev/null +++ b/appview/database/migrations/0006_phase2_catalog_subjects.sql @@ -0,0 +1,150 @@ +-- Phase 2: replace Podcast-Index-keyed records with strongRef subjects and +-- introduce the episode/podcast catalogs. Every subject-bearing collection +-- (comment, recommendation, bookmark, episode_state, subscription) now +-- references a catalog record via {subject_uri, subject_cid}. +-- +-- Nothing is in production, so we drop legacy columns without a backfill. + +-- ----- Catalog tables ---------------------------------------------------- + +CREATE TABLE "episode_catalog" ( + "id" BIGSERIAL PRIMARY KEY, + "podcast_guid" VARCHAR(512) NOT NULL, + "episode_guid" VARCHAR(512) NOT NULL, + "at_uri" VARCHAR(1024) NOT NULL, + "cid" VARCHAR(256) NOT NULL, + "rkey" VARCHAR(512) NOT NULL, + "title" VARCHAR(1024) NOT NULL, + "published_at" VARCHAR(64) NOT NULL, + "feed_url" VARCHAR(2048), + "enclosure_url" VARCHAR(2048), + "duration_s" INTEGER, + "podcast_index_feed_id" BIGINT, + "podcast_index_episode_id" BIGINT, + "indexed_at" TIMESTAMPTZ NOT NULL DEFAULT NOW() +); +CREATE UNIQUE INDEX "idx_episode_catalog_guids" ON "episode_catalog" ("podcast_guid", "episode_guid"); +CREATE UNIQUE INDEX "idx_episode_catalog_at_uri" ON "episode_catalog" ("at_uri"); +CREATE INDEX "idx_episode_catalog_pi" ON "episode_catalog" ("podcast_index_feed_id", "podcast_index_episode_id"); + +CREATE TABLE "podcast_catalog" ( + "id" BIGSERIAL PRIMARY KEY, + "podcast_guid" VARCHAR(512) NOT NULL, + "at_uri" VARCHAR(1024) NOT NULL, + "cid" VARCHAR(256) NOT NULL, + "rkey" VARCHAR(512) NOT NULL, + "title" VARCHAR(1024) NOT NULL, + "feed_url" VARCHAR(2048), + "author" VARCHAR(1024), + "artwork_url" VARCHAR(2048), + "language" VARCHAR(32), + "podcast_index_feed_id" BIGINT, + "indexed_at" TIMESTAMPTZ NOT NULL DEFAULT NOW() +); +CREATE UNIQUE INDEX "idx_podcast_catalog_guid" ON "podcast_catalog" ("podcast_guid"); +CREATE UNIQUE INDEX "idx_podcast_catalog_at_uri" ON "podcast_catalog" ("at_uri"); +CREATE INDEX "idx_podcast_catalog_pi" ON "podcast_catalog" ("podcast_index_feed_id"); + +-- ----- Comments ---------------------------------------------------------- +-- Drop everything that used to live on the row but is now resolved through +-- the catalog. Keep the legacy reply_root/reply_parent because they're +-- strongRef components into other comments, not into episodes. + +DROP INDEX IF EXISTS "idx_comments_episode"; +ALTER TABLE "comments" + DROP COLUMN "feed_id", + DROP COLUMN "episode_id", + DROP COLUMN "episode_guid", + DROP COLUMN "podcast_guid", + ADD COLUMN "subject_uri" VARCHAR(1024) NOT NULL, + ADD COLUMN "subject_cid" VARCHAR(256) NOT NULL; +CREATE INDEX "idx_comments_subject" ON "comments" ("subject_uri"); + +-- ----- Recommendations --------------------------------------------------- + +DROP INDEX IF EXISTS "idx_recommendations_episode"; +ALTER TABLE "recommendations" + DROP COLUMN "feed_id", + DROP COLUMN "episode_id", + DROP COLUMN "episode_guid", + DROP COLUMN "podcast_guid", + ADD COLUMN "subject_uri" VARCHAR(1024) NOT NULL, + ADD COLUMN "subject_cid" VARCHAR(256) NOT NULL; +CREATE INDEX "idx_recommendations_subject" ON "recommendations" ("subject_uri"); + +-- ----- Bookmarks --------------------------------------------------------- + +DROP INDEX IF EXISTS "idx_bookmarks_episode"; +ALTER TABLE "bookmarks" + DROP COLUMN "feed_id", + DROP COLUMN "episode_id", + DROP COLUMN "episode_guid", + DROP COLUMN "podcast_guid", + ADD COLUMN "subject_uri" VARCHAR(1024) NOT NULL, + ADD COLUMN "subject_cid" VARCHAR(256) NOT NULL; +CREATE INDEX "idx_bookmarks_subject" ON "bookmarks" ("subject_uri"); + +-- ----- Episode states ---------------------------------------------------- + +DROP INDEX IF EXISTS "idx_episode_states_episode"; +ALTER TABLE "episode_states" + DROP COLUMN "feed_id", + DROP COLUMN "episode_id", + DROP COLUMN "episode_guid", + DROP COLUMN "podcast_guid", + ADD COLUMN "subject_uri" VARCHAR(1024) NOT NULL, + ADD COLUMN "subject_cid" VARCHAR(256) NOT NULL; +CREATE UNIQUE INDEX "idx_episode_states_did_subject" ON "episode_states" ("did", "subject_uri"); + +-- ----- Subscriptions ----------------------------------------------------- + +DROP INDEX IF EXISTS "idx_subscriptions_feed_id"; +ALTER TABLE "subscriptions" + DROP COLUMN "feed_id", + DROP COLUMN "feed_url", + DROP COLUMN "podcast_guid", + ADD COLUMN "subject_uri" VARCHAR(1024) NOT NULL, + ADD COLUMN "subject_cid" VARCHAR(256) NOT NULL; +CREATE INDEX "idx_subscriptions_subject" ON "subscriptions" ("subject_uri"); + +-- ----- Stats keyed by AT-URI -------------------------------------------- +-- The old stats tables were keyed on PI integer IDs. Drop and recreate +-- keyed on the catalog AT-URI so the entire data model speaks one identifier. + +DROP TABLE IF EXISTS "podcast_stats"; +DROP TABLE IF EXISTS "episode_stats"; + +CREATE TABLE "episode_stats" ( + "subject_uri" VARCHAR(1024) PRIMARY KEY, + "comment_count" INTEGER NOT NULL DEFAULT 0, + "recommendation_count" INTEGER NOT NULL DEFAULT 0, + "bookmark_count" INTEGER NOT NULL DEFAULT 0, + "like_count" INTEGER NOT NULL DEFAULT 0, + "last_updated" TIMESTAMPTZ NOT NULL DEFAULT NOW() +); + +CREATE TABLE "podcast_stats" ( + "subject_uri" VARCHAR(1024) PRIMARY KEY, + "subscriber_count" INTEGER NOT NULL DEFAULT 0, + "comment_count" INTEGER NOT NULL DEFAULT 0, + "recommendation_count" INTEGER NOT NULL DEFAULT 0, + "last_updated" TIMESTAMPTZ NOT NULL DEFAULT NOW() +); + +-- ----- Comment likes ----------------------------------------------------- + +CREATE TABLE "comment_likes" ( + "id" BIGSERIAL PRIMARY KEY, + "did" VARCHAR(255) NOT NULL, + "rkey" VARCHAR(512) NOT NULL, + "cid" VARCHAR(256), + "at_uri" VARCHAR(1024) NOT NULL, + "subject_uri" VARCHAR(1024) NOT NULL, + "subject_cid" VARCHAR(256) NOT NULL, + "created_at" VARCHAR(64) NOT NULL, + "indexed_at" TIMESTAMPTZ NOT NULL DEFAULT NOW() +); +CREATE UNIQUE INDEX "idx_comment_likes_did_rkey" ON "comment_likes" ("did", "rkey"); +CREATE INDEX "idx_comment_likes_subject" ON "comment_likes" ("subject_uri"); +-- One like per (user, comment) — a re-like is a no-op upsert against this. +CREATE UNIQUE INDEX "idx_comment_likes_did_subject" ON "comment_likes" ("did", "subject_uri"); diff --git a/appview/database/migrations/0007_phase3_notifications.sql b/appview/database/migrations/0007_phase3_notifications.sql new file mode 100644 index 0000000..ffb2a54 --- /dev/null +++ b/appview/database/migrations/0007_phase3_notifications.sql @@ -0,0 +1,67 @@ +-- Phase 3: notifications, viewer-state, per-reason prefs, APNs device tokens. +-- +-- notifications — one row per (recipient, source) social event +-- notification_state — per-viewer "seenAt" watermark for unread counts +-- notification_prefs — per-(viewer, reason) toggle; missing rows mean enabled +-- device_tokens — APNs registrations, one row per (did, token) +-- push_outbox — pending APNs deliveries, drained by push_dispatcher + +CREATE TABLE "notifications" ( + "id" BIGSERIAL PRIMARY KEY, + "recipient_did" VARCHAR(255) NOT NULL, + "reason" VARCHAR(32) NOT NULL, + "actor_did" VARCHAR(255) NOT NULL, + "subject_uri" VARCHAR(1024) NOT NULL, + "source_uri" VARCHAR(1024) NOT NULL, + "created_at" TIMESTAMPTZ NOT NULL DEFAULT NOW() +); +-- The primary read pattern is "show me my newest notifications"; +-- recipient + id-desc covers it without touching created_at. +CREATE INDEX "idx_notifications_recipient_id" + ON "notifications" ("recipient_did", "id" DESC); +-- Dedupe scan against recent rows: same (recipient, actor, reason, source) +-- inside a small time window should be a no-op rather than a duplicate. +CREATE UNIQUE INDEX "idx_notifications_dedupe" + ON "notifications" ("recipient_did", "actor_did", "reason", "source_uri"); + +CREATE TABLE "notification_state" ( + "did" VARCHAR(255) PRIMARY KEY, + "seen_at" TIMESTAMPTZ NOT NULL DEFAULT NOW() +); + +CREATE TABLE "notification_prefs" ( + "did" VARCHAR(255) NOT NULL, + "reason" VARCHAR(32) NOT NULL, + "enabled" BOOLEAN NOT NULL DEFAULT TRUE, + PRIMARY KEY ("did", "reason") +); + +CREATE TABLE "device_tokens" ( + "id" BIGSERIAL PRIMARY KEY, + "did" VARCHAR(255) NOT NULL, + "token" VARCHAR(512) NOT NULL, + "environment" VARCHAR(16) NOT NULL, + "bundle_id" VARCHAR(255), + "created_at" TIMESTAMPTZ NOT NULL DEFAULT NOW(), + "last_seen_at" TIMESTAMPTZ NOT NULL DEFAULT NOW(), + "invalid_at" TIMESTAMPTZ +); +CREATE UNIQUE INDEX "idx_device_tokens_did_token" + ON "device_tokens" ("did", "token"); +CREATE INDEX "idx_device_tokens_did" + ON "device_tokens" ("did") + WHERE "invalid_at" IS NULL; + +CREATE TABLE "push_outbox" ( + "id" BIGSERIAL PRIMARY KEY, + "notification_id" BIGINT NOT NULL REFERENCES "notifications"("id") ON DELETE CASCADE, + "recipient_did" VARCHAR(255) NOT NULL, + "payload" JSONB NOT NULL, + "attempts" INTEGER NOT NULL DEFAULT 0, + "queued_at" TIMESTAMPTZ NOT NULL DEFAULT NOW(), + "sent_at" TIMESTAMPTZ, + "last_error" TEXT +); +CREATE INDEX "idx_push_outbox_pending" + ON "push_outbox" ("queued_at") + WHERE "sent_at" IS NULL; diff --git a/appview/database/migrations/0008_phase4_moderation.sql b/appview/database/migrations/0008_phase4_moderation.sql new file mode 100644 index 0000000..4419200 --- /dev/null +++ b/appview/database/migrations/0008_phase4_moderation.sql @@ -0,0 +1,37 @@ +-- Phase 4: moderation primitives. +-- +-- thread_settings — per-root-comment reply restrictions written by xyz.effem.feed.threadgate. +-- labels — moderation labels keyed (src_did, subject_uri, val). Effem's own labeler +-- plus any external labelers the AppView imports from. +-- comments.self_labels — author-applied content warnings. +-- profiles.labeler_subscriptions — DIDs whose labels this user trusts. + +CREATE TABLE "thread_settings" ( + "root_uri" VARCHAR(1024) PRIMARY KEY, + "root_did" VARCHAR(255) NOT NULL, + "rkey" VARCHAR(512) NOT NULL, + "allow" JSONB, + "created_at" VARCHAR(64) NOT NULL, + "indexed_at" TIMESTAMPTZ NOT NULL DEFAULT NOW() +); +CREATE INDEX "idx_thread_settings_root_did" ON "thread_settings" ("root_did"); + +CREATE TABLE "labels" ( + "id" BIGSERIAL PRIMARY KEY, + "src_did" VARCHAR(255) NOT NULL, + "subject_uri" VARCHAR(1024) NOT NULL, + "val" VARCHAR(128) NOT NULL, + "neg" BOOLEAN NOT NULL DEFAULT FALSE, + "created_at" VARCHAR(64) NOT NULL, + "indexed_at" TIMESTAMPTZ NOT NULL DEFAULT NOW() +); +-- One row per (src, subject, val) — re-applying a label idempotently updates +-- the existing row (incl. flipping `neg`) instead of producing duplicates. +CREATE UNIQUE INDEX "idx_labels_src_subject_val" ON "labels" ("src_did", "subject_uri", "val"); +CREATE INDEX "idx_labels_subject" ON "labels" ("subject_uri"); + +ALTER TABLE "comments" + ADD COLUMN "self_labels" JSONB; + +ALTER TABLE "profiles" + ADD COLUMN "labeler_subscriptions" JSONB; diff --git a/appview/database/models.go b/appview/database/models.go index 3879007..f36ada6 100644 --- a/appview/database/models.go +++ b/appview/database/models.go @@ -10,36 +10,39 @@ type FirehoseCursor struct { func (FirehoseCursor) TableName() string { return "firehose_cursor" } +// Subscription stores a user's podcast subscription, referencing a podcast +// catalog record via strongRef. type Subscription struct { - ID uint `gorm:"primaryKey" json:"-"` - DID string `gorm:"column:did;size:255;not null;index:idx_subscriptions_did_rkey,unique;index:idx_subscriptions_did" json:"did"` - Rkey string `gorm:"size:512;not null;index:idx_subscriptions_did_rkey,unique" json:"rkey"` - FeedID int64 `gorm:"not null;index:idx_subscriptions_feed_id" json:"feed_id"` - FeedURL string `gorm:"size:2048" json:"feed_url"` - PodcastGuid string `gorm:"size:512" json:"podcast_guid"` - CreatedAt string `gorm:"size:64;not null" json:"created_at"` - IndexedAt time.Time `gorm:"autoCreateTime" json:"-"` + ID uint `gorm:"primaryKey" json:"-"` + DID string `gorm:"column:did;size:255;not null;index:idx_subscriptions_did_rkey,unique;index:idx_subscriptions_did" json:"did"` + Rkey string `gorm:"size:512;not null;index:idx_subscriptions_did_rkey,unique" json:"rkey"` + SubjectURI string `gorm:"size:1024;not null;index:idx_subscriptions_subject" json:"subject_uri"` + SubjectCID string `gorm:"column:subject_cid;size:256;not null" json:"subject_cid"` + CreatedAt string `gorm:"size:64;not null" json:"created_at"` + IndexedAt time.Time `gorm:"autoCreateTime" json:"-"` } func (Subscription) TableName() string { return "subscriptions" } +// Comment stores a comment on an episode (or reply to another comment). +// `subject_uri` / `subject_cid` reference the episode's catalog record; +// `reply_root` / `reply_parent` reference other comments via strongRef. type Comment struct { - ID uint `gorm:"primaryKey" json:"-"` - DID string `gorm:"column:did;size:255;not null;index:idx_comments_did_rkey,unique;index:idx_comments_did" json:"did"` - Rkey string `gorm:"size:512;not null;index:idx_comments_did_rkey,unique" json:"rkey"` - CID string `gorm:"column:cid;size:256" json:"cid"` - ATURI string `gorm:"column:at_uri;size:1024;index;not null" json:"at_uri"` - FeedID int64 `gorm:"not null;index:idx_comments_episode" json:"feed_id"` - EpisodeID int64 `gorm:"not null;index:idx_comments_episode" json:"episode_id"` - EpisodeGuid string `gorm:"size:512" json:"episode_guid"` - PodcastGuid string `gorm:"size:512" json:"podcast_guid"` - Text string `gorm:"type:text;not null" json:"text"` - TimestampS *int `gorm:"index" json:"timestamp_s"` - ReplyRoot string `gorm:"size:1024;index:idx_comments_reply_root" json:"reply_root"` - ReplyRootCID string `gorm:"column:reply_root_cid;size:256" json:"reply_root_cid"` - ReplyParent string `gorm:"size:1024" json:"reply_parent"` - ReplyParentCID string `gorm:"column:reply_parent_cid;size:256" json:"reply_parent_cid"` - Facets []byte `gorm:"type:jsonb" json:"facets"` + ID uint `gorm:"primaryKey" json:"-"` + DID string `gorm:"column:did;size:255;not null;index:idx_comments_did_rkey,unique;index:idx_comments_did" json:"did"` + Rkey string `gorm:"size:512;not null;index:idx_comments_did_rkey,unique" json:"rkey"` + CID string `gorm:"column:cid;size:256" json:"cid"` + ATURI string `gorm:"column:at_uri;size:1024;index;not null" json:"at_uri"` + SubjectURI string `gorm:"size:1024;not null;index:idx_comments_subject" json:"subject_uri"` + SubjectCID string `gorm:"column:subject_cid;size:256;not null" json:"subject_cid"` + Text string `gorm:"type:text;not null" json:"text"` + TimestampS *int `gorm:"index" json:"timestamp_s"` + ReplyRoot string `gorm:"size:1024;index:idx_comments_reply_root" json:"reply_root"` + ReplyRootCID string `gorm:"column:reply_root_cid;size:256" json:"reply_root_cid"` + ReplyParent string `gorm:"size:1024" json:"reply_parent"` + ReplyParentCID string `gorm:"column:reply_parent_cid;size:256" json:"reply_parent_cid"` + Facets []byte `gorm:"type:jsonb" json:"facets"` + SelfLabels []byte `gorm:"column:self_labels;type:jsonb" json:"self_labels"` // Removed is set to true when an admin takes the content down. Reads // filter it out; the row stays for audit. Indexer struct-updates skip // this field because its zero value is false, so a firehose re-index of @@ -52,14 +55,12 @@ type Comment struct { func (Comment) TableName() string { return "comments" } type Recommendation struct { - ID uint `gorm:"primaryKey" json:"-"` - DID string `gorm:"column:did;size:255;not null;index:idx_recommendations_did_rkey,unique;index:idx_recommendations_did" json:"did"` - Rkey string `gorm:"size:512;not null;index:idx_recommendations_did_rkey,unique" json:"rkey"` - FeedID int64 `gorm:"not null;index:idx_recommendations_episode" json:"feed_id"` - EpisodeID int64 `gorm:"not null;index:idx_recommendations_episode" json:"episode_id"` - EpisodeGuid string `gorm:"size:512" json:"episode_guid"` - PodcastGuid string `gorm:"size:512" json:"podcast_guid"` - Text string `gorm:"type:text" json:"text"` + ID uint `gorm:"primaryKey" json:"-"` + DID string `gorm:"column:did;size:255;not null;index:idx_recommendations_did_rkey,unique;index:idx_recommendations_did" json:"did"` + Rkey string `gorm:"size:512;not null;index:idx_recommendations_did_rkey,unique" json:"rkey"` + SubjectURI string `gorm:"size:1024;not null;index:idx_recommendations_subject" json:"subject_uri"` + SubjectCID string `gorm:"column:subject_cid;size:256;not null" json:"subject_cid"` + Text string `gorm:"type:text" json:"text"` // See Comment.Removed for semantics. Removed bool `gorm:"not null;default:false" json:"removed"` CreatedAt string `gorm:"size:64;not null" json:"created_at"` @@ -82,26 +83,28 @@ type PodcastList struct { func (PodcastList) TableName() string { return "podcast_lists" } type Bookmark struct { - ID uint `gorm:"primaryKey" json:"-"` - DID string `gorm:"column:did;size:255;not null;index:idx_bookmarks_did_rkey,unique;index:idx_bookmarks_did" json:"did"` - Rkey string `gorm:"size:512;not null;index:idx_bookmarks_did_rkey,unique" json:"rkey"` - FeedID int64 `gorm:"not null;index:idx_bookmarks_episode" json:"feed_id"` - EpisodeID int64 `gorm:"not null;index:idx_bookmarks_episode" json:"episode_id"` - EpisodeGuid string `gorm:"size:512" json:"episode_guid"` - PodcastGuid string `gorm:"size:512" json:"podcast_guid"` - TimestampS *int `gorm:"index" json:"timestamp"` - CreatedAt string `gorm:"size:64;not null" json:"created_at"` - IndexedAt time.Time `gorm:"autoCreateTime" json:"-"` + ID uint `gorm:"primaryKey" json:"-"` + DID string `gorm:"column:did;size:255;not null;index:idx_bookmarks_did_rkey,unique;index:idx_bookmarks_did" json:"did"` + Rkey string `gorm:"size:512;not null;index:idx_bookmarks_did_rkey,unique" json:"rkey"` + SubjectURI string `gorm:"size:1024;not null;index:idx_bookmarks_subject" json:"subject_uri"` + SubjectCID string `gorm:"column:subject_cid;size:256;not null" json:"subject_cid"` + TimestampS *int `gorm:"index" json:"timestamp"` + CreatedAt string `gorm:"size:64;not null" json:"created_at"` + IndexedAt time.Time `gorm:"autoCreateTime" json:"-"` } func (Bookmark) TableName() string { return "bookmarks" } type Profile struct { - DID string `gorm:"column:did;primaryKey;size:255" json:"did"` - DisplayName string `gorm:"size:640" json:"display_name"` - Description string `gorm:"type:text" json:"description"` - FavoriteGenres []byte `gorm:"type:jsonb" json:"-"` - IndexedAt time.Time `gorm:"autoCreateTime" json:"-"` + DID string `gorm:"column:did;primaryKey;size:255" json:"did"` + Handle string `gorm:"size:255" json:"handle"` + HandleResolvedAt *time.Time `gorm:"column:handle_resolved_at" json:"-"` + DisplayName string `gorm:"size:640" json:"display_name"` + Description string `gorm:"type:text" json:"description"` + FavoriteGenres []byte `gorm:"type:jsonb" json:"-"` + AvatarCID string `gorm:"column:avatar_cid;size:256" json:"avatar_cid"` + LabelerSubscriptions []byte `gorm:"column:labeler_subscriptions;type:jsonb" json:"-"` + IndexedAt time.Time `gorm:"autoCreateTime" json:"-"` } func (Profile) TableName() string { return "profiles" } @@ -117,43 +120,43 @@ type PICache struct { func (PICache) TableName() string { return "pi_cache" } +// PodcastStats keyed on the podcast's catalog AT-URI. The same key is used +// for any aggregation we want to expose to clients. type PodcastStats struct { - FeedID int64 `gorm:"primaryKey"` - SubscriberCount int `gorm:"default:0"` - CommentCount int `gorm:"default:0"` - RecommendationCount int `gorm:"default:0"` - LastUpdated time.Time `gorm:"autoUpdateTime"` + SubjectURI string `gorm:"primaryKey;size:1024" json:"subject_uri"` + SubscriberCount int `gorm:"not null;default:0" json:"subscriber_count"` + CommentCount int `gorm:"not null;default:0" json:"comment_count"` + RecommendationCount int `gorm:"not null;default:0" json:"recommendation_count"` + LastUpdated time.Time `gorm:"autoUpdateTime" json:"last_updated"` } func (PodcastStats) TableName() string { return "podcast_stats" } type EpisodeStats struct { - EpisodeID int64 `gorm:"primaryKey"` - FeedID int64 `gorm:"not null;index:idx_episode_stats_feed"` - CommentCount int `gorm:"default:0"` - RecommendationCount int `gorm:"default:0"` - BookmarkCount int `gorm:"default:0"` - LastUpdated time.Time `gorm:"autoUpdateTime"` + SubjectURI string `gorm:"primaryKey;size:1024" json:"subject_uri"` + CommentCount int `gorm:"not null;default:0" json:"comment_count"` + RecommendationCount int `gorm:"not null;default:0" json:"recommendation_count"` + BookmarkCount int `gorm:"not null;default:0" json:"bookmark_count"` + LikeCount int `gorm:"not null;default:0" json:"like_count"` + LastUpdated time.Time `gorm:"autoUpdateTime" json:"last_updated"` } func (EpisodeStats) TableName() string { return "episode_stats" } type EpisodeState struct { - ID uint `gorm:"primaryKey" json:"-"` - DID string `gorm:"column:did;size:255;not null;index:idx_episode_states_did_rkey,unique;index:idx_episode_states_did" json:"did"` - Rkey string `gorm:"size:512;not null;index:idx_episode_states_did_rkey,unique" json:"rkey"` - FeedID int64 `gorm:"not null;index:idx_episode_states_episode,unique" json:"feed_id"` - EpisodeID int64 `gorm:"not null;index:idx_episode_states_episode,unique" json:"episode_id"` - EpisodeGuid string `gorm:"size:512" json:"episode_guid"` - PodcastGuid string `gorm:"size:512" json:"podcast_guid"` - PositionS *int `json:"position_s"` - DurationS *int `json:"duration_s"` - Played bool `gorm:"default:false" json:"played"` - Saved bool `gorm:"default:false" json:"saved"` - Hidden bool `gorm:"default:false" json:"hidden"` - UpdatedAt string `gorm:"size:64" json:"updated_at"` - CreatedAt string `gorm:"size:64;not null" json:"created_at"` - IndexedAt time.Time `gorm:"autoCreateTime" json:"-"` + ID uint `gorm:"primaryKey" json:"-"` + DID string `gorm:"column:did;size:255;not null;index:idx_episode_states_did_rkey,unique;index:idx_episode_states_did" json:"did"` + Rkey string `gorm:"size:512;not null;index:idx_episode_states_did_rkey,unique" json:"rkey"` + SubjectURI string `gorm:"size:1024;not null;index:idx_episode_states_did_subject,unique" json:"subject_uri"` + SubjectCID string `gorm:"column:subject_cid;size:256;not null" json:"subject_cid"` + PositionS *int `json:"position_s"` + DurationS *int `json:"duration_s"` + Played bool `gorm:"default:false" json:"played"` + Saved bool `gorm:"default:false" json:"saved"` + Hidden bool `gorm:"default:false" json:"hidden"` + UpdatedAt string `gorm:"size:64" json:"updated_at"` + CreatedAt string `gorm:"size:64;not null" json:"created_at"` + IndexedAt time.Time `gorm:"autoCreateTime" json:"-"` } func (EpisodeState) TableName() string { return "episode_states" } @@ -202,3 +205,212 @@ type AdminAudit struct { } func (AdminAudit) TableName() string { return "admin_audit" } + +// IndexerError records a single record the indexer could not process. The +// watchdog retries unresolved rows; admins inspect them via the listIndexerErrors +// XRPC. Uniqueness is on (did, collection, rkey) so a re-encounter updates the +// existing row instead of producing a duplicate. +type IndexerError struct { + ID uint `gorm:"primaryKey" json:"id"` + DID string `gorm:"column:did;size:255;not null;index:idx_indexer_errors_record,unique" json:"did"` + Collection string `gorm:"size:255;not null;index:idx_indexer_errors_record,unique" json:"collection"` + Rkey string `gorm:"size:512;not null;index:idx_indexer_errors_record,unique" json:"rkey"` + CID string `gorm:"column:cid;size:256" json:"cid,omitempty"` + ErrorMessage string `gorm:"type:text;not null" json:"error_message"` + Attempts int `gorm:"not null;default:1" json:"attempts"` + LastAttemptAt time.Time `gorm:"not null" json:"last_attempt_at"` + ResolvedAt *time.Time `gorm:"column:resolved_at" json:"resolved_at,omitempty"` + CreatedAt time.Time `gorm:"autoCreateTime" json:"created_at"` +} + +func (IndexerError) TableName() string { return "indexer_errors" } + +// AccountStatus tracks PDS/Relay activation state per DID, updated from the +// firehose #account event. Active = false means the account is takendown, +// suspended, or deactivated; read handlers filter such DIDs out. +type AccountStatus struct { + DID string `gorm:"column:did;primaryKey;size:255" json:"did"` + Active bool `gorm:"not null;default:true" json:"active"` + Status string `gorm:"size:64" json:"status,omitempty"` + UpdatedAt time.Time `gorm:"not null" json:"updated_at"` +} + +func (AccountStatus) TableName() string { return "account_status" } + +// DIDPDS caches the PDS service endpoint resolved from a DID's identity +// document. Used to construct blob URLs and to call listRecords during +// backfill. Re-resolved opportunistically when the cached row is older than +// the freshness window (defined at the resolver layer). +type DIDPDS struct { + DID string `gorm:"column:did;primaryKey;size:255" json:"did"` + PDSEndpoint string `gorm:"column:pds_endpoint;size:2048;not null" json:"pds_endpoint"` + ResolvedAt time.Time `gorm:"column:resolved_at;not null" json:"resolved_at"` +} + +func (DIDPDS) TableName() string { return "did_pds" } + +// EpisodeCatalog is the canonical catalog record for a podcast episode. Every +// comment, recommendation, bookmark, and episode-state record references one +// of these via {subject_uri, subject_cid}. The (podcast_guid, episode_guid) +// pair is the natural key; at_uri / cid are derived deterministically so the +// same episode always lands at the same strongRef regardless of who publishes +// the record. +type EpisodeCatalog struct { + ID uint `gorm:"primaryKey" json:"-"` + PodcastGuid string `gorm:"column:podcast_guid;size:512;not null;index:idx_episode_catalog_guids,unique" json:"podcast_guid"` + EpisodeGuid string `gorm:"column:episode_guid;size:512;not null;index:idx_episode_catalog_guids,unique" json:"episode_guid"` + ATURI string `gorm:"column:at_uri;size:1024;not null;index:idx_episode_catalog_at_uri,unique" json:"at_uri"` + CID string `gorm:"column:cid;size:256;not null" json:"cid"` + Rkey string `gorm:"size:512;not null" json:"rkey"` + Title string `gorm:"size:1024;not null" json:"title"` + PublishedAt string `gorm:"size:64;not null" json:"published_at"` + FeedURL string `gorm:"column:feed_url;size:2048" json:"feed_url,omitempty"` + EnclosureURL string `gorm:"column:enclosure_url;size:2048" json:"enclosure_url,omitempty"` + DurationS *int `gorm:"column:duration_s" json:"duration_s,omitempty"` + PodcastIndexFeedID *int64 `gorm:"column:podcast_index_feed_id;index:idx_episode_catalog_pi" json:"podcast_index_feed_id,omitempty"` + PodcastIndexEpisodeID *int64 `gorm:"column:podcast_index_episode_id;index:idx_episode_catalog_pi" json:"podcast_index_episode_id,omitempty"` + IndexedAt time.Time `gorm:"autoCreateTime" json:"indexed_at"` +} + +func (EpisodeCatalog) TableName() string { return "episode_catalog" } + +// PodcastCatalog mirrors EpisodeCatalog but at the show level. Subjects of +// subscriptions and feed-level engagement. +type PodcastCatalog struct { + ID uint `gorm:"primaryKey" json:"-"` + PodcastGuid string `gorm:"column:podcast_guid;size:512;not null;index:idx_podcast_catalog_guid,unique" json:"podcast_guid"` + ATURI string `gorm:"column:at_uri;size:1024;not null;index:idx_podcast_catalog_at_uri,unique" json:"at_uri"` + CID string `gorm:"column:cid;size:256;not null" json:"cid"` + Rkey string `gorm:"size:512;not null" json:"rkey"` + Title string `gorm:"size:1024;not null" json:"title"` + FeedURL string `gorm:"column:feed_url;size:2048" json:"feed_url,omitempty"` + Author string `gorm:"size:1024" json:"author,omitempty"` + ArtworkURL string `gorm:"column:artwork_url;size:2048" json:"artwork_url,omitempty"` + Language string `gorm:"size:32" json:"language,omitempty"` + PodcastIndexFeedID *int64 `gorm:"column:podcast_index_feed_id;index:idx_podcast_catalog_pi" json:"podcast_index_feed_id,omitempty"` + IndexedAt time.Time `gorm:"autoCreateTime" json:"indexed_at"` +} + +func (PodcastCatalog) TableName() string { return "podcast_catalog" } + +// CommentLike captures an xyz.effem.feed.commentLike record. Uniqueness on +// (did, subject_uri) means a re-like is a no-op upsert against the same row; +// the firehose event for an unlike (delete) drops the row outright. +type CommentLike struct { + ID uint `gorm:"primaryKey" json:"-"` + DID string `gorm:"column:did;size:255;not null;index:idx_comment_likes_did_rkey,unique;index:idx_comment_likes_did_subject,unique" json:"did"` + Rkey string `gorm:"size:512;not null;index:idx_comment_likes_did_rkey,unique" json:"rkey"` + CID string `gorm:"column:cid;size:256" json:"cid,omitempty"` + ATURI string `gorm:"column:at_uri;size:1024;not null" json:"at_uri"` + SubjectURI string `gorm:"size:1024;not null;index:idx_comment_likes_subject;index:idx_comment_likes_did_subject,unique" json:"subject_uri"` + SubjectCID string `gorm:"column:subject_cid;size:256;not null" json:"subject_cid"` + CreatedAt string `gorm:"size:64;not null" json:"created_at"` + IndexedAt time.Time `gorm:"autoCreateTime" json:"-"` +} + +func (CommentLike) TableName() string { return "comment_likes" } + +// Notification represents one user-visible event the AppView surfaces to +// the recipient (a reply to one of their comments, a mention, a like on a +// comment they wrote). Uniqueness on (recipient, actor, reason, source_uri) +// means re-encountering the same source record after a firehose replay +// produces a no-op upsert, not a duplicate row. +type Notification struct { + ID uint `gorm:"primaryKey" json:"id"` + RecipientDID string `gorm:"column:recipient_did;size:255;not null;index:idx_notifications_recipient_id;index:idx_notifications_dedupe,unique" json:"recipient_did"` + Reason string `gorm:"size:32;not null;index:idx_notifications_dedupe,unique" json:"reason"` + ActorDID string `gorm:"column:actor_did;size:255;not null;index:idx_notifications_dedupe,unique" json:"actor_did"` + SubjectURI string `gorm:"size:1024;not null" json:"subject_uri"` + SourceURI string `gorm:"size:1024;not null;index:idx_notifications_dedupe,unique" json:"source_uri"` + CreatedAt time.Time `gorm:"not null;autoCreateTime" json:"created_at"` +} + +func (Notification) TableName() string { return "notifications" } + +// NotificationState holds the viewer's last-seen watermark. Notifications +// older than seen_at render as already-read. +type NotificationState struct { + DID string `gorm:"column:did;primaryKey;size:255" json:"did"` + SeenAt time.Time `gorm:"column:seen_at;not null" json:"seen_at"` +} + +func (NotificationState) TableName() string { return "notification_state" } + +// NotificationPref toggles a single reason on or off for a viewer. Missing +// rows mean "enabled" so a fresh user receives every category by default. +type NotificationPref struct { + DID string `gorm:"column:did;primaryKey;size:255" json:"did"` + Reason string `gorm:"primaryKey;size:32" json:"reason"` + Enabled bool `gorm:"not null;default:true" json:"enabled"` +} + +func (NotificationPref) TableName() string { return "notification_prefs" } + +// DeviceToken is one APNs registration. A user with two iOS devices has two +// rows here. `invalid_at` is stamped by the APNs feedback loop so the +// dispatcher stops sending to dead tokens. +type DeviceToken struct { + ID uint `gorm:"primaryKey" json:"id"` + DID string `gorm:"column:did;size:255;not null;index:idx_device_tokens_did_token,unique" json:"did"` + Token string `gorm:"size:512;not null;index:idx_device_tokens_did_token,unique" json:"token"` + Environment string `gorm:"size:16;not null" json:"environment"` + BundleID string `gorm:"column:bundle_id;size:255" json:"bundle_id,omitempty"` + CreatedAt time.Time `gorm:"not null;autoCreateTime" json:"created_at"` + LastSeenAt time.Time `gorm:"column:last_seen_at;not null" json:"last_seen_at"` + InvalidAt *time.Time `gorm:"column:invalid_at" json:"invalid_at,omitempty"` +} + +func (DeviceToken) TableName() string { return "device_tokens" } + +// PushOutbox is the durable queue between the indexer (which writes +// notifications) and the dispatcher (which sends APNs). Decoupling the two +// means a transient APNs outage doesn't block the firehose. +type PushOutbox struct { + ID uint `gorm:"primaryKey" json:"id"` + NotificationID uint `gorm:"column:notification_id;not null" json:"notification_id"` + RecipientDID string `gorm:"column:recipient_did;size:255;not null" json:"recipient_did"` + Payload []byte `gorm:"type:jsonb;not null" json:"-"` + Attempts int `gorm:"not null;default:0" json:"attempts"` + QueuedAt time.Time `gorm:"column:queued_at;not null;autoCreateTime" json:"queued_at"` + SentAt *time.Time `gorm:"column:sent_at" json:"sent_at,omitempty"` + LastError string `gorm:"column:last_error;type:text" json:"last_error,omitempty"` +} + +func (PushOutbox) TableName() string { return "push_outbox" } + +// ThreadSettings records the per-thread reply restrictions written by an +// xyz.effem.feed.threadgate record. `allow` is a JSONB-encoded array of +// rule unions (today only {"$type": "xyz.effem.feed.threadgate#mentionRule"}). +// Semantics: +// +// nil — record exists but `allow` was omitted; anyone may reply. +// [] — empty allowlist; only the root comment's author may reply. +// [...] — replies must satisfy at least one rule (or be the root author). +// +// Missing row means no threadgate; anyone may reply. +type ThreadSettings struct { + RootURI string `gorm:"column:root_uri;primaryKey;size:1024" json:"root_uri"` + RootDID string `gorm:"column:root_did;size:255;not null;index:idx_thread_settings_root_did" json:"root_did"` + Rkey string `gorm:"size:512;not null" json:"rkey"` + Allow []byte `gorm:"type:jsonb" json:"-"` + CreatedAt string `gorm:"size:64;not null" json:"created_at"` + IndexedAt time.Time `gorm:"autoCreateTime" json:"-"` +} + +func (ThreadSettings) TableName() string { return "thread_settings" } + +// Label is one row in the AppView's moderation-label store. Source DID +// identifies the labeler (effem's own labeler DID or a labeler the AppView +// has imported from). Uniqueness on (src_did, subject_uri, val) keeps the +// table compact across re-applications. +type Label struct { + ID uint `gorm:"primaryKey" json:"id"` + SrcDID string `gorm:"column:src_did;size:255;not null;index:idx_labels_src_subject_val,unique" json:"src"` + SubjectURI string `gorm:"column:subject_uri;size:1024;not null;index:idx_labels_src_subject_val,unique;index:idx_labels_subject" json:"subject_uri"` + Val string `gorm:"size:128;not null;index:idx_labels_src_subject_val,unique" json:"val"` + Neg bool `gorm:"not null;default:false" json:"neg"` + CreatedAt string `gorm:"size:64;not null" json:"created_at"` + IndexedAt time.Time `gorm:"autoCreateTime" json:"-"` +} + +func (Label) TableName() string { return "labels" } diff --git a/appview/firehose.go b/appview/firehose.go index f0dcde6..9d73108 100644 --- a/appview/firehose.go +++ b/appview/firehose.go @@ -3,6 +3,7 @@ package appview import ( "bytes" "context" + "errors" "fmt" "net/http" "net/url" @@ -23,6 +24,15 @@ import ( "tangled.org/sparrowtek.com/effem-AppView/appview/metrics" ) +// Sentinel errors recorded against indexer_errors when a record can't even +// reach the indexer. These show up in the admin UI so an operator can spot +// firehose-level corruption distinct from indexer-level validation failures. +var ( + errMissingCARBlocks = errors.New("missing CAR blocks for create/update") + errCIDMismatch = errors.New("record CID mismatch between commit op and CAR") + errNilRecordPayload = errors.New("nil record payload in CAR") +) + // firehoseCursorRowID is the singleton row holding the stream seq. We pin it // to 1 so the persist path can run as a single upsert keyed on the PK. const firehoseCursorRowID = 1 @@ -168,8 +178,9 @@ func (srv *Server) RunFirehoseConsumer(ctx context.Context) error { return nil }, // #account surfaces PDS/Relay status changes (takedowns, deactivations). - // We log + count for now; enforcement (hiding records from inactive - // accounts) will land when we add an account-status table. + // Persisted into account_status so read handlers can hide records + // from inactive accounts. Unknown DIDs are implicitly active — we + // only write rows when the relay tells us something changed. RepoAccount: func(evt *comatproto.SyncSubscribeRepos_Account) error { atomic.StoreInt64(&srv.lastSeq, evt.Seq) srv.lastSeqTime.Store(time.Now()) @@ -183,6 +194,9 @@ func (srv *Server) RunFirehoseConsumer(ctx context.Context) error { status = *evt.Status } srv.logger.Debug("account event", "did", evt.Did, "active", evt.Active, "status", status, "seq", evt.Seq) + if err := srv.upsertAccountStatus(ctx, evt.Did, evt.Active, status); err != nil { + srv.logger.Warn("account_status upsert failed", "err", err, "did", evt.Did) + } return nil }, // #info carries out-of-band notices like OutdatedCursor. No Seq field, @@ -264,32 +278,44 @@ func (srv *Server) handleCommit(ctx context.Context, evt *comatproto.SyncSubscri case repomgr.EvtKindCreateRecord, repomgr.EvtKindUpdateRecord: if rr == nil { srv.logger.Warn("missing CAR blocks for create/update", "path", op.Path) + srv.indexer.RecordError(ctx, evt.Repo, collectionName, rkey.String(), "", errMissingCARBlocks) continue } recCID, recCBOR, err := rr.GetRecordBytes(ctx, op.Path) if err != nil { srv.logger.Warn("failed to get record", "err", err, "path", op.Path) metrics.IncIndexerError(collectionName, op.Action) + srv.indexer.RecordError(ctx, evt.Repo, collectionName, rkey.String(), "", err) continue } if op.Cid != nil && lexutil.LexLink(recCID) != *op.Cid { srv.logger.Warn("record CID mismatch", "path", op.Path, "carCID", recCID, "opCID", op.Cid) metrics.IncIndexerError(collectionName, op.Action) + srv.indexer.RecordError(ctx, evt.Repo, collectionName, rkey.String(), recCID.String(), errCIDMismatch) continue } if recCBOR == nil { srv.logger.Warn("nil record payload", "path", op.Path) metrics.IncIndexerError(collectionName, op.Action) + srv.indexer.RecordError(ctx, evt.Repo, collectionName, rkey.String(), recCID.String(), errNilRecordPayload) continue } if err := srv.indexer.IndexRecord(ctx, evt.Repo, collectionName, rkey.String(), recCID.String(), *recCBOR); err != nil { srv.logger.Warn("failed to index record", "err", err, "path", op.Path) metrics.IncIndexerError(collectionName, op.Action) + srv.indexer.RecordError(ctx, evt.Repo, collectionName, rkey.String(), recCID.String(), err) + continue } + // Success — clear any previously recorded error for this record + // so the watchdog drops it. + srv.indexer.MarkErrorResolved(ctx, evt.Repo, collectionName, rkey.String()) case repomgr.EvtKindDeleteRecord: if err := srv.indexer.DeleteRecord(ctx, evt.Repo, collectionName, rkey.String()); err != nil { srv.logger.Warn("failed to delete record", "err", err, "path", op.Path) metrics.IncIndexerError(collectionName, op.Action) + srv.indexer.RecordError(ctx, evt.Repo, collectionName, rkey.String(), "", err) + } else { + srv.indexer.MarkErrorResolved(ctx, evt.Repo, collectionName, rkey.String()) } default: continue diff --git a/appview/handlers/admin.go b/appview/handlers/admin.go index 360cc35..ea9c11c 100644 --- a/appview/handlers/admin.go +++ b/appview/handlers/admin.go @@ -1,6 +1,7 @@ package handlers import ( + "errors" "net/http" "strconv" "time" @@ -8,6 +9,7 @@ import ( "github.com/bluesky-social/indigo/atproto/syntax" "github.com/labstack/echo/v4" "tangled.org/sparrowtek.com/effem-AppView/appview/database" + "tangled.org/sparrowtek.com/effem-AppView/appview/indexer" ) // Admin-moderatable collections. GetReportedContent supports reading any of @@ -24,6 +26,7 @@ const ( const ( auditActionResolveReport = "resolve_report" auditActionRemoveContent = "remove_content" + auditActionBackfillUser = "backfill_user" ) // reportItem is the wire shape of a report row. Reason is included for @@ -199,10 +202,11 @@ func (h *Handlers) GetReportedContent(c echo.Context) error { if err := db.Where("did = ? AND rkey = ?", did, rkey).First(&row).Error; err != nil { return writeError(c, http.StatusNotFound, "NotFound", "comment not found") } + authors := h.hydrateAuthors(ctx, []database.Comment{row}) return c.JSON(http.StatusOK, map[string]any{ "kind": "comment", "removed": row.Removed, - "content": toCommentResponse(row), + "content": toCommentResponse(row, authors[row.DID]), }) case collectionRecommendation: var row database.Recommendation @@ -297,3 +301,112 @@ func (h *Handlers) RemoveContent(c echo.Context) error { return c.JSON(http.StatusOK, map[string]any{"ok": true}) } + +type backfillUserRequest struct { + DID string `json:"did"` +} + +// BackfillUser walks every Effem collection in the named user's PDS and +// re-indexes each record. Idempotent — handlers' FirstOrCreate clauses +// silently no-op on records the AppView already has. Useful for recovering +// from extended firehose downtime or after wiping the AppView DB during +// development. +func (h *Handlers) BackfillUser(c echo.Context) error { + var req backfillUserRequest + if err := c.Bind(&req); err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "invalid JSON body") + } + if req.DID == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "did is required") + } + + result, httpErr := h.runBackfill(c, req.DID) + if httpErr != nil { + return httpErr + } + + h.writeAudit(c.Request().Context(), requestingDID(c), auditActionBackfillUser, req.DID, + map[string]any{ + "counts": result.Counts, + "recorded_errors": result.RecordedErrors, + }, + ) + + return c.JSON(http.StatusOK, result) +} + +// BackfillSelf re-indexes the authenticated user's own records. Intended for +// the iOS app to call on first login from a new device so reinstalls recover +// state without operator intervention. Distinct from the admin BackfillUser +// because it doesn't require admin scope and is implicitly scoped to the +// caller's own DID. +func (h *Handlers) BackfillSelf(c echo.Context) error { + did := requestingDID(c) + if did == "" { + return writeError(c, http.StatusUnauthorized, "AuthRequired", "authenticated DID is required") + } + + result, httpErr := h.runBackfill(c, did) + if httpErr != nil { + return httpErr + } + return c.JSON(http.StatusOK, result) +} + +// runBackfill is the shared body of BackfillUser and BackfillSelf. Returns +// either the result (on success) or an HTTP error to propagate. +func (h *Handlers) runBackfill(c echo.Context, did string) (*indexer.BackfillResult, error) { + result, err := h.indexer.BackfillUser(c.Request().Context(), did) + if err != nil { + if errors.Is(err, indexer.ErrBackfillInProgress) { + return nil, writeError(c, http.StatusConflict, + "BackfillInProgress", "a backfill is already running for this account") + } + return nil, h.internalError(c, "BackfillUser", err) + } + return result, nil +} + +// ListIndexerErrors returns per-record indexing failures, newest first. By +// default only unresolved rows are returned so operators see live problems; +// pass ?resolved=true to see historical ones. Filterable by collection. +func (h *Handlers) ListIndexerErrors(c echo.Context) error { + limit := parseLimit(c.QueryParam("limit"), 50, 200) + cursorRaw := c.QueryParam("cursor") + collection := c.QueryParam("collection") + includeResolved := c.QueryParam("resolved") == "true" + + q := h.db.WithContext(c.Request().Context()). + Model(&database.IndexerError{}). + Order("id DESC"). + Limit(limit + 1) + if !includeResolved { + q = q.Where("resolved_at IS NULL") + } + if collection != "" { + q = q.Where("collection = ?", collection) + } + if cursorRaw != "" { + cur, err := strconv.ParseUint(cursorRaw, 10, 64) + if err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "cursor must be a positive integer") + } + q = q.Where("id < ?", cur) + } + + var rows []database.IndexerError + if err := q.Find(&rows).Error; err != nil { + return h.internalError(c, "ListIndexerErrors.find", err) + } + + nextCursor := "" + if len(rows) > limit { + nextCursor = strconv.FormatUint(uint64(rows[limit-1].ID), 10) + rows = rows[:limit] + } + + return c.JSON(http.StatusOK, map[string]any{ + "errors": rows, + "cursor": nextCursor, + }) +} diff --git a/appview/handlers/bookmark.go b/appview/handlers/bookmark.go index ae56eb5..c2cbb9e 100644 --- a/appview/handlers/bookmark.go +++ b/appview/handlers/bookmark.go @@ -3,8 +3,8 @@ package handlers import ( "net/http" - "tangled.org/sparrowtek.com/effem-AppView/appview/database" "github.com/labstack/echo/v4" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" ) func (h *Handlers) GetBookmarks(c echo.Context) error { @@ -13,24 +13,21 @@ func (h *Handlers) GetBookmarks(c echo.Context) error { return writeError(c, http.StatusBadRequest, "InvalidRequest", "did is required") } - feedID := parseInt64(c.QueryParam("feedId"), 0) - episodeID := parseInt64(c.QueryParam("episodeId"), 0) + subjectURI := c.QueryParam("subject") limit := parseLimit(c.QueryParam("limit"), 50, 100) cursor := c.QueryParam("cursor") - q := h.db.WithContext(c.Request().Context()).Model(&database.Bookmark{}).Where("did = ?", did).Order("rkey DESC").Limit(limit + 1) - if feedID > 0 { - q = q.Where("feed_id = ?", feedID) - } - if episodeID > 0 { - q = q.Where("episode_id = ?", episodeID) + q := h.db.WithContext(c.Request().Context()).Model(&database.Bookmark{}). + Where("did = ?", did). + Order("rkey DESC"). + Limit(limit + 1) + if subjectURI != "" { + q = q.Where("subject_uri = ?", subjectURI) } if cursor != "" { q = q.Where("rkey < ?", cursor) } - q = h.excludeBlockedDIDs(q, requestingDID(c), "did") - var rows []database.Bookmark if err := q.Find(&rows).Error; err != nil { return h.internalError(c, "GetBookmarks.find", err) @@ -42,32 +39,20 @@ func (h *Handlers) GetBookmarks(c echo.Context) error { rows = rows[:limit] } - type episodeRef struct { - FeedID int64 `json:"feed_id"` - EpisodeID int64 `json:"episode_id"` - EpisodeGuid string `json:"episode_guid,omitempty"` - PodcastGuid string `json:"podcast_guid,omitempty"` - } - type bookmark struct { - DID string `json:"did"` - Rkey string `json:"rkey"` - Episode episodeRef `json:"episode"` - Timestamp *int `json:"timestamp,omitempty"` - CreatedAt string `json:"created_at"` + DID string `json:"did"` + Rkey string `json:"rkey"` + Subject strongRefJSON `json:"subject"` + Timestamp *int `json:"timestamp,omitempty"` + CreatedAt string `json:"created_at"` } list := make([]bookmark, 0, len(rows)) for _, row := range rows { list = append(list, bookmark{ - DID: row.DID, - Rkey: row.Rkey, - Episode: episodeRef{ - FeedID: row.FeedID, - EpisodeID: row.EpisodeID, - EpisodeGuid: row.EpisodeGuid, - PodcastGuid: row.PodcastGuid, - }, + DID: row.DID, + Rkey: row.Rkey, + Subject: strongRefJSON{URI: row.SubjectURI, CID: row.SubjectCID}, Timestamp: row.TimestampS, CreatedAt: row.CreatedAt, }) diff --git a/appview/handlers/catalog.go b/appview/handlers/catalog.go new file mode 100644 index 0000000..b818b88 --- /dev/null +++ b/appview/handlers/catalog.go @@ -0,0 +1,202 @@ +package handlers + +import ( + "encoding/json" + "errors" + "io/fs" + "net/http" + "path" + "strings" + + "github.com/labstack/echo/v4" + "gorm.io/gorm" + "tangled.org/sparrowtek.com/effem-AppView/appview/catalog" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" + "tangled.org/sparrowtek.com/effem-AppView/lexicons" +) + +// ResolveEpisode returns the canonical strongRef + catalog row for a +// podcast episode. Callers either pass `feedId` + `episodeId` (Podcast +// Index identifiers, the fast path the iOS app uses) or `podcastGuid` + +// `episodeGuid` (the canonical natural key). The first form triggers a PI +// lookup on cache miss; the second only succeeds against an already-known +// episode unless a future resolver can resolve from guids alone. +func (h *Handlers) ResolveEpisode(c echo.Context) error { + ctx := c.Request().Context() + + feedID := parseInt64(c.QueryParam("feedId"), 0) + episodeID := parseInt64(c.QueryParam("episodeId"), 0) + podcastGuid := c.QueryParam("podcastGuid") + episodeGuid := c.QueryParam("episodeGuid") + + cat := h.catalog + if cat == nil { + return writeError(c, http.StatusInternalServerError, "Misconfigured", "catalog service unavailable") + } + + // Fast path: known PI IDs. Resolve the episode metadata via the PI + // proxy, then upsert the catalog row. + if episodeID > 0 { + // Try the existing row first; the PI hit happens only on miss. + if row, err := cat.LookupEpisodeByPI(ctx, feedID, episodeID); err == nil { + return c.JSON(http.StatusOK, episodeCatalogResponse(cat, row)) + } else if !errors.Is(err, gorm.ErrRecordNotFound) { + return h.internalError(c, "ResolveEpisode.lookupByPI", err) + } + + fields, err := h.piResolver.ResolveEpisodeFromPIID(ctx, episodeID) + if err != nil { + return writeError(c, http.StatusNotFound, "NotFound", "episode not found in Podcast Index") + } + row, err := cat.EnsureEpisodeFromRecord(ctx, *fields) + if err != nil { + return h.internalError(c, "ResolveEpisode.upsert", err) + } + return c.JSON(http.StatusOK, episodeCatalogResponse(cat, row)) + } + + if podcastGuid == "" || episodeGuid == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", + "provide either episodeId or (podcastGuid, episodeGuid)") + } + + row, err := cat.ResolveEpisode(ctx, + catalog.EpisodeIdentity{PodcastGuid: podcastGuid, EpisodeGuid: episodeGuid}) + if err != nil { + if errors.Is(err, catalog.ErrUnknownEpisode) { + return writeError(c, http.StatusNotFound, "NotFound", "episode is not in the catalog") + } + return h.internalError(c, "ResolveEpisode", err) + } + return c.JSON(http.StatusOK, episodeCatalogResponse(cat, row)) +} + +// ResolvePodcast returns the canonical strongRef + catalog row for a +// podcast. Same two-path API as ResolveEpisode. +func (h *Handlers) ResolvePodcast(c echo.Context) error { + ctx := c.Request().Context() + + feedID := parseInt64(c.QueryParam("feedId"), 0) + podcastGuid := c.QueryParam("podcastGuid") + + cat := h.catalog + if cat == nil { + return writeError(c, http.StatusInternalServerError, "Misconfigured", "catalog service unavailable") + } + + if feedID > 0 { + if row, err := cat.LookupPodcastByPI(ctx, feedID); err == nil { + return c.JSON(http.StatusOK, podcastCatalogResponse(cat, row)) + } else if !errors.Is(err, gorm.ErrRecordNotFound) { + return h.internalError(c, "ResolvePodcast.lookupByPI", err) + } + + fields, err := h.piResolver.ResolvePodcastFromPIID(ctx, feedID) + if err != nil { + return writeError(c, http.StatusNotFound, "NotFound", "podcast not found in Podcast Index") + } + row, err := cat.EnsurePodcastFromRecord(ctx, *fields) + if err != nil { + return h.internalError(c, "ResolvePodcast.upsert", err) + } + return c.JSON(http.StatusOK, podcastCatalogResponse(cat, row)) + } + + if podcastGuid == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", + "provide either feedId or podcastGuid") + } + + row, err := h.catalog.ResolvePodcast(ctx, + catalog.PodcastIdentity{PodcastGuid: podcastGuid}) + if err != nil { + if errors.Is(err, catalog.ErrUnknownPodcast) { + return writeError(c, http.StatusNotFound, "NotFound", "podcast is not in the catalog") + } + return h.internalError(c, "ResolvePodcast", err) + } + return c.JSON(http.StatusOK, podcastCatalogResponse(cat, row)) +} + +// GetEpisodeRecord returns the canonical episode record for a catalog +// AT-URI. Used by other AppViews to dereference strongRefs and verify CIDs. +func (h *Handlers) GetEpisodeRecord(c echo.Context) error { + uri := c.QueryParam("uri") + if uri == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "uri is required") + } + row, err := h.catalog.LookupEpisodeByURI(c.Request().Context(), uri) + if err != nil { + return writeError(c, http.StatusNotFound, "NotFound", "episode not found") + } + return c.JSON(http.StatusOK, map[string]any{ + "uri": row.ATURI, + "cid": row.CID, + "value": h.catalog.EpisodeRecord(row), + }) +} + +// GetPodcastRecord mirrors GetEpisodeRecord for podcasts. +func (h *Handlers) GetPodcastRecord(c echo.Context) error { + uri := c.QueryParam("uri") + if uri == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "uri is required") + } + row, err := h.catalog.LookupPodcastByURI(c.Request().Context(), uri) + if err != nil { + return writeError(c, http.StatusNotFound, "NotFound", "podcast not found") + } + return c.JSON(http.StatusOK, map[string]any{ + "uri": row.ATURI, + "cid": row.CID, + "value": h.catalog.PodcastRecord(row), + }) +} + +func episodeCatalogResponse(cat *catalog.Service, row *database.EpisodeCatalog) map[string]any { + return map[string]any{ + "subject": map[string]any{"uri": row.ATURI, "cid": row.CID}, + "record": cat.EpisodeRecord(row), + } +} + +func podcastCatalogResponse(cat *catalog.Service, row *database.PodcastCatalog) map[string]any { + return map[string]any{ + "subject": map[string]any{"uri": row.ATURI, "cid": row.CID}, + "record": cat.PodcastRecord(row), + } +} + +// DescribeLexicons returns every Effem-namespace lexicon known to this +// AppView so other implementations can subscribe without out-of-band +// schema sharing. The response is `{lexicons: [{id, schema}]}` keyed on +// NSID. +func (h *Handlers) DescribeLexicons(c echo.Context) error { + out := make([]map[string]any, 0) + err := fs.WalkDir(lexicons.FS, ".", func(p string, d fs.DirEntry, err error) error { + if err != nil { + return err + } + if d.IsDir() || !strings.HasSuffix(p, ".json") { + return nil + } + data, readErr := lexicons.FS.ReadFile(p) + if readErr != nil { + return readErr + } + var lex map[string]any + if jsonErr := json.Unmarshal(data, &lex); jsonErr != nil { + return jsonErr + } + id, _ := lex["id"].(string) + if id == "" { + id = strings.TrimSuffix(path.Base(p), ".json") + } + out = append(out, map[string]any{"id": id, "schema": lex}) + return nil + }) + if err != nil { + return h.internalError(c, "DescribeLexicons", err) + } + return c.JSON(http.StatusOK, map[string]any{"lexicons": out}) +} diff --git a/appview/handlers/comment.go b/appview/handlers/comment.go index 227f98d..0bf41e1 100644 --- a/appview/handlers/comment.go +++ b/appview/handlers/comment.go @@ -1,74 +1,183 @@ package handlers import ( + "context" "encoding/json" + "fmt" "net/http" "tangled.org/sparrowtek.com/effem-AppView/appview/database" "github.com/labstack/echo/v4" + "gorm.io/gorm" ) -type commentEpisodeRef struct { - FeedID int64 `json:"feed_id"` - EpisodeID int64 `json:"episode_id"` - EpisodeGuid string `json:"episode_guid,omitempty"` - PodcastGuid string `json:"podcast_guid,omitempty"` +type strongRefJSON struct { + URI string `json:"uri"` + CID string `json:"cid"` +} + +type commentAuthor struct { + DID string `json:"did"` + Handle string `json:"handle,omitempty"` + DisplayName string `json:"display_name,omitempty"` + Avatar string `json:"avatar,omitempty"` } type commentResponse struct { - DID string `json:"did"` - Rkey string `json:"rkey"` - CID string `json:"cid,omitempty"` - Episode commentEpisodeRef `json:"episode"` - Text string `json:"text"` - Timestamp *int `json:"timestamp,omitempty"` - ReplyRoot string `json:"reply_root,omitempty"` - ReplyRootCID string `json:"reply_root_cid,omitempty"` - ReplyParent string `json:"reply_parent,omitempty"` - ReplyParentCID string `json:"reply_parent_cid,omitempty"` - Facets json.RawMessage `json:"facets,omitempty"` - CreatedAt string `json:"created_at"` -} - -func toCommentResponse(row database.Comment) commentResponse { + DID string `json:"did"` + Rkey string `json:"rkey"` + CID string `json:"cid,omitempty"` + URI string `json:"uri"` + Subject strongRefJSON `json:"subject"` + Text string `json:"text"` + Timestamp *int `json:"timestamp,omitempty"` + Reply *commentReply `json:"reply,omitempty"` + Facets json.RawMessage `json:"facets,omitempty"` + SelfLabels []string `json:"self_labels,omitempty"` + Labels []labelView `json:"labels,omitempty"` + CreatedAt string `json:"created_at"` + Author *commentAuthor `json:"author,omitempty"` + LikeCount int `json:"like_count"` + ViewerLike string `json:"viewer_like_uri,omitempty"` +} + +type commentReply struct { + Root strongRefJSON `json:"root"` + Parent strongRefJSON `json:"parent"` +} + +func toCommentResponse(row database.Comment, author *commentAuthor) commentResponse { var facets json.RawMessage if len(row.Facets) > 0 { facets = row.Facets } - return commentResponse{ + r := commentResponse{ DID: row.DID, Rkey: row.Rkey, CID: row.CID, - Episode: commentEpisodeRef{ - FeedID: row.FeedID, - EpisodeID: row.EpisodeID, - EpisodeGuid: row.EpisodeGuid, - PodcastGuid: row.PodcastGuid, + URI: row.ATURI, + Subject: strongRefJSON{ + URI: row.SubjectURI, + CID: row.SubjectCID, }, - Text: row.Text, - Timestamp: row.TimestampS, - ReplyRoot: row.ReplyRoot, - ReplyRootCID: row.ReplyRootCID, - ReplyParent: row.ReplyParent, - ReplyParentCID: row.ReplyParentCID, - Facets: facets, - CreatedAt: row.CreatedAt, + Text: row.Text, + Timestamp: row.TimestampS, + Facets: facets, + SelfLabels: decodeSelfLabels(row.SelfLabels), + CreatedAt: row.CreatedAt, + Author: author, + } + if row.ReplyRoot != "" && row.ReplyParent != "" { + r.Reply = &commentReply{ + Root: strongRefJSON{URI: row.ReplyRoot, CID: row.ReplyRootCID}, + Parent: strongRefJSON{URI: row.ReplyParent, CID: row.ReplyParentCID}, + } + } + return r +} + +// decodeSelfLabels turns the JSONB array `["sexual", "violence"]` stored on +// comments.self_labels back into a []string. Returns nil (which JSON-omits +// the field) for missing or malformed data. +func decodeSelfLabels(raw []byte) []string { + if len(raw) == 0 { + return nil + } + var out []string + if err := json.Unmarshal(raw, &out); err != nil { + return nil + } + if len(out) == 0 { + return nil } + return out +} + +// hydrateAuthors fetches profile + cached PDS endpoint for every distinct +// author DID in `rows` and returns a `did -> commentAuthor` map. DIDs without +// a profile record are still included with `did` populated so the client can +// fall back to showing the raw identifier instead of rendering an anonymous +// row. +// +// Handle and avatar are read from cached identity data (populated by the +// IdentityResolver on the indexer side and by `xyz.effem.actor.profile` +// firehose events). DIDs we haven't resolved yet get an `author` with only +// `did` and any display name — the client shows a placeholder avatar and +// no @handle until the cache warms up. +func (h *Handlers) hydrateAuthors(ctx context.Context, rows []database.Comment) map[string]*commentAuthor { + if len(rows) == 0 { + return map[string]*commentAuthor{} + } + + seen := make(map[string]struct{}, len(rows)) + dids := make([]string, 0, len(rows)) + for _, row := range rows { + if _, ok := seen[row.DID]; ok { + continue + } + seen[row.DID] = struct{}{} + dids = append(dids, row.DID) + } + + profiles := make([]database.Profile, 0, len(dids)) + if err := h.db.WithContext(ctx). + Where("did IN ?", dids). + Find(&profiles).Error; err != nil { + h.logger.Warn("hydrateAuthors profile lookup failed", "err", err) + } + + pdsEndpoints := make(map[string]string, len(dids)) + var pdsRows []database.DIDPDS + if err := h.db.WithContext(ctx). + Where("did IN ?", dids). + Find(&pdsRows).Error; err != nil { + h.logger.Warn("hydrateAuthors did_pds lookup failed", "err", err) + } + for _, row := range pdsRows { + pdsEndpoints[row.DID] = row.PDSEndpoint + } + + authors := make(map[string]*commentAuthor, len(dids)) + for _, p := range profiles { + authors[p.DID] = &commentAuthor{ + DID: p.DID, + Handle: p.Handle, + DisplayName: p.DisplayName, + Avatar: avatarURL(pdsEndpoints[p.DID], p.DID, p.AvatarCID), + } + } + for _, did := range dids { + if _, ok := authors[did]; !ok { + authors[did] = &commentAuthor{DID: did} + } + } + return authors +} + +// avatarURL builds the canonical getBlob URL for an indexed avatar. Returns +// empty string if any input is missing — the client uses its placeholder +// rendering for empty avatars. +func avatarURL(pdsEndpoint, did, cid string) string { + if pdsEndpoint == "" || did == "" || cid == "" { + return "" + } + return fmt.Sprintf("%s/xrpc/com.atproto.sync.getBlob?did=%s&cid=%s", pdsEndpoint, did, cid) } func (h *Handlers) GetComments(c echo.Context) error { - feedID := parseInt64(c.QueryParam("feedId"), 0) - episodeID := parseInt64(c.QueryParam("episodeId"), 0) - if feedID <= 0 || episodeID <= 0 { - return writeError(c, http.StatusBadRequest, "InvalidRequest", "feedId and episodeId are required") + subjectURI := c.QueryParam("subject") + if subjectURI == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "subject is required") } limit := parseLimit(c.QueryParam("limit"), 50, 100) cursor := c.QueryParam("cursor") + reqDID := requestingDID(c) + ctx := c.Request().Context() - query := h.db.WithContext(c.Request().Context()). - Where("feed_id = ? AND episode_id = ?", feedID, episodeID). + query := h.db.WithContext(ctx). + Where("subject_uri = ?", subjectURI). Where("removed = ?", false). Order("rkey DESC"). Limit(limit + 1) @@ -76,7 +185,8 @@ func (h *Handlers) GetComments(c echo.Context) error { query = query.Where("rkey < ?", cursor) } - query = h.excludeBlockedDIDs(query, requestingDID(c), "did") + query = h.excludeBlockedDIDs(query, reqDID, "did") + query = h.excludeInactiveAccounts(query, "did") var comments []database.Comment if err := query.Find(&comments).Error; err != nil { @@ -89,9 +199,22 @@ func (h *Handlers) GetComments(c echo.Context) error { comments = comments[:limit] } + authors := h.hydrateAuthors(ctx, comments) + likeCounts, viewerLikes := h.hydrateCommentLikes(ctx, comments, reqDID) + commentURIs := commentURIList(comments) + labels := h.hydrateLabels(ctx, commentURIs, reqDID) + list := make([]commentResponse, 0, len(comments)) for _, row := range comments { - list = append(list, toCommentResponse(row)) + rowLabels := labels[row.ATURI] + if hasHideLabel(rowLabels) { + continue + } + resp := toCommentResponse(row, authors[row.DID]) + resp.LikeCount = likeCounts[row.ATURI] + resp.ViewerLike = viewerLikes[row.ATURI] + resp.Labels = rowLabels + list = append(list, resp) } return c.JSON(http.StatusOK, map[string]any{ @@ -100,6 +223,67 @@ func (h *Handlers) GetComments(c echo.Context) error { }) } +// commentURIList extracts the at_uri column from a list of comments. Kept +// as a tiny helper because hydrateLabels and the eventual thread variant +// both need the same projection. +func commentURIList(comments []database.Comment) []string { + uris := make([]string, 0, len(comments)) + for _, c := range comments { + uris = append(uris, c.ATURI) + } + return uris +} + +// hydrateCommentLikes returns two maps keyed on comment AT-URI: +// - likeCounts: total like_count per comment (any user). +// - viewerLikes: the AT-URI of the requesting user's own like row, if any. +// Empty for unauthenticated callers. +func (h *Handlers) hydrateCommentLikes(ctx context.Context, comments []database.Comment, viewerDID string) (map[string]int, map[string]string) { + counts := make(map[string]int, len(comments)) + viewer := make(map[string]string, len(comments)) + if len(comments) == 0 { + return counts, viewer + } + + uris := make([]string, 0, len(comments)) + for _, c := range comments { + uris = append(uris, c.ATURI) + } + + type countRow struct { + SubjectURI string + N int64 + } + var rows []countRow + if err := h.db.WithContext(ctx). + Model(&database.CommentLike{}). + Select("subject_uri, COUNT(*) AS n"). + Where("subject_uri IN ?", uris). + Group("subject_uri"). + Scan(&rows).Error; err != nil { + h.logger.Warn("hydrateCommentLikes count failed", "err", err) + return counts, viewer + } + for _, r := range rows { + counts[r.SubjectURI] = int(r.N) + } + + if viewerDID == "" { + return counts, viewer + } + var viewerRows []database.CommentLike + if err := h.db.WithContext(ctx). + Where("did = ? AND subject_uri IN ?", viewerDID, uris). + Find(&viewerRows).Error; err != nil { + h.logger.Warn("hydrateCommentLikes viewer lookup failed", "err", err) + return counts, viewer + } + for _, row := range viewerRows { + viewer[row.SubjectURI] = row.ATURI + } + return counts, viewer +} + func (h *Handlers) GetCommentThread(c echo.Context) error { uri := c.QueryParam("uri") if uri == "" { @@ -111,43 +295,23 @@ func (h *Handlers) GetCommentThread(c echo.Context) error { rootQuery := h.db.WithContext(ctx).Where("at_uri = ?", uri).Where("removed = ?", false) rootQuery = h.excludeBlockedDIDs(rootQuery, reqDID, "did") + rootQuery = h.excludeInactiveAccounts(rootQuery, "did") var root database.Comment if err := rootQuery.First(&root).Error; err != nil { return writeError(c, http.StatusNotFound, "NotFound", "comment not found") } - limit := parseLimit(c.QueryParam("limit"), 50, 200) - cursor := c.QueryParam("cursor") + depth := parseLimit(c.QueryParam("depth"), 6, 10) - // Order and cursor both by rkey. AT Protocol TIDs are monotonically - // time-sortable, so this gives chronological order without the skip / - // duplicate hazard that a cursor on rkey with ORDER BY created_at would - // create when two replies share a timestamp. - repliesQuery := h.db.WithContext(ctx). - Where("reply_root = ?", root.ATURI). - Where("removed = ?", false). - Order("rkey ASC"). - Limit(limit + 1) - if cursor != "" { - repliesQuery = repliesQuery.Where("rkey > ?", cursor) - } - repliesQuery = h.excludeBlockedDIDs(repliesQuery, reqDID, "did") - - var replies []database.Comment - if err := repliesQuery.Find(&replies).Error; err != nil { + replies, err := h.fetchThreadReplies(ctx, root.ATURI, reqDID, depth) + if err != nil { return h.internalError(c, "GetCommentThread.replies", err) } - nextCursor := "" - if len(replies) > limit { - nextCursor = replies[limit-1].Rkey - replies = replies[:limit] - } - // reply_count is the raw thread size — it does not account for the // caller's block filter, so the number can exceed the replies they - // actually see. Clients should treat it as an upper bound. + // actually see. visible_reply_count below is the post-filter count. var replyCount int64 if err := h.db.WithContext(ctx). Model(&database.Comment{}). @@ -157,20 +321,249 @@ func (h *Handlers) GetCommentThread(c echo.Context) error { return h.internalError(c, "GetCommentThread.count", err) } - replyNodes := make([]map[string]any, 0, len(replies)) - for _, row := range replies { - replyNodes = append(replyNodes, map[string]any{ - "comment": toCommentResponse(row), - "replies": []any{}, + gate, err := h.loadThreadSettings(ctx, root.ATURI) + if err != nil { + return h.internalError(c, "GetCommentThread.gate", err) + } + replies = applyThreadgate(root, gate, replies) + + allComments := append([]database.Comment{root}, replies...) + authors := h.hydrateAuthors(ctx, allComments) + labels := h.hydrateLabels(ctx, commentURIList(allComments), reqDID) + if hasHideLabel(labels[root.ATURI]) { + // !hide on the root removes the thread entirely from the public + // surface; the row remains for the author and admin audit. + return writeError(c, http.StatusNotFound, "NotFound", "comment not found") + } + replies = filterHiddenReplies(replies, labels) + + rootNode := buildThreadTree(root, replies, authors, labels) + visibleCount := countTreeReplies(rootNode) + + payload := map[string]any{ + "thread": rootNode, + "reply_count": replyCount, + "visible_reply_count": visibleCount, + } + if gate != nil { + payload["thread_settings"] = gate + } + + return c.JSON(http.StatusOK, payload) +} + +// applyThreadgate filters replies down to those allowed by the root +// comment's threadgate. Returns `replies` unchanged when gate is nil +// (no record) or gate.AllowSet is false (record present but `allow` was +// omitted). Otherwise builds an allowed-DID set from each rule and drops +// any reply whose author isn't in the set. The root comment's own author +// is always allowed. +func applyThreadgate(root database.Comment, gate *threadSettingsView, replies []database.Comment) []database.Comment { + if gate == nil || !gate.AllowSet { + return replies + } + + allowed := map[string]struct{}{root.DID: {}} + for _, rule := range gate.Rules { + switch asStringFromAny(rule["$type"]) { + case "xyz.effem.feed.threadgate#mentionRule": + for did := range mentionedDIDs(root.Facets) { + allowed[did] = struct{}{} + } + } + } + + out := make([]database.Comment, 0, len(replies)) + for _, r := range replies { + if _, ok := allowed[r.DID]; ok { + out = append(out, r) + } + } + return out +} + +// asStringFromAny is a local string-from-any without importing the indexer's +// asString. Centralised here so handlers stay self-contained. +func asStringFromAny(v any) string { + s, _ := v.(string) + return s +} + +// mentionedDIDs scans the comment's facets blob for DIDs mentioned via +// app.bsky.richtext.facet#mention. The wire shape: +// +// [{"index":..., "features":[{"$type":"app.bsky.richtext.facet#mention","did":"..."}]}] +// +// Returns an empty map when facets are missing or unparseable. Centralised +// here rather than borrowed from the indexer to keep handlers free of +// indexer internals. +func mentionedDIDs(raw []byte) map[string]struct{} { + out := map[string]struct{}{} + if len(raw) == 0 { + return out + } + var facets []map[string]any + if err := json.Unmarshal(raw, &facets); err != nil { + return out + } + for _, facet := range facets { + features, ok := facet["features"].([]any) + if !ok { + continue + } + for _, f := range features { + feature, ok := f.(map[string]any) + if !ok { + continue + } + if asStringFromAny(feature["$type"]) != "app.bsky.richtext.facet#mention" { + continue + } + if did := asStringFromAny(feature["did"]); did != "" { + out[did] = struct{}{} + } + } + } + return out +} + +// filterHiddenReplies drops replies whose label set contains !hide. +func filterHiddenReplies(replies []database.Comment, labels map[string][]labelView) []database.Comment { + out := make([]database.Comment, 0, len(replies)) + for _, r := range replies { + if hasHideLabel(labels[r.ATURI]) { + continue + } + out = append(out, r) + } + return out +} + +// fetchThreadReplies pulls every reply under rootURI (any depth) in a single +// query using a recursive CTE. Sorting by rkey is the AT Proto–native way to +// order by record creation time without a separate created_at index lookup. +// +// Inactive accounts (takedown/deactivation) are filtered inside the CTE so +// their entire subtree drops together. Block filtering happens in Go after +// the fetch — orphaned subtrees fall off naturally during buildThreadTree +// because the walker only descends from parents it found. +func (h *Handlers) fetchThreadReplies(ctx context.Context, rootURI, reqDID string, maxDepth int) ([]database.Comment, error) { + const cte = ` +WITH RECURSIVE thread AS ( + SELECT c.*, 1 AS depth + FROM comments c + WHERE c.reply_parent = ? + AND c.removed = FALSE + AND c.did NOT IN (SELECT did FROM account_status WHERE active = FALSE) + UNION ALL + SELECT c.*, t.depth + 1 + FROM comments c + JOIN thread t ON c.reply_parent = t.at_uri + WHERE c.removed = FALSE + AND c.did NOT IN (SELECT did FROM account_status WHERE active = FALSE) + AND t.depth < ? +) +SELECT * FROM thread +` + + q := h.db.WithContext(ctx).Raw(cte, rootURI, maxDepth) + + var rows []database.Comment + if err := q.Scan(&rows).Error; err != nil { + return nil, err + } + + if reqDID == "" { + return rows, nil + } + + return filterBlockedReplies(h.db.WithContext(ctx), rows, reqDID), nil +} + +// filterBlockedReplies removes rows authored by users on either side of a +// block relationship with reqDID. Orphaned subtrees (children of a dropped +// row) fall off naturally during buildThreadTree because the tree walker +// only descends from parents it found. +func filterBlockedReplies(db *gorm.DB, rows []database.Comment, reqDID string) []database.Comment { + if len(rows) == 0 || reqDID == "" { + return rows + } + + var blockedByMe []string + _ = db.Model(&database.Block{}).Where("did = ?", reqDID).Pluck("subject_did", &blockedByMe) + var blockingMe []string + _ = db.Model(&database.Block{}).Where("subject_did = ?", reqDID).Pluck("did", &blockingMe) + + blocked := make(map[string]struct{}, len(blockedByMe)+len(blockingMe)) + for _, d := range blockedByMe { + blocked[d] = struct{}{} + } + for _, d := range blockingMe { + blocked[d] = struct{}{} + } + + if len(blocked) == 0 { + return rows + } + + out := make([]database.Comment, 0, len(rows)) + for _, row := range rows { + if _, hit := blocked[row.DID]; hit { + continue + } + out = append(out, row) + } + return out +} + +type threadNode struct { + Comment commentResponse `json:"comment"` + Replies []*threadNode `json:"replies"` +} + +func buildThreadTree(root database.Comment, replies []database.Comment, authors map[string]*commentAuthor, labels map[string][]labelView) *threadNode { + byParent := make(map[string][]database.Comment, len(replies)) + for _, reply := range replies { + parent := reply.ReplyParent + if parent == "" { + parent = root.ATURI + } + byParent[parent] = append(byParent[parent], reply) + } + + rootResp := toCommentResponse(root, authors[root.DID]) + rootResp.Labels = labels[root.ATURI] + rootNode := &threadNode{ + Comment: rootResp, + Replies: attachChildren(root.ATURI, byParent, authors, labels), + } + return rootNode +} + +func attachChildren(parentURI string, byParent map[string][]database.Comment, authors map[string]*commentAuthor, labels map[string][]labelView) []*threadNode { + children := byParent[parentURI] + if len(children) == 0 { + return []*threadNode{} + } + out := make([]*threadNode, 0, len(children)) + for _, child := range children { + childResp := toCommentResponse(child, authors[child.DID]) + childResp.Labels = labels[child.ATURI] + out = append(out, &threadNode{ + Comment: childResp, + Replies: attachChildren(child.ATURI, byParent, authors, labels), }) } + return out +} - return c.JSON(http.StatusOK, map[string]any{ - "thread": map[string]any{ - "comment": toCommentResponse(root), - "replies": replyNodes, - }, - "reply_count": replyCount, - "cursor": nextCursor, - }) +func countTreeReplies(node *threadNode) int { + if node == nil { + return 0 + } + count := 0 + for _, child := range node.Replies { + count += 1 + countTreeReplies(child) + } + return count } diff --git a/appview/handlers/devices.go b/appview/handlers/devices.go new file mode 100644 index 0000000..b9852f3 --- /dev/null +++ b/appview/handlers/devices.go @@ -0,0 +1,93 @@ +package handlers + +import ( + "net/http" + "strings" + "time" + + "github.com/labstack/echo/v4" + "gorm.io/gorm/clause" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" +) + +type registerDeviceRequest struct { + Token string `json:"token"` + Environment string `json:"environment"` + BundleID string `json:"bundle_id,omitempty"` +} + +// RegisterDevice stores an APNs device token for the authenticated user. +// Idempotent on (did, token): re-registration refreshes `last_seen_at` and +// clears any prior `invalid_at` stamp so a device that was soft-deleted +// can be revived after the user reinstalls. +func (h *Handlers) RegisterDevice(c echo.Context) error { + did := requestingDID(c) + if did == "" { + return writeError(c, http.StatusUnauthorized, "AuthRequired", "authenticated DID is required") + } + var req registerDeviceRequest + if err := c.Bind(&req); err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "invalid JSON body") + } + req.Token = strings.TrimSpace(req.Token) + if req.Token == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "token is required") + } + if req.Environment != "sandbox" && req.Environment != "production" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "environment must be sandbox or production") + } + + now := time.Now().UTC() + row := database.DeviceToken{ + DID: did, + Token: req.Token, + Environment: req.Environment, + BundleID: req.BundleID, + CreatedAt: now, + LastSeenAt: now, + } + err := h.db.WithContext(c.Request().Context()).Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "did"}, {Name: "token"}}, + DoUpdates: clause.Assignments(map[string]any{ + "environment": req.Environment, + "bundle_id": req.BundleID, + "last_seen_at": now, + "invalid_at": nil, + }), + }).Create(&row).Error + if err != nil { + return h.internalError(c, "RegisterDevice", err) + } + return c.JSON(http.StatusOK, map[string]any{"ok": true}) +} + +type unregisterDeviceRequest struct { + Token string `json:"token"` +} + +// UnregisterDevice soft-deletes the (did, token) row. Soft rather than hard +// so we can still diagnose missing notifications later — the dispatcher +// just stops sending to it. +func (h *Handlers) UnregisterDevice(c echo.Context) error { + did := requestingDID(c) + if did == "" { + return writeError(c, http.StatusUnauthorized, "AuthRequired", "authenticated DID is required") + } + var req unregisterDeviceRequest + if err := c.Bind(&req); err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "invalid JSON body") + } + req.Token = strings.TrimSpace(req.Token) + if req.Token == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "token is required") + } + + now := time.Now().UTC() + if err := h.db.WithContext(c.Request().Context()). + Model(&database.DeviceToken{}). + Where("did = ? AND token = ?", did, req.Token). + Update("invalid_at", &now).Error; err != nil { + return h.internalError(c, "UnregisterDevice", err) + } + return c.JSON(http.StatusOK, map[string]any{"ok": true}) +} diff --git a/appview/handlers/episode_state.go b/appview/handlers/episode_state.go index 9144dae..580ebe9 100644 --- a/appview/handlers/episode_state.go +++ b/appview/handlers/episode_state.go @@ -3,8 +3,8 @@ package handlers import ( "net/http" - "tangled.org/sparrowtek.com/effem-AppView/appview/database" "github.com/labstack/echo/v4" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" ) func (h *Handlers) GetEpisodeStates(c echo.Context) error { @@ -13,14 +13,18 @@ func (h *Handlers) GetEpisodeStates(c echo.Context) error { return writeError(c, http.StatusBadRequest, "InvalidRequest", "did is required") } - feedID := parseInt64(c.QueryParam("feedId"), 0) + subjectURI := c.QueryParam("subject") limit := parseLimit(c.QueryParam("limit"), 50, 100) cursor := c.QueryParam("cursor") - q := h.db.WithContext(c.Request().Context()).Model(&database.EpisodeState{}).Where("did = ?", did).Order("rkey DESC").Limit(limit + 1) + q := h.db.WithContext(c.Request().Context()). + Model(&database.EpisodeState{}). + Where("did = ?", did). + Order("rkey DESC"). + Limit(limit + 1) - if feedID > 0 { - q = q.Where("feed_id = ?", feedID) + if subjectURI != "" { + q = q.Where("subject_uri = ?", subjectURI) } if c.QueryParam("saved") == "true" { q = q.Where("saved = ?", true) diff --git a/appview/handlers/episodes.go b/appview/handlers/episodes.go index b48513a..274c9f3 100644 --- a/appview/handlers/episodes.go +++ b/appview/handlers/episodes.go @@ -79,7 +79,12 @@ func (h *Handlers) GetEpisode(c echo.Context) error { } feedID2 := parseInt64(c.QueryParam("feedId"), 0) - payload["social"] = h.episodeSocialCounts(c.Request().Context(), feedID2, episodeID) + social := h.episodeSocialCounts(c.Request().Context(), feedID2, episodeID) + // Nest social inside the `episode` object so clients decode it on the + // same model GetEpisodes uses for each item. + if episode, ok := payload["episode"].(map[string]any); ok { + episode["social"] = social + } setPublicCache(c, 120) return c.JSON(http.StatusOK, payload) diff --git a/appview/handlers/handlers.go b/appview/handlers/handlers.go index c09feb9..e0c46a2 100644 --- a/appview/handlers/handlers.go +++ b/appview/handlers/handlers.go @@ -10,19 +10,45 @@ import ( "github.com/labstack/echo/v4" "gorm.io/gorm" + "tangled.org/sparrowtek.com/effem-AppView/appview/catalog" "tangled.org/sparrowtek.com/effem-AppView/appview/database" "tangled.org/sparrowtek.com/effem-AppView/appview/httpmw" + "tangled.org/sparrowtek.com/effem-AppView/appview/indexer" "tangled.org/sparrowtek.com/effem-AppView/appview/podcastindex" ) type Handlers struct { - db *gorm.DB - pi *podcastindex.CachedClient - logger *slog.Logger + db *gorm.DB + pi *podcastindex.CachedClient + logger *slog.Logger + indexer *indexer.Indexer + catalog *catalog.Service + piResolver *catalog.PIResolver + // labelerDID is the source DID stamped on labels emitted by the + // effem-internal admin label endpoints, and the always-trusted source + // for label reads. Resolved at startup from Config.LabelerDID (with a + // "did:web:labeler.effem.app" fallback applied in NewServer). + labelerDID string } -func New(db *gorm.DB, pi *podcastindex.CachedClient, logger *slog.Logger) *Handlers { - return &Handlers{db: db, pi: pi, logger: logger} +func New( + db *gorm.DB, + pi *podcastindex.CachedClient, + indexer *indexer.Indexer, + catalogSvc *catalog.Service, + piResolver *catalog.PIResolver, + labelerDID string, + logger *slog.Logger, +) *Handlers { + return &Handlers{ + db: db, + pi: pi, + logger: logger, + indexer: indexer, + catalog: catalogSvc, + piResolver: piResolver, + labelerDID: labelerDID, + } } func parseInt64(raw string, fallback int64) int64 { @@ -137,6 +163,21 @@ func (h *Handlers) excludeBlockedDIDs(q *gorm.DB, requesterDID, didColumn string return q } +// excludeInactiveAccounts drops rows authored by DIDs whose `account_status` +// row marks them inactive (takedown, deactivation, suspension). DIDs not +// present in account_status are implicitly active — we never wrote a row +// because the relay never said anything about them. +// +// didColumn is validated against allowedBlockColumns for the same reason as +// excludeBlockedDIDs: defence in depth against an upstream caller passing +// user input. +func (h *Handlers) excludeInactiveAccounts(q *gorm.DB, didColumn string) *gorm.DB { + if _, ok := allowedBlockColumns[didColumn]; !ok { + panic("excludeInactiveAccounts: disallowed didColumn " + didColumn) + } + return q.Where(didColumn+" NOT IN (SELECT did FROM account_status WHERE active = FALSE)") +} + // writeAudit records an admin action in the admin_audit table. Details is // marshaled to JSON; a nil details argument stores SQL NULL. Audit write // failures are logged but do not fail the surrounding admin call — the diff --git a/appview/handlers/inbox.go b/appview/handlers/inbox.go index e01e938..e5faa1a 100644 --- a/appview/handlers/inbox.go +++ b/appview/handlers/inbox.go @@ -23,6 +23,8 @@ func (h *Handlers) GetInbox(c echo.Context) error { return writeError(c, http.StatusBadRequest, "InvalidRequest", "did is required") } + ctx := c.Request().Context() + limit := parseLimit(c.QueryParam("limit"), 50, 100) offset := 0 if cursor := c.QueryParam("cursor"); cursor != "" { @@ -33,7 +35,7 @@ func (h *Handlers) GetInbox(c echo.Context) error { } var subs []database.Subscription - if err := h.db.WithContext(c.Request().Context()). + if err := h.db.WithContext(ctx). Where("did = ?", did). Order("rkey DESC"). Limit(inboxSubscriptionCap). @@ -45,6 +47,28 @@ func (h *Handlers) GetInbox(c echo.Context) error { return c.JSON(http.StatusOK, map[string]any{"items": []any{}, "cursor": ""}) } + // Resolve each subscription's subject AT-URI back to a Podcast Index + // feed ID so we can pull new episodes from the PI proxy. Subjects + // without a matching catalog row (or without a PI cross-reference) + // are skipped — those podcasts predate PI indexing or were registered + // from another source. + subjectURIs := make([]string, 0, len(subs)) + for _, sub := range subs { + subjectURIs = append(subjectURIs, sub.SubjectURI) + } + var podcastRows []database.PodcastCatalog + if err := h.db.WithContext(ctx). + Where("at_uri IN ?", subjectURIs). + Find(&podcastRows).Error; err != nil { + return h.internalError(c, "GetInbox.catalog", err) + } + feedIDByURI := make(map[string]int64, len(podcastRows)) + for _, row := range podcastRows { + if row.PodcastIndexFeedID != nil && *row.PodcastIndexFeedID > 0 { + feedIDByURI[row.ATURI] = *row.PodcastIndexFeedID + } + } + type seenKey struct { FeedID int64 EpisodeID int64 @@ -53,9 +77,13 @@ func (h *Handlers) GetInbox(c echo.Context) error { items := make([]map[string]any, 0) for _, sub := range subs { - raw, err := h.pi.GetEpisodesByFeedID(sub.FeedID, 5) + feedID, ok := feedIDByURI[sub.SubjectURI] + if !ok { + continue + } + raw, err := h.pi.GetEpisodesByFeedID(feedID, 5) if err != nil { - h.logger.Warn("inbox episodes fetch failed", "feedId", sub.FeedID, "err", err) + h.logger.Warn("inbox episodes fetch failed", "feedId", feedID, "err", err) continue } payload, err := decodePayload(raw) @@ -76,14 +104,14 @@ func (h *Handlers) GetInbox(c echo.Context) error { if episodeID <= 0 { continue } - key := seenKey{FeedID: sub.FeedID, EpisodeID: episodeID} + key := seenKey{FeedID: feedID, EpisodeID: episodeID} if _, found := seen[key]; found { continue } seen[key] = struct{}{} if feedIDFromItem(item) <= 0 { - item["feedId"] = sub.FeedID + item["feedId"] = feedID } items = append(items, item) diff --git a/appview/handlers/labels.go b/appview/handlers/labels.go new file mode 100644 index 0000000..58d6607 --- /dev/null +++ b/appview/handlers/labels.go @@ -0,0 +1,463 @@ +package handlers + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "net/url" + "strconv" + "strings" + "time" + + "github.com/bluesky-social/indigo/atproto/syntax" + "github.com/labstack/echo/v4" + "gorm.io/gorm" + "gorm.io/gorm/clause" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" +) + +const auditActionApplyLabel = "apply_label" +const auditActionRemoveLabel = "remove_label" +const auditActionImportLabels = "import_labels" + +type labelView struct { + ID uint `json:"id"` + Src string `json:"src"` + SubjectURI string `json:"subject_uri"` + Val string `json:"val"` + Neg bool `json:"neg"` + CreatedAt string `json:"created_at"` +} + +func toLabelView(row database.Label) labelView { + return labelView{ + ID: row.ID, + Src: row.SrcDID, + SubjectURI: row.SubjectURI, + Val: row.Val, + Neg: row.Neg, + CreatedAt: row.CreatedAt, + } +} + +type applyLabelRequest struct { + SubjectURI string `json:"subject_uri"` + Val string `json:"val"` + Neg bool `json:"neg"` +} + +// ApplyLabel stamps a moderation label onto a subject AT-URI using the +// AppView's own labeler DID as the source. Idempotent on (src, subject, val): +// re-applying the same label flips `neg` or refreshes the timestamp without +// producing a duplicate row. +func (h *Handlers) ApplyLabel(c echo.Context) error { + var req applyLabelRequest + if err := c.Bind(&req); err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "invalid JSON body") + } + if req.SubjectURI == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "subject_uri is required") + } + if _, err := syntax.ParseATURI(req.SubjectURI); err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "subject_uri is not a valid at-uri") + } + val := strings.TrimSpace(req.Val) + if val == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "val is required") + } + if len(val) > 128 { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "val exceeds 128 characters") + } + + now := time.Now().UTC().Format(time.RFC3339Nano) + row := database.Label{ + SrcDID: h.labelerDID, + SubjectURI: req.SubjectURI, + Val: val, + Neg: req.Neg, + CreatedAt: now, + } + + ctx := c.Request().Context() + // UPSERT on (src_did, subject_uri, val): a re-apply updates `neg` and + // `created_at` in place. The unique index makes this race-safe. + res := h.db.WithContext(ctx). + Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "src_did"}, {Name: "subject_uri"}, {Name: "val"}}, + DoUpdates: clause.AssignmentColumns([]string{"neg", "created_at"}), + }). + Create(&row) + if res.Error != nil { + return h.internalError(c, "ApplyLabel.upsert", res.Error) + } + + adminDID := requestingDID(c) + h.writeAudit(ctx, adminDID, auditActionApplyLabel, req.SubjectURI, map[string]any{ + "val": val, + "neg": req.Neg, + }) + + return c.JSON(http.StatusOK, map[string]any{"id": row.ID}) +} + +type removeLabelRequest struct { + SubjectURI string `json:"subject_uri"` + Val string `json:"val"` +} + +// RemoveLabel hard-deletes a label row owned by effem's own labeler. Use +// ApplyLabel with neg=true instead when you want an audit-friendly +// negation; RemoveLabel only exists to undo mistakes. +func (h *Handlers) RemoveLabel(c echo.Context) error { + var req removeLabelRequest + if err := c.Bind(&req); err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "invalid JSON body") + } + if req.SubjectURI == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "subject_uri is required") + } + val := strings.TrimSpace(req.Val) + if val == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "val is required") + } + + ctx := c.Request().Context() + res := h.db.WithContext(ctx). + Where("src_did = ? AND subject_uri = ? AND val = ?", h.labelerDID, req.SubjectURI, val). + Delete(&database.Label{}) + if res.Error != nil { + return h.internalError(c, "RemoveLabel.delete", res.Error) + } + + adminDID := requestingDID(c) + h.writeAudit(ctx, adminDID, auditActionRemoveLabel, req.SubjectURI, map[string]any{ + "val": val, + "deleted": res.RowsAffected, + }) + + return c.JSON(http.StatusOK, map[string]any{"deleted": res.RowsAffected}) +} + +// ListLabels paginates the label store. Newest-first by id. +func (h *Handlers) ListLabels(c echo.Context) error { + src := strings.TrimSpace(c.QueryParam("src")) + subject := strings.TrimSpace(c.QueryParam("subject")) + val := strings.TrimSpace(c.QueryParam("val")) + limit := parseLimit(c.QueryParam("limit"), 100, 500) + cursorRaw := strings.TrimSpace(c.QueryParam("cursor")) + + q := h.db.WithContext(c.Request().Context()). + Model(&database.Label{}). + Order("id DESC"). + Limit(limit + 1) + if src != "" { + q = q.Where("src_did = ?", src) + } + if subject != "" { + q = q.Where("subject_uri = ?", subject) + } + if val != "" { + q = q.Where("val = ?", val) + } + if cursorRaw != "" { + cur, err := strconv.ParseUint(cursorRaw, 10, 64) + if err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "cursor must be a positive integer") + } + q = q.Where("id < ?", cur) + } + + var rows []database.Label + if err := q.Find(&rows).Error; err != nil { + return h.internalError(c, "ListLabels.find", err) + } + + nextCursor := "" + if len(rows) > limit { + nextCursor = strconv.FormatUint(uint64(rows[limit-1].ID), 10) + rows = rows[:limit] + } + + items := make([]labelView, 0, len(rows)) + for _, r := range rows { + items = append(items, toLabelView(r)) + } + + return c.JSON(http.StatusOK, map[string]any{ + "labels": items, + "cursor": nextCursor, + }) +} + +type importLabelsRequest struct { + LabelerDID string `json:"labeler_did"` + Endpoint string `json:"endpoint"` + URIPatterns []string `json:"uri_patterns"` + Limit int `json:"limit"` +} + +// importedLabelEntry mirrors the wire shape returned by +// com.atproto.label.queryLabels. +type importedLabelEntry struct { + Src string `json:"src"` + URI string `json:"uri"` + Val string `json:"val"` + Neg bool `json:"neg"` + CTS string `json:"cts"` +} + +// importLabelClient is the HTTP client used by ImportLabels. Set to a +// non-nil function in tests to inject a deterministic upstream response. +// In production it is left nil so the handler builds a default client per +// call with a tight timeout. +var importLabelClient func(ctx context.Context, url string) (*http.Response, error) + +// ImportLabels pulls labels from an external labeler and stores them in +// the local labels table with src_did = labeler_did. Real-time +// subscription to a labeler's #subscribeLabels stream is future work; +// this gives operators a manual import that's enough to verify the +// read-path filter end-to-end against a real external labeler. +func (h *Handlers) ImportLabels(c echo.Context) error { + var req importLabelsRequest + if err := c.Bind(&req); err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "invalid JSON body") + } + if req.LabelerDID == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "labeler_did is required") + } + if _, err := syntax.ParseDID(req.LabelerDID); err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "labeler_did is not a valid DID") + } + if req.Endpoint == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "endpoint is required") + } + parsedEndpoint, err := url.Parse(req.Endpoint) + if err != nil || parsedEndpoint.Scheme == "" || parsedEndpoint.Host == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "endpoint is not a valid URL") + } + if parsedEndpoint.Scheme != "http" && parsedEndpoint.Scheme != "https" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "endpoint scheme must be http or https") + } + + patterns := req.URIPatterns + if len(patterns) == 0 { + patterns = []string{"at://*"} + } + limit := req.Limit + if limit <= 0 || limit > 250 { + limit = 250 + } + + // Build the upstream URL: /xrpc/com.atproto.label.queryLabels + // with one `uriPatterns=` per pattern and a `sources=` so + // the upstream filters by its own DID. + upstream := *parsedEndpoint + upstream.Path = strings.TrimRight(upstream.Path, "/") + "/xrpc/com.atproto.label.queryLabels" + qry := upstream.Query() + qry.Set("limit", strconv.Itoa(limit)) + qry.Set("sources", req.LabelerDID) + for _, p := range patterns { + qry.Add("uriPatterns", p) + } + upstream.RawQuery = qry.Encode() + + ctx := c.Request().Context() + resp, err := fetchUpstreamLabels(ctx, upstream.String()) + if err != nil { + h.logger.Warn("ImportLabels upstream fetch failed", "err", err, "endpoint", upstream.String()) + return writeError(c, http.StatusBadGateway, "UpstreamError", "labeler endpoint did not respond") + } + defer resp.Body.Close() + + if resp.StatusCode >= 400 { + body, _ := io.ReadAll(io.LimitReader(resp.Body, 2048)) + h.logger.Warn("ImportLabels upstream returned error", + "status", resp.StatusCode, + "body", string(body), + ) + return writeError(c, http.StatusBadGateway, "UpstreamError", + fmt.Sprintf("labeler endpoint returned %d", resp.StatusCode)) + } + + var upstreamBody struct { + Cursor string `json:"cursor"` + Labels []importedLabelEntry `json:"labels"` + } + if err := json.NewDecoder(resp.Body).Decode(&upstreamBody); err != nil { + return writeError(c, http.StatusBadGateway, "UpstreamError", "labeler response was not valid JSON") + } + + imported, skipped := 0, 0 + for _, entry := range upstreamBody.Labels { + // Defensive: enforce the requested source so a misbehaving labeler + // can't import labels claiming a different src into our store. + if entry.Src != "" && entry.Src != req.LabelerDID { + skipped++ + continue + } + if entry.URI == "" || entry.Val == "" { + skipped++ + continue + } + if _, err := syntax.ParseATURI(entry.URI); err != nil { + skipped++ + continue + } + row := database.Label{ + SrcDID: req.LabelerDID, + SubjectURI: entry.URI, + Val: entry.Val, + Neg: entry.Neg, + CreatedAt: entry.CTS, + } + if row.CreatedAt == "" { + row.CreatedAt = time.Now().UTC().Format(time.RFC3339Nano) + } + res := h.db.WithContext(ctx). + Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "src_did"}, {Name: "subject_uri"}, {Name: "val"}}, + DoUpdates: clause.AssignmentColumns([]string{"neg", "created_at"}), + }). + Create(&row) + if res.Error != nil { + h.logger.Warn("ImportLabels upsert failed", + "err", res.Error, + "src", req.LabelerDID, + "uri", entry.URI, + "val", entry.Val, + ) + skipped++ + continue + } + imported++ + } + + adminDID := requestingDID(c) + h.writeAudit(ctx, adminDID, auditActionImportLabels, req.LabelerDID, map[string]any{ + "endpoint": req.Endpoint, + "imported": imported, + "skipped": skipped, + }) + + return c.JSON(http.StatusOK, map[string]any{ + "imported": imported, + "skipped": skipped, + }) +} + +// fetchUpstreamLabels makes the HTTP GET against a labeler's queryLabels +// endpoint. Centralised so tests can swap in a fake transport. +func fetchUpstreamLabels(ctx context.Context, url string) (*http.Response, error) { + if importLabelClient != nil { + return importLabelClient(ctx, url) + } + req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil) + if err != nil { + return nil, err + } + client := &http.Client{Timeout: 15 * time.Second} + return client.Do(req) +} + +// hydrateLabels loads moderation labels for the supplied set of subject +// AT-URIs and returns a map keyed by subject URI. The result includes only +// labels that the viewer should see: effem's own labels are always +// included, and labels emitted by `src_did`s the viewer has subscribed to +// (per `profiles.labeler_subscriptions`) are merged in. +// +// Public reads (empty viewerDID) get effem's own labels only. +func (h *Handlers) hydrateLabels(ctx context.Context, subjectURIs []string, viewerDID string) map[string][]labelView { + out := make(map[string][]labelView, len(subjectURIs)) + if len(subjectURIs) == 0 { + return out + } + + trusted := []string{h.labelerDID} + if viewerDID != "" { + var profile database.Profile + err := h.db.WithContext(ctx).Where("did = ?", viewerDID).First(&profile).Error + if err == nil && len(profile.LabelerSubscriptions) > 0 { + var dids []string + if err := json.Unmarshal(profile.LabelerSubscriptions, &dids); err == nil { + for _, d := range dids { + if d != "" && d != h.labelerDID { + trusted = append(trusted, d) + } + } + } + } + } + + var rows []database.Label + if err := h.db.WithContext(ctx). + Where("subject_uri IN ?", subjectURIs). + Where("src_did IN ?", trusted). + Where("neg = ?", false). + Find(&rows).Error; err != nil { + h.logger.Warn("hydrateLabels failed", "err", err) + return out + } + + for _, row := range rows { + out[row.SubjectURI] = append(out[row.SubjectURI], toLabelView(row)) + } + return out +} + +// hasHideLabel reports whether the supplied label set contains the !hide +// directive. !hide-labeled subjects are dropped from public read responses +// entirely; the row remains in the database for audit and for the author's +// own queries. +func hasHideLabel(labels []labelView) bool { + for _, l := range labels { + if l.Val == "!hide" { + return true + } + } + return false +} + +// loadThreadSettings reads the threadgate associated with a root comment. +// Returns (nil, nil) when no threadgate has been published. Returns +// (gate, nil) where gate.RawAllow is JSON-decoded for inspection. +// +// gorm.ErrRecordNotFound is the explicit "no gate" case; any other DB +// error is bubbled up. +func (h *Handlers) loadThreadSettings(ctx context.Context, rootURI string) (*threadSettingsView, error) { + var row database.ThreadSettings + err := h.db.WithContext(ctx).Where("root_uri = ?", rootURI).First(&row).Error + if err == gorm.ErrRecordNotFound { + return nil, nil + } + if err != nil { + return nil, err + } + + view := &threadSettingsView{ + Rkey: row.Rkey, + AllowSet: false, + } + if len(row.Allow) > 0 { + var rules []map[string]any + if err := json.Unmarshal(row.Allow, &rules); err == nil { + view.AllowSet = true + view.Rules = rules + } + } + return view, nil +} + +// threadSettingsView is the wire shape for the thread_settings field on +// xyz.effem.feed.getCommentThread responses. The iOS / web client reads: +// +// - missing field → no threadgate, replies open to anyone +// - present, AllowSet=false → threadgate exists with no `allow`, open +// - present, AllowSet=true, Rules=[] → locked (only root author may reply) +// - present, AllowSet=true, Rules=[...] → restricted, see Rules +type threadSettingsView struct { + Rkey string `json:"rkey,omitempty"` + AllowSet bool `json:"allow_set"` + Rules []map[string]any `json:"allow,omitempty"` +} diff --git a/appview/handlers/notifications.go b/appview/handlers/notifications.go new file mode 100644 index 0000000..b3cdbf4 --- /dev/null +++ b/appview/handlers/notifications.go @@ -0,0 +1,284 @@ +package handlers + +import ( + "context" + "errors" + "net/http" + "strconv" + "time" + + "github.com/labstack/echo/v4" + "gorm.io/gorm" + "gorm.io/gorm/clause" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" +) + +// Valid notification reason values. Mirrors indexer.NotificationReason* +// but the handlers package can't import indexer cleanly (it would create +// a cycle), so the string set is duplicated here as a small const. +var allowedNotificationReasons = map[string]struct{}{ + "reply": {}, + "mention": {}, + "like": {}, +} + +type notificationItem struct { + ID uint `json:"id"` + Reason string `json:"reason"` + ActorDID string `json:"actor_did"` + SubjectURI string `json:"subject_uri"` + SourceURI string `json:"source_uri"` + CreatedAt string `json:"created_at"` + IsRead bool `json:"is_read"` + Actor *commentAuthor `json:"actor,omitempty"` +} + +// ListNotifications returns the authenticated user's notifications, +// newest-first. Cursor is the id of the oldest row from the previous page. +func (h *Handlers) ListNotifications(c echo.Context) error { + recipient := requestingDID(c) + if recipient == "" { + return writeError(c, http.StatusUnauthorized, "AuthRequired", "authenticated DID is required") + } + + limit := parseLimit(c.QueryParam("limit"), 50, 100) + cursorRaw := c.QueryParam("cursor") + reasons := c.QueryParams()["reasons"] + + ctx := c.Request().Context() + q := h.db.WithContext(ctx). + Model(&database.Notification{}). + Where("recipient_did = ?", recipient). + Order("id DESC"). + Limit(limit + 1) + if cursorRaw != "" { + cur, err := strconv.ParseUint(cursorRaw, 10, 64) + if err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "cursor must be a positive integer") + } + q = q.Where("id < ?", cur) + } + if len(reasons) > 0 { + for _, r := range reasons { + if _, ok := allowedNotificationReasons[r]; !ok { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "unknown reason: "+r) + } + } + q = q.Where("reason IN ?", reasons) + } + + var rows []database.Notification + if err := q.Find(&rows).Error; err != nil { + return h.internalError(c, "ListNotifications.find", err) + } + + nextCursor := "" + if len(rows) > limit { + nextCursor = strconv.FormatUint(uint64(rows[limit-1].ID), 10) + rows = rows[:limit] + } + + seenAt := h.lookupSeenAt(ctx, recipient) + actors := h.hydrateNotificationActors(ctx, rows) + + items := make([]notificationItem, 0, len(rows)) + for _, row := range rows { + items = append(items, notificationItem{ + ID: row.ID, + Reason: row.Reason, + ActorDID: row.ActorDID, + SubjectURI: row.SubjectURI, + SourceURI: row.SourceURI, + CreatedAt: row.CreatedAt.UTC().Format(time.RFC3339Nano), + IsRead: !seenAt.IsZero() && !row.CreatedAt.After(seenAt), + Actor: actors[row.ActorDID], + }) + } + + resp := map[string]any{ + "notifications": items, + "cursor": nextCursor, + } + if !seenAt.IsZero() { + resp["seen_at"] = seenAt.UTC().Format(time.RFC3339Nano) + } + return c.JSON(http.StatusOK, resp) +} + +func (h *Handlers) lookupSeenAt(ctx context.Context, did string) time.Time { + var state database.NotificationState + if err := h.db.WithContext(ctx).Where("did = ?", did).First(&state).Error; err != nil { + return time.Time{} + } + return state.SeenAt +} + +// hydrateNotificationActors batch-fetches commentAuthor info for every +// distinct actor DID in `rows`. Same shape as the comment hydrator so the +// iOS client can reuse its avatar / handle rendering. +func (h *Handlers) hydrateNotificationActors(ctx context.Context, rows []database.Notification) map[string]*commentAuthor { + if len(rows) == 0 { + return map[string]*commentAuthor{} + } + seen := map[string]struct{}{} + dids := make([]string, 0, len(rows)) + for _, row := range rows { + if _, ok := seen[row.ActorDID]; ok { + continue + } + seen[row.ActorDID] = struct{}{} + dids = append(dids, row.ActorDID) + } + + profiles := make([]database.Profile, 0, len(dids)) + if err := h.db.WithContext(ctx).Where("did IN ?", dids).Find(&profiles).Error; err != nil { + h.logger.Warn("hydrateNotificationActors profile lookup failed", "err", err) + } + var pdsRows []database.DIDPDS + _ = h.db.WithContext(ctx).Where("did IN ?", dids).Find(&pdsRows).Error + pdsEndpoints := map[string]string{} + for _, row := range pdsRows { + pdsEndpoints[row.DID] = row.PDSEndpoint + } + + out := map[string]*commentAuthor{} + for _, p := range profiles { + out[p.DID] = &commentAuthor{ + DID: p.DID, + Handle: p.Handle, + DisplayName: p.DisplayName, + Avatar: avatarURL(pdsEndpoints[p.DID], p.DID, p.AvatarCID), + } + } + for _, did := range dids { + if _, ok := out[did]; !ok { + out[did] = &commentAuthor{DID: did} + } + } + return out +} + +type updateSeenRequest struct { + SeenAt string `json:"seen_at"` +} + +// UpdateSeen advances the viewer's watermark. Idempotent — older calls +// (older seenAt timestamp than what's stored) are no-ops. +func (h *Handlers) UpdateSeen(c echo.Context) error { + did := requestingDID(c) + if did == "" { + return writeError(c, http.StatusUnauthorized, "AuthRequired", "authenticated DID is required") + } + var req updateSeenRequest + if err := c.Bind(&req); err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "invalid JSON body") + } + if req.SeenAt == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "seen_at is required") + } + seenAt, err := time.Parse(time.RFC3339Nano, req.SeenAt) + if err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "seen_at must be RFC3339") + } + + ctx := c.Request().Context() + row := database.NotificationState{DID: did, SeenAt: seenAt} + // Only advance the watermark — never roll it back. + err = h.db.WithContext(ctx).Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "did"}}, + DoUpdates: clause.Assignments(map[string]any{ + "seen_at": clause.Expr{ + SQL: "GREATEST(notification_state.seen_at, ?)", + Vars: []any{seenAt}, + }, + }), + }).Create(&row).Error + if err != nil { + return h.internalError(c, "UpdateSeen", err) + } + return c.JSON(http.StatusOK, map[string]any{"ok": true}) +} + +type preference struct { + Reason string `json:"reason"` + Enabled bool `json:"enabled"` +} + +type setPreferencesRequest struct { + Preferences []preference `json:"preferences"` +} + +// GetPreferences returns one entry per known reason. Reasons with no row +// in notification_prefs default to enabled. +func (h *Handlers) GetPreferences(c echo.Context) error { + did := requestingDID(c) + if did == "" { + return writeError(c, http.StatusUnauthorized, "AuthRequired", "authenticated DID is required") + } + + var rows []database.NotificationPref + if err := h.db.WithContext(c.Request().Context()). + Where("did = ?", did). + Find(&rows).Error; err != nil { + return h.internalError(c, "GetPreferences", err) + } + stored := map[string]bool{} + for _, row := range rows { + stored[row.Reason] = row.Enabled + } + + prefs := make([]preference, 0, len(allowedNotificationReasons)) + for reason := range allowedNotificationReasons { + enabled := true + if v, ok := stored[reason]; ok { + enabled = v + } + prefs = append(prefs, preference{Reason: reason, Enabled: enabled}) + } + return c.JSON(http.StatusOK, map[string]any{"preferences": prefs}) +} + +// SetPreferences replaces every preference for the viewer. Missing reasons +// in the request body revert to the default (enabled) by deletion. +func (h *Handlers) SetPreferences(c echo.Context) error { + did := requestingDID(c) + if did == "" { + return writeError(c, http.StatusUnauthorized, "AuthRequired", "authenticated DID is required") + } + var req setPreferencesRequest + if err := c.Bind(&req); err != nil { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "invalid JSON body") + } + + for _, p := range req.Preferences { + if _, ok := allowedNotificationReasons[p.Reason]; !ok { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "unknown reason: "+p.Reason) + } + } + + ctx := c.Request().Context() + err := h.db.WithContext(ctx).Transaction(func(tx *gorm.DB) error { + if err := tx.Where("did = ?", did).Delete(&database.NotificationPref{}).Error; err != nil { + return err + } + if len(req.Preferences) == 0 { + return nil + } + rows := make([]database.NotificationPref, 0, len(req.Preferences)) + for _, p := range req.Preferences { + rows = append(rows, database.NotificationPref{ + DID: did, + Reason: p.Reason, + Enabled: p.Enabled, + }) + } + return tx.Create(&rows).Error + }) + if err != nil { + if errors.Is(err, gorm.ErrInvalidData) { + return writeError(c, http.StatusBadRequest, "InvalidRequest", err.Error()) + } + return h.internalError(c, "SetPreferences", err) + } + return c.JSON(http.StatusOK, map[string]any{"ok": true}) +} diff --git a/appview/handlers/profile.go b/appview/handlers/profile.go index a8745f5..b6ce58e 100644 --- a/appview/handlers/profile.go +++ b/appview/handlers/profile.go @@ -14,8 +14,21 @@ func (h *Handlers) GetProfile(c echo.Context) error { return writeError(c, http.StatusBadRequest, "InvalidRequest", "did is required") } + ctx := c.Request().Context() + + // Reject profile reads for accounts the relay has marked inactive + // (takendown / suspended / deactivated). The row may still exist, but + // we treat the account as gone from the public surface. + var inactiveCount int64 + if err := h.db.WithContext(ctx). + Model(&database.AccountStatus{}). + Where("did = ? AND active = FALSE", did). + Count(&inactiveCount).Error; err == nil && inactiveCount > 0 { + return writeError(c, http.StatusNotFound, "NotFound", "profile not found") + } + var profile database.Profile - if err := h.db.WithContext(c.Request().Context()).Where("did = ?", did).First(&profile).Error; err != nil { + if err := h.db.WithContext(ctx).Where("did = ?", did).First(&profile).Error; err != nil { return writeError(c, http.StatusNotFound, "NotFound", "profile not found") } @@ -23,6 +36,10 @@ func (h *Handlers) GetProfile(c echo.Context) error { if len(profile.FavoriteGenres) > 0 { _ = json.Unmarshal(profile.FavoriteGenres, &genres) } + labelerSubs := []string{} + if len(profile.LabelerSubscriptions) > 0 { + _ = json.Unmarshal(profile.LabelerSubscriptions, &labelerSubs) + } var subscriptionCount int64 _ = h.db.WithContext(c.Request().Context()).Model(&database.Subscription{}).Where("did = ?", did).Count(&subscriptionCount).Error @@ -37,14 +54,15 @@ func (h *Handlers) GetProfile(c echo.Context) error { return c.JSON(http.StatusOK, map[string]any{ "profile": map[string]any{ - "did": profile.DID, - "display_name": profile.DisplayName, - "description": profile.Description, - "favorite_genres": genres, - "subscription_count": subscriptionCount, - "comment_count": commentCount, - "recommendation_count": recommendationCount, - "list_count": listCount, + "did": profile.DID, + "display_name": profile.DisplayName, + "description": profile.Description, + "favorite_genres": genres, + "labeler_subscriptions": labelerSubs, + "subscription_count": subscriptionCount, + "comment_count": commentCount, + "recommendation_count": recommendationCount, + "list_count": listCount, }, }) } diff --git a/appview/handlers/recommendation.go b/appview/handlers/recommendation.go index 342d1d5..0750e76 100644 --- a/appview/handlers/recommendation.go +++ b/appview/handlers/recommendation.go @@ -4,13 +4,12 @@ import ( "net/http" "time" - "tangled.org/sparrowtek.com/effem-AppView/appview/database" "github.com/labstack/echo/v4" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" ) func (h *Handlers) GetRecommendations(c echo.Context) error { - feedID := parseInt64(c.QueryParam("feedId"), 0) - episodeID := parseInt64(c.QueryParam("episodeId"), 0) + subjectURI := c.QueryParam("subject") limit := parseLimit(c.QueryParam("limit"), 50, 100) cursor := c.QueryParam("cursor") @@ -19,17 +18,15 @@ func (h *Handlers) GetRecommendations(c echo.Context) error { Where("removed = ?", false). Order("rkey DESC"). Limit(limit + 1) - if feedID > 0 { - q = q.Where("feed_id = ?", feedID) - } - if episodeID > 0 { - q = q.Where("episode_id = ?", episodeID) + if subjectURI != "" { + q = q.Where("subject_uri = ?", subjectURI) } if cursor != "" { q = q.Where("rkey < ?", cursor) } q = h.excludeBlockedDIDs(q, requestingDID(c), "did") + q = h.excludeInactiveAccounts(q, "did") var rows []database.Recommendation if err := q.Find(&rows).Error; err != nil { @@ -42,32 +39,20 @@ func (h *Handlers) GetRecommendations(c echo.Context) error { rows = rows[:limit] } - type episodeRef struct { - FeedID int64 `json:"feed_id"` - EpisodeID int64 `json:"episode_id"` - EpisodeGuid string `json:"episode_guid,omitempty"` - PodcastGuid string `json:"podcast_guid,omitempty"` - } - type recommendation struct { - DID string `json:"did"` - Rkey string `json:"rkey"` - Episode episodeRef `json:"episode"` - Text string `json:"text,omitempty"` - CreatedAt string `json:"created_at"` + DID string `json:"did"` + Rkey string `json:"rkey"` + Subject strongRefJSON `json:"subject"` + Text string `json:"text,omitempty"` + CreatedAt string `json:"created_at"` } list := make([]recommendation, 0, len(rows)) for _, row := range rows { list = append(list, recommendation{ - DID: row.DID, - Rkey: row.Rkey, - Episode: episodeRef{ - FeedID: row.FeedID, - EpisodeID: row.EpisodeID, - EpisodeGuid: row.EpisodeGuid, - PodcastGuid: row.PodcastGuid, - }, + DID: row.DID, + Rkey: row.Rkey, + Subject: strongRefJSON{URI: row.SubjectURI, CID: row.SubjectCID}, Text: row.Text, CreatedAt: row.CreatedAt, }) @@ -77,9 +62,8 @@ func (h *Handlers) GetRecommendations(c echo.Context) error { } type popularRow struct { - FeedID int64 `json:"feed_id"` - EpisodeID int64 `json:"episode_id"` - Count int64 `json:"count"` + SubjectURI string `json:"subject_uri"` + Count int64 `json:"count"` } func (h *Handlers) GetPopular(c echo.Context) error { @@ -106,8 +90,8 @@ func (h *Handlers) GetPopular(c echo.Context) error { } var rows []popularRow - err := q.Select("feed_id, episode_id, COUNT(*) as count"). - Group("feed_id, episode_id"). + err := q.Select("subject_uri, COUNT(*) as count"). + Group("subject_uri"). Order("count DESC"). Limit(limit). Scan(&rows).Error diff --git a/appview/handlers/social.go b/appview/handlers/social.go index a1e4003..7548ec7 100644 --- a/appview/handlers/social.go +++ b/appview/handlers/social.go @@ -3,9 +3,11 @@ package handlers import ( "context" "encoding/json" + "errors" "sort" "strconv" + "gorm.io/gorm" "tangled.org/sparrowtek.com/effem-AppView/appview/database" ) @@ -20,6 +22,7 @@ type EpisodeSocialCounts struct { CommentCount int `json:"comment_count"` RecommendationCount int `json:"recommendation_count"` BookmarkCount int `json:"bookmark_count"` + LikeCount int `json:"like_count"` } func anyToInt64(v any) int64 { @@ -103,28 +106,42 @@ func feedIDFromItem(item map[string]any) int64 { return 0 } +// podcastSocialCounts returns aggregate counts for the podcast identified by +// the given Podcast Index feed ID. It looks up the corresponding catalog row +// to find the canonical subject_uri; podcasts that have not yet been +// registered in the catalog return all-zero counts. func (h *Handlers) podcastSocialCounts(ctx context.Context, feedID int64) SocialCounts { if feedID <= 0 { - return SocialCounts{} + return SocialCounts{SubscribedByFollowing: []string{}} + } + + var podcastRow database.PodcastCatalog + if err := h.db.WithContext(ctx). + Where("podcast_index_feed_id = ?", feedID). + First(&podcastRow).Error; err != nil { + return SocialCounts{SubscribedByFollowing: []string{}} } var stats database.PodcastStats - if err := h.db.WithContext(ctx).Where("feed_id = ?", feedID).First(&stats).Error; err == nil { + if err := h.db.WithContext(ctx). + Where("subject_uri = ?", podcastRow.ATURI). + First(&stats).Error; err == nil { return SocialCounts{ SubscriberCount: stats.SubscriberCount, CommentCount: stats.CommentCount, RecommendationCount: stats.RecommendationCount, SubscribedByFollowing: []string{}, } + } else if !errors.Is(err, gorm.ErrRecordNotFound) { + h.logger.Warn("podcastSocialCounts stats fetch failed", "err", err, "feedId", feedID) } - var subCount int64 - _ = h.db.WithContext(ctx).Model(&database.Subscription{}).Where("feed_id = ?", feedID).Count(&subCount).Error - var commentCount int64 - _ = h.db.WithContext(ctx).Model(&database.Comment{}).Where("feed_id = ?", feedID).Count(&commentCount).Error - var recCount int64 - _ = h.db.WithContext(ctx).Model(&database.Recommendation{}).Where("feed_id = ?", feedID).Count(&recCount).Error - + // Fall back to counting on demand when stats haven't been computed yet. + var subCount, commentCount, recCount int64 + _ = h.db.WithContext(ctx).Model(&database.Subscription{}).Where("subject_uri = ?", podcastRow.ATURI).Count(&subCount).Error + // Per-podcast comment / recommendation counts require joining each + // episode's stats; for now expose just the subscription count and let + // callers that need rollups subscribe to the per-episode stats stream. return SocialCounts{ SubscriberCount: int(subCount), CommentCount: int(commentCount), @@ -144,35 +161,57 @@ func sortEpisodeItemsByPublishedDesc(items []map[string]any) { }) } +// episodeSocialCounts returns aggregate counts for one episode identified by +// PI IDs. Translates to the catalog AT-URI; episodes not yet in the catalog +// return all-zero counts. func (h *Handlers) episodeSocialCounts(ctx context.Context, feedID, episodeID int64) EpisodeSocialCounts { if episodeID <= 0 { return EpisodeSocialCounts{} } + var row database.EpisodeCatalog + q := h.db.WithContext(ctx).Where("podcast_index_episode_id = ?", episodeID) + if feedID > 0 { + q = q.Where("podcast_index_feed_id = ?", feedID) + } + if err := q.First(&row).Error; err != nil { + return EpisodeSocialCounts{} + } + var stats database.EpisodeStats - if err := h.db.WithContext(ctx).Where("episode_id = ?", episodeID).First(&stats).Error; err == nil { + if err := h.db.WithContext(ctx). + Where("subject_uri = ?", row.ATURI). + First(&stats).Error; err == nil { return EpisodeSocialCounts{ CommentCount: stats.CommentCount, RecommendationCount: stats.RecommendationCount, BookmarkCount: stats.BookmarkCount, + LikeCount: stats.LikeCount, } + } else if !errors.Is(err, gorm.ErrRecordNotFound) { + h.logger.Warn("episodeSocialCounts stats fetch failed", + "err", err, "feedId", feedID, "episodeId", episodeID) } - q := h.db.WithContext(ctx) - if feedID > 0 { - q = q.Where("feed_id = ?", feedID) - } - + // Fall back to counting on demand. Stats are eventually-consistent so + // this path keeps the response correct between stats refreshes. + db := h.db.WithContext(ctx) var commentCount int64 - _ = q.Model(&database.Comment{}).Where("episode_id = ?", episodeID).Count(&commentCount).Error - var recommendationCount int64 - _ = q.Model(&database.Recommendation{}).Where("episode_id = ?", episodeID).Count(&recommendationCount).Error + _ = db.Model(&database.Comment{}). + Where("subject_uri = ? AND removed = ?", row.ATURI, false). + Count(&commentCount).Error + var recCount int64 + _ = db.Model(&database.Recommendation{}). + Where("subject_uri = ? AND removed = ?", row.ATURI, false). + Count(&recCount).Error var bookmarkCount int64 - _ = q.Model(&database.Bookmark{}).Where("episode_id = ?", episodeID).Count(&bookmarkCount).Error + _ = db.Model(&database.Bookmark{}). + Where("subject_uri = ?", row.ATURI). + Count(&bookmarkCount).Error return EpisodeSocialCounts{ CommentCount: int(commentCount), - RecommendationCount: int(recommendationCount), + RecommendationCount: int(recCount), BookmarkCount: int(bookmarkCount), } } diff --git a/appview/handlers/subscribe.go b/appview/handlers/subscribe.go new file mode 100644 index 0000000..4db5815 --- /dev/null +++ b/appview/handlers/subscribe.go @@ -0,0 +1,118 @@ +package handlers + +import ( + "encoding/json" + "net/http" + "time" + + "github.com/gorilla/websocket" + "github.com/labstack/echo/v4" + "tangled.org/sparrowtek.com/effem-AppView/appview/indexer" +) + +// wsUpgrader negotiates the WebSocket handshake. Origin checks happen at +// the CORS middleware layer and again here as a belt-and-braces: only +// origins on the allow-list reach the upgrader. +var wsUpgrader = websocket.Upgrader{ + ReadBufferSize: 1024, + WriteBufferSize: 4096, + // AT Proto subscriptions are read-only from the client's perspective — + // we accept the upgrade regardless of origin and rely on the auth + // middleware to gate access. The browser CORS path runs separately. + CheckOrigin: func(r *http.Request) bool { return true }, +} + +const ( + // wsWriteTimeout caps how long a single frame write can take. Slow + // consumers get disconnected rather than back-pressuring the bus. + wsWriteTimeout = 10 * time.Second + // wsPingInterval keeps proxies and load balancers from idling the + // connection out. 30s is well under the typical 60s TCP idle timeout. + wsPingInterval = 30 * time.Second + // wsPongTimeout is how long we wait for a pong response before + // treating the connection as dead. + wsPongTimeout = 60 * time.Second +) + +// SubscribeComments streams real-time create/delete/like/unlike events for +// one episode's comment stream. The client passes `?subject=at://...` and +// keeps the socket open while the relevant view is foregrounded. +// +// Each frame on the wire is a JSON object `{"type": "...", ...}` matching +// the xyz.effem.feed.subscribeComments lexicon's message union. No headers +// or framing past what gorilla/websocket gives us. +func (h *Handlers) SubscribeComments(c echo.Context) error { + subject := c.QueryParam("subject") + if subject == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "subject is required") + } + + if h.indexer == nil || h.indexer.Events() == nil { + return writeError(c, http.StatusInternalServerError, "Misconfigured", "event bus unavailable") + } + + conn, err := wsUpgrader.Upgrade(c.Response(), c.Request(), nil) + if err != nil { + // Upgrade already wrote a response on failure; just log. + h.logger.Warn("subscribeComments upgrade failed", "err", err) + return nil + } + defer conn.Close() + + events, unsubscribe := h.indexer.Events().Subscribe(subject) + defer unsubscribe() + + // Reader goroutine: drains incoming frames purely so the pong + // handler fires. We don't expect the client to send anything + // meaningful — any read error tears the connection down. + readerDone := make(chan struct{}) + go func() { + defer close(readerDone) + conn.SetReadDeadline(time.Now().Add(wsPongTimeout)) + conn.SetPongHandler(func(string) error { + conn.SetReadDeadline(time.Now().Add(wsPongTimeout)) + return nil + }) + for { + if _, _, err := conn.NextReader(); err != nil { + return + } + } + }() + + ticker := time.NewTicker(wsPingInterval) + defer ticker.Stop() + + ctx := c.Request().Context() + + for { + select { + case <-ctx.Done(): + return nil + case <-readerDone: + return nil + case evt, ok := <-events: + if !ok { + return nil + } + if err := writeEvent(conn, evt); err != nil { + h.logger.Debug("subscribeComments write failed", "err", err, "subject", subject) + return nil + } + case <-ticker.C: + conn.SetWriteDeadline(time.Now().Add(wsWriteTimeout)) + if err := conn.WriteMessage(websocket.PingMessage, nil); err != nil { + return nil + } + } + } +} + +func writeEvent(conn *websocket.Conn, evt indexer.Event) error { + conn.SetWriteDeadline(time.Now().Add(wsWriteTimeout)) + payload, err := json.Marshal(evt) + if err != nil { + return err + } + return conn.WriteMessage(websocket.TextMessage, payload) +} diff --git a/appview/handlers/subscription.go b/appview/handlers/subscription.go index ca7a7cf..728a203 100644 --- a/appview/handlers/subscription.go +++ b/appview/handlers/subscription.go @@ -3,8 +3,8 @@ package handlers import ( "net/http" - "tangled.org/sparrowtek.com/effem-AppView/appview/database" "github.com/labstack/echo/v4" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" ) func (h *Handlers) GetSubscriptions(c echo.Context) error { @@ -32,29 +32,19 @@ func (h *Handlers) GetSubscriptions(c echo.Context) error { rows = rows[:limit] } - type podcastRef struct { - FeedID int64 `json:"feed_id"` - FeedURL string `json:"feed_url,omitempty"` - PodcastGuid string `json:"podcast_guid,omitempty"` - } - type subscription struct { - DID string `json:"did"` - Rkey string `json:"rkey"` - Podcast podcastRef `json:"podcast"` - CreatedAt string `json:"created_at"` + DID string `json:"did"` + Rkey string `json:"rkey"` + Subject strongRefJSON `json:"subject"` + CreatedAt string `json:"created_at"` } list := make([]subscription, 0, len(rows)) for _, row := range rows { list = append(list, subscription{ - DID: row.DID, - Rkey: row.Rkey, - Podcast: podcastRef{ - FeedID: row.FeedID, - FeedURL: row.FeedURL, - PodcastGuid: row.PodcastGuid, - }, + DID: row.DID, + Rkey: row.Rkey, + Subject: strongRefJSON{URI: row.SubjectURI, CID: row.SubjectCID}, CreatedAt: row.CreatedAt, }) } @@ -66,16 +56,16 @@ func (h *Handlers) GetSubscriptions(c echo.Context) error { } func (h *Handlers) GetSubscribers(c echo.Context) error { - feedID := parseInt64(c.QueryParam("feedId"), 0) - if feedID <= 0 { - return writeError(c, http.StatusBadRequest, "InvalidRequest", "feedId is required") + subjectURI := c.QueryParam("subject") + if subjectURI == "" { + return writeError(c, http.StatusBadRequest, "InvalidRequest", "subject is required") } limit := parseLimit(c.QueryParam("limit"), 50, 100) cursor := c.QueryParam("cursor") q := h.db.WithContext(c.Request().Context()). - Where("feed_id = ?", feedID). + Where("subject_uri = ?", subjectURI). Order("rkey DESC"). Limit(limit + 1) if cursor != "" { @@ -105,12 +95,15 @@ func (h *Handlers) GetSubscribers(c echo.Context) error { } var count int64 - if err := h.db.WithContext(c.Request().Context()).Model(&database.Subscription{}).Where("feed_id = ?", feedID).Count(&count).Error; err != nil { + if err := h.db.WithContext(c.Request().Context()). + Model(&database.Subscription{}). + Where("subject_uri = ?", subjectURI). + Count(&count).Error; err != nil { return h.internalError(c, "GetSubscribers.count", err) } return c.JSON(http.StatusOK, map[string]any{ - "feed_id": feedID, + "subject": subjectURI, "subscribers": list, "count": count, "cursor": nextCursor, diff --git a/appview/indexer/backfill.go b/appview/indexer/backfill.go new file mode 100644 index 0000000..7480d7b --- /dev/null +++ b/appview/indexer/backfill.go @@ -0,0 +1,206 @@ +package indexer + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "io" + "net/http" + "net/url" + "strings" + "sync" + "time" +) + +// ErrBackfillInProgress is returned when a backfill is already running for +// the requested DID. Callers should surface this as a 409 Conflict so the +// client knows to wait and retry instead of retrying immediately. +var ErrBackfillInProgress = errors.New("backfill already in progress for did") + +// backfillLocks ensures only one BackfillUser per DID runs at a time. We +// use sync.Map rather than a regular map+Mutex because the access pattern +// is "load-or-store, then delete on completion" — exactly its sweet spot. +var backfillLocks sync.Map + +// BackfillResult summarises a backfill run for one user. Counts is keyed by +// collection NSID. RecordedErrors counts records that the indexer rejected +// but that have been captured in indexer_errors for retry. +type BackfillResult struct { + DID string `json:"did"` + Counts map[string]int `json:"counts"` + RecordedErrors int `json:"recorded_errors"` +} + +const ( + // backfillPageSize is the listRecords page limit. Most PDS implementations + // cap at 100; we send 100 so we hit the upper bound and minimise round + // trips. + backfillPageSize = 100 + // backfillHTTPTimeout bounds a single listRecords call. Most PDS instances + // respond in well under a second; 30s is forgiving for a cold-cached one. + backfillHTTPTimeout = 30 * time.Second +) + +// BackfillUser walks every Effem-namespaced collection in the given user's +// PDS via com.atproto.repo.listRecords and pushes each record through the +// indexer. Safe to call repeatedly — IndexParsedRecord's FirstOrCreate keeps +// it idempotent. Returns ErrBackfillInProgress if another goroutine is +// already backfilling the same DID. +func (idx *Indexer) BackfillUser(ctx context.Context, did string) (*BackfillResult, error) { + if did == "" { + return nil, fmt.Errorf("did is empty") + } + + if _, loaded := backfillLocks.LoadOrStore(did, struct{}{}); loaded { + return nil, ErrBackfillInProgress + } + defer backfillLocks.Delete(did) + + pds, err := idx.identity.PDSEndpoint(ctx, did) + if err != nil { + return nil, fmt.Errorf("resolve pds for %s: %w", did, err) + } + + client := &http.Client{Timeout: backfillHTTPTimeout} + + result := &BackfillResult{ + DID: did, + Counts: make(map[string]int, len(effemCollections)), + } + + for _, collection := range effemCollections { + count, errs, err := idx.backfillCollection(ctx, client, pds, did, collection) + if err != nil { + return result, fmt.Errorf("backfill %s: %w", collection, err) + } + result.Counts[collection] = count + result.RecordedErrors += errs + } + + return result, nil +} + +// backfillCollection paginates listRecords for a single collection and feeds +// each record through the indexer. Returns the number of records processed +// successfully and the number of records that the indexer rejected (recorded +// in indexer_errors for the watchdog to retry). +func (idx *Indexer) backfillCollection( + ctx context.Context, + client *http.Client, + pds, did, collection string, +) (int, int, error) { + var cursor string + var processed, recordedErrors int + + for { + page, err := listRecords(ctx, client, pds, did, collection, cursor) + if err != nil { + return processed, recordedErrors, err + } + + for _, rec := range page.Records { + rkey := rkeyFromURI(rec.URI) + if rkey == "" { + idx.logger.Warn("backfill: skipping record with unparseable URI", + "uri", rec.URI, "did", did, "collection", collection) + continue + } + if err := idx.IndexParsedRecord(ctx, did, collection, rkey, rec.CID, rec.Value); err != nil { + idx.logger.Warn("backfill: index failed", + "err", err, "did", did, "collection", collection, "rkey", rkey) + idx.RecordError(ctx, did, collection, rkey, rec.CID, err) + recordedErrors++ + continue + } + idx.MarkErrorResolved(ctx, did, collection, rkey) + processed++ + } + + if page.Cursor == "" || len(page.Records) == 0 { + break + } + cursor = page.Cursor + } + + return processed, recordedErrors, nil +} + +type listRecordsRecord struct { + URI string `json:"uri"` + CID string `json:"cid"` + Value map[string]any `json:"value"` +} + +type listRecordsResponse struct { + Cursor string `json:"cursor"` + Records []listRecordsRecord `json:"records"` +} + +// listRecords calls com.atproto.repo.listRecords against the PDS directly so +// we get records as plain JSON maps. Indigo's typed bindings would force a +// CBOR round-trip via LexiconTypeDecoder; for backfill we want the simplest +// path that produces a `map[string]any` per record. +func listRecords( + ctx context.Context, + client *http.Client, + pds, did, collection, cursor string, +) (*listRecordsResponse, error) { + q := url.Values{} + q.Set("repo", did) + q.Set("collection", collection) + q.Set("limit", fmt.Sprintf("%d", backfillPageSize)) + if cursor != "" { + q.Set("cursor", cursor) + } + + endpoint := strings.TrimRight(pds, "/") + "/xrpc/com.atproto.repo.listRecords?" + q.Encode() + req, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil) + if err != nil { + return nil, err + } + req.Header.Set("Accept", "application/json") + + resp, err := client.Do(req) + if err != nil { + return nil, err + } + defer resp.Body.Close() + + body, err := io.ReadAll(resp.Body) + if err != nil { + return nil, err + } + if resp.StatusCode != http.StatusOK { + return nil, fmt.Errorf("listRecords %s %s: HTTP %d: %s", + did, collection, resp.StatusCode, truncate(string(body), 256)) + } + + var out listRecordsResponse + if err := json.Unmarshal(body, &out); err != nil { + return nil, fmt.Errorf("decode listRecords response: %w", err) + } + return &out, nil +} + +// rkeyFromURI extracts the rkey from an AT-URI like +// "at://did:plc:abc/xyz.effem.feed.comment/3jx2…". Returns empty string if +// the URI doesn't follow the expected shape. +func rkeyFromURI(uri string) string { + const prefix = "at://" + if !strings.HasPrefix(uri, prefix) { + return "" + } + parts := strings.Split(uri[len(prefix):], "/") + if len(parts) < 3 { + return "" + } + return parts[2] +} + +func truncate(s string, n int) string { + if len(s) <= n { + return s + } + return s[:n] + "…" +} diff --git a/appview/indexer/bookmark.go b/appview/indexer/bookmark.go index 493cdc5..506b3c6 100644 --- a/appview/indexer/bookmark.go +++ b/appview/indexer/bookmark.go @@ -7,39 +7,31 @@ import ( "tangled.org/sparrowtek.com/effem-AppView/appview/database" ) -func (idx *Indexer) indexBookmark(ctx context.Context, did, rkey string, rec map[string]any) error { - episode, ok := asMap(rec["episode"]) - if !ok { - return fmt.Errorf("bookmark missing episode object") - } - feedID, ok := asInt64(episode["feedId"]) - if !ok || feedID <= 0 { - return fmt.Errorf("bookmark missing valid episode.feedId") - } - episodeID, ok := asInt64(episode["episodeId"]) - if !ok || episodeID <= 0 { - return fmt.Errorf("bookmark missing valid episode.episodeId") +func (idx *Indexer) indexBookmark(ctx context.Context, did, rkey, cid string, rec map[string]any) error { + subject, err := parseStrongRef(rec["subject"]) + if err != nil { + return fmt.Errorf("bookmark subject: %w", err) } bookmark := database.Bookmark{ - DID: did, - Rkey: rkey, - FeedID: feedID, - EpisodeID: episodeID, - EpisodeGuid: asString(episode["episodeGuid"]), - PodcastGuid: asString(episode["podcastGuid"]), - CreatedAt: asString(rec["createdAt"]), + DID: did, + Rkey: rkey, + SubjectURI: subject.URI, + SubjectCID: subject.CID, + CreatedAt: asString(rec["createdAt"]), } if ts, ok := asInt64(rec["timestamp"]); ok { tsInt := int(ts) bookmark.TimestampS = &tsInt } + _ = cid // bookmark records don't carry their own CID on the DB row + db := idx.db.WithContext(ctx) if err := db.Where("did = ? AND rkey = ?", did, rkey).Assign(bookmark).FirstOrCreate(&bookmark).Error; err != nil { return err } - return idx.refreshEpisodeStats(ctx, feedID, episodeID) + return idx.refreshEpisodeStats(ctx, bookmark.SubjectURI) } func (idx *Indexer) indexProfile(ctx context.Context, did string, rec map[string]any) error { @@ -53,12 +45,65 @@ func (idx *Indexer) indexProfile(ctx context.Context, did string, rec map[string } profile := database.Profile{ - DID: did, - DisplayName: asString(rec["displayName"]), - Description: asString(rec["description"]), - FavoriteGenres: jsonBytes(genres), + DID: did, + DisplayName: asString(rec["displayName"]), + Description: asString(rec["description"]), + FavoriteGenres: jsonBytes(genres), + AvatarCID: extractBlobCID(rec["avatar"]), + LabelerSubscriptions: extractDIDArray(rec["labelerSubscriptions"]), } db := idx.db.WithContext(ctx) + // Assign explicit columns so a profile record that omits the avatar + // doesn't clobber a handle the IdentityResolver wrote earlier (handle / + // handle_resolved_at are not in this struct). return db.Where("did = ?", did).Assign(profile).FirstOrCreate(&profile).Error } + +// extractDIDArray normalizes a field expected to hold an array of DID +// strings. Skips empty / duplicate entries so a malformed item can't taint +// the whole list. Returns nil on missing input so the JSONB column stays +// SQL NULL (which the read path treats as "no subscriptions"). +func extractDIDArray(v any) []byte { + raw, ok := v.([]any) + if !ok || len(raw) == 0 { + return nil + } + out := make([]string, 0, len(raw)) + seen := make(map[string]struct{}, len(raw)) + for _, item := range raw { + s := asString(item) + if s == "" { + continue + } + if _, dup := seen[s]; dup { + continue + } + seen[s] = struct{}{} + out = append(out, s) + } + if len(out) == 0 { + return nil + } + return jsonBytes(out) +} + +// extractBlobCID pulls the CID out of an AT Proto blob value (the +// `{$type: blob, ref: {$link: }, mimeType, size}` shape produced by +// atdata.UnmarshalCBOR). Returns empty string for any non-blob input. +func extractBlobCID(v any) string { + blob, ok := v.(map[string]any) + if !ok { + return "" + } + ref, ok := blob["ref"].(map[string]any) + if !ok { + // CBOR-decoded blobs sometimes surface ref as a plain string link; + // accept that shape too. + return asString(blob["ref"]) + } + if link := asString(ref["$link"]); link != "" { + return link + } + return asString(ref["cid"]) +} diff --git a/appview/indexer/comment.go b/appview/indexer/comment.go index 5f96da4..efd245d 100644 --- a/appview/indexer/comment.go +++ b/appview/indexer/comment.go @@ -4,8 +4,8 @@ import ( "context" "fmt" - "tangled.org/sparrowtek.com/effem-AppView/appview/database" "github.com/bluesky-social/indigo/atproto/syntax" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" ) const ( @@ -14,17 +14,9 @@ const ( ) func (idx *Indexer) indexComment(ctx context.Context, did, rkey, cid string, rec map[string]any) error { - episode, ok := asMap(rec["episode"]) - if !ok { - return fmt.Errorf("comment missing episode object") - } - feedID, ok := asInt64(episode["feedId"]) - if !ok || feedID <= 0 { - return fmt.Errorf("comment missing valid episode.feedId") - } - episodeID, ok := asInt64(episode["episodeId"]) - if !ok || episodeID <= 0 { - return fmt.Errorf("comment missing valid episode.episodeId") + subject, err := parseStrongRef(rec["subject"]) + if err != nil { + return fmt.Errorf("comment subject: %w", err) } text := asString(rec["text"]) @@ -38,17 +30,16 @@ func (idx *Indexer) indexComment(ctx context.Context, did, rkey, cid string, rec } comment := database.Comment{ - DID: did, - Rkey: rkey, - CID: cid, - ATURI: fmt.Sprintf("at://%s/xyz.effem.feed.comment/%s", did, rkey), - FeedID: feedID, - EpisodeID: episodeID, - EpisodeGuid: asString(episode["episodeGuid"]), - PodcastGuid: asString(episode["podcastGuid"]), - Text: text, - CreatedAt: createdAt, - Facets: jsonBytes(rec["facets"]), + DID: did, + Rkey: rkey, + CID: cid, + ATURI: fmt.Sprintf("at://%s/xyz.effem.feed.comment/%s", did, rkey), + SubjectURI: subject.URI, + SubjectCID: subject.CID, + Text: text, + CreatedAt: createdAt, + Facets: jsonBytes(rec["facets"]), + SelfLabels: extractSelfLabelValues(rec["labels"]), } if ts, ok := asInt64(rec["timestamp"]); ok { @@ -60,38 +51,177 @@ func (idx *Indexer) indexComment(ctx context.Context, did, rkey, cid string, rec } if reply, ok := asMap(rec["reply"]); ok { - if root, ok := asMap(reply["root"]); ok { - rootURI := asString(root["uri"]) - rootCID := asString(root["cid"]) - if _, err := syntax.ParseATURI(rootURI); err != nil { - return fmt.Errorf("invalid reply.root.uri %q: %w", rootURI, err) - } - if _, err := syntax.ParseCID(rootCID); err != nil { - return fmt.Errorf("invalid reply.root.cid %q: %w", rootCID, err) - } - comment.ReplyRoot = rootURI - comment.ReplyRootCID = rootCID + if rootRef, err := parseStrongRefOptional(reply["root"], "reply.root"); err != nil { + return err + } else if rootRef != nil { + comment.ReplyRoot = rootRef.URI + comment.ReplyRootCID = rootRef.CID } - if parent, ok := asMap(reply["parent"]); ok { - parentURI := asString(parent["uri"]) - parentCID := asString(parent["cid"]) - if _, err := syntax.ParseATURI(parentURI); err != nil { - return fmt.Errorf("invalid reply.parent.uri %q: %w", parentURI, err) - } - if _, err := syntax.ParseCID(parentCID); err != nil { - return fmt.Errorf("invalid reply.parent.cid %q: %w", parentCID, err) - } - comment.ReplyParent = parentURI - comment.ReplyParentCID = parentCID + if parentRef, err := parseStrongRefOptional(reply["parent"], "reply.parent"); err != nil { + return err + } else if parentRef != nil { + comment.ReplyParent = parentRef.URI + comment.ReplyParentCID = parentRef.CID } } db := idx.db.WithContext(ctx) + var existing database.Comment + created := false + if err := db.Where("did = ? AND rkey = ?", did, rkey).First(&existing).Error; err != nil { + created = true + } if err := db.Where("did = ? AND rkey = ?", did, rkey).Assign(comment).FirstOrCreate(&comment).Error; err != nil { return err } - if err := idx.refreshPodcastStats(ctx, feedID); err != nil { + if err := idx.refreshEpisodeStats(ctx, comment.SubjectURI); err != nil { return err } - return idx.refreshEpisodeStats(ctx, feedID, episodeID) + if created { + idx.events.Broadcast(Event{ + Type: EventCreated, + SubjectURI: comment.SubjectURI, + Comment: commentWireValue(comment), + }) + idx.notifyReply(ctx, comment) + idx.notifyMentions(ctx, comment) + } + return nil +} + +// commentWireValue produces the inline-friendly comment shape for the +// websocket frame. Mirrors the public `commentResponse` shape on the read +// path so clients can apply WS events the same way they apply paginated +// results. +func commentWireValue(c database.Comment) map[string]any { + value := map[string]any{ + "did": c.DID, + "rkey": c.Rkey, + "cid": c.CID, + "uri": c.ATURI, + "subject": map[string]any{"uri": c.SubjectURI, "cid": c.SubjectCID}, + "text": c.Text, + "created_at": c.CreatedAt, + "like_count": 0, + } + if c.TimestampS != nil { + value["timestamp"] = *c.TimestampS + } + if c.ReplyRoot != "" && c.ReplyParent != "" { + value["reply"] = map[string]any{ + "root": map[string]any{"uri": c.ReplyRoot, "cid": c.ReplyRootCID}, + "parent": map[string]any{"uri": c.ReplyParent, "cid": c.ReplyParentCID}, + } + } + if len(c.Facets) > 0 { + value["facets"] = c.Facets + } + if len(c.SelfLabels) > 0 { + value["self_labels"] = c.SelfLabels + } + return value +} + +// extractSelfLabelValues normalizes a com.atproto.label.defs#selfLabels union +// into the canonical JSON `["val1", "val2"]` form stored in the comments +// table. Returns nil when the input is missing or malformed so the column +// stays SQL NULL — easier to filter than an empty JSON array. +func extractSelfLabelValues(v any) []byte { + wrapper, ok := v.(map[string]any) + if !ok { + return nil + } + values, ok := wrapper["values"].([]any) + if !ok || len(values) == 0 { + return nil + } + out := make([]string, 0, len(values)) + seen := make(map[string]struct{}, len(values)) + for _, item := range values { + entry, ok := item.(map[string]any) + if !ok { + continue + } + val := asString(entry["val"]) + if val == "" { + continue + } + if _, dup := seen[val]; dup { + continue + } + seen[val] = struct{}{} + out = append(out, val) + } + if len(out) == 0 { + return nil + } + return jsonBytes(out) +} + +func (idx *Indexer) indexCommentLike(ctx context.Context, did, rkey, cid string, rec map[string]any) error { + subject, err := parseStrongRef(rec["subject"]) + if err != nil { + return fmt.Errorf("commentLike subject: %w", err) + } + createdAt := asString(rec["createdAt"]) + if _, err := syntax.ParseDatetime(createdAt); err != nil { + return fmt.Errorf("invalid commentLike createdAt %q: %w", createdAt, err) + } + + like := database.CommentLike{ + DID: did, + Rkey: rkey, + CID: cid, + ATURI: fmt.Sprintf("at://%s/xyz.effem.feed.commentLike/%s", did, rkey), + SubjectURI: subject.URI, + SubjectCID: subject.CID, + CreatedAt: createdAt, + } + + db := idx.db.WithContext(ctx) + existing := database.CommentLike{} + created := false + if err := db.Where("did = ? AND rkey = ?", did, rkey).First(&existing).Error; err != nil { + created = true + } + if err := db.Where("did = ? AND rkey = ?", did, rkey).Assign(like).FirstOrCreate(&like).Error; err != nil { + return err + } + if err := idx.refreshEpisodeStatsForComment(ctx, like.SubjectURI); err != nil { + return err + } + + // Broadcast the live like-count change to subscribers of the comment's + // episode so heart buttons across the room toggle in real time. + if episodeURI, count, ok := idx.likeContextForComment(ctx, like.SubjectURI); ok { + idx.events.Broadcast(Event{ + Type: EventLiked, + SubjectURI: episodeURI, + CommentURI: like.SubjectURI, + LikeCount: count, + ActorDID: did, + }) + } + if created { + idx.notifyLike(ctx, like) + } + return nil +} + +// likeContextForComment returns the episode subject AT-URI that owns +// commentURI and the current like_count, both used to populate live +// like/unlike websocket frames. +func (idx *Indexer) likeContextForComment(ctx context.Context, commentURI string) (string, int, bool) { + var c database.Comment + if err := idx.db.WithContext(ctx).Where("at_uri = ?", commentURI).First(&c).Error; err != nil { + return "", 0, false + } + var n int64 + if err := idx.db.WithContext(ctx). + Model(&database.CommentLike{}). + Where("subject_uri = ?", commentURI). + Count(&n).Error; err != nil { + return "", 0, false + } + return c.SubjectURI, int(n), true } diff --git a/appview/indexer/episode_state.go b/appview/indexer/episode_state.go index 731384d..b88f3e2 100644 --- a/appview/indexer/episode_state.go +++ b/appview/indexer/episode_state.go @@ -7,32 +7,22 @@ import ( "tangled.org/sparrowtek.com/effem-AppView/appview/database" ) -func (idx *Indexer) indexEpisodeState(ctx context.Context, did, rkey string, rec map[string]any) error { - episode, ok := asMap(rec["episode"]) - if !ok { - return fmt.Errorf("episodeState missing episode object") - } - feedID, ok := asInt64(episode["feedId"]) - if !ok || feedID <= 0 { - return fmt.Errorf("episodeState missing valid episode.feedId") - } - episodeID, ok := asInt64(episode["episodeId"]) - if !ok || episodeID <= 0 { - return fmt.Errorf("episodeState missing valid episode.episodeId") +func (idx *Indexer) indexEpisodeState(ctx context.Context, did, rkey, cid string, rec map[string]any) error { + subject, err := parseStrongRef(rec["subject"]) + if err != nil { + return fmt.Errorf("episodeState subject: %w", err) } state := database.EpisodeState{ - DID: did, - Rkey: rkey, - FeedID: feedID, - EpisodeID: episodeID, - EpisodeGuid: asString(episode["episodeGuid"]), - PodcastGuid: asString(episode["podcastGuid"]), - Played: asBool(rec["played"]), - Saved: asBool(rec["saved"]), - Hidden: asBool(rec["hidden"]), - CreatedAt: asString(rec["createdAt"]), - UpdatedAt: asString(rec["updatedAt"]), + DID: did, + Rkey: rkey, + SubjectURI: subject.URI, + SubjectCID: subject.CID, + Played: asBool(rec["played"]), + Saved: asBool(rec["saved"]), + Hidden: asBool(rec["hidden"]), + CreatedAt: asString(rec["createdAt"]), + UpdatedAt: asString(rec["updatedAt"]), } if pos, ok := asInt64(rec["positionS"]); ok { posInt := int(pos) @@ -43,6 +33,8 @@ func (idx *Indexer) indexEpisodeState(ctx context.Context, did, rkey string, rec state.DurationS = &durInt } + _ = cid + db := idx.db.WithContext(ctx) return db.Where("did = ? AND rkey = ?", did, rkey).Assign(state).FirstOrCreate(&state).Error } diff --git a/appview/indexer/errors.go b/appview/indexer/errors.go new file mode 100644 index 0000000..9b2e55a --- /dev/null +++ b/appview/indexer/errors.go @@ -0,0 +1,64 @@ +package indexer + +import ( + "context" + "time" + + "gorm.io/gorm" + "gorm.io/gorm/clause" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" +) + +// RecordError captures a per-record failure in the indexer_errors table. +// The row is unique on (did, collection, rkey) — a re-encounter increments +// `attempts` and updates `error_message` / `last_attempt_at` in place rather +// than spawning a new row. ResolvedAt is left untouched so a previously +// resolved record that fails again surfaces as a fresh problem. +// +// The function tolerates a nil error (no-op) so callers don't have to guard. +func (idx *Indexer) RecordError(ctx context.Context, did, collection, rkey, cid string, recordErr error) { + if recordErr == nil { + return + } + + now := time.Now() + row := database.IndexerError{ + DID: did, + Collection: collection, + Rkey: rkey, + CID: cid, + ErrorMessage: recordErr.Error(), + Attempts: 1, + LastAttemptAt: now, + } + + // Upsert keyed on the unique (did, collection, rkey) index. On conflict + // bump the attempt counter via raw SQL so the increment is atomic; GORM's + // AssignmentColumns would overwrite with the literal "1" from the new row. + err := idx.db.WithContext(ctx).Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "did"}, {Name: "collection"}, {Name: "rkey"}}, + DoUpdates: clause.Assignments(map[string]any{ + "cid": cid, + "error_message": recordErr.Error(), + "attempts": gorm.Expr("indexer_errors.attempts + 1"), + "last_attempt_at": now, + }), + }).Create(&row).Error + if err != nil { + idx.logger.Warn("indexer_errors upsert failed", + "err", err, "did", did, "collection", collection, "rkey", rkey) + } +} + +// MarkErrorResolved stamps `resolved_at` so the watchdog stops retrying and +// the admin listing can hide it by default. No-op if no row exists. +func (idx *Indexer) MarkErrorResolved(ctx context.Context, did, collection, rkey string) { + now := time.Now() + if err := idx.db.WithContext(ctx). + Model(&database.IndexerError{}). + Where("did = ? AND collection = ? AND rkey = ?", did, collection, rkey). + Update("resolved_at", now).Error; err != nil { + idx.logger.Warn("indexer_errors resolve failed", + "err", err, "did", did, "collection", collection, "rkey", rkey) + } +} diff --git a/appview/indexer/events.go b/appview/indexer/events.go new file mode 100644 index 0000000..f8019fc --- /dev/null +++ b/appview/indexer/events.go @@ -0,0 +1,93 @@ +package indexer + +import ( + "sync" +) + +// Event is one comment-stream message. Type names mirror the +// xyz.effem.feed.subscribeComments lexicon union members. +type Event struct { + Type string `json:"type"` + SubjectURI string `json:"-"` // routing key; never sent on the wire + Comment any `json:"comment,omitempty"` + URI string `json:"uri,omitempty"` + CommentURI string `json:"commentUri,omitempty"` + LikeCount int `json:"likeCount,omitempty"` + ActorDID string `json:"actorDid,omitempty"` +} + +// Event type constants. Re-exported for use by the websocket handler. +const ( + EventCreated = "created" + EventDeleted = "deleted" + EventLiked = "liked" + EventUnliked = "unliked" +) + +// EventBus is the in-process pub/sub the indexer uses to push live updates +// to anyone listening over the comment-subscribe websocket. Subscribers +// register interest in a specific subject AT-URI; broadcasts only fan out +// to matching listeners. No persistence — disconnects miss any events +// that fired during the gap, but the catalog + comment table are the +// durable source of truth. +type EventBus struct { + mu sync.RWMutex + subscribers map[string]map[chan Event]struct{} +} + +func NewEventBus() *EventBus { + return &EventBus{ + subscribers: map[string]map[chan Event]struct{}{}, + } +} + +// Subscribe registers a fresh listener for the given subject. Buffer size +// is small (32) — a websocket handler that can't keep up gets dropped +// with a non-blocking send rather than back-pressuring the indexer. +func (b *EventBus) Subscribe(subjectURI string) (<-chan Event, func()) { + ch := make(chan Event, 32) + b.mu.Lock() + set := b.subscribers[subjectURI] + if set == nil { + set = map[chan Event]struct{}{} + b.subscribers[subjectURI] = set + } + set[ch] = struct{}{} + b.mu.Unlock() + + return ch, func() { + b.mu.Lock() + if set := b.subscribers[subjectURI]; set != nil { + delete(set, ch) + if len(set) == 0 { + delete(b.subscribers, subjectURI) + } + } + b.mu.Unlock() + close(ch) + } +} + +// Broadcast publishes one event to every subscriber on its subject. Sends +// are non-blocking — a slow consumer simply misses the event. The websocket +// handler treats a missed event as a "reload from the server" signal. +func (b *EventBus) Broadcast(evt Event) { + if evt.SubjectURI == "" { + return + } + b.mu.RLock() + set := b.subscribers[evt.SubjectURI] + listeners := make([]chan Event, 0, len(set)) + for ch := range set { + listeners = append(listeners, ch) + } + b.mu.RUnlock() + + for _, ch := range listeners { + select { + case ch <- evt: + default: + // drop on saturation; client polling backstops the gap + } + } +} diff --git a/appview/indexer/identity.go b/appview/indexer/identity.go new file mode 100644 index 0000000..c486f6b --- /dev/null +++ b/appview/indexer/identity.go @@ -0,0 +1,161 @@ +package indexer + +import ( + "context" + "errors" + "fmt" + "time" + + "github.com/bluesky-social/indigo/atproto/identity" + "github.com/bluesky-social/indigo/atproto/syntax" + "gorm.io/gorm" + "gorm.io/gorm/clause" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" +) + +// identityFreshness is how long a cached PDS endpoint or handle is trusted +// before we re-resolve. PLC docs change rarely, so 24h is a reasonable +// balance between freshness and load on plc.directory. +const identityFreshness = 24 * time.Hour + +// IdentityResolver wraps indigo's identity.Directory with two persistent +// caches: did_pds (DID → PDS endpoint) and profiles.handle (DID → handle). +// The Directory's in-memory cache disappears on restart; the DB caches +// survive. +type IdentityResolver struct { + directory identity.Directory + db *gorm.DB +} + +// NewIdentityResolver returns a resolver backed by indigo's DefaultDirectory +// (PLC + standard handle resolution). +func NewIdentityResolver(db *gorm.DB) *IdentityResolver { + return &IdentityResolver{ + directory: identity.DefaultDirectory(), + db: db, + } +} + +// ensureIdentityCached refreshes the DID→PDS and DID→handle caches in the +// background. Used by IndexRecord to warm the cache so the next comment +// read can render a handle and an avatar URL without blocking on a PLC +// hit. Errors are intentionally dropped; the next attempt will retry. +func (idx *Indexer) ensureIdentityCached(ctx context.Context, did string) { + if idx.identity == nil || did == "" { + return + } + // Detach from the request context so a slow PLC lookup doesn't block + // the indexer loop. PLC has reasonable internal timeouts; we add our + // own outer bound as belt-and-braces. + go func() { + bgCtx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + _, _ = idx.identity.PDSEndpoint(bgCtx, did) + }() +} + +// PDSEndpoint returns the PDS service URL for a DID, hitting the cache first +// and falling back to a network lookup. The returned endpoint omits any +// trailing slash so callers can append paths cleanly. +// +// On a fresh resolution we also opportunistically refresh the handle in +// `profiles` so the comment hydrator doesn't need a second lookup. +func (r *IdentityResolver) PDSEndpoint(ctx context.Context, did string) (string, error) { + if did == "" { + return "", errors.New("did is empty") + } + + var row database.DIDPDS + err := r.db.WithContext(ctx).Where("did = ?", did).First(&row).Error + if err == nil && time.Since(row.ResolvedAt) < identityFreshness { + return row.PDSEndpoint, nil + } + + ident, err := r.lookup(ctx, did) + if err != nil { + return "", err + } + + endpoint := ident.PDSEndpoint() + if endpoint == "" { + return "", fmt.Errorf("identity for %s has no PDS endpoint", did) + } + + r.cacheDIDPDS(ctx, did, endpoint) + r.cacheHandle(ctx, did, ident.Handle.String()) + + return endpoint, nil +} + +// Handle returns the cached handle for a DID, refreshing from the directory +// if the cached value is stale or missing. Returns the special +// `handle.invalid` placeholder if the DID's handle doesn't bidirectionally +// verify — callers can render this as the raw DID instead. +func (r *IdentityResolver) Handle(ctx context.Context, did string) (string, error) { + if did == "" { + return "", errors.New("did is empty") + } + + var profile database.Profile + err := r.db.WithContext(ctx).Where("did = ?", did).First(&profile).Error + if err == nil && profile.HandleResolvedAt != nil && time.Since(*profile.HandleResolvedAt) < identityFreshness && profile.Handle != "" { + return profile.Handle, nil + } + + ident, err := r.lookup(ctx, did) + if err != nil { + return "", err + } + + handle := ident.Handle.String() + r.cacheHandle(ctx, did, handle) + return handle, nil +} + +func (r *IdentityResolver) lookup(ctx context.Context, did string) (*identity.Identity, error) { + parsedDID, err := syntax.ParseDID(did) + if err != nil { + return nil, fmt.Errorf("invalid did %q: %w", did, err) + } + return r.directory.LookupDID(ctx, parsedDID) +} + +func (r *IdentityResolver) cacheDIDPDS(ctx context.Context, did, endpoint string) { + row := database.DIDPDS{ + DID: did, + PDSEndpoint: endpoint, + ResolvedAt: time.Now(), + } + if err := r.db.WithContext(ctx).Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "did"}}, + DoUpdates: clause.AssignmentColumns([]string{"pds_endpoint", "resolved_at"}), + }).Create(&row).Error; err != nil { + // Caching is best-effort; fall through with the live value rather + // than failing the caller. + _ = err + } +} + +func (r *IdentityResolver) cacheHandle(ctx context.Context, did, handle string) { + if handle == "" { + return + } + now := time.Now() + // Use a partial update so we never clobber display_name / description + // the indexer wrote from the user's xyz.effem.actor.profile record. + // We also create the row with `now` for ResolvedAt; the assignment + // columns include it so re-resolution refreshes the timestamp. + row := database.Profile{ + DID: did, + Handle: handle, + HandleResolvedAt: &now, + } + if err := r.db.WithContext(ctx).Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "did"}}, + DoUpdates: clause.AssignmentColumns([]string{ + "handle", "handle_resolved_at", + }), + }).Create(&row).Error; err != nil { + _ = err + } +} diff --git a/appview/indexer/indexer.go b/appview/indexer/indexer.go index 9535718..d57a82a 100644 --- a/appview/indexer/indexer.go +++ b/appview/indexer/indexer.go @@ -6,6 +6,7 @@ import ( "fmt" "log/slog" + "tangled.org/sparrowtek.com/effem-AppView/appview/catalog" "tangled.org/sparrowtek.com/effem-AppView/appview/database" "github.com/bluesky-social/indigo/atproto/atdata" "github.com/bluesky-social/indigo/atproto/syntax" @@ -13,14 +14,44 @@ import ( ) type Indexer struct { - db *gorm.DB - logger *slog.Logger + db *gorm.DB + logger *slog.Logger + identity *IdentityResolver + catalog *catalog.Service + events *EventBus } -func New(db *gorm.DB, logger *slog.Logger) *Indexer { - return &Indexer{db: db, logger: logger.With("component", "indexer")} +func New(db *gorm.DB, catalogSvc *catalog.Service, logger *slog.Logger) *Indexer { + return &Indexer{ + db: db, + logger: logger.With("component", "indexer"), + identity: NewIdentityResolver(db), + catalog: catalogSvc, + events: NewEventBus(), + } +} + +// Catalog exposes the catalog service for handlers that need to look up +// strongRefs / subject metadata at query time. +func (idx *Indexer) Catalog() *catalog.Service { + return idx.catalog +} + +// Identity exposes the resolver so other packages (handlers, backfill) can +// reuse the cached DID→PDS / DID→handle lookups. +func (idx *Indexer) Identity() *IdentityResolver { + return idx.identity +} + +// Events exposes the in-process event bus so the websocket handler can +// subscribe to per-subject comment streams. +func (idx *Indexer) Events() *EventBus { + return idx.events } +// IndexRecord decodes CBOR record bytes from the firehose and dispatches to +// the per-collection indexer. Backfill paths that already have a parsed map +// should call IndexParsedRecord directly. func (idx *Indexer) IndexRecord(ctx context.Context, did, collection, rkey, cid string, data []byte) error { if _, err := syntax.ParseDID(did); err != nil { return fmt.Errorf("invalid did %q: %w", did, err) @@ -37,19 +68,36 @@ func (idx *Indexer) IndexRecord(ctx context.Context, did, collection, rkey, cid return fmt.Errorf("decoding record cbor: %w", err) } + return idx.IndexParsedRecord(ctx, did, collection, rkey, cid, rec) +} + +// IndexParsedRecord dispatches an already-decoded record map to the +// per-collection indexer. This is the entry point used by the backfill +// path, which fetches records as JSON via listRecords and never sees CBOR. +// +// Caller is responsible for DID / rkey / CID validation if the source is +// untrusted; backfill trusts the PDS response. The identity cache is +// warmed here so both firehose and backfill paths benefit. +func (idx *Indexer) IndexParsedRecord(ctx context.Context, did, collection, rkey, cid string, rec map[string]any) error { + idx.ensureIdentityCached(ctx, did) + switch collection { case "xyz.effem.feed.subscription": - return idx.indexSubscription(ctx, did, rkey, rec) + return idx.indexSubscription(ctx, did, rkey, cid, rec) case "xyz.effem.feed.comment": return idx.indexComment(ctx, did, rkey, cid, rec) + case "xyz.effem.feed.commentLike": + return idx.indexCommentLike(ctx, did, rkey, cid, rec) + case "xyz.effem.feed.threadgate": + return idx.indexThreadgate(ctx, did, rkey, cid, rec) case "xyz.effem.feed.recommendation": - return idx.indexRecommendation(ctx, did, rkey, rec) + return idx.indexRecommendation(ctx, did, rkey, cid, rec) case "xyz.effem.feed.list": return idx.indexList(ctx, did, rkey, rec) case "xyz.effem.feed.bookmark": - return idx.indexBookmark(ctx, did, rkey, rec) + return idx.indexBookmark(ctx, did, rkey, cid, rec) case "xyz.effem.feed.episodeState": - return idx.indexEpisodeState(ctx, did, rkey, rec) + return idx.indexEpisodeState(ctx, did, rkey, cid, rec) case "xyz.effem.actor.profile": return idx.indexProfile(ctx, did, rec) case "xyz.effem.graph.block": @@ -62,6 +110,29 @@ func (idx *Indexer) IndexRecord(ctx context.Context, did, collection, rkey, cid } } +// effemCollections is the set of NSIDs the backfill walker enumerates against +// a user's PDS. Order matters only for log readability. +var effemCollections = []string{ + "xyz.effem.actor.profile", + "xyz.effem.feed.subscription", + "xyz.effem.feed.comment", + "xyz.effem.feed.commentLike", + "xyz.effem.feed.threadgate", + "xyz.effem.feed.recommendation", + "xyz.effem.feed.list", + "xyz.effem.feed.bookmark", + "xyz.effem.feed.episodeState", + "xyz.effem.graph.block", + "xyz.effem.moderation.report", +} + +// EffemCollections returns the NSIDs the backfill walker covers. +func EffemCollections() []string { + out := make([]string, len(effemCollections)) + copy(out, effemCollections) + return out +} + func (idx *Indexer) DeleteRecord(ctx context.Context, did, collection, rkey string) error { if _, err := syntax.ParseDID(did); err != nil { return fmt.Errorf("invalid did %q: %w", did, err) @@ -80,7 +151,8 @@ func (idx *Indexer) DeleteRecord(ctx context.Context, did, collection, rkey stri if err := db.Delete(&row).Error; err != nil { return err } - return idx.refreshPodcastStats(ctx, row.FeedID) + return idx.refreshPodcastStats(ctx, row.SubjectURI) + case "xyz.effem.feed.comment": var row database.Comment if err := db.Where("did = ? AND rkey = ?", did, rkey).First(&row).Error; err != nil { @@ -89,22 +161,50 @@ func (idx *Indexer) DeleteRecord(ctx context.Context, did, collection, rkey stri if err := db.Delete(&row).Error; err != nil { return err } - if err := idx.refreshPodcastStats(ctx, row.FeedID); err != nil { + if err := idx.refreshEpisodeStats(ctx, row.SubjectURI); err != nil { return err } - return idx.refreshEpisodeStats(ctx, row.FeedID, row.EpisodeID) - case "xyz.effem.feed.recommendation": - var row database.Recommendation + idx.events.Broadcast(Event{ + Type: EventDeleted, + SubjectURI: row.SubjectURI, + URI: row.ATURI, + }) + return nil + case "xyz.effem.feed.commentLike": + var row database.CommentLike if err := db.Where("did = ? AND rkey = ?", did, rkey).First(&row).Error; err != nil { return nil } if err := db.Delete(&row).Error; err != nil { return err } - if err := idx.refreshPodcastStats(ctx, row.FeedID); err != nil { + // A commentLike's subject is the comment, not the episode. Refresh + // the episode that owns the comment, derived via the comment's row. + if err := idx.refreshEpisodeStatsForComment(ctx, row.SubjectURI); err != nil { + return err + } + if episodeURI, count, ok := idx.likeContextForComment(ctx, row.SubjectURI); ok { + idx.events.Broadcast(Event{ + Type: EventUnliked, + SubjectURI: episodeURI, + CommentURI: row.SubjectURI, + LikeCount: count, + ActorDID: did, + }) + } + return nil + case "xyz.effem.feed.threadgate": + // Threadgate deletion restores the open state for the gated thread. + return db.Where("root_did = ? AND rkey = ?", did, rkey).Delete(&database.ThreadSettings{}).Error + case "xyz.effem.feed.recommendation": + var row database.Recommendation + if err := db.Where("did = ? AND rkey = ?", did, rkey).First(&row).Error; err != nil { + return nil + } + if err := db.Delete(&row).Error; err != nil { return err } - return idx.refreshEpisodeStats(ctx, row.FeedID, row.EpisodeID) + return idx.refreshEpisodeStats(ctx, row.SubjectURI) case "xyz.effem.feed.list": return db.Where("did = ? AND rkey = ?", did, rkey).Delete(&database.PodcastList{}).Error case "xyz.effem.feed.bookmark": @@ -115,7 +215,7 @@ func (idx *Indexer) DeleteRecord(ctx context.Context, did, collection, rkey stri if err := db.Delete(&row).Error; err != nil { return err } - return idx.refreshEpisodeStats(ctx, row.FeedID, row.EpisodeID) + return idx.refreshEpisodeStats(ctx, row.SubjectURI) case "xyz.effem.feed.episodeState": return db.Where("did = ? AND rkey = ?", did, rkey).Delete(&database.EpisodeState{}).Error case "xyz.effem.actor.profile": diff --git a/appview/indexer/notifications.go b/appview/indexer/notifications.go new file mode 100644 index 0000000..9d1f43f --- /dev/null +++ b/appview/indexer/notifications.go @@ -0,0 +1,187 @@ +package indexer + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "time" + + "gorm.io/gorm" + "gorm.io/gorm/clause" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" +) + +// Notification reason constants. Mirrors the lexicon's knownValues so the +// indexer, handlers, and clients agree on one vocabulary. +const ( + NotificationReasonReply = "reply" + NotificationReasonMention = "mention" + NotificationReasonLike = "like" +) + +// notify writes a notification row plus a matching push_outbox row, both +// inside a single transaction. The composite unique index on +// (recipient_did, actor_did, reason, source_uri) makes re-encountering the +// same source record idempotent — a firehose replay can't deliver the +// same notification twice. +// +// recipient must be set; actor == recipient is a no-op (you don't get +// notifications for your own actions). Disabled reasons in +// notification_prefs are skipped silently. +func (idx *Indexer) notify( + ctx context.Context, + recipientDID, actorDID, reason, subjectURI, sourceURI string, +) { + if recipientDID == "" || actorDID == "" || sourceURI == "" || reason == "" { + return + } + if recipientDID == actorDID { + return + } + + if !idx.notificationsEnabled(ctx, recipientDID, reason) { + return + } + + row := database.Notification{ + RecipientDID: recipientDID, + Reason: reason, + ActorDID: actorDID, + SubjectURI: subjectURI, + SourceURI: sourceURI, + CreatedAt: time.Now().UTC(), + } + + err := idx.db.WithContext(ctx).Transaction(func(tx *gorm.DB) error { + // DoNothing on conflict — the existing row already represents this + // event. RowsAffected will be zero so the outbox enqueue skips. + res := tx.Clauses(clause.OnConflict{ + Columns: []clause.Column{{Name: "recipient_did"}, {Name: "actor_did"}, {Name: "reason"}, {Name: "source_uri"}}, + DoNothing: true, + }).Create(&row) + if res.Error != nil { + return res.Error + } + if res.RowsAffected == 0 { + return nil + } + payload, err := json.Marshal(map[string]any{ + "reason": reason, + "actor_did": actorDID, + "subject_uri": subjectURI, + "source_uri": sourceURI, + }) + if err != nil { + return fmt.Errorf("push_outbox payload: %w", err) + } + outbox := database.PushOutbox{ + NotificationID: row.ID, + RecipientDID: recipientDID, + Payload: payload, + } + return tx.Create(&outbox).Error + }) + if err != nil { + idx.logger.Warn("notify failed", + "err", err, + "recipient", recipientDID, + "actor", actorDID, + "reason", reason, + ) + } +} + +// notificationsEnabled reports whether the recipient has the given reason +// enabled. Missing rows count as enabled (default-allow), matching the +// lexicon contract for `getPreferences`. +func (idx *Indexer) notificationsEnabled(ctx context.Context, did, reason string) bool { + var pref database.NotificationPref + err := idx.db.WithContext(ctx). + Where("did = ? AND reason = ?", did, reason). + First(&pref).Error + if errors.Is(err, gorm.ErrRecordNotFound) { + return true + } + if err != nil { + // Default-allow on infrastructure errors so prefs don't act as a + // silent global mute. + return true + } + return pref.Enabled +} + +// notifyReply enqueues a reply notification to the parent comment's +// author. Looks up the parent's DID via the comments table — replies to +// comments that haven't been indexed yet skip silently and the watchdog +// will reapply once the parent lands. +func (idx *Indexer) notifyReply(ctx context.Context, comment database.Comment) { + if comment.ReplyParent == "" { + return + } + var parent database.Comment + err := idx.db.WithContext(ctx).Where("at_uri = ?", comment.ReplyParent).First(&parent).Error + if err != nil { + // Parent not yet indexed; nothing to notify against. The watchdog + // retry path re-runs the indexer once the parent lands. + return + } + idx.notify(ctx, parent.DID, comment.DID, NotificationReasonReply, parent.ATURI, comment.ATURI) +} + +// notifyMentions scans the comment's facets for mention features and +// enqueues a mention notification per unique DID. Reuses Bluesky's +// app.bsky.richtext.facet shape so iOS / web clients can reuse the same +// rendering logic for both surfaces. +func (idx *Indexer) notifyMentions(ctx context.Context, comment database.Comment) { + if len(comment.Facets) == 0 { + return + } + mentioned := extractMentionedDIDs(comment.Facets) + for did := range mentioned { + idx.notify(ctx, did, comment.DID, NotificationReasonMention, comment.ATURI, comment.ATURI) + } +} + +// notifyLike enqueues a like notification to the comment's author. +func (idx *Indexer) notifyLike(ctx context.Context, like database.CommentLike) { + var target database.Comment + if err := idx.db.WithContext(ctx).Where("at_uri = ?", like.SubjectURI).First(&target).Error; err != nil { + return + } + idx.notify(ctx, target.DID, like.DID, NotificationReasonLike, target.ATURI, like.ATURI) +} + +// extractMentionedDIDs walks a JSON-encoded facets array and returns the +// set of DIDs that appear in `mention` features. The lexicon shape is +// borrowed from Bluesky: +// +// [{"index":{...}, "features":[{"$type":"app.bsky.richtext.facet#mention","did":"did:plc:..."}]}, ...] +func extractMentionedDIDs(raw []byte) map[string]struct{} { + mentions := map[string]struct{}{} + var facets []map[string]any + if err := json.Unmarshal(raw, &facets); err != nil { + return mentions + } + for _, facet := range facets { + features, ok := facet["features"].([]any) + if !ok { + continue + } + for _, f := range features { + feature, ok := f.(map[string]any) + if !ok { + continue + } + t, _ := feature["$type"].(string) + if t != "app.bsky.richtext.facet#mention" { + continue + } + did, _ := feature["did"].(string) + if did != "" { + mentions[did] = struct{}{} + } + } + } + return mentions +} diff --git a/appview/indexer/recommendation.go b/appview/indexer/recommendation.go index 0364d12..00b3fe1 100644 --- a/appview/indexer/recommendation.go +++ b/appview/indexer/recommendation.go @@ -7,37 +7,26 @@ import ( "tangled.org/sparrowtek.com/effem-AppView/appview/database" ) -func (idx *Indexer) indexRecommendation(ctx context.Context, did, rkey string, rec map[string]any) error { - subject, ok := asMap(rec["subject"]) - if !ok { - return fmt.Errorf("recommendation missing subject") - } - feedID, ok := asInt64(subject["feedId"]) - if !ok || feedID <= 0 { - return fmt.Errorf("recommendation missing valid subject.feedId") - } - episodeID, ok := asInt64(subject["episodeId"]) - if !ok || episodeID <= 0 { - return fmt.Errorf("recommendation missing valid subject.episodeId") +func (idx *Indexer) indexRecommendation(ctx context.Context, did, rkey, cid string, rec map[string]any) error { + subject, err := parseStrongRef(rec["subject"]) + if err != nil { + return fmt.Errorf("recommendation subject: %w", err) } reco := database.Recommendation{ - DID: did, - Rkey: rkey, - FeedID: feedID, - EpisodeID: episodeID, - EpisodeGuid: asString(subject["episodeGuid"]), - PodcastGuid: asString(subject["podcastGuid"]), - Text: asString(rec["text"]), - CreatedAt: asString(rec["createdAt"]), + DID: did, + Rkey: rkey, + SubjectURI: subject.URI, + SubjectCID: subject.CID, + Text: asString(rec["text"]), + CreatedAt: asString(rec["createdAt"]), } + _ = cid + db := idx.db.WithContext(ctx) if err := db.Where("did = ? AND rkey = ?", did, rkey).Assign(reco).FirstOrCreate(&reco).Error; err != nil { return err } - if err := idx.refreshPodcastStats(ctx, feedID); err != nil { - return err - } - return idx.refreshEpisodeStats(ctx, feedID, episodeID) + return idx.refreshEpisodeStats(ctx, reco.SubjectURI) } diff --git a/appview/indexer/stats.go b/appview/indexer/stats.go index 4c28db2..d2939a2 100644 --- a/appview/indexer/stats.go +++ b/appview/indexer/stats.go @@ -2,56 +2,49 @@ package indexer import ( "context" + "errors" "time" - "tangled.org/sparrowtek.com/effem-AppView/appview/database" + "gorm.io/gorm" "gorm.io/gorm/clause" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" ) -func (idx *Indexer) refreshPodcastStats(ctx context.Context, feedID int64) error { - if feedID <= 0 { +// refreshPodcastStats recomputes the aggregate counts for one podcast, +// identified by the catalog AT-URI. Subscriptions, comments, and +// recommendations against the podcast all contribute. +func (idx *Indexer) refreshPodcastStats(ctx context.Context, subjectURI string) error { + if subjectURI == "" { return nil } db := idx.db.WithContext(ctx) var subCount int64 - if err := db.Model(&database.Subscription{}).Where("feed_id = ?", feedID).Count(&subCount).Error; err != nil { - return err - } - - var commentCount int64 - if err := db.Model(&database.Comment{}). - Where("feed_id = ?", feedID). - Where("removed = ?", false). - Count(&commentCount).Error; err != nil { - return err - } - - var recommendationCount int64 - if err := db.Model(&database.Recommendation{}). - Where("feed_id = ?", feedID). - Where("removed = ?", false). - Count(&recommendationCount).Error; err != nil { + if err := db.Model(&database.Subscription{}).Where("subject_uri = ?", subjectURI).Count(&subCount).Error; err != nil { return err } row := database.PodcastStats{ - FeedID: feedID, - SubscriberCount: int(subCount), - CommentCount: int(commentCount), - RecommendationCount: int(recommendationCount), - LastUpdated: time.Now().UTC(), + SubjectURI: subjectURI, + SubscriberCount: int(subCount), + // Podcast-level comment/recommendation aggregates are derived by + // rolling up every episode that belongs to the same podcast. Left + // at zero until we wire that join — it's a cheap follow-up that + // doesn't affect the per-episode counts most surfaces care about. + LastUpdated: time.Now().UTC(), } return db.Clauses(clause.OnConflict{ - Columns: []clause.Column{{Name: "feed_id"}}, - DoUpdates: clause.AssignmentColumns([]string{"subscriber_count", "comment_count", "recommendation_count", "last_updated"}), + Columns: []clause.Column{{Name: "subject_uri"}}, + DoUpdates: clause.AssignmentColumns([]string{"subscriber_count", "last_updated"}), }).Create(&row).Error } -func (idx *Indexer) refreshEpisodeStats(ctx context.Context, feedID, episodeID int64) error { - if feedID <= 0 || episodeID <= 0 { +// refreshEpisodeStats recomputes the aggregate counts for one episode, +// identified by its catalog AT-URI. +func (idx *Indexer) refreshEpisodeStats(ctx context.Context, subjectURI string) error { + if subjectURI == "" { return nil } @@ -59,7 +52,7 @@ func (idx *Indexer) refreshEpisodeStats(ctx context.Context, feedID, episodeID i var commentCount int64 if err := db.Model(&database.Comment{}). - Where("feed_id = ? AND episode_id = ?", feedID, episodeID). + Where("subject_uri = ?", subjectURI). Where("removed = ?", false). Count(&commentCount).Error; err != nil { return err @@ -67,28 +60,62 @@ func (idx *Indexer) refreshEpisodeStats(ctx context.Context, feedID, episodeID i var recommendationCount int64 if err := db.Model(&database.Recommendation{}). - Where("feed_id = ? AND episode_id = ?", feedID, episodeID). + Where("subject_uri = ?", subjectURI). Where("removed = ?", false). Count(&recommendationCount).Error; err != nil { return err } var bookmarkCount int64 - if err := db.Model(&database.Bookmark{}).Where("feed_id = ? AND episode_id = ?", feedID, episodeID).Count(&bookmarkCount).Error; err != nil { + if err := db.Model(&database.Bookmark{}). + Where("subject_uri = ?", subjectURI). + Count(&bookmarkCount).Error; err != nil { + return err + } + + // Like count rolls up commentLike records against any of this episode's + // comments. We aggregate via a subquery rather than a JOIN to keep the + // stats refresh self-contained. + var likeCount int64 + if err := db.Model(&database.CommentLike{}). + Where("subject_uri IN (?)", + db.Model(&database.Comment{}). + Select("at_uri"). + Where("subject_uri = ?", subjectURI). + Where("removed = ?", false), + ). + Count(&likeCount).Error; err != nil { return err } row := database.EpisodeStats{ - EpisodeID: episodeID, - FeedID: feedID, + SubjectURI: subjectURI, CommentCount: int(commentCount), RecommendationCount: int(recommendationCount), BookmarkCount: int(bookmarkCount), + LikeCount: int(likeCount), LastUpdated: time.Now().UTC(), } return db.Clauses(clause.OnConflict{ - Columns: []clause.Column{{Name: "episode_id"}}, - DoUpdates: clause.AssignmentColumns([]string{"feed_id", "comment_count", "recommendation_count", "bookmark_count", "last_updated"}), + Columns: []clause.Column{{Name: "subject_uri"}}, + DoUpdates: clause.AssignmentColumns([]string{"comment_count", "recommendation_count", "bookmark_count", "like_count", "last_updated"}), }).Create(&row).Error } + +// refreshEpisodeStatsForComment refreshes the stats of the episode that owns +// the given comment. Used when a commentLike fires — the like's subject is +// the comment, not the episode, so we walk one step to find the right key. +func (idx *Indexer) refreshEpisodeStatsForComment(ctx context.Context, commentURI string) error { + if commentURI == "" { + return nil + } + var c database.Comment + if err := idx.db.WithContext(ctx).Where("at_uri = ?", commentURI).First(&c).Error; err != nil { + if errors.Is(err, gorm.ErrRecordNotFound) { + return nil + } + return err + } + return idx.refreshEpisodeStats(ctx, c.SubjectURI) +} diff --git a/appview/indexer/strongref.go b/appview/indexer/strongref.go new file mode 100644 index 0000000..7d383b2 --- /dev/null +++ b/appview/indexer/strongref.go @@ -0,0 +1,51 @@ +package indexer + +import ( + "fmt" + + "github.com/bluesky-social/indigo/atproto/syntax" +) + +// strongRef mirrors com.atproto.repo.strongRef on the wire shape `{uri, cid}`. +type strongRef struct { + URI string + CID string +} + +// parseStrongRef extracts and validates a strongRef from a CBOR-decoded +// record field. Returns an error if either the URI or CID is missing or +// fails syntax validation. +func parseStrongRef(v any) (strongRef, error) { + m, ok := v.(map[string]any) + if !ok { + return strongRef{}, fmt.Errorf("expected object, got %T", v) + } + uri := asString(m["uri"]) + cidStr := asString(m["cid"]) + if uri == "" { + return strongRef{}, fmt.Errorf("missing uri") + } + if cidStr == "" { + return strongRef{}, fmt.Errorf("missing cid") + } + if _, err := syntax.ParseATURI(uri); err != nil { + return strongRef{}, fmt.Errorf("invalid at-uri %q: %w", uri, err) + } + if _, err := syntax.ParseCID(cidStr); err != nil { + return strongRef{}, fmt.Errorf("invalid cid %q: %w", cidStr, err) + } + return strongRef{URI: uri, CID: cidStr}, nil +} + +// parseStrongRefOptional returns (nil, nil) when v is missing. Field name is +// used for richer error messages on partial structures. +func parseStrongRefOptional(v any, field string) (*strongRef, error) { + if v == nil { + return nil, nil + } + ref, err := parseStrongRef(v) + if err != nil { + return nil, fmt.Errorf("%s: %w", field, err) + } + return &ref, nil +} diff --git a/appview/indexer/subscription.go b/appview/indexer/subscription.go index 649de3d..8870915 100644 --- a/appview/indexer/subscription.go +++ b/appview/indexer/subscription.go @@ -7,28 +7,25 @@ import ( "tangled.org/sparrowtek.com/effem-AppView/appview/database" ) -func (idx *Indexer) indexSubscription(ctx context.Context, did, rkey string, rec map[string]any) error { - podcast, ok := asMap(rec["podcast"]) - if !ok { - return fmt.Errorf("subscription missing podcast object") - } - feedID, ok := asInt64(podcast["feedId"]) - if !ok || feedID <= 0 { - return fmt.Errorf("subscription missing valid podcast.feedId") +func (idx *Indexer) indexSubscription(ctx context.Context, did, rkey, cid string, rec map[string]any) error { + subject, err := parseStrongRef(rec["subject"]) + if err != nil { + return fmt.Errorf("subscription subject: %w", err) } sub := database.Subscription{ - DID: did, - Rkey: rkey, - FeedID: feedID, - FeedURL: asString(podcast["feedUrl"]), - PodcastGuid: asString(podcast["podcastGuid"]), - CreatedAt: asString(rec["createdAt"]), + DID: did, + Rkey: rkey, + SubjectURI: subject.URI, + SubjectCID: subject.CID, + CreatedAt: asString(rec["createdAt"]), } + _ = cid + db := idx.db.WithContext(ctx) if err := db.Where("did = ? AND rkey = ?", did, rkey).Assign(sub).FirstOrCreate(&sub).Error; err != nil { return err } - return idx.refreshPodcastStats(ctx, feedID) + return idx.refreshPodcastStats(ctx, sub.SubjectURI) } diff --git a/appview/indexer/threadgate.go b/appview/indexer/threadgate.go new file mode 100644 index 0000000..f29db10 --- /dev/null +++ b/appview/indexer/threadgate.go @@ -0,0 +1,109 @@ +package indexer + +import ( + "context" + "encoding/json" + "fmt" + + "github.com/bluesky-social/indigo/atproto/syntax" + "tangled.org/sparrowtek.com/effem-AppView/appview/database" +) + +// indexThreadgate persists a xyz.effem.feed.threadgate record into the +// thread_settings table. +// +// Two convention checks happen before the row is written: +// +// 1. The owning DID of the threadgate must equal the owning DID of the +// comment it gates. AT Proto repos are scoped to their owner, so this +// is also enforced by the writer's authorization — but we recheck on +// the read path because a malicious relay could replay arbitrary +// records. +// 2. The threadgate's rkey must equal the gated comment's rkey. This is +// the AT Proto-native way to look up the threadgate for a thread +// without an extra index, and matches Bluesky's app.bsky.feed.threadgate +// convention. +// +// A threadgate that arrives before its comment lands in indexer_errors so +// the watchdog retries once the parent is indexed. +func (idx *Indexer) indexThreadgate(ctx context.Context, did, rkey, _ string, rec map[string]any) error { + subject, err := parseStrongRef(rec["comment"]) + if err != nil { + return fmt.Errorf("threadgate comment: %w", err) + } + + createdAt := asString(rec["createdAt"]) + if _, err := syntax.ParseDatetime(createdAt); err != nil { + return fmt.Errorf("invalid threadgate createdAt %q: %w", createdAt, err) + } + + parsedURI, err := syntax.ParseATURI(subject.URI) + if err != nil { + return fmt.Errorf("threadgate comment uri: %w", err) + } + if parsedURI.Authority().String() != did { + return fmt.Errorf("threadgate did %q does not match comment did %q", did, parsedURI.Authority().String()) + } + if parsedURI.RecordKey().String() != rkey { + return fmt.Errorf("threadgate rkey %q does not match comment rkey %q", rkey, parsedURI.RecordKey().String()) + } + + // Verify the referenced comment actually exists in our index. If not, + // surface as an indexer error so the watchdog re-runs after the + // comment lands. Once a re-run succeeds, the watchdog clears the error. + var comment database.Comment + if err := idx.db.WithContext(ctx).Where("at_uri = ?", subject.URI).First(&comment).Error; err != nil { + return fmt.Errorf("threadgate references unknown comment %s", subject.URI) + } + + allow, err := normalizeAllow(rec["allow"]) + if err != nil { + return fmt.Errorf("threadgate allow: %w", err) + } + + row := database.ThreadSettings{ + RootURI: subject.URI, + RootDID: did, + Rkey: rkey, + Allow: allow, + CreatedAt: createdAt, + } + db := idx.db.WithContext(ctx) + return db.Where("root_uri = ?", subject.URI).Assign(row).FirstOrCreate(&row).Error +} + +// normalizeAllow turns the user-supplied allow value into the canonical +// JSONB-encoded shape stored in `thread_settings.allow`. The result is: +// +// - nil when allow was omitted (no restriction; anyone may reply). +// - JSON `[]` when allow was an empty array (nobody may reply except +// the root author). +// - JSON array of {$type: "..."} unions for each valid rule. +// +// Unknown rule types are dropped silently so a threadgate written by a +// future client doesn't bounce out of the indexer entirely; the known +// rules in the same array are still honored. +func normalizeAllow(v any) ([]byte, error) { + if v == nil { + return nil, nil + } + raw, ok := v.([]any) + if !ok { + return nil, fmt.Errorf("expected array, got %T", v) + } + clean := make([]map[string]any, 0, len(raw)) + for _, entry := range raw { + m, ok := entry.(map[string]any) + if !ok { + continue + } + t := asString(m["$type"]) + switch t { + case "xyz.effem.feed.threadgate#mentionRule": + clean = append(clean, map[string]any{"$type": t}) + default: + // Drop unknown rule types; logging happens at the caller if needed. + } + } + return json.Marshal(clean) +} diff --git a/appview/indexer/watchdog.go b/appview/indexer/watchdog.go new file mode 100644 index 0000000..bf77711 --- /dev/null +++ b/appview/indexer/watchdog.go @@ -0,0 +1,156 @@ +package indexer + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "net/url" + "strings" + "time" + + "tangled.org/sparrowtek.com/effem-AppView/appview/database" +) + +const ( + // watchdogInterval is how often the retry sweep runs. Hourly keeps load + // on user PDS instances low while ensuring transient errors (DNS hiccup, + // brief PDS outage) recover within the same day. + watchdogInterval = time.Hour + // watchdogMaxAttempts caps how many times we re-fetch a failing record + // before giving up. Five attempts spread across five hours is plenty + // for transient causes; persistent failures point at a real schema or + // validation bug worth investigating manually. + watchdogMaxAttempts = 5 + // watchdogMinAge ensures we don't retry a row the firehose only just + // recorded an error against — give the upstream a moment to settle. + watchdogMinAge = 5 * time.Minute + // watchdogBatchSize is how many rows the sweep pulls per tick. Sized so + // a single tick comfortably finishes within the interval even on a + // pathologically slow PDS. + watchdogBatchSize = 100 +) + +// RunWatchdog ticks every watchdogInterval, pulling unresolved indexer_errors +// rows and re-fetching each record from its origin PDS. On success the row's +// resolved_at is stamped. On failure the row's attempts counter is bumped +// (via the upsert path in RecordError) so we eventually stop retrying. +// +// Returns only when ctx is cancelled. +func (idx *Indexer) RunWatchdog(ctx context.Context) { + ticker := time.NewTicker(watchdogInterval) + defer ticker.Stop() + + for { + select { + case <-ctx.Done(): + return + case <-ticker.C: + idx.retryFailedRecords(ctx) + } + } +} + +func (idx *Indexer) retryFailedRecords(ctx context.Context) { + var rows []database.IndexerError + cutoff := time.Now().Add(-watchdogMinAge) + err := idx.db.WithContext(ctx). + Where("resolved_at IS NULL"). + Where("attempts < ?", watchdogMaxAttempts). + Where("last_attempt_at < ?", cutoff). + Order("last_attempt_at ASC"). + Limit(watchdogBatchSize). + Find(&rows).Error + if err != nil { + idx.logger.Warn("watchdog query failed", "err", err) + return + } + + if len(rows) == 0 { + return + } + + client := &http.Client{Timeout: 30 * time.Second} + idx.logger.Info("watchdog retrying records", "count", len(rows)) + + for _, row := range rows { + if err := idx.retryRecord(ctx, client, row); err != nil { + idx.logger.Debug("watchdog retry failed", + "did", row.DID, "collection", row.Collection, + "rkey", row.Rkey, "err", err) + idx.RecordError(ctx, row.DID, row.Collection, row.Rkey, row.CID, err) + continue + } + idx.MarkErrorResolved(ctx, row.DID, row.Collection, row.Rkey) + } +} + +func (idx *Indexer) retryRecord(ctx context.Context, client *http.Client, row database.IndexerError) error { + pds, err := idx.identity.PDSEndpoint(ctx, row.DID) + if err != nil { + return fmt.Errorf("resolve pds: %w", err) + } + + rec, cid, err := getRecord(ctx, client, pds, row.DID, row.Collection, row.Rkey) + if err != nil { + return fmt.Errorf("getRecord: %w", err) + } + + useCID := cid + if useCID == "" { + useCID = row.CID + } + + return idx.IndexParsedRecord(ctx, row.DID, row.Collection, row.Rkey, useCID, rec) +} + +type getRecordResponse struct { + URI string `json:"uri"` + CID string `json:"cid"` + Value map[string]any `json:"value"` +} + +// getRecord fetches a single record from a PDS as JSON. Returns the parsed +// value map plus the record's current CID — useful when the cached CID in +// indexer_errors no longer matches (the record was edited via putRecord). +func getRecord( + ctx context.Context, + client *http.Client, + pds, did, collection, rkey string, +) (map[string]any, string, error) { + q := url.Values{} + q.Set("repo", did) + q.Set("collection", collection) + q.Set("rkey", rkey) + + endpoint := strings.TrimRight(pds, "/") + "/xrpc/com.atproto.repo.getRecord?" + q.Encode() + req, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil) + if err != nil { + return nil, "", err + } + req.Header.Set("Accept", "application/json") + + resp, err := client.Do(req) + if err != nil { + return nil, "", err + } + defer resp.Body.Close() + + body, err := io.ReadAll(resp.Body) + if err != nil { + return nil, "", err + } + if resp.StatusCode == http.StatusNotFound { + return nil, "", fmt.Errorf("record gone from PDS (deleted upstream)") + } + if resp.StatusCode != http.StatusOK { + return nil, "", fmt.Errorf("HTTP %d: %s", resp.StatusCode, truncate(string(body), 256)) + } + + var out getRecordResponse + if err := json.Unmarshal(body, &out); err != nil { + return nil, "", fmt.Errorf("decode: %w", err) + } + return out.Value, out.CID, nil +} diff --git a/appview/server.go b/appview/server.go index e4a5ce1..84cf69d 100644 --- a/appview/server.go +++ b/appview/server.go @@ -16,6 +16,8 @@ import ( "gorm.io/driver/postgres" "gorm.io/gorm" gormlogger "gorm.io/gorm/logger" + "tangled.org/sparrowtek.com/effem-AppView/appview/apns" + "tangled.org/sparrowtek.com/effem-AppView/appview/catalog" "tangled.org/sparrowtek.com/effem-AppView/appview/database" "tangled.org/sparrowtek.com/effem-AppView/appview/handlers" "tangled.org/sparrowtek.com/effem-AppView/appview/httpmw" @@ -29,6 +31,9 @@ type Server struct { echo *echo.Echo pi *podcastindex.CachedClient indexer *indexer.Indexer + catalog *catalog.Service + piResolver *catalog.PIResolver + apns *apns.Dispatcher config Config logger *slog.Logger lastSeq int64 @@ -81,6 +86,18 @@ func NewServer(cfg Config) (*Server, error) { piClient := podcastindex.NewClient(cfg.PIKey, cfg.PISecret) cachedPI := podcastindex.NewCachedClient(piClient, db) + piResolver := catalog.NewPIResolver(cachedPI) + catalogSvc := catalog.New(db, piResolver, catalog.Config{CatalogDID: cfg.CatalogDID}) + + apnsDispatcher, err := apns.New(db, logger, apns.Config{ + KeyPath: cfg.APNsKeyPath, + KeyID: cfg.APNsKeyID, + TeamID: cfg.APNsTeamID, + BundleID: cfg.APNsBundleID, + }) + if err != nil { + return nil, fmt.Errorf("apns dispatcher init: %w", err) + } authz := httpmw.NewTokenAuthorizer(cfg.AuthReadTokens, cfg.AuthAdminTokens). WithAdminDIDs(cfg.AdminDIDs) rateLimiter, err := httpmw.NewPrincipalRateLimiter(httpmw.RateLimiterConfig{ @@ -134,12 +151,15 @@ func NewServer(cfg Config) (*Server, error) { e.Use(metrics.HTTPMiddleware()) srv := &Server{ - db: db, - echo: e, - pi: cachedPI, - indexer: indexer.New(db, logger), - config: cfg, - logger: logger, + db: db, + echo: e, + pi: cachedPI, + indexer: indexer.New(db, catalogSvc, logger), + catalog: catalogSvc, + piResolver: piResolver, + apns: apnsDispatcher, + config: cfg, + logger: logger, } // Expose firehose lag as a scrape-time gauge. Reading lastSeqTime via the @@ -159,7 +179,11 @@ func NewServer(cfg Config) (*Server, error) { } func (srv *Server) registerRoutes() { - h := handlers.New(srv.db, srv.pi, srv.logger) + labelerDID := srv.config.LabelerDID + if labelerDID == "" { + labelerDID = "did:web:labeler.effem.app" + } + h := handlers.New(srv.db, srv.pi, srv.indexer, srv.catalog, srv.piResolver, labelerDID, srv.logger) srv.echo.GET("/_health", srv.handleHealth) @@ -179,6 +203,21 @@ func (srv *Server) registerRoutes() { xrpc.GET("/xyz.effem.feed.getRecommendations", h.GetRecommendations) xrpc.GET("/xyz.effem.feed.getPopular", h.GetPopular) + xrpc.GET("/xyz.effem.feed.resolveEpisode", h.ResolveEpisode) + xrpc.GET("/xyz.effem.feed.resolvePodcast", h.ResolvePodcast) + xrpc.GET("/xyz.effem.feed.getEpisodeRecord", h.GetEpisodeRecord) + xrpc.GET("/xyz.effem.feed.getPodcastRecord", h.GetPodcastRecord) + xrpc.GET("/xyz.effem.feed.describeLexicons", h.DescribeLexicons) + + xrpc.GET("/xyz.effem.feed.subscribeComments", h.SubscribeComments) + + xrpc.GET("/xyz.effem.notification.listNotifications", h.ListNotifications) + xrpc.POST("/xyz.effem.notification.updateSeen", h.UpdateSeen) + xrpc.GET("/xyz.effem.notification.getPreferences", h.GetPreferences) + xrpc.POST("/xyz.effem.notification.setPreferences", h.SetPreferences) + xrpc.POST("/xyz.effem.notification.registerDevice", h.RegisterDevice) + xrpc.POST("/xyz.effem.notification.unregisterDevice", h.UnregisterDevice) + xrpc.GET("/xyz.effem.feed.getList", h.GetList) xrpc.GET("/xyz.effem.feed.getLists", h.GetLists, httpmw.RequireQueryDID("did")) @@ -189,6 +228,8 @@ func (srv *Server) registerRoutes() { xrpc.GET("/xyz.effem.actor.getProfile", h.GetProfile) + xrpc.POST("/xyz.effem.actor.backfillSelf", h.BackfillSelf) + xrpc.GET("/xyz.effem.feed.getInbox", h.GetInbox, httpmw.RequireQueryDID("did")) xrpc.GET("/xyz.effem.search.podcasts", h.SearchPodcasts) @@ -214,6 +255,33 @@ func (srv *Server) registerRoutes() { admin.POST("/xyz.effem.admin.resolveReport", h.ResolveReport) admin.GET("/xyz.effem.admin.getReportedContent", h.GetReportedContent) admin.POST("/xyz.effem.admin.removeContent", h.RemoveContent) + admin.GET("/xyz.effem.admin.listIndexerErrors", h.ListIndexerErrors) + admin.POST("/xyz.effem.admin.backfillUser", h.BackfillUser) + admin.POST("/xyz.effem.admin.applyLabel", h.ApplyLabel) + admin.POST("/xyz.effem.admin.removeLabel", h.RemoveLabel) + admin.GET("/xyz.effem.admin.listLabels", h.ListLabels) + admin.POST("/xyz.effem.admin.importLabels", h.ImportLabels) +} + +// RunWatchdog ticks an hourly sweep over indexer_errors, re-fetching each +// failing record from its origin PDS. Returns only when ctx is cancelled. +func (srv *Server) RunWatchdog(ctx context.Context) { + srv.logger.Info("watchdog starting") + srv.indexer.RunWatchdog(ctx) + srv.logger.Info("watchdog stopped") +} + +// RunPushDispatcher drains push_outbox to APNs. Returns only when ctx is +// cancelled. The dispatcher is safe to start regardless of whether APNs is +// configured — when it isn't, queued rows are marked as sent in-place so +// the table doesn't grow unbounded during development. +func (srv *Server) RunPushDispatcher(ctx context.Context) { + if srv.apns == nil { + return + } + srv.logger.Info("push dispatcher starting", "enabled", srv.apns.Enabled()) + srv.apns.Run(ctx) + srv.logger.Info("push dispatcher stopped") } func (srv *Server) RunAPI(ctx context.Context) error { diff --git a/cmd/effem-appview/main.go b/cmd/effem-appview/main.go index 711a148..c7756af 100644 --- a/cmd/effem-appview/main.go +++ b/cmd/effem-appview/main.go @@ -150,6 +150,36 @@ func main() { EnvVars: []string{"EFFEM_RATE_LIMIT_SUB_BURST"}, Usage: "Per-DID burst request capacity (sub tier)", }, + &cli.StringFlag{ + Name: "apns-key-path", + EnvVars: []string{"EFFEM_APNS_KEY_PATH"}, + Usage: "Path to the APNs .p8 auth key. Push is disabled when empty.", + }, + &cli.StringFlag{ + Name: "apns-key-id", + EnvVars: []string{"EFFEM_APNS_KEY_ID"}, + Usage: "APNs Key ID from the Apple Developer portal.", + }, + &cli.StringFlag{ + Name: "apns-team-id", + EnvVars: []string{"EFFEM_APNS_TEAM_ID"}, + Usage: "Apple Developer Team ID that owns the bundle.", + }, + &cli.StringFlag{ + Name: "apns-bundle-id", + EnvVars: []string{"EFFEM_APNS_BUNDLE_ID"}, + Usage: "iOS app bundle identifier (APNs topic).", + }, + &cli.StringFlag{ + Name: "catalog-did", + EnvVars: []string{"EFFEM_CATALOG_DID"}, + Usage: "DID that owns the episode/podcast catalog repo. Defaults to did:web:catalog.effem.app.", + }, + &cli.StringFlag{ + Name: "labeler-did", + EnvVars: []string{"EFFEM_LABELER_DID"}, + Usage: "DID stamped as `src` on labels emitted by xyz.effem.admin.applyLabel. Defaults to did:web:labeler.effem.app.", + }, }, Action: run, } @@ -197,6 +227,12 @@ func run(cctx *cli.Context) error { RateLimitBurst: cctx.Int("rate-limit-burst"), RateLimitSubRPS: cctx.Float64("rate-limit-sub-rps"), RateLimitSubBurst: cctx.Int("rate-limit-sub-burst"), + APNsKeyPath: cctx.String("apns-key-path"), + APNsKeyID: cctx.String("apns-key-id"), + APNsTeamID: cctx.String("apns-team-id"), + APNsBundleID: cctx.String("apns-bundle-id"), + CatalogDID: cctx.String("catalog-did"), + LabelerDID: cctx.String("labeler-did"), } srv, err := appview.NewServer(cfg) @@ -215,8 +251,25 @@ func run(cctx *cli.Context) error { close(firehoseDone) }() + // Background sweep that retries indexer_errors. Cheaper to start + // alongside the firehose since both share the indexer + DB pool. + watchdogDone := make(chan struct{}) + go func() { + srv.RunWatchdog(ctx) + close(watchdogDone) + }() + + // Push dispatcher consumes push_outbox and ships notifications to APNs. + pushDone := make(chan struct{}) + go func() { + srv.RunPushDispatcher(ctx) + close(pushDone) + }() + apiErr := srv.RunAPI(ctx) <-firehoseDone + <-watchdogDone + <-pushDone if apiErr != nil { return apiErr diff --git a/go.mod b/go.mod index 5894d98..a4a78c5 100644 --- a/go.mod +++ b/go.mod @@ -25,6 +25,7 @@ require ( github.com/gocql/gocql v1.7.0 // indirect github.com/gogo/protobuf v1.3.2 // indirect github.com/golang-jwt/jwt v3.2.2+incompatible // indirect + github.com/golang-jwt/jwt/v4 v4.4.1 // indirect github.com/golang/snappy v0.0.4 // indirect github.com/google/uuid v1.4.0 // indirect github.com/hailocab/go-hostpool v0.0.0-20160125115350-e80d13ce29ed // indirect @@ -79,6 +80,7 @@ require ( github.com/prometheus/common v0.45.0 // indirect github.com/prometheus/procfs v0.12.0 // indirect github.com/russross/blackfriday/v2 v2.1.0 // indirect + github.com/sideshow/apns2 v0.25.0 // indirect github.com/spaolacci/murmur3 v1.1.0 // indirect github.com/valyala/bytebufferpool v1.0.0 // indirect github.com/valyala/fasttemplate v1.2.2 // indirect diff --git a/go.sum b/go.sum index 73d944a..1ec0c3b 100644 --- a/go.sum +++ b/go.sum @@ -1,6 +1,8 @@ github.com/BurntSushi/toml v0.3.1/go.mod h1:xHWCNGjB5oqiDr8zfno3MHue2Ht5sIBksp03qcyfWMU= github.com/RussellLuo/slidingwindow v0.0.0-20200528002341-535bb99d338b h1:5/++qT1/z812ZqBvqQt6ToRswSuPZ/B33m6xVHRzADU= github.com/RussellLuo/slidingwindow v0.0.0-20200528002341-535bb99d338b/go.mod h1:4+EPqMRApwwE/6yo6CxiHoSnBzjRr3jsqer7frxP8y4= +github.com/alecthomas/template v0.0.0-20190718012654-fb15b899a751/go.mod h1:LOuyumcjzFXgccqObfd/Ljyb9UuFJ6TxHnclSeseNhc= +github.com/alecthomas/units v0.0.0-20201120081800-1786d5ef83d4/go.mod h1:OMCwj8VM1Kc9e19TLln2VL61YJF0x1XFtfdL4JdbSyE= github.com/alexbrainman/goissue34681 v0.0.0-20191006012335-3fc7a47baff5 h1:iW0a5ljuFxkLGPNem5Ui+KBjFJzKg4Fv2fnxe4dvzpM= github.com/alexbrainman/goissue34681 v0.0.0-20191006012335-3fc7a47baff5/go.mod h1:Y2QMoi1vgtOIfc+6DhrMOGkLoGzqSV2rKp4Sm+opsyA= github.com/benbjohnson/clock v1.1.0/go.mod h1:J11/hYXuz8f4ySSvYwY0FKfm+ezbsZBKZxNJlLklBHA= @@ -46,6 +48,8 @@ github.com/gogo/protobuf v1.3.2 h1:Ov1cvc58UF3b5XjBnZv7+opcTcQFZebYjWzi34vdm4Q= github.com/gogo/protobuf v1.3.2/go.mod h1:P1XiOD3dCwIKUDQYPy72D8LYyHL2YPYrpS2s69NZV8Q= github.com/golang-jwt/jwt v3.2.2+incompatible h1:IfV12K8xAKAnZqdXVzCZ+TOjboZ2keLg81eXfW3O+oY= github.com/golang-jwt/jwt v3.2.2+incompatible/go.mod h1:8pz2t5EyA70fFQQSrl6XZXzqecmYZeUEB8OUGHkxJ+I= +github.com/golang-jwt/jwt/v4 v4.4.1 h1:pC5DB52sCeK48Wlb9oPcdhnjkz1TKt1D/P7WKJ0kUcQ= +github.com/golang-jwt/jwt/v4 v4.4.1/go.mod h1:m21LjoU+eqJr34lmDMbreY2eSTRJ1cv77w39/MY0Ch0= github.com/golang/snappy v0.0.3/go.mod h1:/XxbfmMg8lxefKM7IXC3fBNl/7bRcc72aCRzEWrmP2Q= github.com/golang/snappy v0.0.4 h1:yAGX7huGHXlcLOEtBnF4w7FQwA26wojNCwOYAEhLjQM= github.com/golang/snappy v0.0.4/go.mod h1:/XxbfmMg8lxefKM7IXC3fBNl/7bRcc72aCRzEWrmP2Q= @@ -251,6 +255,8 @@ github.com/russross/blackfriday/v2 v2.0.1/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQD github.com/russross/blackfriday/v2 v2.1.0 h1:JIOH55/0cWyOuilr9/qlrm0BSXldqnqwMsf35Ld67mk= github.com/russross/blackfriday/v2 v2.1.0/go.mod h1:+Rmxgy9KzJVeS9/2gXHxylqXiyQDYRxCVz55jmeOWTM= github.com/shurcooL/sanitized_anchor_name v1.0.0/go.mod h1:1NzhyTcUVG4SuEtjjoZeVRXNmyL/1OwPU0+IJeTBvfc= +github.com/sideshow/apns2 v0.25.0 h1:XOzanncO9MQxkb03T/2uU2KcdVjYiIf0TMLzec0FTW4= +github.com/sideshow/apns2 v0.25.0/go.mod h1:7Fceu+sL0XscxrfLSkAoH6UtvKefq3Kq1n4W3ayQZqE= github.com/smartystreets/assertions v1.2.0 h1:42S6lae5dvLc7BrLu/0ugRtcFVjoJNMC/N3yZFZkDFs= github.com/smartystreets/assertions v1.2.0/go.mod h1:tcbTF8ujkAEcZ8TElKY+i30BzYlVhC/LOxJk7iOWnoo= github.com/smartystreets/goconvey v1.7.2 h1:9RBaZCeXEQ3UselpuwUQHltGVXvdwm6cv1hgR6gDIPg= @@ -312,6 +318,7 @@ go.uber.org/zap v1.16.0/go.mod h1:MA8QOfq0BHJwdXa996Y4dYkAqRKB8/1K1QMMZVaNZjQ= go.uber.org/zap v1.19.1/go.mod h1:j3DNczoxDZroyBnOT1L/Q79cfUMGZxlv/9dzN7SM1rI= go.uber.org/zap v1.26.0 h1:sI7k6L95XOKS281NhVKOFCUNIvv9e0w4BF8N3u+tCRo= go.uber.org/zap v1.26.0/go.mod h1:dtElttAiwGvoJ/vj4IwHBS/gXsEu/pZ50mUIRWuG0so= +golang.org/x/crypto v0.0.0-20170512130425-ab89591268e0/go.mod h1:6SG95UA2DQfeDnfUPMdvaQW0Q7yPrPDi9nlGo2tz2b4= golang.org/x/crypto v0.0.0-20190308221718-c2843e01d9a2/go.mod h1:djNgcEr1/C05ACkg1iLfiJU5Ep61QUkGW8qpdssI0+w= golang.org/x/crypto v0.0.0-20190510104115-cbcb75029529/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI= golang.org/x/crypto v0.0.0-20191011191535-87dc89f01550/go.mod h1:yigFU9vqHzYiE8UmvKecakEJjdnWj3jj499lnFckfCI= @@ -334,6 +341,7 @@ golang.org/x/net v0.0.0-20190620200207-3b0461eec859/go.mod h1:z5CRVTTTmAJ677TzLL golang.org/x/net v0.0.0-20200226121028-0de0cce0169b/go.mod h1:z5CRVTTTmAJ677TzLLGU+0bjPO0LkuOLi4/5GtJWs/s= golang.org/x/net v0.0.0-20201021035429-f5854403a974/go.mod h1:sp8m0HH+o8qH0wwXwYZr8TS3Oi6o0r6Gce1SSxlDquU= golang.org/x/net v0.0.0-20210405180319-a5a99cb37ef4/go.mod h1:p54w0d4576C0XHj96bSt6lcn1PtDYWL6XObtHCRCNQM= +golang.org/x/net v0.0.0-20220403103023-749bd193bc2b/go.mod h1:CfG3xpIq0wQ8r1q4Su4UZFWDARRcnwPjda9FqA0JpMk= golang.org/x/net v0.24.0 h1:1PcaxkF854Fu3+lvBIx5SYn9wRlBzzcnHZSiaFFAb0w= golang.org/x/net v0.24.0/go.mod h1:2Q7sJY5mzlzWjKtYUEXSlBWCdyaioyXzRB2RtU8KVE8= golang.org/x/sync v0.0.0-20190423024810-112230192c58/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM= @@ -350,15 +358,19 @@ golang.org/x/sys v0.0.0-20200930185726-fdedc70b468f/go.mod h1:h1NjWce9XRLGQEsW7w golang.org/x/sys v0.0.0-20201119102817-f84b799fce68/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= golang.org/x/sys v0.0.0-20210330210617-4fbd30eecc44/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= golang.org/x/sys v0.0.0-20210510120138-977fb7262007/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.0.0-20210615035016-665e8c7367d1/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.0.0-20210630005230-0f9fa26af87c/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= +golang.org/x/sys v0.0.0-20211216021012-1d35b9e2eb4e/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.0.0-20220811171246-fbc7d0a398ab/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.5.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg= golang.org/x/sys v0.22.0 h1:RI27ohtqKCnwULzJLqkv897zojh5/DwS/ENaMzUOaWI= golang.org/x/sys v0.22.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA= golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo= +golang.org/x/term v0.0.0-20210927222741-03fcf44c2211/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8= golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ= golang.org/x/text v0.3.3/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ= +golang.org/x/text v0.3.7/go.mod h1:u+2+/6zg+i71rQMx5EYifcz6MCKuco9NR6JIITiCfzQ= golang.org/x/text v0.14.0 h1:ScX5w1eTa3QqT8oi6+ziP7dTV1S2+ALU0bI+0zXKWiQ= golang.org/x/text v0.14.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU= golang.org/x/text v0.20.0 h1:gK/Kv2otX8gz+wn7Rmb3vT96ZwuoxnQlY+HlJVj7Qug= @@ -386,6 +398,7 @@ golang.org/x/xerrors v0.0.0-20231012003039-104605ab7028 h1:+cNy6SZtPcJQH3LJVLOSm golang.org/x/xerrors v0.0.0-20231012003039-104605ab7028/go.mod h1:NDW/Ps6MPRej6fsCIbMTohpP40sJ/P/vI1MoTEGwX90= google.golang.org/protobuf v1.33.0 h1:uNO2rsAINq/JlFpSdYEKIZ0uKD/R9cpdv0T+yoGwGmI= google.golang.org/protobuf v1.33.0/go.mod h1:c6P6GXX6sHbq/GpV6MGZEdwhWPcYBgnhAHhKbcUYpos= +gopkg.in/alecthomas/kingpin.v2 v2.2.6/go.mod h1:FMv+mEhP44yOT+4EoQTLFTRgOQ1FBLkstjWtayDeSgw= gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= gopkg.in/check.v1 v1.0.0-20180628173108-788fd7840127/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk= diff --git a/lexicons/lexicons.go b/lexicons/lexicons.go new file mode 100644 index 0000000..042b40d --- /dev/null +++ b/lexicons/lexicons.go @@ -0,0 +1,9 @@ +// Package lexicons exposes the Effem-namespace lexicon JSON files as an +// embedded FS so the AppView can serve them at runtime (e.g. via the +// xyz.effem.feed.describeLexicons XRPC) and ship them as part of the binary. +package lexicons + +import "embed" + +//go:embed xyz/effem/*/*.json +var FS embed.FS diff --git a/lexicons/xyz/effem/actor/profile.json b/lexicons/xyz/effem/actor/profile.json index cc9621d..9beac71 100644 --- a/lexicons/xyz/effem/actor/profile.json +++ b/lexicons/xyz/effem/actor/profile.json @@ -19,6 +19,12 @@ "maxLength": 2560, "maxGraphemes": 256 }, + "avatar": { + "type": "blob", + "accept": ["image/png", "image/jpeg", "image/webp"], + "maxSize": 1000000, + "description": "Profile avatar. PNG, JPEG, or WebP, up to 1 MB." + }, "favoriteGenres": { "type": "array", "maxLength": 10, @@ -29,6 +35,12 @@ "maxLength": 320, "maxGraphemes": 32 } + }, + "labelerSubscriptions": { + "type": "array", + "maxLength": 20, + "items": { "type": "string", "format": "did" }, + "description": "DIDs of labeler services this user trusts. Labels emitted by these DIDs appear in this user's read responses; labels from non-subscribed labelers do not (effem's own labeler is always applied)." } } } diff --git a/lexicons/xyz/effem/admin/applyLabel.json b/lexicons/xyz/effem/admin/applyLabel.json new file mode 100644 index 0000000..09dc149 --- /dev/null +++ b/lexicons/xyz/effem/admin/applyLabel.json @@ -0,0 +1,44 @@ +{ + "lexicon": 1, + "id": "xyz.effem.admin.applyLabel", + "defs": { + "main": { + "type": "procedure", + "description": "Apply a moderation label to a subject AT-URI. Source is effem's own labeler DID (configured server-side). Idempotent on (src, subject_uri, val): re-applying the same label updates the timestamp.", + "input": { + "encoding": "application/json", + "schema": { + "type": "object", + "required": ["subject_uri", "val"], + "properties": { + "subject_uri": { + "type": "string", + "format": "at-uri", + "description": "AT-URI of the subject record being labeled (typically a xyz.effem.feed.comment)." + }, + "val": { + "type": "string", + "maxLength": 128, + "description": "Label value. Conventional values: !hide / !warn to drive renderer behavior; spam / abuse / etc. for descriptive labels." + }, + "neg": { + "type": "boolean", + "default": false, + "description": "True to record this row as a negation (un-label). A negation cancels a prior label of the same (src, subject, val)." + } + } + } + }, + "output": { + "encoding": "application/json", + "schema": { + "type": "object", + "required": ["id"], + "properties": { + "id": { "type": "integer" } + } + } + } + } + } +} diff --git a/lexicons/xyz/effem/admin/importLabels.json b/lexicons/xyz/effem/admin/importLabels.json new file mode 100644 index 0000000..b0cccc5 --- /dev/null +++ b/lexicons/xyz/effem/admin/importLabels.json @@ -0,0 +1,54 @@ +{ + "lexicon": 1, + "id": "xyz.effem.admin.importLabels", + "defs": { + "main": { + "type": "procedure", + "description": "Pull labels from an external labeler's com.atproto.label.queryLabels endpoint and import them into the AppView's label store. Imported rows are stamped with the labeler's DID as `src`, so subscribed-labeler filtering on read responses includes them automatically.", + "input": { + "encoding": "application/json", + "schema": { + "type": "object", + "required": ["labeler_did", "endpoint"], + "properties": { + "labeler_did": { + "type": "string", + "format": "did", + "description": "DID of the labeler whose labels to import. Stored as the `src` column of every imported row." + }, + "endpoint": { + "type": "string", + "format": "uri", + "description": "Base URL of the labeler's service (e.g. https://labeler.example.com). The procedure appends /xrpc/com.atproto.label.queryLabels." + }, + "uri_patterns": { + "type": "array", + "minLength": 1, + "maxLength": 20, + "items": { "type": "string", "maxLength": 1024 }, + "description": "URI glob patterns to request from the upstream labeler. Defaults to `at://*` when omitted." + }, + "limit": { + "type": "integer", + "minimum": 1, + "maximum": 250, + "default": 250, + "description": "Maximum labels to pull per upstream page." + } + } + } + }, + "output": { + "encoding": "application/json", + "schema": { + "type": "object", + "required": ["imported", "skipped"], + "properties": { + "imported": { "type": "integer", "description": "Count of new or updated rows." }, + "skipped": { "type": "integer", "description": "Count of rows the upstream returned that were already up-to-date." } + } + } + } + } + } +} diff --git a/lexicons/xyz/effem/admin/listLabels.json b/lexicons/xyz/effem/admin/listLabels.json new file mode 100644 index 0000000..d043978 --- /dev/null +++ b/lexicons/xyz/effem/admin/listLabels.json @@ -0,0 +1,58 @@ +{ + "lexicon": 1, + "id": "xyz.effem.admin.listLabels", + "defs": { + "main": { + "type": "query", + "description": "List labels in the AppView's label store. Filter by source DID, subject URI, or label value. Newest-first paginated by id.", + "parameters": { + "type": "params", + "properties": { + "src": { + "type": "string", + "format": "did", + "description": "Filter to labels emitted by a single labeler DID. Omit for all sources." + }, + "subject": { + "type": "string", + "format": "at-uri", + "description": "Filter to labels on a single subject. Omit for all subjects." + }, + "val": { + "type": "string", + "maxLength": 128, + "description": "Filter to a single label value." + }, + "limit": { "type": "integer", "minimum": 1, "maximum": 500, "default": 100 }, + "cursor": { "type": "string", "description": "Pagination cursor (last id returned)." } + } + }, + "output": { + "encoding": "application/json", + "schema": { + "type": "object", + "required": ["labels"], + "properties": { + "labels": { + "type": "array", + "items": { "type": "ref", "ref": "#labelView" } + }, + "cursor": { "type": "string" } + } + } + } + }, + "labelView": { + "type": "object", + "required": ["id", "src", "subject_uri", "val", "neg", "created_at"], + "properties": { + "id": { "type": "integer" }, + "src": { "type": "string", "format": "did" }, + "subject_uri": { "type": "string", "format": "at-uri" }, + "val": { "type": "string" }, + "neg": { "type": "boolean" }, + "created_at": { "type": "string", "format": "datetime" } + } + } + } +} diff --git a/lexicons/xyz/effem/admin/removeLabel.json b/lexicons/xyz/effem/admin/removeLabel.json new file mode 100644 index 0000000..d7f5cbc --- /dev/null +++ b/lexicons/xyz/effem/admin/removeLabel.json @@ -0,0 +1,21 @@ +{ + "lexicon": 1, + "id": "xyz.effem.admin.removeLabel", + "defs": { + "main": { + "type": "procedure", + "description": "Hard-delete a label row owned by effem's own labeler. Use applyLabel with neg=true instead to record an audit-friendly negation; use removeLabel only to undo a mistake.", + "input": { + "encoding": "application/json", + "schema": { + "type": "object", + "required": ["subject_uri", "val"], + "properties": { + "subject_uri": { "type": "string", "format": "at-uri" }, + "val": { "type": "string", "maxLength": 128 } + } + } + } + } + } +} diff --git a/lexicons/xyz/effem/feed/bookmark.json b/lexicons/xyz/effem/feed/bookmark.json index 19f469e..da96ba5 100644 --- a/lexicons/xyz/effem/feed/bookmark.json +++ b/lexicons/xyz/effem/feed/bookmark.json @@ -8,16 +8,17 @@ "key": "tid", "record": { "type": "object", - "required": ["episode", "createdAt"], + "required": ["subject", "createdAt"], "properties": { - "episode": { + "subject": { "type": "ref", - "ref": "xyz.effem.feed.defs#episodeRef" + "ref": "com.atproto.repo.strongRef", + "description": "strongRef to an xyz.effem.feed.episode record." }, "timestamp": { "type": "integer", "minimum": 0, - "description": "Episode timestamp in seconds." + "description": "Episode playhead in seconds at which the bookmark was placed." }, "createdAt": { "type": "string", diff --git a/lexicons/xyz/effem/feed/comment.json b/lexicons/xyz/effem/feed/comment.json index 3a5360c..1e3622b 100644 --- a/lexicons/xyz/effem/feed/comment.json +++ b/lexicons/xyz/effem/feed/comment.json @@ -4,64 +4,69 @@ "defs": { "main": { "type": "record", - "description": "A comment on a podcast episode.", + "description": "A comment on an episode.", "key": "tid", "record": { "type": "object", - "required": ["episode", "text", "createdAt"], + "required": ["subject", "text", "createdAt"], "properties": { - "episode": { + "subject": { "type": "ref", - "ref": "xyz.effem.feed.defs#episodeRef" + "ref": "com.atproto.repo.strongRef", + "description": "strongRef to an xyz.effem.feed.episode record." }, "text": { "type": "string", "minLength": 1, "minGraphemes": 1, "maxLength": 3000, - "maxGraphemes": 300, - "description": "The comment text." + "maxGraphemes": 300 + }, + "facets": { + "type": "array", + "items": { + "type": "ref", + "ref": "app.bsky.richtext.facet" + }, + "description": "Rich-text annotations (mentions, links, hashtags)." }, "reply": { "type": "ref", - "ref": "#replyRef", - "description": "If this comment is a reply to another comment." + "ref": "#replyRef" }, "timestamp": { "type": "integer", "minimum": 0, - "description": "Episode timestamp in seconds this comment refers to (optional)." + "description": "Episode playhead in seconds at the moment the comment was composed." }, - "facets": { + "langs": { "type": "array", - "items": { - "type": "ref", - "ref": "app.bsky.richtext.facet" - }, - "description": "Rich text facets (mentions, links, hashtags). Reuses Bluesky's facet schema." + "items": { "type": "string", "format": "language" }, + "maxLength": 3 + }, + "labels": { + "type": "union", + "refs": ["com.atproto.label.defs#selfLabels"], + "description": "Author-applied content warning labels. Renderers should respect these to gate sensitive content behind a tap-to-reveal." }, "createdAt": { "type": "string", "format": "datetime" + }, + "updatedAt": { + "type": "string", + "format": "datetime", + "description": "Stamped when the author edits the comment via putRecord." } } } }, "replyRef": { "type": "object", - "description": "Reference to a parent comment.", "required": ["root", "parent"], "properties": { - "root": { - "type": "ref", - "ref": "com.atproto.repo.strongRef", - "description": "AT URI + CID of the root comment in the thread." - }, - "parent": { - "type": "ref", - "ref": "com.atproto.repo.strongRef", - "description": "AT URI + CID of the direct parent comment." - } + "root": { "type": "ref", "ref": "com.atproto.repo.strongRef" }, + "parent": { "type": "ref", "ref": "com.atproto.repo.strongRef" } } } } diff --git a/lexicons/xyz/effem/feed/commentLike.json b/lexicons/xyz/effem/feed/commentLike.json new file mode 100644 index 0000000..f14c40d --- /dev/null +++ b/lexicons/xyz/effem/feed/commentLike.json @@ -0,0 +1,26 @@ +{ + "lexicon": 1, + "id": "xyz.effem.feed.commentLike", + "defs": { + "main": { + "type": "record", + "description": "A like on an xyz.effem.feed.comment record.", + "key": "tid", + "record": { + "type": "object", + "required": ["subject", "createdAt"], + "properties": { + "subject": { + "type": "ref", + "ref": "com.atproto.repo.strongRef", + "description": "strongRef to the xyz.effem.feed.comment being liked." + }, + "createdAt": { + "type": "string", + "format": "datetime" + } + } + } + } + } +} diff --git a/lexicons/xyz/effem/feed/defs.json b/lexicons/xyz/effem/feed/defs.json deleted file mode 100644 index 7be842f..0000000 --- a/lexicons/xyz/effem/feed/defs.json +++ /dev/null @@ -1,52 +0,0 @@ -{ - "lexicon": 1, - "id": "xyz.effem.feed.defs", - "defs": { - "podcastRef": { - "type": "object", - "description": "Canonical reference to a podcast using Podcast Index identifiers.", - "required": ["feedId"], - "properties": { - "feedId": { - "type": "integer", - "minimum": 1, - "description": "Podcast Index feed ID (primary key)." - }, - "feedUrl": { - "type": "string", - "format": "uri", - "description": "RSS feed URL (fallback identifier)." - }, - "podcastGuid": { - "type": "string", - "description": "Podcasting 2.0 podcast:guid value, if available." - } - } - }, - "episodeRef": { - "type": "object", - "description": "Canonical reference to a podcast episode.", - "required": ["feedId", "episodeId"], - "properties": { - "feedId": { - "type": "integer", - "minimum": 1, - "description": "Podcast Index feed ID." - }, - "episodeId": { - "type": "integer", - "minimum": 1, - "description": "Podcast Index episode ID (primary key)." - }, - "episodeGuid": { - "type": "string", - "description": "Episode GUID from the RSS feed." - }, - "podcastGuid": { - "type": "string", - "description": "Podcasting 2.0 podcast:guid." - } - } - } - } -} diff --git a/lexicons/xyz/effem/feed/episode.json b/lexicons/xyz/effem/feed/episode.json new file mode 100644 index 0000000..58c5a97 --- /dev/null +++ b/lexicons/xyz/effem/feed/episode.json @@ -0,0 +1,32 @@ +{ + "lexicon": 1, + "id": "xyz.effem.feed.episode", + "defs": { + "main": { + "type": "record", + "description": "Canonical record for a podcast episode. Subject of comments, recommendations, bookmarks, and episode-state records via com.atproto.repo.strongRef.", + "key": "any", + "record": { + "type": "object", + "required": ["podcastGuid", "episodeGuid", "title", "publishedAt"], + "properties": { + "podcastGuid": { "type": "string", "description": "Podcasting 2.0 podcast:guid." }, + "episodeGuid": { "type": "string" }, + "title": { "type": "string", "maxLength": 1024 }, + "publishedAt": { "type": "string", "format": "datetime" }, + "feedUrl": { "type": "string", "format": "uri" }, + "enclosureUrl": { "type": "string", "format": "uri" }, + "durationS": { "type": "integer", "minimum": 0 }, + "podcastIndex": { "type": "ref", "ref": "#podcastIndexRef", "description": "Cross-reference for clients that index by Podcast Index IDs." } + } + } + }, + "podcastIndexRef": { + "type": "object", + "properties": { + "feedId": { "type": "integer", "minimum": 1 }, + "episodeId": { "type": "integer", "minimum": 1 } + } + } + } +} diff --git a/lexicons/xyz/effem/feed/episodeState.json b/lexicons/xyz/effem/feed/episodeState.json index 74facb0..8944d25 100644 --- a/lexicons/xyz/effem/feed/episodeState.json +++ b/lexicons/xyz/effem/feed/episodeState.json @@ -4,15 +4,16 @@ "defs": { "main": { "type": "record", - "description": "Per-user, per-episode playback and library state. The rkey is the Podcast Index episode ID, giving one record per episode per user.", + "description": "Per-user, per-episode playback and library state. The rkey is derived from the episode's catalog rkey, giving one record per episode per user.", "key": "any", "record": { "type": "object", - "required": ["episode", "createdAt"], + "required": ["subject", "createdAt"], "properties": { - "episode": { + "subject": { "type": "ref", - "ref": "xyz.effem.feed.defs#episodeRef" + "ref": "com.atproto.repo.strongRef", + "description": "strongRef to an xyz.effem.feed.episode record." }, "positionS": { "type": "integer", diff --git a/lexicons/xyz/effem/feed/list.json b/lexicons/xyz/effem/feed/list.json index 4c5b8c4..017ae4e 100644 --- a/lexicons/xyz/effem/feed/list.json +++ b/lexicons/xyz/effem/feed/list.json @@ -30,7 +30,8 @@ "maxLength": 200, "items": { "type": "ref", - "ref": "xyz.effem.feed.defs#podcastRef" + "ref": "com.atproto.repo.strongRef", + "description": "strongRef to an xyz.effem.feed.podcast record." } }, "createdAt": { diff --git a/lexicons/xyz/effem/feed/podcast.json b/lexicons/xyz/effem/feed/podcast.json new file mode 100644 index 0000000..dbc9cc3 --- /dev/null +++ b/lexicons/xyz/effem/feed/podcast.json @@ -0,0 +1,30 @@ +{ + "lexicon": 1, + "id": "xyz.effem.feed.podcast", + "defs": { + "main": { + "type": "record", + "description": "Canonical record for a podcast feed. Subject of subscriptions and feed-level engagement via com.atproto.repo.strongRef.", + "key": "any", + "record": { + "type": "object", + "required": ["podcastGuid", "title"], + "properties": { + "podcastGuid": { "type": "string", "description": "Podcasting 2.0 podcast:guid." }, + "title": { "type": "string", "maxLength": 1024 }, + "feedUrl": { "type": "string", "format": "uri" }, + "author": { "type": "string", "maxLength": 1024 }, + "artworkUrl": { "type": "string", "format": "uri" }, + "language": { "type": "string", "maxLength": 32 }, + "podcastIndex": { "type": "ref", "ref": "#podcastIndexRef" } + } + } + }, + "podcastIndexRef": { + "type": "object", + "properties": { + "feedId": { "type": "integer", "minimum": 1 } + } + } + } +} diff --git a/lexicons/xyz/effem/feed/recommendation.json b/lexicons/xyz/effem/feed/recommendation.json index c75fec6..de53fa6 100644 --- a/lexicons/xyz/effem/feed/recommendation.json +++ b/lexicons/xyz/effem/feed/recommendation.json @@ -4,7 +4,7 @@ "defs": { "main": { "type": "record", - "description": "A recommendation (like/upvote) for a podcast episode.", + "description": "A recommendation for a podcast episode.", "key": "tid", "record": { "type": "object", @@ -12,8 +12,8 @@ "properties": { "subject": { "type": "ref", - "ref": "xyz.effem.feed.defs#episodeRef", - "description": "The episode being recommended." + "ref": "com.atproto.repo.strongRef", + "description": "strongRef to an xyz.effem.feed.episode record." }, "text": { "type": "string", diff --git a/lexicons/xyz/effem/feed/subscribeComments.json b/lexicons/xyz/effem/feed/subscribeComments.json new file mode 100644 index 0000000..6e26dcd --- /dev/null +++ b/lexicons/xyz/effem/feed/subscribeComments.json @@ -0,0 +1,59 @@ +{ + "lexicon": 1, + "id": "xyz.effem.feed.subscribeComments", + "defs": { + "main": { + "type": "subscription", + "description": "Streams comment create and delete events for one episode in real time. Filtered by subject (the episode's catalog AT-URI). Connection stays open until the client closes it; the server emits one frame per indexed event after the firehose round-trip completes.", + "parameters": { + "type": "params", + "required": ["subject"], + "properties": { + "subject": { + "type": "string", + "format": "at-uri", + "description": "AT-URI of the xyz.effem.feed.episode catalog record whose comments you want to follow." + } + } + }, + "message": { + "schema": { + "type": "union", + "refs": ["#created", "#deleted", "#liked", "#unliked"] + } + } + }, + "created": { + "type": "object", + "required": ["comment"], + "properties": { + "comment": { "type": "unknown", "description": "The newly-indexed comment record, in the shape returned by xyz.effem.feed.getComments." } + } + }, + "deleted": { + "type": "object", + "required": ["uri"], + "properties": { + "uri": { "type": "string", "format": "at-uri" } + } + }, + "liked": { + "type": "object", + "required": ["commentUri", "likeCount"], + "properties": { + "commentUri": { "type": "string", "format": "at-uri" }, + "likeCount": { "type": "integer", "minimum": 0 }, + "actorDid": { "type": "string", "format": "did" } + } + }, + "unliked": { + "type": "object", + "required": ["commentUri", "likeCount"], + "properties": { + "commentUri": { "type": "string", "format": "at-uri" }, + "likeCount": { "type": "integer", "minimum": 0 }, + "actorDid": { "type": "string", "format": "did" } + } + } + } +} diff --git a/lexicons/xyz/effem/feed/subscription.json b/lexicons/xyz/effem/feed/subscription.json index e133106..bbbccc4 100644 --- a/lexicons/xyz/effem/feed/subscription.json +++ b/lexicons/xyz/effem/feed/subscription.json @@ -4,15 +4,16 @@ "defs": { "main": { "type": "record", - "description": "Record declaring a podcast subscription. Stored in the user's AT Proto repo.", + "description": "Record declaring a podcast subscription.", "key": "tid", "record": { "type": "object", - "required": ["podcast", "createdAt"], + "required": ["subject", "createdAt"], "properties": { - "podcast": { + "subject": { "type": "ref", - "ref": "xyz.effem.feed.defs#podcastRef" + "ref": "com.atproto.repo.strongRef", + "description": "strongRef to an xyz.effem.feed.podcast record." }, "createdAt": { "type": "string", diff --git a/lexicons/xyz/effem/feed/threadgate.json b/lexicons/xyz/effem/feed/threadgate.json new file mode 100644 index 0000000..3975092 --- /dev/null +++ b/lexicons/xyz/effem/feed/threadgate.json @@ -0,0 +1,37 @@ +{ + "lexicon": 1, + "id": "xyz.effem.feed.threadgate", + "defs": { + "main": { + "type": "record", + "description": "Restricts who may reply under a comment thread. The threadgate's rkey MUST match the root comment's rkey, and only the root comment's author may publish one. Absence of the record means anyone may reply. Empty `allow` means nobody may reply.", + "key": "tid", + "record": { + "type": "object", + "required": ["comment", "createdAt"], + "properties": { + "comment": { + "type": "ref", + "ref": "com.atproto.repo.strongRef", + "description": "strongRef to the root xyz.effem.feed.comment whose replies this record governs." + }, + "allow": { + "type": "array", + "maxLength": 5, + "items": { + "type": "union", + "refs": ["#mentionRule"] + }, + "description": "Allowlist rules. A reply is permitted if it matches any listed rule (or if the replier is the root comment's author). Omit the field to permit anyone; supply an empty array to permit nobody." + }, + "createdAt": { "type": "string", "format": "datetime" } + } + } + }, + "mentionRule": { + "type": "object", + "description": "Replies from DIDs explicitly @-mentioned in the root comment's facets are permitted.", + "properties": {} + } + } +} diff --git a/lexicons/xyz/effem/notification/getPreferences.json b/lexicons/xyz/effem/notification/getPreferences.json new file mode 100644 index 0000000..83cc399 --- /dev/null +++ b/lexicons/xyz/effem/notification/getPreferences.json @@ -0,0 +1,31 @@ +{ + "lexicon": 1, + "id": "xyz.effem.notification.getPreferences", + "defs": { + "main": { + "type": "query", + "description": "Returns the authenticated user's per-reason notification preferences. Missing reasons are treated as enabled.", + "output": { + "encoding": "application/json", + "schema": { + "type": "object", + "required": ["preferences"], + "properties": { + "preferences": { + "type": "array", + "items": { "type": "ref", "ref": "#preference" } + } + } + } + } + }, + "preference": { + "type": "object", + "required": ["reason", "enabled"], + "properties": { + "reason": { "type": "string", "knownValues": ["reply", "mention", "like"] }, + "enabled": { "type": "boolean" } + } + } + } +} diff --git a/lexicons/xyz/effem/notification/listNotifications.json b/lexicons/xyz/effem/notification/listNotifications.json new file mode 100644 index 0000000..c3d0a2f --- /dev/null +++ b/lexicons/xyz/effem/notification/listNotifications.json @@ -0,0 +1,69 @@ +{ + "lexicon": 1, + "id": "xyz.effem.notification.listNotifications", + "defs": { + "main": { + "type": "query", + "description": "Returns notifications for the authenticated user, newest-first.", + "parameters": { + "type": "params", + "properties": { + "limit": { "type": "integer", "minimum": 1, "maximum": 100, "default": 50 }, + "cursor": { "type": "string" }, + "reasons": { + "type": "array", + "items": { "type": "string" }, + "description": "Optional filter on notification reason (reply, mention, like)." + } + } + }, + "output": { + "encoding": "application/json", + "schema": { + "type": "object", + "required": ["notifications"], + "properties": { + "notifications": { + "type": "array", + "items": { "type": "ref", "ref": "#notification" } + }, + "cursor": { "type": "string" }, + "seenAt": { "type": "string", "format": "datetime", "description": "Watermark at which the viewer last marked notifications as seen." } + } + } + } + }, + "notification": { + "type": "object", + "required": ["id", "reason", "actorDid", "subjectUri", "sourceUri", "createdAt", "isRead"], + "properties": { + "id": { "type": "integer", "minimum": 1 }, + "reason": { + "type": "string", + "description": "Why the notification was generated.", + "knownValues": ["reply", "mention", "like"] + }, + "actorDid": { "type": "string", "format": "did", "description": "DID of the user whose action triggered the notification." }, + "subjectUri": { "type": "string", "format": "at-uri", "description": "URI of the record the viewer cares about (e.g. the comment that was replied to)." }, + "sourceUri": { "type": "string", "format": "at-uri", "description": "URI of the record that triggered the notification (e.g. the reply itself)." }, + "createdAt": { "type": "string", "format": "datetime" }, + "isRead": { "type": "boolean", "description": "Has the viewer already advanced their seen-watermark past this row?" }, + "actor": { + "type": "ref", + "ref": "#actorView", + "description": "Hydrated actor info (display name, handle, avatar). Optional — clients should fall back to actorDid." + } + } + }, + "actorView": { + "type": "object", + "required": ["did"], + "properties": { + "did": { "type": "string", "format": "did" }, + "handle": { "type": "string" }, + "displayName": { "type": "string" }, + "avatar": { "type": "string", "format": "uri" } + } + } + } +} diff --git a/lexicons/xyz/effem/notification/registerDevice.json b/lexicons/xyz/effem/notification/registerDevice.json new file mode 100644 index 0000000..9bee101 --- /dev/null +++ b/lexicons/xyz/effem/notification/registerDevice.json @@ -0,0 +1,22 @@ +{ + "lexicon": 1, + "id": "xyz.effem.notification.registerDevice", + "defs": { + "main": { + "type": "procedure", + "description": "Registers an APNs device token for the authenticated user. Subsequent notification events for this user will fan out to all of their registered devices.", + "input": { + "encoding": "application/json", + "schema": { + "type": "object", + "required": ["token", "environment"], + "properties": { + "token": { "type": "string", "description": "Hex-encoded APNs device token." }, + "environment": { "type": "string", "knownValues": ["sandbox", "production"] }, + "bundleId": { "type": "string", "description": "App bundle identifier the token belongs to." } + } + } + } + } + } +} diff --git a/lexicons/xyz/effem/notification/setPreferences.json b/lexicons/xyz/effem/notification/setPreferences.json new file mode 100644 index 0000000..48ad72d --- /dev/null +++ b/lexicons/xyz/effem/notification/setPreferences.json @@ -0,0 +1,23 @@ +{ + "lexicon": 1, + "id": "xyz.effem.notification.setPreferences", + "defs": { + "main": { + "type": "procedure", + "description": "Replaces the authenticated user's notification preferences. Pass the full set every time — missing reasons revert to enabled.", + "input": { + "encoding": "application/json", + "schema": { + "type": "object", + "required": ["preferences"], + "properties": { + "preferences": { + "type": "array", + "items": { "type": "ref", "ref": "xyz.effem.notification.getPreferences#preference" } + } + } + } + } + } + } +} diff --git a/lexicons/xyz/effem/notification/unregisterDevice.json b/lexicons/xyz/effem/notification/unregisterDevice.json new file mode 100644 index 0000000..1363397 --- /dev/null +++ b/lexicons/xyz/effem/notification/unregisterDevice.json @@ -0,0 +1,20 @@ +{ + "lexicon": 1, + "id": "xyz.effem.notification.unregisterDevice", + "defs": { + "main": { + "type": "procedure", + "description": "Removes an APNs device token from the authenticated user. Called on logout and on APNs token-invalidation feedback.", + "input": { + "encoding": "application/json", + "schema": { + "type": "object", + "required": ["token"], + "properties": { + "token": { "type": "string", "description": "Hex-encoded APNs device token." } + } + } + } + } + } +} diff --git a/lexicons/xyz/effem/notification/updateSeen.json b/lexicons/xyz/effem/notification/updateSeen.json new file mode 100644 index 0000000..226e2df --- /dev/null +++ b/lexicons/xyz/effem/notification/updateSeen.json @@ -0,0 +1,20 @@ +{ + "lexicon": 1, + "id": "xyz.effem.notification.updateSeen", + "defs": { + "main": { + "type": "procedure", + "description": "Advances the viewer's seen-watermark. Notifications older than seenAt render as already-read.", + "input": { + "encoding": "application/json", + "schema": { + "type": "object", + "required": ["seenAt"], + "properties": { + "seenAt": { "type": "string", "format": "datetime" } + } + } + } + } + } +} -- 2.51.2