diff --git a/docs/PRD_AUTHOR_OWNED_POSTS.md b/docs/PRD_AUTHOR_OWNED_POSTS.md index df9c759..0a3dcbf 100644 --- a/docs/PRD_AUTHOR_OWNED_POSTS.md +++ b/docs/PRD_AUTHOR_OWNED_POSTS.md @@ -38,7 +38,18 @@ author-supplied created_at, delete-to-evade); per-origin-PDS quota explicitly deferred to Beta. Rev 2.6 (2026-08-08): task-3 second-opinion — fingerprint normalized to resolved-DID scope, release decoupled from request context, admission wiring -fail-loud, ActorClass fail-closed.** +fail-loud, ActorClass fail-closed. +Rev 2.7 (2026-08-08): task-5 plan review — post.getStatus pulled forward into +task 5 as the T2 observation surface (unauthenticated; mild disclosure of +rejected-post status accepted, owner-flagged); hosted-community detection = +community credential presence, NEVER hosted_by_did (attacker-controlled for +firehose-indexed communities); deleted_accounts marker table (migration 036 — +account deletion previously left no marker, so swept admissions could be +recreated by replayed events); author-delete resurrection loop closed (driver +excludes tombstoned posts; decider refuses them); §5.1 keeps the deprecated +community.post collection subscribed until task 8's drain; §9's T2 list +re-scoped — accepted-state arcs prove at T1 (no T2 community holds +credentials), T2 contracts assert consumer semantics via getStatus.** **Supersedes** the write-path architecture in `docs/federation-prd.md`: that document solves cross-instance posting by service-auth-forwarding the write to diff --git a/internal/api/handlers/post/getstatus.go b/internal/api/handlers/post/getstatus.go new file mode 100644 index 0000000..21996f6 --- /dev/null +++ b/internal/api/handlers/post/getstatus.go @@ -0,0 +1,36 @@ +package post + +import ( + "net/http" + + "Coves/internal/core/posts" +) + +// RED STUB (task 5, cycle 1). Signature only — HandleGetStatus writes nothing, +// so every assertion in getstatus_integration_test.go fails on the response +// rather than on a missing symbol. The implementation is GREEN's. + +// GetStatusHandler serves social.coves.community.post.getStatus: one +// community's decision about one post (docs/PRD_AUTHOR_OWNED_POSTS.md §3.4). +// +// UNAUTHENTICATED, deliberately, and the trade is recorded rather than hidden. +// The caller with the strongest need is an author on a DIFFERENT server whose +// post is pending on this one (§7): they have no account here, so there is no +// session to require, and service-auth is Beta scope. The cost is that a +// rejected post's status is mildly disclosed to anyone who can name its URI — +// accepted by the owner in PRD rev 2.7. That the route carries no auth +// middleware is declared in internal/api/routes/registration_test.go, which is +// the only place the whole HTTP surface is enumerated. +type GetStatusHandler struct { + service posts.StatusService +} + +// NewGetStatusHandler creates a new getStatus handler. +func NewGetStatusHandler(service posts.StatusService) *GetStatusHandler { + return &GetStatusHandler{service: service} +} + +// HandleGetStatus handles +// GET /xrpc/social.coves.community.post.getStatus?post=at://...&community=did:... +func (h *GetStatusHandler) HandleGetStatus(w http.ResponseWriter, r *http.Request) { +} diff --git a/internal/api/handlers/post/getstatus_integration_test.go b/internal/api/handlers/post/getstatus_integration_test.go new file mode 100644 index 0000000..bde4523 --- /dev/null +++ b/internal/api/handlers/post/getstatus_integration_test.go @@ -0,0 +1,433 @@ +//go:build integration + +package post_test + +import ( + "context" + "database/sql" + "encoding/json" + "net/http" + "net/http/httptest" + "net/url" + "testing" + "time" + + "Coves/internal/api/handlers/post" + "Coves/internal/core/posts" + "Coves/internal/db/postgres" + "Coves/tests/fixtures" + "Coves/tests/testkit" + + _ "github.com/lib/pq" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// What social.coves.community.post.getStatus answers, over the real admissions +// table (docs/PRD_AUTHOR_OWNED_POSTS.md §3.4, pulled into task 5 by rev 2.7). +// +// This is at T1 rather than in a handler unit test with a fake service, and the +// reason is the whole point of the endpoint. getStatus exists so that a client +// can learn a decision that lives NOWHERE ELSE: a rejection writes no community +// record (§3.3), so unlike acceptance and removal there is no firehose event, no +// repo record, and no other endpoint carrying it. The only source of truth is a +// row in community_post_admissions, and a test that faked the service would +// prove the JSON shape while leaving open the question that matters — whether +// the five statuses the table can actually hold each come out as something a +// client can act on. So every case below seeds its state through the real +// repository's own mutations, in the same way the consumer will. +// +// The pipeline tier then uses this endpoint as its observation surface for the +// consumer contracts (§9, rev 2.7): no T2 community holds credentials, so the +// accepted-state arcs prove here and at the repository, and T2 asserts consumer +// semantics by asking getStatus what the AppView concluded. + +// statusStack is the getStatus endpoint over the real admissions store, plus the +// repository the tests seed through. +type statusStack struct { + handler *post.GetStatusHandler + admissions posts.AdmissionRepository +} + +func newStatusStack(db *sql.DB) statusStack { + admissions := postgres.NewAdmissionRepository(db) + return statusStack{ + handler: post.NewGetStatusHandler(posts.NewStatusService(admissions)), + admissions: admissions, + } +} + +// statusSubject is one (community, post) pair with the post row seeded in the +// AUTHOR's repo shape — at:///social.coves.community.postv2/ — +// because that is where a post lives now (§3.1) and a URI in the old +// community-repo shape would exercise a normalization path this endpoint must +// not have. +type statusSubject struct { + CommunityDID string + AuthorDID string + PostURI string +} + +func newStatusSubject(t *testing.T, db *sql.DB) statusSubject { + t.Helper() + + ctx := context.Background() + name := testkit.UniqueIDWithPrefix(t, "stat") + + communityDID, err := fixtures.Community(ctx, db, name, "owner"+name) + require.NoErrorf(t, err, "seeding community %s", name) + + authorDID := fixtures.DID(testkit.UniqueID(t)) + rkey := testkit.TID() + postURI := "at://" + authorDID + "/social.coves.community.postv2/" + rkey + + _, err = db.ExecContext(ctx, ` + INSERT INTO posts (uri, cid, rkey, author_did, community_did, title, created_at) + VALUES ($1, $2, $3, $4, $5, $6, NOW()) + `, postURI, "bafyreistatusseed", rkey, authorDID, communityDID, "a post whose status someone is asking about") + require.NoError(t, err, "seeding the post row the admission is about") + + return statusSubject{CommunityDID: communityDID, AuthorDID: authorDID, PostURI: postURI} +} + +// getStatus drives the handler with NO Authorization header, which is half the +// contract: the caller this endpoint is built for is an author on another +// server with no account here (§7). That the ROUTE carries no auth middleware +// is declared in internal/api/routes/registration_test.go; what is proven here +// is that the handler itself serves a complete answer to an anonymous caller +// rather than degrading to an empty or partial one. +func getStatus(t *testing.T, h *post.GetStatusHandler, postURI, communityDID string) *httptest.ResponseRecorder { + t.Helper() + + target := "/xrpc/social.coves.community.post.getStatus?post=" + + url.QueryEscape(postURI) + "&community=" + url.QueryEscape(communityDID) + rec := httptest.NewRecorder() + h.HandleGetStatus(rec, httptest.NewRequest(http.MethodGet, target, nil)) + return rec +} + +// decodeStatus reads a 200 body as a generic map, so that a MISSING optional +// field and a field present-but-null are distinguishable. That distinction is +// load-bearing: the lexicon marks decisionCode, decisionAt and acceptanceUri +// optional, and a client that meets `"decisionCode": null` on a pending post has +// been handed a decision that does not exist. +func decodeStatus(t *testing.T, rec *httptest.ResponseRecorder) map[string]interface{} { + t.Helper() + + require.Equalf(t, http.StatusOK, rec.Code, "status = %d, want 200 (body: %s)", rec.Code, rec.Body.String()) + assert.Equal(t, "application/json", rec.Header().Get("Content-Type")) + + var body map[string]interface{} + require.NoErrorf(t, json.Unmarshal(rec.Body.Bytes(), &body), + "decoding the getStatus response (body: %q)", rec.Body.String()) + return body +} + +func TestGetStatus_Pending(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + stack := newStatusStack(db) + subject := newStatusSubject(t, db) + + _, err := stack.admissions.UpsertPending(ctx, posts.UpsertPendingCommand{ + CommunityDID: subject.CommunityDID, + PostURI: subject.PostURI, + EvaluatedCID: "bafyreipendingcontent", + }) + require.NoError(t, err) + + body := decodeStatus(t, getStatus(t, stack.handler, subject.PostURI, subject.CommunityDID)) + + assert.Equal(t, "pending", body["status"]) + + // A pending post has been decided about by nobody, and the response must + // say exactly that by carrying no decision fields at all. A client polling + // this endpoint for the accepted transition (§7's UX) reads their presence + // as "the wait is over". + assert.NotContains(t, body, "decisionCode", "a pending post carries no decision") + assert.NotContains(t, body, "decisionAt", "a pending post carries no decision time") + assert.NotContains(t, body, "acceptanceUri", "a pending post has no acceptance record to point at") +} + +func TestGetStatus_AcceptedNamesTheAcceptanceRecord(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + stack := newStatusStack(db) + subject := newStatusSubject(t, db) + + const acceptedCID = "bafyreiacceptedcontent" + rkey := testkit.TID() + acceptanceURI := "at://" + subject.CommunityDID + "/social.coves.community.acceptance/" + rkey + + _, err := stack.admissions.UpsertPending(ctx, posts.UpsertPendingCommand{ + CommunityDID: subject.CommunityDID, + PostURI: subject.PostURI, + EvaluatedCID: acceptedCID, + }) + require.NoError(t, err) + + result, err := stack.admissions.ApplyAcceptance(ctx, posts.ApplyAcceptanceCommand{ + CommunityDID: subject.CommunityDID, + PostURI: subject.PostURI, + AcceptanceURI: acceptanceURI, + AcceptanceRkey: rkey, + PinnedCID: acceptedCID, + Watermark: posts.CommunityWatermark{Rev: testkit.TID()}, + }) + require.NoError(t, err) + require.Equal(t, posts.AdmissionApplied, result.Outcome, "fixture: the acceptance must have applied") + + body := decodeStatus(t, getStatus(t, stack.handler, subject.PostURI, subject.CommunityDID)) + + assert.Equal(t, "accepted", body["status"]) + + // The acceptance URI is what turns this answer from a claim into something + // verifiable: the caller can go read the community's signed attestation + // instead of trusting this AppView's summary of it. Omitting it would make + // getStatus the authority on a fact it is only reporting. + assert.Equal(t, acceptanceURI, body["acceptanceUri"], + "an accepted post must name its acceptance record so the caller can read the signed attestation") + assert.NotContains(t, body, "decisionCode", "acceptance is not a refusal and carries no code") +} + +func TestGetStatus_RejectedCarriesTheReasonAndWhen(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + stack := newStatusStack(db) + subject := newStatusSubject(t, db) + + const judgedCID = "bafyreirejectedcontent" + _, err := stack.admissions.UpsertPending(ctx, posts.UpsertPendingCommand{ + CommunityDID: subject.CommunityDID, + PostURI: subject.PostURI, + EvaluatedCID: judgedCID, + }) + require.NoError(t, err) + + before := time.Now().Add(-time.Minute) + result, err := stack.admissions.RecordRejection(ctx, posts.RecordRejectionCommand{ + CommunityDID: subject.CommunityDID, + PostURI: subject.PostURI, + DecisionCode: string(posts.DecisionRateLimitExceeded), + JudgedCID: judgedCID, + Redrivable: false, + }) + require.NoError(t, err) + require.Equal(t, posts.AdmissionApplied, result.Outcome, "fixture: the rejection must have applied") + + body := decodeStatus(t, getStatus(t, stack.handler, subject.PostURI, subject.CommunityDID)) + + assert.Equal(t, "rejected", body["status"]) + + // THIS is the case the endpoint exists for. A rejection writes no community + // record (§3.3), so there is no acceptance to read, no removal to read, and + // nothing on the firehose — an author whose post vanished has no other way + // to learn it was refused, or why. A `rejected` with no code is a status + // that tells them only that asking was pointless. + assert.Equal(t, string(posts.DecisionRateLimitExceeded), body["decisionCode"], + "a rejection is invisible everywhere else, so the code is the only explanation the author will ever get") + + decisionAt, ok := body["decisionAt"].(string) + require.Truef(t, ok, "decisionAt must be present and a string on a rejected post; got %#v", body["decisionAt"]) + parsed, err := time.Parse(time.RFC3339, decisionAt) + require.NoErrorf(t, err, "decisionAt %q must be an RFC 3339 datetime, per the lexicon's datetime format", decisionAt) + assert.Truef(t, parsed.After(before), "decisionAt (%s) must record when the decision was made", parsed) + + assert.NotContains(t, body, "acceptanceUri", "a rejected post was never accepted, so there is no record to name") +} + +func TestGetStatus_RemovedCarriesTheModerationCode(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + stack := newStatusStack(db) + subject := newStatusSubject(t, db) + + _, err := stack.admissions.UpsertPending(ctx, posts.UpsertPendingCommand{ + CommunityDID: subject.CommunityDID, + PostURI: subject.PostURI, + EvaluatedCID: "bafyreiremovedcontent", + }) + require.NoError(t, err) + + result, err := stack.admissions.ApplyRemoval(ctx, posts.ApplyRemovalCommand{ + CommunityDID: subject.CommunityDID, + PostURI: subject.PostURI, + DecisionCode: string(posts.DecisionRuleViolation), + Watermark: posts.CommunityWatermark{Rev: testkit.TID()}, + }) + require.NoError(t, err) + require.Equal(t, posts.AdmissionApplied, result.Outcome, "fixture: the removal must have applied") + + body := decodeStatus(t, getStatus(t, stack.handler, subject.PostURI, subject.CommunityDID)) + + assert.Equal(t, "removed", body["status"]) + assert.Equal(t, string(posts.DecisionRuleViolation), body["decisionCode"], + "a removal's code is what #removedPost renders to the author; a removal without one is an unexplained moderation act") + assert.NotContains(t, body, "acceptanceUri", + "a removal deletes the acceptance in the same commit (§3.3), so naming one would point at a record that no longer exists") +} + +func TestGetStatus_PendingReacceptance(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + stack := newStatusStack(db) + subject := newStatusSubject(t, db) + + const originalCID = "bafyreioriginalcontent" + rkey := testkit.TID() + acceptanceURI := "at://" + subject.CommunityDID + "/social.coves.community.acceptance/" + rkey + + _, err := stack.admissions.UpsertPending(ctx, posts.UpsertPendingCommand{ + CommunityDID: subject.CommunityDID, + PostURI: subject.PostURI, + EvaluatedCID: originalCID, + }) + require.NoError(t, err) + _, err = stack.admissions.ApplyAcceptance(ctx, posts.ApplyAcceptanceCommand{ + CommunityDID: subject.CommunityDID, + PostURI: subject.PostURI, + AcceptanceURI: acceptanceURI, + AcceptanceRkey: rkey, + PinnedCID: originalCID, + Watermark: posts.CommunityWatermark{Rev: testkit.TID()}, + }) + require.NoError(t, err) + + // The author edits. The standing acceptance now pins content that is no + // longer current, and §5.5 forbids rendering the new CID under it. + _, err = stack.admissions.UpsertPending(ctx, posts.UpsertPendingCommand{ + CommunityDID: subject.CommunityDID, + PostURI: subject.PostURI, + EvaluatedCID: "bafyreieditedcontent", + }) + require.NoError(t, err) + + body := decodeStatus(t, getStatus(t, stack.handler, subject.PostURI, subject.CommunityDID)) + + // The status is reported verbatim rather than collapsed into "pending". An + // author who edited an accepted post and is shown plain `pending` cannot + // tell that from a post that was never accepted at all, and the two have + // completely different next steps. + assert.Equal(t, "pending_reacceptance", body["status"], + "pending_reacceptance must not be flattened into pending: the author needs to know their edit un-published an accepted post") +} + +func TestGetStatus_UnknownSubjectIsNotFound(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + stack := newStatusStack(db) + subject := newStatusSubject(t, db) + + // The post and the community both exist; what does not exist is a decision. + // That is the ordinary state of a post the community has never been offered, + // and it must be answered as not-found rather than invented as `pending` — + // pending is a promise that someone is going to decide. + rec := getStatus(t, stack.handler, subject.PostURI, subject.CommunityDID) + + require.Equalf(t, http.StatusNotFound, rec.Code, + "a subject with no admission row must be 404, not a fabricated status (body: %s)", rec.Body.String()) + + var body map[string]interface{} + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &body)) + assert.Equal(t, "NotFound", body["error"], + "the XRPC error name is what a client switches on; the shared mapper spells post-not-found as NotFound (errors.go)") +} + +func TestGetStatus_RequiresBothHalvesOfTheSubject(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + stack := newStatusStack(db) + subject := newStatusSubject(t, db) + + // A post carries independent decisions from several communities (§2), so + // "the status of this post" is not a question with one answer. A request + // missing either half has to be refused rather than answered about whichever + // row happens to be found first. + cases := []struct { + name string + target string + }{ + {"no post", "/xrpc/social.coves.community.post.getStatus?community=" + url.QueryEscape(subject.CommunityDID)}, + {"no community", "/xrpc/social.coves.community.post.getStatus?post=" + url.QueryEscape(subject.PostURI)}, + {"neither", "/xrpc/social.coves.community.post.getStatus"}, + {"empty post", "/xrpc/social.coves.community.post.getStatus?post=&community=" + url.QueryEscape(subject.CommunityDID)}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + rec := httptest.NewRecorder() + stack.handler.HandleGetStatus(rec, httptest.NewRequest(http.MethodGet, tc.target, nil)) + assert.Equalf(t, http.StatusBadRequest, rec.Code, + "an incomplete subject must be 400 (body: %s)", rec.Body.String()) + }) + } +} + +func TestGetStatus_ScopesTheAnswerToTheNamedCommunity(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + stack := newStatusStack(db) + + // One post, two communities, opposite decisions. This is the fork case the + // per-(community, post) key exists for (§2, §6.1), and it is the sharpest + // available proof that the community parameter is actually part of the + // lookup rather than decoration on a post-scoped query. + accepting := newStatusSubject(t, db) + + otherName := testkit.UniqueIDWithPrefix(t, "statfork") + forkDID, err := fixtures.Community(ctx, db, otherName, "owner"+otherName) + require.NoError(t, err) + + const cid = "bafyreitwocommunities" + rkey := testkit.TID() + acceptanceURI := "at://" + accepting.CommunityDID + "/social.coves.community.acceptance/" + rkey + + for _, communityDID := range []string{accepting.CommunityDID, forkDID} { + _, err = stack.admissions.UpsertPending(ctx, posts.UpsertPendingCommand{ + CommunityDID: communityDID, + PostURI: accepting.PostURI, + EvaluatedCID: cid, + }) + require.NoError(t, err) + } + + _, err = stack.admissions.ApplyAcceptance(ctx, posts.ApplyAcceptanceCommand{ + CommunityDID: accepting.CommunityDID, + PostURI: accepting.PostURI, + AcceptanceURI: acceptanceURI, + AcceptanceRkey: rkey, + PinnedCID: cid, + Watermark: posts.CommunityWatermark{Rev: testkit.TID()}, + }) + require.NoError(t, err) + + _, err = stack.admissions.ApplyRemoval(ctx, posts.ApplyRemovalCommand{ + CommunityDID: forkDID, + PostURI: accepting.PostURI, + DecisionCode: string(posts.DecisionOffTopic), + Watermark: posts.CommunityWatermark{Rev: testkit.TID()}, + }) + require.NoError(t, err) + + accepted := decodeStatus(t, getStatus(t, stack.handler, accepting.PostURI, accepting.CommunityDID)) + assert.Equal(t, "accepted", accepted["status"]) + + removed := decodeStatus(t, getStatus(t, stack.handler, accepting.PostURI, forkDID)) + assert.Equal(t, "removed", removed["status"], + "the same post is accepted in one community and removed in another; an answer that ignored the community parameter would report one of them everywhere") + assert.Equal(t, string(posts.DecisionOffTopic), removed["decisionCode"]) +} diff --git a/internal/api/routes/registration_test.go b/internal/api/routes/registration_test.go index 5a4f2f5..5815685 100644 --- a/internal/api/routes/registration_test.go +++ b/internal/api/routes/registration_test.go @@ -182,6 +182,17 @@ var declaredRoutes = []declaredRoute{ {http.MethodPost, "/xrpc/social.coves.community.post.create", authRequired, 0, false}, {http.MethodPost, "/xrpc/social.coves.community.post.delete", authRequired, 0, false}, {http.MethodGet, "/xrpc/social.coves.community.post.get", authOptional, 0, false}, + // getStatus takes no auth at all, unlike post.get beside it, and the + // asymmetry is the decision this line exists to hold. The caller it is for + // is an author on ANOTHER server asking this host whether it accepted their + // post (PRD §7): they have no account here, so there is no session to + // require and no viewer state to personalise. A rejection is AppView-local + // and writes no community record (§3.3), so this endpoint is the only way + // that answer is reachable at all. The accepted cost, recorded in PRD rev + // 2.7: anyone who can name a post URI learns its status in a community. + // Adding OptionalAuth here would be harmless; adding RequireAuth would make + // the cross-server case unanswerable, which is why it is declared. + {http.MethodGet, "/xrpc/social.coves.community.post.getStatus", authNone, 0, false}, // RegisterVoteRoutes — social.coves.feed.vote.* {http.MethodPost, "/xrpc/social.coves.feed.vote.create", authRequired, 0, false}, diff --git a/internal/atproto/jetstream/acceptance_consumer_test.go b/internal/atproto/jetstream/acceptance_consumer_test.go new file mode 100644 index 0000000..534f7d5 --- /dev/null +++ b/internal/atproto/jetstream/acceptance_consumer_test.go @@ -0,0 +1,544 @@ +//go:build integration + +package jetstream + +import ( + "context" + "database/sql" + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + "Coves/internal/atproto/identity" + "Coves/internal/core/posts" + "Coves/internal/core/users" + "Coves/internal/db/postgres" + "Coves/tests/testkit" + + _ "github.com/lib/pq" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// Ingesting the two records a COMMUNITY writes about a post: +// social.coves.community.acceptance and social.coves.community.removal +// (docs/PRD_AUTHOR_OWNED_POSTS.md §5.4, §5.2). +// +// These are the records that decide what a community shows. A post claiming a +// community and lacking an acceptance is never rendered in it (§2), so an +// acceptance event is the moment a post becomes visible and a removal event is +// the moment it stops — which makes two properties non-negotiable here: +// +// - THE REPO DID IS THE COMMUNITY, and it must be a community this AppView has +// indexed. Nothing else in the event says which community decided. An +// acceptance from an arbitrary repo, taken at face value, is a stranger +// publishing into someone else's feed. +// - THE PINNED CID IS PART OF THE DECISION. An acceptance is a strongRef, and +// agreeing to at://x/postv2/y is not agreeing to whatever that URI holds +// later. A consumer that dropped the CID comparison would let an author edit +// content past moderation after approval. +// +// AND CONVERGENCE IS NOT FREE. An acceptance can arrive for a post this AppView +// has never seen, and redrive alone cannot fix that: bounded retries cannot +// manufacture a post event that a relay-coverage gap will never deliver. §5.4's +// direct fetch is the mechanism, and the second half of this file is about the +// ways that fetch must refuse rather than the way it succeeds — it is an +// outbound request whose destination is chosen by a stranger's record. + +const ( + accPrefix = "did:plc:acc" + accCommunity = accPrefix + "community" + accAuthor = accPrefix + "author" + accOutsider = accPrefix + "outsider" +) + +// accFixture is a consumer wired for community-repo events, plus the stores the +// assertions read. +type accFixture struct { + consumer *PostEventConsumer + admissions posts.AdmissionRepository + db *sql.DB +} + +func newAccFixture(t *testing.T, db *sql.DB, opts ...PostEventConsumerOption) accFixture { + t.Helper() + + insertBridgedUser(t, db, accAuthor, "accauthor.test") + insertBridgedCommunity(t, db, accCommunity, "acccommunity.test", accAuthor) + + us := newMockUserService() + us.users[accAuthor] = &users.User{DID: accAuthor, Handle: "accauthor.test"} + + admissions := postgres.NewAdmissionRepository(db) + wired := append([]PostEventConsumerOption{ + WithAdmissions(admissions), + WithDeletedAccounts(postgres.NewDeletedAccountRepository(db)), + }, opts...) + + return accFixture{ + consumer: NewPostEventConsumer( + postgres.NewPostRepository(db), + postgres.NewCommunityRepository(db), + us, + db, + wired..., + ), + admissions: admissions, + db: db, + } +} + +// accPostURI is the author-repo URI an acceptance points at. +func accPostURI(rkey string) string { return "at://" + accAuthor + "/" + PostV2Collection + "/" + rkey } + +// indexPV2 puts a post in the index the ordinary way — through the consumer — +// so these tests start from a state the pipeline actually produces. +func (f accFixture) indexPV2(t *testing.T, rkey, cid string, timeUS int64) string { + t.Helper() + uri := accPostURI(rkey) + require.NoError(t, f.consumer.HandleEvent(context.Background(), pv2Event( + accAuthor, "create", rkey, testkit.TID(), cid, timeUS, + pv2Record(accCommunity, "a post awaiting a decision", "body"), + )), "fixture: indexing the subject post") + return uri +} + +// acceptanceEvent builds a community-repo acceptance commit. +// +// The rkey is derived from the subject rather than invented, because that is +// what the writers do (§3.2): one post has exactly one acceptance rkey per +// community, forever, which is what makes three independent writers converge on +// putRecord of the same record instead of allocating duplicate TIDs. +func acceptanceEvent(communityDID, postURI, pinnedCID, rev string, timeUS int64) *JetstreamEvent { + return revCommitEvent(communityDID, posts.AcceptanceCollection, "create", + posts.SubjectRkey(postURI), rev, "bafyreiacceptancerecord", timeUS, + map[string]interface{}{ + "$type": posts.AcceptanceCollection, + "subject": map[string]interface{}{"uri": postURI, "cid": pinnedCID}, + "createdAt": "2026-03-01T00:00:00Z", + }) +} + +// removalEvent builds a community-repo removal commit. +func removalEvent(communityDID, postURI, pinnedCID, code, rev string, timeUS int64) *JetstreamEvent { + return revCommitEvent(communityDID, posts.RemovalCollection, "create", + posts.SubjectRkey(postURI), rev, "bafyreiremovalrecord", timeUS, + map[string]interface{}{ + "$type": posts.RemovalCollection, + "subject": map[string]interface{}{"uri": postURI, "cid": pinnedCID}, + "code": code, + "createdAt": "2026-03-01T00:00:00Z", + }) +} + +func TestAcceptanceConsumer_MatchingCID_AcceptsAndStampsTheWatermark(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + f := newAccFixture(t, db) + ctx := context.Background() + base := time.Now().UnixMicro() + + const cid = "bafyreiaccmatch" + uri := f.indexPV2(t, "accmatch", cid, base) + + rev := testkit.TID() + require.NoError(t, f.consumer.HandleEvent(ctx, acceptanceEvent(accCommunity, uri, cid, rev, base+1_000_000))) + + admission, err := f.admissions.Get(ctx, accCommunity, uri) + require.NoError(t, err) + require.NotNil(t, admission) + + assert.Equal(t, posts.AdmissionStatusAccepted, admission.Status, + "an acceptance pinning the CID the AppView has indexed is the community agreeing to exactly this content") + assertNullableStringPV2(t, cid, admission.AcceptedCID, "accepted_cid must be the CID the acceptance pinned") + assertNullableStringPV2(t, posts.SubjectRkey(uri), admission.AcceptanceRkey, + "the acceptance rkey is deterministic from the subject; storing a different one breaks the one-record-per-subject convergence the writers rely on") + + // The §5.2 tuple, and specifically its second half. The rank is derived from + // the OPERATION by the repository, never taken from the wire: a put ranks + // above a delete so that the removal commit {acceptance-delete, removal-put} + // converges on removed and the restore commit {removal-delete, + // acceptance-put} converges on accepted, whichever half the consumer sees + // first. A caller-supplied rank would let one mislabelled event reorder a + // commit permanently. + require.NotNil(t, admission.LastCommunityEvent, "a community event must stamp the subject-scoped watermark") + assert.Equal(t, rev, admission.LastCommunityEvent.Rev, + "the watermark rev must be the COMMIT's rev — it is the only clock that orders acceptance against removal, since they are different record URIs about the same subject") + assert.Equal(t, posts.CommunityOpPut, admission.LastCommunityEvent.OpRank, + "a record write ranks as a put; the rank is the operation's kind, derived repo-side") +} + +func TestAcceptanceConsumer_ReplayIsANoOpAndLeavesTheRowUntouched(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + f := newAccFixture(t, db) + ctx := context.Background() + base := time.Now().UnixMicro() + + const cid = "bafyreiaccreplay" + uri := f.indexPV2(t, "accreplay", cid, base) + + // The redelivery is guaranteed, not hypothetical: the connector rewinds its + // cursor after every reconnect, the AppView consumes overlapping feeds, and + // the dead-letter redriver replays. This exact commit WILL arrive twice. + event := acceptanceEvent(accCommunity, uri, cid, testkit.TID(), base+1_000_000) + require.NoError(t, f.consumer.HandleEvent(ctx, event)) + + before := readAdmissionRow(t, db, accCommunity, uri) + + require.NoError(t, f.consumer.HandleEvent(ctx, event), + "an equal watermark is a replay, and a replay is a no-op — not an error the connector logs as a failure") + + after := readAdmissionRow(t, db, accCommunity, uri) + assert.Equal(t, before, after, + "a replayed acceptance must leave the row byte-identical. Re-stamping decision_at or updated_at would make the moderation audit trail a function of how many feeds happened to carry the event") +} + +func TestAcceptanceConsumer_FromANonCommunityRepo_IsRefusedAndRedrivable(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + f := newAccFixture(t, db) + ctx := context.Background() + base := time.Now().UnixMicro() + + const cid = "bafyreiaccoutsider" + uri := f.indexPV2(t, "accoutsider", cid, base) + + // accOutsider is a DID with no communities row. Nothing in the record says + // which community decided — the repo IS the claim — so accepting this would + // let anyone with a PDS publish into any feed by writing a record that names + // someone else's post. + err := f.consumer.HandleEvent(ctx, acceptanceEvent(accOutsider, uri, cid, testkit.TID(), base+1_000_000)) + require.Error(t, err, "an acceptance from a repo that is not an indexed community must not be applied") + + // Transient, not permanent, and the reason is delivery order rather than + // leniency: BigSky preserves order within a repo, not across repos, so a + // community's first acceptance can genuinely outrun its own profile event. + // Marking this permanent would spend the redrive budget that resolves the + // race and discard a legitimate decision. + assert.NotErrorIs(t, err, ErrPermanentEvent, + "an unindexed community repo is an ordering failure; the redrive resolves it once the community profile arrives") + + assert.Zero(t, countRows(t, db, + `SELECT count(*) FROM community_post_admissions WHERE post_uri = $1 AND community_did = $2`, uri, accOutsider), + "the refused acceptance must not have opened an admission row for the outsider repo") +} + +func TestRemovalConsumer_RemovesWithTheCodeItCarries(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + f := newAccFixture(t, db) + ctx := context.Background() + base := time.Now().UnixMicro() + + const cid = "bafyreiaccremove" + uri := f.indexPV2(t, "accremove", cid, base) + + revs := increasingTIDs(t, 2) + require.NoError(t, f.consumer.HandleEvent(ctx, acceptanceEvent(accCommunity, uri, cid, revs[0], base+1_000_000))) + require.NoError(t, f.consumer.HandleEvent(ctx, removalEvent( + accCommunity, uri, cid, string(posts.DecisionRuleViolation), revs[1], base+2_000_000))) + + admission, err := f.admissions.Get(ctx, accCommunity, uri) + require.NoError(t, err) + require.NotNil(t, admission) + + assert.Equal(t, posts.AdmissionStatusRemoved, admission.Status) + assertNullableStringPV2(t, string(posts.DecisionRuleViolation), admission.DecisionCode, + "the removal's code is what #removedPost renders to the author; dropping it turns a moderation act into an unexplained disappearance") + assert.NotNil(t, admission.DecisionAt, "a removal must record when it happened") +} + +func TestRemovalConsumer_PreemptiveRemovalCreatesTheRow(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + f := newAccFixture(t, db) + ctx := context.Background() + base := time.Now().UnixMicro() + + const cid = "bafyreiaccpreempt" + uri := f.indexPV2(t, "accpreempt", cid, base) + + // A removal with no prior acceptance is VALID (§5.4). A community that has + // decided in advance about a post — an author it is about to ban, content it + // has already seen elsewhere — must be able to say so, and a consumer that + // required an acceptance first would drop exactly the decisions a community + // most wants to make early. + require.NoError(t, f.consumer.HandleEvent(ctx, removalEvent( + accCommunity, uri, cid, string(posts.DecisionSpam), testkit.TID(), base+1_000_000))) + + admission, err := f.admissions.Get(ctx, accCommunity, uri) + require.NoErrorf(t, err, "a pre-emptive removal must create the admission row it decides") + require.NotNil(t, admission) + assert.Equal(t, posts.AdmissionStatusRemoved, admission.Status) + assertNullableStringPV2(t, string(posts.DecisionSpam), admission.DecisionCode, "decision_code") +} + +// --------------------------------------------------------------------------- +// §5.4 direct fetch: acceptance before post +// --------------------------------------------------------------------------- + +// fakeAuthorPDS is an httptest server answering com.atproto.repo.getRecord for +// the author's postv2 record, standing in for the PDS a DID document points at. +// +// It asserts the request shape as it serves, because the fetch is the one place +// the AppView reads a record without the firehose: a getRecord aimed at the +// wrong repo or collection would return someone else's record, and the CID check +// downstream would happily verify it. +func fakeAuthorPDS(t *testing.T, expectRepo string, handler http.HandlerFunc) *httptest.Server { + t.Helper() + return httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + assert.Equal(t, "/xrpc/com.atproto.repo.getRecord", r.URL.Path) + assert.Equal(t, expectRepo, r.URL.Query().Get("repo")) + assert.Equal(t, PostV2Collection, r.URL.Query().Get("collection")) + handler(w, r) + })) +} + +// serveRecord writes a getRecord response body. +func serveRecord(t *testing.T, w http.ResponseWriter, uri, cid string, value map[string]interface{}) { + t.Helper() + w.Header().Set("Content-Type", "application/json") + require.NoError(t, json.NewEncoder(w).Encode(map[string]interface{}{ + "uri": uri, "cid": cid, "value": value, + })) +} + +// newFetcherAt builds a DirectPostFetcher pointed at srv, with the SSRF guard +// stood down. +// +// Standing it down is REQUIRED and is the whole reason the seam exists: +// httptest listens on loopback, which the guard blocks by design, so a fetcher +// that could not be relaxed could not be tested against a fake PDS at all. The +// guard's default is asserted separately, and behaviourally, in +// TestDirectPostFetcher_RefusesAPrivateHostByDefault. +func newFetcherAt(t *testing.T, authorDID, pdsURL string) *DirectPostFetcher { + t.Helper() + fetcher := NewDirectPostFetcher(&mockIdentityResolverForUser{ + identities: map[string]*identity.Identity{ + authorDID: {DID: authorDID, Handle: "accauthor.test", PDSURL: pdsURL}, + }, + }) + fetcher.allowPrivateHosts = true + return fetcher +} + +func TestAcceptanceConsumer_UnindexedPost_IsFetchedDirectlyAndAccepted(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + base := time.Now().UnixMicro() + + const cid = "bafyreiaccfetched" + rkey := "accfetch" + uri := accPostURI(rkey) + + srv := fakeAuthorPDS(t, accAuthor, func(w http.ResponseWriter, r *http.Request) { + serveRecord(t, w, uri, cid, pv2Record(accCommunity, "fetched straight from the PDS", "body")) + }) + defer srv.Close() + + f := newAccFixture(t, db, WithPostRecordFetcher(newFetcherAt(t, accAuthor, srv.URL))) + + // The post was NEVER indexed — no create event ever arrived, and none ever + // will if the relay does not crawl the author's PDS. This is the case §5.4 + // says redrive cannot solve: retries cannot manufacture an event nobody is + // going to send. Without the fetch, convergence requires full relay + // coverage, which is a bet rather than a guarantee. + require.NoError(t, f.consumer.HandleEvent(ctx, acceptanceEvent(accCommunity, uri, cid, testkit.TID(), base))) + + authorDID, communityDID, storedCID, _, _ := readPV2Post(t, db, uri) + assert.Equal(t, accAuthor, authorDID, "the fetched post is attributed to the repo it was read from") + assert.Equal(t, accCommunity, communityDID) + assert.Equal(t, cid, storedCID) + + admission, err := f.admissions.Get(ctx, accCommunity, uri) + require.NoError(t, err) + require.NotNil(t, admission) + assert.Equal(t, posts.AdmissionStatusAccepted, admission.Status, + "the fetch exists so the acceptance can be APPLIED; indexing the post and leaving the decision pending would solve half the problem and leave the post invisible") +} + +func TestAcceptanceConsumer_FetchedCIDMismatch_IsPermanentlyRefused(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + base := time.Now().UnixMicro() + + rkey := "accmismatch" + uri := accPostURI(rkey) + + srv := fakeAuthorPDS(t, accAuthor, func(w http.ResponseWriter, r *http.Request) { + // The PDS serves the CURRENT version. The acceptance pins an older one. + serveRecord(t, w, uri, "bafyreiaccnowcurrent", pv2Record(accCommunity, "the version now at that rkey", "body")) + }) + defer srv.Close() + + f := newAccFixture(t, db, WithPostRecordFetcher(newFetcherAt(t, accAuthor, srv.URL))) + + err := f.consumer.HandleEvent(ctx, acceptanceEvent(accCommunity, uri, "bafyreiaccpinnedold", testkit.TID(), base)) + + // The CID check is what makes the fetch trustworthy at all. Without it the + // AppView indexes whatever the author's PDS chooses to serve under that + // rkey — the author (or whoever holds their keys) picks the content, and the + // community's signed acceptance is made to cover it retroactively. + require.Error(t, err, "a fetched record whose CID is not the one the acceptance pinned must never be indexed under that acceptance") + assert.ErrorIs(t, err, ErrPermanentEvent, + "the pinned version is gone from the repo and no retry brings it back; the connector must dead-letter this with its redrive budget already spent rather than re-fetching the same mismatch ten times") + + assert.Zero(t, countRows(t, db, `SELECT count(*) FROM posts WHERE uri = $1`, uri), + "the unverified record must not be indexed") +} + +func TestAcceptanceConsumer_FetchedRecordNamingAnotherCommunity_IsPermanentlyRefused(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + base := time.Now().UnixMicro() + + const cid = "bafyreiaccwrongcommunity" + rkey := "accwrongcomm" + uri := accPostURI(rkey) + + srv := fakeAuthorPDS(t, accAuthor, func(w http.ResponseWriter, r *http.Request) { + // The record was submitted to a DIFFERENT community. + serveRecord(t, w, uri, cid, pv2Record("did:plc:accsomewhereelse", "submitted elsewhere", "body")) + }) + defer srv.Close() + + f := newAccFixture(t, db, WithPostRecordFetcher(newFetcherAt(t, accAuthor, srv.URL))) + + err := f.consumer.HandleEvent(ctx, acceptanceEvent(accCommunity, uri, cid, testkit.TID(), base)) + + // Cross-community acceptance is the privileged fork/import flow, and §10.2 + // is explicit that it is deliberately NOT built: the data model supports it, + // the flow that exercises it is future scope. Until it exists, a community + // accepting a post that names someone else is a community pulling another + // community's content into its feed on its own say-so. + require.Error(t, err, "a community may not accept a post whose record names a different community — the fork/import flow is deliberately not built (§10.2)") + assert.ErrorIs(t, err, ErrPermanentEvent, + "the record's community field is immutable across updates (§3.1), so this can never become valid; retrying it is pure noise") + + assert.Zero(t, countRows(t, db, `SELECT count(*) FROM posts WHERE uri = $1`, uri), + "the refused record must not be indexed") +} + +func TestDirectPostFetcher_RefusesAnOversizedBody(t *testing.T) { + t.Parallel() + + uri := accPostURI("accoversized") + + // A post record has a lexicon-bounded size; a PDS streaming megabytes is + // either broken or hostile. The cap has to be enforced by the FETCHER + // because the DID document that chose this host is attacker-controlled — any + // stranger who writes an acceptance record picks where this request goes, + // and an unbounded read there is a memory-exhaustion primitive handed to the + // public. + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + _, _ = w.Write([]byte(`{"uri":"` + uri + `","cid":"bafyreiaccbig","value":{"$type":"` + + PostV2Collection + `","community":"` + accCommunity + `","createdAt":"2026-03-01T00:00:00Z","content":"`)) + chunk := strings.Repeat("A", 64*1024) + for i := 0; i < 64; i++ { // 4 MiB of content + if _, err := w.Write([]byte(chunk)); err != nil { + return + } + } + _, _ = w.Write([]byte(`"}}`)) + })) + defer srv.Close() + + fetched, err := newFetcherAt(t, accAuthor, srv.URL).FetchPost(context.Background(), uri) + require.Error(t, err, "a response past the size cap must be an error, not a truncated record parsed as if it were whole") + assert.Nil(t, fetched) +} + +func TestDirectPostFetcher_RefusesAPrivateHostByDefault(t *testing.T) { + t.Parallel() + + uri := accPostURI("accssrf") + + var reached bool + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + reached = true + serveRecord(t, w, uri, "bafyreiaccssrf", pv2Record(accCommunity, "should never be read", "body")) + })) + defer srv.Close() + + // The default constructor, with nothing stood down. httptest listens on + // loopback, which is precisely the class of address the guard blocks. + // + // This is asserted behaviourally rather than by reading the flag because the + // flag is not the protection — a fetcher that stored allowPrivateHosts and + // then built its client from a hardcoded false (or true) would pass a field + // check and fail this. + fetcher := NewDirectPostFetcher(&mockIdentityResolverForUser{ + identities: map[string]*identity.Identity{ + accAuthor: {DID: accAuthor, Handle: "accauthor.test", PDSURL: srv.URL}, + }, + }) + require.Falsef(t, fetcher.allowPrivateHosts, + "NewDirectPostFetcher must default to SSRF protection ON: the PDS this dials is named by a DID document anyone can publish") + + fetched, err := fetcher.FetchPost(context.Background(), uri) + require.Error(t, err, + "a PDS resolving to a private address must be refused: an acceptance record is public and unauthenticated input, so this fetch is a request forger pointed at whatever the AppView can reach on its own network") + assert.Nil(t, fetched) + assert.Falsef(t, reached, "the guard must refuse before the request is made, not after the server has already answered") +} + +// readAdmissionRow returns every mutable column of one admission row as a +// comparable value, so "the row did not change" can be asserted as a whole +// rather than field by field — a new column added later is covered without +// anyone remembering to extend an assertion. +func readAdmissionRow(t *testing.T, db *sql.DB, communityDID, postURI string) []interface{} { + t.Helper() + + var ( + status string + acceptanceURI, acceptanceRkey, accepted *string + decisionCode, evaluatedCID, rev *string + decisionAt *time.Time + opRank *int16 + redrivable bool + createdAt, updatedAt time.Time + ) + require.NoError(t, db.QueryRow(` + SELECT status, acceptance_uri, acceptance_rkey, accepted_cid, decision_code, decision_at, + evaluated_cid, redrivable, last_community_rev, last_community_op_rank, created_at, updated_at + FROM community_post_admissions WHERE community_did = $1 AND post_uri = $2 + `, communityDID, postURI).Scan( + &status, &acceptanceURI, &acceptanceRkey, &accepted, &decisionCode, &decisionAt, + &evaluatedCID, &redrivable, &rev, &opRank, &createdAt, &updatedAt, + )) + + deref := func(p *string) interface{} { + if p == nil { + return nil + } + return *p + } + var decidedAt interface{} + if decisionAt != nil { + decidedAt = decisionAt.UTC() + } + var rank interface{} + if opRank != nil { + rank = *opRank + } + return []interface{}{ + status, deref(acceptanceURI), deref(acceptanceRkey), deref(accepted), deref(decisionCode), + decidedAt, deref(evaluatedCID), redrivable, deref(rev), rank, createdAt.UTC(), updatedAt.UTC(), + } +} diff --git a/internal/atproto/jetstream/authorpost.go b/internal/atproto/jetstream/authorpost.go new file mode 100644 index 0000000..2c44009 --- /dev/null +++ b/internal/atproto/jetstream/authorpost.go @@ -0,0 +1,123 @@ +package jetstream + +import ( + "context" + "net/http" + + "Coves/internal/atproto/identity" + "Coves/internal/core/posts" +) + +// RED STUB (task 5, cycle 1). Declarations only — every function body here +// returns zero values, so the tests describing author-owned post ingestion +// compile and fail on their assertions rather than on missing symbols. The +// implementations, the HandleEvent dispatch for the three new collections, and +// the consumerWantedCollections entries are GREEN's. + +// PostV2Collection is the author-repo post record of +// docs/PRD_AUTHOR_OWNED_POSTS.md §3.1 — the §3.0 successor to the deprecated +// social.coves.community.post. +// +// The NSID is new rather than reused, and that is a safety property, not +// bookkeeping: a consumer built against the published community.post derives +// community = repo DID for that collection, so feeding it author-repo records +// under the same name would have it index authors as communities. A new NSID +// makes a stale consumer ignore the records entirely, which is the correct +// failure mode. +const PostV2Collection = "social.coves.community.postv2" + +// DeletedAccountLookup reports whether a DID names an account this AppView was +// asked to erase (migration 036, PRD rev 2.7). +// +// It exists because "no users row" stopped being an answer. Under author-owned +// posts an unknown author is a NORMAL state that must still index (§5.3), so +// the absence of a profile can no longer stand in for "this account is gone" — +// and without a marker, a redriven post event or a replayed acceptance quietly +// recreates the very rows the deletion swept. +// +// A lookup FAILURE must never be read as "not deleted". Failing open here means +// a database blip re-indexes an erased account's content, which is the one +// outcome a deletion is supposed to make impossible. +type DeletedAccountLookup interface { + IsAccountDeleted(ctx context.Context, did string) (bool, error) +} + +// WithAdmissions installs the per-(community, post) admission store. Without +// it, postv2 and acceptance/removal events have nowhere to record a decision. +func WithAdmissions(admissions posts.AdmissionRepository) PostEventConsumerOption { + return func(c *PostEventConsumer) { c.admissions = admissions } +} + +// WithDeletedAccounts installs the erased-account gate. Without it, no gate +// runs and every event indexes — the pre-036 behaviour. +func WithDeletedAccounts(lookup DeletedAccountLookup) PostEventConsumerOption { + return func(c *PostEventConsumer) { c.deletedAccounts = lookup } +} + +// WithPostRecordFetcher installs the §5.4 direct fetch used when an acceptance +// names a post this AppView has never indexed. +func WithPostRecordFetcher(fetcher PostRecordFetcher) PostEventConsumerOption { + return func(c *PostEventConsumer) { c.postFetcher = fetcher } +} + +// FetchedPost is one author-repo record read directly from its PDS. +// +// It carries the CID separately from the record because the CID is what the +// caller VERIFIES: an acceptance pins a strongRef, and a fetch that returned +// only the record body would leave the consumer indexing whatever the author's +// PDS felt like serving under that rkey. +type FetchedPost struct { + URI string + CID string + Record map[string]interface{} +} + +// PostRecordFetcher reads an author's post record straight from their PDS. +// +// This is what makes firehose-only ingestion actually CONVERGE (§5.4). +// Acceptance-before-post does not converge by dead-letter redrive alone: +// bounded retries cannot manufacture a post event that a relay-coverage gap +// will never deliver. Redrive stays the backstop for transient failures; this +// is the mechanism. +type PostRecordFetcher interface { + // FetchPost resolves the repo DID in postURI to a PDS and reads the record. + // The returned CID is the PDS's, unverified — checking it against the CID an + // acceptance pinned is the caller's job, because only the caller knows what + // was pinned. + FetchPost(ctx context.Context, postURI string) (*FetchedPost, error) +} + +// DirectPostFetcher is the production PostRecordFetcher: DID resolution, then +// com.atproto.repo.getRecord over an SSRF-guarded client with a size cap. +type DirectPostFetcher struct { + resolver identity.Resolver + + // allowPrivateHosts disables the SSRF protection that blocks private and + // loopback addresses. NEVER set outside tests. It exists for the same reason + // blueskypost's allowPrivateHost does: this package's own tests point the + // fetcher at an httptest server, which necessarily listens on loopback and + // would otherwise be refused by the guard that must stay on in production. + // + // The guard is not decorative here. The DID document that names the PDS is + // attacker-controlled — anyone can publish one — so an unguarded fetcher is + // a request forger pointed at whatever is reachable from the AppView's + // network, driven by any stranger who writes an acceptance record. + allowPrivateHosts bool +} + +// NewDirectPostFetcher wires the §5.4 fetch. SSRF protection is ON and there is +// no parameter to turn it off: a constructor that accepted a boolean is a +// constructor someone eventually passes true to from production wiring. +func NewDirectPostFetcher(resolver identity.Resolver) *DirectPostFetcher { + return &DirectPostFetcher{resolver: resolver} +} + +// httpClient builds the guarded client for one fetch. Declared here so the +// guard is derived from allowPrivateHosts at call time rather than baked into a +// client at construction, where a test seam could not reach it. +func (f *DirectPostFetcher) httpClient() *http.Client { return nil } + +// FetchPost implements PostRecordFetcher. +func (f *DirectPostFetcher) FetchPost(ctx context.Context, postURI string) (*FetchedPost, error) { + return nil, nil +} diff --git a/internal/atproto/jetstream/post_consumer.go b/internal/atproto/jetstream/post_consumer.go index 7ad9651..dc2b6b6 100644 --- a/internal/atproto/jetstream/post_consumer.go +++ b/internal/atproto/jetstream/post_consumer.go @@ -29,6 +29,19 @@ type PostEventConsumer struct { // before its author's profile. The identity is admitted only when its PDS // passes bridgeTrust. identityResolver identity.Resolver + + // RED STUB fields (task 5, cycle 1) — the collaborators author-owned post + // ingestion needs. Declared here so the options in authorpost.go compile; + // nothing reads them yet. See docs/PRD_AUTHOR_OWNED_POSTS.md §5.3-§5.6. + // + // admissions holds the per-(community, post) decision state. nil means the + // consumer is running in its pre-034 shape and records no admissions. + admissions posts.AdmissionRepository + // deletedAccounts gates events from erased accounts. nil means no gate. + deletedAccounts DeletedAccountLookup + // postFetcher resolves an acceptance whose subject was never indexed. nil + // means the dead-letter queue is the only convergence mechanism. + postFetcher PostRecordFetcher } // PostEventConsumerOption configures optional PostEventConsumer behaviour. diff --git a/internal/atproto/jetstream/postv2_consumer_test.go b/internal/atproto/jetstream/postv2_consumer_test.go new file mode 100644 index 0000000..254903f --- /dev/null +++ b/internal/atproto/jetstream/postv2_consumer_test.go @@ -0,0 +1,494 @@ +//go:build integration + +package jetstream + +import ( + "context" + "database/sql" + "testing" + "time" + + "Coves/internal/core/posts" + "Coves/internal/core/users" + "Coves/internal/db/postgres" + "Coves/tests/testkit" + + _ "github.com/lib/pq" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// Ingesting social.coves.community.postv2 — the post record that now lives in +// the AUTHOR's repo (docs/PRD_AUTHOR_OWNED_POSTS.md §3.1, §5.3). +// +// The collection is new, and so is almost everything about how an event from it +// is read. Three inversions drive every case below: +// +// - AUTHORSHIP COMES FROM THE REPO, not from the record. There is no `author` +// field to read and none to trust; event.Did IS the author. The old +// consumer's central security check — repo DID must equal the record's +// community — inverts into its opposite: the repo DID must NOT be the +// community, and the community is a claim the record makes. +// - AN UNKNOWN AUTHOR IS NORMAL. Open federated posting means the author of a +// post may be someone this AppView has never indexed, so "no users row" can +// no longer refuse the event (§5.3, migration 034 dropped the FK). +// - THE POST ROW IS NO LONGER THE DECISION. Whether a community shows the post +// lives in community_post_admissions, and a postv2 event's job is to record +// content plus a pending admission — never to decide anything. +// +// A REFUSAL AND A SKIP ARE DIFFERENT ANSWERS, and several cases turn on which +// one is being asserted. The connector dead-letters exactly what a handler +// returns as an error: nil is "handled, nothing more to do" and never reaches +// the queue; an error wrapped in ErrPermanentEvent is dead-lettered with its +// redrive budget already spent; any other error is dead-lettered retryable. +// "No dead letter" below therefore means a nil return, and it is asserted +// wherever an event must be dropped without the queue growing. + +const ( + pv2Prefix = "did:plc:pv2" + pv2Community = pv2Prefix + "community" + pv2Author = pv2Prefix + "author" + pv2Other = pv2Prefix + "otherauthor" +) + +// pv2Fixture is a wired consumer plus the stores the assertions read. +type pv2Fixture struct { + consumer *PostEventConsumer + admissions posts.AdmissionRepository + db *sql.DB + users *mockUserService +} + +// newPV2Fixture indexes a community and returns a consumer wired with the three +// collaborators author-owned ingestion needs, all real: the admissions store and +// the deleted-account lookup both run against this test's Postgres clone, so a +// gate that reads the wrong table fails here rather than passing against a map. +func newPV2Fixture(t *testing.T, db *sql.DB) pv2Fixture { + t.Helper() + + insertBridgedUser(t, db, pv2Author, "pv2author.test") + insertBridgedCommunity(t, db, pv2Community, "pv2community.test", pv2Author) + + us := newMockUserService() + us.users[pv2Author] = &users.User{DID: pv2Author, Handle: "pv2author.test"} + + admissions := postgres.NewAdmissionRepository(db) + consumer := NewPostEventConsumer( + postgres.NewPostRepository(db), + postgres.NewCommunityRepository(db), + us, + db, + WithAdmissions(admissions), + WithDeletedAccounts(postgres.NewDeletedAccountRepository(db)), + ) + + return pv2Fixture{consumer: consumer, admissions: admissions, db: db, users: us} +} + +// pv2URI is the AT-URI of an author-repo post: the author's DID is the +// authority, which is the whole point of the flip. +func pv2URI(authorDID, rkey string) string { + return "at://" + authorDID + "/" + PostV2Collection + "/" + rkey +} + +// pv2Record builds a postv2 record body. It carries NO author field, by +// construction — the lexicon has none (§3.1), and a consumer that still read one +// would be reading a field only a forger would bother to send. +func pv2Record(communityDID, title, content string) map[string]interface{} { + return map[string]interface{}{ + "$type": PostV2Collection, + "community": communityDID, + "title": title, + "content": content, + "createdAt": "2026-03-01T00:00:00Z", + } +} + +// pv2Event builds a commit event in the AUTHOR's repo. +func pv2Event(authorDID, op, rkey, rev, cid string, timeUS int64, record map[string]interface{}) *JetstreamEvent { + return revCommitEvent(authorDID, PostV2Collection, op, rkey, rev, cid, timeUS, record) +} + +// readPV2Post returns the indexed row's identity columns, or fails naming the +// URI that is missing. +func readPV2Post(t *testing.T, db *sql.DB, uri string) (authorDID, communityDID, cid, title string, deletedAt *time.Time) { + t.Helper() + err := db.QueryRow( + `SELECT author_did, community_did, cid, title, deleted_at FROM posts WHERE uri = $1`, uri, + ).Scan(&authorDID, &communityDID, &cid, &title, &deletedAt) + require.NoErrorf(t, err, "no post row for %s", uri) + return authorDID, communityDID, cid, title, deletedAt +} + +func countRows(t *testing.T, db *sql.DB, query string, args ...interface{}) int { + t.Helper() + var n int + require.NoError(t, db.QueryRow(query, args...).Scan(&n)) + return n +} + +// markAccountDeleted writes the migration-036 erasure marker directly, which is +// what userRepo.Delete leaves behind. +func markAccountDeleted(t *testing.T, db *sql.DB, did string) { + t.Helper() + _, err := db.Exec( + `INSERT INTO deleted_accounts (did, deleted_at) VALUES ($1, NOW()) ON CONFLICT (did) DO NOTHING`, did) + require.NoErrorf(t, err, "marking %s deleted", did) +} + +func TestPostV2Consumer_Create_IndexesTheAuthorsPostAndOpensAPendingAdmission(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + f := newPV2Fixture(t, db) + ctx := context.Background() + + const cid = "bafyreipv2create" + rkey := "pv2create" + uri := pv2URI(pv2Author, rkey) + + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "create", rkey, testkit.TID(), cid, time.Now().UnixMicro(), + pv2Record(pv2Community, "an author-signed post", "words the author is accountable for"), + ))) + + authorDID, communityDID, storedCID, title, deletedAt := readPV2Post(t, db, uri) + + // The author is the repo, not a field. If this ever reads from the record + // instead, any repo can claim any author — which is exactly the + // impersonation power the flip removed (§1). + assert.Equal(t, pv2Author, authorDID, + "author_did must come from event.Did: the record has no author field, and deriving one from anywhere else restores the forgery the flip removed") + assert.Equal(t, pv2Community, communityDID, + "community_did comes from the record — it is the author's submission target, a claim the community has not yet agreed to") + assert.Equal(t, cid, storedCID) + assert.Equal(t, "an author-signed post", title) + assert.Nil(t, deletedAt) + + admission, err := f.admissions.Get(ctx, pv2Community, uri) + require.NoErrorf(t, err, "indexing a postv2 must open the admission row the community will decide against") + require.NotNil(t, admission) + + // PENDING, not accepted. The post claims the community; the community has + // said nothing. §2 is explicit that a post lacking an acceptance is never + // shown in that community, and an indexer that opened the row as anything + // else would publish speech the community never agreed to carry. + assert.Equal(t, posts.AdmissionStatusPending, admission.Status) + assertNullableStringPV2(t, cid, admission.EvaluatedCID, + "evaluated_cid must be the CID this event carried: it is what the next decision judges and what an acceptance's pinned CID is compared against") + + // An author-repo event orders by the per-record rev gate, never by the + // community watermark. Stamping one here would let an author's edit outrank + // a moderator's removal — two repos, two unrelated revision clocks (§5.2). + assert.Nilf(t, admission.LastCommunityEvent, + "an author-repo event must not advance the community watermark; got %+v", admission.LastCommunityEvent) +} + +func TestPostV2Consumer_DeletedAuthor_IsDroppedWithoutADeadLetter(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + f := newPV2Fixture(t, db) + ctx := context.Background() + + // The account was erased. Migration 036's marker is the only thing that says + // so: the users row is gone, and under §5.3 a missing users row means + // "federated author we have not indexed", which indexes normally. Without + // the marker this event silently re-creates the content the deletion swept — + // and it WILL arrive, because dead-letter redrives and overlapping feeds + // replay events long after the account is gone. + markAccountDeleted(t, db, pv2Other) + + rkey := "pv2deleted" + uri := pv2URI(pv2Other, rkey) + + err := f.consumer.HandleEvent(ctx, pv2Event( + pv2Other, "create", rkey, testkit.TID(), "bafyreipv2deleted", time.Now().UnixMicro(), + pv2Record(pv2Community, "a post from an erased account", "content the AppView was asked to forget"), + )) + + // Nil, not an error. The connector dead-letters whatever a handler returns, + // so refusing this event with an error would fill the queue with rows that + // redrive, fail identically, and retire — turning every erased account into + // a permanent stream of operational noise. The event is not a failure; it is + // an event with nothing to do. + require.NoError(t, err, + "an event from an erased account must be dropped as a no-op: returning an error dead-letters it, and the queue exists for failures, not for events the AppView correctly ignores") + + assert.Zero(t, countRows(t, db, `SELECT count(*) FROM posts WHERE uri = $1`, uri), + "the post of an erased account must not be indexed") + assert.Zero(t, countRows(t, db, `SELECT count(*) FROM community_post_admissions WHERE post_uri = $1`, uri), + "no admission row either: an admission for an erased account's post is the row migration 036 exists to stop being recreated") +} + +func TestPostV2Consumer_UnknownAuthorIndexesAnyway(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + f := newPV2Fixture(t, db) + ctx := context.Background() + + // The §5.3 flip, and the case that separates "erased" from "never seen". + // pv2Other has no users row and no erasure marker — the ordinary state of an + // author on someone else's server. The old consumer refused this event, which + // the test architecture recorded as "federated authors cannot currently be + // indexed"; open federated posting makes that refusal a bug. + rkey := "pv2unknown" + uri := pv2URI(pv2Other, rkey) + + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Other, "create", rkey, testkit.TID(), "bafyreipv2unknown", time.Now().UnixMicro(), + pv2Record(pv2Community, "a post from a federated stranger", "posted from a PDS we have never met"), + )), "an author this AppView has never indexed must not block ingestion: that refusal is what made cross-server posting impossible") + + authorDID, _, _, _, _ := readPV2Post(t, db, uri) + assert.Equal(t, pv2Other, authorDID) + + admission, err := f.admissions.Get(ctx, pv2Community, uri) + require.NoError(t, err) + require.NotNil(t, admission) + assert.Equal(t, posts.AdmissionStatusPending, admission.Status) +} + +func TestPostV2Consumer_UnknownCommunity_IsRedrivable(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + f := newPV2Fixture(t, db) + ctx := context.Background() + + const ghostCommunity = "did:plc:pv2ghostcommunity" + rkey := "pv2ghost" + + err := f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "create", rkey, testkit.TID(), "bafyreipv2ghost", time.Now().UnixMicro(), + pv2Record(ghostCommunity, "aimed at a community we have not indexed", "body"), + )) + + require.Error(t, err, "a post naming a community this AppView has never indexed cannot open an admission row against it") + + // The classification is the assertion. BigSky preserves order within a repo, + // not across repos, so a post can genuinely arrive before the community's own + // profile event — this is an ORDERING failure, and marking it permanent would + // discard every post that merely arrived early, with the redrive that would + // have fixed it already spent. + assert.NotErrorIs(t, err, ErrPermanentEvent, + "community-not-found is an ordering failure and must stay transient so the redrive succeeds once the community arrives") + + assert.Zero(t, countRows(t, db, `SELECT count(*) FROM posts WHERE uri = $1`, pv2URI(pv2Author, rkey)), + "the refused post must not have been indexed") +} + +func TestPostV2Consumer_UpdateChangingCommunity_IgnoresTheWholeEvent(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + f := newPV2Fixture(t, db) + ctx := context.Background() + + otherName := "pv2secondcommunity" + const secondCommunity = pv2Prefix + "community2" + insertBridgedCommunity(t, db, secondCommunity, otherName+".test", pv2Author) + + rkey := "pv2retarget" + uri := pv2URI(pv2Author, rkey) + base := time.Now().UnixMicro() + revs := increasingTIDs(t, 2) + + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "create", rkey, revs[0], "bafyreipv2original", base, + pv2Record(pv2Community, "original title", "original body"), + ))) + + // The retarget attempt: same record, new community, and new content riding + // along. §3.1 says the ENTIRE event is invalid — discard it, do not merely + // keep the old community value. Applying the content while ignoring the + // community would leave the first community's admission holding a CID it + // never evaluated, silently publishing content nobody judged under a + // standing acceptance. + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "update", rkey, revs[1], "bafyreipv2retargeted", base+1_000_000, + pv2Record(secondCommunity, "retargeted title", "retargeted body"), + )), "an update that changes community is invalid, not an infrastructure failure: it must be skipped, not dead-lettered") + + _, communityDID, cid, title, _ := readPV2Post(t, db, uri) + assert.Equal(t, pv2Community, communityDID, "the community must not move; retargeting a post means writing a new record") + assert.Equalf(t, "bafyreipv2original", cid, + "the whole event is invalid, so the CID must not move either — a moved CID under an unmoved community is content the community never evaluated") + assert.Equal(t, "original title", title, "the content half of a rejected event must be rejected with it") + + assert.Zero(t, countRows(t, db, + `SELECT count(*) FROM community_post_admissions WHERE post_uri = $1 AND community_did = $2`, uri, secondCommunity), + "an ignored retarget must not open an admission in the community it named: that row would be a dangling decision about a post that never claimed this community") +} + +func TestPostV2Consumer_UpdateWithNewContent_ReopensAnAcceptedAdmission(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + f := newPV2Fixture(t, db) + ctx := context.Background() + + rkey := "pv2edit" + uri := pv2URI(pv2Author, rkey) + base := time.Now().UnixMicro() + revs := increasingTIDs(t, 2) + + const originalCID = "bafyreipv2editv1" + const editedCID = "bafyreipv2editv2" + + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "create", rkey, revs[0], originalCID, base, + pv2Record(pv2Community, "before the edit", "the version the community judged"), + ))) + + // The community accepts the original, pinning that exact CID. + acceptanceRkey := testkit.TID() + accepted, err := f.admissions.ApplyAcceptance(ctx, posts.ApplyAcceptanceCommand{ + CommunityDID: pv2Community, + PostURI: uri, + AcceptanceURI: "at://" + pv2Community + "/social.coves.community.acceptance/" + acceptanceRkey, + AcceptanceRkey: acceptanceRkey, + PinnedCID: originalCID, + Watermark: posts.CommunityWatermark{Rev: testkit.TID()}, + }) + require.NoError(t, err) + require.Equal(t, posts.AdmissionApplied, accepted.Outcome, "fixture: the acceptance must stand before the edit arrives") + + // The author edits. The standing acceptance now pins content that is no + // longer current, and §5.5 forbids rendering the new CID under it. + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "update", rkey, revs[1], editedCID, base+1_000_000, + pv2Record(pv2Community, "after the edit", "words the community has not seen"), + ))) + + _, _, storedCID, title, _ := readPV2Post(t, db, uri) + assert.Equal(t, editedCID, storedCID, "the edit's content must be indexed — it is what the community will re-judge") + assert.Equal(t, "after the edit", title) + + admission, err := f.admissions.Get(ctx, pv2Community, uri) + require.NoError(t, err) + require.NotNil(t, admission) + + assert.Equal(t, posts.AdmissionStatusPendingReacceptance, admission.Status, + "an edit under a standing acceptance must reopen the decision: auto-rendering the new CID under the old acceptance would let an author swap content past moderation after approval") + assertNullableStringPV2(t, editedCID, admission.EvaluatedCID, + "evaluated_cid must follow the content, or the re-decision judges the version the author replaced") + assertNullableStringPV2(t, originalCID, admission.AcceptedCID, + "the acceptance still pins the CID the community actually agreed to; moving it here would forge agreement to the edit") +} + +func TestPostV2Consumer_EditOfARemovedPost_IsSkippedNotDeadLettered(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + f := newPV2Fixture(t, db) + ctx := context.Background() + + rkey := "pv2removededit" + uri := pv2URI(pv2Author, rkey) + base := time.Now().UnixMicro() + revs := increasingTIDs(t, 2) + + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "create", rkey, revs[0], "bafyreipv2removedv1", base, + pv2Record(pv2Community, "will be removed", "body"), + ))) + + // The precondition is asserted, not assumed. ApplyRemoval below creates the + // admission row when the subject is absent, so without this the whole test + // would pass against a consumer that ignores postv2 entirely — the edit + // would be a no-op for the wrong reason and the final assertion would read + // back a row nothing had ever contested. + _, _, _, _, deletedAt := readPV2Post(t, db, uri) + require.Nil(t, deletedAt, "fixture: the post must be indexed and live before the removal lands") + + removed, err := f.admissions.ApplyRemoval(ctx, posts.ApplyRemovalCommand{ + CommunityDID: pv2Community, + PostURI: uri, + DecisionCode: string(posts.DecisionRuleViolation), + Watermark: posts.CommunityWatermark{Rev: testkit.TID()}, + }) + require.NoError(t, err) + require.Equal(t, posts.AdmissionApplied, removed.Outcome, "fixture: the removal must stand") + + // §5.5: removal is terminal against author-repo events. The repository + // answers this edit with skipped_terminal — a value, not an error — and the + // consumer must pass that through as success. Mapping a CAS skip onto an + // error return is the mistake migration 033's precedent exists to prevent: + // it routes the system WORKING into the dead-letter queue, where every + // redrive re-runs a decision that will refuse identically forever. + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "update", rkey, revs[1], "bafyreipv2removedv2", base+1_000_000, + pv2Record(pv2Community, "edited while removed", "laundering attempt"), + )), "a terminal admission skip is an outcome, not a failure: returning an error would dead-letter healthy skips") + + admission, err := f.admissions.Get(ctx, pv2Community, uri) + require.NoError(t, err) + require.NotNil(t, admission) + assert.Equal(t, posts.AdmissionStatusRemoved, admission.Status, + "editing a removed post must not reopen it; that is how a removed post gets laundered back through auto-acceptance") +} + +func TestPostV2Consumer_Delete_TombstonesTheRow(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + f := newPV2Fixture(t, db) + ctx := context.Background() + + rkey := "pv2delete" + uri := pv2URI(pv2Author, rkey) + base := time.Now().UnixMicro() + revs := increasingTIDs(t, 2) + + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "create", rkey, revs[0], "bafyreipv2delete", base, + pv2Record(pv2Community, "to be deleted by its author", "body that must survive the tombstone"), + ))) + + // A delete carries no record, exactly as Jetstream delivers it. + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "delete", rkey, revs[1], "", base+1_000_000, nil, + ))) + + _, _, _, title, deletedAt := readPV2Post(t, db, uri) + require.NotNil(t, deletedAt, + "an author delete must SOFT-delete: the row is the rev gate's tombstone, the comment thread's parent, and what moderation still reads") + assert.Equal(t, "to be deleted by its author", title, + "a soft delete must not blank the content") + + // The host-side half — the community observing the tombstone and deleting + // its acceptance (§5.3) — is task 5b's scope and deliberately not asserted + // here. What matters at this point is that the tombstone exists for that + // sweep to find. +} + +// assertNullableStringPV2 asserts a nullable column holds exactly want. +// +// Named apart from the postgres package's helper of the same shape because this +// package has its own; the duplication is two lines against an import cycle. +func assertNullableStringPV2(t *testing.T, want string, got *string, what string) { + t.Helper() + if !assert.NotNilf(t, got, "%s: want %q, got NULL", what, want) { + return + } + assert.Equalf(t, want, *got, "%s", what) +} + +// increasingTIDs returns n real atProto TIDs whose lexicographic order is their +// generation order — what a repo's successive commits actually carry, and what +// the rev gate compares. Invented revs would prove the gate works on invented +// data. +func increasingTIDs(t *testing.T, n int) []string { + t.Helper() + revs := make([]string, n) + for i := range revs { + revs[i] = testkit.TID() + if i > 0 { + require.Greaterf(t, revs[i], revs[i-1], + "testkit.TID must emit lexicographically increasing revs; got %q after %q", revs[i], revs[i-1]) + } + } + return revs +} diff --git a/internal/core/posts/status.go b/internal/core/posts/status.go new file mode 100644 index 0000000..ca851cc --- /dev/null +++ b/internal/core/posts/status.go @@ -0,0 +1,83 @@ +package posts + +import ( + "context" + "time" +) + +// RED STUB (task 5, cycle 1). Signatures only — every method returns zero +// values so the tests that describe this surface compile and fail on their +// assertions rather than on a missing symbol. The implementation is GREEN's. + +// The read side of an admission decision: social.coves.community.post.getStatus +// (docs/PRD_AUTHOR_OWNED_POSTS.md §3.4). +// +// It exists because a rejection is AppView-LOCAL. §3.3 is explicit that a +// submission refused before it was ever accepted writes NO community record — +// spam must not be archived forever in the repo of the community that refused +// it — so there is nothing on the firehose for an author's client to read, and +// "did it get in, and if not, why" has no answer except to ask the host. +// +// It is deliberately its own service rather than a method on Service. Service +// is the write path plus post hydration and is implemented by test doubles all +// over the suite; a status query needs the admissions repository and nothing +// else, and widening the big interface to reach it would make every one of +// those doubles carry a method it has no opinion about. + +// PostStatus is one community's answer about one post. +// +// The optional fields are pointers rather than zero strings because their +// absence is meaningful and different from emptiness: a pending post has no +// decision to report, and rendering an empty code would tell an author their +// post was refused for a reason nobody can name. +type PostStatus struct { + // Status is the admission state (§6.1), verbatim: the same vocabulary the + // admissions table and the consumer speak, so a client that switches on it + // is switching on the real state machine and not a display translation. + Status AdmissionStatus + + // DecisionCode is set for rejected and removed, and only for those. It is + // the vocabulary of DecisionCode — both the codes a community publishes in + // a removal record and the admission-time codes that never reach a repo. + DecisionCode *string + + // DecisionAt is when the decision above was made. + DecisionAt *time.Time + + // AcceptanceURI names the live community acceptance record, so a client can + // go read the signed attestation rather than taking this AppView's word for + // it. Set only while an acceptance stands. + AcceptanceURI *string +} + +// GetStatusRequest names one subject: which community's answer, about which +// post. +// +// Both halves are required and neither has a default. A post can hold +// independent decisions from several communities (§2, forks), so "the status of +// this post" is not a question with one answer, and a request that omitted the +// community would have to invent one. +type GetStatusRequest struct { + PostURI string + CommunityDID string +} + +// StatusService answers getStatus. +type StatusService interface { + // GetStatus returns one community's decision about one post, or ErrNotFound + // when the community has never seen it. + GetStatus(ctx context.Context, req GetStatusRequest) (*PostStatus, error) +} + +type statusService struct { + admissions AdmissionRepository +} + +// NewStatusService wires the status query over the admissions store. +func NewStatusService(admissions AdmissionRepository) StatusService { + return &statusService{admissions: admissions} +} + +func (s *statusService) GetStatus(ctx context.Context, req GetStatusRequest) (*PostStatus, error) { + return nil, nil +} diff --git a/internal/db/postgres/admission_repo_schema_test.go b/internal/db/postgres/admission_repo_schema_test.go index 498d943..7f613fb 100644 --- a/internal/db/postgres/admission_repo_schema_test.go +++ b/internal/db/postgres/admission_repo_schema_test.go @@ -318,13 +318,15 @@ func TestMigration034_DownRestoresTheAuthorForeignKeyUnvalidated(t *testing.T) { require.NoError(t, err, "with fk_author dropped, a federated author's post must index even though no users row exists for them") - // The expected-version parameter is the tripwire, and it has fired once - // already: migration 035 (post_submissions) now sits on top of 034, so it - // has to come off first. Rolling back explicitly, one asserted step at a - // time, is what keeps the assertions below pointed at 034's Down rather than - // at whatever happens to be newest. + // The expected-version parameter is the tripwire, and it has now fired + // twice: migration 035 (post_submissions) and 036 (deleted_accounts) both + // sit on top of 034, so both have to come off first. Rolling back + // explicitly, one asserted step at a time, is what keeps the assertions + // below pointed at 034's Down rather than at whatever happens to be newest. + require.EqualValues(t, 36, testkit.MigrateDownOne(t, db, 36), + "036 sits on top of 034 and must be rolled back first; asserting which migration came off is what stops this test drifting onto a newer one") require.EqualValues(t, 35, testkit.MigrateDownOne(t, db, 35), - "035 sits on top of 034 and must be rolled back first; asserting which migration came off is what stops this test drifting onto a newer one") + "035 sits on top of 034 and must be rolled back next; asserting which migration came off is what stops this test drifting onto a newer one") assert.EqualValues(t, 34, testkit.MigrateDownOne(t, db, 34), "this test asserts on 034's Down section; rolling back a different migration would prove nothing about it") diff --git a/internal/db/postgres/deleted_account_repo.go b/internal/db/postgres/deleted_account_repo.go new file mode 100644 index 0000000..a9d3808 --- /dev/null +++ b/internal/db/postgres/deleted_account_repo.go @@ -0,0 +1,35 @@ +package postgres + +import ( + "context" + "database/sql" +) + +// RED STUB (task 5, cycle 1). Signatures only; the query is GREEN's. + +// DeletedAccountRepository reads the migration-036 erasure markers. +// +// It satisfies jetstream.DeletedAccountLookup structurally rather than by +// importing it: the interface is declared where it is CONSUMED (the ingestion +// consumer), which is what keeps the storage layer from depending on the +// firehose layer for a single method. +type DeletedAccountRepository struct { + db *sql.DB +} + +// NewDeletedAccountRepository wires the lookup over the AppView database. +func NewDeletedAccountRepository(db *sql.DB) *DeletedAccountRepository { + return &DeletedAccountRepository{db: db} +} + +// IsAccountDeleted reports whether this DID names an account the AppView was +// asked to erase. +// +// A query failure must come back as an error, never as false. Under +// author-owned posts an unknown author indexes normally (§5.3), so a false here +// is indistinguishable from a healthy answer — a database blip would silently +// re-index the content a deletion erased, which is the exact outcome the marker +// table exists to prevent. +func (r *DeletedAccountRepository) IsAccountDeleted(ctx context.Context, did string) (bool, error) { + return false, nil +} diff --git a/internal/db/postgres/deleted_accounts_schema_test.go b/internal/db/postgres/deleted_accounts_schema_test.go new file mode 100644 index 0000000..7875f99 --- /dev/null +++ b/internal/db/postgres/deleted_accounts_schema_test.go @@ -0,0 +1,212 @@ +//go:build integration + +package postgres + +import ( + "context" + "testing" + "time" + + "Coves/internal/core/users" + "Coves/tests/fixtures" + "Coves/tests/testkit" + + _ "github.com/lib/pq" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// Migration 036's marker table, and the thing it exists to make impossible. +// +// Account deletion used to leave NO trace. userRepo.Delete removes the users +// row, the posts, and (since 034) the admission rows — and then the firehose +// redelivers a post event for that same author, or a dead letter for it is +// redriven, and every one of those swept rows comes straight back. The AppView +// re-indexes the content of an account it was asked to erase, and nothing in +// the schema can tell it not to: an absent users row is indistinguishable from +// an author who simply has not been indexed yet, which under author-owned posts +// (§5.3) is a state the consumer is REQUIRED to accept. +// +// deleted_accounts is what makes those two cases distinguishable. A row here +// means "this DID was erased on purpose"; no row means "never seen". The +// ingestion gate that reads it is tested in internal/atproto/jetstream; what is +// tested here is the half that has to be true for the gate to mean anything: +// the marker is written, it is written ATOMICALLY WITH the deletion, and +// re-registration clears it. +// +// See docs/PRD_AUTHOR_OWNED_POSTS.md rev 2.7 (§5 status header). + +const deletedAccountsTable = "deleted_accounts" + +func TestDeletedAccountsTable_Columns(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + + requireTableExists(t, db, deletedAccountsTable) + + type columnShape struct { + dataType string + nullable bool + } + want := map[string]columnShape{ + "did": {"text", false}, + "deleted_at": {"timestamp with time zone", false}, + "deleted_rev": {"text", true}, + } + + rows, err := db.QueryContext(ctx, ` + SELECT column_name, data_type, is_nullable + FROM information_schema.columns + WHERE table_schema = current_schema() AND table_name = $1 + `, deletedAccountsTable) + require.NoError(t, err) + defer func() { _ = rows.Close() }() + + got := map[string]columnShape{} + for rows.Next() { + var name, dataType, isNullable string + require.NoError(t, rows.Scan(&name, &dataType, &isNullable)) + got[name] = columnShape{dataType: dataType, nullable: isNullable == "YES"} + } + require.NoError(t, rows.Err()) + + for name, wantShape := range want { + gotShape, ok := got[name] + if !assert.Truef(t, ok, "%s.%s is missing", deletedAccountsTable, name) { + continue + } + assert.Equalf(t, wantShape.dataType, gotShape.dataType, "%s.%s type", deletedAccountsTable, name) + assert.Equalf(t, wantShape.nullable, gotShape.nullable, "%s.%s nullability", deletedAccountsTable, name) + } + + assert.Equal(t, []string{"did"}, primaryKeyColumns(t, db, deletedAccountsTable), + "the DID is the whole key: one marker per account, so a re-delete updates rather than accumulating rows the gate would have to deduplicate") + + // deleted_at is NOT NULL because the marker's only job is to be READ by a + // consumer deciding whether to index an event, and a marker with no time is + // a marker that cannot participate in any retention or audit answer later. + // deleted_rev is nullable because nothing knows the account's repo revision + // at AppView-deletion time — the deletion is a local administrative act, not + // a commit — and a column that had to be filled would be filled with a lie. + assert.Falsef(t, got["deleted_at"].nullable, "deleted_at must be NOT NULL") + assert.Truef(t, got["deleted_rev"].nullable, "deleted_rev must be nullable") +} + +func TestUserRepo_Delete_LeavesADeletionMarker(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + requireTableExists(t, db, deletedAccountsTable) + + handle := testkit.UniqueIDWithPrefix(t, "delmark") + did := fixtures.DID(handle) + fixtures.User(t, db, handle+".test", did) + + before := time.Now().Add(-time.Second) + require.NoError(t, NewUserRepository(db).Delete(ctx, did)) + + var deletedAt time.Time + var deletedRev *string + err := db.QueryRowContext(ctx, + `SELECT deleted_at, deleted_rev FROM deleted_accounts WHERE did = $1`, did, + ).Scan(&deletedAt, &deletedRev) + require.NoErrorf(t, err, + "deleting %s left no marker row. Without one, a redriven post event for this author re-indexes the content the deletion erased, "+ + "and the consumer cannot tell an erased account from one that has simply not been indexed yet", did) + + assert.Truef(t, deletedAt.After(before), "deleted_at (%s) must record when the deletion happened", deletedAt) + assert.Nil(t, deletedRev, + "deleted_rev must be left NULL: the AppView does not know the account's repo revision at deletion time, and inventing one would put a fabricated watermark where a real comparison happens") +} + +func TestUserRepo_Delete_MarkerIsWrittenInTheSameTransaction(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + requireTableExists(t, db, deletedAccountsTable) + + // A DID with content but NO users row. Delete sweeps the content and then + // finds nothing to delete from users, which is the one failure the method + // already reports (users.ErrUserNotFound) — and it makes this the cheapest + // honest probe of atomicity there is. + // + // The claim under test is narrow and load-bearing: the marker INSERT must be + // a statement of the deletion transaction, not a separate write afterwards. + // A marker written outside it survives a rollback, and then a DID that was + // never actually erased is permanently refused by the ingestion gate — the + // account's own future posts stop indexing, silently, with no row anywhere + // explaining why. + name := testkit.UniqueIDWithPrefix(t, "delatomic") + communityDID, err := fixtures.Community(ctx, db, name, "owner"+name) + require.NoError(t, err) + + ghostAuthor := fixtures.DID(testkit.UniqueID(t)) + postURI := "at://" + ghostAuthor + "/social.coves.community.postv2/" + testkit.TID() + _, err = db.ExecContext(ctx, ` + INSERT INTO community_post_admissions (community_did, post_uri, status, created_at, updated_at) + VALUES ($1, $2, 'pending', NOW(), NOW()) + `, communityDID, postURI) + require.NoError(t, err) + + deleteErr := NewUserRepository(db).Delete(ctx, ghostAuthor) + require.ErrorIsf(t, deleteErr, users.ErrUserNotFound, + "fixture: deleting a DID with no users row must fail, which is what gives this test a rolled-back transaction to inspect") + + var markers int + require.NoError(t, db.QueryRowContext(ctx, + `SELECT count(*) FROM deleted_accounts WHERE did = $1`, ghostAuthor).Scan(&markers)) + assert.Zerof(t, markers, + "the deletion FAILED and rolled back, but a marker for %s survived: the marker insert is running outside the deletion transaction. "+ + "A marker for an account that still exists is worse than no marker at all — the ingestion gate refuses every future event from a live account", ghostAuthor) + + // And the rollback really did roll back, so the surviving marker above could + // only have come from a write outside the transaction. + var admissions int + require.NoError(t, db.QueryRowContext(ctx, + `SELECT count(*) FROM community_post_admissions WHERE post_uri = $1`, postURI).Scan(&admissions)) + require.Equal(t, 1, admissions, + "fixture: the failed deletion must have rolled its content sweep back, or this test proves nothing about where the marker was written") +} + +func TestUserRepo_Create_ClearsTheDeletionMarker(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + requireTableExists(t, db, deletedAccountsTable) + + // Re-registration is the marker's exit. A DID that comes back — the same + // person signing up again on the same PDS, or an account restored after a + // mistaken deletion — must index normally, and a marker left behind would + // make the AppView refuse their new posts forever with nothing to show for + // it. Create is the assertion point because it is the one statement both + // service paths funnel through: IndexUser calls CreateUser (service.go:457) + // and RegisterAccount ends in the same repository insert. + repo := NewUserRepository(db) + + handle := testkit.UniqueIDWithPrefix(t, "rereg") + did := fixtures.DID(handle) + fixtures.User(t, db, handle+".test", did) + require.NoError(t, repo.Delete(ctx, did)) + + var markers int + require.NoError(t, db.QueryRowContext(ctx, + `SELECT count(*) FROM deleted_accounts WHERE did = $1`, did).Scan(&markers)) + require.Equal(t, 1, markers, "fixture: the deletion must have left a marker for the re-registration to clear") + + _, err := repo.Create(ctx, &users.User{ + DID: did, + Handle: handle + ".test", + PDSURL: testkit.Endpoints().PDS.BaseURL, + }) + require.NoError(t, err, "a deleted DID must be able to register again") + + require.NoError(t, db.QueryRowContext(ctx, + `SELECT count(*) FROM deleted_accounts WHERE did = $1`, did).Scan(&markers)) + assert.Zerof(t, markers, + "re-registering %s left the deletion marker standing. The ingestion gate reads this table, so the account would index its profile and then have every post it writes silently dropped", did) +} -- 2.51.2 From 9005b5d5d51718ea0b1aea28523b87b07461efcb Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 00:02:51 -0700 Subject: [PATCH 02/17] =?UTF-8?q?feat(ingestion):=20GREEN=20=E2=80=94=20au?= =?UTF-8?q?thor-owned=20post=20ingestion,=20enablers=20+=20consumers?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Task 5a green gate. All 25 reds pass; the only remaining failure is cmd/contract-manifest, which is EXPECTED and is task 5b's work (the three new collections need //coves:ingestion-contract markers in tests/e2e). A. Migration 036 enablers deleted_accounts(did PK, deleted_at NOT NULL, deleted_rev nullable). userRepo.Delete writes the marker as step 0 of the SAME transaction, so a rolled-back deletion cannot leave a marker naming a live account. userRepo.Create clears it (now transactional) — the marker's exit is re-registration, and both service paths funnel through that insert. postgres.DeletedAccountRepository reads it; a query failure is an error, never a false. B. post.getStatus Lexicon getStatus.json (query; description records why it is deliberately unauthenticated). posts.statusService over the admissions repo; decision fields gated on rejected/removed so a pending post never carries a code. Handler omits absent optionals entirely rather than emitting null. Route registered with NO auth middleware via a new variadic RegisterPostRoutes option, keeping every existing caller compiling. C. postv2 consumer (authorpost.go) author = event.Did, enforced by a record type that HAS no author field. Erasure gate runs before anything touches the database. Community immutability discards the whole update event; unknown community stays transient; unknown author indexes anyway with bounded, non-fatal opportunistic hydration (5s, verified handles only). UpsertPending follows the gated content write, so a refused event never moves evaluated_cid backwards. Delete tombstones. D. acceptance/removal consumer + §5.4 direct fetch Repo DID must be an indexed community (transient if not — the profile event can genuinely arrive second). Tuple CAS applied through the task-2 repo; every skip returns nil. Pre-emptive removal creates its row. Acceptance-before-post converges by DirectPostFetcher: DID-resolved PDS, SSRF-guarded client, 1 MiB cap detected rather than truncated, pinned-CID verification, and a PERMANENT refusal when the fetched record names another community. That last check is also enforced on the already- indexed path, or it would be bypassable by posting before accepting. Acceptance deletions resolve their subject through acceptance_rkey; removal deletions are no-ops because their paired put outranks them. E. consumerWantedCollections ConsumerPosts keeps social.coves.community.post and gains postv2, acceptance and removal. Production wiring passes the admissions repo, the erasure lookup and the fetcher; the SSRF-relaxed fetcher constructor is named for what it does and reachable only under IS_DEV_ENV. Shared refactor, behaviour-preserving: the rev gate + content UPDATE and the gated insert are now one implementation each (applyPostContentUpdate, indexPostIfRevWins), used by both the community-repo and author-repo paths, which differ in who may claim what — not in how an edit is applied. BridgeTrust re-keyed to the AUTHOR's PDS on the postv2 path, default-deny for an unhydrated author. Co-Authored-By: Claude Fable 5 --- cmd/server/consumers.go | 23 +- cmd/server/routes.go | 3 +- cmd/server/wiring.go | 21 +- internal/api/handlers/post/getstatus.go | 75 +- internal/api/routes/post.go | 40 + internal/atproto/jetstream/authorpost.go | 954 +++++++++++++++++- internal/atproto/jetstream/feeds.go | 13 + internal/atproto/jetstream/post_consumer.go | 422 +++++--- .../coves/community/post/getStatus.json | 58 ++ internal/core/posts/status.go | 46 +- .../036_create_deleted_accounts.sql | 60 ++ internal/db/postgres/deleted_account_repo.go | 11 +- internal/db/postgres/user_repo.go | 62 +- 13 files changed, 1606 insertions(+), 182 deletions(-) create mode 100644 internal/atproto/lexicon/social/coves/community/post/getStatus.json create mode 100644 internal/db/migrations/036_create_deleted_accounts.sql diff --git a/cmd/server/consumers.go b/cmd/server/consumers.go index d09c7d0..668b742 100644 --- a/cmd/server/consumers.go +++ b/cmd/server/consumers.go @@ -2,6 +2,7 @@ package main import ( "Coves/internal/atproto/jetstream" + postgresRepo "Coves/internal/db/postgres" "context" "errors" "fmt" @@ -160,12 +161,30 @@ func (a *application) registerFeedConsumers() []feedConsumer { a.identityResolver, jetstream.WithCommunityRevGate(a.revGate)), }) - // Posts created in community repositories. + // Posts, in both shapes, plus the community records that decide about + // them: the deprecated community-repo post, the author-repo postv2, and + // the acceptance/removal pair. One consumer, because they write the same + // admission row and an acceptance is meaningless without the post it pins. + // + // The direct fetcher is what makes acceptance-before-post converge without + // full relay coverage (PRD §5.4). It dials a PDS named by a DID document + // anyone can publish, so its SSRF guard stays on in production and the + // stood-down constructor is reachable only under IS_DEV_ENV — where the + // hermetic stack's PDS is a private address the guard would otherwise + // refuse. + postFetcher := jetstream.NewDirectPostFetcher(a.identityResolver) + if a.cfg.IsDevEnv { + slog.Warn("direct post fetch has SSRF protection DISABLED (IS_DEV_ENV); this must never be set in production") + postFetcher = jetstream.NewDevDirectPostFetcher(a.identityResolver) + } consumers = append(consumers, feedConsumer{ name: jetstream.ConsumerPosts, handler: jetstream.NewPostEventConsumer(a.postRepo, a.communityRepo, a.userService, a.db, jetstream.WithPostBridgeTrust(a.bridgeTrust), - jetstream.WithPostIdentityResolver(a.identityResolver)), + jetstream.WithPostIdentityResolver(a.identityResolver), + jetstream.WithAdmissions(a.admissionRepo), + jetstream.WithDeletedAccounts(postgresRepo.NewDeletedAccountRepository(a.db)), + jetstream.WithPostRecordFetcher(postFetcher)), }) // Aggregators: service declarations and authorization records, following diff --git a/cmd/server/routes.go b/cmd/server/routes.go index 8eef425..3bca2a3 100644 --- a/cmd/server/routes.go +++ b/cmd/server/routes.go @@ -73,7 +73,8 @@ func registerXRPCRoutes(r chi.Router, app *application) { // Posts accept dual auth so aggregator bots can publish with a service // JWT or API key rather than a user's OAuth session. routes.RegisterPostRoutes(r, app.postService, app.voteService, app.blueskyService, - app.dualAuth, app.authMiddleware) + app.dualAuth, app.authMiddleware, + routes.WithPostStatusService(app.postStatusService)) routes.RegisterVoteRoutes(r, app.voteService, app.authMiddleware) routes.RegisterUserBlockRoutes(r, app.userBlockService, app.authMiddleware) diff --git a/cmd/server/wiring.go b/cmd/server/wiring.go index d57e8eb..51a75a5 100644 --- a/cmd/server/wiring.go +++ b/cmd/server/wiring.go @@ -97,11 +97,19 @@ type application struct { commentRepo comments.Repository userBlockRepo userblocks.Repository aggregatorRepo aggregators.Repository + // admissionRepo is shared by the ingestion consumer, which WRITES the + // per-(community, post) decisions, and the status query, which reads them. + admissionRepo posts.AdmissionRepository // Domain services - userService users.UserService - communityService communities.Service - postService posts.Service + userService users.UserService + communityService communities.Service + postService posts.Service + // postStatusService answers post.getStatus. Separate from postService + // because a status query needs the admissions store and nothing else, and + // widening the write-path interface to reach it would make every test + // double of posts.Service carry a method it has no opinion about. + postStatusService posts.StatusService voteService votes.Service commentService comments.Service userBlockService userblocks.Service @@ -251,6 +259,7 @@ func (a *application) buildRepositories() { a.commentRepo = postgresRepo.NewCommentRepository(a.db) a.userBlockRepo = postgresRepo.NewUserBlockRepository(a.db) a.aggregatorRepo = postgresRepo.NewAggregatorRepository(a.db) + a.admissionRepo = postgresRepo.NewAdmissionRepository(a.db) } func (a *application) buildServices(ctx context.Context) error { @@ -348,6 +357,12 @@ func (a *application) buildServices(ctx context.Context) error { }), ) + // getStatus is how an author on another server learns what happened to + // their post. It is the ONLY way a rejection is reachable — a submission + // refused before it was ever accepted writes no community record, so there + // is nothing on the firehose to read. + a.postStatusService = posts.NewStatusService(a.admissionRepo) + // Subject existence is deliberately not validated: the vote is written to // the user's own PDS regardless, and the Jetstream consumer only updates // counts for subjects that still exist. Checking here would trade a diff --git a/internal/api/handlers/post/getstatus.go b/internal/api/handlers/post/getstatus.go index 21996f6..49c6545 100644 --- a/internal/api/handlers/post/getstatus.go +++ b/internal/api/handlers/post/getstatus.go @@ -1,15 +1,14 @@ package post import ( + "encoding/json" + "log" "net/http" + "time" "Coves/internal/core/posts" ) -// RED STUB (task 5, cycle 1). Signature only — HandleGetStatus writes nothing, -// so every assertion in getstatus_integration_test.go fails on the response -// rather than on a missing symbol. The implementation is GREEN's. - // GetStatusHandler serves social.coves.community.post.getStatus: one // community's decision about one post (docs/PRD_AUTHOR_OWNED_POSTS.md §3.4). // @@ -33,4 +32,72 @@ func NewGetStatusHandler(service posts.StatusService) *GetStatusHandler { // HandleGetStatus handles // GET /xrpc/social.coves.community.post.getStatus?post=at://...&community=did:... func (h *GetStatusHandler) HandleGetStatus(w http.ResponseWriter, r *http.Request) { + if r.Method != http.MethodGet { + http.Error(w, "Method not allowed", http.StatusMethodNotAllowed) + return + } + + postURI := r.URL.Query().Get("post") + communityDID := r.URL.Query().Get("community") + + // Both halves are refused rather than defaulted. A post carries independent + // decisions from several communities (§2), so an incomplete subject is not + // an under-specified question with an obvious answer — it is a different + // question, and answering about whichever row turned up first would report + // one community's verdict as another's. + if postURI == "" { + writeError(w, http.StatusBadRequest, "InvalidRequest", "post parameter is required") + return + } + if communityDID == "" { + writeError(w, http.StatusBadRequest, "InvalidRequest", "community parameter is required") + return + } + if len(postURI) > maxURILength { + writeError(w, http.StatusBadRequest, "InvalidRequest", "post URI exceeds maximum length") + return + } + if len(communityDID) > maxURILength { + writeError(w, http.StatusBadRequest, "InvalidRequest", "community DID exceeds maximum length") + return + } + + status, err := h.service.GetStatus(r.Context(), posts.GetStatusRequest{ + PostURI: postURI, + CommunityDID: communityDID, + }) + if err != nil { + handleServiceError(w, err) + return + } + + // Built field by field so an absent optional field is ABSENT from the JSON + // rather than present and null. The distinction is the client's: a caller + // polling for the accepted transition reads `"decisionCode": null` as a + // decision that was made, when in fact none exists. + body := map[string]interface{}{"status": string(status.Status)} + if status.DecisionCode != nil { + body["decisionCode"] = *status.DecisionCode + } + if status.DecisionAt != nil { + body["decisionAt"] = status.DecisionAt.UTC().Format(time.RFC3339) + } + if status.AcceptanceURI != nil { + body["acceptanceUri"] = *status.AcceptanceURI + } + + // Pre-encoded so an encoding failure still yields a proper error response + // rather than a 200 with a truncated body (mirrors post.get). + responseBytes, err := json.Marshal(body) + if err != nil { + log.Printf("ERROR: Failed to encode getStatus response: %v", err) + writeError(w, http.StatusInternalServerError, "InternalServerError", "Failed to encode response") + return + } + + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusOK) + if _, err := w.Write(responseBytes); err != nil { + log.Printf("ERROR: Failed to write getStatus response: %v", err) + } } diff --git a/internal/api/routes/post.go b/internal/api/routes/post.go index 53917c6..67dfd3b 100644 --- a/internal/api/routes/post.go +++ b/internal/api/routes/post.go @@ -10,6 +10,28 @@ import ( "github.com/go-chi/chi/v5" ) +// PostRouteOption supplies a collaborator that only some of the post routes +// need. +// +// It is variadic rather than another positional parameter because the status +// query is served by its OWN service (posts.StatusService, which needs the +// admissions store and nothing else) and every existing caller — including the +// minimal wirings in tests — would otherwise have to name a dependency it has +// no opinion about. +type PostRouteOption func(*postRouteConfig) + +type postRouteConfig struct { + statusService posts.StatusService +} + +// WithPostStatusService supplies the service behind +// social.coves.community.post.getStatus. The route is registered either way, so +// that the HTTP surface does not silently change shape with the wiring; without +// this option the handler has no service to call. +func WithPostStatusService(service posts.StatusService) PostRouteOption { + return func(c *postRouteConfig) { c.statusService = service } +} + // RegisterPostRoutes registers post-related XRPC endpoints on the router // Implements social.coves.community.post.* lexicon endpoints // authMiddleware can be either OAuthAuthMiddleware or DualAuthMiddleware (used for @@ -22,7 +44,12 @@ func RegisterPostRoutes( blueskyService blueskypost.Service, authMiddleware middleware.AuthMiddleware, oauthMiddleware *middleware.OAuthAuthMiddleware, + opts ...PostRouteOption, ) { + var cfg postRouteConfig + for _, opt := range opts { + opt(&cfg) + } // oauthMiddleware.OptionalAuth gates the public get endpoint below. A nil value is a // wiring bug (minimal/test setups) that would otherwise panic on the first request to // post.get; fail fast at registration with a clear message instead. @@ -50,6 +77,19 @@ func RegisterPostRoutes( // Used for feed-skeleton hydration and permalink / cold-load rendering. r.With(oauthMiddleware.OptionalAuth).Get("/xrpc/social.coves.community.post.get", getHandler.HandleGet) + // social.coves.community.post.getStatus - one community's admission + // decision about one post. + // + // NO auth middleware at all, unlike post.get beside it. The caller this is + // built for is an author on ANOTHER server asking this host whether it + // accepted their post (PRD §7): they have no account here, so there is no + // session to require and no viewer state to personalise. OptionalAuth would + // be harmless but pointless — the answer does not vary by viewer — while + // RequireAuth would make the cross-server case unanswerable, which is the + // asymmetry internal/api/routes/registration_test.go declares. + statusHandler := post.NewGetStatusHandler(cfg.statusService) + r.Get("/xrpc/social.coves.community.post.getStatus", statusHandler.HandleGetStatus) + // Future endpoints (Beta): // r.With(authMiddleware.RequireAuth).Post("/xrpc/social.coves.community.post.update", updateHandler.HandleUpdate) // r.Get("/xrpc/social.coves.community.post.list", listHandler.HandleList) diff --git a/internal/atproto/jetstream/authorpost.go b/internal/atproto/jetstream/authorpost.go index 2c44009..60acd84 100644 --- a/internal/atproto/jetstream/authorpost.go +++ b/internal/atproto/jetstream/authorpost.go @@ -2,17 +2,40 @@ package jetstream import ( "context" + "database/sql" + "encoding/json" + "errors" + "fmt" + "io" + "log" "net/http" + "net/url" + "strconv" + "strings" + "time" "Coves/internal/atproto/identity" + "Coves/internal/atproto/oauth" + "Coves/internal/core/communities" "Coves/internal/core/posts" + "Coves/internal/core/users" ) -// RED STUB (task 5, cycle 1). Declarations only — every function body here -// returns zero values, so the tests describing author-owned post ingestion -// compile and fail on their assertions rather than on missing symbols. The -// implementations, the HandleEvent dispatch for the three new collections, and -// the consumerWantedCollections entries are GREEN's. +// Ingesting author-owned posts and the community records that decide about +// them (docs/PRD_AUTHOR_OWNED_POSTS.md §5.3-§5.6). +// +// Three collections land here, and they invert each other: +// +// - social.coves.community.postv2 arrives from the AUTHOR's repo, so +// event.Did IS the author and the community is a claim the record makes. +// - social.coves.community.acceptance and .removal arrive from the +// COMMUNITY's repo, so event.Did IS the community and the post is a +// subject the record names. +// +// The old community.post path (still in post_consumer.go) checks that the repo +// DID EQUALS the record's community. Here that check splits in two opposite +// directions, which is why these handlers are their own file rather than more +// branches in the old ones. // PostV2Collection is the author-repo post record of // docs/PRD_AUTHOR_OWNED_POSTS.md §3.1 — the §3.0 successor to the deprecated @@ -60,6 +83,10 @@ func WithPostRecordFetcher(fetcher PostRecordFetcher) PostEventConsumerOption { return func(c *PostEventConsumer) { c.postFetcher = fetcher } } +// --------------------------------------------------------------------------- +// §5.4 direct fetch +// --------------------------------------------------------------------------- + // FetchedPost is one author-repo record read directly from its PDS. // // It carries the CID separately from the record because the CID is what the @@ -105,6 +132,19 @@ type DirectPostFetcher struct { allowPrivateHosts bool } +// maxFetchedRecordBytes bounds how much of a PDS getRecord response is read. +// +// A post record has a lexicon-bounded size, so a PDS streaming megabytes is +// either broken or hostile — and the host is chosen by a stranger's record, so +// an unbounded read here is a memory-exhaustion primitive handed to the public. +// The cap mirrors users.maxProfileResponseBytes, which bounds the same call for +// the same reason. +const maxFetchedRecordBytes = 1 << 20 // 1 MiB + +// maxFetchErrorDetailBytes caps how much of a failing PDS response is echoed +// into the logs, so a hostile host cannot flood them. +const maxFetchErrorDetailBytes = 256 + // NewDirectPostFetcher wires the §5.4 fetch. SSRF protection is ON and there is // no parameter to turn it off: a constructor that accepted a boolean is a // constructor someone eventually passes true to from production wiring. @@ -112,12 +152,912 @@ func NewDirectPostFetcher(resolver identity.Resolver) *DirectPostFetcher { return &DirectPostFetcher{resolver: resolver} } +// NewDevDirectPostFetcher builds a fetcher with the SSRF guard STOOD DOWN, for +// development and hermetic test stacks whose PDS is reachable only on a private +// address. +// +// It is a separate constructor rather than a flag on the safe one so that the +// dangerous choice has to be named at the call site, where a reviewer sees it, +// and so that no production wiring can reach it by passing a variable that +// happens to be true. Its one caller is gated on IS_DEV_ENV. +func NewDevDirectPostFetcher(resolver identity.Resolver) *DirectPostFetcher { + return &DirectPostFetcher{resolver: resolver, allowPrivateHosts: true} +} + // httpClient builds the guarded client for one fetch. Declared here so the // guard is derived from allowPrivateHosts at call time rather than baked into a // client at construction, where a test seam could not reach it. -func (f *DirectPostFetcher) httpClient() *http.Client { return nil } +func (f *DirectPostFetcher) httpClient() *http.Client { + return oauth.NewSSRFSafeHTTPClient(f.allowPrivateHosts) +} // FetchPost implements PostRecordFetcher. func (f *DirectPostFetcher) FetchPost(ctx context.Context, postURI string) (*FetchedPost, error) { - return nil, nil + repoDID, collection, rkey, ok := parseRecordURI(postURI) + if !ok { + return nil, fmt.Errorf("cannot fetch %q: not an at:// record URI", postURI) + } + if f.resolver == nil { + return nil, fmt.Errorf("cannot fetch %s: no identity resolver is wired", postURI) + } + + // The PDS is resolved from the DID document rather than taken from anything + // the acceptance record said. The record names a subject; where that + // subject's repo lives is a fact about the DID, and letting a record assert + // it would let the record choose the host this request goes to. + resolved, err := f.resolver.Resolve(ctx, repoDID) + if err != nil { + return nil, fmt.Errorf("resolving the repo of %s: %w", postURI, err) + } + if resolved == nil || resolved.PDSURL == "" { + return nil, fmt.Errorf("resolving the repo of %s: no PDS endpoint in the DID document", postURI) + } + + endpoint := strings.TrimSuffix(resolved.PDSURL, "/") + "/xrpc/com.atproto.repo.getRecord?repo=" + + url.QueryEscape(repoDID) + "&collection=" + url.QueryEscape(collection) + + "&rkey=" + url.QueryEscape(rkey) + + req, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil) + if err != nil { + return nil, fmt.Errorf("building the getRecord request for %s: %w", postURI, err) + } + + resp, err := f.httpClient().Do(req) + if err != nil { + return nil, fmt.Errorf("fetching %s from its PDS: %w", postURI, err) + } + defer func() { _ = resp.Body.Close() }() + + // One byte past the cap, so an over-cap body is DETECTED rather than + // silently truncated and then parsed as if it were whole. A truncated + // record that happened to parse would be indexed as the author's content. + body, err := io.ReadAll(io.LimitReader(resp.Body, maxFetchedRecordBytes+1)) + if err != nil { + return nil, fmt.Errorf("reading the getRecord response for %s: %w", postURI, err) + } + if len(body) > maxFetchedRecordBytes { + return nil, fmt.Errorf("the PDS serving %s returned more than %d bytes", postURI, maxFetchedRecordBytes) + } + + if resp.StatusCode != http.StatusOK { + detail := string(body) + if len(detail) > maxFetchErrorDetailBytes { + detail = detail[:maxFetchErrorDetailBytes] + } + // Quoted so control characters and ANSI escapes from a hostile PDS + // cannot corrupt log output. + return nil, fmt.Errorf("the PDS serving %s answered getRecord with status %d: %s", + postURI, resp.StatusCode, strconv.Quote(detail)) + } + + var parsed struct { + URI string `json:"uri"` + CID string `json:"cid"` + Value map[string]interface{} `json:"value"` + } + if err := json.Unmarshal(body, &parsed); err != nil { + return nil, fmt.Errorf("parsing the getRecord response for %s: %w", postURI, err) + } + if parsed.Value == nil { + return nil, fmt.Errorf("the getRecord response for %s carried no record value", postURI) + } + if parsed.CID == "" { + // Without a CID there is nothing to verify the pinned reference + // against, and an unverified record is exactly what the fetch must + // never index. + return nil, fmt.Errorf("the getRecord response for %s carried no CID", postURI) + } + + return &FetchedPost{URI: postURI, CID: parsed.CID, Record: parsed.Value}, nil +} + +// --------------------------------------------------------------------------- +// The author-repo post record +// --------------------------------------------------------------------------- + +// AuthorPostRecord is a social.coves.community.postv2 record as it arrives from +// Jetstream. +// +// It has NO author field, and that absence is enforced by the type rather than +// by discipline: authorship comes from the repo the record lives in, so a +// struct that could hold an author is a struct someone eventually reads one +// from — which is precisely the impersonation the flip removed. The lexicon has +// no such property either, so a record carrying one is a forger's field and is +// ignored here by construction. +type AuthorPostRecord struct { + Title *string `json:"title,omitempty"` + Content *string `json:"content,omitempty"` + Embed map[string]interface{} `json:"embed,omitempty"` + Labels *posts.SelfLabels `json:"labels,omitempty"` + BridgedStats *BridgedStatsFromJetstream `json:"bridgedStats,omitempty"` + Type string `json:"$type"` + Community string `json:"community"` + CreatedAt string `json:"createdAt"` + Facets []interface{} `json:"facets,omitempty"` +} + +// parseAuthorPostRecord converts a raw Jetstream record map into an +// AuthorPostRecord, refusing the shapes that can never become valid. +func parseAuthorPostRecord(record map[string]interface{}) (*AuthorPostRecord, error) { + recordJSON, err := json.Marshal(record) + if err != nil { + return nil, fmt.Errorf("failed to marshal postv2 record: %w", err) + } + + var parsed AuthorPostRecord + if err := json.Unmarshal(recordJSON, &parsed); err != nil { + // PERMANENT: the record's shape doesn't match the lexicon (wrong field + // types); replaying the identical bytes can never parse differently. + return nil, fmt.Errorf("%w: failed to unmarshal postv2 record: %v", ErrPermanentEvent, err) + } + + // PERMANENT for the same reason: a record missing a required field is + // structurally invalid forever. `community` is the submission target, so a + // record without one names no admission subject at all. + if parsed.Community == "" { + return nil, fmt.Errorf("%w: postv2 record missing community field", ErrPermanentEvent) + } + if parsed.CreatedAt == "" { + return nil, fmt.Errorf("%w: postv2 record missing createdAt field", ErrPermanentEvent) + } + + return &parsed, nil +} + +// --------------------------------------------------------------------------- +// postv2: the author's own post record +// --------------------------------------------------------------------------- + +// handleAuthorPostEvent routes one social.coves.community.postv2 commit. +// +// event.Did is the AUTHOR, unconditionally and for every operation below. +func (c *PostEventConsumer) handleAuthorPostEvent(ctx context.Context, event *JetstreamEvent, commit *CommitEvent) error { + authorDID := event.Did + + // THE ERASURE GATE, and it runs before anything else touches the database — + // before parsing, before hydration, before the rev gate. An event from an + // account this AppView was asked to forget has nothing to do, and the + // cheapest way to guarantee that is to leave before any code path that + // could write. Ordering it after a parse would also mean a malformed record + // from an erased account dead-letters, which is operational noise about + // content nobody may keep. + erased, err := c.authorWasErased(ctx, authorDID) + if err != nil { + return err + } + if erased { + // Nil, not an error. The connector dead-letters whatever a handler + // returns, so refusing here would fill the queue with rows that redrive, + // fail identically and retire — every erased account becoming a + // permanent stream of noise. This is not a failure; it is an event with + // nothing to do. + log.Printf("INFO: dropping %s %s for erased account %s (migration 036 marker)", + PostV2Collection, commit.Operation, authorDID) + return nil + } + + switch commit.Operation { + case "create", "update": + return c.upsertAuthorPost(ctx, authorDID, commit, event.TimeUS) + case "delete": + return c.tombstoneRecord(ctx, recordURI(authorDID, PostV2Collection, commit.RKey), commit.Rev) + } + return nil +} + +// canRecordAdmissions reports whether this consumer has somewhere to put a +// decision. +// +// Every one of the three author-owned collections exists to write an admission +// row, so a consumer built without the store cannot handle any of them — it is +// running in its pre-034 shape. Ignoring the events is the honest answer: +// indexing a postv2 with no admission row would publish a post no community +// ever decided about, which is worse than not indexing it at all. It is logged +// because in production this is always a wiring bug. +func (c *PostEventConsumer) canRecordAdmissions(collection string) bool { + if c.admissions != nil { + return true + } + log.Printf("WARNING: ignoring %s event - this consumer has no admissions store wired", collection) + return false +} + +// authorWasErased reports whether this DID carries a migration-036 erasure +// marker. A lookup failure is an ERROR, never a false: failing open would +// re-index the content a deletion erased, which is the one outcome the marker +// exists to prevent. With no lookup wired the gate is absent and everything +// indexes, which is the pre-036 behaviour. +func (c *PostEventConsumer) authorWasErased(ctx context.Context, did string) (bool, error) { + if c.deletedAccounts == nil { + return false, nil + } + erased, err := c.deletedAccounts.IsAccountDeleted(ctx, did) + if err != nil { + return false, fmt.Errorf("checking the erasure marker for %s: %w", did, err) + } + return erased, nil +} + +// upsertAuthorPost indexes an author-repo post and opens (or refreshes) the +// pending admission the community will decide against. +// +// A postv2 event never DECIDES anything. It records content plus the fact that +// the author submitted it; whether the community shows the post lives in +// community_post_admissions and is written only by community events. +func (c *PostEventConsumer) upsertAuthorPost(ctx context.Context, authorDID string, commit *CommitEvent, timeUS int64) error { + if commit.Record == nil { + return fmt.Errorf("%w: postv2 %s event missing record data", ErrPermanentEvent, commit.Operation) + } + + record, err := parseAuthorPostRecord(commit.Record) + if err != nil { + return err + } + + uri := recordURI(authorDID, PostV2Collection, commit.RKey) + + // The community must be one this AppView has indexed, or there is no + // subject to open an admission against. + // + // Deliberately NOT permanent: BigSky preserves order within a repo, not + // across repos, so a post can genuinely arrive before the community's own + // profile event. Marking this permanent would discard every post that + // merely arrived early, with the redrive that would have fixed it already + // spent. + if _, err := c.communityRepo.GetByDID(ctx, record.Community); err != nil { + if communities.IsNotFound(err) { + log.Printf("Error: cannot index %s before its community %s is indexed", uri, record.Community) + return fmt.Errorf("community not found: %s - cannot index post before community", record.Community) + } + return fmt.Errorf("%w: failed to verify community %s exists: %v", errValidationInfra, record.Community, err) + } + + stored, found, err := c.loadStoredPost(ctx, uri) + if err != nil { + return err + } + + // IMMUTABILITY (§3.1): an update that changes `community` invalidates the + // WHOLE event. Not merely the community field — applying the content while + // keeping the old community would leave the first community's admission + // holding a CID it never evaluated, publishing content nobody judged under + // a standing acceptance. Retargeting a post means writing a new record. + // + // A skip, not an error: an invalid record from a stranger's repo is not an + // infrastructure failure, and dead-lettering it would retry a record that + // can never become valid. + if found && stored.communityDID != record.Community { + log.Printf("🚨 SECURITY: ignoring the whole %s update for %s - community is immutable (stored %s, incoming %s)", + PostV2Collection, uri, stored.communityDID, record.Community) + return nil + } + + // Provenance for bridgedStats is keyed on the AUTHOR's PDS now, because the + // record lives in the author's repo. The community's host has no say over + // what an author asserts about their own record any more, so checking the + // community's PDS (as the community-repo path does) would trust the wrong + // party entirely. An author this AppView holds no row for — the ordinary + // unhydrated federated author — has no provenance to prove, so the gate + // default-denies. + up, down, asOf := c.trustedBridgedStats(ctx, authorDID, record.BridgedStats, uri) + + facetsJSON, embedJSON, labelsJSON, err := serializePostContent( + sanitizeFacets(record.Facets, record.Content, uri), record.Embed, record.Labels) + if err != nil { + return err + } + + var applied bool + if found { + applied, err = c.applyPostContentUpdate(ctx, postContentUpdate{ + uri: uri, storedID: stored.id, rev: commit.Rev, cid: commit.CID, + title: record.Title, content: record.Content, + facets: facetsJSON, embed: embedJSON, labels: labelsJSON, + bridgedUpvotes: up, bridgedDownvotes: down, bridgedAsOf: asOf, + storedAsOf: stored.bridgedAsOf, storedDeletedAt: stored.deletedAt, + storedIndexedAt: stored.indexedAt, timeUS: timeUS, + }) + if err != nil { + return err + } + } else { + applied, err = c.insertAuthorPost(ctx, authorPostInsert{ + uri: uri, authorDID: authorDID, record: record, commit: commit, timeUS: timeUS, + facets: facetsJSON, embed: embedJSON, labels: labelsJSON, + bridgedUpvotes: up, bridgedDownvotes: down, bridgedAsOf: asOf, + }) + if err != nil { + return err + } + } + + if !applied { + // The rev gate or the recency guard refused this event: a newer state is + // already indexed. Opening or refreshing an admission from it would move + // evaluated_cid BACKWARDS onto content the row no longer holds, which is + // how an accepted post gets flipped to pending_reacceptance by a + // duplicate delivery. + return nil + } + + // The author is looked up only after the content is safely indexed, and a + // failure here is never fatal. Under §5.3 an author this AppView has never + // seen is a normal state that must index anyway, so hydration is an + // enrichment: getting a profile row for them is nice, and not getting one + // must not cost the post. + c.hydrateAuthorOpportunistically(ctx, authorDID) + + // The admission row: PENDING, always. The post claims the community; the + // community has said nothing. A row opened as anything else would publish + // speech the community never agreed to carry. + // + // It follows the content write rather than sharing its transaction, which + // leaves one bounded window: a failure between the two indexes the post + // without opening its admission, and the redrive is then rev-gated away, so + // the repair needs the record re-emitted. The alternative — opening the + // admission first — trades that for a phantom moderation-queue entry for a + // post that was never indexed, which is the worse of the two because it is + // invisible to the operator rather than visible as a dead letter. + if _, err := c.admissions.UpsertPending(ctx, posts.UpsertPendingCommand{ + CommunityDID: record.Community, + PostURI: uri, + EvaluatedCID: commit.CID, + }); err != nil { + return fmt.Errorf("recording the pending admission for %s in %s: %w", uri, record.Community, err) + } + + log.Printf("✓ Indexed author post: %s (author: %s, community: %s)", uri, authorDID, record.Community) + return nil +} + +// hydrateAuthorOpportunistically indexes a minimal profile for an author this +// AppView has not seen, so their posts are not permanently authorless. +// +// Bounded and non-fatal, both deliberately. Bounded because it is an outbound +// resolution on the hot firehose path driven by an identifier a stranger chose; +// non-fatal because §5.3 makes indexing the post the obligation and the profile +// merely an enrichment — refusing the event over a slow PLC lookup would +// reinstate exactly the refusal the flip removed. +func (c *PostEventConsumer) hydrateAuthorOpportunistically(ctx context.Context, authorDID string) { + if c.identityResolver == nil { + return + } + + if _, err := c.userService.GetUserByDID(ctx, authorDID); err == nil { + return // already indexed + } else if !errors.Is(err, users.ErrUserNotFound) { + log.Printf("debug: skipping author hydration for %s (lookup failed: %v)", authorDID, err) + return + } + + hydrateCtx, cancel := context.WithTimeout(ctx, authorHydrationTimeout) + defer cancel() + + resolved, err := c.identityResolver.Resolve(hydrateCtx, authorDID) + if err != nil || resolved == nil { + log.Printf("debug: could not resolve post author %s for hydration: %v", authorDID, err) + return + } + // The resolver returns a bidirectionally verified handle, or the reserved + // "handle.invalid" when verification failed. Indexing the latter would + // write a placeholder into a column with a uniqueness constraint, so the + // second unverifiable author would collide with the first. + if resolved.DID != authorDID || resolved.Handle == "" || resolved.Handle == invalidHandle { + log.Printf("debug: not hydrating post author %s (unverified identity)", authorDID) + return + } + + if err := c.userService.IndexUser(hydrateCtx, resolved.DID, resolved.Handle, resolved.PDSURL); err != nil { + log.Printf("debug: could not hydrate post author %s: %v", authorDID, err) + } +} + +// invalidHandle is the reserved handle atProto identity resolution reports when +// a DID's handle cannot be bidirectionally verified. +const invalidHandle = "handle.invalid" + +// authorHydrationTimeout bounds the opportunistic identity resolution above. +// Short on purpose: it is an enrichment on the firehose path, so it must never +// become the reason events back up. +const authorHydrationTimeout = 5 * time.Second + +// trustedBridgedStats returns the bridged aggregate to apply for an author-repo +// post, or a nil asOf meaning "leave the stored bridged columns alone". +// +// Default-deny at every step: no aggregate, no users row for the author, a PDS +// outside the trusted bridge set, or an aggregate failing input hygiene all +// return nothing to apply. +func (c *PostEventConsumer) trustedBridgedStats(ctx context.Context, authorDID string, stats *BridgedStatsFromJetstream, uri string) (int, int, *time.Time) { + if stats == nil { + return 0, 0, nil + } + + author, err := c.userService.GetUserByDID(ctx, authorDID) + if err != nil || author == nil { + log.Printf("debug: ignoring bridgedStats on %s (no indexed author %s to prove provenance)", uri, authorDID) + return 0, 0, nil + } + if !c.bridgeTrust.TrustsPDS(author.PDSURL) { + log.Printf("debug: ignoring bridgedStats on %s from untrusted author repo %s", uri, authorDID) + return 0, 0, nil + } + up, down, asOf, ok := validatedBridgedStats(stats, uri) + if !ok { + return 0, 0, nil + } + return up, down, &asOf +} + +// authorPostInsert is everything the first indexing of an author-repo post +// needs, gathered so the insert reads as one decision rather than a dozen +// positional arguments. +type authorPostInsert struct { + uri string + authorDID string + record *AuthorPostRecord + commit *CommitEvent + timeUS int64 + facets sql.NullString + embed sql.NullString + labels sql.NullString + bridgedUpvotes int + bridgedDownvotes int + bridgedAsOf *time.Time +} + +// insertAuthorPost indexes a post the AppView has never held. It reports +// whether the write applied — false means the rev gate refused the event. +func (c *PostEventConsumer) insertAuthorPost(ctx context.Context, in authorPostInsert) (bool, error) { + createdAt := parseRecordCreatedAt(in.record.CreatedAt, in.uri) + + post := &posts.Post{ + URI: in.uri, + CID: in.commit.CID, + RKey: in.commit.RKey, + // THE AUTHOR IS THE REPO. Not a field, not a lookup — deriving it from + // anywhere else is what would let any repo claim any author. + AuthorDID: in.authorDID, + // The community is a CLAIM the record makes: the author's submission + // target, which the community has not yet agreed to. + CommunityDID: in.record.Community, + Title: in.record.Title, + Content: in.record.Content, + ContentFacets: nullableString(in.facets), + Embed: nullableString(in.embed), + ContentLabels: nullableString(in.labels), + CreatedAt: createdAt, + IndexedAt: indexedAtForEvent(in.timeUS), + } + if in.bridgedAsOf != nil { + post.BridgedUpvoteCount = in.bridgedUpvotes + post.BridgedDownvoteCount = in.bridgedDownvotes + post.BridgedStatsAsOf = in.bridgedAsOf + post.Score = in.bridgedUpvotes - in.bridgedDownvotes + } + + // The rev gate decides inside this transaction, so a refusal and the writes + // it refuses can never half-apply. A gate skip surfaces as applied=false. + applied, err := c.indexPostIfRevWins(ctx, post, in.commit.Rev) + if err != nil { + return false, fmt.Errorf("failed to index author post %s: %w", in.uri, err) + } + return applied, nil +} + +// --------------------------------------------------------------------------- +// acceptance and removal: the community's decision records +// --------------------------------------------------------------------------- + +// communityDecisionRecord is an acceptance or a removal as it arrives from +// Jetstream. Both name their subject by strongRef; only a removal carries a +// code. +type communityDecisionRecord struct { + Type string `json:"$type"` + Subject struct { + URI string `json:"uri"` + CID string `json:"cid"` + } `json:"subject"` + Code string `json:"code,omitempty"` + Reason string `json:"reason,omitempty"` + CreatedAt string `json:"createdAt"` +} + +// parseCommunityDecision converts a raw acceptance/removal record, refusing the +// shapes that can never become valid. +func parseCommunityDecision(record map[string]interface{}, collection string) (*communityDecisionRecord, error) { + recordJSON, err := json.Marshal(record) + if err != nil { + return nil, fmt.Errorf("failed to marshal %s record: %w", collection, err) + } + + var parsed communityDecisionRecord + if err := json.Unmarshal(recordJSON, &parsed); err != nil { + return nil, fmt.Errorf("%w: failed to unmarshal %s record: %v", ErrPermanentEvent, collection, err) + } + if parsed.Subject.URI == "" { + return nil, fmt.Errorf("%w: %s record names no subject", ErrPermanentEvent, collection) + } + // The pinned CID is half the decision, not decoration: agreeing to a URI is + // not agreeing to whatever that URI holds later. A record without one gives + // the consumer nothing to compare against the indexed content. + if parsed.Subject.CID == "" { + return nil, fmt.Errorf("%w: %s record for %s pins no CID", ErrPermanentEvent, collection, parsed.Subject.URI) + } + if collection == posts.RemovalCollection && parsed.Code == "" { + return nil, fmt.Errorf("%w: removal record for %s carries no code", ErrPermanentEvent, parsed.Subject.URI) + } + return &parsed, nil +} + +// handleCommunityDecisionEvent routes one acceptance or removal commit. +// +// event.Did is the COMMUNITY, and that is the only thing in the event that says +// which community decided — the record names a post, not a decider. So the repo +// has to BE an indexed community: taking an arbitrary repo at its word would let +// anyone with a PDS publish into any feed by writing a record about someone +// else's post. +func (c *PostEventConsumer) handleCommunityDecisionEvent(ctx context.Context, event *JetstreamEvent, commit *CommitEvent) error { + communityDID := event.Did + + if _, err := c.communityRepo.GetByDID(ctx, communityDID); err != nil { + if communities.IsNotFound(err) { + // Transient, and the reason is delivery order rather than leniency: + // a community's first acceptance can genuinely outrun its own + // profile event, and marking this permanent would spend the redrive + // budget that resolves the race and discard a real decision. + log.Printf("🚨 SECURITY: refusing %s from %s - not an indexed community repo", + commit.Collection, communityDID) + return fmt.Errorf("community not found: %s - cannot apply a %s from a repo that is not an indexed community", + communityDID, commit.Collection) + } + return fmt.Errorf("%w: failed to verify community %s exists: %v", errValidationInfra, communityDID, err) + } + + if commit.Operation == "delete" { + return c.applyCommunityDecisionDelete(ctx, communityDID, commit) + } + + if commit.Record == nil { + return fmt.Errorf("%w: %s %s event missing record data", ErrPermanentEvent, commit.Collection, commit.Operation) + } + decision, err := parseCommunityDecision(commit.Record, commit.Collection) + if err != nil { + return err + } + + // The subject's author, and therefore the erasure gate. An admission row + // for an erased account's post is exactly the row migration 036 exists to + // stop being recreated, and an acceptance replayed months later is one of + // the two ways it comes back. + subjectAuthor, subjectCollection, _, ok := parseRecordURI(decision.Subject.URI) + if !ok { + return fmt.Errorf("%w: %s names subject %q, which is not an at:// record URI", + ErrPermanentEvent, commit.Collection, decision.Subject.URI) + } + erased, err := c.authorWasErased(ctx, subjectAuthor) + if err != nil { + return err + } + if erased { + log.Printf("INFO: dropping %s %s about %s - its author was erased", + commit.Collection, commit.Operation, decision.Subject.URI) + return nil + } + + watermark := posts.CommunityWatermark{Rev: commit.Rev} + + switch commit.Collection { + case posts.AcceptanceCollection: + return c.applyAcceptance(ctx, communityDID, commit, decision, subjectCollection, watermark) + case posts.RemovalCollection: + return c.applyRemoval(ctx, communityDID, decision, watermark) + } + return nil +} + +// applyAcceptance records a community's agreement to exactly one version of one +// post, converging on the subject first when the AppView has never seen it. +func (c *PostEventConsumer) applyAcceptance( + ctx context.Context, + communityDID string, + commit *CommitEvent, + decision *communityDecisionRecord, + subjectCollection string, + watermark posts.CommunityWatermark, +) error { + indexedCommunity, indexed, err := c.indexedPostCommunity(ctx, decision.Subject.URI) + if err != nil { + return err + } + switch { + case !indexed: + if err := c.convergeOnAcceptedSubject(ctx, communityDID, decision, subjectCollection); err != nil { + return err + } + case indexedCommunity != communityDID: + // The same refusal the fetch path makes, on the path where the post is + // already indexed. Both are the §10.2 rule: a community accepting a + // post that names a DIFFERENT community is the fork/import flow, which + // the data model supports and nothing is built for — so today it is a + // community pulling another community's content into its feed on its + // own say-so. Enforcing it in only one of the two places would leave + // the whole check bypassable by getting the post indexed first, which + // an attacker controls: they simply post before they accept. + // + // PERMANENT: a post's community is immutable across updates (§3.1), so + // no retry makes this valid. + return fmt.Errorf("%w: %s was submitted to community %s, but the acceptance came from %s (the fork/import flow is not built)", + ErrPermanentEvent, decision.Subject.URI, indexedCommunity, communityDID) + } + + result, err := c.admissions.ApplyAcceptance(ctx, posts.ApplyAcceptanceCommand{ + CommunityDID: communityDID, + PostURI: decision.Subject.URI, + AcceptanceURI: recordURI(communityDID, posts.AcceptanceCollection, commit.RKey), + AcceptanceRkey: commit.RKey, + PinnedCID: decision.Subject.CID, + Watermark: watermark, + }) + if err != nil { + return fmt.Errorf("applying the acceptance of %s in %s: %w", decision.Subject.URI, communityDID, err) + } + logAdmissionOutcome(posts.AcceptanceCollection, communityDID, decision.Subject.URI, result.Outcome) + return nil +} + +// convergeOnAcceptedSubject reads an accepted post straight from its author's +// PDS and indexes it (§5.4). +// +// Redrive alone cannot solve acceptance-before-post: bounded retries cannot +// manufacture an event that a relay-coverage gap will never deliver. This is +// the mechanism that makes convergence a guarantee rather than a bet on full +// relay coverage — and because it is an outbound request whose destination is +// chosen by a stranger's record, most of what follows is refusals. +func (c *PostEventConsumer) convergeOnAcceptedSubject( + ctx context.Context, + communityDID string, + decision *communityDecisionRecord, + subjectCollection string, +) error { + if subjectCollection != PostV2Collection { + // An acceptance is about an author-repo post. A subject in any other + // collection is not a thing this community can accept, and no retry + // changes which collection a URI names. + return fmt.Errorf("%w: acceptance names subject %s, which is not a %s record", + ErrPermanentEvent, decision.Subject.URI, PostV2Collection) + } + if c.postFetcher == nil { + // Without the fetch the only convergence mechanism left is redrive, so + // the event must stay retryable rather than be dropped. + return fmt.Errorf("acceptance for unindexed post %s: no direct fetcher is wired, so only redrive can converge", + decision.Subject.URI) + } + + fetched, err := c.postFetcher.FetchPost(ctx, decision.Subject.URI) + if err != nil { + // Transient: a PDS that is down, slow, or briefly unreachable is the + // ordinary case, and the redrive is what it is for. + return fmt.Errorf("fetching the accepted post %s directly: %w", decision.Subject.URI, err) + } + + // THE CID CHECK IS WHAT MAKES THE FETCH TRUSTWORTHY AT ALL. Without it the + // AppView indexes whatever the author's PDS chooses to serve under that + // rkey — the author (or whoever holds their keys) picks the content, and + // the community's signed acceptance is made to cover it retroactively. + // + // PERMANENT: the pinned version is gone from the repo and no retry brings + // it back, so re-fetching the same mismatch ten times is pure noise. + if fetched.CID != decision.Subject.CID { + return fmt.Errorf("%w: the PDS serving %s returned CID %s, but the acceptance pinned %s", + ErrPermanentEvent, decision.Subject.URI, fetched.CID, decision.Subject.CID) + } + + record, err := parseAuthorPostRecord(fetched.Record) + if err != nil { + return err + } + + // Cross-community acceptance is the privileged fork/import flow, and §10.2 + // is explicit that it is deliberately NOT built. Until it exists, a + // community accepting a post that names someone else is a community pulling + // another community's content into its feed on its own say-so. + // + // PERMANENT: the record's community field is immutable across updates + // (§3.1), so this can never become valid. + if record.Community != communityDID { + return fmt.Errorf("%w: %s names community %s, but the acceptance came from %s (the fork/import flow is not built)", + ErrPermanentEvent, decision.Subject.URI, record.Community, communityDID) + } + + authorDID, _, rkey, _ := parseRecordURI(decision.Subject.URI) + + facetsJSON, embedJSON, labelsJSON, err := serializePostContent( + sanitizeFacets(record.Facets, record.Content, decision.Subject.URI), record.Embed, record.Labels) + if err != nil { + return err + } + up, down, asOf := c.trustedBridgedStats(ctx, authorDID, record.BridgedStats, decision.Subject.URI) + + // The fetch is not a firehose event, so it carries no rev to gate on and no + // event time to stamp: an empty rev bypasses the gate (which is correct — a + // later real event for this record still wins on its own rev) and the + // watermark falls back to wall clock. + if _, err := c.insertAuthorPost(ctx, authorPostInsert{ + uri: decision.Subject.URI, + authorDID: authorDID, + record: record, + commit: &CommitEvent{ + Operation: "create", Collection: PostV2Collection, + RKey: rkey, CID: fetched.CID, + }, + facets: facetsJSON, embed: embedJSON, labels: labelsJSON, + bridgedUpvotes: up, bridgedDownvotes: down, bridgedAsOf: asOf, + }); err != nil { + return err + } + + // The pending admission comes with it. ApplyAcceptance would create a row + // on its own, but one with no evaluated content: recording what was indexed + // is what lets the next author edit be recognised as an edit. + if _, err := c.admissions.UpsertPending(ctx, posts.UpsertPendingCommand{ + CommunityDID: communityDID, + PostURI: decision.Subject.URI, + EvaluatedCID: fetched.CID, + }); err != nil { + return fmt.Errorf("recording the pending admission for fetched post %s: %w", decision.Subject.URI, err) + } + + c.hydrateAuthorOpportunistically(ctx, authorDID) + log.Printf("✓ Converged on accepted post %s by direct fetch (author: %s, community: %s)", + decision.Subject.URI, authorDID, communityDID) + return nil +} + +// applyRemoval records a community's moderation decision about a post. +// +// A removal with no prior acceptance is VALID: a community that has decided in +// advance about a post — an author it is about to ban, content it has already +// seen elsewhere — must be able to say so, and requiring an acceptance first +// would drop exactly the decisions a community most wants to make early. No +// direct fetch either: a removed post is not rendered, so there is nothing to +// converge on, and fetching content in order to hide it would hand a moderation +// record the power to make the AppView dial an arbitrary host. +func (c *PostEventConsumer) applyRemoval( + ctx context.Context, + communityDID string, + decision *communityDecisionRecord, + watermark posts.CommunityWatermark, +) error { + result, err := c.admissions.ApplyRemoval(ctx, posts.ApplyRemovalCommand{ + CommunityDID: communityDID, + PostURI: decision.Subject.URI, + DecisionCode: decision.Code, + Watermark: watermark, + }) + if err != nil { + return fmt.Errorf("applying the removal of %s in %s: %w", decision.Subject.URI, communityDID, err) + } + logAdmissionOutcome(posts.RemovalCollection, communityDID, decision.Subject.URI, result.Outcome) + return nil +} + +// applyCommunityDecisionDelete applies the withdrawal of an acceptance or a +// removal. +// +// A delete event carries NO record, so the subject cannot be read from it — and +// the rkey is a SHA-256 digest of the subject URI (§3.2), which is one-way. The +// subject is therefore recoverable only from state the AppView already holds: +// acceptance_rkey, which is stored exactly while an acceptance stands, i.e. +// exactly when an acceptance deletion has something to withdraw. +// +// A removal deletion has no such column and needs none. Every moderation commit +// is a PAIR (§3.3) — the removal commit is {acceptance-delete, removal-put} and +// the restore commit is {removal-delete, acceptance-put} — and the put half +// carries the subject in-record and outranks its paired delete under the §5.2 +// tuple. So the put alone converges the row whichever half arrives first, and +// an unresolvable delete is a no-op rather than a lost transition. The lone +// acceptance deletion, which the host writes when an author deletes their post +// (§5.3), is the one delete that arrives unpaired — and it is the one this +// lookup resolves. +func (c *PostEventConsumer) applyCommunityDecisionDelete(ctx context.Context, communityDID string, commit *CommitEvent) error { + if commit.Collection != posts.AcceptanceCollection { + log.Printf("INFO: %s deletion in %s carries no subject and is superseded by its paired write; skipping", + commit.Collection, communityDID) + return nil + } + + var postURI string + err := c.db.QueryRowContext(ctx, + `SELECT post_uri FROM community_post_admissions + WHERE community_did = $1 AND acceptance_rkey = $2`, + communityDID, commit.RKey, + ).Scan(&postURI) + if errors.Is(err, sql.ErrNoRows) { + // No acceptance of that rkey stands here — the removal half of the same + // commit already cleared it, or this AppView never saw the acceptance. + // Either way there is nothing to withdraw. + log.Printf("INFO: acceptance deletion %s/%s matches no standing acceptance; nothing to withdraw", + communityDID, commit.RKey) + return nil + } + if err != nil { + return fmt.Errorf("resolving the subject of acceptance deletion %s/%s: %w", communityDID, commit.RKey, err) + } + + result, err := c.admissions.ApplyAcceptanceDelete(ctx, posts.CommunityDeleteCommand{ + CommunityDID: communityDID, + PostURI: postURI, + Watermark: posts.CommunityWatermark{Rev: commit.Rev}, + }) + if err != nil { + return fmt.Errorf("withdrawing the acceptance of %s in %s: %w", postURI, communityDID, err) + } + logAdmissionOutcome(posts.AcceptanceCollection+"#delete", communityDID, postURI, result.Outcome) + return nil +} + +// logAdmissionOutcome records what a community event DID, including the skips. +// +// A skip is the ordering gate working — a multi-feed duplicate, a dead-letter +// redrive, an event superseded by its own commit's other half — so it is logged +// rather than returned as an error, which would bury healthy skips in the +// dead-letter queue among genuine failures. +func logAdmissionOutcome(collection, communityDID, postURI string, outcome posts.AdmissionOutcome) { + if outcome == posts.AdmissionApplied { + log.Printf("✓ Applied %s for %s in %s", collection, postURI, communityDID) + return + } + log.Printf("admission: %s for %s in %s was %s (an outcome, not a failure)", + collection, postURI, communityDID, outcome) +} + +// --------------------------------------------------------------------------- +// small shared helpers +// --------------------------------------------------------------------------- + +// recordURI builds the AT-URI of a record from its repo, collection and rkey. +func recordURI(repoDID, collection, rkey string) string { + return fmt.Sprintf("at://%s/%s/%s", repoDID, collection, rkey) +} + +// parseRecordURI splits at:////. It reports ok=false +// for anything else, including URIs with extra path segments — a subject the +// AppView cannot address is a subject it must refuse rather than guess at. +func parseRecordURI(uri string) (repoDID, collection, rkey string, ok bool) { + rest, found := strings.CutPrefix(uri, "at://") + if !found { + return "", "", "", false + } + parts := strings.Split(rest, "/") + if len(parts) != 3 || parts[0] == "" || parts[1] == "" || parts[2] == "" { + return "", "", "", false + } + return parts[0], parts[1], parts[2], true +} + +// indexedPostCommunity returns the community an indexed post was submitted to. +// +// Soft-deleted rows COUNT as indexed: a tombstoned post has been seen, and +// treating it as absent would send the direct fetch to resurrect content its +// author deleted. +func (c *PostEventConsumer) indexedPostCommunity(ctx context.Context, uri string) (string, bool, error) { + var communityDID string + err := c.db.QueryRowContext(ctx, `SELECT community_did FROM posts WHERE uri = $1`, uri).Scan(&communityDID) + if errors.Is(err, sql.ErrNoRows) { + return "", false, nil + } + if err != nil { + return "", false, fmt.Errorf("reading the indexed community of %s: %w", uri, err) + } + return communityDID, true, nil +} + +// nullableString converts a serialized JSON column back to the pointer shape +// posts.Post uses, where nil means "the record carried none". +func nullableString(v sql.NullString) *string { + if !v.Valid { + return nil + } + s := v.String + return &s } diff --git a/internal/atproto/jetstream/feeds.go b/internal/atproto/jetstream/feeds.go index 3109e15..161bc79 100644 --- a/internal/atproto/jetstream/feeds.go +++ b/internal/atproto/jetstream/feeds.go @@ -63,8 +63,21 @@ var consumerWantedCollections = map[string][]string{ "social.coves.community.subscription", "social.coves.community.block", }, + // One consumer for all four post-related collections, because they decide + // about each other: an acceptance is meaningless without the postv2 it + // pins, and both write the same admission row. Splitting them across + // connectors would give the two halves independent cursors and independent + // dead letters for one conversation. + // + // social.coves.community.post stays subscribed even though it is DEPRECATED + // (§3.0): the records already written to community repos keep arriving, and + // dropping the filter would silently stop indexing edits and deletes of + // every post that exists today. ConsumerPosts: { "social.coves.community.post", + "social.coves.community.postv2", + "social.coves.community.acceptance", + "social.coves.community.removal", }, ConsumerAggregators: { "social.coves.aggregator.service", diff --git a/internal/atproto/jetstream/post_consumer.go b/internal/atproto/jetstream/post_consumer.go index dc2b6b6..b5ce0ee 100644 --- a/internal/atproto/jetstream/post_consumer.go +++ b/internal/atproto/jetstream/post_consumer.go @@ -94,8 +94,10 @@ func (c *PostEventConsumer) HandleEvent(ctx context.Context, event *JetstreamEve commit := event.Commit - // Handle post record operations - if commit.Collection == "social.coves.community.post" { + switch commit.Collection { + // The DEPRECATED community-repo post (§3.0). Here the repo DID must EQUAL + // the record's community; the three collections below invert that. + case "social.coves.community.post": switch commit.Operation { case "create": return c.createPost(ctx, event.Did, commit, event.TimeUS) @@ -104,6 +106,22 @@ func (c *PostEventConsumer) HandleEvent(ctx context.Context, event *JetstreamEve case "delete": return c.deletePost(ctx, event.Did, commit) } + + // The author-repo post: event.Did IS the author, and the community is a + // claim the record makes (authorpost.go). + case PostV2Collection: + if !c.canRecordAdmissions(commit.Collection) { + return nil + } + return c.handleAuthorPostEvent(ctx, event, commit) + + // The community's decision records: event.Did IS the community, and the + // post is a subject the record names (authorpost.go). + case posts.AcceptanceCollection, posts.RemovalCollection: + if !c.canRecordAdmissions(commit.Collection) { + return nil + } + return c.handleCommunityDecisionEvent(ctx, event, commit) } // Silently ignore other operations and other collections @@ -161,22 +179,7 @@ func (c *PostEventConsumer) createPost(ctx context.Context, repoDID string, comm // Format: at://community_did/social.coves.community.post/rkey uri := fmt.Sprintf("at://%s/social.coves.community.post/%s", repoDID, commit.RKey) - // Parse timestamp from record - createdAt, err := time.Parse(time.RFC3339, postRecord.CreatedAt) - if err != nil { - // Fallback to current time if parsing fails - log.Printf("Warning: Failed to parse createdAt timestamp, using current time: %v", err) - createdAt = time.Now() - } - - // SECURITY: Clamp future timestamps to now. created_at drives the "new" sort - // and the hot-rank age, so a record asserting a future date (hostile or - // clock-skewed federated repo) could otherwise pin itself to the top of - // feeds until wall-clock catches up. - if now := time.Now(); createdAt.After(now) { - log.Printf("Warning: post %s has future createdAt %s, clamping to now", uri, postRecord.CreatedAt) - createdAt = now - } + createdAt := parseRecordCreatedAt(postRecord.CreatedAt, uri) // Build post entity post := &posts.Post{ @@ -218,36 +221,17 @@ func (c *PostEventConsumer) createPost(ctx context.Context, repoDID string, comm // Serialize JSON fields (facets, embed, labels) // Return error if any non-empty field fails to serialize (prevents silent data loss) - postRecord.Facets = sanitizedPostFacets(postRecord, uri) - if postRecord.Facets != nil { - facetsJSON, marshalErr := json.Marshal(postRecord.Facets) - if marshalErr != nil { - return fmt.Errorf("failed to serialize facets: %w", marshalErr) - } - facetsStr := string(facetsJSON) - post.ContentFacets = &facetsStr - } - - if postRecord.Embed != nil { - embedJSON, marshalErr := json.Marshal(postRecord.Embed) - if marshalErr != nil { - return fmt.Errorf("failed to serialize embed: %w", marshalErr) - } - embedStr := string(embedJSON) - post.Embed = &embedStr - } - - if postRecord.Labels != nil { - labelsJSON, marshalErr := json.Marshal(postRecord.Labels) - if marshalErr != nil { - return fmt.Errorf("failed to serialize labels: %w", marshalErr) - } - labelsStr := string(labelsJSON) - post.ContentLabels = &labelsStr + facetsJSON, embedJSON, labelsJSON, err := serializePostContent( + sanitizedPostFacets(postRecord, uri), postRecord.Embed, postRecord.Labels) + if err != nil { + return err } + post.ContentFacets = nullableString(facetsJSON) + post.Embed = nullableString(embedJSON) + post.ContentLabels = nullableString(labelsJSON) // Atomically: Rev-gate + Index post + Reconcile comment count for out-of-order arrivals - if err := c.indexPostAndReconcileCounts(ctx, post, commit.Rev); err != nil { + if _, err := c.indexPostIfRevWins(ctx, post, commit.Rev); err != nil { return fmt.Errorf("failed to index post and reconcile counts: %w", err) } @@ -259,10 +243,18 @@ func (c *PostEventConsumer) createPost(ctx context.Context, repoDID string, comm // deletePost handles post deletion events from Jetstream // Soft-deletes the post in AppView database by setting deleted_at timestamp func (c *PostEventConsumer) deletePost(ctx context.Context, repoDID string, commit *CommitEvent) error { - // Build AT-URI for this post // Format: at://community_did/social.coves.community.post/rkey - uri := fmt.Sprintf("at://%s/social.coves.community.post/%s", repoDID, commit.RKey) + return c.tombstoneRecord(ctx, fmt.Sprintf("at://%s/social.coves.community.post/%s", repoDID, commit.RKey), commit.Rev) +} +// tombstoneRecord soft-deletes the post at uri under the rev gate. +// +// SOFT, never hard, whichever repo the record lived in: the row is the rev +// gate's tombstone, the comment thread's parent, and what moderation still +// reads. It is shared by the community-repo and author-repo delete paths +// because a deletion is the one operation where the two are identical — the +// URI already says whose repo it was. +func (c *PostEventConsumer) tombstoneRecord(ctx context.Context, uri, rev string) error { // REV GATE + soft delete in one transaction (the repo's SoftDelete is not // transaction-aware, and the delete's rev must be recorded atomically with // the tombstone: it is what rejects a stale cross-feed copy of the CREATE @@ -279,12 +271,12 @@ func (c *PostEventConsumer) deletePost(ctx context.Context, repoDID string, comm } }() - won, err := tryAdvanceRecordRev(ctx, tx, uri, commit.Rev) + won, err := tryAdvanceRecordRev(ctx, tx, uri, rev) if err != nil { return err } if !won { - logSkippedStaleRev(ConsumerPosts, "delete", uri, commit.Rev) + logSkippedStaleRev(ConsumerPosts, "delete", uri, rev) return nil } @@ -300,7 +292,7 @@ func (c *PostEventConsumer) deletePost(ctx context.Context, repoDID string, comm return fmt.Errorf("failed to commit post delete transaction: %w", err) } - log.Printf("✓ Deleted post: %s (community: %s, rkey: %s)", uri, repoDID, commit.RKey) + log.Printf("✓ Deleted post: %s", uri) return nil } @@ -340,78 +332,28 @@ func (c *PostEventConsumer) updatePost(ctx context.Context, repoDID string, comm uri := fmt.Sprintf("at://%s/social.coves.community.post/%s", repoDID, commit.RKey) // Fetch the stored row so we can enforce immutability and run the asOf regression guard. - var ( - storedID int64 - storedCommunityDID string - storedAuthorDID string - storedDeletedAt *time.Time - storedAsOf *time.Time - storedIndexedAt time.Time - ) - err = c.db.QueryRowContext(ctx, - `SELECT id, community_did, author_did, deleted_at, bridged_stats_as_of, indexed_at FROM posts WHERE uri = $1`, - uri, - ).Scan(&storedID, &storedCommunityDID, &storedAuthorDID, &storedDeletedAt, &storedAsOf, &storedIndexedAt) - if errors.Is(err, sql.ErrNoRows) { - // Not indexed yet (out-of-order delivery). Jetstream will replay CREATE; skip. - log.Printf("Update event for non-indexed post: %s (will be indexed on CREATE)", uri) - return nil - } + stored, found, err := c.loadStoredPost(ctx, uri) if err != nil { - return fmt.Errorf("failed to load stored post for update: %w", err) - } - - // Skip soft-deleted rows: a deleted post should not be resurrected by an edit. - if storedDeletedAt != nil { - log.Printf("Update event for soft-deleted post: %s (skipping)", uri) - return nil + return err } - - // RECENCY GUARD: a redriven (DeadLetterRedriver) or rewound update can arrive - // AFTER a newer update was already indexed. indexed_at is the watermark of the - // last applied event for this row (event time, see indexedAtForEvent); an event - // whose time_us is not strictly newer must be skipped, or a stale replay would - // silently revert newer content. Skipping is SUCCESS (the newer state wins) — - // returning an error would re-dead-letter an event that must never be applied. - // This Go pre-check exists for clean logging; the UPDATE below repeats the - // comparison atomically so a concurrent newer write between this read and the - // write still cannot be clobbered. - if evTime, ok := eventTime(timeUS); ok && !storedIndexedAt.Before(evTime) { - log.Printf("INFO: skipping stale post update for %s (event time %s <= last indexed %s; newer state already applied)", - uri, evTime.Format(time.RFC3339Nano), storedIndexedAt.Format(time.RFC3339Nano)) + if !found { + // Not indexed yet (out-of-order delivery). Jetstream will replay CREATE; skip. + log.Printf("Update event for non-indexed post: %s (will be indexed on CREATE)", uri) return nil } // SECURITY: community and author are immutable. Reassignment is rejected (skipped). - if storedCommunityDID != postRecord.Community || storedAuthorDID != postRecord.Author { + if stored.communityDID != postRecord.Community || stored.authorDID != postRecord.Author { log.Printf("🚨 SECURITY: Rejecting post update - community/author reassignment is not allowed: %s (stored community=%s author=%s; incoming community=%s author=%s)", - uri, storedCommunityDID, storedAuthorDID, postRecord.Community, postRecord.Author) + uri, stored.communityDID, stored.authorDID, postRecord.Community, postRecord.Author) return nil } // Serialize optional JSON content fields (return on failure to avoid silent data loss). - var facetsJSON, embedJSON, labelsJSON sql.NullString - postRecord.Facets = sanitizedPostFacets(postRecord, uri) - if postRecord.Facets != nil { - b, marshalErr := json.Marshal(postRecord.Facets) - if marshalErr != nil { - return fmt.Errorf("failed to serialize facets: %w", marshalErr) - } - facetsJSON.String, facetsJSON.Valid = string(b), true - } - if postRecord.Embed != nil { - b, marshalErr := json.Marshal(postRecord.Embed) - if marshalErr != nil { - return fmt.Errorf("failed to serialize embed: %w", marshalErr) - } - embedJSON.String, embedJSON.Valid = string(b), true - } - if postRecord.Labels != nil { - b, marshalErr := json.Marshal(postRecord.Labels) - if marshalErr != nil { - return fmt.Errorf("failed to serialize labels: %w", marshalErr) - } - labelsJSON.String, labelsJSON.Valid = string(b), true + facetsJSON, embedJSON, labelsJSON, err := serializePostContent( + sanitizedPostFacets(postRecord, uri), postRecord.Embed, postRecord.Labels) + if err != nil { + return err } // Decide the candidate bridged aggregate to hand to the atomic UPDATE. It is applied @@ -431,20 +373,124 @@ func (c *PostEventConsumer) updatePost(ctx context.Context, repoDID string, comm if c.bridgeTrust.TrustsPDS(community.PDSURL) { if up, down, asOf, ok := validatedBridgedStats(postRecord.BridgedStats, uri); ok { incomingUp, incomingDown, incomingAsOf = up, down, &asOf - // Best-effort log only (the write is authoritative and atomic): a - // strictly-older asOf is dropped by the SQL guard. Kept at debug because - // the bridge re-sends the same asOf on every content edit, so this is - // noise, not an anomaly. - if storedAsOf != nil && asOf.Before(*storedAsOf) { - log.Printf("debug: ignoring strictly-older bridgedStats for %s (incoming asOf %s < stored %s)", - uri, asOf.Format(time.RFC3339), storedAsOf.Format(time.RFC3339)) - } } } else { log.Printf("debug: ignoring bridgedStats on post %s from untrusted repo %s (not a trusted bridge PDS)", uri, repoDID) } } + if _, err := c.applyPostContentUpdate(ctx, postContentUpdate{ + uri: uri, storedID: stored.id, rev: commit.Rev, cid: commit.CID, + title: postRecord.Title, content: postRecord.Content, + facets: facetsJSON, embed: embedJSON, labels: labelsJSON, + bridgedUpvotes: incomingUp, bridgedDownvotes: incomingDown, bridgedAsOf: incomingAsOf, + storedAsOf: stored.bridgedAsOf, storedDeletedAt: stored.deletedAt, + storedIndexedAt: stored.indexedAt, timeUS: timeUS, + }); err != nil { + return err + } + return nil +} + +// storedPost is the slice of an indexed post row the write paths need: the +// identity to update, the columns immutability is checked against, and the two +// watermarks (bridged asOf, indexed_at) the guards compare. +type storedPost struct { + id int64 + communityDID string + authorDID string + deletedAt *time.Time + bridgedAsOf *time.Time + indexedAt time.Time +} + +// loadStoredPost reads the row for uri. found=false means the post has never +// been indexed, which is an ordinary out-of-order arrival rather than an error. +func (c *PostEventConsumer) loadStoredPost(ctx context.Context, uri string) (storedPost, bool, error) { + var stored storedPost + err := c.db.QueryRowContext(ctx, + `SELECT id, community_did, author_did, deleted_at, bridged_stats_as_of, indexed_at FROM posts WHERE uri = $1`, + uri, + ).Scan(&stored.id, &stored.communityDID, &stored.authorDID, + &stored.deletedAt, &stored.bridgedAsOf, &stored.indexedAt) + if errors.Is(err, sql.ErrNoRows) { + return storedPost{}, false, nil + } + if err != nil { + return storedPost{}, false, fmt.Errorf("failed to load stored post %s: %w", uri, err) + } + return stored, true, nil +} + +// postContentUpdate is one already-validated edit of an indexed post. +// +// It exists so the community-repo and author-repo paths share ONE content +// write. What differs between them is who may claim what — the repo/community +// check inverts, and bridgedStats provenance keys on a different repo — and all +// of that is settled by the caller before it gets here. What does not differ is +// how an edit is applied: the same rev gate, the same recency guard, the same +// atomic bridged-stats regression rule. Two copies of that would drift. +type postContentUpdate struct { + uri string + storedID int64 + rev string + cid string + + title *string + content *string + facets sql.NullString + embed sql.NullString + labels sql.NullString + + bridgedUpvotes int + bridgedDownvotes int + // bridgedAsOf nil means "leave the stored bridged columns alone". + bridgedAsOf *time.Time + + storedAsOf *time.Time + storedDeletedAt *time.Time + storedIndexedAt time.Time + timeUS int64 +} + +// applyPostContentUpdate runs the rev gate and the atomic content UPDATE. +// +// It reports whether the write APPLIED. A false with no error is a skip — the +// stored row already holds a newer state — and every skip here is the system +// working: multi-feed duplicates, dead-letter redrives, and edits of posts +// deleted between the load and the write all land in it. Returning any of them +// as an error would dead-letter healthy events. +func (c *PostEventConsumer) applyPostContentUpdate(ctx context.Context, in postContentUpdate) (bool, error) { + // Skip soft-deleted rows: a deleted post should not be resurrected by an edit. + if in.storedDeletedAt != nil { + log.Printf("Update event for soft-deleted post: %s (skipping)", in.uri) + return false, nil + } + + // RECENCY GUARD: a redriven (DeadLetterRedriver) or rewound update can arrive + // AFTER a newer update was already indexed. indexed_at is the watermark of the + // last applied event for this row (event time, see indexedAtForEvent); an event + // whose time_us is not strictly newer must be skipped, or a stale replay would + // silently revert newer content. Skipping is SUCCESS (the newer state wins) — + // returning an error would re-dead-letter an event that must never be applied. + // This Go pre-check exists for clean logging; the UPDATE below repeats the + // comparison atomically so a concurrent newer write between this read and the + // write still cannot be clobbered. + if evTime, ok := eventTime(in.timeUS); ok && !in.storedIndexedAt.Before(evTime) { + log.Printf("INFO: skipping stale post update for %s (event time %s <= last indexed %s; newer state already applied)", + in.uri, evTime.Format(time.RFC3339Nano), in.storedIndexedAt.Format(time.RFC3339Nano)) + return false, nil + } + + // Best-effort log only (the write is authoritative and atomic): a + // strictly-older asOf is dropped by the SQL guard. Kept at debug because + // the bridge re-sends the same asOf on every content edit, so this is + // noise, not an anomaly. + if in.bridgedAsOf != nil && in.storedAsOf != nil && in.bridgedAsOf.Before(*in.storedAsOf) { + log.Printf("debug: ignoring strictly-older bridgedStats for %s (incoming asOf %s < stored %s)", + in.uri, in.bridgedAsOf.Format(time.RFC3339), in.storedAsOf.Format(time.RFC3339)) + } + // Single atomic UPDATE. edited_at is bumped only when content actually changed (so a // debounced stats-only refresh does not mark the post edited). The bridged columns // and the inclusive score move together via a shared applies-guard: apply the @@ -502,7 +548,7 @@ func (c *PostEventConsumer) updatePost(ctx context.Context, repoDID string, comm // Only rev, assigned by the repo itself, orders events across feeds. tx, err := c.db.BeginTx(ctx, nil) if err != nil { - return fmt.Errorf("failed to begin transaction: %w", err) + return false, fmt.Errorf("failed to begin transaction: %w", err) } defer func() { if rollbackErr := tx.Rollback(); rollbackErr != nil && rollbackErr != sql.ErrTxDone { @@ -510,23 +556,23 @@ func (c *PostEventConsumer) updatePost(ctx context.Context, repoDID string, comm } }() - won, err := tryAdvanceRecordRev(ctx, tx, uri, commit.Rev) + won, err := tryAdvanceRecordRev(ctx, tx, in.uri, in.rev) if err != nil { - return err + return false, err } if !won { - logSkippedStaleRev(ConsumerPosts, "update", uri, commit.Rev) - return nil + logSkippedStaleRev(ConsumerPosts, "update", in.uri, in.rev) + return false, nil } result, err := tx.ExecContext(ctx, updateQuery, - storedID, commit.CID, postRecord.Title, postRecord.Content, - facetsJSON, embedJSON, labelsJSON, - incomingUp, incomingDown, incomingAsOf, - timeUS, + in.storedID, in.cid, in.title, in.content, + in.facets, in.embed, in.labels, + in.bridgedUpvotes, in.bridgedDownvotes, in.bridgedAsOf, + in.timeUS, ) if err != nil { - return fmt.Errorf("failed to update post: %w", err) + return false, fmt.Errorf("failed to update post: %w", err) } // A post can be soft-deleted — or overtaken by a concurrent NEWER update (recency @@ -536,25 +582,26 @@ func (c *PostEventConsumer) updatePost(ctx context.Context, repoDID string, comm // current state supersedes this event. rowsAffected, err := result.RowsAffected() if err != nil { - return fmt.Errorf("failed to check post update result: %w", err) + return false, fmt.Errorf("failed to check post update result: %w", err) } if rowsAffected == 0 { // The deferred rollback also reverts the gate advance — conservative: a // replay re-evaluates against whatever state superseded this event. - log.Printf("Update event for post that was deleted or superseded by a newer update between load and write: %s (skipping)", uri) - return nil + log.Printf("Update event for post that was deleted or superseded by a newer update between load and write: %s (skipping)", in.uri) + return false, nil } if err := tx.Commit(); err != nil { - return fmt.Errorf("failed to commit post update transaction: %w", err) + return false, fmt.Errorf("failed to commit post update transaction: %w", err) } - if incomingAsOf != nil { - log.Printf("✓ Updated post: %s (bridgedStats candidate applied if newer-or-equal: up=%d down=%d)", uri, incomingUp, incomingDown) + if in.bridgedAsOf != nil { + log.Printf("✓ Updated post: %s (bridgedStats candidate applied if newer-or-equal: up=%d down=%d)", + in.uri, in.bridgedUpvotes, in.bridgedDownvotes) } else { - log.Printf("✓ Updated post: %s", uri) + log.Printf("✓ Updated post: %s", in.uri) } - return nil + return true, nil } // parseBridgedAsOf parses a bridgedStats.asOf timestamp, logging (and returning the @@ -568,12 +615,17 @@ func parseBridgedAsOf(asOf, uri string) (time.Time, error) { return t, nil } -// indexPostAndReconcileCounts atomically indexes a post and reconciles comment counts -// This fixes the race condition where comments arrive before their parent post -func (c *PostEventConsumer) indexPostAndReconcileCounts(ctx context.Context, post *posts.Post, rev string) error { +// indexPostIfRevWins atomically indexes a post and reconciles comment counts. +// This fixes the race condition where comments arrive before their parent post. +// +// It reports whether the insert APPLIED: false means the rev gate refused the +// event, or the row already existed. Callers that must not act on content they +// did not write — the author-repo path, which opens an admission from the CID +// it just indexed — read that flag rather than assuming the write happened. +func (c *PostEventConsumer) indexPostIfRevWins(ctx context.Context, post *posts.Post, rev string) (bool, error) { tx, err := c.db.BeginTx(ctx, nil) if err != nil { - return fmt.Errorf("failed to begin transaction: %w", err) + return false, fmt.Errorf("failed to begin transaction: %w", err) } defer func() { if rollbackErr := tx.Rollback(); rollbackErr != nil && rollbackErr != sql.ErrTxDone { @@ -588,11 +640,11 @@ func (c *PostEventConsumer) indexPostAndReconcileCounts(ctx context.Context, pos // and writes commit or roll back together. won, err := tryAdvanceRecordRev(ctx, tx, post.URI, rev) if err != nil { - return err + return false, err } if !won { logSkippedStaleRev(ConsumerPosts, "create", post.URI, rev) - return nil + return false, nil } // 1. Insert the post (idempotent with RETURNING clause) @@ -654,13 +706,15 @@ func (c *PostEventConsumer) indexPostAndReconcileCounts(ctx context.Context, pos // machinery already exists; see comment_consumer.go). log.Printf("Post already indexed: %s (idempotent)", post.URI) if commitErr := tx.Commit(); commitErr != nil { - return fmt.Errorf("failed to commit transaction: %w", commitErr) + return false, fmt.Errorf("failed to commit transaction: %w", commitErr) } - return nil + // Reported as NOT applied: no content was written, so a caller that + // would record what it just indexed has nothing new to record. + return false, nil } if insertErr != nil { - return fmt.Errorf("failed to insert post: %w", insertErr) + return false, fmt.Errorf("failed to insert post: %w", insertErr) } // 2. Reconcile comment_count for this newly inserted post @@ -689,15 +743,15 @@ func (c *PostEventConsumer) indexPostAndReconcileCounts(ctx context.Context, pos // Reconciliation failure is a critical error - it means comment_count will be incorrect // This could cause data inconsistency where the displayed count doesn't match reality // Roll back the transaction to maintain consistency - return fmt.Errorf("failed to reconcile comment_count for %s: %w", post.URI, reconcileErr) + return false, fmt.Errorf("failed to reconcile comment_count for %s: %w", post.URI, reconcileErr) } // Commit transaction if err := tx.Commit(); err != nil { - return fmt.Errorf("failed to commit transaction: %w", err) + return false, fmt.Errorf("failed to commit transaction: %w", err) } - return nil + return true, nil } // errValidationInfra marks a post-validation failure caused by an infrastructure fault @@ -823,27 +877,85 @@ type BridgedStatsFromJetstream struct { AsOf string `json:"asOf"` } -// sanitizedPostFacets drops facets whose byte ranges fall outside the post's +// sanitizedPostFacets sanitizes the facets on a community-repo post record. +// +// A record-shaped wrapper over sanitizeFacets, kept because the author-repo +// record type deliberately has no author field and so cannot be the same type: +// the shared work is the range checking, not the unwrapping. +func sanitizedPostFacets(postRecord *PostRecordFromJetstream, uri string) []interface{} { + return sanitizeFacets(postRecord.Facets, postRecord.Content, uri) +} + +// sanitizeFacets drops facets whose byte ranges fall outside the post's // content (or are otherwise structurally invalid) before indexing. Firehose // records from federated repos cannot be rejected back to their author, and // clients must never receive ranges that slice outside the content, so invalid // facets are dropped rather than failing the event. Returns nil when no // facets survive, preserving the callers' nil-means-absent serialization. -func sanitizedPostFacets(postRecord *PostRecordFromJetstream, uri string) []interface{} { - if postRecord.Facets == nil { +func sanitizeFacets(facets []interface{}, content *string, uri string) []interface{} { + if facets == nil { return nil } contentByteLen := 0 - if postRecord.Content != nil { - contentByteLen = len(*postRecord.Content) + if content != nil { + contentByteLen = len(*content) } - kept, dropped := richtext.SanitizeFacets(postRecord.Facets, contentByteLen) + kept, dropped := richtext.SanitizeFacets(facets, contentByteLen) if dropped > 0 { log.Printf("Warning: dropped %d invalid facet(s) on post %s during indexing", dropped, uri) } return kept } +// serializePostContent marshals the three optional JSON columns a post record +// carries. A marshal failure is returned rather than swallowed: silently +// dropping facets, an embed, or labels would index a post that reads as though +// its author never sent them. +func serializePostContent(facets []interface{}, embed map[string]interface{}, labels *posts.SelfLabels) (facetsJSON, embedJSON, labelsJSON sql.NullString, err error) { + if facets != nil { + b, marshalErr := json.Marshal(facets) + if marshalErr != nil { + return facetsJSON, embedJSON, labelsJSON, fmt.Errorf("failed to serialize facets: %w", marshalErr) + } + facetsJSON.String, facetsJSON.Valid = string(b), true + } + if embed != nil { + b, marshalErr := json.Marshal(embed) + if marshalErr != nil { + return facetsJSON, embedJSON, labelsJSON, fmt.Errorf("failed to serialize embed: %w", marshalErr) + } + embedJSON.String, embedJSON.Valid = string(b), true + } + if labels != nil { + b, marshalErr := json.Marshal(labels) + if marshalErr != nil { + return facetsJSON, embedJSON, labelsJSON, fmt.Errorf("failed to serialize labels: %w", marshalErr) + } + labelsJSON.String, labelsJSON.Valid = string(b), true + } + return facetsJSON, embedJSON, labelsJSON, nil +} + +// parseRecordCreatedAt reads a record's author-supplied createdAt, falling back +// to now when it does not parse. +// +// SECURITY: future timestamps are clamped to now. created_at drives the "new" +// sort and the hot-rank age, so a record asserting a future date (hostile or +// clock-skewed federated repo) could otherwise pin itself to the top of feeds +// until wall-clock catches up. +func parseRecordCreatedAt(raw, uri string) time.Time { + createdAt, err := time.Parse(time.RFC3339, raw) + if err != nil { + log.Printf("Warning: Failed to parse createdAt timestamp for %s, using current time: %v", uri, err) + return time.Now() + } + if now := time.Now(); createdAt.After(now) { + log.Printf("Warning: post %s has future createdAt %s, clamping to now", uri, raw) + return now + } + return createdAt +} + // parsePostRecord converts a raw Jetstream record map to a PostRecordFromJetstream func parsePostRecord(record map[string]interface{}) (*PostRecordFromJetstream, error) { // Marshal to JSON and back to ensure proper type conversion diff --git a/internal/atproto/lexicon/social/coves/community/post/getStatus.json b/internal/atproto/lexicon/social/coves/community/post/getStatus.json new file mode 100644 index 0000000..bf72fda --- /dev/null +++ b/internal/atproto/lexicon/social/coves/community/post/getStatus.json @@ -0,0 +1,58 @@ +{ + "lexicon": 1, + "id": "social.coves.community.post.getStatus", + "defs": { + "main": { + "type": "query", + "description": "Get one community's admission decision about one post. Intentionally UNAUTHENTICATED: the caller with the strongest need is an author on another server whose post is pending on this host, and they have no account here to authenticate with. It is also the only way a rejection is reachable at all - a submission refused before it was ever accepted writes no community record, so there is no repository record and no firehose event carrying it, and without this endpoint an author whose post vanished could never learn that it was refused or why. The accepted cost is that anyone who can name a post AT-URI learns its status in a community. Both parameters are required: a post carries independent decisions from several communities, so there is no single status of a post.", + "parameters": { + "type": "params", + "required": ["post", "community"], + "properties": { + "post": { + "type": "string", + "format": "at-uri", + "description": "AT-URI of the post, in the author's repository" + }, + "community": { + "type": "string", + "format": "did", + "description": "DID of the community whose decision is being asked about" + } + } + }, + "output": { + "encoding": "application/json", + "schema": { + "type": "object", + "required": ["status"], + "properties": { + "status": { + "type": "string", + "knownValues": ["pending", "accepted", "pending_reacceptance", "rejected", "removed"], + "description": "The community's decision state. Reported verbatim rather than collapsed: an author who edited an accepted post needs pending_reacceptance to be distinguishable from a post that was never accepted, because the two have completely different next steps." + }, + "decisionCode": { + "type": "string", + "description": "Why the post was refused. Present only for rejected and removed. The vocabulary is open and spans both the codes a community publishes in a removal record and the admission-time codes that never reach a repository." + }, + "decisionAt": { + "type": "string", + "format": "datetime", + "description": "When the refusal above was decided" + }, + "acceptanceUri": { + "type": "string", + "format": "at-uri", + "description": "AT-URI of the live community acceptance record, so the caller can read the signed attestation rather than trusting this AppView's summary of it. Present only while an acceptance stands." + } + } + } + }, + "errors": [ + {"name": "InvalidRequest", "description": "A missing or malformed post or community parameter"}, + {"name": "NotFound", "description": "This community has no decision about this post"} + ] + } + } +} diff --git a/internal/core/posts/status.go b/internal/core/posts/status.go index ca851cc..865c9c4 100644 --- a/internal/core/posts/status.go +++ b/internal/core/posts/status.go @@ -2,13 +2,10 @@ package posts import ( "context" + "strings" "time" ) -// RED STUB (task 5, cycle 1). Signatures only — every method returns zero -// values so the tests that describe this surface compile and fail on their -// assertions rather than on a missing symbol. The implementation is GREEN's. - // The read side of an admission decision: social.coves.community.post.getStatus // (docs/PRD_AUTHOR_OWNED_POSTS.md §3.4). // @@ -78,6 +75,45 @@ func NewStatusService(admissions AdmissionRepository) StatusService { return &statusService{admissions: admissions} } +// GetStatus reads one community's decision about one post. +// +// Both halves of the subject are required rather than defaulted, because a +// post genuinely carries independent decisions from several communities (§2) +// and answering about whichever row was found first would report one +// community's verdict as though it were another's. func (s *statusService) GetStatus(ctx context.Context, req GetStatusRequest) (*PostStatus, error) { - return nil, nil + if strings.TrimSpace(req.PostURI) == "" { + return nil, NewValidationError("post", "post URI is required") + } + if strings.TrimSpace(req.CommunityDID) == "" { + return nil, NewValidationError("community", "community DID is required") + } + + admission, err := s.admissions.Get(ctx, req.CommunityDID, req.PostURI) + if err != nil { + // ErrNotFound travels out unchanged: a subject the community has never + // been offered is a genuine 404, not a status to invent. Reporting it + // as `pending` would promise the author that somebody is going to + // decide. + return nil, err + } + + status := &PostStatus{ + Status: admission.Status, + // The live acceptance record, and only while one stands. The repository + // clears these columns on removal and never sets them on a rejection, + // so this is the acceptance a caller can actually go and read. + AcceptanceURI: admission.AcceptanceURI, + } + + // The decision fields are gated on the status rather than copied blind. + // They describe a REFUSAL, and the two statuses above are the only ones a + // refusal produces; surfacing a code beside `pending` would tell an author + // their post was refused while it is still waiting. + if admission.Status == AdmissionStatusRejected || admission.Status == AdmissionStatusRemoved { + status.DecisionCode = admission.DecisionCode + status.DecisionAt = admission.DecisionAt + } + + return status, nil } diff --git a/internal/db/migrations/036_create_deleted_accounts.sql b/internal/db/migrations/036_create_deleted_accounts.sql new file mode 100644 index 0000000..153d4c4 --- /dev/null +++ b/internal/db/migrations/036_create_deleted_accounts.sql @@ -0,0 +1,60 @@ +-- +goose Up +-- The erasure marker: proof that a DID was deleted ON PURPOSE +-- (docs/PRD_AUTHOR_OWNED_POSTS.md §5.3, rev 2.7). +-- +-- WHY THIS EXISTS. Account deletion used to leave no trace. userRepo.Delete +-- removes the users row, the posts, and (since migration 034) the admission +-- rows — and then the firehose redelivers a post event for that same author, +-- or a dead letter for one is redriven, and every swept row comes straight +-- back. Nothing in the schema could tell the consumer not to re-index it. +-- +-- The absence of a users row cannot carry that meaning, because under +-- author-owned posts it already means something else and something normal: a +-- post record now lives in the AUTHOR's repo, so its author may be someone +-- this AppView has never indexed, and §5.3 REQUIRES that event to index +-- anyway. "No users row" is therefore the ordinary state of a federated +-- author, and reading it as "erased" would refuse the open federated posting +-- the whole design exists to enable. +-- +-- A row here means "this DID was erased on purpose"; no row means "never +-- seen". That is the entire distinction, and it is why the table holds a DID +-- and almost nothing else. +-- +-- WHY NO FOREIGN KEY. The marker outlives the users row by construction — it +-- is written in the same transaction that deletes it — so a reference to +-- users(did) could never be satisfied. It is deliberately not scoped to +-- accounts this AppView hosts either: an erasure request may name a DID whose +-- repo lives elsewhere. +-- +-- HOW IT IS CLEARED. Re-registration. A DID that comes back — the same person +-- signing up again, or an account restored after a mistaken deletion — must +-- index normally, so the repository's user INSERT removes the marker in the +-- same transaction. A marker left standing would make the AppView accept the +-- account's profile and then silently drop every post it writes, forever, +-- with nothing anywhere explaining why. +CREATE TABLE deleted_accounts ( + -- The DID is the whole key: one marker per account, so a re-delete + -- updates in place rather than accumulating rows the ingestion gate would + -- have to deduplicate on every event it reads. + did TEXT PRIMARY KEY, + + -- NOT NULL because the marker's only job is to be READ by a consumer + -- deciding whether to index an event, and a marker with no time cannot + -- participate in any retention or audit answer later. + deleted_at TIMESTAMPTZ NOT NULL DEFAULT NOW(), + + -- Nullable, and expected to stay NULL for AppView-initiated deletions: + -- nothing knows the account's repo revision at deletion time, because the + -- deletion is a local administrative act rather than a commit. A column + -- that had to be filled would be filled with a fabricated watermark, at + -- the one place a real comparison happens. It exists for the future case + -- where an erasure IS observed as a repo event carrying a rev. + deleted_rev TEXT +); + +COMMENT ON TABLE deleted_accounts IS 'Erasure markers: DIDs deleted on purpose, so ingestion can tell an erased account from a federated author it has never indexed (PRD_AUTHOR_OWNED_POSTS 5.3)'; +COMMENT ON COLUMN deleted_accounts.deleted_at IS 'When the deletion happened; read by retention and audit, never by the ingestion gate itself'; +COMMENT ON COLUMN deleted_accounts.deleted_rev IS 'Repo revision the erasure was observed at, when one is known; NULL for AppView-initiated deletions'; + +-- +goose Down +DROP TABLE IF EXISTS deleted_accounts; diff --git a/internal/db/postgres/deleted_account_repo.go b/internal/db/postgres/deleted_account_repo.go index a9d3808..1f9c71c 100644 --- a/internal/db/postgres/deleted_account_repo.go +++ b/internal/db/postgres/deleted_account_repo.go @@ -3,10 +3,9 @@ package postgres import ( "context" "database/sql" + "fmt" ) -// RED STUB (task 5, cycle 1). Signatures only; the query is GREEN's. - // DeletedAccountRepository reads the migration-036 erasure markers. // // It satisfies jetstream.DeletedAccountLookup structurally rather than by @@ -31,5 +30,11 @@ func NewDeletedAccountRepository(db *sql.DB) *DeletedAccountRepository { // re-index the content a deletion erased, which is the exact outcome the marker // table exists to prevent. func (r *DeletedAccountRepository) IsAccountDeleted(ctx context.Context, did string) (bool, error) { - return false, nil + var deleted bool + if err := r.db.QueryRowContext(ctx, + `SELECT EXISTS (SELECT 1 FROM deleted_accounts WHERE did = $1)`, did, + ).Scan(&deleted); err != nil { + return false, fmt.Errorf("checking whether %s was erased: %w", did, err) + } + return deleted, nil } diff --git a/internal/db/postgres/user_repo.go b/internal/db/postgres/user_repo.go index a843af1..f65d3df 100644 --- a/internal/db/postgres/user_repo.go +++ b/internal/db/postgres/user_repo.go @@ -21,14 +21,44 @@ func NewUserRepository(db *sql.DB) users.UserRepository { return &postgresUserRepo{db: db} } -// Create inserts a new user into the users table +// Create inserts a new user into the users table. +// +// It also clears any migration-036 erasure marker for the DID, in the same +// transaction, because registering IS the marker's exit. A DID that comes back +// — the same person signing up again, or an account restored after a mistaken +// deletion — must index normally, and a marker left standing would have the +// ingestion gate silently drop every post the returning account writes. Both +// service paths funnel through here (IndexUser via CreateUser, and +// RegisterAccount), which is why the clear lives at the repository statement +// rather than in either of them. func (r *postgresUserRepo) Create(ctx context.Context, user *users.User) (*users.User, error) { + tx, err := r.db.BeginTx(ctx, nil) + if err != nil { + return nil, fmt.Errorf("failed to start transaction creating user did=%s: %w", user.DID, err) + } + defer func() { + if err := tx.Rollback(); err != nil && err != sql.ErrTxDone { + slog.Error("failed to rollback user create transaction", + slog.String("did", user.DID), + slog.String("error", err.Error()), + ) + } + }() + + // Ordered before the insert so that a failing insert — a duplicate DID or a + // taken handle — rolls the clear back with it. Clearing a marker for an + // account that did not actually re-register would silently re-open + // ingestion for content the AppView was asked to forget. + if _, err := tx.ExecContext(ctx, `DELETE FROM deleted_accounts WHERE did = $1`, user.DID); err != nil { + return nil, fmt.Errorf("failed to clear deletion marker for did=%s: %w", user.DID, err) + } + query := ` INSERT INTO users (did, handle, pds_url) VALUES ($1, $2, $3) RETURNING did, handle, pds_url, created_at, updated_at` - err := r.db.QueryRowContext(ctx, query, user.DID, user.Handle, user.PDSURL). + err = tx.QueryRowContext(ctx, query, user.DID, user.Handle, user.PDSURL). Scan(&user.DID, &user.Handle, &user.PDSURL, &user.CreatedAt, &user.UpdatedAt) if err != nil { // Check for unique constraint violations @@ -43,6 +73,10 @@ func (r *postgresUserRepo) Create(ctx context.Context, user *users.User) (*users return nil, fmt.Errorf("failed to create user: %w", err) } + if err := tx.Commit(); err != nil { + return nil, fmt.Errorf("failed to commit user create transaction for did=%s: %w", user.DID, err) + } + return user, nil } @@ -251,6 +285,30 @@ func (r *postgresUserRepo) Delete(ctx context.Context, did string) error { } }() + // 0. Record the erasure marker (migration 036). + // + // It goes FIRST and inside this transaction, both deliberately. Inside, + // because a marker that survived a rolled-back deletion would name an + // account that still exists — and the ingestion gate reads this table, so + // that account's future posts would be dropped forever with no row + // anywhere explaining it. First, because every statement below erases + // content, and the marker is what stops the firehose putting it back: a + // redriven post event or a replayed acceptance for this DID arrives long + // after the sweep, and without a marker the consumer cannot tell an erased + // account from a federated author it has simply never indexed (§5.3). + // + // deleted_rev is left NULL: an AppView-initiated deletion is a local + // administrative act, not a repo commit, so there is no revision to record + // and inventing one would put a fabricated watermark where real + // comparisons happen. A re-delete refreshes the timestamp rather than + // erroring, so the sweep stays idempotent. + if _, err := tx.ExecContext(ctx, ` + INSERT INTO deleted_accounts (did, deleted_at) VALUES ($1, NOW()) + ON CONFLICT (did) DO UPDATE SET deleted_at = NOW() + `, did); err != nil { + return fmt.Errorf("failed to record deletion marker for did=%s: %w", did, err) + } + // Delete in correct order to avoid foreign key violations // Tables without FK constraints on user_did are deleted first -- 2.51.2 From 3a9ee330f4164d240fc5d487523d36cead9acf8e Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 00:27:25 -0700 Subject: [PATCH 03/17] =?UTF-8?q?test(ingestion):=20RED=20cycle=202=20?= =?UTF-8?q?=E2=80=94=205b=20driver/factory/decider=20+=203=20T2=20contract?= =?UTF-8?q?s?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Task 5b red gate. 28 failing tests, zero pre-existing breakage. contract-manifest is GREEN again: all three new collections now carry markers, and tier discipline passes (no forbidden imports, no testkit.DB). JOB 1 — 5b (a) AdmissionRepository.ListPendingSubjects (T1, internal/db/postgres): undecided statuses only, oldest first, LIMIT-bounded; EXCLUDES tombstoned and never-indexed posts (F22 resurrection loop) and communities with no stored credentials (F2/F3). The hosted test is the sharp one — fixtures.Community already sets hosted_by_did to this instance while storing no credentials, which is exactly the shape a firehose-indexed community has, so a hosted_by_did implementation passes everything else and fails only there. Plus a catalog assertion for the partial index the cross-community scan needs. (b) CommunityRepoFactory (T1, real PDS): credentialed community yields a client that completes an authenticated round trip; uncredentialed yields ErrCommunityNotHosted specifically, because a generic error reads as transient and would re-list every remote community forever. (c) AdmissionDecider (T0 — every collaborator is an interface and nothing here needs a socket; tier is chosen by what a test needs out of process, not by which task it belongs to). Actor classification, the stricter-fallback on a failed IsAggregator lookup, tombstoned/absent posts refused BEFORE the policy runs, and infra failures returning undecided with no code minted. (d) QueueDriver (T0): every subject once per pass, grouped by community, per-subject backoff on deferral (injected clock), a failing subject not aborting the pass, and a snapshot whose lastPassAt is nil until the first pass — the only signal a dead driver produces. (e) DeleteAcceptance (T1, real PDS): withdraws a standing acceptance, writes NO removal record (an author deleting their own post is not a moderation act), skips when nothing stands, idempotent under redelivery. Consumer side: the tombstone triggers exactly one sweep, fires for no unaccepted post, and a failed sweep cannot hold the local tombstone hostage. (f) acceptanceQueue health block, omitted entirely when no driver runs. JOB 2 — the three T2 contracts (tests/e2e/author_post_contract_test.go) postv2 (outer frame), acceptance and removal, observed through getStatus because post.get stays status-agnostic until task 7. The retarget and phantom-acceptance negatives are bounded by a later commit in the SAME repo, per TestPostIngestion's ordering argument. The removal is written as one applyWrites commit — two commits would let a consumer that cannot order them pass. The deterministic rkey is re-derived from stdlib rather than importing posts.SubjectRkey: no test in this tier imports internal/core, and a test calling the production helper could not detect a bug in it. T2 is compile-verified only (go vet -tags e2e + the manifest gate); no AppView is serving locally and a full run needs the hermetic stack. Co-Authored-By: Claude Fable 5 --- cmd/server/acceptance_queue_health_test.go | 120 ++++ cmd/server/health.go | 37 ++ internal/atproto/jetstream/authorpost.go | 22 + internal/atproto/jetstream/post_consumer.go | 3 + .../atproto/jetstream/postv2_consumer_test.go | 182 +++++- internal/core/posts/acceptance_delete_test.go | 148 +++++ internal/core/posts/admissions.go | 39 ++ internal/core/posts/community_repo_factory.go | 55 ++ .../core/posts/community_repo_factory_test.go | 108 ++++ internal/core/posts/community_writer.go | 41 ++ internal/core/posts/decider.go | 99 ++++ internal/core/posts/decider_test.go | 311 +++++++++++ internal/core/posts/engine_matrix_test.go | 20 + internal/core/posts/queue.go | 168 ++++++ internal/core/posts/queue_test.go | 334 +++++++++++ internal/db/postgres/admission_queue_repo.go | 29 + .../db/postgres/admission_queue_repo_test.go | 279 ++++++++++ tests/e2e/author_post_contract_test.go | 517 ++++++++++++++++++ 18 files changed, 2509 insertions(+), 3 deletions(-) create mode 100644 cmd/server/acceptance_queue_health_test.go create mode 100644 internal/core/posts/acceptance_delete_test.go create mode 100644 internal/core/posts/community_repo_factory.go create mode 100644 internal/core/posts/community_repo_factory_test.go create mode 100644 internal/core/posts/decider.go create mode 100644 internal/core/posts/decider_test.go create mode 100644 internal/core/posts/queue.go create mode 100644 internal/core/posts/queue_test.go create mode 100644 internal/db/postgres/admission_queue_repo.go create mode 100644 internal/db/postgres/admission_queue_repo_test.go create mode 100644 tests/e2e/author_post_contract_test.go diff --git a/cmd/server/acceptance_queue_health_test.go b/cmd/server/acceptance_queue_health_test.go new file mode 100644 index 0000000..87092bb --- /dev/null +++ b/cmd/server/acceptance_queue_health_test.go @@ -0,0 +1,120 @@ +package main + +import ( + "encoding/json" + "testing" + "time" + + "Coves/internal/core/posts" +) + +// The acceptance driver's entry in /health/consumers. +// +// The driver is the one moving part in this system with no natural symptom of +// its own. A consumer that dies disconnects, and the connector says so; a driver +// that dies simply stops running, produces no error and no log line, and every +// consequence shows up somewhere else entirely — as posts that never become +// visible, days later, in a community whose moderators assume nobody is posting. +// These three fields are the only place that failure is visible before a user +// reports it. + +func TestBuildAcceptanceQueueHealth_ReportsBacklogAndAges(t *testing.T) { + now := time.Date(2026, 8, 8, 12, 0, 0, 0, time.UTC) + oldest := now.Add(-30 * time.Minute) + lastPass := now.Add(-45 * time.Second) + + queue := buildAcceptanceQueueHealth(posts.QueueSnapshot{ + PendingBacklog: 12, + OldestPendingAt: &oldest, + LastPassAt: &lastPass, + LastPassDeferred: 4, + LastPassFailed: 1, + }, now) + + if queue.PendingBacklog != 12 { + t.Errorf("expected pendingBacklog 12, got %d", queue.PendingBacklog) + } + + // The age, not the timestamp. A backlog that is merely BIG is a busy + // instance; a backlog whose oldest entry keeps getting older is an engine + // that has stopped settling anything, and only the age says which of those + // is happening without the reader doing arithmetic against their own clock. + if queue.OldestPendingAgeSeconds == nil { + t.Fatal("expected oldestPendingAgeSeconds to be set for a non-empty backlog") + } + if *queue.OldestPendingAgeSeconds != 1800 { + t.Errorf("expected oldestPendingAgeSeconds 1800, got %d", *queue.OldestPendingAgeSeconds) + } + + if queue.LastPassAt == nil || !queue.LastPassAt.Equal(lastPass) { + t.Errorf("expected lastPassAt %v, got %v", lastPass, queue.LastPassAt) + } + if queue.LastPassDeferred != 4 { + t.Errorf("expected lastPassDeferred 4, got %d", queue.LastPassDeferred) + } + if queue.LastPassFailed != 1 { + t.Errorf("expected lastPassFailed 1, got %d", queue.LastPassFailed) + } +} + +func TestBuildAcceptanceQueueHealth_OmitsAgesItCannotHonestlyReport(t *testing.T) { + now := time.Date(2026, 8, 8, 12, 0, 0, 0, time.UTC) + + // A driver that has never run, with an empty backlog. Both omissions matter + // and for the same reason: a zero would be READ, and read wrongly. + // oldestPendingAgeSeconds of 0 says "something arrived just now" when in + // fact nothing is waiting, and a zero-valued lastPassAt renders as the epoch + // — which looks like a driver that has been dead since 1970 rather than one + // that started a minute ago. + queue := buildAcceptanceQueueHealth(posts.QueueSnapshot{}, now) + + if queue.PendingBacklog != 0 { + t.Errorf("expected an empty backlog, got %d", queue.PendingBacklog) + } + if queue.OldestPendingAgeSeconds != nil { + t.Errorf("an empty backlog has no oldest entry; got age %d", *queue.OldestPendingAgeSeconds) + } + if queue.LastPassAt != nil { + t.Errorf("a driver that has never run must not claim a last pass; got %v", *queue.LastPassAt) + } + + encoded, err := json.Marshal(queue) + if err != nil { + t.Fatalf("marshalling the queue health: %v", err) + } + body := string(encoded) + for _, omitted := range []string{"oldestPendingAgeSeconds", "lastPassAt"} { + if contains(body, omitted) { + t.Errorf("%s must be omitted from the JSON when unset, not serialised as null: %s", omitted, body) + } + } +} + +func TestConsumerHealthResponse_OmitsTheQueueEntirelyWhenNoDriverRuns(t *testing.T) { + // An AppView that hosts no communities runs no acceptance driver at all, and + // its health response must not carry an all-zero queue — that reads as a + // driver that is running and settling nothing, which is the exact shape of + // the failure an operator is watching for. + encoded, err := json.Marshal(consumerHealthResponse{Status: "ok"}) + if err != nil { + t.Fatalf("marshalling the response: %v", err) + } + if contains(string(encoded), "acceptanceQueue") { + t.Errorf("acceptanceQueue must be absent when no driver is wired: %s", encoded) + } +} + +// contains is strings.Contains, spelled locally so this file's imports stay the +// two it genuinely needs. +func contains(haystack, needle string) bool { + return len(haystack) >= len(needle) && indexOf(haystack, needle) >= 0 +} + +func indexOf(haystack, needle string) int { + for i := 0; i+len(needle) <= len(haystack); i++ { + if haystack[i:i+len(needle)] == needle { + return i + } + } + return -1 +} diff --git a/cmd/server/health.go b/cmd/server/health.go index 2a28e44..b4261e6 100644 --- a/cmd/server/health.go +++ b/cmd/server/health.go @@ -2,6 +2,7 @@ package main import ( "Coves/internal/atproto/jetstream" + "Coves/internal/core/posts" "encoding/json" "log/slog" "net/http" @@ -34,6 +35,42 @@ type consumerHealthResponse struct { // would look healthier the sicker the database gets. DeadLetterBacklogUnknown bool `json:"deadLetterBacklogUnknown,omitempty"` Consumers []consumerHealth `json:"consumers"` + + // AcceptanceQueue reports the acceptance engine's driver, and is omitted + // entirely on a deployment that runs no driver (one hosting no communities + // has nothing to accept). Omitted rather than zeroed: an all-zero queue and + // an absent one mean different things, and only one of them is worth waking + // somebody for. + AcceptanceQueue *acceptanceQueueHealth `json:"acceptanceQueue,omitempty"` +} + +// acceptanceQueueHealth is the acceptance driver's entry in the response. +// +// The two age fields answer the two questions an operator has, and neither can +// be derived from the backlog size alone. A big backlog on a busy instance is +// healthy; a backlog whose OLDEST entry keeps getting older is an engine that +// has stopped settling anything. And a driver that has died produces no error +// and no log — it simply stops — so lastPassAt is the only signal that the pass +// is still happening at all. +type acceptanceQueueHealth struct { + PendingBacklog int `json:"pendingBacklog"` + // OldestPendingAgeSeconds is omitted when the backlog is empty: there is no + // oldest entry, and reporting 0 would read as "something arrived just now". + OldestPendingAgeSeconds *int64 `json:"oldestPendingAgeSeconds,omitempty"` + // LastPassAt is omitted until the first pass completes, distinguishing "the + // driver has never run" from "the driver ran and found nothing". + LastPassAt *time.Time `json:"lastPassAt,omitempty"` + LastPassDeferred int `json:"lastPassDeferred"` + LastPassFailed int `json:"lastPassFailed"` +} + +// buildAcceptanceQueueHealth renders one driver snapshot. +// +// RED STUB (task 5, cycle 2). A separate pure function rather than another +// parameter on buildConsumerHealthResponse: the two have no shared logic, and +// widening that signature would touch every existing call site to say nothing. +func buildAcceptanceQueueHealth(snapshot posts.QueueSnapshot, now time.Time) acceptanceQueueHealth { + return acceptanceQueueHealth{} } // buildConsumerHealthResponse is the pure decision core of /health/consumers, diff --git a/internal/atproto/jetstream/authorpost.go b/internal/atproto/jetstream/authorpost.go index 60acd84..0a11c32 100644 --- a/internal/atproto/jetstream/authorpost.go +++ b/internal/atproto/jetstream/authorpost.go @@ -83,6 +83,28 @@ func WithPostRecordFetcher(fetcher PostRecordFetcher) PostEventConsumerOption { return func(c *PostEventConsumer) { c.postFetcher = fetcher } } +// AcceptanceDeleter withdraws a community's acceptance of a post. Satisfied by +// posts.CommunityRecordWriter. +// +// RED STUB (task 5, cycle 2). Narrowed to one method because that is all the +// tombstone path needs: the consumer must never write an acceptance, a removal +// or a repin — those are the ENGINE's verdicts, and a consumer holding the full +// writer is one edit away from making one. +type AcceptanceDeleter interface { + DeleteAcceptance(ctx context.Context, cmd posts.CommunityAcceptanceDeleteCommand) (posts.CommunityWriteResult, error) +} + +// WithAcceptanceCleanup installs the host-side sweep that withdraws a +// community's acceptance when the AUTHOR deletes their post (§5.3). +// +// Only the HOST can do this — the acceptance lives in the community's repo and +// needs its keys — so the sweep is silently a no-op for every community this +// AppView does not host, and that is the common case on any instance that is +// not the community's home. nil disables it entirely. +func WithAcceptanceCleanup(deleter AcceptanceDeleter) PostEventConsumerOption { + return func(c *PostEventConsumer) { c.acceptanceCleanup = deleter } +} + // --------------------------------------------------------------------------- // §5.4 direct fetch // --------------------------------------------------------------------------- diff --git a/internal/atproto/jetstream/post_consumer.go b/internal/atproto/jetstream/post_consumer.go index b5ce0ee..f9896fc 100644 --- a/internal/atproto/jetstream/post_consumer.go +++ b/internal/atproto/jetstream/post_consumer.go @@ -42,6 +42,9 @@ type PostEventConsumer struct { // postFetcher resolves an acceptance whose subject was never indexed. nil // means the dead-letter queue is the only convergence mechanism. postFetcher PostRecordFetcher + // acceptanceCleanup withdraws a hosted community's acceptance when the + // author tombstones the post. nil means no sweep runs. + acceptanceCleanup AcceptanceDeleter } // PostEventConsumerOption configures optional PostEventConsumer behaviour. diff --git a/internal/atproto/jetstream/postv2_consumer_test.go b/internal/atproto/jetstream/postv2_consumer_test.go index 254903f..dd8bed4 100644 --- a/internal/atproto/jetstream/postv2_consumer_test.go +++ b/internal/atproto/jetstream/postv2_consumer_test.go @@ -5,6 +5,7 @@ package jetstream import ( "context" "database/sql" + "errors" "testing" "time" @@ -459,9 +460,184 @@ func TestPostV2Consumer_Delete_TombstonesTheRow(t *testing.T) { "a soft delete must not blank the content") // The host-side half — the community observing the tombstone and deleting - // its acceptance (§5.3) — is task 5b's scope and deliberately not asserted - // here. What matters at this point is that the tombstone exists for that - // sweep to find. + // its acceptance (§5.3) — is asserted separately below, since it only runs + // on the instance that HOSTS the community. +} + +// recordingAcceptanceDeleter is the host-side sweep, observed rather than +// performed. +// +// A fake here and not a real community repo, deliberately: what the CONSUMER +// owes is that it asks for the right subject, exactly once, and only when it +// should. Whether the ask reaches the PDS correctly — the shaped delete, the +// skip on a missing record, the committed rev — is the writer's own contract +// and is proven against a real PDS in +// internal/core/posts/acceptance_delete_test.go. Wiring a credentialed +// community in here would re-prove that and make this test unable to say +// anything about the case that matters most: the sweep NOT firing. +type recordingAcceptanceDeleter struct { + calls []posts.CommunityAcceptanceDeleteCommand + err error +} + +func (d *recordingAcceptanceDeleter) DeleteAcceptance( + _ context.Context, cmd posts.CommunityAcceptanceDeleteCommand, +) (posts.CommunityWriteResult, error) { + d.calls = append(d.calls, cmd) + if d.err != nil { + return posts.CommunityWriteResult{}, d.err + } + return posts.CommunityWriteResult{Rev: testkit.TID()}, nil +} + +func TestPostV2Consumer_Delete_WithdrawsTheHostedCommunitysAcceptance(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + + sweep := &recordingAcceptanceDeleter{} + f := newPV2Fixture(t, db) + f.consumer = NewPostEventConsumer( + postgres.NewPostRepository(db), + postgres.NewCommunityRepository(db), + f.users, + db, + WithAdmissions(f.admissions), + WithDeletedAccounts(postgres.NewDeletedAccountRepository(db)), + WithAcceptanceCleanup(sweep), + ) + + rkey := "pv2sweep" + uri := pv2URI(pv2Author, rkey) + base := time.Now().UnixMicro() + revs := increasingTIDs(t, 2) + + const cid = "bafyreipv2sweep" + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "create", rkey, revs[0], cid, base, + pv2Record(pv2Community, "accepted, then withdrawn by its author", "body"), + ))) + + acceptanceRkey := testkit.TID() + accepted, err := f.admissions.ApplyAcceptance(ctx, posts.ApplyAcceptanceCommand{ + CommunityDID: pv2Community, + PostURI: uri, + AcceptanceURI: "at://" + pv2Community + "/social.coves.community.acceptance/" + acceptanceRkey, + AcceptanceRkey: acceptanceRkey, + PinnedCID: cid, + Watermark: posts.CommunityWatermark{Rev: testkit.TID()}, + }) + require.NoError(t, err) + require.Equal(t, posts.AdmissionApplied, accepted.Outcome, "fixture: an acceptance must stand for the sweep to withdraw") + + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "delete", rkey, revs[1], "", base+1_000_000, nil, + ))) + + // The acceptance points at a record nobody can fetch now. The community repo + // is the curated index the whole portability argument rests on, so leaving + // it standing means the CAR permanently cites content the author withdrew, + // and a peer replaying it shows a post that no longer exists. + require.Lenf(t, sweep.calls, 1, + "the tombstone must trigger exactly one acceptance withdrawal; got %d", len(sweep.calls)) + assert.Equal(t, pv2Community, sweep.calls[0].CommunityDID) + assert.Equal(t, uri, sweep.calls[0].PostURI, + "the sweep must name the tombstoned post; the acceptance rkey is derived from this URI, so a wrong subject deletes a different post's acceptance") +} + +func TestPostV2Consumer_Delete_DoesNotSweepWhenNoAcceptanceStands(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + + sweep := &recordingAcceptanceDeleter{} + f := newPV2Fixture(t, db) + f.consumer = NewPostEventConsumer( + postgres.NewPostRepository(db), + postgres.NewCommunityRepository(db), + f.users, + db, + WithAdmissions(f.admissions), + WithDeletedAccounts(postgres.NewDeletedAccountRepository(db)), + WithAcceptanceCleanup(sweep), + ) + + rkey := "pv2nosweep" + base := time.Now().UnixMicro() + revs := increasingTIDs(t, 2) + + // Indexed and pending: the community never accepted it, so there is nothing + // in its repo to withdraw. This is the COMMON case — most posts a community + // sees were never accepted by it — and a sweep that fired anyway would put + // one pointless authenticated PDS round trip behind every delete event on + // the network. + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "create", rkey, revs[0], "bafyreipv2nosweep", base, + pv2Record(pv2Community, "never accepted", "body"), + ))) + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "delete", rkey, revs[1], "", base+1_000_000, nil, + ))) + + assert.Emptyf(t, sweep.calls, + "the sweep fired for a post that was never accepted (%d calls): the admission row is the AppView's own record of whether an acceptance stands, "+ + "and consulting it is what keeps this from being a PDS round trip per delete event", len(sweep.calls)) +} + +func TestPostV2Consumer_Delete_SurvivesAFailedSweep(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + + sweep := &recordingAcceptanceDeleter{err: errors.New("the community's PDS is unreachable")} + f := newPV2Fixture(t, db) + f.consumer = NewPostEventConsumer( + postgres.NewPostRepository(db), + postgres.NewCommunityRepository(db), + f.users, + db, + WithAdmissions(f.admissions), + WithDeletedAccounts(postgres.NewDeletedAccountRepository(db)), + WithAcceptanceCleanup(sweep), + ) + + rkey := "pv2sweepfail" + uri := pv2URI(pv2Author, rkey) + base := time.Now().UnixMicro() + revs := increasingTIDs(t, 2) + + const cid = "bafyreipv2sweepfail" + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "create", rkey, revs[0], cid, base, + pv2Record(pv2Community, "the sweep will fail", "body"), + ))) + acceptanceRkey := testkit.TID() + _, err := f.admissions.ApplyAcceptance(ctx, posts.ApplyAcceptanceCommand{ + CommunityDID: pv2Community, + PostURI: uri, + AcceptanceURI: "at://" + pv2Community + "/social.coves.community.acceptance/" + acceptanceRkey, + AcceptanceRkey: acceptanceRkey, + PinnedCID: cid, + Watermark: posts.CommunityWatermark{Rev: testkit.TID()}, + }) + require.NoError(t, err) + + deleteErr := f.consumer.HandleEvent(ctx, pv2Event( + pv2Author, "delete", rkey, revs[1], "", base+1_000_000, nil, + )) + + // THE TOMBSTONE IS THE LOCAL TRUTH AND MUST LAND REGARDLESS. The author + // asked for their post to be gone; a community PDS that cannot be reached + // must not keep this AppView serving it. The acceptance withdrawal is + // best-effort cleanup of a REMOTE repo, and the engine's own passes revisit + // it — so the only question here is whether a failed sweep can hold the + // deletion hostage. + _, _, _, _, deletedAt := readPV2Post(t, db, uri) + require.NotNilf(t, deletedAt, + "the post was not tombstoned because the acceptance sweep failed: an unreachable community PDS would keep this AppView serving content its author deleted (sweep error: %v)", deleteErr) } // assertNullableStringPV2 asserts a nullable column holds exactly want. diff --git a/internal/core/posts/acceptance_delete_test.go b/internal/core/posts/acceptance_delete_test.go new file mode 100644 index 0000000..7dbfea7 --- /dev/null +++ b/internal/core/posts/acceptance_delete_test.go @@ -0,0 +1,148 @@ +//go:build integration + +package posts_test + +import ( + "context" + "testing" + + "Coves/internal/core/posts" + "Coves/tests/testkit" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// DeleteAcceptance: the writer's fifth method, and the one the author's own +// deletion needs (docs/PRD_AUTHOR_OWNED_POSTS.md §5.3). +// +// # WHY THE OTHER FOUR CANNOT DO THIS +// +// When an author tombstones their post, the community's acceptance is left +// pointing at a record nobody can fetch. The community repo is the curated index +// its whole portability argument rests on, so leaving the acceptance standing +// means the CAR permanently cites content the author withdrew, and any peer +// replaying it shows a post that no longer exists. +// +// WriteRemoval would clear it — and would also publish a signed moderation +// event, with a reason code, that never happened. That record is portable and +// permanent: a peer reading the community's repo would see the community +// declaring it removed this post, when in fact the author deleted it. The +// distinction is the difference between an accurate public record and a +// defamatory one, which is why this is its own method rather than a removal with +// a special code. +// +// # AGAINST A REAL PDS, FOR ONE REASON +// +// The PDS answers a delete of a missing record with a 500 (task 4's locked +// decisions: the writers are STATE-SHAPED because the PDS refuses the wrong +// shape). A fake would happily accept the delete and this file would prove +// nothing about the case that actually happens — a redelivered tombstone event +// meeting an acceptance an earlier pass already withdrew. + +func TestDeleteAcceptance_WithdrawsAStandingAcceptance(t *testing.T) { + t.Parallel() + + f := newEngineFixture(t) + ctx := context.Background() + + post := f.publishPost(t, "a post its author will delete") + f.seedPending(t, post.URI, post.CID) + outcome, err := f.process(t, post.URI) + require.NoError(t, err) + require.Equal(t, posts.EngineAccepted, outcome, "fixture: the acceptance must stand before it can be withdrawn") + + rkey := posts.SubjectRkey(post.URI) + standing := f.acceptanceOf(t, post.URI) + require.NotEmpty(t, standing.CID, "fixture: the acceptance record must exist") + + result, err := f.writer.DeleteAcceptance(ctx, posts.CommunityAcceptanceDeleteCommand{ + CommunityDID: f.community.DID, + PostURI: post.URI, + }) + require.NoError(t, err) + assert.False(t, result.Skipped, "an acceptance that was actually there is a delete, not a skip") + assert.NotEmptyf(t, result.Rev, + "the delete must report the rev it committed in: that is the §5.2 watermark the firehose copy of this same deletion will be compared against, "+ + "and without it the AppView cannot stamp its own event") + + f.assertRecordAbsent(t, posts.AcceptanceCollection, rkey, "the withdrawn acceptance") + + // And NOTHING was published in its place. A removal record here would put a + // moderation act on the public firehose that no moderator performed. + f.assertRecordAbsent(t, posts.RemovalCollection, rkey, + "a removal record — the author deleted their own post, which is not the community removing it") +} + +func TestDeleteAcceptance_IsASkipWhenThereIsNothingToWithdraw(t *testing.T) { + t.Parallel() + + f := newEngineFixture(t) + ctx := context.Background() + + post := f.publishPost(t, "a post that was never accepted") + + // No acceptance was ever written. This is the ordinary case, not an edge + // one: the sweep runs on every tombstone event, and most posts a community + // sees were never accepted by it — plus every tombstone event is redelivered + // at least once by the connector's cursor rewind. + result, err := f.writer.DeleteAcceptance(ctx, posts.CommunityAcceptanceDeleteCommand{ + CommunityDID: f.community.DID, + PostURI: post.URI, + }) + + require.NoError(t, err, + "deleting an acceptance that is not there must be a no-op, not an error: the PDS answers a delete of a missing record with a 500, "+ + "so an unshaped delete would dead-letter every redelivered tombstone") + assert.True(t, result.Skipped, "nothing was written, and the result must say so") +} + +func TestDeleteAcceptance_IsIdempotentUnderRedelivery(t *testing.T) { + t.Parallel() + + f := newEngineFixture(t) + ctx := context.Background() + + post := f.publishPost(t, "a post whose tombstone arrives twice") + f.seedPending(t, post.URI, post.CID) + _, err := f.process(t, post.URI) + require.NoError(t, err) + + cmd := posts.CommunityAcceptanceDeleteCommand{CommunityDID: f.community.DID, PostURI: post.URI} + + first, err := f.writer.DeleteAcceptance(ctx, cmd) + require.NoError(t, err) + require.False(t, first.Skipped) + + // The connector rewinds its cursor five seconds after every reconnect, so + // the identical tombstone commit is guaranteed to be redelivered. + second, err := f.writer.DeleteAcceptance(ctx, cmd) + require.NoError(t, err, "a redelivered tombstone must not fail the sweep") + assert.True(t, second.Skipped, "the second attempt found nothing to withdraw") + + f.assertRecordAbsent(t, posts.AcceptanceCollection, posts.SubjectRkey(post.URI), "the withdrawn acceptance") +} + +func TestDeleteAcceptance_RefusesASubjectItCannotKey(t *testing.T) { + t.Parallel() + + f := newEngineFixture(t) + + // The rkey is derived from the subject URI, so an empty or malformed subject + // hashes to a perfectly valid-looking key pointing at nothing — and a delete + // aimed at the wrong rkey in a community's own repo is a write, not a read. + // Every other writer validates its command; this one must too. + for _, badURI := range []string{"", "not-an-at-uri", "at://"} { + _, err := f.writer.DeleteAcceptance(context.Background(), posts.CommunityAcceptanceDeleteCommand{ + CommunityDID: f.community.DID, + PostURI: badURI, + }) + assert.Errorf(t, err, "a subject URI of %q must be refused rather than hashed into a key", badURI) + } + + _, err := f.writer.DeleteAcceptance(context.Background(), posts.CommunityAcceptanceDeleteCommand{ + CommunityDID: "", + PostURI: "at://" + testkit.UniqueID(t) + "/social.coves.community.postv2/x", + }) + assert.Error(t, err, "a delete with no community names no repo to delete from") +} diff --git a/internal/core/posts/admissions.go b/internal/core/posts/admissions.go index 660563f..630398d 100644 --- a/internal/core/posts/admissions.go +++ b/internal/core/posts/admissions.go @@ -340,6 +340,45 @@ type AdmissionRepository interface { // one is an error rather than a silent reset to the first page, which would // make a moderator re-review what they had already cleared. ListByStatusForCommunity(ctx context.Context, communityDID string, status AdmissionStatus, limit int, cursor *string) ([]*Admission, *string, error) + + // ListPendingSubjects returns the work the acceptance engine still owes a + // decision on, ACROSS communities, oldest first and bounded by limit. + // + // It is a different question from ListByStatusForCommunity, which serves a + // moderator looking at one community. This serves the engine's driver, which + // has no community in hand — it is asking "what is undecided anywhere that I + // can actually decide", and the two halves of that qualification are why this + // is a query rather than a filter over the other one: + // + // - ONLY COMMUNITIES THIS APPVIEW HOSTS. Hosting means holding the + // community's PDS credentials, because writing the acceptance is the + // whole point and a community whose keys we do not have can never be + // satisfied. Every pass would re-list it, hand it to the engine, and + // collect the same credential failure forever. + // - ONLY SUBJECTS WHOSE POST STILL STANDS. An admission whose post row is + // tombstoned or absent must not be handed to the engine: accepting a + // deleted post writes an acceptance for content that no longer exists, + // and the tombstone sweep would then delete it again — a resurrection + // loop between two components each behaving correctly on its own. + // + // Both exclusions are also enforced by the decider (a driver is a queue, not + // a security boundary), but doing it here is what keeps the queue's DEPTH an + // honest signal: a backlog full of subjects nothing can ever settle looks + // identical, from the outside, to an engine that has stopped working. + ListPendingSubjects(ctx context.Context, limit int) ([]PendingSubject, error) +} + +// PendingSubject is one (community, post) pair the engine still owes a decision. +// +// It carries CreatedAt because the driver's health surface reports the age of +// the oldest undecided subject, and that number is the queue's only early +// warning: a backlog that is merely BIG is a busy instance, while a backlog +// whose oldest entry keeps getting older is an engine that has stopped settling +// anything. The ordering the query already applies makes carrying it free. +type PendingSubject struct { + CommunityDID string + PostURI string + CreatedAt time.Time } // DecisionCode is the reason a post was refused or removed — the value stored diff --git a/internal/core/posts/community_repo_factory.go b/internal/core/posts/community_repo_factory.go new file mode 100644 index 0000000..8d6b0fa --- /dev/null +++ b/internal/core/posts/community_repo_factory.go @@ -0,0 +1,55 @@ +package posts + +import ( + "context" + "errors" + + "Coves/internal/core/communities" +) + +// RED STUB (task 5, cycle 2). Signatures only; the body is GREEN's. + +// ErrCommunityNotHosted reports that this AppView does not hold the community's +// PDS credentials, so it cannot write records into that community's repo. +// +// It is a PERMANENT SKIP, not a deferral, and the distinction decides whether a +// backlog drains or grows forever. A deferral means "look again later", which is +// right for an expired token — the refresh will fix it. Not being the host is +// not a transient condition: no retry, no redrive and no amount of waiting turns +// another instance's community into one this AppView can sign for. A driver that +// treated the two the same would re-offer every remote community's posts on +// every pass, forever, and the genuine deferrals would be invisible underneath. +var ErrCommunityNotHosted = errors.New("community is not hosted by this AppView") + +// CommunityCredentialSource supplies a community's repo credentials, refreshing +// them if they are close to expiry. Satisfied by communities.Service. +type CommunityCredentialSource interface { + GetByDID(ctx context.Context, did string) (*communities.Community, error) + EnsureFreshToken(ctx context.Context, community *communities.Community) (*communities.Community, error) +} + +// NewCommunityRepoFactory builds the production CommunityRepoFactory: a +// credential lookup, a token refresh, and a PDS client bound to the community's +// own repo. +// +// # HOSTING IS CREDENTIAL PRESENCE, NEVER hosted_by_did +// +// The obvious test — does communities.hosted_by_did name this instance — is +// wrong, and dangerously so. That column is populated from the community's own +// PROFILE RECORD when a community is indexed from the firehose, which means it +// is a claim made by whoever controls that repo. Anyone can write a community +// profile naming this AppView as its host. Trusting it would have the factory +// hand back a repo client for a community this instance has no keys for; every +// write would then fail at the PDS, but only after the engine had already +// decided, and a hostile community could aim that traffic wherever its PDS URL +// pointed. +// +// Credentials cannot be claimed. Either this AppView holds the community's +// refresh token — which happens exactly once, when it provisioned the account +// through social.coves.community.create — or it does not. That is the honest +// question, and it is the only one this factory asks. +func NewCommunityRepoFactory(communities CommunityCredentialSource) CommunityRepoFactory { + return func(ctx context.Context, communityDID string) (CommunityRepo, error) { + return nil, nil + } +} diff --git a/internal/core/posts/community_repo_factory_test.go b/internal/core/posts/community_repo_factory_test.go new file mode 100644 index 0000000..52f626c --- /dev/null +++ b/internal/core/posts/community_repo_factory_test.go @@ -0,0 +1,108 @@ +//go:build integration + +package posts_test + +import ( + "context" + "testing" + + "Coves/internal/core/posts" + "Coves/tests/fixtures" + "Coves/tests/testkit" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// The production CommunityRepoFactory: how the engine gets a signing session on +// a community's repo, and — the part with teeth — how it decides it has no +// business having one. +// +// # WHY THE REFUSAL IS THE INTERESTING HALF +// +// This is the AppView's only answer to "may I write into this community's +// repository". Get it wrong in the permissive direction and the engine starts +// deciding about communities it does not host: it runs the policy, reaches a +// verdict, and then discovers at the PDS that it holds no keys — after the +// decision, and pointed at whatever host that community's record named. +// +// The tempting implementation is a hosted_by_did comparison, and it is the wrong +// one for a reason that is invisible from the column name. hosted_by_did is +// copied out of a community's own PROFILE RECORD as it is indexed from the +// firehose, so it is a claim made by whoever controls that repo. Anyone can +// publish a community profile naming this AppView as its host. Every fixture +// community in this suite has exactly that shape, which is what the second test +// below exploits. +// +// Credentials cannot be claimed: a stored refresh token exists only because this +// AppView provisioned the account itself, through +// social.coves.community.create. That is the honest question, and it is the only +// one the factory is allowed to ask. + +func TestCommunityRepoFactory_OpensAHostedCommunitysRepo(t *testing.T) { + t.Parallel() + + fixture := newPostFixture(t) + factory := posts.NewCommunityRepoFactory(fixture.communityService) + + repo, err := factory(context.Background(), fixture.community.DID) + require.NoError(t, err, "the AppView provisioned this community's account, so it holds its credentials") + require.NotNil(t, repo) + + assert.Equal(t, fixture.community.DID, repo.DID(), + "the repo's DID is the authority half of every record URI the writers produce; a client bound to the wrong repo "+ + "would mint acceptance URIs under a DID that never signed them") + + // An authenticated round trip, not merely a constructed struct. A factory + // that returned a client with an empty or stale token would satisfy every + // assertion above and then fail on the engine's first write — which is the + // point at which a verdict has already been reached. + commit, err := repo.GetLatestCommit(context.Background()) + require.NoError(t, err, + "the returned client must be able to talk to the repo: an unauthenticated one fails later, after the engine has already decided") + require.NotNil(t, commit) +} + +func TestCommunityRepoFactory_RefusesACommunityItOnlyClaimsToHost(t *testing.T) { + t.Parallel() + + fixture := newPostFixture(t) + + // A community indexed from the firehose: it names this instance in + // hosted_by_did — the claim any repo on the network can make — and stores no + // credentials, because nothing here ever provisioned it. + label := testkit.UniqueIDWithPrefix(t, "claim") + claimant, err := fixtures.Community(context.Background(), fixture.db, label, "owner"+label) + require.NoError(t, err) + + var claimedHost string + require.NoError(t, fixture.db.QueryRow( + `SELECT hosted_by_did FROM communities WHERE did = $1`, claimant).Scan(&claimedHost)) + require.NotEmpty(t, claimedHost, + "fixture: the community must CLAIM a host, or this proves nothing about which signal the factory trusts") + + repo, err := posts.NewCommunityRepoFactory(fixture.communityService)(context.Background(), claimant) + + require.Error(t, err, + "a community whose credentials this AppView does not hold must be refused, whatever its profile record claims about who hosts it") + assert.Nil(t, repo) + + // The SPELLING of the refusal is what the driver switches on. A permanent + // skip tells it to stop offering this subject; a generic error reads as + // transient, so every pass would re-list every remote community's posts, + // forever, and the deferrals worth looking at would be buried underneath. + assert.ErrorIs(t, err, posts.ErrCommunityNotHosted, + "the refusal must be ErrCommunityNotHosted: not hosting is permanent, and an unclassified error would be retried until the heat death of the queue") +} + +func TestCommunityRepoFactory_RefusesACommunityNobodyHasIndexed(t *testing.T) { + t.Parallel() + + fixture := newPostFixture(t) + + repo, err := posts.NewCommunityRepoFactory(fixture.communityService)( + context.Background(), "did:plc:aaaaaaaaaanevercommunity") + + require.Error(t, err, "a DID naming no indexed community cannot yield a repo client") + assert.Nil(t, repo) +} diff --git a/internal/core/posts/community_writer.go b/internal/core/posts/community_writer.go index 8942c84..1b45ebc 100644 --- a/internal/core/posts/community_writer.go +++ b/internal/core/posts/community_writer.go @@ -173,6 +173,42 @@ type CommunityRecordWriter interface { // meaning "when this community accepted this post" rather than being // restamped every time a bridge refreshes its vote counts. RepinAcceptance(ctx context.Context, cmd CommunityWriteCommand) (CommunityWriteResult, error) + + // DeleteAcceptance withdraws this community's acceptance WITHOUT writing a + // removal, which is the one shape the other four cannot express. + // + // It exists for the author's own deletion (§5.3): the author tombstones + // their post, and the community's acceptance now points at a record that no + // longer exists. Leaving it standing means the community's repo — the + // curated index its whole portability argument rests on — permanently cites + // content nobody can fetch, and any peer replaying that CAR would show a + // post the author withdrew. + // + // IT IS NOT A REMOVAL, and conflating the two would be a factual error the + // firehose carries forever. A removal record is a MODERATION act, signed by + // the community, carrying a reason code, portable and auditable. An author + // deleting their own post is not the community judging anything, and + // publishing a removal for it would put a moderation event in the public + // record that never happened. + // + // Deleting a record that is not there is a no-op reported as a skip, not an + // error: the sweep is idempotent by necessity — every tombstone event may be + // redelivered, and the acceptance may already have been withdrawn by an + // earlier pass. + DeleteAcceptance(ctx context.Context, cmd CommunityAcceptanceDeleteCommand) (CommunityWriteResult, error) +} + +// CommunityAcceptanceDeleteCommand withdraws an acceptance. +// +// It carries no CID, deliberately. Every other command pins one because it is +// making a claim about a specific version; this one is undoing a claim, and the +// subject it is undoing it for is identified by URI — the same URI the +// deterministic rkey is derived from. A CID here would suggest the delete is +// conditional on a version, which it is not: the post is gone, whatever version +// the acceptance happened to pin. +type CommunityAcceptanceDeleteCommand struct { + CommunityDID string + PostURI string } // communityRecordWriter is the production writer over real repos. @@ -317,6 +353,11 @@ func (w *communityRecordWriter) RepinAcceptance(ctx context.Context, cmd Communi return w.pinAcceptance(ctx, cmd, acceptanceMustExist) } +// DeleteAcceptance is a RED STUB (task 5, cycle 2); the body is GREEN's. +func (w *communityRecordWriter) DeleteAcceptance(ctx context.Context, cmd CommunityAcceptanceDeleteCommand) (CommunityWriteResult, error) { + return CommunityWriteResult{}, nil +} + // pinAcceptance makes an acceptance of cmd.PostCID stand at the subject's rkey. // // THE PRE-READ DECIDES EVERYTHING. Whether there is work to do at all, what the diff --git a/internal/core/posts/decider.go b/internal/core/posts/decider.go new file mode 100644 index 0000000..fda07c6 --- /dev/null +++ b/internal/core/posts/decider.go @@ -0,0 +1,99 @@ +package posts + +import ( + "context" +) + +// RED STUB (task 5, cycle 2). Signatures only; the body is GREEN's. + +// The production AdmissionDecider: the adapter that turns "decide about this +// indexed post" into the AdmissionRequest evaluateAdmissionPolicy already +// answers (docs/PRD_AUTHOR_OWNED_POSTS.md §5.6). +// +// The engine's input is an admission row — a community DID and a post URI, and +// nothing about who wrote the post or what class of actor they are. Everything +// the policy needs beyond that has to be recovered from the index, and the two +// recoveries have opposite failure rules, which is most of what this component +// is: +// +// - THE POST. Absent or tombstoned means there is nothing to decide, and the +// answer must NOT be an admission. This is the second half of the +// resurrection guard: the driver already excludes tombstoned subjects, but a +// post can be deleted between the listing and the decision, and a decider +// that admitted it would write an acceptance for content the tombstone sweep +// is about to delete — each component correct, the pair looping. +// - THE ACTOR CLASS. A misclassification is a privilege decision. Trusted +// aggregators skip visibility, ban and authorization entirely, so guessing +// UPWARD on a failed lookup would hand the widest privileges in the system +// to whoever made the lookup fail. Every uncertain path therefore falls to +// the STRICTER class, matching CreatePost's existing behaviour (service.go +// step 3 treats a failed IsAggregator lookup as an ordinary user). +// +// It reuses evaluateAdmissionPolicy rather than admitPost, and that is the split +// task 3 built for: admitPost RESERVES a ledger slot, and the engine is not a +// submission. Reserving here would charge an author's quota for a firehose +// redelivery and then refuse the redecision as a duplicate of the very post it +// is redeciding. + +// PostLookup reads the indexed post a decision is about. Satisfied by +// Repository. +type PostLookup interface { + GetByURI(ctx context.Context, uri string) (*Post, error) +} + +// AggregatorLookup reports whether a DID is a registered aggregator. Satisfied +// by aggregators.Service. +type AggregatorLookup interface { + IsAggregator(ctx context.Context, did string) (bool, error) +} + +// DeciderDeps is everything the production decider reads. +// +// A struct rather than six positional parameters, and not only for readability: +// two of these are aggregator collaborators with very different jobs — +// Authorizer answers "may this aggregator post here", Aggregators answers "is +// this DID an aggregator at all" — and in production both are the same object. +// Positionally they would be adjacent, same-shaped, and silently swappable. +type DeciderDeps struct { + // Posts reads the indexed post the decision is about. + Posts PostLookup + + // Communities resolves the community the admission row names. + Communities CommunityLookup + + // Authorizer checks a registered aggregator's authorization and quota. May + // be nil on a deployment with no aggregator support, in which case no author + // is ever classified as a registered aggregator. + Authorizer AggregatorAuthorizer + + // Aggregators classifies an author. May be nil, with the same consequence. + Aggregators AggregatorLookup + + // Policy is the ban lookup, ledger, limits and clock admitPost already uses. + Policy AdmissionPolicy + + // TrustedAggregatorDIDs is the set from TRUSTED_AGGREGATOR_DIDS, resolved + // ONCE at construction rather than read per decision. + // + // Reading the environment inside the decision would repeat the mistake + // ActorClass's doc comment describes: it hides the most consequential input + // to a security decision from the place that makes it, and it makes the + // trusted branch untestable alongside t.Parallel, since Go's testing package + // refuses t.Setenv there. + TrustedAggregatorDIDs map[string]bool +} + +// AdmissionEngineDecider is the production AdmissionDecider. +type AdmissionEngineDecider struct { + deps DeciderDeps +} + +// NewAdmissionEngineDecider wires the decider. +func NewAdmissionEngineDecider(deps DeciderDeps) *AdmissionEngineDecider { + return &AdmissionEngineDecider{deps: deps} +} + +// DecideAdmission implements AdmissionDecider. +func (d *AdmissionEngineDecider) DecideAdmission(ctx context.Context, communityDID, postURI string) (AdmissionDecision, error) { + return AdmissionDecision{}, nil +} diff --git a/internal/core/posts/decider_test.go b/internal/core/posts/decider_test.go new file mode 100644 index 0000000..3ef13f3 --- /dev/null +++ b/internal/core/posts/decider_test.go @@ -0,0 +1,311 @@ +package posts + +import ( + "context" + "errors" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// The production AdmissionDecider (docs/PRD_AUTHOR_OWNED_POSTS.md §5.6): the +// adapter between "decide about this indexed post" and the policy matrix +// admit_matrix_test.go already pins. +// +// THIS IS T0 AND NOT T1, deliberately, though the driver and repository halves +// of this task are T1. Every collaborator here is an interface — a post lookup, +// an aggregator lookup, the injected policy — and the tier is chosen by what a +// test NEEDS out of process (docs/TEST_ARCHITECTURE.md), not by which task it +// belongs to. Nothing below opens a socket, and a Postgres-backed version of +// these cases would prove the same things more slowly while making the actor +// matrix awkward to enumerate. +// +// What is asserted here is exactly the seam the policy matrix cannot see: which +// AdmissionRequest gets built. The matrix takes a request and proves the +// verdict; this proves the request is the right one, which is where the two +// genuinely dangerous mistakes live. +// +// - THE ACTOR CLASS IS A PRIVILEGE DECISION. A trusted aggregator skips +// visibility, ban and authorization entirely. Anything that guesses UPWARD +// when it is unsure hands the widest privileges in the system to whoever can +// make a lookup fail. +// - A DELETED POST HAS NOTHING TO DECIDE. Admitting one writes an acceptance +// for content that no longer exists, and the host-side tombstone sweep then +// deletes that acceptance — two components looping against each other, each +// correct alone. + +const ( + deciderCommunityDID = "did:plc:decidercommunity" + deciderAuthorDID = "did:plc:deciderauthor" + deciderTrustedDID = "did:plc:decidertrustedbot" + deciderPostURI = "at://" + deciderAuthorDID + "/social.coves.community.postv2/decided" +) + +// stubPostLookup answers with one post, or with a failure. +type stubPostLookup struct { + post *Post + err error + + calls int +} + +func (s *stubPostLookup) GetByURI(_ context.Context, _ string) (*Post, error) { + s.calls++ + if s.err != nil { + return nil, s.err + } + if s.post == nil { + return nil, ErrNotFound + } + return s.post, nil +} + +// stubAggregatorLookup is the IsAggregator classification lookup. +type stubAggregatorLookup struct { + registered map[string]bool + err error + + calls int +} + +func (s *stubAggregatorLookup) IsAggregator(_ context.Context, did string) (bool, error) { + s.calls++ + if s.err != nil { + return false, s.err + } + return s.registered[did], nil +} + +// deciderHarness is the default world: a live post by an ordinary author in a +// public community, with the same policy stubs the admit matrix uses. +type deciderHarness struct { + *admitHarness + + posts *stubPostLookup + aggregators *stubAggregatorLookup + trusted map[string]bool +} + +func newDeciderHarness() *deciderHarness { + base := newAdmitHarness() + base.communities.community.DID = deciderCommunityDID + return &deciderHarness{ + admitHarness: base, + posts: &stubPostLookup{post: &Post{ + URI: deciderPostURI, + CID: "bafyreidecided", + AuthorDID: deciderAuthorDID, + CommunityDID: deciderCommunityDID, + }}, + aggregators: &stubAggregatorLookup{registered: map[string]bool{}}, + trusted: map[string]bool{}, + } +} + +func (h *deciderHarness) decider() *AdmissionEngineDecider { + return NewAdmissionEngineDecider(DeciderDeps{ + Posts: h.posts, + Communities: h.communities, + Authorizer: h.aggregators2(), + Aggregators: h.aggregators, + Policy: AdmissionPolicy{ + Ledger: h.ledger, + Bans: h.bans, + Limits: h.limits, + Now: h.clock(), + }, + TrustedAggregatorDIDs: h.trusted, + }) +} + +// aggregators2 is the admit harness's authorization stub, named apart from the +// classification lookup because the two answer different questions and only +// this file holds both at once. +func (h *deciderHarness) aggregators2() *stubAggregatorAuthorizer { return h.admitHarness.aggregators } + +func (h *deciderHarness) decide(t *testing.T) (AdmissionDecision, error) { + t.Helper() + return h.decider().DecideAdmission(context.Background(), deciderCommunityDID, deciderPostURI) +} + +func TestDecider_AdmitsAnOrdinaryPostInAPublicCommunity(t *testing.T) { + t.Parallel() + + h := newDeciderHarness() + decision, err := h.decide(t) + + require.NoError(t, err) + assert.True(t, decision.Admitted(), "a live post by an unbanned author in a public community is admitted: %+v", decision) + + // An admission has to be EARNED, not defaulted. AdmissionDecision's zero + // value reports Admitted() true — deliberately, since a decision is an + // admission only when there is neither a code nor a cause — so a decider + // that returned early, or was never implemented, produces exactly the + // verdict above. These two assertions are what separate "the policy ran and + // said yes" from "nothing ran at all". + assert.Positivef(t, h.posts.calls, + "the decision was reached without reading the post it is about (%d lookups)", h.posts.calls) + assert.Positivef(t, h.communities.resolveCalls, + "the decision was reached without resolving the community (%d lookups): an admission nobody evaluated is the zero value, not a verdict", h.communities.resolveCalls) + + // The ledger is what separates this from admitPost, and the separation is + // the whole reason task 3 split the policy out. The engine is not a + // submission — it is re-deciding a post that already exists, often one it + // has decided before — so reserving here would charge the author's quota for + // a firehose redelivery and then refuse the redecision as a duplicate of the + // very post it is redeciding. + assert.Emptyf(t, h.ledger.reserveCalls, + "the decider must NOT reserve a ledger slot: it re-decides existing posts, and a redelivery would consume quota and then be refused as its own duplicate") + assert.Nil(t, decision.Reservation, "a decision that reserved nothing must not report a reservation") +} + +func TestDecider_RefusesATombstonedPostWithoutRunningPolicy(t *testing.T) { + t.Parallel() + + h := newDeciderHarness() + deleted := time.Date(2026, 8, 1, 11, 0, 0, 0, time.UTC) + h.posts.post.DeletedAt = &deleted + + decision, err := h.decide(t) + + // Not an admission, whichever way it is spelled. The driver already excludes + // tombstoned subjects, but a post can be deleted between the listing and the + // decision, and this is the guard that closes that window. + assert.Falsef(t, decision.Admitted(), + "a tombstoned post must never be admitted: the acceptance would pin content that no longer exists, and the tombstone sweep would then delete the acceptance — forever, once per pass: %+v", decision) + if err == nil { + assert.NotEmpty(t, decision.Code, + "a refusal that is not an error must carry a code; a decision with neither reads as an admission to Admitted()") + } + + // And the policy never ran. Evaluating a deleted post costs a community + // resolve, a ban lookup and a quota count to answer a question that has no + // content behind it — and worse, it can produce a REJECTION code that gets + // written onto the row as this community's verdict about a post nobody can + // read. + assert.Zerof(t, h.communities.resolveCalls, + "the tombstone check must run BEFORE the policy: %d community lookups happened for a post that no longer exists", h.communities.resolveCalls) + assert.Zero(t, h.bans.calls, "no ban lookup for a deleted post") +} + +func TestDecider_RefusesAPostItCannotFind(t *testing.T) { + t.Parallel() + + h := newDeciderHarness() + h.posts.post = nil // GetByURI answers ErrNotFound + + decision, err := h.decide(t) + + assert.Falsef(t, decision.Admitted(), + "an admission row whose post was never indexed has no content to judge and must not be admitted: %+v", decision) + if err == nil { + assert.NotEmpty(t, decision.Code, "a refusal must carry a code") + } + assert.Zero(t, h.communities.resolveCalls, "a subject with no post must not reach the policy") +} + +func TestDecider_ClassifiesTheAuthorFromTheTrustedSet(t *testing.T) { + t.Parallel() + + // A trusted aggregator skips visibility, ban and authorization (admit.go's + // check order). Proving the class was applied means proving those lookups + // did NOT happen — the verdict alone cannot distinguish "trusted, so skipped" + // from "an ordinary user who happened to pass every check". + h := newDeciderHarness() + h.posts.post.AuthorDID = deciderTrustedDID + h.trusted[deciderTrustedDID] = true + h.bans.membership = banned() + + decision, err := h.decide(t) + + require.NoError(t, err) + assert.Truef(t, decision.Admitted(), + "a trusted aggregator skips the ban check entirely; a refusal here means the author was classified as an ordinary user: %+v", decision) + // The class can only have been applied if the AUTHOR was read, and the + // author only comes from the post. Without this, a decider that returned the + // zero value would satisfy every assertion below by never looking at + // anything. + assert.Positivef(t, h.posts.calls, + "the author's class was decided without reading the post that names them (%d lookups)", h.posts.calls) + assert.Zerof(t, h.bans.calls, + "the ban lookup ran for a TRUSTED aggregator (%d calls): the class was not applied", h.bans.calls) + assert.Zerof(t, h.aggregators.calls, + "a DID in the trusted set must not cost an IsAggregator lookup — the set is checked first, and the lookup is the expensive path") +} + +func TestDecider_ClassifiesARegisteredAggregator(t *testing.T) { + t.Parallel() + + h := newDeciderHarness() + h.aggregators.registered[deciderAuthorDID] = true + + decision, err := h.decide(t) + + require.NoError(t, err) + assert.True(t, decision.Admitted(), "an authorized aggregator is admitted: %+v", decision) + assert.Positivef(t, h.aggregators2().calls, + "a registered aggregator must be held to the community's authorization record; %d authorization checks ran", h.aggregators2().calls) + assert.Zero(t, h.bans.calls, "an aggregator is not held to member bans") +} + +func TestDecider_FallsToTheStricterClassWhenTheLookupFails(t *testing.T) { + t.Parallel() + + // The classification lookup is down. There are two ways to be wrong here and + // only one of them is survivable: guessing "aggregator" skips the ban and + // visibility checks, so a database blip would become a window in which every + // banned author's posts are accepted. Guessing "user" costs an aggregator + // some refused posts until the lookup recovers. CreatePost already chose the + // second (service.go step 3), and the engine must not disagree with the write + // path about who someone is. + h := newDeciderHarness() + h.aggregators.err = errors.New("aggregators table unreachable") + h.bans.membership = banned() + + decision, err := h.decide(t) + + require.NoError(t, err, "a failed CLASSIFICATION is not a failed decision: the stricter class is a safe answer, not an outage") + assert.Falsef(t, decision.Admitted(), + "a failed IsAggregator lookup must fall to ActorUser, and this author is banned — an admission here means the failure was resolved upward into aggregator privileges: %+v", decision) + assert.Equal(t, DecisionAuthorBanned, decision.Code, + "falling to the user class means the ban check runs and answers") + assert.Positive(t, h.bans.calls, "the stricter class must actually apply the checks it implies") +} + +func TestDecider_IsUndecidedWhenThePolicyCannotBeEvaluated(t *testing.T) { + t.Parallel() + + // An infrastructure failure inside the policy — the ban lookup is down — must + // come back as UNDECIDED, never as a code. The engine writes a decision code + // onto the admission row and sets redrivable=false for policy refusals, so a + // Postgres blip minted as a code becomes a permanent verdict about somebody's + // post that nothing will ever retry. + h := newDeciderHarness() + h.bans.err = errors.New("memberships unreachable") + + decision, err := h.decide(t) + + require.Error(t, err, "a policy that could not be evaluated must report an error, not a verdict") + assert.False(t, decision.Admitted(), "an undecided answer is not an admission") + assert.Emptyf(t, decision.Code, + "an infrastructure failure minted the decision code %q: the engine persists codes and marks policy refusals non-redrivable, so an outage would become a permanent refusal", decision.Code) +} + +func TestDecider_IsUndecidedWhenThePostLookupFails(t *testing.T) { + t.Parallel() + + // The mirror of the tombstone case, and it must NOT collapse into it. "The + // post is gone" and "I could not read the post" look the same from the call + // site and mean opposite things: the first is terminal, the second clears. + h := newDeciderHarness() + h.posts.err = errors.New("posts table unreachable") + + decision, err := h.decide(t) + + require.Error(t, err, "a post lookup that FAILED is not a post that is absent") + assert.False(t, decision.Admitted()) + assert.Emptyf(t, decision.Code, + "a failed post lookup minted the code %q; an unreadable post must be retried, not permanently refused", decision.Code) +} diff --git a/internal/core/posts/engine_matrix_test.go b/internal/core/posts/engine_matrix_test.go index 9b81ac4..987dc15 100644 --- a/internal/core/posts/engine_matrix_test.go +++ b/internal/core/posts/engine_matrix_test.go @@ -161,6 +161,17 @@ func (w *fakeWriter) RepinAcceptance(_ context.Context, _ CommunityWriteCommand) return CommunityWriteResult{}, nil } +// DeleteAcceptance is the author-deletion sweep, which the ENGINE never +// performs — an author withdrawing their own post is not a verdict. It records +// its call for the same reason every other method here does: the recorder is +// how this file asserts which repo write each verdict produced, so an engine +// that started withdrawing acceptances would show up as an unexpected entry +// rather than as a silent behaviour change. +func (w *fakeWriter) DeleteAcceptance(_ context.Context, _ CommunityAcceptanceDeleteCommand) (CommunityWriteResult, error) { + w.rec.record("DeleteAcceptance") + return CommunityWriteResult{}, nil +} + // fakeRefresher counts forced credential renewals. type fakeRefresher struct { rec *engineRecorder @@ -250,6 +261,15 @@ func (a *fakeAdmissions) ListByStatusForCommunity(_ context.Context, _ string, _ return nil, nil, nil } +// ListPendingSubjects is the queue driver's backlog query. The engine never +// calls it — the driver does, and drives the engine with what it returns — so +// this exists to satisfy the interface and records the call, which is itself an +// assertion: an engine that started listing its own work would show up here. +func (a *fakeAdmissions) ListPendingSubjects(_ context.Context, _ int) ([]PendingSubject, error) { + a.rec.record("ListPendingSubjects") + return nil, nil +} + // --------------------------------------------------------------------------- // Harness // --------------------------------------------------------------------------- diff --git a/internal/core/posts/queue.go b/internal/core/posts/queue.go new file mode 100644 index 0000000..f0c539a --- /dev/null +++ b/internal/core/posts/queue.go @@ -0,0 +1,168 @@ +package posts + +import ( + "context" + "time" +) + +// RED STUB (task 5, cycle 2). Signatures only; every body returns zero values. + +// The acceptance engine's driver: the thing that decides WHEN the engine runs +// and on what (docs/PRD_AUTHOR_OWNED_POSTS.md §5.6, §8). +// +// The engine settles one subject. Nothing until now decided which subjects, in +// what order, or how often — the fast path and the firehose consumer both push +// work at it, and neither can see a subject that was left pending because a +// credential expired or a lookup blipped. This is the pull side: a periodic pass +// over the undecided backlog that eventually reaches every stranded row. +// +// # IT IS A SINGLE GOROUTINE, AND THAT IS THE PER-COMMUNITY SERIALIZATION +// +// Task 4 recorded the requirement: serialize the queue per community DID, +// because swapCommit is repo-global and sibling workers on one busy community +// would starve each other's removals. In v1 that requirement is satisfied +// trivially and completely — one goroutine walks one ordered list, so no two +// subjects of any community are ever in flight together. The ordering matters +// anyway: the list is grouped by community so that when this DOES grow a worker +// pool, the partition it has to shard on is already the shape of the data, +// rather than something a later change has to introduce. +// +// # A DEFERRAL IS NOT A FAILURE, AND MUST NOT BECOME A SPIN +// +// EngineDeferred is the common outcome, not the exceptional one: a community +// whose token expired, a post whose content CID has not been decoded yet, a +// policy lookup that could not be reached. Every one of those is "look again +// later", and a driver that took it as "try again now" would turn one wedged +// subject into a hot loop against a PDS that is already unhappy. Deferred +// subjects therefore carry a per-subject backoff, and the same subject is never +// touched twice in one pass — §8's edit-debounce falls out of the same rule, +// since a post being edited in a storm coalesces into one pending_reacceptance +// row that this pass sees exactly once. + +// AdmissionProcessor settles one subject. Satisfied by *AcceptanceEngine. +type AdmissionProcessor interface { + ProcessAdmission(ctx context.Context, communityDID, postURI string) (EngineOutcome, error) +} + +// PendingSubjectLister supplies the backlog. Satisfied by AdmissionRepository. +type PendingSubjectLister interface { + ListPendingSubjects(ctx context.Context, limit int) ([]PendingSubject, error) +} + +// PassReport is what one pass did, for logs and for the health surface. +// +// Deferred and Failed are counted separately because they mean opposite things +// to an operator. A pass that defers everything is usually a credential or +// connectivity problem that will clear; a pass that FAILS everything is a bug. +// Collapsing them into one number would make the first look like the second at +// exactly the moment somebody is deciding whether to page. +type PassReport struct { + // Listed is how many subjects the backlog query returned. + Listed int + + // Processed is how many were handed to the engine — Listed minus those the + // backoff held back. + Processed int + + // Settled is how many reached a verdict: accepted, rejected, removed or + // repinned. + Settled int + + // Deferred is how many the engine declined to decide yet. + Deferred int + + // Failed is how many returned an error. + Failed int + + // StartedAt is when the pass began. + StartedAt time.Time +} + +// QueueSnapshot is the driver's health surface: what the backlog looks like and +// when it was last worked. +// +// LastPassAt is the liveness signal and the reason the snapshot exists at all. +// A driver that has stopped running produces no logs and no errors — it simply +// stops, and every symptom of that appears somewhere else, as posts that never +// become visible. An operator needs one number that says the pass is happening. +type QueueSnapshot struct { + PendingBacklog int + + // OldestPendingAt is the created_at of the oldest subject the last pass + // listed, or nil when the backlog was empty. + OldestPendingAt *time.Time + + // LastPassAt is nil until the first pass completes — distinguishing "the + // driver has never run" from "the driver ran and found nothing", which look + // identical in every other field. + LastPassAt *time.Time + + LastPassDeferred int + LastPassFailed int +} + +// QueueDriverOption configures the driver. +type QueueDriverOption func(*QueueDriver) + +// WithQueueBatchSize bounds how many subjects one pass lists. +func WithQueueBatchSize(size int) QueueDriverOption { + return func(d *QueueDriver) { d.batchSize = size } +} + +// WithQueueBackoff sets the delay before a DEFERRED subject is offered to the +// engine again, and the ceiling that delay grows to. +// +// It is per-subject rather than global: one community with dead credentials must +// not slow down the queue for everyone else, which is exactly what a global +// backoff would do. +func WithQueueBackoff(base, max time.Duration) QueueDriverOption { + return func(d *QueueDriver) { d.backoffBase, d.backoffMax = base, max } +} + +// QueueDriver walks the undecided backlog and feeds the engine. +type QueueDriver struct { + subjects PendingSubjectLister + engine AdmissionProcessor + now Clock + batchSize int + + backoffBase time.Duration + backoffMax time.Duration + + // deferrals holds the per-subject retry-not-before times. In-memory on + // purpose: it is a politeness hint, not state anything is allowed to depend + // on, so a restart that forgets it costs one extra attempt per subject and + // nothing else. + deferrals map[PendingSubject]time.Time + + snapshot QueueSnapshot +} + +// NewQueueDriver wires the driver. +func NewQueueDriver(subjects PendingSubjectLister, engine AdmissionProcessor, now Clock, opts ...QueueDriverOption) *QueueDriver { + d := &QueueDriver{ + subjects: subjects, + engine: engine, + now: now, + deferrals: make(map[PendingSubject]time.Time), + } + for _, opt := range opts { + opt(d) + } + return d +} + +// RunPass processes one batch of the backlog. +// +// It returns an error ONLY when the backlog itself could not be read. A subject +// the engine failed on is counted and the pass continues: one poisonous row +// must not stop every other community's posts from being decided, which is what +// an early return would do — and the row is still in the backlog next pass. +func (d *QueueDriver) RunPass(ctx context.Context) (PassReport, error) { + return PassReport{}, nil +} + +// Snapshot returns the driver's health surface as of the last completed pass. +func (d *QueueDriver) Snapshot() QueueSnapshot { + return QueueSnapshot{} +} diff --git a/internal/core/posts/queue_test.go b/internal/core/posts/queue_test.go new file mode 100644 index 0000000..0315873 --- /dev/null +++ b/internal/core/posts/queue_test.go @@ -0,0 +1,334 @@ +package posts + +import ( + "context" + "errors" + "sync" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// The acceptance engine's queue driver (docs/PRD_AUTHOR_OWNED_POSTS.md §5.6, +// §8): what turns "the engine can settle a subject" into "every stranded +// subject eventually gets settled". +// +// T0, because the driver is a scheduling policy over two interfaces and nothing +// it decides needs a database to be true. The backlog QUERY it drives is T1 +// (internal/db/postgres/admission_queue_repo_test.go); this is the half that +// decides what to do with the answer. +// +// # THE THREE PROPERTIES, AND WHY EACH IS A REAL FAILURE MODE +// +// - ONE SUBJECT AT A TIME, GROUPED BY COMMUNITY. swapCommit is repo-global, so +// two workers on one community's repo starve each other's writes — task 4 +// recorded this before the driver existed. v1 satisfies it trivially with a +// single goroutine, but the ORDER still matters: the partition a worker pool +// would eventually shard on has to already be the shape of the data. +// - A DEFERRAL IS NOT A RETRY SIGNAL. EngineDeferred is the common outcome — +// an expired token, an undecoded CID, an unreachable lookup — and a driver +// that re-offered it immediately would hammer a PDS that is already +// unhappy. This is the difference between a queue and a spin loop. +// - ONE BAD SUBJECT MUST NOT STOP THE PASS. An error on one row aborting the +// loop means one poisoned admission freezes every other community's posts, +// and the symptom appears as "posts stopped becoming visible" with nothing +// pointing at the row that caused it. + +// fakeSubjects is the backlog. +type fakeSubjects struct { + batches [][]PendingSubject + err error + + limits []int + calls int +} + +func (f *fakeSubjects) ListPendingSubjects(_ context.Context, limit int) ([]PendingSubject, error) { + f.limits = append(f.limits, limit) + f.calls++ + if f.err != nil { + return nil, f.err + } + if len(f.batches) == 0 { + return nil, nil + } + batch := f.batches[0] + if len(f.batches) > 1 { + f.batches = f.batches[1:] + } + return batch, nil +} + +// fakeEngine records the subjects it was handed, in order, and answers each +// with a scripted outcome. +type fakeEngine struct { + mu sync.Mutex + seen []PendingSubject + outcomes map[string]EngineOutcome + errs map[string]error +} + +func newFakeEngine() *fakeEngine { + return &fakeEngine{ + outcomes: map[string]EngineOutcome{}, + errs: map[string]error{}, + } +} + +func (e *fakeEngine) ProcessAdmission(_ context.Context, communityDID, postURI string) (EngineOutcome, error) { + e.mu.Lock() + defer e.mu.Unlock() + e.seen = append(e.seen, PendingSubject{CommunityDID: communityDID, PostURI: postURI}) + if err := e.errs[postURI]; err != nil { + return EngineDeferred, err + } + if outcome, ok := e.outcomes[postURI]; ok { + return outcome, nil + } + return EngineAccepted, nil +} + +func (e *fakeEngine) uris() []string { + e.mu.Lock() + defer e.mu.Unlock() + uris := make([]string, 0, len(e.seen)) + for _, subject := range e.seen { + uris = append(uris, subject.PostURI) + } + return uris +} + +func (e *fakeEngine) communityOrder() []string { + e.mu.Lock() + defer e.mu.Unlock() + var order []string + for _, subject := range e.seen { + if len(order) == 0 || order[len(order)-1] != subject.CommunityDID { + order = append(order, subject.CommunityDID) + } + } + return order +} + +// queueClock is a mutable instant the driver reads through Clock. +type queueClock struct{ at time.Time } + +func (c *queueClock) now() Clock { return func() time.Time { return c.at } } +func (c *queueClock) advance(d time.Duration) { c.at = c.at.Add(d) } +func newQueueClock() *queueClock { return &queueClock{at: time.Date(2026, 8, 8, 9, 0, 0, 0, time.UTC)} } +func subject(community, rkey string) PendingSubject { + return PendingSubject{ + CommunityDID: community, + PostURI: "at://did:plc:queueauthor/social.coves.community.postv2/" + rkey, + CreatedAt: time.Date(2026, 8, 8, 8, 0, 0, 0, time.UTC), + } +} + +func TestQueueDriver_ProcessesEverySubjectOncePerPass(t *testing.T) { + t.Parallel() + + clock := newQueueClock() + first, second := subject("did:plc:qa", "a"), subject("did:plc:qa", "b") + subjects := &fakeSubjects{batches: [][]PendingSubject{{first, second}}} + engine := newFakeEngine() + + report, err := NewQueueDriver(subjects, engine, clock.now(), WithQueueBatchSize(10)). + RunPass(context.Background()) + + require.NoError(t, err) + assert.Equal(t, []string{first.PostURI, second.PostURI}, engine.uris(), + "every listed subject must be offered to the engine, in the order the backlog returned them") + assert.Equal(t, 2, report.Listed) + assert.Equal(t, 2, report.Processed) + assert.Equal(t, 2, report.Settled) + assert.Equal(t, []int{10}, subjects.limits, + "the batch size must reach the query: a driver that listed the whole backlog would hold one transaction open across a table that grows forever") +} + +func TestQueueDriver_GroupsWorkByCommunity(t *testing.T) { + t.Parallel() + + clock := newQueueClock() + // Interleaved on the way in — the backlog is ordered by age across + // communities, so this is the shape it genuinely returns. + subjects := &fakeSubjects{batches: [][]PendingSubject{{ + subject("did:plc:qa", "a1"), + subject("did:plc:qb", "b1"), + subject("did:plc:qa", "a2"), + subject("did:plc:qb", "b2"), + }}} + engine := newFakeEngine() + + _, err := NewQueueDriver(subjects, engine, clock.now()).RunPass(context.Background()) + require.NoError(t, err) + + // Each community's subjects arrive contiguously. In v1 the single goroutine + // already makes this safe, so what the assertion protects is the NEXT + // version: the moment this grows a worker pool, the shard key has to be the + // community, and a driver whose output interleaves communities is one that + // would be sharded on nothing. + order := engine.communityOrder() + assert.Len(t, order, 2, + "a community's subjects must be contiguous; interleaving them (%v) means a future worker pool has no partition to shard on and would run two writers against one repo, where swapCommit is repo-global", order) + assert.ElementsMatch(t, []string{"did:plc:qa", "did:plc:qb"}, order) + assert.Len(t, engine.uris(), 4, "grouping must not drop or duplicate work") +} + +func TestQueueDriver_DoesNotTouchOneSubjectTwiceInAPass(t *testing.T) { + t.Parallel() + + clock := newQueueClock() + hot := subject("did:plc:qa", "hot") + // The same subject twice in one listing: exactly what a duplicated row or a + // racing edit produces, and the shape §8's edit-debounce has to survive. + subjects := &fakeSubjects{batches: [][]PendingSubject{{hot, hot, subject("did:plc:qa", "cold")}}} + engine := newFakeEngine() + engine.outcomes[hot.PostURI] = EngineDeferred + + report, err := NewQueueDriver(subjects, engine, clock.now()).RunPass(context.Background()) + require.NoError(t, err) + + seen := 0 + for _, uri := range engine.uris() { + if uri == hot.PostURI { + seen++ + } + } + assert.Equalf(t, 1, seen, + "a subject was processed %d times in one pass. §8's edit-debounce is exactly this rule: a post being edited in a storm coalesces "+ + "into one pending_reacceptance row, and a driver that re-decided it per occurrence would re-run the whole policy per keystroke", seen) + assert.Equal(t, 1, report.Deferred) +} + +func TestQueueDriver_HoldsADeferredSubjectBackUntilItsBackoffElapses(t *testing.T) { + t.Parallel() + + clock := newQueueClock() + stuck := subject("did:plc:qa", "stuck") + healthy := subject("did:plc:qb", "healthy") + // The same backlog every pass — which is what the query genuinely returns + // while the subject stays undecided. + subjects := &fakeSubjects{batches: [][]PendingSubject{{stuck, healthy}}} + engine := newFakeEngine() + engine.outcomes[stuck.PostURI] = EngineDeferred + + driver := NewQueueDriver(subjects, engine, clock.now(), + WithQueueBackoff(time.Minute, 10*time.Minute)) + ctx := context.Background() + + first, err := driver.RunPass(ctx) + require.NoError(t, err) + assert.Equal(t, 2, first.Processed) + assert.Equal(t, 1, first.Deferred) + + // Immediately again. A community with a dead token defers on every pass, and + // a driver that re-offered it at once would turn a scheduled job into a spin + // against a PDS that is already failing. + clock.advance(time.Second) + second, err := driver.RunPass(ctx) + require.NoError(t, err) + + countIn := func(uris []string, target string) int { + n := 0 + for _, uri := range uris { + if uri == target { + n++ + } + } + return n + } + assert.Equalf(t, 1, countIn(engine.uris(), stuck.PostURI), + "a subject deferred one second ago was offered again; the backoff exists so one wedged community does not become a hot loop") + + // The healthy subject keeps flowing. This is the half that makes the backoff + // per-SUBJECT rather than global: one community's dead credentials must not + // stall everyone else's posts, which is precisely what a global pause does. + assert.Equalf(t, 2, countIn(engine.uris(), healthy.PostURI), + "the backoff must be per-subject: a healthy subject stopped being processed because a DIFFERENT one deferred") + assert.Equal(t, 1, second.Processed, "only the healthy subject was due") + + // Past the backoff, the deferred subject is due again — a deferral is "look + // again later", and a driver that never looked again would strand it forever. + clock.advance(2 * time.Minute) + third, err := driver.RunPass(ctx) + require.NoError(t, err) + assert.Equalf(t, 2, countIn(engine.uris(), stuck.PostURI), + "a deferred subject must be retried once its backoff elapses; holding it forever is how a recovered community's posts stay pending") + assert.Equal(t, 2, third.Processed) +} + +func TestQueueDriver_AFailedSubjectDoesNotAbortThePass(t *testing.T) { + t.Parallel() + + clock := newQueueClock() + poison := subject("did:plc:qa", "poison") + after := subject("did:plc:qb", "after") + subjects := &fakeSubjects{batches: [][]PendingSubject{{poison, after}}} + engine := newFakeEngine() + engine.errs[poison.PostURI] = errors.New("the decision blew up") + + report, err := NewQueueDriver(subjects, engine, clock.now()).RunPass(context.Background()) + + require.NoError(t, err, + "a pass is only an error when the BACKLOG could not be read; a subject that failed is a counted outcome") + assert.Containsf(t, engine.uris(), after.PostURI, + "the pass stopped at the first failing subject. One poisoned admission would then freeze every other community's posts, "+ + "and the symptom — 'posts stopped appearing' — points nowhere near the row that caused it") + assert.Equal(t, 1, report.Failed) + assert.Equal(t, 1, report.Settled) + + // Failures and deferrals are counted apart because they mean opposite things + // to whoever is deciding whether to page: a pass that defers everything is + // usually credentials and will clear, a pass that FAILS everything is a bug. + assert.Zero(t, report.Deferred, "an error is not a deferral") +} + +func TestQueueDriver_ReportsAnUnreadableBacklog(t *testing.T) { + t.Parallel() + + clock := newQueueClock() + subjects := &fakeSubjects{err: errors.New("admissions table unreachable")} + + report, err := NewQueueDriver(subjects, newFakeEngine(), clock.now()).RunPass(context.Background()) + + require.Error(t, err, "a backlog that could not be read is the one failure a pass has nothing to do about") + assert.Zero(t, report.Processed) +} + +func TestQueueDriver_SnapshotReportsTheBacklogAndTheLastPass(t *testing.T) { + t.Parallel() + + clock := newQueueClock() + oldest := subject("did:plc:qa", "oldest") + oldest.CreatedAt = clock.at.Add(-90 * time.Minute) + newer := subject("did:plc:qa", "newer") + newer.CreatedAt = clock.at.Add(-5 * time.Minute) + + subjects := &fakeSubjects{batches: [][]PendingSubject{{oldest, newer}}} + engine := newFakeEngine() + engine.outcomes[newer.PostURI] = EngineDeferred + + driver := NewQueueDriver(subjects, engine, clock.now()) + + // Before the first pass, LastPassAt is nil — the one field that distinguishes + // "the driver has never run" from "the driver ran and found nothing". A + // driver that has died produces no error and no log; it simply stops, and + // every symptom shows up somewhere else as posts that never become visible. + require.Nil(t, driver.Snapshot().LastPassAt, + "before any pass, lastPassAt must be nil: a zero time would read as a pass that happened at the epoch") + + _, err := driver.RunPass(context.Background()) + require.NoError(t, err) + + snapshot := driver.Snapshot() + assert.Equal(t, 2, snapshot.PendingBacklog) + require.NotNil(t, snapshot.OldestPendingAt) + assert.Equal(t, oldest.CreatedAt, *snapshot.OldestPendingAt, + "the oldest undecided subject's age is the queue's only early warning: a backlog that is merely big is a busy instance, one whose oldest entry keeps ageing is an engine that has stopped settling anything") + require.NotNil(t, snapshot.LastPassAt) + assert.Equal(t, clock.at, *snapshot.LastPassAt) + assert.Equal(t, 1, snapshot.LastPassDeferred) + assert.Zero(t, snapshot.LastPassFailed) +} diff --git a/internal/db/postgres/admission_queue_repo.go b/internal/db/postgres/admission_queue_repo.go new file mode 100644 index 0000000..584bbaf --- /dev/null +++ b/internal/db/postgres/admission_queue_repo.go @@ -0,0 +1,29 @@ +package postgres + +import ( + "context" + + "Coves/internal/core/posts" +) + +// RED STUB (task 5, cycle 2). Signature only; the query is GREEN's. + +// ListPendingSubjects returns the acceptance engine's backlog: subjects this +// AppView can actually settle, oldest first. +// +// The two exclusions are documented on the interface +// (posts.AdmissionRepository); what belongs here is how they are SPELLED, since +// both are joins and a join written the obvious way gets one of them wrong: +// +// - HOSTED is a credential test — communities.pds_refresh_token_encrypted IS +// NOT NULL — and never communities.hosted_by_did, which is a claim copied +// out of a firehose-indexed community's own profile record and is therefore +// attacker-controlled (see posts.NewCommunityRepoFactory). +// - THE POST MUST STAND, which is an INNER JOIN on posts plus a +// deleted_at IS NULL predicate. Both halves are needed and they exclude +// different rows: an admission can legitimately exist with NO post row at +// all (an acceptance that arrived before its subject), and a LEFT JOIN would +// let those through as decidable when there is nothing to decide about. +func (r *postgresAdmissionRepo) ListPendingSubjects(ctx context.Context, limit int) ([]posts.PendingSubject, error) { + return nil, nil +} diff --git a/internal/db/postgres/admission_queue_repo_test.go b/internal/db/postgres/admission_queue_repo_test.go new file mode 100644 index 0000000..2043b19 --- /dev/null +++ b/internal/db/postgres/admission_queue_repo_test.go @@ -0,0 +1,279 @@ +//go:build integration + +package postgres + +import ( + "context" + "database/sql" + "strings" + "testing" + "time" + + "Coves/internal/core/posts" + "Coves/tests/fixtures" + "Coves/tests/testkit" + + _ "github.com/lib/pq" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// The acceptance engine's backlog query (docs/PRD_AUTHOR_OWNED_POSTS.md §5.6). +// +// ListPendingSubjects is a work queue, and the assertions that matter here are +// about what it LEAVES OUT rather than what it returns. Every excluded class is +// a subject the engine can never settle, and including one does not produce a +// visible failure — it produces a pass that does a little useless work, forever, +// on every instance, growing with the network. The two that would hurt most: +// +// - A community whose credentials this AppView does not hold. The engine's +// entire job is to write an acceptance into that community's repo. Handing +// it a community it cannot sign for means every pass ends in the same +// credential failure, and the genuine deferrals — the ones an operator needs +// to see — are buried under them. +// - A subject whose post has been tombstoned. Accepting a deleted post writes +// an acceptance for content that no longer exists, and the host-side +// tombstone sweep then deletes that acceptance; the next pass re-lists the +// same still-pending subject and does it again. Two components, each correct +// in isolation, looping against a PDS. + +// hostedCommunity seeds a community this AppView genuinely hosts. +// +// The credential column is written directly rather than through +// UpdateCredentials, and the value is not a real token, because the query's +// question is PRESENCE and nothing decrypts it here. Going through +// UpdateCredentials would drag in the encryption_keys row and pgp_sym_encrypt to +// prove something about a predicate that only reads IS NOT NULL. +func hostedCommunity(t *testing.T, db *sql.DB, name string) string { + t.Helper() + + did := uncredentialedCommunity(t, db, name) + _, err := db.ExecContext(context.Background(), + `UPDATE communities SET pds_refresh_token_encrypted = $2 WHERE did = $1`, + did, []byte("a stored refresh token")) + require.NoErrorf(t, err, "granting %s credentials", did) + return did +} + +// uncredentialedCommunity seeds a community this AppView does NOT host. +// +// It is fixtures.Community unchanged, and that is the trap worth naming: +// fixtures.Community sets hosted_by_did to this instance's DID while storing no +// credentials at all. Every community indexed from the firehose has exactly that +// shape, because hosted_by_did is copied out of the community's own profile +// record — a claim by whoever controls that repo. So an implementation that +// asked "does hosted_by_did name us" would return this community, and an +// attacker could put their community in any AppView's work queue by writing one +// field. +func uncredentialedCommunity(t *testing.T, db *sql.DB, name string) string { + t.Helper() + + label := testkit.UniqueIDWithPrefix(t, name) + did, err := fixtures.Community(context.Background(), db, label, "owner"+label) + require.NoErrorf(t, err, "seeding community %s", label) + return did +} + +// pendingSubjectIn seeds a live post and an admission row in the given status, +// with created_at set to age ago so ordering is assertable without sleeping. +func pendingSubjectIn(t *testing.T, db *sql.DB, communityDID, status string, age time.Duration) string { + t.Helper() + + uri := livePost(t, db, communityDID) + seedAdmission(t, db, communityDID, uri, status, age) + return uri +} + +// livePost inserts an author-repo post row that is indexed and not deleted. +func livePost(t *testing.T, db *sql.DB, communityDID string) string { + t.Helper() + + authorDID := fixtures.DID(testkit.UniqueID(t)) + rkey := testkit.TID() + uri := "at://" + authorDID + "/social.coves.community.postv2/" + rkey + _, err := db.ExecContext(context.Background(), ` + INSERT INTO posts (uri, cid, rkey, author_did, community_did, title, created_at) + VALUES ($1, $2, $3, $4, $5, $6, NOW()) + `, uri, "bafyreiqueue"+rkey, rkey, authorDID, communityDID, "a post awaiting a verdict") + require.NoError(t, err) + return uri +} + +// seedAdmission writes one admission row directly, so a test can produce a +// status the repository's own mutations would refuse to reach in one step. +func seedAdmission(t *testing.T, db *sql.DB, communityDID, postURI, status string, age time.Duration) { + t.Helper() + + var decisionCode any + if status == "rejected" || status == "removed" { + decisionCode = string(posts.DecisionRuleViolation) + } + _, err := db.ExecContext(context.Background(), ` + INSERT INTO community_post_admissions + (community_did, post_uri, status, decision_code, evaluated_cid, created_at, updated_at) + VALUES ($1, $2, $3, $4, $5, NOW() - $6::interval, NOW()) + `, communityDID, postURI, status, decisionCode, "bafyreievaluated", age.String()) + require.NoErrorf(t, err, "seeding a %s admission for %s", status, postURI) +} + +// subjectURIs reduces a result to the post URIs it named, for set comparison. +func subjectURIs(subjects []posts.PendingSubject) []string { + uris := make([]string, 0, len(subjects)) + for _, subject := range subjects { + uris = append(uris, subject.PostURI) + } + return uris +} + +func TestListPendingSubjects_ReturnsOnlyTheUndecidedStatuses(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + community := hostedCommunity(t, db, "queuestat") + + pending := pendingSubjectIn(t, db, community, "pending", 4*time.Minute) + reacceptance := pendingSubjectIn(t, db, community, "pending_reacceptance", 3*time.Minute) + accepted := pendingSubjectIn(t, db, community, "accepted", 2*time.Minute) + rejected := pendingSubjectIn(t, db, community, "rejected", time.Minute) + removed := pendingSubjectIn(t, db, community, "removed", 30*time.Second) + + subjects, err := NewAdmissionRepository(db).ListPendingSubjects(context.Background(), 50) + require.NoError(t, err) + + // Both undecided states, and both for the same reason: §5.6 names + // pending and pending_reacceptance as the two the engine owes an answer for. + // Leaving pending_reacceptance out would strand every edited post in a state + // where the old acceptance no longer applies and no new one is ever written — + // the post silently disappears from the community for good. + assert.ElementsMatch(t, []string{pending, reacceptance}, subjectURIs(subjects), + "the backlog is exactly the undecided states") + + for _, settled := range []string{accepted, rejected, removed} { + assert.NotContainsf(t, subjectURIs(subjects), settled, + "a settled subject (%s) must not be re-offered: re-deciding a removed row is how a removed post is laundered back into a feed", settled) + } +} + +func TestListPendingSubjects_IsOldestFirstAndBounded(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + community := hostedCommunity(t, db, "queueorder") + + oldest := pendingSubjectIn(t, db, community, "pending", 10*time.Minute) + middle := pendingSubjectIn(t, db, community, "pending", 5*time.Minute) + newest := pendingSubjectIn(t, db, community, "pending", time.Minute) + + repo := NewAdmissionRepository(db) + + all, err := repo.ListPendingSubjects(context.Background(), 50) + require.NoError(t, err) + assert.Equal(t, []string{oldest, middle, newest}, subjectURIs(all), + "oldest first: a queue served newest-first starves its own backlog, and the post that has waited longest is the one whose author is already wondering where it went") + + // The bound is not a nicety. This query runs on a timer against a table that + // grows with every submission the instance has ever seen, and a pass that + // listed the whole backlog would hold one transaction open across it and then + // try to settle all of it inside a single bounded cycle. + page, err := repo.ListPendingSubjects(context.Background(), 2) + require.NoError(t, err) + assert.Equal(t, []string{oldest, middle}, subjectURIs(page), + "the limit must bound the result, and must take the oldest — a limit applied after an unordered scan would let a busy instance never reach its oldest work") + + // created_at travels with the subject because the health surface reports the + // age of the oldest undecided row, which is the queue's only early warning. + require.NotEmpty(t, all) + assert.WithinDuration(t, time.Now().Add(-10*time.Minute), all[0].CreatedAt, time.Minute, + "the subject must carry the created_at the ordering was made on") +} + +func TestListPendingSubjects_ExcludesDeletedAndUnindexedPosts(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + community := hostedCommunity(t, db, "queueghost") + + live := pendingSubjectIn(t, db, community, "pending", 3*time.Minute) + + // A post the author deleted. The admission row survives the tombstone (the + // community's decision about it is history), but there is nothing left to + // decide. + tombstoned := pendingSubjectIn(t, db, community, "pending", 2*time.Minute) + _, err := db.ExecContext(ctx, `UPDATE posts SET deleted_at = NOW() WHERE uri = $1`, tombstoned) + require.NoError(t, err) + + // An admission with NO post row at all — an acceptance that arrived before + // its subject, which §5.4 says is a state the system genuinely reaches and + // which is why community_post_admissions carries no foreign key to posts. + // This is the row a LEFT JOIN would let through. + unindexed := "at://" + fixtures.DID(testkit.UniqueID(t)) + "/social.coves.community.postv2/" + testkit.TID() + seedAdmission(t, db, community, unindexed, "pending", time.Minute) + + subjects, err := NewAdmissionRepository(db).ListPendingSubjects(ctx, 50) + require.NoError(t, err) + + assert.Equal(t, []string{live}, subjectURIs(subjects)) + assert.NotContainsf(t, subjectURIs(subjects), tombstoned, + "a tombstoned post (%s) must not be offered for a decision: accepting it writes an acceptance for content that no longer exists, "+ + "and the host-side tombstone sweep then deletes that acceptance — the two loop against each other forever", tombstoned) + assert.NotContainsf(t, subjectURIs(subjects), unindexed, + "an admission whose post was never indexed (%s) has no content to judge; a LEFT JOIN would offer it every pass and the engine would defer it every pass", unindexed) +} + +func TestListPendingSubjects_ExcludesCommunitiesThisAppViewCannotSignFor(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + hosted := hostedCommunity(t, db, "queuehosted") + claimant := uncredentialedCommunity(t, db, "queueclaim") + + ours := pendingSubjectIn(t, db, hosted, "pending", 2*time.Minute) + theirs := pendingSubjectIn(t, db, claimant, "pending", time.Minute) + + // The claimant's row asserts what an attacker-controlled profile record can + // assert. hosted_by_did is copied out of a firehose-indexed community's own + // profile, so any repo on the network can name this instance as its host; + // only the credential column reflects something this AppView did itself. + var claimedHost string + require.NoError(t, db.QueryRow(`SELECT hosted_by_did FROM communities WHERE did = $1`, claimant).Scan(&claimedHost)) + require.Equal(t, fixtures.InstanceDID(), claimedHost, + "fixture: the uncredentialed community must CLAIM this instance as its host, or this test proves nothing about which signal is trusted") + + subjects, err := NewAdmissionRepository(db).ListPendingSubjects(context.Background(), 50) + require.NoError(t, err) + + assert.Equal(t, []string{ours}, subjectURIs(subjects)) + assert.NotContainsf(t, subjectURIs(subjects), theirs, + "a community with NO stored credentials (%s) was offered to the engine even though it only CLAIMS this instance as its host. "+ + "Hosting is credential presence — pds_refresh_token_encrypted, which exists only because this AppView provisioned the account — "+ + "and never hosted_by_did, which anyone can write into their own profile record", claimant) +} + +func TestAdmissionsTable_PendingQueueIndex(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + requireTableExists(t, db, admissionsTable) + + // The existing queue index leads with community_did — right for a + // moderator's view of ONE community, useless for this query, which asks + // "what is undecided anywhere" and would scan the whole table through it. + // A partial index over the two undecided statuses is what keeps a periodic + // pass from getting more expensive with every post the instance ever + // indexed, and nothing about a missing one FAILS: it just quietly degrades. + var matched string + for name, definition := range indexDefinitions(t, db, admissionsTable) { + predicate := normalizePredicate(indexPredicate(definition)) + if predicate == "" || !strings.Contains(predicate, "pending") || !strings.Contains(predicate, "pending_reacceptance") { + continue + } + if strings.HasPrefix(indexColumns(definition), "created_at") { + matched = name + } + } + assert.NotEmptyf(t, matched, + "no partial index on (created_at) restricted to the undecided statuses; ListPendingSubjects runs on a timer over a table that grows "+ + "with every submission, and the existing (community_did, status, created_at) index cannot serve a cross-community scan. Indexes found: %v", + indexDefinitions(t, db, admissionsTable)) +} diff --git a/tests/e2e/author_post_contract_test.go b/tests/e2e/author_post_contract_test.go new file mode 100644 index 0000000..645a935 --- /dev/null +++ b/tests/e2e/author_post_contract_test.go @@ -0,0 +1,517 @@ +//go:build e2e + +package e2e + +import ( + "context" + "crypto/sha256" + "encoding/base32" + "net/url" + "strings" + "testing" + "time" + + "Coves/tests/testkit" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// The author-owned post pipeline: the three collections that replace the +// community-repo post record (docs/PRD_AUTHOR_OWNED_POSTS.md §3, §5). +// +// # THE SHAPE IS INVERTED FROM post_contract_test.go, AND THAT IS THE POINT +// +// The deprecated social.coves.community.post lives in the COMMUNITY's repo and +// names its author in a field. Everything here is the other way round: +// +// - social.coves.community.postv2 lives in the AUTHOR's repo and has NO author +// field. The repo the commit arrived in IS the author, so the old contract's +// central proof — a repo forging a post for another community — has no +// analogue; what replaces it is that the community field is a CLAIM, and a +// post making it is not visible in that community until the community says +// otherwise. +// - social.coves.community.acceptance and .removal live in the COMMUNITY's +// repo and are the community saying otherwise. They are written here by the +// test, holding the community's own session, because that is exactly what +// the production engine does with the community's credentials — and no +// community in this tier HAS credentials in the AppView (post_admission_ +// contract_test.go's standing ceiling), so the engine cannot be driven from +// out here. Writing the records directly is not a shortcut around the +// engine; it is the only way to feed the consumers the events a credentialed +// production engine would emit. +// +// # THE OBSERVATION SURFACE IS getStatus, NOT post.get +// +// PRD rev 2.7 pulled social.coves.community.post.getStatus forward into this +// task precisely so these contracts have something to watch. post.get is +// status-agnostic until task 7 rebuilds the read paths behind the centralized +// visibility predicate (§6.2), so asserting "a pending post is invisible" here +// would be asserting a behaviour that does not exist yet and is not this task's +// to build. What IS true today is asserted; what task 7 owes is named where it +// would otherwise look like a gap. +// +// # THE rkey IS COMPUTED HERE RATHER THAN IMPORTED +// +// An acceptance's record key is the unpadded lowercase base32 encoding of the +// SHA-256 digest of the subject AT-URI (§3.2), and posts.SubjectRkey is the +// production implementation. This file re-derives it from stdlib instead of +// importing that package, for two reasons that point the same way: no test in +// this tier imports internal/core/... at all, and — more usefully — a test that +// called the production helper could not detect a bug in it, because both sides +// of the comparison would move together. Re-deriving makes these contracts an +// independent check of the derivation, so a change to either fails here. + +const ( + postV2Collection = "social.coves.community.postv2" + acceptanceCollection = "social.coves.community.acceptance" + removalCollection = "social.coves.community.removal" + postGetStatusEndpoint = "social.coves.community.post.getStatus" +) + +// subjectRkeyEncoding mirrors posts.SubjectRkey's encoding: RFC 4648 base32 with +// the padding dropped, because '=' is outside the atProto record-key charset. +var subjectRkeyEncoding = base32.StdEncoding.WithPadding(base32.NoPadding) + +// subjectRkey is the record key a community's acceptance and removal for one +// post share. See this file's header for why it is not posts.SubjectRkey. +func subjectRkey(postURI string) string { + digest := sha256.Sum256([]byte(postURI)) + return strings.ToLower(subjectRkeyEncoding.EncodeToString(digest[:])) +} + +// postStatusView is the slice of getStatus's response the contracts observe. +type postStatusView struct { + Status string `json:"status"` + DecisionCode string `json:"decisionCode"` + DecisionAt string `json:"decisionAt"` + AcceptanceURI string `json:"acceptanceUri"` +} + +// PostStatus asks the community host what it decided about a post. +// +// A subject the community has never seen is a not-found StatusError, which +// testkit.PendingIfNotFound turns into "no decision yet" inside a probe — the +// same shape every other read in this tier uses. +func (p *pipeline) PostStatus(ctx context.Context, postURI, communityDID string) (postStatusView, error) { + var view postStatusView + err := p.AppView.Query(ctx, postGetStatusEndpoint, url.Values{ + "post": {postURI}, + "community": {communityDID}, + }, &view) + return view, err +} + +// authorPostURI renders the AT-URI a postv2 record has once committed: the +// AUTHOR's DID is the authority, which is the whole shape of this domain. +func authorPostURI(authorDID, rkey string) string { + return "at://" + authorDID + "/" + postV2Collection + "/" + rkey +} + +// postV2Record builds a social.coves.community.postv2 record. There is no +// author field, by construction — the lexicon has none (§3.1), and a consumer +// that still read one would be reading a field only a forger would send. +func postV2Record(communityDID, title, content string) map[string]any { + return map[string]any{ + "$type": postV2Collection, + "community": communityDID, + "title": title, + "content": content, + "createdAt": time.Now().UTC().Format(time.RFC3339), + } +} + +// acceptanceRecord builds the community's attestation, pinning the exact +// version it accepted. +func acceptanceRecord(postURI, postCID string) map[string]any { + return map[string]any{ + "$type": acceptanceCollection, + "subject": map[string]any{"uri": postURI, "cid": postCID}, + "createdAt": time.Now().UTC().Format(time.RFC3339), + } +} + +// removalRecord builds the community's record that a post has been removed. +func removalRecord(postURI, postCID, code string) map[string]any { + return map[string]any{ + "$type": removalCollection, + "subject": map[string]any{"uri": postURI, "cid": postCID}, + "code": code, + "createdAt": time.Now().UTC().Format(time.RFC3339), + } +} + +// awaitStatus waits for getStatus to report want, and returns the view it saw. +func awaitStatus(t *testing.T, p *pipeline, postURI, communityDID, want, description string) postStatusView { + t.Helper() + + var observed postStatusView + p.Await(t, description, func() (bool, error) { + view, err := p.PostStatus(context.Background(), postURI, communityDID) + if pending, wrapped := testkit.PendingIfNotFound(err); wrapped != nil || pending { + return false, wrapped + } + observed = view + return view.Status == want, nil + }) + return observed +} + +// statusAbsent reports whether the community has no decision at all about a +// post — a 404 from getStatus, which is different from every status value. +func statusAbsent(p *pipeline, postURI, communityDID string) func() (bool, error) { + return func() (bool, error) { + _, err := p.PostStatus(context.Background(), postURI, communityDID) + if err == nil { + return false, nil + } + if testkit.IsNotFound(err) { + return true, nil + } + return false, err + } +} + +// TestAuthorPostIngestion is the pipeline proof for the author-repo post record, +// and the outer frame the other two contracts hang off. +// +// coves:ingestion-contract social.coves.community.postv2 +// +// create → the post is indexed, attributed to the REPO it came from, and the +// community it names holds it as `pending` +// retarget→ an update changing `community` is discarded WHOLE: the new +// community never sees it and the old one's decision is untouched +// delete → the post stops being served, and STAYS gone (Holds, §3.4a) +// +// # WHY `pending` IS THE INTERESTING ASSERTION +// +// Under the old model an indexed post was a visible post. Here a post claiming a +// community is never shown in it until an acceptance exists (§2), so the state +// this contract has to prove is the one that did not exist before: indexed, +// attributed, and NOT yet admitted. A consumer that opened the row as anything +// other than pending would publish speech the community never agreed to carry — +// and getStatus is the only surface that can tell the difference, because +// post.get is status-agnostic until task 7. +func TestAuthorPostIngestion(t *testing.T) { + p := newPipeline(t) + + author := p.IndexedAccount(t, "av") + community := indexedCommunity(t, p, "av", author.DID) + elsewhere := indexedCommunity(t, p, "ae", author.DID) + + title := "author-owned " + testkit.UniqueID(t) + rkey := testkit.TID() + uri := authorPostURI(author.DID, rkey) + + // Written into the AUTHOR's own repo, with the author's own session. That is + // the entire point of the flip: the repo signature is the authorship anchor, + // so nobody but this account can produce this record. + record := author.PutRecord(t, postV2Collection, rkey, + postV2Record(community.DID, title, "words the author is accountable for")) + + view := awaitStatus(t, p, uri, community.DID, "pending", + "the author's post to reach the community's admission state via the consumers") + + assert.Empty(t, view.AcceptanceURI, "a pending post has no acceptance record to point at") + assert.Empty(t, view.DecisionCode, "a pending post has been refused by nobody") + + // The post itself is served, attributed to the repo it arrived in. post.get + // is status-agnostic today — task 7 owes the centralized visibility + // predicate that makes a pending post invisible to non-authors (§6.2) — so + // what is asserted here is what IS true: the record was indexed, and its + // author is the DID that signed the commit rather than a field somebody + // could have written. + served, err := p.Post(context.Background(), uri) + require.NoError(t, err) + require.Falsef(t, served.NotFound, "the indexed post must be served by post.get: %+v", served) + assert.Equalf(t, author.DID, served.Author.DID, + "authorship must come from the repo the commit arrived in; the postv2 record carries no author field at all, so a different DID here means one was invented") + assert.Equal(t, community.DID, served.Community.DID) + assert.Equal(t, record.CID, served.CID, "the indexed CID must be the commit's") + assert.Equal(t, title, served.Record["title"]) + assert.Nilf(t, served.Record["author"], + "the record must not carry an author field: it is the field whose removal makes authorship unforgeable (§3.1)") + + // ---- retarget: the whole event is invalid ------------------------------ + // §3.1 is explicit — a consumer must DISCARD an update that changes + // `community`, not merely retain the old value. Retargeting a post means + // writing a new record, and applying the content half while ignoring the + // community half would leave the original community's admission holding a + // CID it never evaluated. + retargeted := "retargeted " + testkit.UniqueID(t) + author.PutRecord(t, postV2Collection, rkey, + postV2Record(elsewhere.DID, retargeted, "aimed at a community that never received this post")) + + // Bounded by a later event in the SAME repo, the way TestPostIngestion + // bounds its spoof: a second post by this author, committed after the + // retarget, cannot overtake it — the PDS sequencer orders a repo's own + // commits and Jetstream serializes per repo. When the bystander is visible, + // the retarget has already been through the consumer. + bystander := testkit.TID() + bystanderURI := authorPostURI(author.DID, bystander) + author.PutRecord(t, postV2Collection, bystander, + postV2Record(community.DID, "bystander "+testkit.UniqueID(t), "committed after the retarget")) + awaitStatus(t, p, bystanderURI, community.DID, "pending", + "a later post in the same repo, which bounds the retarget's rejection") + + absentElsewhere := statusAbsent(p, uri, elsewhere.DID) + absent, err := absentElsewhere() + require.NoError(t, err) + require.Truef(t, absent, + "the retargeted update opened an admission in %s. A post's community is immutable across updates (§3.1); "+ + "honouring the change lets an author move a post into any community by editing it, with no acceptance and no moderator involved", + elsewhere.DID) + p.Holds(t, "the retargeted community to stay undecided", absentElsewhere) + + original, err := p.PostStatus(context.Background(), uri, community.DID) + require.NoError(t, err) + assert.Equal(t, "pending", original.Status, + "the original community's decision must be untouched by an invalid update") + + unchanged, err := p.Post(context.Background(), uri) + require.NoError(t, err) + assert.Equalf(t, title, unchanged.Record["title"], + "the CONTENT of a discarded event must be discarded with it: applying the new title while refusing the new community would leave the community holding a CID it never judged") + assert.NotEqual(t, retargeted, unchanged.Record["title"]) + + // ---- delete ------------------------------------------------------------- + author.DeleteExistingRecord(t, postV2Collection, rkey) + + gone := func() (bool, error) { + v, err := p.Post(context.Background(), uri) + if err != nil { + return false, err + } + return v.NotFound, nil + } + p.Await(t, "the author's deleted post to stop being served", gone) + p.Holds(t, "the deleted post to stay deleted", gone) +} + +// TestCommunityAcceptanceIngestion is the pipeline proof for the community's +// attestation record. +// +// coves:ingestion-contract social.coves.community.acceptance +// +// accept → the pending post becomes `accepted` and names the acceptance +// edit → the standing acceptance no longer covers the content, and the +// post falls back to `pending_reacceptance` +// unaccept → deleting the acceptance returns the post to `pending` +// unknown subject → an acceptance naming a post that does not exist admits +// nothing, and keeps admitting nothing (Holds) +// +// # THE CID IS THE SUBSTANCE OF THE ATTESTATION +// +// An acceptance is a strongRef, and agreeing to at://x/postv2/y is not agreeing +// to whatever that URI holds tomorrow. The edit step is the whole reason the +// pinned CID exists: if the AppView rendered new content under an old +// acceptance, an author could get anything approved and then swap it, which is +// the single most valuable attack against a moderation system built on pointers. +func TestCommunityAcceptanceIngestion(t *testing.T) { + p := newPipeline(t) + + author := p.IndexedAccount(t, "cv") + community := indexedCommunity(t, p, "cv", author.DID) + + rkey := testkit.TID() + uri := authorPostURI(author.DID, rkey) + record := author.PutRecord(t, postV2Collection, rkey, + postV2Record(community.DID, "accept me "+testkit.UniqueID(t), "content the community will attest to")) + + awaitStatus(t, p, uri, community.DID, "pending", "the post to be indexed and awaiting a decision") + + // The community attests, from its own repo, at the deterministic rkey. One + // post has exactly one acceptance rkey per community, forever — which is + // what makes the three independent production writers (§3.2) converge on + // putRecord of the same record instead of minting duplicates. + acceptRkey := subjectRkey(uri) + community.PutRecord(t, acceptanceCollection, acceptRkey, acceptanceRecord(uri, record.CID)) + + accepted := awaitStatus(t, p, uri, community.DID, "accepted", + "the community's acceptance to admit the post") + assert.Equalf(t, "at://"+community.DID+"/"+acceptanceCollection+"/"+acceptRkey, accepted.AcceptanceURI, + "the reported acceptance URI must resolve to the record that actually stands, or a client following it to verify the attestation gets nothing") + assert.Empty(t, accepted.DecisionCode, "an acceptance is not a refusal") + + // ---- the edit that un-accepts ------------------------------------------ + edited := author.PutRecord(t, postV2Collection, rkey, + postV2Record(community.DID, "edited after acceptance "+testkit.UniqueID(t), + "content the community has never seen")) + require.NotEqualf(t, record.CID, edited.CID, + "the edit produced an identical CID, so this step would prove nothing about re-acceptance") + + reacceptance := awaitStatus(t, p, uri, community.DID, "pending_reacceptance", + "the edit to fall out from under the standing acceptance") + assert.NotEmptyf(t, reacceptance.AcceptanceURI, + "the acceptance record still STANDS in the community's repo — only its pinned CID no longer matches — so getStatus must keep naming it; "+ + "clearing it would tell the author their acceptance was withdrawn, which is a different and untrue thing") + + // ---- withdrawing the acceptance ---------------------------------------- + // Deleting the acceptance with NO removal in the same commit is the author- + // deletion sweep's shape (§5.3), and it means "no longer accepted", not + // "removed": the post falls back to undecided rather than to a moderation + // state nobody entered. + community.DeleteExistingRecord(t, acceptanceCollection, acceptRkey) + + withdrawn := awaitStatus(t, p, uri, community.DID, "pending", + "the deleted acceptance to return the post to undecided") + assert.Emptyf(t, withdrawn.DecisionCode, + "an acceptance deleted on its own is not a removal; minting a decision code here would put a moderation act in the record that no moderator performed") + + // ---- an acceptance for a post that does not exist ---------------------- + // The §5.4 direct fetch resolves the subject author's PDS and reads the + // record when the AppView has not indexed it. Here there is no record to + // read, and the CID cannot be verified against anything — so nothing may be + // indexed and nothing may be admitted. Bounded by a later commit in the SAME + // repo, then held. + phantomURI := authorPostURI(author.DID, testkit.TID()) + community.PutRecord(t, acceptanceCollection, subjectRkey(phantomURI), + acceptanceRecord(phantomURI, "bafyreiaphantomcidthatnobodyminted")) + + laterRkey := testkit.TID() + laterURI := authorPostURI(author.DID, laterRkey) + author.PutRecord(t, postV2Collection, laterRkey, + postV2Record(community.DID, "later "+testkit.UniqueID(t), "committed after the phantom acceptance")) + community.PutRecord(t, acceptanceCollection, subjectRkey(laterURI), acceptanceRecord(laterURI, "bafyreiplaceholder")) + p.Await(t, "a later acceptance in the same repo, which bounds the phantom's rejection", func() (bool, error) { + _, err := p.PostStatus(context.Background(), laterURI, community.DID) + return testkit.PendingIfNotFound(err) + }) + + phantomAbsent := func() (bool, error) { + view, err := p.PostStatus(context.Background(), phantomURI, community.DID) + if err != nil { + if testkit.IsNotFound(err) { + return true, nil + } + return false, err + } + return view.Status != "accepted", nil + } + nothingAdmitted, err := phantomAbsent() + require.NoError(t, err) + require.True(t, nothingAdmitted, + "an acceptance naming a post that exists nowhere was admitted. The pinned CID is unverifiable against a record nobody wrote, "+ + "so admitting it means the AppView will render whatever eventually appears at that URI under an attestation made before it existed") + p.Holds(t, "the phantom subject to stay unadmitted", phantomAbsent) + + served, err := p.Post(context.Background(), phantomURI) + require.NoError(t, err) + assert.Truef(t, served.NotFound, + "a post that only an acceptance ever mentioned must not be indexed: the direct fetch had nothing to verify the pinned CID against") +} + +// TestCommunityRemovalIngestion is the pipeline proof for the moderation record. +// +// coves:ingestion-contract social.coves.community.removal +// +// remove → an accepted post becomes `removed` and carries the code +// pre-emptive → a removal with no prior acceptance is valid and indexes +// terminal → an author edit after removal does NOT reopen the decision +// +// # THE REMOVAL COMMIT IS ATOMIC, AND THE TEST WRITES IT THAT WAY +// +// §3.3 requires the acceptance's deletion and the removal's write to reach the +// firehose in ONE com.atproto.repo.applyWrites commit, so the firehose never +// carries a half-completed moderation action. Sending them as two commits would +// let this contract pass against a consumer that cannot order them — which is +// the exact defect §5.2's composite watermark exists to prevent — so the test +// uses applyWrites directly rather than two testkit calls. +func TestCommunityRemovalIngestion(t *testing.T) { + p := newPipeline(t) + + author := p.IndexedAccount(t, "rv") + community := indexedCommunity(t, p, "rv", author.DID) + + rkey := testkit.TID() + uri := authorPostURI(author.DID, rkey) + record := author.PutRecord(t, postV2Collection, rkey, + postV2Record(community.DID, "remove me "+testkit.UniqueID(t), "content a moderator will take down")) + + awaitStatus(t, p, uri, community.DID, "pending", "the post to be indexed") + + subject := subjectRkey(uri) + community.PutRecord(t, acceptanceCollection, subject, acceptanceRecord(uri, record.CID)) + awaitStatus(t, p, uri, community.DID, "accepted", "the post to be accepted before it is removed") + + // One commit, both operations. rank(delete) < rank(put) inside a commit + // (§5.2), so whichever half the consumer applies second either lands or is + // skipped as not-greater — and the subject converges on `removed` either way. + applyWrites(t, community, []map[string]any{ + {"$type": "com.atproto.repo.applyWrites#delete", "collection": acceptanceCollection, "rkey": subject}, + { + "$type": "com.atproto.repo.applyWrites#create", + "collection": removalCollection, + "rkey": subject, + "value": removalRecord(uri, record.CID, "rule-violation"), + }, + }) + + removed := awaitStatus(t, p, uri, community.DID, "removed", + "the atomic removal commit to take the post down") + assert.Equalf(t, "rule-violation", removed.DecisionCode, + "the removal's code is what a client renders to the author; a removal without one is an unexplained disappearance") + assert.NotEmpty(t, removed.DecisionAt, "a removal must record when it happened") + assert.Emptyf(t, removed.AcceptanceURI, + "the acceptance was deleted in the same commit, so naming one would point a verifying client at a record that no longer exists") + + // ---- removal is terminal across author edits --------------------------- + // §5.5: `removed` is exited only by a moderator restore. An author edit + // updates audit metadata and nothing else — otherwise editing a removed post + // would launder it straight back through auto-acceptance, which is the most + // obvious way to defeat moderation on a system where the author holds the + // record. + author.PutRecord(t, postV2Collection, rkey, + postV2Record(community.DID, "edited while removed "+testkit.UniqueID(t), "an attempt to start over")) + + stillRemoved := func() (bool, error) { + view, err := p.PostStatus(context.Background(), uri, community.DID) + if err != nil { + return false, err + } + return view.Status == "removed", nil + } + p.Holds(t, "the removed post to stay removed across an author edit", stillRemoved) + + // ---- pre-emptive removal ------------------------------------------------ + // A removal with no prior acceptance is valid (§5.4): a community may decide + // in advance about content it has already seen elsewhere, or about an author + // it is banning. A consumer requiring an acceptance first would drop exactly + // the decisions a community most wants to make early. + preRkey := testkit.TID() + preURI := authorPostURI(author.DID, preRkey) + preRecord := author.PutRecord(t, postV2Collection, preRkey, + postV2Record(community.DID, "pre-emptively removed "+testkit.UniqueID(t), "never accepted at all")) + awaitStatus(t, p, preURI, community.DID, "pending", "the pre-emptive subject to be indexed") + + community.PutRecord(t, removalCollection, subjectRkey(preURI), + removalRecord(preURI, preRecord.CID, "spam")) + + preRemoved := awaitStatus(t, p, preURI, community.DID, "removed", + "a removal with no prior acceptance to take the post down") + assert.Equal(t, "spam", preRemoved.DecisionCode) +} + +// applyWrites commits several repo operations together, which is the only way +// to produce the atomic moderation commit §3.3 requires. +// +// testkit has no applyWrites helper — every other contract in this tier writes +// one record at a time — so this speaks the XRPC procedure directly through the +// account's own client. validate is false for the reason task 4 recorded: the +// PDS refuses validate:true for lexicons it has not been served, and these three +// are not published to it. +func applyWrites(t *testing.T, community provisionedCommunity, writes []map[string]any) { + t.Helper() + + ctx, cancel := context.WithTimeout(context.Background(), contractBudget) + defer cancel() + + err := community.XRPC().Procedure(ctx, "com.atproto.repo.applyWrites", map[string]any{ + "repo": community.DID, + "validate": false, + "writes": writes, + }, nil) + require.NoErrorf(t, err, + "com.atproto.repo.applyWrites was refused. The removal commit must carry the acceptance's delete and the removal's create TOGETHER (§3.3); "+ + "splitting them into two commits would let a consumer that cannot order them pass this contract") +} -- 2.51.2 From 78dac39a1cc57c23e255fe3c0fd0cea9961bfdc2 Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 01:19:28 -0700 Subject: [PATCH 04/17] =?UTF-8?q?feat(ingestion):=20GREEN=20cycle=202=20?= =?UTF-8?q?=E2=80=94=20queue=20driver,=20decider,=20factory,=20sweep?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit All 42 cycle-2 reds pass at T0/T1. The three T2 contracts pass only with a one-line fix to the RED helper they share — see BLOCKER below; the file is committed exactly as RED wrote it. (a) ListPendingSubjects + its index Two INNER JOINs, both exclusions in SQL: posts (deleted_at IS NULL) and communities (pds_refresh_token_encrypted IS NOT NULL — credential presence, never hosted_by_did, which any repo can claim about itself). Migration 036 gains the partial index on (created_at) over the two undecided statuses and is renamed to match its widened scope; a 037 would have broken admission_repo_schema_test's asserted 36→35→34 chain. (b) CommunityRepoFactory + credential refresher Credential presence is the hosting test; absence is ErrCommunityNotHosted (permanent), while an unindexed community stays an ordinary error because it may simply not have arrived. Token renewed before the client is built, since a client is bound to the token it was constructed with. (c) AdmissionEngineDecider Post read first; tombstoned or absent returns UNDECIDED wrapping the new ErrSubjectGone rather than a code — the engine turns a code on a pending_reacceptance row into a REMOVAL record, and an author deleting their own post is not the community removing it. Actor classification checks the trusted set first (no lookup), and every uncertain path falls to ActorUser. TrustedAggregatorDIDs is now one helper shared with CreatePost so the two paths cannot disagree about who is privileged. (d) QueueDriver One goroutine, grouped by community (the partition a future worker pool must shard on), per-subject exponential backoff on deferral keyed by (community, post) — NOT by PendingSubject, whose time.Time field makes map equality fragile. Errors checked before outcomes, since a failing engine returns EngineDeferred alongside them. Snapshot is mutex-guarded: the health handler reads it while the job writes. (e) DeleteAcceptance + the tombstone sweep State-shaped single-op applyWrites, swapCommit-guarded, absence reported as a skip. The consumer sweeps only when the admission row says an acceptance stands, only when the tombstone actually applied (not on every redelivery), and never lets a failed sweep hold the local tombstone hostage. (f) Wiring: engine + driver behind ACCEPTANCE_QUEUE_INTERVAL (0 disables, and a nil driver is what omits the health block); acceptanceQueue in /health/consumers via an option, since an existing test pins the handler's arity. TWO PRODUCTION BUGS the T2 contracts caught, both in the READ path: - post.get refused every postv2 URI ("invalid collection in URI"), making author-owned posts unfetchable by the endpoint that hydrates every feed. parsePostURIParts now accepts both collections; the DELETE path narrows back to the community-repo one, because it uses the authority as a community DID and would otherwise try to delete from the AUTHOR's repo. - the synthesized `record` map hardcoded $type community.post and an `author` field, so every postv2 read handed back exactly the field whose removal makes authorship unforgeable. The shape now follows the URI. Co-Authored-By: Claude Fable 5 --- .env.ci | 15 ++ .env.dev.example | 27 +++ .env.prod.example | 27 +++ cmd/server/consumers.go | 8 +- cmd/server/health.go | 71 +++++- cmd/server/jobs.go | 53 +++++ cmd/server/main.go | 7 + cmd/server/routes.go | 10 +- cmd/server/wiring.go | 66 +++++- internal/atproto/jetstream/authorpost.go | 106 ++++++++- internal/atproto/jetstream/post_consumer.go | 27 ++- internal/config/config.go | 64 ++++++ internal/core/posts/community_repo_factory.go | 92 +++++++- internal/core/posts/community_writer.go | 109 ++++++++- internal/core/posts/decider.go | 143 +++++++++++- internal/core/posts/queue.go | 216 ++++++++++++++++-- internal/core/posts/service.go | 64 ++++-- ... 036_deleted_accounts_and_queue_index.sql} | 30 +++ internal/db/postgres/admission_queue_repo.go | 64 +++++- internal/db/postgres/post_repo.go | 27 ++- 20 files changed, 1160 insertions(+), 66 deletions(-) rename internal/db/migrations/{036_create_deleted_accounts.sql => 036_deleted_accounts_and_queue_index.sql} (67%) diff --git a/.env.ci b/.env.ci index 807e963..ade1ef7 100644 --- a/.env.ci +++ b/.env.ci @@ -167,3 +167,18 @@ IMAGE_PROXY_MAX_SOURCE_SIZE_MB=10 # Observability # ============================================================================= OTEL_ENABLED=false + +# ============================================================================= +# Acceptance queue +# ============================================================================= +# CI: absent from .env.dev, which takes the 1m default. The pipeline tier's +# per-probe budget is 45 seconds, so a minute-long cadence means a contract that +# ever needed the driver would time out before its first pass — and would do it +# as a flake rather than as a failure that names the driver. Two seconds keeps +# the pass inside every probe window. +# +# No contract currently DRIVES the engine — no community in the pipeline tier +# holds credentials in the AppView, so the backlog query correctly returns +# nothing — which makes this a liveness setting rather than a functional one: +# the driver runs, against the real database, on every CI boot. +ACCEPTANCE_QUEUE_INTERVAL=2s diff --git a/.env.dev.example b/.env.dev.example index b4e9567..62762c3 100644 --- a/.env.dev.example +++ b/.env.dev.example @@ -212,3 +212,30 @@ OTEL_ENABLED=false # clears itself; there is no sweeper, and the damage is bounded by this # window. # POST_SUBMISSIONS_DEDUPE_WINDOW=1h + +# ----------------------------------------------------------------------------- +# Acceptance queue (the engine's pull side) +# ----------------------------------------------------------------------------- +# A post claiming a community is not visible in it until the community writes an +# acceptance record (docs/PRD_AUTHOR_OWNED_POSTS.md section 5.6). The write path +# and the firehose consumer both push work at the engine that writes those, and +# neither can see a subject left undecided because a credential expired or a +# lookup blipped. This periodic pass is the only thing that reaches those, so its +# cadence is the worst case delay before a stranded post becomes visible. +# +# It is a no-op on an instance that hosts no communities: both the backlog query +# and the repo factory key on STORED PDS CREDENTIALS, which exist only for +# communities this AppView provisioned itself. +# +# How often the pass runs (default: 1m). Set to 0 to DISABLE the driver +# entirely, which is supported rather than a misconfiguration — an AppView that +# hosts no communities can accept nothing. Disabling it also removes the +# acceptanceQueue block from /health/consumers, so an absent driver and an idle +# one stay distinguishable. +# ACCEPTANCE_QUEUE_INTERVAL=1m +# +# How many subjects one pass takes (default: 50). Unlike the quotas above this +# one cannot fail open — the backlog query substitutes its own page size for a +# non-positive value and clamps an over-large one — so it is not validated at +# startup. +# ACCEPTANCE_QUEUE_BATCH_SIZE=50 diff --git a/.env.prod.example b/.env.prod.example index f986583..c0e1e10 100644 --- a/.env.prod.example +++ b/.env.prod.example @@ -407,6 +407,33 @@ OTEL_ENABLED=false # window. # POST_SUBMISSIONS_DEDUPE_WINDOW=1h +# ----------------------------------------------------------------------------- +# Acceptance queue (the engine's pull side) +# ----------------------------------------------------------------------------- +# A post claiming a community is not visible in it until the community writes an +# acceptance record (docs/PRD_AUTHOR_OWNED_POSTS.md section 5.6). The write path +# and the firehose consumer both push work at the engine that writes those, and +# neither can see a subject left undecided because a credential expired or a +# lookup blipped. This periodic pass is the only thing that reaches those, so its +# cadence is the worst case delay before a stranded post becomes visible. +# +# It is a no-op on an instance that hosts no communities: both the backlog query +# and the repo factory key on STORED PDS CREDENTIALS, which exist only for +# communities this AppView provisioned itself. +# +# How often the pass runs (default: 1m). Set to 0 to DISABLE the driver +# entirely, which is supported rather than a misconfiguration — an AppView that +# hosts no communities can accept nothing. Disabling it also removes the +# acceptanceQueue block from /health/consumers, so an absent driver and an idle +# one stay distinguishable. +# ACCEPTANCE_QUEUE_INTERVAL=1m +# +# How many subjects one pass takes (default: 50). Unlike the quotas above this +# one cannot fail open — the backlog query substitutes its own page size for a +# non-positive value and clamps an over-large one — so it is not validated at +# startup. +# ACCEPTANCE_QUEUE_BATCH_SIZE=50 + # ============================================================================= # Optional: Versioning # ============================================================================= diff --git a/cmd/server/consumers.go b/cmd/server/consumers.go index 668b742..2e29419 100644 --- a/cmd/server/consumers.go +++ b/cmd/server/consumers.go @@ -184,7 +184,13 @@ func (a *application) registerFeedConsumers() []feedConsumer { jetstream.WithPostIdentityResolver(a.identityResolver), jetstream.WithAdmissions(a.admissionRepo), jetstream.WithDeletedAccounts(postgresRepo.NewDeletedAccountRepository(a.db)), - jetstream.WithPostRecordFetcher(postFetcher)), + jetstream.WithPostRecordFetcher(postFetcher), + // The host-side half of an author's own deletion (§5.3): when the + // author tombstones a post this instance's community accepted, the + // acceptance in that community's repo is withdrawn. It refuses + // itself for every community this AppView does not host, which on + // most instances is all of them. + jetstream.WithAcceptanceCleanup(a.communityWriter)), }) // Aggregators: service declarations and authorization records, following diff --git a/cmd/server/health.go b/cmd/server/health.go index b4261e6..48698e4 100644 --- a/cmd/server/health.go +++ b/cmd/server/health.go @@ -66,11 +66,33 @@ type acceptanceQueueHealth struct { // buildAcceptanceQueueHealth renders one driver snapshot. // -// RED STUB (task 5, cycle 2). A separate pure function rather than another -// parameter on buildConsumerHealthResponse: the two have no shared logic, and -// widening that signature would touch every existing call site to say nothing. +// A separate pure function rather than another parameter on +// buildConsumerHealthResponse: the two have no shared logic, and widening that +// signature would touch every existing call site to say nothing. +// +// Both optional fields are OMITTED rather than zeroed when there is nothing to +// report, because a zero here would be read, and read wrongly: an age of 0 says +// "something arrived just now" when in fact nothing is waiting, and a +// zero-valued timestamp renders as the epoch, which looks like a driver that +// has been dead since 1970 rather than one that started a minute ago. func buildAcceptanceQueueHealth(snapshot posts.QueueSnapshot, now time.Time) acceptanceQueueHealth { - return acceptanceQueueHealth{} + queue := acceptanceQueueHealth{ + PendingBacklog: snapshot.PendingBacklog, + LastPassAt: snapshot.LastPassAt, + LastPassDeferred: snapshot.LastPassDeferred, + LastPassFailed: snapshot.LastPassFailed, + } + + // The AGE, not the timestamp. A backlog that is merely big is a busy + // instance; a backlog whose oldest entry keeps getting older is an engine + // that has stopped settling anything — and only the age says which is + // happening without the reader doing arithmetic against their own clock. + if snapshot.OldestPendingAt != nil { + age := int64(now.Sub(*snapshot.OldestPendingAt).Seconds()) + queue.OldestPendingAgeSeconds = &age + } + + return queue } // buildConsumerHealthResponse is the pure decision core of /health/consumers, @@ -127,7 +149,39 @@ func buildConsumerHealthResponse(statuses []jetstream.ConnectorStatus, backlogs // and the dead letter backlog per consumer. Responds 503 when any consumer // has been disconnected longer than consumerStalledThreshold (indexing is // stalled) so monitoring can alert on it. -func consumerHealthHandler(connectors []*jetstream.Connector, deadLetterQueue jetstream.DeadLetterQueue) http.HandlerFunc { +// acceptanceQueueReporter is the driver, narrowed to the one method health +// needs. Nil means no driver runs on this deployment, and the queue block is +// omitted entirely rather than reported as all-zero — an all-zero queue reads +// as a driver that is running and settling nothing, which is precisely the +// failure an operator is watching for. +type acceptanceQueueReporter interface { + Snapshot() posts.QueueSnapshot +} + +// consumerHealthOption adds a surface to /health/consumers that not every +// deployment has. +// +// An option rather than another parameter because the queue is genuinely +// optional — an AppView hosting no communities runs no driver — and because a +// nil third argument at every existing call site would say nothing while +// reading as an omission. +type consumerHealthOption func(*consumerHealthConfig) + +type consumerHealthConfig struct { + acceptanceQueue acceptanceQueueReporter +} + +// withAcceptanceQueue reports the acceptance driver alongside the consumers. +func withAcceptanceQueue(queue acceptanceQueueReporter) consumerHealthOption { + return func(c *consumerHealthConfig) { c.acceptanceQueue = queue } +} + +func consumerHealthHandler(connectors []*jetstream.Connector, deadLetterQueue jetstream.DeadLetterQueue, opts ...consumerHealthOption) http.HandlerFunc { + var cfg consumerHealthConfig + for _, opt := range opts { + opt(&cfg) + } + return func(w http.ResponseWriter, r *http.Request) { backlogs, err := deadLetterQueue.CountDeadLetters(r.Context()) backlogUnknown := err != nil @@ -142,7 +196,12 @@ func consumerHealthHandler(connectors []*jetstream.Connector, deadLetterQueue je statuses = append(statuses, connector.Status()) } - response, httpCode := buildConsumerHealthResponse(statuses, backlogs, backlogUnknown, time.Now()) + now := time.Now() + response, httpCode := buildConsumerHealthResponse(statuses, backlogs, backlogUnknown, now) + if cfg.acceptanceQueue != nil { + acceptance := buildAcceptanceQueueHealth(cfg.acceptanceQueue.Snapshot(), now) + response.AcceptanceQueue = &acceptance + } w.Header().Set("Content-Type", "application/json") w.WriteHeader(httpCode) diff --git a/cmd/server/jobs.go b/cmd/server/jobs.go index e79ef50..55abf43 100644 --- a/cmd/server/jobs.go +++ b/cmd/server/jobs.go @@ -6,6 +6,8 @@ import ( "log/slog" "sync" "time" + + "Coves/internal/core/posts" ) const ( @@ -130,6 +132,57 @@ type expiringTokenRefresher interface { RefreshExpiringTokens(ctx context.Context, expiryBuffer time.Duration) (int, []error) } +// acceptanceQueuePass is one walk of the acceptance engine's backlog. +// Declared as an interface so this job body is testable without an engine, a +// PDS or a database behind it. +type acceptanceQueuePass interface { + RunPass(ctx context.Context) (posts.PassReport, error) +} + +// startAcceptanceQueueJob walks the undecided admission backlog on an interval. +// +// It is the PULL half of admission (docs/PRD_AUTHOR_OWNED_POSTS.md §5.6). The +// synchronous write path and the firehose consumer both push work at the +// engine, and neither can see a subject that was left undecided because a +// credential expired or a lookup blipped — this pass is the only thing that +// eventually reaches those, so a deployment where it stops running is one where +// posts quietly stay invisible. +// +// Nothing here fails the process: a pass that could not read the backlog is +// logged and the next tick tries again, because the reason a backlog is +// unreadable is almost always the database being briefly unavailable, which the +// rest of the AppView is already reporting. +func startAcceptanceQueueJob(ctx context.Context, wg *sync.WaitGroup, queue acceptanceQueuePass, interval time.Duration) { + if queue == nil || interval <= 0 { + return + } + + runTicker(ctx, wg, "acceptance-queue", interval, func(ctx context.Context) { + report, err := queue.RunPass(ctx) + if err != nil { + if !errors.Is(err, context.Canceled) { + slog.Error("acceptance queue pass failed", "error", err) + } + return + } + + // Deferrals and failures are reported apart because they mean opposite + // things to whoever is deciding whether to page: a pass that defers + // everything is usually credentials and will clear, while a pass that + // FAILS everything is a bug. A quiet pass logs nothing, which is the + // common case on an instance that hosts no communities. + if report.Processed > 0 || report.Failed > 0 { + slog.Info("acceptance queue pass completed", + "listed", report.Listed, + "processed", report.Processed, + "settled", report.Settled, + "deferred", report.Deferred, + "failed", report.Failed, + ) + } + }) +} + // startAggregatorTokenRefreshJob proactively refreshes aggregator OAuth tokens // before they expire. // diff --git a/cmd/server/main.go b/cmd/server/main.go index a724579..6d88924 100644 --- a/cmd/server/main.go +++ b/cmd/server/main.go @@ -96,6 +96,13 @@ func run() error { startOAuthCleanupJob(backgroundCtx, &backgroundWG, sessionStore) startAggregatorTokenRefreshJob(backgroundCtx, &backgroundWG, app.apiKeyService) + // Nil when the driver is disabled, and passed as a typed nil would be a + // non-nil interface — so the guard is here rather than inside the job. + if app.acceptanceQueue != nil { + startAcceptanceQueueJob(backgroundCtx, &backgroundWG, + app.acceptanceQueue, cfg.Submissions.AcceptanceQueueInterval) + } + consumers, err := startConsumers(backgroundCtx, &backgroundWG, app) if err != nil { // Some connectors may already be running. Drain them under the same diff --git a/cmd/server/routes.go b/cmd/server/routes.go index 3bca2a3..5c5d21d 100644 --- a/cmd/server/routes.go +++ b/cmd/server/routes.go @@ -170,5 +170,13 @@ func registerWebRoutes(r chi.Router, app *application) { func registerHealthRoutes(r chi.Router, app *application, consumers *consumerSet) { r.Get("/health", livenessHandler) r.Get("/xrpc/_health", livenessHandler) - r.Get("/health/consumers", consumerHealthHandler(consumers.connectors, app.jetstreamState)) + // The option is added only when a driver actually runs. Passing a typed nil + // pointer instead would be non-nil to an interface comparison, and the + // response would carry an all-zero queue — which reads as a driver that is + // running and settling nothing, the exact failure an operator watches for. + var healthOptions []consumerHealthOption + if app.acceptanceQueue != nil { + healthOptions = append(healthOptions, withAcceptanceQueue(app.acceptanceQueue)) + } + r.Get("/health/consumers", consumerHealthHandler(consumers.connectors, app.jetstreamState, healthOptions...)) } diff --git a/cmd/server/wiring.go b/cmd/server/wiring.go index 51a75a5..5a1ecf8 100644 --- a/cmd/server/wiring.go +++ b/cmd/server/wiring.go @@ -109,7 +109,15 @@ type application struct { // because a status query needs the admissions store and nothing else, and // widening the write-path interface to reach it would make every test // double of posts.Service carry a method it has no opinion about. - postStatusService posts.StatusService + postStatusService posts.StatusService + // communityWriter publishes acceptances and removals into the repos of + // communities this AppView hosts. Shared by the acceptance engine, which + // writes verdicts, and the post consumer, which withdraws an acceptance + // when the author deletes the post it covers. + communityWriter posts.CommunityRecordWriter + // acceptanceQueue walks the undecided backlog. nil when the driver is + // disabled (ACCEPTANCE_QUEUE_INTERVAL=0). + acceptanceQueue *posts.QueueDriver voteService votes.Service commentService comments.Service userBlockService userblocks.Service @@ -363,6 +371,8 @@ func (a *application) buildServices(ctx context.Context) error { // is nothing on the firehose to read. a.postStatusService = posts.NewStatusService(a.admissionRepo) + a.buildAcceptanceEngine() + // Subject existence is deliberately not validated: the vote is written to // the user's own PDS regardless, and the Jetstream consumer only updates // counts for subjects that still exist. Checking here would trade a @@ -395,6 +405,60 @@ func (a *application) buildServices(ctx context.Context) error { return nil } +// buildAcceptanceEngine wires the §5.6 acceptance engine and the driver that +// feeds it. +// +// EVERYTHING HERE IS ABOUT WRITING INTO A COMMUNITY'S OWN REPO, which only that +// community's host can do — so on an instance that hosts nothing, all of it is +// a no-op that costs one query a minute: the repo factory refuses with +// ErrCommunityNotHosted and the backlog query returns nothing. Both are keyed on +// STORED CREDENTIALS rather than on communities.hosted_by_did, which is copied +// out of a community's own profile record and can therefore be claimed by any +// repo on the network. +func (a *application) buildAcceptanceEngine() { + repoFactory := posts.NewCommunityRepoFactory(a.communityService) + a.communityWriter = posts.NewCommunityRecordWriter(repoFactory, time.Now) + + decider := posts.NewAdmissionEngineDecider(posts.DeciderDeps{ + Posts: a.postRepo, + Communities: a.communityService, + Authorizer: a.aggregatorService, + Aggregators: a.aggregatorService, + Policy: posts.AdmissionPolicy{ + Ledger: postgresRepo.NewSubmissionLedger(a.db), + Bans: a.communityService, + Limits: posts.SubmissionLimits{ + MaxPerAuthorPerCommunity: a.cfg.Submissions.MaxPerAuthorPerCommunity, + Window: a.cfg.Submissions.Window, + DedupeWindow: a.cfg.Submissions.DedupeWindow, + }, + Now: time.Now, + }, + // Resolved ONCE, here, rather than read per decision — and through the + // same helper the write path uses, so the two cannot drift into + // disagreeing about who is privileged. + TrustedAggregatorDIDs: posts.TrustedAggregatorDIDs(), + }) + + engine := posts.NewAcceptanceEngine( + a.admissionRepo, decider, a.communityWriter, + posts.NewCommunityCredentialRefresher(a.communityService)) + + // Zero DISABLES the driver, and leaving the field nil is what makes + // /health/consumers omit the queue block entirely. An all-zero queue and an + // absent one mean different things: the first reads as a driver that is + // running and settling nothing, which is the exact failure an operator + // watches for. + if a.cfg.Submissions.AcceptanceQueueInterval <= 0 { + slog.Warn("acceptance queue driver disabled (ACCEPTANCE_QUEUE_INTERVAL=0); " + + "posts left undecided by the fast path and the firehose will not be revisited") + return + } + + a.acceptanceQueue = posts.NewQueueDriver(a.admissionRepo, engine, time.Now, + posts.WithQueueBatchSize(a.cfg.Submissions.AcceptanceQueueBatchSize)) +} + // adminReportAlertOptions builds the operator-alert wiring for admin reports. // // Alerting is opt-in (TELEGRAM_ALERTS_ENABLED): most operators running their diff --git a/internal/atproto/jetstream/authorpost.go b/internal/atproto/jetstream/authorpost.go index 0a11c32..25487e2 100644 --- a/internal/atproto/jetstream/authorpost.go +++ b/internal/atproto/jetstream/authorpost.go @@ -47,7 +47,12 @@ import ( // under the same name would have it index authors as communities. A new NSID // makes a stale consumer ignore the records entirely, which is the correct // failure mode. -const PostV2Collection = "social.coves.community.postv2" +// +// It is an alias for the domain's constant rather than a second spelling of the +// string: the read path resolves post URIs against the same name, and two +// literals would let the indexer and the reader drift into disagreeing about +// what a post record is called. +const PostV2Collection = posts.PostV2Collection // DeletedAccountLookup reports whether a DID names an account this AppView was // asked to erase (migration 036, PRD rev 2.7). @@ -362,11 +367,108 @@ func (c *PostEventConsumer) handleAuthorPostEvent(ctx context.Context, event *Je case "create", "update": return c.upsertAuthorPost(ctx, authorDID, commit, event.TimeUS) case "delete": - return c.tombstoneRecord(ctx, recordURI(authorDID, PostV2Collection, commit.RKey), commit.Rev) + return c.tombstoneAuthorPost(ctx, authorDID, commit) } return nil } +// tombstoneAuthorPost soft-deletes the author's post and then withdraws any +// acceptance a community this AppView HOSTS still holds for it (§5.3). +// +// THE ORDER IS THE CONTRACT. The tombstone is the local truth and lands first: +// the author asked for their post to be gone, and a community PDS that cannot +// be reached must not keep this AppView serving it. The withdrawal is +// best-effort cleanup of a REMOTE repo, and it is deliberately not allowed to +// hold the deletion hostage. +func (c *PostEventConsumer) tombstoneAuthorPost(ctx context.Context, authorDID string, commit *CommitEvent) error { + uri := recordURI(authorDID, PostV2Collection, commit.RKey) + + // Read before the tombstone, because the community is what says WHOSE + // acceptance to withdraw and a delete event carries no record to read it + // from. The soft delete leaves the row in place, so this could equally run + // afterwards; doing it first keeps the sweep off the path when the post was + // never indexed here at all. + stored, indexed, err := c.loadStoredPost(ctx, uri) + if err != nil { + return err + } + + applied, err := c.tombstoneRecordIfRevWins(ctx, uri, commit.Rev) + if err != nil { + return err + } + if !applied || !indexed { + // A gate skip means this deletion was already applied — the sweep ran + // with it — so re-sweeping would put one authenticated PDS round trip + // behind every redelivery of every tombstone on the network. + return nil + } + + c.withdrawAcceptance(ctx, stored.communityDID, uri) + return nil +} + +// withdrawAcceptance asks the community's host to delete its acceptance of a +// post whose author has just deleted it. +// +// It is a NO-OP unless three things are true, and each exclusion removes a +// large class of pointless work: +// +// - a sweep is wired at all (nil on any build without the community writer); +// - THIS AppView holds the community's credentials — the acceptance lives in +// the community's repo and needs its keys, so on any instance that is not +// the community's home this is silently not our job, which is the common +// case; +// - an acceptance actually stands, per the AppView's own admission row. +// Consulting it is what keeps this from being a PDS round trip per delete +// event, since most posts a community sees it never accepted. +// +// A failure is LOGGED AND SWALLOWED. Returning it would dead-letter an event +// whose local half already committed, and the redrive would then be rejected by +// the rev gate — so the retry could never reach this code again anyway. The +// standing acceptance is left pointing at a deleted record until something +// revisits it, which nothing currently does: that gap is real, bounded to +// hosted communities, and named here rather than hidden. +func (c *PostEventConsumer) withdrawAcceptance(ctx context.Context, communityDID, postURI string) { + if c.acceptanceCleanup == nil || communityDID == "" { + return + } + + admission, err := c.admissions.Get(ctx, communityDID, postURI) + if err != nil { + if !errors.Is(err, posts.ErrNotFound) { + log.Printf("[ACCEPTANCE-SWEEP] Warning: could not read the admission of %s in %s: %v", + postURI, communityDID, err) + } + return + } + // The URI, not the status. `accepted` and `pending_reacceptance` both have a + // live acceptance record standing in the community's repo — the second + // merely pins content the author has since edited — and both must be + // withdrawn when the subject itself is deleted. + if admission == nil || admission.AcceptanceURI == nil { + return + } + + result, err := c.acceptanceCleanup.DeleteAcceptance(ctx, posts.CommunityAcceptanceDeleteCommand{ + CommunityDID: communityDID, + PostURI: postURI, + }) + switch { + case errors.Is(err, posts.ErrCommunityNotHosted): + // Not this instance's community. Expected, and not worth a warning: + // every AppView sees the deletions of every community it indexes. + log.Printf("debug: not withdrawing the acceptance of %s — %s is hosted elsewhere", postURI, communityDID) + case err != nil: + log.Printf("[ACCEPTANCE-SWEEP] Warning: could not withdraw the acceptance of %s in %s; "+ + "the record now cites a deleted post: %v", postURI, communityDID, err) + case result.Skipped: + log.Printf("debug: no acceptance of %s stood in %s to withdraw", postURI, communityDID) + default: + log.Printf("✓ Withdrew the acceptance of deleted post %s in %s", postURI, communityDID) + } +} + // canRecordAdmissions reports whether this consumer has somewhere to put a // decision. // diff --git a/internal/atproto/jetstream/post_consumer.go b/internal/atproto/jetstream/post_consumer.go index f9896fc..5ee6dbb 100644 --- a/internal/atproto/jetstream/post_consumer.go +++ b/internal/atproto/jetstream/post_consumer.go @@ -247,17 +247,26 @@ func (c *PostEventConsumer) createPost(ctx context.Context, repoDID string, comm // Soft-deletes the post in AppView database by setting deleted_at timestamp func (c *PostEventConsumer) deletePost(ctx context.Context, repoDID string, commit *CommitEvent) error { // Format: at://community_did/social.coves.community.post/rkey - return c.tombstoneRecord(ctx, fmt.Sprintf("at://%s/social.coves.community.post/%s", repoDID, commit.RKey), commit.Rev) + _, err := c.tombstoneRecordIfRevWins(ctx, + fmt.Sprintf("at://%s/social.coves.community.post/%s", repoDID, commit.RKey), commit.Rev) + return err } -// tombstoneRecord soft-deletes the post at uri under the rev gate. +// tombstoneRecordIfRevWins soft-deletes the post at uri under the rev gate, and +// reports whether the deletion APPLIED. // // SOFT, never hard, whichever repo the record lived in: the row is the rev // gate's tombstone, the comment thread's parent, and what moderation still // reads. It is shared by the community-repo and author-repo delete paths // because a deletion is the one operation where the two are identical — the // URI already says whose repo it was. -func (c *PostEventConsumer) tombstoneRecord(ctx context.Context, uri, rev string) error { +// +// The applied flag exists for the author-repo path's acceptance sweep, which +// must fire once per deletion rather than once per DELIVERY of it: the +// connector rewinds its cursor after every reconnect, so a tombstone that +// re-swept on each redelivery would put an authenticated PDS round trip behind +// every replayed event. +func (c *PostEventConsumer) tombstoneRecordIfRevWins(ctx context.Context, uri, rev string) (bool, error) { // REV GATE + soft delete in one transaction (the repo's SoftDelete is not // transaction-aware, and the delete's rev must be recorded atomically with // the tombstone: it is what rejects a stale cross-feed copy of the CREATE @@ -266,7 +275,7 @@ func (c *PostEventConsumer) tombstoneRecord(ctx context.Context, uri, rev string // record is rejected too. tx, err := c.db.BeginTx(ctx, nil) if err != nil { - return fmt.Errorf("failed to begin transaction: %w", err) + return false, fmt.Errorf("failed to begin transaction: %w", err) } defer func() { if rollbackErr := tx.Rollback(); rollbackErr != nil && rollbackErr != sql.ErrTxDone { @@ -276,11 +285,11 @@ func (c *PostEventConsumer) tombstoneRecord(ctx context.Context, uri, rev string won, err := tryAdvanceRecordRev(ctx, tx, uri, rev) if err != nil { - return err + return false, err } if !won { logSkippedStaleRev(ConsumerPosts, "delete", uri, rev) - return nil + return false, nil } // Same statement as postRepo.SoftDelete, inlined for transactionality. @@ -288,15 +297,15 @@ func (c *PostEventConsumer) tombstoneRecord(ctx context.Context, uri, rev string if _, err := tx.ExecContext(ctx, `UPDATE posts SET deleted_at = NOW() WHERE uri = $1 AND deleted_at IS NULL`, uri, ); err != nil { - return fmt.Errorf("failed to soft delete post: %w", err) + return false, fmt.Errorf("failed to soft delete post: %w", err) } if err := tx.Commit(); err != nil { - return fmt.Errorf("failed to commit post delete transaction: %w", err) + return false, fmt.Errorf("failed to commit post delete transaction: %w", err) } log.Printf("✓ Deleted post: %s", uri) - return nil + return true, nil } // updatePost handles post record update events from Jetstream. diff --git a/internal/config/config.go b/internal/config/config.go index 1ba1f98..eb5d0bb 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -355,6 +355,31 @@ type SubmissionsConfig struct { // repeat. It is separate from Window because the two answer different // questions: one bounds volume, the other catches retries. DedupeWindow time.Duration + + // AcceptanceQueueInterval is how often the acceptance engine's driver walks + // the undecided backlog (docs/PRD_AUTHOR_OWNED_POSTS.md §5.6). + // + // It is the PULL side of admission. The synchronous fast path and the + // firehose consumer both push work at the engine, and neither can see a + // subject that was left undecided because a credential expired or a lookup + // blipped — this pass is what eventually reaches those, so its cadence is + // the worst-case delay before a stranded post becomes visible. + // + // Zero DISABLES the driver, which is a supported deployment rather than a + // misconfiguration: an AppView that hosts no communities can accept nothing + // and has no backlog to walk. It is the one submission setting Validate does + // not require to be positive, for exactly that reason. + AcceptanceQueueInterval time.Duration + + // AcceptanceQueueBatchSize bounds how many subjects one pass lists. The + // backlog table grows with every submission the instance has ever seen, so + // an unbounded pass would hold a transaction open across all of it and then + // try to settle it inside a single cycle. + // + // Unlike the quotas above it is not validated, because it cannot fail open: + // the backlog query substitutes its own page size for a non-positive value + // and clamps an over-large one, so a bound exists whatever is set here. + AcceptanceQueueBatchSize int } // TokenEndpointEnabled reports whether the signup-token endpoint can operate. @@ -641,6 +666,19 @@ const ( defaultMaxSubmissionsPerCommunity = 10 defaultSubmissionWindow = time.Hour defaultSubmissionDedupeWindow = time.Hour + + // defaultAcceptanceQueueInterval is the backlog pass's cadence. A minute is + // the compromise the two failure modes point at from opposite directions: + // the pass is the only thing that reaches a subject nothing else will + // retry, so a long interval is a long wait for an author whose post got + // stuck, while a short one repeatedly scans a backlog that is usually empty. + defaultAcceptanceQueueInterval = time.Minute + + // defaultAcceptanceQueueBatch is how many subjects one pass takes. Small + // enough that a pass fits comfortably inside its own interval even when + // every subject needs a PDS round trip, since the backlog is drained across + // passes rather than in one. + defaultAcceptanceQueueBatch = 50 ) func (c *Config) loadSubmissions() error { @@ -657,10 +695,21 @@ func (c *Config) loadSubmissions() error { return err } + queueInterval, err := durationVar("ACCEPTANCE_QUEUE_INTERVAL", defaultAcceptanceQueueInterval) + if err != nil { + return err + } + queueBatch, err := intVar("ACCEPTANCE_QUEUE_BATCH_SIZE", defaultAcceptanceQueueBatch) + if err != nil { + return err + } + c.Submissions = SubmissionsConfig{ MaxPerAuthorPerCommunity: maxPerCommunity, Window: window, DedupeWindow: dedupeWindow, + AcceptanceQueueInterval: queueInterval, + AcceptanceQueueBatchSize: queueBatch, } return nil } @@ -771,6 +820,21 @@ func (c *Config) Validate() error { "it scopes the ledger's uniqueness bucket, and without a width every repost collides with the original forever", c.Submissions.DedupeWindow)) } + // The interval is checked for being NEGATIVE rather than non-positive, + // unlike the three above: zero is the documented way to disable the driver + // on an instance that hosts no communities, while a negative one is a + // time.Ticker panic waiting for the first boot after a typo. + if c.Submissions.AcceptanceQueueInterval < 0 { + problems = append(problems, fmt.Sprintf( + "ACCEPTANCE_QUEUE_INTERVAL cannot be negative (got %s); use 0 to disable the acceptance queue driver", + c.Submissions.AcceptanceQueueInterval)) + } + // The batch size is deliberately NOT validated, unlike every quota above. + // The quotas fail open when unset — an absent limit is no limit — so an + // omission there has to stop the boot. A batch size does not: the backlog + // query substitutes its own page for a non-positive limit and clamps an + // over-large one, so the bound exists whatever this value is, and the worst + // an omission costs is a different page size. if !c.IsDevEnv { switch { diff --git a/internal/core/posts/community_repo_factory.go b/internal/core/posts/community_repo_factory.go index 8d6b0fa..4a1eca8 100644 --- a/internal/core/posts/community_repo_factory.go +++ b/internal/core/posts/community_repo_factory.go @@ -3,12 +3,12 @@ package posts import ( "context" "errors" + "fmt" + "Coves/internal/atproto/pds" "Coves/internal/core/communities" ) -// RED STUB (task 5, cycle 2). Signatures only; the body is GREEN's. - // ErrCommunityNotHosted reports that this AppView does not hold the community's // PDS credentials, so it cannot write records into that community's repo. // @@ -28,6 +28,39 @@ type CommunityCredentialSource interface { EnsureFreshToken(ctx context.Context, community *communities.Community) (*communities.Community, error) } +// NewCommunityCredentialRefresher is the production CredentialRefresher: the +// engine's one forced renewal after a write comes back 401. +// +// KNOWN LIMITATION, recorded rather than hidden. EnsureFreshToken renews only a +// token that is within its expiry BUFFER, so a token the PDS has rejected for +// any other reason — revoked, invalidated by a password change, rotated out of +// band — is re-fetched unchanged and the retry fails identically. The subject +// then defers and the next pass tries again, which is correct but slower than +// it could be. Repairing it properly means a force-renew path on +// communities.Service, which is a wider change than this task, and the current +// behaviour is at worst the behaviour of having no refresher at all. +func NewCommunityCredentialRefresher(source CommunityCredentialSource) CredentialRefresher { + return credentialRefresher{source: source} +} + +type credentialRefresher struct { + source CommunityCredentialSource +} + +func (r credentialRefresher) RefreshCommunityCredentials(ctx context.Context, communityDID string) error { + community, err := r.source.GetByDID(ctx, communityDID) + if err != nil { + return fmt.Errorf("re-reading the credentials of %s: %w", communityDID, err) + } + if community == nil { + return fmt.Errorf("re-reading the credentials of %s: no such community is indexed", communityDID) + } + if _, err := r.source.EnsureFreshToken(ctx, community); err != nil { + return fmt.Errorf("renewing the credentials of %s: %w", communityDID, err) + } + return nil +} + // NewCommunityRepoFactory builds the production CommunityRepoFactory: a // credential lookup, a token refresh, and a PDS client bound to the community's // own repo. @@ -48,8 +81,59 @@ type CommunityCredentialSource interface { // refresh token — which happens exactly once, when it provisioned the account // through social.coves.community.create — or it does not. That is the honest // question, and it is the only one this factory asks. -func NewCommunityRepoFactory(communities CommunityCredentialSource) CommunityRepoFactory { +func NewCommunityRepoFactory(source CommunityCredentialSource) CommunityRepoFactory { return func(ctx context.Context, communityDID string) (CommunityRepo, error) { - return nil, nil + community, err := source.GetByDID(ctx, communityDID) + if err != nil { + if communities.IsNotFound(err) { + // Deliberately NOT ErrCommunityNotHosted. A community nobody has + // indexed may simply not have arrived yet — cross-repo delivery + // order is not guaranteed — and spelling an ordering artefact as + // a permanent skip would abandon a subject that resolves on its + // own once the profile lands. + return nil, fmt.Errorf("opening the repo of %s: no community with that DID is indexed: %w", + communityDID, err) + } + return nil, fmt.Errorf("opening the repo of %s: %w", communityDID, err) + } + if community == nil { + return nil, fmt.Errorf("opening the repo of %s: the community lookup returned nothing", communityDID) + } + + // THE HOSTING TEST, and the only one this factory is allowed to make. + // A stored refresh token exists exactly when this AppView provisioned + // the account itself; nothing a remote repo can publish creates one. + // See the note above for why hosted_by_did is not consulted. + if community.PDSRefreshToken == "" { + return nil, fmt.Errorf("opening the repo of %s: %w", communityDID, ErrCommunityNotHosted) + } + + // The token is renewed BEFORE the client is built rather than after a + // write fails, because a client is bound to the access token it was + // constructed with: refreshing afterwards would leave this caller + // holding the stale one. + fresh, err := source.EnsureFreshToken(ctx, community) + if err != nil { + return nil, fmt.Errorf("refreshing the credentials of %s: %w", communityDID, err) + } + if fresh == nil || fresh.PDSAccessToken == "" { + return nil, fmt.Errorf("refreshing the credentials of %s: no access token came back", communityDID) + } + + client, err := pds.NewFromAccessToken(fresh.PDSURL, fresh.DID, fresh.PDSAccessToken) + if err != nil { + return nil, fmt.Errorf("building a PDS client for %s: %w", communityDID, err) + } + + // The community-repo writers need the commit rev and applyWrites, and + // neither is on the base Client. Asserted rather than assumed: a + // transport that lost either would otherwise fail at the first + // moderation commit, which is after a verdict has already been reached. + repo, ok := client.(CommunityRepo) + if !ok { + return nil, fmt.Errorf("building a PDS client for %s: the client does not implement the community-repo "+ + "write surface (commit rev + applyWrites)", communityDID) + } + return repo, nil } } diff --git a/internal/core/posts/community_writer.go b/internal/core/posts/community_writer.go index 1b45ebc..d715540 100644 --- a/internal/core/posts/community_writer.go +++ b/internal/core/posts/community_writer.go @@ -21,6 +21,17 @@ import ( // because a re-fire must not mint a new record CID. const ( + // PostV2Collection is the AUTHOR-repo collection a post record lives in + // under author-owned posts (§3.1) — the successor to the deprecated + // community-repo social.coves.community.post. + // + // It lives beside the two community-repo collections because the three are + // one vocabulary: an acceptance's subject is a record in this collection, + // and the ingestion consumer re-exports this constant rather than declaring + // its own so that the reader and the writer cannot come to disagree about + // what a post record is called. + PostV2Collection = "social.coves.community.postv2" + // AcceptanceCollection is the community-repo collection holding a // community's attestation that it accepts a post. AcceptanceCollection = "social.coves.community.acceptance" @@ -353,9 +364,83 @@ func (w *communityRecordWriter) RepinAcceptance(ctx context.Context, cmd Communi return w.pinAcceptance(ctx, cmd, acceptanceMustExist) } -// DeleteAcceptance is a RED STUB (task 5, cycle 2); the body is GREEN's. +// DeleteAcceptance withdraws a standing acceptance and writes nothing in its +// place. +// +// STATE-SHAPED, like every other writer here and for the same reason task 4 +// recorded: the PDS has no tolerant delete, and answers a delete of a missing +// record with a 500. So absence is read first and reported as a skip. That is +// not an edge case — it is the COMMON one, because the sweep fires on every +// tombstone event and most posts a community sees were never accepted by it, +// and because the connector rewinds its cursor after every reconnect so each +// tombstone arrives at least twice. +// +// It is a batch of one rather than a putRecord-shaped call because applyWrites +// is where a delete can be guarded by swapCommit: the pre-read that found the +// acceptance is only true until somebody else writes, and the guard is what +// turns a concurrent restore into a detected conflict instead of a silently +// deleted fresh acceptance. func (w *communityRecordWriter) DeleteAcceptance(ctx context.Context, cmd CommunityAcceptanceDeleteCommand) (CommunityWriteResult, error) { - return CommunityWriteResult{}, nil + if err := validateAcceptanceDeleteCommand(cmd); err != nil { + return CommunityWriteResult{}, err + } + + repo, err := w.openRepo(ctx, cmd.CommunityDID) + if err != nil { + return CommunityWriteResult{}, err + } + + rkey := SubjectRkey(cmd.PostURI) + uri := recordURI(repo.DID(), AcceptanceCollection, rkey) + + for attempt := 0; ; attempt++ { + // The head is read before the record, for the same reason commitPair + // reads it first: a swapCommit read afterwards could be newer than the + // state the batch was shaped from, guarding the commit against a + // revision that already contains the change the shape assumed absent. + head, err := repo.GetLatestCommit(ctx) + if err != nil { + return CommunityWriteResult{}, fmt.Errorf("reading the head of %s: %w", cmd.CommunityDID, err) + } + + standing, err := readStandingRecord(ctx, repo, AcceptanceCollection, rkey) + if err != nil { + return CommunityWriteResult{}, err + } + if standing == nil { + // Nothing to withdraw. The head is still reported as the Rev, on the + // same catch-up reasoning the other writers' skips use: a row + // stranded by an earlier failed stamp can be caught up from it. + return CommunityWriteResult{URI: uri, RKey: rkey, Rev: head.Rev, Skipped: true}, nil + } + + result, err := repo.ApplyWrites(ctx, []pds.Write{{ + Op: pds.WriteOpDelete, + Collection: AcceptanceCollection, + RKey: rkey, + }}, head.CID) + if err == nil { + // No CID: a delete leaves no record to name. The Rev is the §5.2 + // watermark the firehose copy of this same deletion is compared + // against, so it is the one field that must be here. + return CommunityWriteResult{URI: uri, RKey: rkey, Rev: result.CommitRev}, nil + } + + // A lost swapCommit and a 500 are the same fact from two directions: the + // state this batch was shaped from is not the state the PDS is in. Both + // are answered by reading again, never by resending the same shape — + // here that matters most for the 500, which is what a delete of a record + // somebody else removed between the pre-read and the commit looks like. + staleShape := errors.Is(err, pds.ErrSwapConflict) || errors.Is(err, pds.ErrServerError) + if !staleShape || attempt >= swapRetryLimit { + return CommunityWriteResult{}, fmt.Errorf("withdrawing the acceptance of %s in %s: %w", + cmd.PostURI, cmd.CommunityDID, err) + } + if err := w.backoff(ctx, attempt); err != nil { + return CommunityWriteResult{}, fmt.Errorf("withdrawing the acceptance of %s in %s: %w", + cmd.PostURI, cmd.CommunityDID, err) + } + } } // pinAcceptance makes an acceptance of cmd.PostCID stand at the subject's rkey. @@ -776,6 +861,26 @@ func validateWriteCommand(cmd CommunityWriteCommand) error { return validateSubjectURI("acceptance write", cmd.PostURI) } +// validateAcceptanceDeleteCommand refuses a withdrawal that names no repo or no +// subject. +// +// The subject check is not symmetry with the other writers — it is the point. +// The rkey is a DIGEST of the subject URI, so a malformed or empty subject +// hashes to a perfectly well-formed key pointing at something else entirely, +// and a delete aimed at the wrong rkey in a community's own repo is a WRITE. A +// validation that only the create paths performed would leave the one operation +// that destroys data unchecked. +func validateAcceptanceDeleteCommand(cmd CommunityAcceptanceDeleteCommand) error { + switch { + case cmd.CommunityDID == "": + return fmt.Errorf("acceptance withdrawal: %w", NewValidationError("communityDID", "is required")) + case cmd.PostURI == "": + return fmt.Errorf("acceptance withdrawal: %w", NewValidationError("postURI", + "is required — the record key is derived from it, so an empty subject deletes a well-formed key belonging to nothing")) + } + return validateSubjectURI("acceptance withdrawal", cmd.PostURI) +} + // validateRemovalCommand refuses a removal with no reason code. `code` is // required by the lexicon and is what a client renders in #removedPost and what // the author is told. diff --git a/internal/core/posts/decider.go b/internal/core/posts/decider.go index fda07c6..6700d33 100644 --- a/internal/core/posts/decider.go +++ b/internal/core/posts/decider.go @@ -2,10 +2,13 @@ package posts import ( "context" + "errors" + "fmt" + "log" + "os" + "strings" ) -// RED STUB (task 5, cycle 2). Signatures only; the body is GREEN's. - // The production AdmissionDecider: the adapter that turns "decide about this // indexed post" into the AdmissionRequest evaluateAdmissionPolicy already // answers (docs/PRD_AUTHOR_OWNED_POSTS.md §5.6). @@ -35,6 +38,37 @@ import ( // redelivery and then refuse the redecision as a duplicate of the very post it // is redeciding. +// TrustedAggregatorDIDs reads the trusted-actor allowlist out of the process +// environment: TRUSTED_AGGREGATOR_DIDS, comma-separated, falling back to the +// legacy single-DID KAGI_AGGREGATOR_DID. +// +// It exists so the WRITE path and the ENGINE cannot disagree about who is +// trusted. Those two decide about the same post at different moments, and a +// trusted actor skips visibility, ban and authorization entirely — so two +// spellings of "read this variable" drifting apart would mean the same author +// is privileged on one path and not the other, which is the least debuggable +// shape a permission bug can take. +// +// It is called ONCE, at wiring time, and its result handed to whoever needs it. +// Reading the environment inside a decision would hide the most consequential +// input to a security decision from the place that makes it, and would make the +// trusted branch untestable alongside t.Parallel — Go's testing package refuses +// t.Setenv there. +func TrustedAggregatorDIDs() map[string]bool { + raw := os.Getenv("TRUSTED_AGGREGATOR_DIDS") + if raw == "" { + raw = os.Getenv("KAGI_AGGREGATOR_DID") + } + + trusted := map[string]bool{} + for _, did := range strings.Split(raw, ",") { + if did = strings.TrimSpace(did); did != "" { + trusted[did] = true + } + } + return trusted +} + // PostLookup reads the indexed post a decision is about. Satisfied by // Repository. type PostLookup interface { @@ -93,7 +127,110 @@ func NewAdmissionEngineDecider(deps DeciderDeps) *AdmissionEngineDecider { return &AdmissionEngineDecider{deps: deps} } +// ErrSubjectGone reports that the post an admission row names no longer stands: +// it was tombstoned by its author, or was never indexed at all. +// +// It travels as an ERROR rather than as a DecisionCode, and the choice is the +// difference between a correct record and a defamatory one. The engine turns a +// code on a `pending_reacceptance` row into a REMOVAL — a signed, portable +// moderation act published to the firehose — and an author deleting their own +// post is not the community removing it. There is no code that can be minted +// here without risking that, so the decider declines to decide instead: nothing +// is written, and the subject leaves the backlog on its own, because +// ListPendingSubjects excludes tombstoned and unindexed posts. +var ErrSubjectGone = errors.New("the post this admission names no longer stands") + // DecideAdmission implements AdmissionDecider. func (d *AdmissionEngineDecider) DecideAdmission(ctx context.Context, communityDID, postURI string) (AdmissionDecision, error) { - return AdmissionDecision{}, nil + // THE POST FIRST, and everything else after it. Two things come out of this + // lookup and both gate the policy: whether there is any content to judge, + // and who wrote it — and the author is what the actor class is derived + // from, so nothing about privilege can be decided before this returns. + post, err := d.deps.Posts.GetByURI(ctx, postURI) + switch { + case err != nil && IsNotFound(err): + // Absent. An admission row can legitimately exist with no post — an + // acceptance that arrived before its subject (§5.4) — so this is a + // normal state rather than a corruption, and there is simply nothing to + // judge yet. + return undecided(fmt.Errorf("deciding %s for %s: %w: it was never indexed", + postURI, communityDID, ErrSubjectGone)) + case err != nil: + // A lookup that FAILED is not a post that is absent. The two look + // identical from here and mean opposite things: the first clears, the + // second does not, and collapsing them would let a Postgres blip refuse + // somebody's post. + return undecided(fmt.Errorf("deciding %s for %s: reading the post: %w", postURI, communityDID, err)) + case post == nil: + return undecided(fmt.Errorf("deciding %s for %s: %w: it was never indexed", + postURI, communityDID, ErrSubjectGone)) + case post.DeletedAt != nil: + // Tombstoned. The driver already excludes these, but a post can be + // deleted between the listing and the decision, and admitting one would + // write an acceptance for content that no longer exists — which the + // host-side tombstone sweep would then delete, once per pass, forever. + return undecided(fmt.Errorf("deciding %s for %s: %w: its author deleted it", + postURI, communityDID, ErrSubjectGone)) + } + + return evaluateAdmissionPolicy(ctx, admissionDeps{ + communities: d.deps.Communities, + bans: d.deps.Policy.Bans, + aggregators: d.deps.Authorizer, + ledger: d.deps.Policy.Ledger, + limits: d.deps.Policy.Limits, + now: d.deps.Policy.Now, + }, AdmissionRequest{ + Actor: d.classify(ctx, post.AuthorDID), + AuthorDID: post.AuthorDID, + // The community DID, which resolves to itself. The engine's input is an + // admission row, and its key is already the resolved DID — there is no + // client-typed handle anywhere on this path to resolve. + Community: communityDID, + // EMPTY, and it must stay empty. Fingerprint is the dedupe key, read + // only by reserveSubmission, and this path deliberately never reserves: + // the engine re-decides posts that already exist, so a ledger row here + // would charge an author's quota for a firehose redelivery and then + // refuse the redecision as a duplicate of the very post it is + // redeciding. + Fingerprint: "", + }) +} + +// classify decides what class of actor the author is. +// +// EVERY UNCERTAIN PATH FALLS TO ActorUser, the stricter class. A trusted +// aggregator skips visibility, ban and authorization entirely, so resolving a +// failed lookup UPWARD would hand the widest privileges in the system to +// whoever managed to make the lookup fail. Guessing downward costs an +// aggregator some refused posts until the lookup recovers — and CreatePost +// already made exactly this choice (service.go step 3), so the engine agreeing +// with it is also what keeps the write path and the ingestion path from +// disagreeing about who someone is. +func (d *AdmissionEngineDecider) classify(ctx context.Context, authorDID string) ActorClass { + // The trusted set is checked FIRST, which is both the cheaper path and the + // only one that costs nothing: it is an in-memory set resolved at + // construction, so a trusted actor never pays for a database lookup to + // learn what the process already knew. + if d.deps.TrustedAggregatorDIDs[authorDID] { + return ActorTrustedAggregator + } + + // With no aggregator collaborators wired — a deployment with no aggregator + // support at all — nobody can be classified as one, which is the strict + // answer rather than a degraded one. + if d.deps.Aggregators == nil || d.deps.Authorizer == nil { + return ActorUser + } + + registered, err := d.deps.Aggregators.IsAggregator(ctx, authorDID) + if err != nil { + log.Printf("[ADMISSION-DECIDER] Warning: classifying %s fell back to the user class, IsAggregator failed: %v", + authorDID, err) + return ActorUser + } + if registered { + return ActorRegisteredAggregator + } + return ActorUser } diff --git a/internal/core/posts/queue.go b/internal/core/posts/queue.go index f0c539a..7c116ba 100644 --- a/internal/core/posts/queue.go +++ b/internal/core/posts/queue.go @@ -2,11 +2,12 @@ package posts import ( "context" + "fmt" + "log" + "sync" "time" ) -// RED STUB (task 5, cycle 2). Signatures only; every body returns zero values. - // The acceptance engine's driver: the thing that decides WHEN the engine runs // and on what (docs/PRD_AUTHOR_OWNED_POSTS.md §5.6, §8). // @@ -129,22 +130,67 @@ type QueueDriver struct { backoffBase time.Duration backoffMax time.Duration - // deferrals holds the per-subject retry-not-before times. In-memory on - // purpose: it is a politeness hint, not state anything is allowed to depend - // on, so a restart that forgets it costs one extra attempt per subject and - // nothing else. - deferrals map[PendingSubject]time.Time + // deferrals holds the per-subject backoff state. In-memory on purpose: it is + // a politeness hint, not state anything is allowed to depend on, so a + // restart that forgets it costs one extra attempt per subject and nothing + // else. + // + // It is keyed by (community, post) rather than by the whole PendingSubject + // because the third field is a time.Time read back from Postgres, and Go + // compares those by wall clock AND monotonic reading AND location. Two + // reads of one unchanged row can therefore produce values that are equal to + // a human and distinct to a map, which would silently defeat the backoff. + deferrals map[subjectKey]deferral + // mu guards deferrals and snapshot. Snapshot is read by the health handler + // on an HTTP goroutine while RunPass is writing on the job goroutine, so + // this is a genuine race rather than a defensive one. + mu sync.Mutex snapshot QueueSnapshot } +// subjectKey identifies one subject by the two fields that actually name it. +type subjectKey struct { + communityDID string + postURI string +} + +func keyOf(subject PendingSubject) subjectKey { + return subjectKey{communityDID: subject.CommunityDID, postURI: subject.PostURI} +} + +// deferral is how long one subject is held back, and until when. +// +// The delay is carried alongside the deadline so it can GROW: a subject that +// defers repeatedly is one whose community is wedged, and re-offering it on a +// fixed interval would keep a steady trickle of doomed requests pointed at a +// PDS that is already failing. +type deferral struct { + until time.Time + delay time.Duration +} + +// Default queue bounds, applied when the corresponding option is not given. +// +// The batch bound is not a nicety: this query runs on a timer against a table +// that grows with every submission the instance has ever seen, so a driver +// built without one must still not ask for the whole backlog. +const ( + defaultQueueBatchSize = 100 + defaultQueueBackoffBase = time.Minute + defaultQueueBackoffMax = 15 * time.Minute +) + // NewQueueDriver wires the driver. func NewQueueDriver(subjects PendingSubjectLister, engine AdmissionProcessor, now Clock, opts ...QueueDriverOption) *QueueDriver { d := &QueueDriver{ - subjects: subjects, - engine: engine, - now: now, - deferrals: make(map[PendingSubject]time.Time), + subjects: subjects, + engine: engine, + now: now, + batchSize: defaultQueueBatchSize, + backoffBase: defaultQueueBackoffBase, + backoffMax: defaultQueueBackoffMax, + deferrals: make(map[subjectKey]deferral), } for _, opt := range opts { opt(d) @@ -159,10 +205,154 @@ func NewQueueDriver(subjects PendingSubjectLister, engine AdmissionProcessor, no // must not stop every other community's posts from being decided, which is what // an early return would do — and the row is still in the backlog next pass. func (d *QueueDriver) RunPass(ctx context.Context) (PassReport, error) { - return PassReport{}, nil + startedAt := d.now() + report := PassReport{StartedAt: startedAt} + + subjects, err := d.subjects.ListPendingSubjects(ctx, d.batchSize) + if err != nil { + // The one failure a pass has nothing to do about. Every other outcome + // below is per-subject and counted; this one means there is no work + // list to count against. + return report, fmt.Errorf("listing the acceptance backlog: %w", err) + } + report.Listed = len(subjects) + + for _, subject := range groupByCommunity(subjects) { + if d.heldBack(subject) { + continue + } + + outcome, err := d.engine.ProcessAdmission(ctx, subject.CommunityDID, subject.PostURI) + report.Processed++ + + // The ERROR is checked before the outcome, because a failing engine + // returns EngineDeferred alongside it and reading the outcome first + // would file every failure as a deferral — collapsing the two numbers an + // operator uses to decide whether this is credentials or a bug. + switch { + case err != nil: + report.Failed++ + log.Printf("[ACCEPTANCE-QUEUE] Warning: %s in %s could not be settled: %v", + subject.PostURI, subject.CommunityDID, err) + // NOT backed off. A failure is unexplained, so the driver has no + // basis for guessing how long to wait; the next pass re-lists it and + // the row is still there. Deferral is the engine SAYING "later". + case outcome == EngineDeferred: + report.Deferred++ + d.deferSubject(subject, startedAt) + default: + report.Settled++ + d.clearDeferral(subject) + } + } + + d.record(subjects, report, startedAt) + return report, nil } // Snapshot returns the driver's health surface as of the last completed pass. func (d *QueueDriver) Snapshot() QueueSnapshot { - return QueueSnapshot{} + d.mu.Lock() + defer d.mu.Unlock() + return d.snapshot +} + +// heldBack reports whether a subject's backoff has yet to elapse. +func (d *QueueDriver) heldBack(subject PendingSubject) bool { + d.mu.Lock() + defer d.mu.Unlock() + + held, ok := d.deferrals[keyOf(subject)] + return ok && d.now().Before(held.until) +} + +// deferSubject holds a subject back, doubling its wait each consecutive time. +func (d *QueueDriver) deferSubject(subject PendingSubject, at time.Time) { + d.mu.Lock() + defer d.mu.Unlock() + + key := keyOf(subject) + delay := d.backoffBase + if previous, ok := d.deferrals[key]; ok && previous.delay > 0 { + delay = previous.delay * 2 + } + if delay > d.backoffMax { + delay = d.backoffMax + } + d.deferrals[key] = deferral{until: at.Add(delay), delay: delay} +} + +// clearDeferral forgets a settled subject, so the map tracks the backlog rather +// than the history of everything the driver has ever seen. +func (d *QueueDriver) clearDeferral(subject PendingSubject) { + d.mu.Lock() + defer d.mu.Unlock() + delete(d.deferrals, keyOf(subject)) +} + +// record publishes the pass's health surface. +func (d *QueueDriver) record(subjects []PendingSubject, report PassReport, at time.Time) { + d.mu.Lock() + defer d.mu.Unlock() + + snapshot := QueueSnapshot{ + PendingBacklog: report.Listed, + LastPassDeferred: report.Deferred, + LastPassFailed: report.Failed, + } + // Taken as a MINIMUM rather than as subjects[0], even though the query + // orders by age. The oldest entry's age is the queue's only early warning, + // and deriving it from an ordering assumption would make it silently wrong + // the first time anything reorders the list. + for i, subject := range subjects { + if i == 0 || subject.CreatedAt.Before(*snapshot.OldestPendingAt) { + oldest := subject.CreatedAt + snapshot.OldestPendingAt = &oldest + } + } + passedAt := at + snapshot.LastPassAt = &passedAt + + d.snapshot = snapshot +} + +// groupByCommunity returns the subjects with each community's contiguous, and +// with duplicates dropped. +// +// GROUPING. swapCommit is repo-global, so two writers on one community's repo +// starve each other — task 4 recorded that before this driver existed. A single +// goroutine satisfies it today whatever the order; what the grouping protects is +// the NEXT version, where a worker pool has to shard on something, and the only +// safe partition is the community. Output that interleaved communities would be +// output with no partition in it. +// +// DEDUPLICATION. A subject appearing twice in one listing is what a racing edit +// produces, and §8's edit-debounce is exactly this rule: a post edited in a +// storm coalesces into one pending_reacceptance row, and a driver that re-decided +// it per occurrence would re-run the whole policy per keystroke. +// +// The order WITHIN a community is preserved, and so is the order BETWEEN them +// (first appearance wins), so the query's oldest-first discipline survives. +func groupByCommunity(subjects []PendingSubject) []PendingSubject { + var order []string + byCommunity := make(map[string][]PendingSubject) + seen := make(map[subjectKey]bool, len(subjects)) + + for _, subject := range subjects { + if seen[keyOf(subject)] { + continue + } + seen[keyOf(subject)] = true + + if _, known := byCommunity[subject.CommunityDID]; !known { + order = append(order, subject.CommunityDID) + } + byCommunity[subject.CommunityDID] = append(byCommunity[subject.CommunityDID], subject) + } + + grouped := make([]PendingSubject, 0, len(seen)) + for _, communityDID := range order { + grouped = append(grouped, byCommunity[communityDID]...) + } + return grouped } diff --git a/internal/core/posts/service.go b/internal/core/posts/service.go index 8fe70a7..4858bf1 100644 --- a/internal/core/posts/service.go +++ b/internal/core/posts/service.go @@ -9,7 +9,6 @@ import ( "io" "log" "net/http" - "os" "strings" "time" @@ -135,22 +134,11 @@ func (s *postService) CreatePost(ctx context.Context, req CreatePostRequest) (*C return nil, fmt.Errorf("authenticated DID does not match author DID") } - // 3. Determine actor type: trusted aggregator, other aggregator, or regular user - // Check against comma-separated list of trusted aggregator DIDs - trustedDIDs := os.Getenv("TRUSTED_AGGREGATOR_DIDS") - if trustedDIDs == "" { - // Fallback to legacy single DID env var - trustedDIDs = os.Getenv("KAGI_AGGREGATOR_DID") - } - isTrustedAggregator := false - if trustedDIDs != "" { - for _, did := range strings.Split(trustedDIDs, ",") { - if strings.TrimSpace(did) == req.AuthorDID { - isTrustedAggregator = true - break - } - } - } + // 3. Determine actor type: trusted aggregator, other aggregator, or regular user. + // The allowlist is read through the shared helper rather than inline, so + // this path and the acceptance engine's decider cannot drift into disagreeing + // about who is trusted — see TrustedAggregatorDIDs. + isTrustedAggregator := TrustedAggregatorDIDs()[req.AuthorDID] // Check if this is a non-trusted aggregator (requires database lookup) var isOtherAggregator bool @@ -858,8 +846,19 @@ func parsePostURIParts(uri, field string) (authority string, rkey string, err er if authority == "" { return "", "", NewValidationError(field, "invalid post URI: missing authority") } - if collection != postCollection { - return "", "", NewValidationError(field, fmt.Sprintf("invalid collection in URI: expected %s, got %s", postCollection, collection)) + // EITHER post collection is a well-formed post URI. A post now lives in the + // author's repo under social.coves.community.postv2 (§3.1), while every post + // written before the flip is still at the deprecated community-repo NSID, and + // a reader has to be able to name both — refusing postv2 here made the new + // records unfetchable by the endpoint that hydrates every feed. + // + // What the authority MEANS differs between them — the community for the old + // collection, the author for the new — so a caller that goes on to use it as + // one or the other must narrow this itself. parsePostURI does, because it + // writes to the repo the authority names. + if collection != postCollection && collection != PostV2Collection { + return "", "", NewValidationError(field, fmt.Sprintf("invalid collection in URI: expected %s or %s, got %s", + postCollection, PostV2Collection, collection)) } if rkey == "" { return "", "", NewValidationError(field, "invalid post URI: missing rkey") @@ -867,6 +866,22 @@ func parsePostURIParts(uri, field string) (authority string, rkey string, err er return authority, rkey, nil } +// CollectionOfPostURI returns the collection segment of an at:// record URI, or +// "" when the URI is not shaped like one. +// +// It is exported because the two post collections are now indexed into one +// table, so every layer that renders or narrows a post has to ask the same +// question of the same URI — the repository decides which record shape to +// build from it, and this path decides which writes it will accept. A second +// spelling of the split is a second place for the two to disagree. +func CollectionOfPostURI(uri string) string { + parts := strings.Split(strings.TrimPrefix(uri, "at://"), "/") + if len(parts) != 3 { + return "" + } + return parts[1] +} + // requireDIDAuthority enforces that a parsed post-URI authority is a DID (not a handle). // Handles are mutable, so a handle-based URI would break after a community rename, or // mis-resolve if the handle is later reassigned. field names the request parameter for errors. @@ -1107,6 +1122,17 @@ func (s *postService) parsePostURI(uri string) (communityDID string, rkey string if err != nil { return "", "", err } + + // NARROWED to the community-repo collection, which the shared splitter + // deliberately is not. This path treats the authority as the COMMUNITY and + // goes on to open that repo and delete from it — so handed an author-repo + // postv2 URI it would authenticate as the AUTHOR's DID and try to delete a + // record there. The write path moves to postv2 in task 6; until it does, + // refusing is the only correct answer. + if collection := CollectionOfPostURI(uri); collection != postCollection { + return "", "", NewValidationError("uri", fmt.Sprintf( + "deleting a post is only supported for %s URIs, got %s", postCollection, collection)) + } if err := requireDIDAuthority(communityDID, "uri"); err != nil { return "", "", err } diff --git a/internal/db/migrations/036_create_deleted_accounts.sql b/internal/db/migrations/036_deleted_accounts_and_queue_index.sql similarity index 67% rename from internal/db/migrations/036_create_deleted_accounts.sql rename to internal/db/migrations/036_deleted_accounts_and_queue_index.sql index 153d4c4..fdc5f66 100644 --- a/internal/db/migrations/036_create_deleted_accounts.sql +++ b/internal/db/migrations/036_deleted_accounts_and_queue_index.sql @@ -56,5 +56,35 @@ COMMENT ON TABLE deleted_accounts IS 'Erasure markers: DIDs deleted on purpose, COMMENT ON COLUMN deleted_accounts.deleted_at IS 'When the deletion happened; read by retention and audit, never by the ingestion gate itself'; COMMENT ON COLUMN deleted_accounts.deleted_rev IS 'Repo revision the erasure was observed at, when one is known; NULL for AppView-initiated deletions'; +-- The acceptance engine's backlog scan (PRD_AUTHOR_OWNED_POSTS.md §5.6). +-- +-- It rides along in this migration rather than getting one of its own because +-- both halves are the same task's enablers, and an index is not worth a +-- version of its own on a table one migration away. +-- +-- WHY THE EXISTING INDEX CANNOT SERVE IT. Migration 034's +-- (community_did, status, created_at) index leads with the community, which is +-- exactly right for a moderator paging ONE community's queue and useless for +-- the driver's question — "what is undecided ANYWHERE" — which has no community +-- in hand and would scan every community's rows through it. +-- +-- WHY PARTIAL. The undecided rows are a small and roughly constant slice of a +-- table that grows with every submission the instance has ever seen: a settled +-- admission stays forever, and pending ones drain. Restricting the index to the +-- two undecided statuses keeps it proportional to the BACKLOG rather than to +-- history, so a pass that runs on a timer does not get more expensive every day +-- the instance stays up. +-- +-- Nothing FAILS without this index, which is precisely the danger: the query +-- keeps returning correct answers and quietly costs more every week, and the +-- symptom arrives as general database pressure with nothing pointing here. +CREATE INDEX idx_admissions_pending_queue + ON community_post_admissions (created_at) + WHERE status IN ('pending', 'pending_reacceptance'); + +COMMENT ON INDEX idx_admissions_pending_queue IS 'CRITICAL: the acceptance engine cross-community backlog scan, oldest first (PRD_AUTHOR_OWNED_POSTS 5.6)'; + -- +goose Down +DROP INDEX IF EXISTS idx_admissions_pending_queue; + DROP TABLE IF EXISTS deleted_accounts; diff --git a/internal/db/postgres/admission_queue_repo.go b/internal/db/postgres/admission_queue_repo.go index 584bbaf..d808657 100644 --- a/internal/db/postgres/admission_queue_repo.go +++ b/internal/db/postgres/admission_queue_repo.go @@ -2,11 +2,28 @@ package postgres import ( "context" + "fmt" "Coves/internal/core/posts" ) -// RED STUB (task 5, cycle 2). Signature only; the query is GREEN's. +// maxPendingSubjectsPerPass caps what one backlog scan may return, whatever a +// caller asks for. +// +// The bound belongs here rather than only at the call site because the caller +// is a periodic job and the table grows with every submission the instance has +// ever seen: a driver misconfigured with an enormous batch would hold one +// transaction open across the whole backlog and then try to settle all of it +// inside a single bounded cycle. +const maxPendingSubjectsPerPass = 500 + +// defaultPendingSubjectsPerPass is what a non-positive limit means. +// +// A zero limit reads as "no bound" to a LIMIT clause author and as "no work" to +// everyone else, and neither is a useful answer to give a queue. Substituting a +// modest page keeps a driver built without an explicit batch size working +// rather than silently idle. +const defaultPendingSubjectsPerPass = 100 // ListPendingSubjects returns the acceptance engine's backlog: subjects this // AppView can actually settle, oldest first. @@ -25,5 +42,48 @@ import ( // all (an acceptance that arrived before its subject), and a LEFT JOIN would // let those through as decidable when there is nothing to decide about. func (r *postgresAdmissionRepo) ListPendingSubjects(ctx context.Context, limit int) ([]posts.PendingSubject, error) { - return nil, nil + switch { + case limit <= 0: + limit = defaultPendingSubjectsPerPass + case limit > maxPendingSubjectsPerPass: + limit = maxPendingSubjectsPerPass + } + + // Both joins are INNER, and both exclusions are spelled as a join rather + // than as a NOT EXISTS so the planner can use the ordinary primary keys on + // posts and communities. ORDER BY created_at is the queue discipline: a + // queue served newest-first starves its own backlog, and the post that has + // waited longest is the one whose author is already wondering where it went. + const query = ` + SELECT admissions.community_did, admissions.post_uri, admissions.created_at + FROM community_post_admissions AS admissions + JOIN posts + ON posts.uri = admissions.post_uri + AND posts.deleted_at IS NULL + JOIN communities + ON communities.did = admissions.community_did + AND communities.pds_refresh_token_encrypted IS NOT NULL + WHERE admissions.status IN ('pending', 'pending_reacceptance') + ORDER BY admissions.created_at ASC + LIMIT $1 + ` + + rows, err := r.db.QueryContext(ctx, query, limit) + if err != nil { + return nil, fmt.Errorf("listing the acceptance engine's pending subjects: %w", err) + } + defer func() { _ = rows.Close() }() + + subjects := make([]posts.PendingSubject, 0, limit) + for rows.Next() { + var subject posts.PendingSubject + if err := rows.Scan(&subject.CommunityDID, &subject.PostURI, &subject.CreatedAt); err != nil { + return nil, fmt.Errorf("scanning a pending subject: %w", err) + } + subjects = append(subjects, subject) + } + if err := rows.Err(); err != nil { + return nil, fmt.Errorf("reading the pending subjects: %w", err) + } + return subjects, nil } diff --git a/internal/db/postgres/post_repo.go b/internal/db/postgres/post_repo.go index 025b499..f880bb2 100644 --- a/internal/db/postgres/post_repo.go +++ b/internal/db/postgres/post_repo.go @@ -17,6 +17,12 @@ import ( "github.com/lib/pq" ) +// legacyPostCollection is the DEPRECATED community-repo post record (§3.0). A +// post written since the author-owned flip is a posts.PostV2Collection record +// in the author's own repo; both are indexed into the same table, and the URI +// is what says which one a row came from. +const legacyPostCollection = "social.coves.community.post" + type postgresPostRepo struct { db *sql.DB } @@ -484,13 +490,28 @@ func scanPostView(rows *sql.Rows, extraDest ...interface{}) (*posts.PostView, er CommentCount: postView.CommentCount, } - // Build the record (required by lexicon) + // Build the record (required by lexicon). + // + // THE SHAPE FOLLOWS THE URI, because the two post collections are different + // records rather than two spellings of one. A postv2 record lives in the + // AUTHOR's repo and has NO author field — its removal is what makes + // authorship unforgeable (§3.1) — so synthesising one here would hand every + // reader back exactly the field the flip deleted, and a client that trusted + // it would be trusting a value this AppView made up. + collection := posts.CollectionOfPostURI(postView.URI) + if collection == "" { + collection = legacyPostCollection + } record := map[string]interface{}{ - "$type": "social.coves.community.post", + "$type": collection, "community": communityRef.DID, - "author": authorView.DID, "createdAt": postView.CreatedAt.Format(time.RFC3339), } + if collection == legacyPostCollection { + // The deprecated community-repo record DOES carry an author field, and + // it is part of the record a client may verify against the repo. + record["author"] = authorView.DID + } // Add optional fields to record if present if title.Valid { -- 2.51.2 From 7fae5f8dee5ae28f39107f7b5e2ac57d303e6914 Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 01:22:22 -0700 Subject: [PATCH 05/17] test(e2e): fix an unsatisfiable getStatus wait, and gofmt the queue tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit awaitStatus could never report done. PendingIfNotFound returns (done, err) — its nil case means "the read succeeded, that IS the answer" — and the guard read that first value as "pending", so every successful read returned false and the status comparison below was dead code. Each wait polled until the 45s budget expired, spending the contract's 100/min bucket on the way. Replaced with the direct form: a not-found is "no decision yet", any other error is terminal, and a successful read is where the comparison starts. The comment records why PendingIfNotFound is the wrong helper here rather than leaving the next reader to rediscover it. Also gofmt on queue_test.go, which make fmt-check rejected. Verified: gofmt -l over tests/e2e/ and internal/core/posts/ is empty, go vet -tags e2e passes, and contract-manifest still resolves all three markers at their new line numbers. NOTE: make fmt-check still fails on tests/lexicon_fixtures_test.go, which is pre-existing (committed unformatted in 10349f6, task 4) and outside the scope authorized here. Co-Authored-By: Claude Fable 5 --- internal/core/posts/queue_test.go | 6 +++--- tests/e2e/author_post_contract_test.go | 11 +++++++++-- 2 files changed, 12 insertions(+), 5 deletions(-) diff --git a/internal/core/posts/queue_test.go b/internal/core/posts/queue_test.go index 0315873..f32f4c5 100644 --- a/internal/core/posts/queue_test.go +++ b/internal/core/posts/queue_test.go @@ -115,9 +115,9 @@ func (e *fakeEngine) communityOrder() []string { // queueClock is a mutable instant the driver reads through Clock. type queueClock struct{ at time.Time } -func (c *queueClock) now() Clock { return func() time.Time { return c.at } } -func (c *queueClock) advance(d time.Duration) { c.at = c.at.Add(d) } -func newQueueClock() *queueClock { return &queueClock{at: time.Date(2026, 8, 8, 9, 0, 0, 0, time.UTC)} } +func (c *queueClock) now() Clock { return func() time.Time { return c.at } } +func (c *queueClock) advance(d time.Duration) { c.at = c.at.Add(d) } +func newQueueClock() *queueClock { return &queueClock{at: time.Date(2026, 8, 8, 9, 0, 0, 0, time.UTC)} } func subject(community, rkey string) PendingSubject { return PendingSubject{ CommunityDID: community, diff --git a/tests/e2e/author_post_contract_test.go b/tests/e2e/author_post_contract_test.go index 645a935..a634e81 100644 --- a/tests/e2e/author_post_contract_test.go +++ b/tests/e2e/author_post_contract_test.go @@ -148,8 +148,15 @@ func awaitStatus(t *testing.T, p *pipeline, postURI, communityDID, want, descrip var observed postStatusView p.Await(t, description, func() (bool, error) { view, err := p.PostStatus(context.Background(), postURI, communityDID) - if pending, wrapped := testkit.PendingIfNotFound(err); wrapped != nil || pending { - return false, wrapped + // NOT testkit.PendingIfNotFound: it is for probes where a successful read + // IS the answer, so its nil case reports DONE — which here would end the + // wait before the status was compared. This wait needs the opposite: a + // successful read is where the question starts. + if err != nil { + if testkit.IsNotFound(err) { + return false, nil + } + return false, err } observed = view return view.Status == want, nil -- 2.51.2 From 147a2a092dae6d8a0a64bf6c06dde40cc4c15dc5 Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 01:37:09 -0700 Subject: [PATCH 06/17] fix(queue): pace failing subjects like deferred ones, and judge dueness once per pass MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A wedged community's PDS answers the engine with an ERROR, not a deferral, so exempting failures from the backoff left the loudest case as the only one nothing paced. The two stay counted apart — that distinction is what tells an operator credentials from a bug — but they are now backed off identically, because 'how soon is it worth asking again' has the same answer for both. Dueness is also judged against the instant the pass began rather than the clock now, so one pass makes one decision: reading the clock per subject let a slow pass treat its first and last subjects by different rules. Co-Authored-By: Claude Fable 5 --- internal/core/posts/queue.go | 24 ++++++++++++++++++------ 1 file changed, 18 insertions(+), 6 deletions(-) diff --git a/internal/core/posts/queue.go b/internal/core/posts/queue.go index 7c116ba..340e1be 100644 --- a/internal/core/posts/queue.go +++ b/internal/core/posts/queue.go @@ -218,7 +218,7 @@ func (d *QueueDriver) RunPass(ctx context.Context) (PassReport, error) { report.Listed = len(subjects) for _, subject := range groupByCommunity(subjects) { - if d.heldBack(subject) { + if d.heldBack(subject, startedAt) { continue } @@ -229,14 +229,21 @@ func (d *QueueDriver) RunPass(ctx context.Context) (PassReport, error) { // returns EngineDeferred alongside it and reading the outcome first // would file every failure as a deferral — collapsing the two numbers an // operator uses to decide whether this is credentials or a bug. + // + // They are COUNTED apart and BACKED OFF the same. The counters are what + // an operator reads, and there the difference is the whole point: a pass + // that defers everything is usually credentials and will clear, while a + // pass that fails everything is a bug. The backoff answers a different + // question — "how soon is it worth asking again" — and the honest answer + // for a failure is at least as conservative as for a deferral. A wedged + // community's PDS returns errors, not deferrals, so exempting failures + // would leave the loudest case as the one thing nothing paced. switch { case err != nil: report.Failed++ log.Printf("[ACCEPTANCE-QUEUE] Warning: %s in %s could not be settled: %v", subject.PostURI, subject.CommunityDID, err) - // NOT backed off. A failure is unexplained, so the driver has no - // basis for guessing how long to wait; the next pass re-lists it and - // the row is still there. Deferral is the engine SAYING "later". + d.deferSubject(subject, startedAt) case outcome == EngineDeferred: report.Deferred++ d.deferSubject(subject, startedAt) @@ -258,12 +265,17 @@ func (d *QueueDriver) Snapshot() QueueSnapshot { } // heldBack reports whether a subject's backoff has yet to elapse. -func (d *QueueDriver) heldBack(subject PendingSubject) bool { +// +// It is judged against the instant the PASS began rather than against the clock +// now, so one pass makes one decision about what is due. Reading the clock per +// subject would let a long pass treat its first and last subjects by different +// rules, and would make which subjects ran depend on how slow the engine was. +func (d *QueueDriver) heldBack(subject PendingSubject, at time.Time) bool { d.mu.Lock() defer d.mu.Unlock() held, ok := d.deferrals[keyOf(subject)] - return ok && d.now().Before(held.until) + return ok && at.Before(held.until) } // deferSubject holds a subject back, doubling its wait each consecutive time. -- 2.51.2 From fa3807495a3470328a9ac4b19db105789d8d0d86 Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 02:15:30 -0700 Subject: [PATCH 07/17] test(ingestion): pin the handle-flood root causes and the fetch taxonomy MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four reds; T0 fully green and no pre-existing T1 breakage. 1. communities: the profile record CreateCommunity writes carries no `handle`. Read back via getRecord — the failure dumps the actual record, so the evidence is in the output rather than in an argument. Without the field the consumer must resolve one from the DID document, which on an egress-blocked stack yields "handle.invalid" into a UNIQUE column. 2. jetstream: a profile event whose handle is held by a DIFFERENT DID is currently swallowed as an idempotent replay and returns nil. Pinned as a PERMANENT refusal naming the contested handle, with the community absent and the incumbent untouched. Permanent is the load-bearing half: left transient it costs ~4.2s of inline blocking plus ten redrives per delivery, which is the flood. The other half of the narrowing is pinned beside it and PASSES today — a same-DID replay must stay a silent no-op. A fix that widened the refusal to every conflict would dead-letter every community's profile on every cursor rewind, and nothing else in the suite would notice. 3. jetstream: a genuine XRPC RecordNotFound from the author's PDS is classified transient today; pinned as permanent, with the httptest request count asserting one event produces exactly one fetch. Bare 404 and 5xx are pinned as transient and PASS today — they are the guards that stop the fix over-reaching, since a bare 404 usually means the request never reached a PDS at all (users.FetchProfileRecord draws the same distinction). Scope note: driving HandleEvent directly, no dead-letter row is written and no redrive runs — the consumer never touches that table, the connector does. These pin the input the connector switches on (the ErrPermanentEvent wrapping) plus the absence of any retry loop inside the consumer/fetcher. The connector's half is TestConnector_DeadLettersAfterRetryExhaustion. 4. jetstream: a community whose handle resolves to "handle.invalid" must not be stored, classified TRANSIENT (the PLC may be unreachable now and fine in a minute). Mirrors authorpost.go's user-path guard. No conflict with fix 2 — different causes, composing guards: this one refuses before the insert, that one refuses a collision the insert reports. Co-Authored-By: Claude Fable 5 --- .../jetstream/acceptance_consumer_test.go | 141 +++++++++++++ .../community_handle_conflict_test.go | 198 ++++++++++++++++++ .../community_profile_handle_test.go | 90 ++++++++ 3 files changed, 429 insertions(+) create mode 100644 internal/atproto/jetstream/community_handle_conflict_test.go create mode 100644 internal/core/communities/community_profile_handle_test.go diff --git a/internal/atproto/jetstream/acceptance_consumer_test.go b/internal/atproto/jetstream/acceptance_consumer_test.go index 534f7d5..416c2e3 100644 --- a/internal/atproto/jetstream/acceptance_consumer_test.go +++ b/internal/atproto/jetstream/acceptance_consumer_test.go @@ -498,6 +498,147 @@ func TestDirectPostFetcher_RefusesAPrivateHostByDefault(t *testing.T) { assert.Falsef(t, reached, "the guard must refuse before the request is made, not after the server has already answered") } +// --------------------------------------------------------------------------- +// §5.4 direct fetch: classifying what the author's PDS answered +// --------------------------------------------------------------------------- + +// How a failed fetch is CLASSIFIED, which decides what the connector does next. +// +// The consumer returns an error and the connector reads its shape: an error +// wrapping ErrPermanentEvent is dead-lettered with the redrive budget already +// spent, and anything else is retried inline (~4.2 seconds of blocking, per +// event) and then redriven ten times. So the classification is not a label — it +// is the difference between one forensic row and forty pointless refetches +// against somebody else's PDS while this consumer stops indexing. +// +// A non-200 from getRecord is currently one undifferentiated failure, and the +// three cases below need three different answers: +// +// - A GENUINE XRPC RecordNotFound is a definite fact about the repo: the PDS +// was reached, it understood the question, and the record is not there. No +// retry changes that. +// - A BARE 404 — no XRPC error envelope — usually means the request never +// reached a PDS at all: a stale pds_url pointing at a reverse proxy or a +// generic web server, which answers 404 for everything. Treating that as +// proof the record does not exist would permanently discard a post over a +// misconfigured hostname. users.FetchProfileRecord already draws exactly +// this distinction, and for exactly this reason. +// - A 5xx is the PDS saying it is having a bad time. Definitionally transient. +// +// WHAT IS ASSERTED HERE, AND WHAT IS NOT. These drive HandleEvent directly, so +// no dead-letter row is written and no redrive runs — the consumer never touches +// that table; the connector does. What these pin is the input the connector +// switches on (the ErrPermanentEvent wrapping) plus the request count, which +// proves the consumer and fetcher hold no retry loop of their own. The +// connector's half — that a permanent error is dead-lettered exhausted and a +// transient one is retried — is TestConnector_DeadLettersAfterRetryExhaustion. + +// countingAuthorPDS is fakeAuthorPDS with a request tally, so a test can prove +// one event produced exactly one outbound fetch. +func countingAuthorPDS(t *testing.T, expectRepo string, requests *int, handler http.HandlerFunc) *httptest.Server { + t.Helper() + return fakeAuthorPDS(t, expectRepo, func(w http.ResponseWriter, r *http.Request) { + *requests++ + handler(w, r) + }) +} + +func TestAcceptanceConsumer_GenuineRecordNotFound_IsPermanentlyRefusedAfterOneFetch(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + base := time.Now().UnixMicro() + uri := accPostURI("accgone") + + var requests int + srv := countingAuthorPDS(t, accAuthor, &requests, func(w http.ResponseWriter, r *http.Request) { + // The reference PDS' answer for a repo that exists and a record that + // does not: 400 with an XRPC error envelope naming RecordNotFound. + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusBadRequest) + _, _ = w.Write([]byte(`{"error":"RecordNotFound","message":"Could not locate record: ` + uri + `"}`)) + }) + defer srv.Close() + + f := newAccFixture(t, db, WithPostRecordFetcher(newFetcherAt(t, accAuthor, srv.URL))) + + err := f.consumer.HandleEvent(context.Background(), + acceptanceEvent(accCommunity, uri, "bafyreiaccgonepinned", testkit.TID(), base)) + + require.Error(t, err, "an acceptance whose subject the PDS says does not exist cannot be applied") + assert.ErrorIs(t, err, ErrPermanentEvent, + "a genuine RecordNotFound is a definite fact about the repo — the PDS was reached and answered — so no retry can change it. "+ + "Left transient, every one of these costs ~4.2s of inline blocking plus ten redrives, and a community can mint them at will "+ + "by writing acceptances for URIs nobody wrote") + + assert.Equalf(t, 1, requests, + "one event produced %d fetches: the consumer or the fetcher is retrying internally. Retries belong to the connector, "+ + "which can classify and budget them; a loop in here is invisible to it and unbounded", requests) + + assert.Zero(t, countRows(t, db, `SELECT count(*) FROM posts WHERE uri = $1`, uri), + "nothing may be indexed for a subject the PDS says does not exist") +} + +func TestAcceptanceConsumer_BareNotFound_StaysTransient(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + base := time.Now().UnixMicro() + uri := accPostURI("accbare404") + + var requests int + srv := countingAuthorPDS(t, accAuthor, &requests, func(w http.ResponseWriter, r *http.Request) { + // No XRPC envelope. This is what a reverse proxy, a load balancer or a + // generic web server answers — which is what a stale pds_url in a DID + // document points at. + w.Header().Set("Content-Type", "text/html") + w.WriteHeader(http.StatusNotFound) + _, _ = w.Write([]byte("404 Not Found")) + }) + defer srv.Close() + + f := newAccFixture(t, db, WithPostRecordFetcher(newFetcherAt(t, accAuthor, srv.URL))) + + err := f.consumer.HandleEvent(context.Background(), + acceptanceEvent(accCommunity, uri, "bafyreiaccbarepinned", testkit.TID(), base)) + + require.Error(t, err, "a fetch that did not produce a record cannot apply the acceptance") + assert.NotErrorIs(t, err, ErrPermanentEvent, + "a bare 404 carries no XRPC error envelope, which means the request most likely never reached a PDS at all — a stale pds_url "+ + "pointing at a proxy. Reading it as proof the record does not exist permanently discards a real post over a misconfigured "+ + "hostname; users.FetchProfileRecord draws the same distinction for the same reason") + + assert.Equal(t, 1, requests, "one event, one fetch — the connector owns retries") +} + +func TestAcceptanceConsumer_PDSServerError_StaysTransient(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + base := time.Now().UnixMicro() + uri := accPostURI("accpds5xx") + + var requests int + srv := countingAuthorPDS(t, accAuthor, &requests, func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(http.StatusServiceUnavailable) + _, _ = w.Write([]byte(`{"error":"InternalServerError","message":"upstream unavailable"}`)) + }) + defer srv.Close() + + f := newAccFixture(t, db, WithPostRecordFetcher(newFetcherAt(t, accAuthor, srv.URL))) + + err := f.consumer.HandleEvent(context.Background(), + acceptanceEvent(accCommunity, uri, "bafyreiacc5xxpinned", testkit.TID(), base)) + + require.Error(t, err) + assert.NotErrorIs(t, err, ErrPermanentEvent, + "a 5xx is the author's PDS saying it is unwell, which is the definition of transient; discarding the acceptance permanently "+ + "would lose a post because somebody else's server restarted") + + assert.Equal(t, 1, requests, "one event, one fetch — the connector owns retries") +} + // readAdmissionRow returns every mutable column of one admission row as a // comparable value, so "the row did not change" can be asserted as a whole // rather than field by field — a new column added later is covered without diff --git a/internal/atproto/jetstream/community_handle_conflict_test.go b/internal/atproto/jetstream/community_handle_conflict_test.go new file mode 100644 index 0000000..9c0c903 --- /dev/null +++ b/internal/atproto/jetstream/community_handle_conflict_test.go @@ -0,0 +1,198 @@ +//go:build integration + +package jetstream + +import ( + "context" + "fmt" + "testing" + "time" + + "Coves/internal/atproto/identity" + "Coves/internal/db/postgres" + "Coves/tests/fixtures" + "Coves/tests/testkit" + + _ "github.com/lib/pq" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// What a handle collision does to a community profile event. +// +// # THE SWALLOW, AND WHY IT LOOKS REASONABLE +// +// createCommunity ends in repo.Create, and treats communities.IsConflict as +// "already indexed — idempotent, nothing to do" (community_consumer.go). That +// reading is correct for exactly one of the two conflicts IsConflict matches. +// +// - ErrCommunityAlreadyExists means the DID is already in the table. That IS +// an idempotent replay: the community this event describes is indexed, the +// event changed nothing, and returning nil is right. Jetstream redelivers +// constantly, so this path is walked all the time. +// - ErrHandleTaken means a DIFFERENT DID already holds this handle. Nothing +// about that is idempotent. The community in the event was NOT indexed, is +// not in the table under any DID, and never will be — and the AppView says +// nothing, logs it as a successful replay, and moves on. +// +// The consequence is not confined to the community. Posts, comments and votes +// naming it are refused as "community not found", which the taxonomy classifies +// TRANSIENT (correctly — it is ordinarily a delivery race), so each one burns +// the connector's full inline retry budget before dead-lettering. A single +// swallowed collision turns into a sustained flood of 4.2-second blocking +// failures in three other consumers, none of which points anywhere near here. +// +// So the pin is in two halves, and both are needed: the collision must surface, +// and the genuine replay must keep NOT surfacing. A fix that widened the error +// into every conflict would dead-letter every redelivered profile event in the +// system. + +const conflictProfileCollection = "social.coves.community.profile" + +// communityProfileEvent builds the commit a community's profile write produces. +// The handle is carried in the record, which is the shape the AppView's own +// CreateCommunity produces once it stops omitting the field +// (internal/core/communities/community_profile_handle_test.go). +func communityProfileEvent(did, handle, name, rev string) *JetstreamEvent { + return &JetstreamEvent{ + Did: did, + Kind: "commit", + TimeUS: time.Now().UnixMicro(), + Commit: &CommitEvent{ + Rev: rev, + Operation: "create", + Collection: conflictProfileCollection, + RKey: "self", + CID: "bafyconflict" + rev, + Record: map[string]interface{}{ + "$type": conflictProfileCollection, + "handle": handle, + "name": name, + "displayName": "Conflict " + name, + "createdBy": "did:plc:conflicttestcreator", + "hostedBy": "did:web:test.local", + "visibility": "public", + "federation": map[string]interface{}{"allowExternalDiscovery": true}, + "createdAt": time.Now().UTC().Format(time.RFC3339), + }, + }, + } +} + +func TestCommunityConsumer_HandleTakenByAnotherDID_IsAPermanentRefusal(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + // skipVerification: the hostedBy/handle-domain check is a different security + // property with its own tests, and leaving it on would reject these events + // before the conflict path is ever reached. + consumer := NewCommunityEventConsumer(postgres.NewCommunityRepository(db), "did:web:test.local", true, nil) + + suffix := testkit.UniqueID(t) + contested := fmt.Sprintf("c-first%s.test.local", suffix) + incumbent := fixtures.DID("incumbent" + suffix) + newcomer := fixtures.DID("newcomer" + suffix) + + require.NoError(t, consumer.HandleEvent(ctx, communityProfileEvent(incumbent, contested, "first"+suffix, "3lconflicta")), + "fixture: the first community must index cleanly") + + // A SECOND, DIFFERENT community claiming the same handle. In production this + // is what two communities resolving to "handle.invalid" look like, but it is + // equally what a genuine handle race or a hostile duplicate looks like — and + // none of them is an idempotent replay. + err := consumer.HandleEvent(ctx, communityProfileEvent(newcomer, contested, "second"+suffix, "3lconflictb")) + + require.Errorf(t, err, + "a profile event whose handle is already held by a DIFFERENT community (%s) was accepted as an idempotent replay. "+ + "The community was never indexed, and every post naming it will dead-letter as \"community not found\" with "+ + "nothing in the logs pointing here", incumbent) + assert.ErrorIsf(t, err, ErrPermanentEvent, + "the refusal must be PERMANENT. A handle held by another DID does not resolve itself by waiting, so a transient "+ + "classification spends the connector's full inline retry budget (~4.2s, blocking the consumer) and then ten "+ + "redrives, per delivery, forever") + assert.Containsf(t, err.Error(), contested, + "the error must name the contested handle: it is the only thing that makes this diagnosable from a log line") + + // The newcomer is genuinely absent, and the incumbent is untouched — a + // refusal that had partially applied would be worse than the swallow. + var newcomerRows, incumbentRows int + require.NoError(t, db.QueryRow(`SELECT count(*) FROM communities WHERE did = $1`, newcomer).Scan(&newcomerRows)) + assert.Zero(t, newcomerRows, "the refused community must not be indexed") + + require.NoError(t, db.QueryRow( + `SELECT count(*) FROM communities WHERE did = $1 AND handle = $2`, incumbent, contested).Scan(&incumbentRows)) + assert.Equal(t, 1, incumbentRows, "the community that legitimately holds the handle must be untouched by the refusal") +} + +func TestCommunityConsumer_SameDIDReplay_StaysSilent(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + consumer := NewCommunityEventConsumer(postgres.NewCommunityRepository(db), "did:web:test.local", true, nil) + + suffix := testkit.UniqueID(t) + handle := fmt.Sprintf("c-replay%s.test.local", suffix) + did := fixtures.DID("replay" + suffix) + + event := communityProfileEvent(did, handle, "replay"+suffix, "3lreplaya") + require.NoError(t, consumer.HandleEvent(ctx, event)) + + // THE OTHER HALF OF THE NARROWING, and the reason it has to be pinned + // alongside the collision rather than left implied. The connector rewinds + // its cursor five seconds after every reconnect and the AppView consumes + // overlapping feeds, so this exact commit is guaranteed to be redelivered — + // constantly, for every community. A fix that widened the refusal to every + // conflict would dead-letter all of it. + require.NoError(t, consumer.HandleEvent(ctx, event), + "a redelivered profile event for the SAME DID is a genuine idempotent replay and must stay a silent no-op; "+ + "refusing it would dead-letter every community's profile on every cursor rewind") + + var rows int + require.NoError(t, db.QueryRow(`SELECT count(*) FROM communities WHERE did = $1`, did).Scan(&rows)) + assert.Equal(t, 1, rows, "a replay must not produce a second row") +} + +func TestCommunityConsumer_UnverifiableHandle_IsNotStored(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + + // A record with NO handle — the federated shape, where resolution is the + // only option — and a resolver that cannot verify the DID's handle. atProto + // identity resolution reports that as the reserved "handle.invalid" rather + // than as an error, so the consumer receives a perfectly well-formed + // identity naming a handle that is not one. + did := fixtures.DID("unverified" + testkit.UniqueID(t)) + resolver := &mockIdentityResolverForUser{identities: map[string]*identity.Identity{ + did: {DID: did, Handle: invalidHandle, PDSURL: "https://pds.example.invalid"}, + }} + consumer := NewCommunityEventConsumer(postgres.NewCommunityRepository(db), "did:web:test.local", true, resolver) + + event := communityProfileEvent(did, "", "unverified", "3lunverified") + delete(event.Commit.Record, "handle") + + err := consumer.HandleEvent(ctx, event) + + // This is the guard authorpost.go already applies on the user path, for the + // identical reason: "handle.invalid" is a PLACEHOLDER, not a handle, and the + // column it would land in is UNIQUE. Store it once and the next unverifiable + // community collides with it — which is the collision the test above pins, + // arriving from a completely different direction and with no attacker + // involved. + require.Errorf(t, err, + "a community whose handle could not be verified was indexed anyway. \"handle.invalid\" is the reserved "+ + "placeholder for exactly this case, and communities.handle is UNIQUE — so the FIRST one indexed takes the "+ + "placeholder and every later one collides with it") + assert.NotErrorIsf(t, err, ErrPermanentEvent, + "an unverifiable handle is a RESOLUTION failure, not a property of the record: the PLC directory may be "+ + "unreachable right now and verifiable in a minute, so the redrive has to be allowed to succeed") + + var stored int + require.NoError(t, db.QueryRow( + `SELECT count(*) FROM communities WHERE handle = $1`, invalidHandle).Scan(&stored)) + assert.Zerof(t, stored, + "%q was written into communities.handle; the next community that cannot resolve will collide with it", invalidHandle) +} diff --git a/internal/core/communities/community_profile_handle_test.go b/internal/core/communities/community_profile_handle_test.go new file mode 100644 index 0000000..5466529 --- /dev/null +++ b/internal/core/communities/community_profile_handle_test.go @@ -0,0 +1,90 @@ +//go:build integration + +package communities_test + +import ( + "context" + "strings" + "testing" + + "Coves/internal/core/communities" + "Coves/tests/testkit" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// The handle in the record CreateCommunity writes to the PDS. +// +// # WHY A MISSING MAP KEY IS A PIPELINE OUTAGE +// +// The community profile record is the ONLY thing that tells the AppView a +// community exists — the community consumer indexes repos it has never seen, and +// there is no signup step for communities. When the record carries no `handle`, +// the consumer falls through to resolving one from the DID document +// (community_consumer.go createCommunity), and on an egress-blocked stack that +// resolution cannot reach the PLC directory. What comes back is the reserved +// "handle.invalid". +// +// communities.handle carries a UNIQUE constraint. So the FIRST community indexed +// that way takes "handle.invalid", every subsequent one collides with it, and +// the collision is currently swallowed as an idempotent replay — the community +// is silently dropped, and every post, comment and vote naming it dead-letters +// as "community not found". One absent map key, and the visible symptom is a +// flood of unrelated transient failures four layers away. +// +// Writing the handle is not a departure from atProto's "handles are mutable, +// resolve them from DIDs" guidance. That guidance is about trusting a stranger's +// self-reported handle; this AppView PROVISIONED the account and asked the PDS +// for this exact handle, so putting it in the record states a fact it already +// holds. The consumer's resolution path stays for federated communities, where +// it is the only option available. +func TestCreateCommunity_ProfileRecordCarriesTheHandle(t *testing.T) { + t.Parallel() + + service, _, pdsServer := newCommunityService(t) + ctx := context.Background() + + // "c-" plus the name must stay inside the PDS' 18-character local-label cap, + // which is why UniqueIDWithPrefix is the only generator allowed here. + name := testkit.UniqueIDWithPrefix(t, "hc") + require.LessOrEqualf(t, len("c-"+name), testkit.MaxIDLength, + "the generated community name %q makes a handle label the PDS will refuse", name) + + community, err := service.CreateCommunity(ctx, communities.CreateCommunityRequest{ + Name: name, + DisplayName: "Handle Carrier", + Description: "a community whose profile record must name its handle", + Visibility: "public", + CreatedByDID: "did:plc:handlecarriertest", + }) + require.NoError(t, err) + require.NotEmpty(t, community.Handle, "fixture: the service must have derived a handle") + + // Read the record back out of the community's own repo — the bytes the + // firehose will carry, not the service's in-memory view of them. The + // consumer parses exactly this. + session := pdsServer.Login(t, community.Handle, community.PDSPassword) + record := session.GetRecord(t, "social.coves.community.profile", "self") + + handle, ok := record.Value["handle"].(string) + require.Truef(t, ok, + "the profile record carries no `handle` field: %#v.\n"+ + "Without it the consumer must resolve one from the DID document, which on an egress-blocked stack yields "+ + "the reserved \"handle.invalid\" — and communities.handle is UNIQUE, so the second community indexed that "+ + "way collides with the first and is dropped, taking every post that names it into the dead-letter queue "+ + "as \"community not found\"", record.Value) + + assert.Equal(t, community.Handle, handle, + "the record's handle must be the one the service provisioned and reports; a record that disagrees with the "+ + "AppView's own row is worse than a missing field, because the index and the firehose would then be "+ + "describing two different communities") + + // The shape is asserted independently of the service's own string building, + // so a change to the derivation has to be a deliberate edit here too. + assert.Truef(t, strings.HasPrefix(handle, "c-"+name+"."), + "the handle %q must be the c-{name}.{domain} form the provisioner builds; a handle in any other shape does not "+ + "resolve to this community's DID, and every client addressing it by handle 404s while the row looks healthy", handle) + assert.NotContains(t, handle, "handle.invalid", + "the reserved unverifiable-identity handle must never reach a record: it is the value that collides on the UNIQUE constraint") +} -- 2.51.2 From 66bc5255e5b239908d84723172ae2063335c8466 Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 02:22:22 -0700 Subject: [PATCH 08/17] fix(ingestion): close the handle-collision flood and sort the fetch taxonomy MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four fixes for RED's pins at fa38074. Three of them are one causal chain that ended as a dead-letter flood on a lane four layers from the cause. 1. communities: CreateCommunity's profile record now carries `handle`. That record is the ONLY thing that tells an AppView a community exists, so omitting the field left the consumer resolving one from the DID document — which on an egress-blocked stack yields "handle.invalid". Writing it is not a departure from "handles are mutable, resolve from DIDs": that guidance is about trusting a STRANGER's self-reported handle, and this process provisioned the account and asked the PDS for exactly this one. Federated communities still resolve, because there resolution is the only option. 2. jetstream: the community consumer no longer stores an unverifiable handle. Identity resolution reports an unverifiable DID by returning the reserved "handle.invalid" rather than an error, so a well-formed identity naming a non-handle was landing in a UNIQUE column — and the first one to do so made every later unverifiable community collide with it. TRANSIENT, because it is a fact about the resolution and not about the record: the directory may answer fine a minute later. Mirrors the user path's guard in authorpost.go. 3. jetstream: the conflict swallow is narrowed. communities.IsConflict matches two errors that mean opposite things — ErrCommunityAlreadyExists IS an idempotent replay (walked constantly, must stay a silent no-op), while ErrHandleTaken means a DIFFERENT DID holds the handle and the community in the event was never indexed at all. The second is now a PERMANENT refusal naming both DIDs and the handle; an unclassified conflict is reported rather than swallowed, so a future unique constraint cannot inherit the same silence. 4. jetstream: DirectPostFetcher sorts a non-200 getRecord instead of reporting one undifferentiated failure. A genuine XRPC RecordNotFound is permanent — the PDS was reached and answered, and left transient any community could mint unlimited lane-blocking by writing acceptances for URIs nobody wrote. A BARE 404 stays transient: with no envelope the request most likely never reached a PDS, so trusting it would discard a real post over a mistyped hostname. 5xx stays transient by definition. The predicate is EXPORTED from users (IsRecordNotFoundResponse) and FetchProfileRecord now calls it, so the line is drawn once. Why permanence is the load-bearing half of 3 and 4: a transient error costs the connector three inline retries — about 4.2 seconds of a blocked lane that now carries four collections — plus ten redrives, per delivery. ErrPermanentEvent short-circuits all of it. Co-Authored-By: Claude Fable 5 --- internal/atproto/jetstream/authorpost.go | 27 +++++++ .../atproto/jetstream/community_consumer.go | 77 ++++++++++++++++++- internal/core/communities/service.go | 21 ++++- internal/core/users/profile_backfill.go | 44 ++++++++--- 4 files changed, 156 insertions(+), 13 deletions(-) diff --git a/internal/atproto/jetstream/authorpost.go b/internal/atproto/jetstream/authorpost.go index 25487e2..84aaef0 100644 --- a/internal/atproto/jetstream/authorpost.go +++ b/internal/atproto/jetstream/authorpost.go @@ -251,6 +251,33 @@ func (f *DirectPostFetcher) FetchPost(ctx context.Context, postURI string) (*Fet if len(detail) > maxFetchErrorDetailBytes { detail = detail[:maxFetchErrorDetailBytes] } + // THE CLASSIFICATION IS THE EXPENSIVE PART OF THIS FUNCTION. The + // connector reads the returned error's shape: ErrPermanentEvent is + // dead-lettered with its redrive budget already spent, and anything + // else costs three inline retries (~4.2s of a blocked lane that also + // carries posts) plus ten redrives. So a non-200 has to be sorted, not + // merely reported. + // + // A GENUINE XRPC RecordNotFound is a definite fact about the repo: the + // PDS was reached, understood the question, and answered that the + // record is not there. Nothing a retry does changes it — and left + // transient, any community can mint unlimited lane-blocking by writing + // acceptances for URIs nobody ever wrote. + // + // A BARE 404 is the opposite, and the distinction is not pedantry: + // with no XRPC envelope the request most likely never reached a PDS at + // all — a stale pds_url pointing at a reverse proxy or a generic web + // server, both of which 404 everything. Reading that as proof the + // record does not exist would permanently discard a real post over a + // misconfigured hostname. Everything else, 5xx included, is the PDS + // having a bad time and is transient by definition. + // + // users.FetchProfileRecord draws exactly this line for exactly this + // reason; the predicates are shared with it rather than re-derived. + if users.IsRecordNotFoundResponse(resp.StatusCode, body) { + return nil, fmt.Errorf("%w: the PDS serving %s answered getRecord with status %d: %s", + ErrPermanentEvent, postURI, resp.StatusCode, strconv.Quote(detail)) + } // Quoted so control characters and ANSI escapes from a hostile PDS // cannot corrupt log output. return nil, fmt.Errorf("the PDS serving %s answered getRecord with status %d: %s", diff --git a/internal/atproto/jetstream/community_consumer.go b/internal/atproto/jetstream/community_consumer.go index 5a9a517..0e22938 100644 --- a/internal/atproto/jetstream/community_consumer.go +++ b/internal/atproto/jetstream/community_consumer.go @@ -7,6 +7,7 @@ import ( "Coves/internal/core/richtext" "context" "encoding/json" + "errors" "fmt" "log" "net/http" @@ -213,6 +214,25 @@ func (c *CommunityEventConsumer) createCommunity(ctx context.Context, did string if err != nil { return fmt.Errorf("failed to resolve handle from PLC for %s: %w (no fallback - will retry during backfill)", did, err) } + // "handle.invalid" IS NOT A HANDLE. atProto identity resolution + // reports a DID whose handle it could not verify bidirectionally + // by returning that reserved placeholder rather than an error, so + // what arrives here is a perfectly well-formed identity naming a + // non-handle — and communities.handle is UNIQUE. Store it once and + // every subsequent unverifiable community collides with it, which + // the insert reports as a conflict and the swallow below used to + // discard silently. + // + // TRANSIENT, deliberately: this is a fact about the RESOLUTION, not + // about the record. The PLC directory may be unreachable this + // second and answer fine the next, so the redrive has to be allowed + // to succeed. The user path applies the same guard for the same + // reason (authorpost.go, hydrateAuthorOpportunistically). + if identity.Handle == "" || identity.Handle == invalidHandle { + return fmt.Errorf("resolving the handle of community %s: identity resolution returned %q, "+ + "which is the reserved placeholder for an unverifiable handle and must never be stored in a unique column "+ + "(retryable — the directory may verify it later)", did, identity.Handle) + } profile.Handle = identity.Handle // Persist the resolved PDS host: BridgeTrust gates bridgedStats // on the post's community row carrying its repo's PDS URL, and a @@ -300,13 +320,42 @@ func (c *CommunityEventConsumer) createCommunity(ctx context.Context, did string } } - // Index in AppView database + // Index in AppView database. + // + // THE TWO CONFLICTS MEAN OPPOSITE THINGS, and treating them alike is what + // turned one handle collision into a flood of unrelated dead letters. + // communities.IsConflict matches both, so it is too wide to switch on here. _, err = c.repo.Create(ctx, community) if err != nil { - // Check if it already exists (idempotency) - if communities.IsConflict(err) { + switch { + case errors.Is(err, communities.ErrCommunityAlreadyExists): + // The DID is already in the table: a genuine idempotent replay. + // The connector rewinds its cursor after every reconnect and the + // AppView consumes overlapping feeds, so this path is walked + // constantly for every community and must stay a silent no-op. log.Printf("Community already indexed: %s (%s)", community.Handle, community.DID) return nil + + case errors.Is(err, communities.ErrHandleTaken): + // A DIFFERENT DID already holds this handle. Nothing about that is + // idempotent: the community in this event was NOT indexed, is in + // the table under no DID at all, and never will be while the + // incumbent stands. + // + // PERMANENT. A handle held by someone else does not resolve itself + // by waiting, so a transient classification spends the connector's + // full inline retry budget — about 4.2 seconds of blocking, on a + // lane that also carries posts — and then ten redrives, per + // delivery, forever. Both DIDs and the handle go in the message + // because a log line is the only place this is diagnosable from. + return fmt.Errorf("%w: cannot index community %s: handle %q is already held by community %s", + ErrPermanentEvent, community.DID, community.Handle, c.incumbentOfHandle(ctx, community.Handle)) + + case communities.IsConflict(err): + // A conflict this build does not recognise. Reported rather than + // swallowed: the swallow is what hid the handle collision, and a + // new unique constraint would otherwise inherit the same silence. + return fmt.Errorf("failed to index community %s: unclassified conflict: %w", community.DID, err) } return fmt.Errorf("failed to index community: %w", err) } @@ -315,6 +364,28 @@ func (c *CommunityEventConsumer) createCommunity(ctx context.Context, did string return nil } +// incumbentOfHandle names the community that already holds a contested handle, +// for the refusal message. +// +// BEST EFFORT, and never allowed to change the outcome. The refusal is already +// decided by the time this runs; this only fills in the half of the message an +// operator cannot otherwise get — "some other community has it" sends them to +// the database, "did:plc:… has it" sends them to the community. A lookup that +// fails yields a placeholder rather than an error, because replacing a precise +// permanent refusal with a vague transient one would trade the diagnosis for +// the flood this whole change exists to stop. +func (c *CommunityEventConsumer) incumbentOfHandle(ctx context.Context, handle string) string { + const unknown = "an unidentified DID" + if c.repo == nil { + return unknown + } + incumbent, err := c.repo.GetByHandle(ctx, handle) + if err != nil || incumbent == nil { + return unknown + } + return incumbent.DID +} + // updateCommunity updates an existing community from the firehose func (c *CommunityEventConsumer) updateCommunity(ctx context.Context, did string, commit *CommitEvent) error { if commit.Record == nil { diff --git a/internal/core/communities/service.go b/internal/core/communities/service.go index 8c5d629..be227f5 100644 --- a/internal/core/communities/service.go +++ b/internal/core/communities/service.go @@ -195,10 +195,29 @@ func (s *communityService) CreateCommunity(ctx context.Context, req CreateCommun return nil, fmt.Errorf("generated atProto handle is invalid: %w", validateErr) } - // Build community profile record + // Build community profile record. + // + // THE HANDLE IS IN THE RECORD, and its absence used to be a pipeline + // outage. This record is the ONLY thing that tells an AppView a community + // exists — the community consumer indexes repos it has never seen, and + // communities have no signup step — so a record without a handle leaves the + // consumer to resolve one from the DID document. On an egress-blocked stack + // that resolution cannot reach the PLC directory and yields the reserved + // "handle.invalid". communities.handle is UNIQUE, so the first community + // indexed that way takes the placeholder and every later one collides with + // it; the community is dropped, and every post, comment and vote naming it + // dead-letters as "community not found", four layers from the cause. + // + // This is NOT a departure from atProto's "handles are mutable, resolve them + // from DIDs" guidance. That guidance is about trusting a STRANGER's + // self-reported handle. This AppView provisioned the account and asked the + // PDS for exactly this handle, so writing it states a fact this process + // already holds. Consumers still resolve for federated communities, where + // resolution is the only option there is. profile := map[string]interface{}{ "$type": "social.coves.community.profile", "name": req.Name, // Short name for !mentions (e.g., "gaming") + "handle": pdsAccount.Handle, "visibility": req.Visibility, "hostedBy": s.instanceDID, // V2: Instance hosts, community owns "createdBy": req.CreatedByDID, diff --git a/internal/core/users/profile_backfill.go b/internal/core/users/profile_backfill.go index 481fee9..e61bf62 100644 --- a/internal/core/users/profile_backfill.go +++ b/internal/core/users/profile_backfill.go @@ -80,15 +80,9 @@ func FetchProfileRecord(ctx context.Context, client *http.Client, pdsURL, did st switch { case resp.StatusCode == http.StatusOK: // fall through to parse below - case resp.StatusCode == http.StatusNotFound && isXRPCErrorBody(body): - // Some PDS implementations 404 on missing records. Only trust the 404 - // when the body is an XRPC error object — a bare 404 (HTML from a - // reverse proxy or generic web server behind a stale pds_url) means we - // never reached a PDS at all, so it must surface as an error, not be - // silently classified as "user has no profile record". - return nil, nil - case resp.StatusCode == http.StatusBadRequest && isRecordNotFoundBody(body): - // Reference PDS returns 400 RecordNotFound when the repo exists but has no profile + case IsRecordNotFoundResponse(resp.StatusCode, body): + // The PDS was reached and said the record is not there. Absence is a + // normal outcome here, not an error. return nil, nil default: // SECURITY: cap the echoed body so a hostile PDS can't flood our logs, @@ -117,6 +111,38 @@ func FetchProfileRecord(ctx context.Context, client *http.Client, pdsURL, did st return &input, nil } +// IsRecordNotFoundResponse reports whether a getRecord response is a PDS +// saying, definitively, that the record is not there. +// +// It is exported because the answer decides how OTHER callers classify a failed +// fetch, and there must be exactly one line drawn. The firehose consumer's §5.4 +// direct fetch marks a genuine not-found as a PERMANENT event — dead-lettered +// with its redrive budget spent — while everything else stays transient and +// costs the connector three inline retries plus ten redrives. Two +// implementations of "is this a real not-found" would eventually disagree, and +// the disagreement would show up as either discarded posts or a blocked lane. +// +// THE TWO SHAPES IT ACCEPTS, and the one it deliberately does not: +// +// - 400 with an XRPC RecordNotFound (or an InvalidRequest whose message says +// it could not locate the record) — what the reference PDS answers when the +// repo exists and the record does not. +// - 404 WITH an XRPC error envelope — what some other implementations answer. +// - NOT a bare 404. With no envelope the request most likely never reached a +// PDS at all: a stale pds_url pointing at a reverse proxy or a generic web +// server, both of which answer 404 for everything. Trusting that would +// report a record as gone because somebody mistyped a hostname. +func IsRecordNotFoundResponse(statusCode int, body []byte) bool { + switch statusCode { + case http.StatusNotFound: + return isXRPCErrorBody(body) + case http.StatusBadRequest: + return isRecordNotFoundBody(body) + default: + return false + } +} + // isXRPCErrorBody reports whether body is an XRPC error JSON object (has a // non-empty "error" field). Used to distinguish a real PDS not-found response // from a bare 404 served by whatever non-PDS host a stale pds_url points at. -- 2.51.2 From 26d179c3f09212f1e01fc0c19f9f4a2764c6a742 Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 02:33:59 -0700 Subject: [PATCH 09/17] fix(e2e): make the tier poll cadence fit inside the limiter budget MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit contractPollInterval 250ms → 600ms so 100/min buckets outlast the 45s contractBudget on EVERY wait (75 polls < 100): four contracts measured 23.9-24.7s against the 25s cliff on the last gate run, and the posts lane legitimately runs long now that it carries four collections. The premise that justified 250ms — healthy waits never approach the budget — was measured false twice (task 4's comment contract, task 5's cluster). Per-wait comment-contract override folded into the default. Co-Authored-By: Claude Fable 5 --- tests/e2e/comment_contract_test.go | 16 +++------- tests/e2e/contracts_test.go | 51 ++++++++++++++++-------------- 2 files changed, 33 insertions(+), 34 deletions(-) diff --git a/tests/e2e/comment_contract_test.go b/tests/e2e/comment_contract_test.go index bfa39e0..c19806c 100644 --- a/tests/e2e/comment_contract_test.go +++ b/tests/e2e/comment_contract_test.go @@ -273,23 +273,17 @@ func indexedPost(t *testing.T, p *pipeline, community provisionedCommunity, auth postRecord(community.DID, authorDID, title, "a post to hang comments on")) uri := postURI(community.DID, rkey) - // 600ms, not the default 250ms: this wait legitimately runs LONG. Under a - // full-suite `make ci` the posts consumer is draining every parallel - // contract's records at once, and this parent-post index was measured at - // 23.97s on a QUIET stack — right at the ~25s cliff where 250ms polling - // exhausts the global 100/minute bucket and the wait dies as a 429 with - // 20 seconds of budget still unspent (observed 5 consecutive CI runs, - // 2026-08-08). At 600ms the full 45s contractBudget fits inside the - // bucket (75 polls < 100), so the wait fails on the budget or not at all — - // which is what contractPollInterval's own doc says a 429 here should - // mean. Discovery latency on a healthy fast run costs ~350ms extra. + // This wait was the first to hit the 250ms-era limiter cliff (measured + // 23.97s healthy latency; five consecutive gate deaths) and carried its + // own 600ms override until task 5 made that the tier default — see + // contractPollInterval's HISTORY note. p.Await(t, "the post these comments hang off to be indexed", func() (bool, error) { view, err := p.Post(context.Background(), uri) if err != nil { return false, err } return !view.NotFound, nil - }, testkit.WithPollInterval(600*time.Millisecond)) + }) return strongRef{URI: uri, CID: record.CID} } diff --git a/tests/e2e/contracts_test.go b/tests/e2e/contracts_test.go index 8df55d7..5c0a0ae 100644 --- a/tests/e2e/contracts_test.go +++ b/tests/e2e/contracts_test.go @@ -326,29 +326,34 @@ const contractHoldWindow = 5 * time.Second // contractPollInterval is how often a T2 wait re-asks the serving endpoint. // -// Slower than testkit's 100ms default ON PURPOSE, and the reason is the rate -// limiter described in the package doc. Every poll is a request against a -// 100-per-minute budget, so the interval and contractBudget are a pair: -// -// 45s budget ÷ 250ms = 180 polls if a wait runs its FULL length -// -// which is over the 100 a bucket allows. That is deliberate rather than -// overlooked, because of what the two cases cost: -// -// - A wait that SUCCEEDS costs one or two polls. The pipeline delivers in -// well under a second on this stack, and WaitFor probes before it sleeps, -// so a healthy contract never approaches the budget. This is every poll the -// tier issues on a green run. -// - A wait that FAILS was going to fail anyway. Past roughly 25 seconds it -// starts collecting 429s instead of "not yet" — so Await translates that -// status into a message saying so, rather than letting a rate limit -// masquerade as a broken endpoint. -// -// Buying the difference would mean either a 1s interval (adding half a second -// to every wait in the tier for the benefit of runs that are already red) or -// raising the AppView's limit in .env.ci — which would delete the one signal -// that a polling storm is happening at all. Neither trade is worth it. -const contractPollInterval = 250 * time.Millisecond +// Slower than testkit's 100ms default ON PURPOSE, and the interval and +// contractBudget are a pair against the 100-per-minute limiter described in +// the package doc: +// +// 45s budget ÷ 600ms = 75 polls if a wait runs its FULL length +// +// which fits inside the 100 a bucket allows, so a wait always fails on the +// BUDGET (a real timeout with consumer health attached), never on the +// limiter. That invariant is what the 429-explainer in Await promises, and +// it was not always true here. +// +// HISTORY — this was 250ms, on the stated premise that "the pipeline +// delivers in well under a second, so a healthy contract never approaches +// the budget," making the limiter cliff (100 polls ≈ 25s) unreachable for +// green runs. The premise was measured false twice as the suite grew: +// task 4 found the comment contract's parent-post wait at 23.97s healthy +// latency under full-suite load (five consecutive cliff deaths), and task 5 +// measured FOUR contracts clustered at 23.9-24.7s against the 25s wall — +// one scheduling hiccup from red on every gate run. Under `make ci` the +// posts lane legitimately runs tens of seconds behind (it carries four +// collections, and inline dead-letter retries block it by design), so long +// waits are healthy, not hopeless. 600ms buys the full budget for every +// wait at a cost of ~350ms average extra discovery latency on fast runs. +// +// Do NOT "fix" a marginal wait by raising the AppView's limit in .env.ci — +// that would delete the one signal that a genuine polling storm is +// happening at all. +const contractPollInterval = 600 * time.Millisecond // contractHoldPollInterval is how often a Holds re-asks. It is deliberately // four times slower than contractPollInterval, and the reason is arithmetic the -- 2.51.2 From cc5ce2f09ae83ecabc73c4f158ee7a3b82ecc8e8 Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 03:10:52 -0700 Subject: [PATCH 10/17] test(ingestion): pin the admission-durability and erasure-integrity findings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Seven reds plus three characterization guards. T0 clean; no other T1 failures in the package. RED — every one is the same shape, TWO WRITES ONE EVENT: the rev gate makes the first happen exactly once and the second is skipped, reverted or never retried because the event is already counted as done. P2 a lone removal-delete (no paired acceptance-create) leaves the row `removed` forever — the community's repo no longer says so, but the AppView still refuses to serve it and nothing will change that. P3 converge writes its fetched CID over a NEWER one a real event had already recorded; the race is made deterministic by running the post event from inside the fetch handler. P4 a failed UpsertPending orphans the admission: the redelivery is rev-gated, so the row is never created and the post is invisible in its community with nothing left to retry. P9 an acceptance for a tombstoned post is applied — getStatus would report `accepted` for a post no read path serves. P10 a retarget naming an unknown community errors instead of being discarded whole; immutability must outrank the transient unknown-community branch or an author mints redrive load by editing one field. P13 a failed acceptance withdrawal is never retried (GREEN's own comment at withdrawAcceptance names this gap). P8a a firehose-driven IndexUser CLEARS the erasure marker — any repo on the network can un-erase an account by emitting one record. PASSING, kept deliberately: P8b acceptance for a swept author is already gated. P8c an unreadable marker table already fails closed. The registration path still clears the marker — the guard against the obvious wrong fix for P8a (never clearing it at all). NOT IN THIS COMMIT — six items need a decision or a seam, reported separately: P1 (positive case needs a signed CAR this package has no PDS floor for), P5, P6, P7 (CONTRADICTS a green cycle-2 pin — must be resolved before either is written), P11, P12. Co-Authored-By: Claude Fable 5 --- .../jetstream/admission_durability_test.go | 380 ++++++++++++++++++ .../jetstream/erasure_integrity_test.go | 199 +++++++++ 2 files changed, 579 insertions(+) create mode 100644 internal/atproto/jetstream/admission_durability_test.go create mode 100644 internal/atproto/jetstream/erasure_integrity_test.go diff --git a/internal/atproto/jetstream/admission_durability_test.go b/internal/atproto/jetstream/admission_durability_test.go new file mode 100644 index 0000000..b7a86f4 --- /dev/null +++ b/internal/atproto/jetstream/admission_durability_test.go @@ -0,0 +1,380 @@ +//go:build integration + +package jetstream + +import ( + "context" + + "errors" + "net/http" + "testing" + "time" + + "Coves/internal/core/posts" + "Coves/internal/db/postgres" + "Coves/tests/testkit" + + _ "github.com/lib/pq" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// Six ways the admission row and the events that move it can fall out of step. +// +// They share a shape worth naming once, because it is the shape of almost every +// bug in this consumer: TWO WRITES, ONE EVENT. Indexing a post writes the posts +// row AND the admission row; a tombstone writes the tombstone AND withdraws the +// acceptance; converging writes the fetched post AND its pending admission. The +// rev gate exists to make the FIRST of each pair happen exactly once — and every +// case below is the second one being skipped, reverted or never retried because +// the gate already counted the event as done. +// +// The gate is right to be a gate. What is wrong is treating "this event has been +// seen" as "everything this event should have caused has happened", which is +// only true when the effects are atomic with the gate advance. Where they are +// not, the second write needs to be idempotent and unconditional rather than +// gated along with the first. + +// flakyAdmissions wraps the real repository and fails a chosen method a fixed +// number of times before letting it through — a transient database fault, which +// is the ordinary way the second write of a pair goes missing. +type flakyAdmissions struct { + posts.AdmissionRepository + + upsertFailures int + upsertCalls int + err error +} + +func (a *flakyAdmissions) UpsertPending(ctx context.Context, cmd posts.UpsertPendingCommand) (posts.AdmissionResult, error) { + a.upsertCalls++ + if a.upsertFailures > 0 { + a.upsertFailures-- + return posts.AdmissionResult{}, a.err + } + return a.AdmissionRepository.UpsertPending(ctx, cmd) +} + +// flakyDeleter fails the acceptance withdrawal a fixed number of times. +type flakyDeleter struct { + failures int + calls []posts.CommunityAcceptanceDeleteCommand + err error +} + +func (d *flakyDeleter) DeleteAcceptance( + _ context.Context, cmd posts.CommunityAcceptanceDeleteCommand, +) (posts.CommunityWriteResult, error) { + d.calls = append(d.calls, cmd) + if d.failures > 0 { + d.failures-- + return posts.CommunityWriteResult{}, d.err + } + return posts.CommunityWriteResult{Rev: testkit.TID()}, nil +} + +// removalDeleteEvent builds the delete half of a community's removal record, +// arriving on its own — no paired acceptance-create in the same commit. +func removalDeleteEvent(communityDID, postURI, rev string, timeUS int64) *JetstreamEvent { + return revCommitEvent(communityDID, posts.RemovalCollection, "delete", + posts.SubjectRkey(postURI), rev, "", timeUS, nil) +} + +func TestAdmission_LoneRemovalDeleteExitsRemoved(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + f := newAccFixture(t, db) + base := time.Now().UnixMicro() + + const cid = "bafyreiloneremoval" + uri := f.indexPV2(t, "loneremoval", cid, base) + + revs := increasingTIDs(t, 3) + require.NoError(t, f.consumer.HandleEvent(ctx, removalEvent( + accCommunity, uri, cid, string(posts.DecisionRuleViolation), revs[0], base+1_000_000))) + + row, err := f.admissions.Get(ctx, accCommunity, uri) + require.NoError(t, err) + require.Equal(t, posts.AdmissionStatusRemoved, row.Status, "fixture: the removal must stand") + + // A removal delete with NO paired acceptance-create. The restore commit of + // §5.2 is {removal-delete, acceptance-create} together, and treating the + // delete half as a no-op is safe THERE because the create outranks it. But a + // moderator can also simply withdraw a removal — deleting the record and + // writing nothing — and that commit carries only this event. Ignoring it + // leaves the post `removed` forever, with the community's own repo no longer + // saying so: the AppView and the signed record disagree, and only the + // AppView is consulted when the post is served. + require.NoError(t, f.consumer.HandleEvent(ctx, removalDeleteEvent(accCommunity, uri, revs[1], base+2_000_000))) + + row, err = f.admissions.Get(ctx, accCommunity, uri) + require.NoError(t, err) + assert.Equalf(t, posts.AdmissionStatusPending, row.Status, + "a lone removal delete left the post %q. The removal record is gone from the community's repo, so nothing "+ + "published says this post is removed — but the AppView still refuses to show it, and no later event will "+ + "change that, because `removed` is terminal against everything except a community event at a greater watermark", row.Status) + + // The audit fields go with it. A row that says `pending` while still + // carrying a decision code reads, to getStatus and to a moderator, as a post + // that was refused for a reason — which is the state the withdrawal undid. + assert.Nil(t, row.DecisionCode, "the withdrawn removal's code must be cleared with it") + assert.Nil(t, row.DecisionAt, "and its decision time") + assert.Truef(t, row.Redrivable, + "redrivable must be reset: it was set false by a terminal decision that no longer stands, and leaving it false "+ + "means the dead-letter redrive will never revisit this subject") +} + +func TestAdmission_ConvergeMustNotRegressTheEvaluatedCID(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + base := time.Now().UnixMicro() + + rkey := "convergerace" + uri := accPostURI(rkey) + const pinnedCID = "bafyreiconvergev1" + const currentCID = "bafyreiconvergev2" + + var f accFixture + + // THE RACE, MADE DETERMINISTIC. The fetch is in flight when the post's own + // firehose event lands — which is not a rare interleaving, it is the normal + // one: the acceptance and the post are in different repos, so Jetstream + // parallelises them, and the fetch exists precisely because the post event + // has not arrived yet. Running the real event from inside the fetch handler + // puts the two in the exact order the race produces, every time. + srv := fakeAuthorPDS(t, accAuthor, func(w http.ResponseWriter, r *http.Request) { + require.NoError(t, f.consumer.HandleEvent(context.Background(), pv2Event( + accAuthor, "create", rkey, testkit.TID(), currentCID, base+500_000, + pv2Record(accCommunity, "the version that actually arrived", "newer body"), + )), "the racing post event must index cleanly") + + // The PDS answers with the version the acceptance pinned, which by now is + // the OLDER one. + serveRecord(t, w, uri, pinnedCID, pv2Record(accCommunity, "the version the acceptance pinned", "older body")) + }) + defer srv.Close() + + f = newAccFixture(t, db, WithPostRecordFetcher(newFetcherAt(t, accAuthor, srv.URL))) + + _ = f.consumer.HandleEvent(ctx, acceptanceEvent(accCommunity, uri, pinnedCID, testkit.TID(), base)) + + row, err := f.admissions.Get(ctx, accCommunity, uri) + require.NoError(t, err) + require.NotNil(t, row) + + // evaluated_cid is what the NEXT decision judges. Regressed to the pinned + // CID, the engine evaluates content the author has already replaced, and — + // worse — an acceptance written against that verdict pins a version the + // AppView is no longer serving. The row would then report `accepted` for + // content nobody can see. + assertNullableStringPV2(t, currentCID, row.EvaluatedCID, + "the converge path wrote its fetched CID over a NEWER one that a real event had already recorded. "+ + "UpsertPending is last-write-wins, so a fetch that lost the race must not apply — the post event is the "+ + "authority on what content stands, and the fetch is a catch-up") + + assert.Equalf(t, posts.AdmissionStatusPendingReacceptance, row.Status, + "an acceptance pinning a CID the post no longer holds is pending_reacceptance, not accepted: the community "+ + "agreed to a version that has since been replaced") + + _, _, storedCID, _, _ := readPV2Post(t, db, uri) + assert.Equal(t, currentCID, storedCID, "the indexed post must hold the version its own event carried") +} + +func TestAdmission_SurvivesAFailedUpsertAcrossRedelivery(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + + flaky := &flakyAdmissions{ + AdmissionRepository: postgres.NewAdmissionRepository(db), + upsertFailures: 1, + err: errors.New("community_post_admissions is briefly unreachable"), + } + f := newAccFixture(t, db) + f.consumer = NewPostEventConsumer( + postgres.NewPostRepository(db), postgres.NewCommunityRepository(db), + newMockUserService(), db, + WithAdmissions(flaky), + WithDeletedAccounts(postgres.NewDeletedAccountRepository(db)), + ) + + rkey := "upsertflake" + uri := accPostURI(rkey) + rev := testkit.TID() + event := pv2Event(accAuthor, "create", rkey, rev, "bafyreiupsertflake", time.Now().UnixMicro(), + pv2Record(accCommunity, "indexed while the admissions table blipped", "body")) + + // First delivery: the post indexes, the admission does not. The event is + // dead-lettered, which is correct — but the rev gate has already advanced. + require.Error(t, f.consumer.HandleEvent(ctx, event), + "fixture: the first delivery must fail on the admission write") + + // THE REDELIVERY, which is guaranteed: the connector rewinds its cursor + // after every reconnect, the redriver replays dead letters, and overlapping + // feeds carry the same commit. It arrives with the SAME rev, so the gate + // rejects it — and if the admission upsert is gated along with the post + // insert, the row is never created and the post is invisible in its + // community forever, with no error anywhere and nothing left to retry. + require.NoError(t, f.consumer.HandleEvent(ctx, event), + "a redelivery of an already-gated event must not error") + + row, err := f.admissions.Get(ctx, accCommunity, uri) + require.NoErrorf(t, err, + "the admission row was never created. The rev gate makes the POST insert happen once; it must not also suppress "+ + "the admission upsert, which is idempotent and is the only thing that makes the post visible in its "+ + "community. Orphaned this way, nothing revisits it: the gate rejects every future delivery (upsert calls: %d)", + flaky.upsertCalls) + require.NotNil(t, row) + assert.Equal(t, posts.AdmissionStatusPending, row.Status) + assertNullableStringPV2(t, "bafyreiupsertflake", row.EvaluatedCID, "evaluated_cid") + + _, _, _, _, deletedAt := readPV2Post(t, db, uri) + assert.Nil(t, deletedAt, "the post itself indexed on the first delivery and must be untouched") +} + +func TestAdmission_AcceptanceForATombstonedPostIsNotApplied(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + f := newAccFixture(t, db) + base := time.Now().UnixMicro() + + const cid = "bafyreitombaccept" + rkey := "tombaccept" + uri := f.indexPV2(t, rkey, cid, base) + + revs := increasingTIDs(t, 2) + require.NoError(t, f.consumer.HandleEvent(ctx, pv2Event( + accAuthor, "delete", rkey, revs[0], "", base+1_000_000, nil))) + + _, _, _, _, deletedAt := readPV2Post(t, db, uri) + require.NotNil(t, deletedAt, "fixture: the post must be tombstoned") + + before, err := f.admissions.Get(ctx, accCommunity, uri) + require.NoError(t, err) + + // An acceptance can legitimately arrive after the author deleted the post: + // the community decided before it saw the tombstone, and the two events are + // in different repos with no ordering between them. Applying it makes the + // AppView report `accepted` for content it will never serve — and the + // host-side sweep, which already ran with the tombstone, will not run again, + // so the community's repo keeps an acceptance pointing at nothing. + require.NoError(t, f.consumer.HandleEvent(ctx, acceptanceEvent(accCommunity, uri, cid, revs[1], base+2_000_000)), + "an acceptance for a deleted post is a skip, not a failure: refusing it would dead-letter an event that will "+ + "be replayed and refused identically forever") + + after, err := f.admissions.Get(ctx, accCommunity, uri) + require.NoError(t, err) + assert.NotEqualf(t, posts.AdmissionStatusAccepted, after.Status, + "a tombstoned post was accepted. getStatus would report `accepted` for a post no read path will ever serve, "+ + "and the acceptance record in the community's repo now cites a record nobody can fetch") + assert.Equalf(t, before.Status, after.Status, + "the admission of a deleted post must be left exactly as it was") +} + +func TestAdmission_RetargetToAnUnknownCommunityIsAWholeEventSkip(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + f := newAccFixture(t, db) + base := time.Now().UnixMicro() + + const cid = "bafyreiretargetghost" + rkey := "retargetghost" + uri := f.indexPV2(t, rkey, cid, base) + + // An update that changes `community` to a DID nobody has indexed. TWO rules + // meet here and only one of them can be right: + // + // - community immutability (§3.1): discard the WHOLE event, silently, and + // never look at anything else it says; + // - unknown community (§5.3): a transient failure, because the community's + // own profile event may simply not have arrived yet. + // + // Immutability has to outrank it, and the reason is not aesthetic. The + // unknown-community branch is TRANSIENT, so a retarget naming a nonexistent + // DID would dead-letter and redrive ten times, blocking ~4.2 seconds inline + // on each delivery — and it can never succeed, because even once that + // community exists the event is still an illegal retarget. An author can + // mint that load at will by editing one field. + require.NoErrorf(t, f.consumer.HandleEvent(ctx, pv2Event( + accAuthor, "update", rkey, testkit.TID(), "bafyreiretargeted", base+1_000_000, + pv2Record("did:plc:accnevercommunity", "retargeted at a ghost", "body"), + )), "a retarget must be discarded on its own terms, before the community is ever looked up: checking the community "+ + "first turns an invalid event into a retryable one that can never succeed") + + _, communityDID, storedCID, _, _ := readPV2Post(t, db, uri) + assert.Equal(t, accCommunity, communityDID, "the community must not move") + assert.Equal(t, cid, storedCID, "the whole event is invalid, so its content must not be applied either") + + assert.Zero(t, countRows(t, db, + `SELECT count(*) FROM community_post_admissions WHERE post_uri = $1 AND community_did = $2`, + uri, "did:plc:accnevercommunity"), + "no admission may be opened for the community an ignored retarget named") +} + +func TestAdmission_FailedWithdrawalIsRetriedOnRedelivery(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + + sweep := &flakyDeleter{failures: 1, err: errors.New("the community's PDS is briefly unreachable")} + f := newAccFixture(t, db) + f.consumer = NewPostEventConsumer( + postgres.NewPostRepository(db), postgres.NewCommunityRepository(db), + newMockUserService(), db, + WithAdmissions(f.admissions), + WithDeletedAccounts(postgres.NewDeletedAccountRepository(db)), + WithAcceptanceCleanup(sweep), + ) + + base := time.Now().UnixMicro() + const cid = "bafyreiwithdrawflake" + rkey := "withdrawflake" + uri := f.indexPV2(t, rkey, cid, base) + + acceptanceRkey := posts.SubjectRkey(uri) + _, err := f.admissions.ApplyAcceptance(ctx, posts.ApplyAcceptanceCommand{ + CommunityDID: accCommunity, + PostURI: uri, + AcceptanceURI: "at://" + accCommunity + "/" + posts.AcceptanceCollection + "/" + acceptanceRkey, + AcceptanceRkey: acceptanceRkey, + PinnedCID: cid, + Watermark: posts.CommunityWatermark{Rev: testkit.TID()}, + }) + require.NoError(t, err) + + rev := testkit.TID() + tombstone := pv2Event(accAuthor, "delete", rkey, rev, "", base+1_000_000, nil) + + // First delivery: the tombstone lands (correctly — the author's deletion is + // the local truth and must not be held hostage by a remote PDS) and the + // withdrawal fails. The failure is swallowed so the event is not + // dead-lettered, which is also right, because the redrive would be rejected + // by the rev gate anyway. + require.NoError(t, f.consumer.HandleEvent(ctx, tombstone)) + _, _, _, _, deletedAt := readPV2Post(t, db, uri) + require.NotNil(t, deletedAt, "fixture: the tombstone must land regardless of the sweep") + require.Len(t, sweep.calls, 1, "fixture: the first withdrawal must have been attempted") + + // The redelivery, with the same rev. Today the gate skip returns before the + // sweep is reconsidered, so the acceptance is left standing in the + // community's repo pointing at a record nobody can fetch — permanently, + // because nothing else revisits it. The community's CAR, the thing its + // portability argument rests on, now cites content the author withdrew. + require.NoError(t, f.consumer.HandleEvent(ctx, tombstone)) + + assert.Greaterf(t, len(sweep.calls), 1, + "a withdrawal that failed was never retried (%d attempts). The tombstone is gated so it happens once; the "+ + "withdrawal is idempotent and must be reconsidered on every delivery, or a single unreachable PDS leaves "+ + "the acceptance standing forever with nothing scheduled to fix it", len(sweep.calls)) + assert.Equal(t, uri, sweep.calls[len(sweep.calls)-1].PostURI, "the retry must name the same subject") +} diff --git a/internal/atproto/jetstream/erasure_integrity_test.go b/internal/atproto/jetstream/erasure_integrity_test.go new file mode 100644 index 0000000..9fcb154 --- /dev/null +++ b/internal/atproto/jetstream/erasure_integrity_test.go @@ -0,0 +1,199 @@ +//go:build integration + +package jetstream + +import ( + "context" + "database/sql" + "errors" + "testing" + "time" + + "Coves/internal/core/posts" + "Coves/internal/core/users" + "Coves/internal/db/postgres" + "Coves/tests/fixtures" + "Coves/tests/testkit" + + _ "github.com/lib/pq" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// What the migration-036 erasure marker has to survive. +// +// The marker is the only thing that distinguishes "this account was erased on +// purpose" from "this account has never been indexed" — and under author-owned +// posts the second is a NORMAL state that must index freely (§5.3). So every +// path that can clear the marker, or that can act without consulting it, is a +// way for erased content to come back. +// +// The three below are the ones the review found, and they fail differently: +// +// - The marker's EXIT is too wide. Re-registration must clear it, and does; +// but so does any firehose-driven index of the same DID, which is not a +// registration at all — it is a stranger's repo emitting a profile event +// for a DID this AppView was asked to forget. +// - The GATE has a survivor. postv2 events are checked; a replayed acceptance +// is a second door into the same admissions table. +// - The LOOKUP fails open. "I could not read the marker table" and "there is +// no marker" must not be the same answer, because the second one indexes. + +// erasedFixture is a consumer wired with the real erasure lookup plus the real +// users service, so the marker's exit is exercised through the same statement +// production uses rather than through a fake. +type erasedFixture struct { + consumer *PostEventConsumer + userService users.UserService + admissions posts.AdmissionRepository +} + +func newErasedFixture(t *testing.T, db *sql.DB, opts ...PostEventConsumerOption) erasedFixture { + t.Helper() + + insertBridgedUser(t, db, accAuthor, "erasureowner.test") + insertBridgedCommunity(t, db, accCommunity, "erasurecommunity.test", accAuthor) + + admissions := postgres.NewAdmissionRepository(db) + wired := append([]PostEventConsumerOption{ + WithAdmissions(admissions), + WithDeletedAccounts(postgres.NewDeletedAccountRepository(db)), + }, opts...) + + return erasedFixture{ + consumer: NewPostEventConsumer( + postgres.NewPostRepository(db), postgres.NewCommunityRepository(db), + newMockUserService(), db, wired...), + userService: users.NewUserService(postgres.NewUserRepository(db), nil, bskySocialPDS, nil, ""), + admissions: admissions, + } +} + +// failingErasureLookup is a DeletedAccountLookup whose table cannot be read. +type failingErasureLookup struct{ err error } + +func (l failingErasureLookup) IsAccountDeleted(context.Context, string) (bool, error) { + return false, l.err +} + +func TestErasure_FirehoseIndexingDoesNotClearTheMarker(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + f := newErasedFixture(t, db) + + erased := fixtures.DID("erased" + testkit.UniqueID(t)) + markAccountDeleted(t, db, erased) + + // IndexUser is what the firehose calls. It is NOT a registration: the DID + // appearing in a profile or identity event means only that some repo + // somewhere emitted a record — a bridge, a replay, an overlapping feed — + // and any of those can arrive months after the account was erased. + // + // The marker's exit is re-registration, and re-registration is a person + // deliberately signing up again through social.coves.actor.signup. Clearing + // it here makes the erasure undone by the very replays it exists to defend + // against, and undone SILENTLY: the users row reappears, the marker is + // gone, and the next replayed post indexes normally. + err := f.userService.IndexUser(ctx, erased, "erased.test", bskySocialPDS) + + var markers int + require.NoError(t, db.QueryRow(`SELECT count(*) FROM deleted_accounts WHERE did = $1`, erased).Scan(&markers)) + assert.Equalf(t, 1, markers, + "a firehose-driven IndexUser cleared the erasure marker for %s. The marker's only exit is a genuine "+ + "re-registration; clearing it here means any repo on the network can un-erase an account by emitting one "+ + "record naming its DID (IndexUser returned: %v)", erased, err) + + var indexed int + require.NoError(t, db.QueryRow(`SELECT count(*) FROM users WHERE did = $1`, erased).Scan(&indexed)) + assert.Zerof(t, indexed, + "the erased account was re-indexed from the firehose; the row the deletion removed is back") +} + +func TestErasure_RegistrationStillClearsTheMarker(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + repo := postgres.NewUserRepository(db) + + // THE OTHER HALF, and it has to be pinned beside the first or the fix has an + // obvious wrong shape available: never clearing the marker at all. A person + // who deletes their account and later signs up again on the same DID must be + // able to, and a marker left standing would make every post they write + // disappear with nothing to point at. + returning := fixtures.DID("returning" + testkit.UniqueID(t)) + markAccountDeleted(t, db, returning) + + _, err := repo.Create(ctx, &users.User{ + DID: returning, Handle: "returning.test", PDSURL: bskySocialPDS, + }) + require.NoError(t, err, "a deleted DID must be able to register again") + + var markers int + require.NoError(t, db.QueryRow(`SELECT count(*) FROM deleted_accounts WHERE did = $1`, returning).Scan(&markers)) + assert.Zero(t, markers, "a genuine registration is the marker's exit") +} + +func TestErasure_AcceptanceForASweptAuthorIsGated(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + f := newErasedFixture(t, db) + + // The mutation survivor. postv2 events are gated, so the obvious door is + // shut — but an acceptance names the post by URI, and the URI carries the + // author's DID, so a replayed acceptance can recreate the admission row for + // an erased account's post without any postv2 event being involved at all. + erased := fixtures.DID("sweptauthor" + testkit.UniqueID(t)) + markAccountDeleted(t, db, erased) + + uri := "at://" + erased + "/" + PostV2Collection + "/" + testkit.TID() + + err := f.consumer.HandleEvent(ctx, acceptanceEvent(accCommunity, uri, "bafyreiswept", testkit.TID(), time.Now().UnixMicro())) + + require.NoError(t, err, + "an acceptance about an erased account's post must be DROPPED, not refused: refusing it dead-letters an event "+ + "that will be replayed and dropped identically forever") + assert.Zerof(t, countRows(t, db, + `SELECT count(*) FROM community_post_admissions WHERE post_uri = $1`, uri), + "the acceptance recreated the admission row that the account deletion swept — which is the exact resurrection "+ + "migration 036 exists to prevent, arriving through the door the postv2 gate does not cover") + assert.Zero(t, countRows(t, db, `SELECT count(*) FROM posts WHERE uri = $1`, uri), + "and nothing may be indexed for it either") +} + +func TestErasure_AnUnreadableMarkerTableFailsClosed(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + + lookupErr := errors.New("deleted_accounts is unreachable") + f := newErasedFixture(t, db, WithDeletedAccounts(failingErasureLookup{err: lookupErr})) + + author := fixtures.DID("unknownerasure" + testkit.UniqueID(t)) + rkey := "erasureunknown" + uri := "at://" + author + "/" + PostV2Collection + "/" + rkey + + err := f.consumer.HandleEvent(ctx, pv2Event( + author, "create", rkey, testkit.TID(), "bafyreierasureunknown", time.Now().UnixMicro(), + pv2Record(accCommunity, "indexed while the marker table was down", "body"), + )) + + // "I could not read the marker" and "there is no marker" must never be the + // same answer. Under §5.3 an unknown author indexes freely, so a lookup that + // failed open is indistinguishable from a healthy one — and a database blip + // becomes a window in which every erased account's replayed content is + // re-indexed, permanently, with nothing recording that it happened. + require.Errorf(t, err, + "a failed erasure lookup was treated as 'not erased' and the post was indexed anyway. The lookup gates a "+ + "deletion guarantee, so it has to fail CLOSED: a transient error retried is a delay, while a fail-open is "+ + "content coming back that somebody asked to have removed") + assert.Zero(t, countRows(t, db, `SELECT count(*) FROM posts WHERE uri = $1`, uri), + "nothing may be indexed while the gate cannot be consulted") + assert.NotErrorIs(t, err, ErrPermanentEvent, + "an unreachable table is transient: the redrive is what makes failing closed cheap rather than lossy") +} -- 2.51.2 From f4291e18e1248447495f87579a50eaa1b6b52b10 Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 03:21:09 -0700 Subject: [PATCH 11/17] test(ingestion): pin the six adjudicated review items MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 11 new reds; T0 and T1 otherwise clean, mechanical gates green. P7 (contract change, authorized) — TestDecider_FallsToTheStricterClass... replaced by a two-halves pin. The engine now must DEFER on an IsAggregator failure: it stamps the verdict into the admission row with redrivable=false, so a downgrade there is a permanent unretryable refusal for a reason that was never true. The write path keeps the stricter-class fallback, pinned beside it against the composition — CreatePost answers a live client who can retry, and nothing is written down. The comment states that asymmetry as the rule. P1 — jetstream gains a PDS floor (one line, harness_test.go) so the positive case runs against a real repo's real CAR; a hand-built fixture would encode guesses about the verification's internals. Three cases: the endpoint must be sync.getRecord, a PDS echoing the pinned CID over substituted content must be refused (RED — believed today), and the honest recomputation must still converge (green, the positive control). P5 — quota over the real admissions table, counting accepted+pending for (author, community). The ledger cannot serve this: post_submissions is written by CreatePost, so a firehose author has no row and never will — counting it would limit local users and exempt the remote ones §8 is about. New AdmissionRepository.CountRecentAdmissions (stubbed). P6 — no schema change. Over-fetch past a backed-off prefix (RED: a stuck community at the head of the backlog starves everything behind it), plus pruning deferrals for subjects that leave the backlog, via a new QueueSnapshot.DeferredSubjects. P11 — seam added as confirmed: ssrfSafeTransport grows a lookupIP field so a test can make the second resolution disagree with the first. RED with the real diagnosis in the output: the base transport re-resolved the hostname independently. The private-answer guard is pinned separately so closing the window cannot widen what passes through it. P12 — 60/min own limiter declared in the routes table; 2048-byte DID accepted, malformed still 400, Cache-Control: no-store. Guards passing deliberately: the write-path stricter-class half, the direct-fetch positive control, the quota's window-roll / refusals-are-free / trusted-exempt cases, and the SSRF private-answer refusal. Co-Authored-By: Claude Fable 5 --- .../post/getstatus_integration_test.go | 99 ++++++++ internal/api/routes/registration_test.go | 10 +- .../direct_fetch_verification_test.go | 166 ++++++++++++ internal/atproto/jetstream/harness_test.go | 8 +- internal/atproto/oauth/transport.go | 14 +- .../atproto/oauth/transport_toctou_test.go | 142 +++++++++++ internal/core/posts/admissions.go | 16 ++ internal/core/posts/decider.go | 14 ++ internal/core/posts/decider_quota_test.go | 237 ++++++++++++++++++ internal/core/posts/decider_test.go | 68 ++++- internal/core/posts/engine_matrix_test.go | 5 + internal/core/posts/queue.go | 10 + internal/core/posts/queue_test.go | 102 ++++++++ internal/db/postgres/admission_queue_repo.go | 14 ++ 14 files changed, 889 insertions(+), 16 deletions(-) create mode 100644 internal/atproto/jetstream/direct_fetch_verification_test.go create mode 100644 internal/atproto/oauth/transport_toctou_test.go create mode 100644 internal/core/posts/decider_quota_test.go diff --git a/internal/api/handlers/post/getstatus_integration_test.go b/internal/api/handlers/post/getstatus_integration_test.go index bde4523..aabdd3a 100644 --- a/internal/api/handlers/post/getstatus_integration_test.go +++ b/internal/api/handlers/post/getstatus_integration_test.go @@ -9,6 +9,7 @@ import ( "net/http" "net/http/httptest" "net/url" + "strings" "testing" "time" @@ -431,3 +432,101 @@ func TestGetStatus_ScopesTheAnswerToTheNamedCommunity(t *testing.T) { "the same post is accepted in one community and removed in another; an answer that ignored the community parameter would report one of them everywhere") assert.Equal(t, string(posts.DecisionOffTopic), removed["decisionCode"]) } + +func TestGetStatus_AcceptsALegalLongDID(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + stack := newStatusStack(db) + + // A DID may legally run to 2048 bytes (the same fact that killed the + // readable-rkey transform in PRD rev 2.2 and forced the SHA-256 digest). An + // author-owned post URI is authority-scoped, so the author's DID is INSIDE + // the URI this endpoint takes — which means a length cap sized for the old + // community-repo URIs silently makes long-DID authors unqueryable, and only + // them. Nothing else in the system would notice: their posts index fine and + // every other endpoint serves them. + name := testkit.UniqueIDWithPrefix(t, "longdid") + communityDID, err := fixtures.Community(ctx, db, name, "owner"+name) + require.NoError(t, err) + + longDID := "did:web:" + strings.Repeat("a", 2048-len("did:web:")) + require.Len(t, longDID, 2048, "fixture: the DID must be exactly at the legal ceiling") + + rkey := testkit.TID() + postURI := "at://" + longDID + "/social.coves.community.postv2/" + rkey + _, err = db.ExecContext(ctx, ` + INSERT INTO posts (uri, cid, rkey, author_did, community_did, title, created_at) + VALUES ($1, $2, $3, $4, $5, $6, NOW()) + `, postURI, "bafyreilongdid", rkey, longDID, communityDID, "a post by an author with a very long DID") + require.NoError(t, err) + + _, err = stack.admissions.UpsertPending(ctx, posts.UpsertPendingCommand{ + CommunityDID: communityDID, + PostURI: postURI, + EvaluatedCID: "bafyreilongdid", + }) + require.NoError(t, err) + + body := decodeStatus(t, getStatus(t, stack.handler, postURI, communityDID)) + assert.Equal(t, "pending", body["status"], + "a legal 2048-byte DID must be queryable; a cap below the spec's ceiling excludes real authors from the only "+ + "endpoint that can tell them why their post is not visible") +} + +func TestGetStatus_RefusesAMalformedURI(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + stack := newStatusStack(db) + subject := newStatusSubject(t, db) + + // Raising the cap must not become "accept anything long". The URI is parsed + // — a handle-authority URI resolves to whoever holds the handle next, and a + // non-at:// string is a client bug that has to come back as one rather than + // as a silent not-found. + for _, malformed := range []string{ + "not-an-at-uri", + "at://", + "https://example.com/post", + "at://" + strings.Repeat("b", 4096) + "/social.coves.community.postv2/x", + } { + rec := httptest.NewRecorder() + target := "/xrpc/social.coves.community.post.getStatus?post=" + + url.QueryEscape(malformed) + "&community=" + url.QueryEscape(subject.CommunityDID) + stack.handler.HandleGetStatus(rec, httptest.NewRequest(http.MethodGet, target, nil)) + + assert.Equalf(t, http.StatusBadRequest, rec.Code, + "the URI %.40q must be refused as malformed, not answered; a 404 here tells a client with a bug that its post "+ + "does not exist (body: %s)", malformed, rec.Body.String()) + } +} + +func TestGetStatus_IsNotCacheable(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + stack := newStatusStack(db) + subject := newStatusSubject(t, db) + + _, err := stack.admissions.UpsertPending(ctx, posts.UpsertPendingCommand{ + CommunityDID: subject.CommunityDID, + PostURI: subject.PostURI, + EvaluatedCID: "bafyreicacheable", + }) + require.NoError(t, err) + + rec := getStatus(t, stack.handler, subject.PostURI, subject.CommunityDID) + require.Equal(t, http.StatusOK, rec.Code) + + // This endpoint exists to be POLLED for a transition (§7), so a cached + // answer is not a stale nicety — it is the endpoint failing at its only job: + // the client keeps being handed `pending` after the post was accepted and + // stops polling. It is also unauthenticated and reports a moderation + // decision, so an intermediary holding a copy is a disclosure surface that + // outlives the request. + assert.Equalf(t, "no-store", rec.Header().Get("Cache-Control"), + "getStatus must answer Cache-Control: no-store; got %q", rec.Header().Get("Cache-Control")) +} diff --git a/internal/api/routes/registration_test.go b/internal/api/routes/registration_test.go index 5815685..5713051 100644 --- a/internal/api/routes/registration_test.go +++ b/internal/api/routes/registration_test.go @@ -192,7 +192,15 @@ var declaredRoutes = []declaredRoute{ // 2.7: anyone who can name a post URI learns its status in a community. // Adding OptionalAuth here would be harmless; adding RequireAuth would make // the cross-server case unanswerable, which is why it is declared. - {http.MethodGet, "/xrpc/social.coves.community.post.getStatus", authNone, 0, false}, + // getStatus carries its OWN limiter, tighter than the global 100/minute, + // and it is the only unauthenticated route in the product that does. Two + // things make it worth the exception: §7's client UX is to POLL it until a + // post flips to accepted, so the honest traffic shape is repeated requests + // from one caller; and because it takes no auth, an unauthenticated + // stranger can ask about any post URI they can name. The budget is what + // bounds enumeration of a community's rejected posts to something an + // operator would notice. + {http.MethodGet, "/xrpc/social.coves.community.post.getStatus", authNone, 60, false}, // RegisterVoteRoutes — social.coves.feed.vote.* {http.MethodPost, "/xrpc/social.coves.feed.vote.create", authRequired, 0, false}, diff --git a/internal/atproto/jetstream/direct_fetch_verification_test.go b/internal/atproto/jetstream/direct_fetch_verification_test.go new file mode 100644 index 0000000..84c5885 --- /dev/null +++ b/internal/atproto/jetstream/direct_fetch_verification_test.go @@ -0,0 +1,166 @@ +//go:build integration + +package jetstream + +import ( + "context" + "net/http" + "net/http/httptest" + "testing" + "time" + + "Coves/internal/atproto/identity" + "Coves/internal/db/postgres" + "Coves/tests/testkit" + + _ "github.com/lib/pq" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// How the §5.4 direct fetch decides the bytes it got are the bytes the +// acceptance pinned. +// +// # THE ENVELOPE IS NOT EVIDENCE +// +// com.atproto.repo.getRecord answers with JSON: {"uri": ..., "cid": ..., "value": +// {...}}. The `cid` in it is a CLAIM BY THE SERVER, and the server is the +// author's PDS — chosen by a DID document, reached because a stranger wrote an +// acceptance record naming that subject. Comparing the pinned CID against that +// field asks the attacker whether the attacker is lying. +// +// The consequence is the worst one available in this design. The AppView indexes +// whatever `value` contains, under a community's SIGNED acceptance of a CID that +// content does not have. The community attested to one thing; every reader is +// shown another; and the attestation is what the whole trust model rests on. +// +// com.atproto.sync.getRecord answers with a CAR instead — the actual repo blocks +// — so the CID can be RECOMPUTED from the bytes rather than read off a label. A +// server cannot lie about a hash of what it just sent. +// +// # WHY THE POSITIVE CASE NEEDS A REAL REPO +// +// The negatives below are servable by hand. The positive is not: a CAR carries a +// commit root and the MST blocks proving the record's membership, and a fixture +// built here would encode this file's guesses about how the verification walks +// them — passing or failing for reasons unrelated to the property. So the +// positive drives the fetch against a record genuinely written to a genuine repo +// on the test PDS, which is why this package now carries a PDS floor +// (harness_test.go). + +// pinnedResolver points the fetcher at one PDS for one DID. +func pinnedResolver(did, pdsURL string) identity.Resolver { + return &mockIdentityResolverForUser{identities: map[string]*identity.Identity{ + did: {DID: did, Handle: "verify.test", PDSURL: pdsURL}, + }} +} + +func TestDirectFetch_RecomputesTheCIDFromARealRepo(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + pdsServer := testkit.NewPDS(t) + + // A real account, a real record, a real commit. The CID the PDS reports is + // one it derived from bytes it stored, so a verifier that recomputes and one + // that trusts the label agree here — which is exactly what makes this the + // positive control for the negatives below. + author := pdsServer.CreateAccount(t, testkit.WithHandlePrefix("vf")) + insertBridgedUser(t, db, accAuthor, "verifyowner.test") + insertBridgedCommunity(t, db, accCommunity, "verifycommunity.test", accAuthor) + + record := author.CreateRecord(t, PostV2Collection, map[string]any{ + "$type": PostV2Collection, + "community": accCommunity, + "title": "written into a real repo", + "content": "bytes the PDS actually holds", + "createdAt": time.Now().UTC().Format(time.RFC3339), + }) + + fetcher := NewDevDirectPostFetcher(pinnedResolver(author.DID, pdsServer.URL())) + consumer := NewPostEventConsumer( + postgres.NewPostRepository(db), postgres.NewCommunityRepository(db), + newMockUserService(), db, + WithAdmissions(postgres.NewAdmissionRepository(db)), + WithDeletedAccounts(postgres.NewDeletedAccountRepository(db)), + WithPostRecordFetcher(fetcher), + ) + + require.NoError(t, consumer.HandleEvent(ctx, + acceptanceEvent(accCommunity, record.URI, record.CID, testkit.TID(), time.Now().UnixMicro())), + "a record whose recomputed CID matches the pin must converge; if this fails while the negatives pass, the "+ + "verification is refusing everything rather than verifying anything") + + _, _, storedCID, _, _ := readPV2Post(t, db, record.URI) + assert.Equal(t, record.CID, storedCID, "the indexed CID must be the one the repo minted") + + row, err := postgres.NewAdmissionRepository(db).Get(ctx, accCommunity, record.URI) + require.NoError(t, err) + assert.Equal(t, "accepted", string(row.Status)) +} + +func TestDirectFetch_UsesSyncGetRecordNotTheJSONEnvelope(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + uri := accPostURI("verifyendpoint") + + var paths []string + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + paths = append(paths, r.URL.Path) + w.WriteHeader(http.StatusNotFound) + })) + defer srv.Close() + + f := newAccFixture(t, db, WithPostRecordFetcher( + NewDevDirectPostFetcher(pinnedResolver(accAuthor, srv.URL)))) + + _ = f.consumer.HandleEvent(context.Background(), + acceptanceEvent(accCommunity, uri, "bafyreiverifyendpoint", testkit.TID(), time.Now().UnixMicro())) + + require.NotEmpty(t, paths, "the fetch must have reached the PDS at all") + assert.Containsf(t, paths, "/xrpc/com.atproto.sync.getRecord", + "the fetch asked %v. repo.getRecord returns a server-authored JSON envelope whose `cid` is a claim; "+ + "sync.getRecord returns the repo's own blocks, which is the only answer a hostile PDS cannot fabricate", paths) + assert.NotContainsf(t, paths, "/xrpc/com.atproto.repo.getRecord", + "the fetch still used repo.getRecord (%v); as long as it does, the CID check compares the pin against a number "+ + "the same server chose", paths) +} + +func TestDirectFetch_RefusesAPDSThatLiesAboutTheCID(t *testing.T) { + t.Parallel() + + db := testkit.DB(t) + ctx := context.Background() + uri := accPostURI("verifyliar") + const pinned = "bafyreiverifypinnedversion" + + // THE ATTACK, in its simplest form: the server echoes the pinned CID back + // and serves whatever content it likes underneath. Against an envelope- + // trusting verifier this succeeds completely and silently — the post indexes, + // the admission goes to accepted, and the community's signed acceptance now + // covers content it never saw. + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + serveRecord(t, w, uri, pinned, + pv2Record(accCommunity, "content the community never evaluated", "substituted after acceptance")) + })) + defer srv.Close() + + f := newAccFixture(t, db, WithPostRecordFetcher( + NewDevDirectPostFetcher(pinnedResolver(accAuthor, srv.URL)))) + + err := f.consumer.HandleEvent(ctx, + acceptanceEvent(accCommunity, uri, pinned, testkit.TID(), time.Now().UnixMicro())) + + require.Error(t, err, + "a PDS claiming the pinned CID over arbitrary content was believed. Nothing about that response is verifiable: "+ + "the CID is a field the same server wrote, so trusting it lets the author's host substitute any content under "+ + "the community's signed attestation") + assert.Zero(t, countRows(t, db, `SELECT count(*) FROM posts WHERE uri = $1`, uri), + "the substituted content must not be indexed") + assert.Zero(t, countRows(t, db, + `SELECT count(*) FROM community_post_admissions WHERE post_uri = $1 AND status = 'accepted'`, uri), + "and no acceptance may be recorded for it") +} diff --git a/internal/atproto/jetstream/harness_test.go b/internal/atproto/jetstream/harness_test.go index 8f9fd6b..e4381e1 100644 --- a/internal/atproto/jetstream/harness_test.go +++ b/internal/atproto/jetstream/harness_test.go @@ -14,6 +14,12 @@ import ( // It lives in a tagged file because a TestMain applies to the whole test // binary: the untagged unit build of this package needs nothing out of // process, and must not be made to probe Postgres before it can run. +// The PDS floor is here for ONE property: §5.4's direct fetch must verify a +// record by recomputing its CID from the bytes the repo actually holds, and the +// only honest source of those bytes is a real repo. A hand-built CAR fixture +// would encode this package's guesses about the verification's internals and +// fail for reasons that have nothing to do with the property — see +// direct_fetch_verification_test.go. func TestMain(m *testing.M) { - os.Exit(testkit.Main(m, testkit.RequirePostgres)) + os.Exit(testkit.Main(m, testkit.RequirePostgres, testkit.RequirePDS)) } diff --git a/internal/atproto/oauth/transport.go b/internal/atproto/oauth/transport.go index a04def1..9e7c463 100644 --- a/internal/atproto/oauth/transport.go +++ b/internal/atproto/oauth/transport.go @@ -11,6 +11,18 @@ import ( type ssrfSafeTransport struct { base *http.Transport allowPrivate bool // For dev/testing only + + // lookupIP resolves a hostname. A field so a test can drive the + // check-then-dial window that the guard has to close; nil means net.LookupIP. + lookupIP func(host string) ([]net.IP, error) +} + +// resolveHost is the transport's one name lookup per request. +func (t *ssrfSafeTransport) resolveHost(host string) ([]net.IP, error) { + if t.lookupIP != nil { + return t.lookupIP(host) + } + return net.LookupIP(host) } // isPrivateIP checks if an IP is in a private/reserved range @@ -54,7 +66,7 @@ func (t *ssrfSafeTransport) RoundTrip(req *http.Request) (*http.Response, error) host := req.URL.Hostname() // Resolve hostname to IP - ips, err := net.LookupIP(host) + ips, err := t.resolveHost(host) if err != nil { return nil, fmt.Errorf("failed to resolve host: %w", err) } diff --git a/internal/atproto/oauth/transport_toctou_test.go b/internal/atproto/oauth/transport_toctou_test.go new file mode 100644 index 0000000..a8b5731 --- /dev/null +++ b/internal/atproto/oauth/transport_toctou_test.go @@ -0,0 +1,142 @@ +package oauth + +import ( + "net" + "net/http" + "net/http/httptest" + "strings" + "sync" + "testing" +) + +// The window between checking an address and connecting to it. +// +// # WHY A SECOND LOOKUP IS A SECOND DECISION +// +// RoundTrip resolves the hostname, walks the answers, and refuses the request if +// any of them is private. Then it hands the URL — the HOSTNAME, not the vetted +// address — to the base transport, which resolves it AGAIN before dialling. +// Nothing binds the second answer to the first. +// +// That is a real capability, not a theoretical one, and DNS rebinding is the +// name for it: an attacker controls the zone, answers the first query with a +// public address and the second with 169.254.169.254, and the guard approves a +// host it never connects to. Every input that reaches this transport is chosen +// by a stranger — a DID document's PDS endpoint, an acceptance record's subject +// — so the attacker also picks when to flip. +// +// The fix is to stop passing a name to the dialler. Vet the addresses once, then +// dial ONE OF THE VETTED ADDRESSES, ignoring the hostname at connect time. +// +// This is asserted through the lookup seam rather than against real DNS, because +// the property is precisely about the SECOND resolution: a test that could not +// make the two answers differ could not tell a fixed transport from a broken one. + +// flippingResolver answers safely the first time and privately afterwards — +// the smallest rebinding attack there is. +type flippingResolver struct { + mu sync.Mutex + safe net.IP + private net.IP + calls int +} + +func (r *flippingResolver) lookup(string) ([]net.IP, error) { + r.mu.Lock() + defer r.mu.Unlock() + r.calls++ + if r.calls == 1 { + return []net.IP{r.safe}, nil + } + return []net.IP{r.private}, nil +} + +func (r *flippingResolver) lookups() int { + r.mu.Lock() + defer r.mu.Unlock() + return r.calls +} + +func TestSSRFTransport_DialsOnlyTheAddressItVetted(t *testing.T) { + // The "safe" address is the loopback the test server actually listens on. + // Using a real listener means a transport that dials the vetted address + // genuinely connects, so the test distinguishes "refused" from "connected to + // the right place" rather than only observing failures. + var reached bool + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + reached = true + w.WriteHeader(http.StatusOK) + })) + defer server.Close() + + host, port, err := net.SplitHostPort(strings.TrimPrefix(server.URL, "http://")) + if err != nil { + t.Fatalf("splitting the test server address: %v", err) + } + safe := net.ParseIP(host) + if safe == nil { + t.Fatalf("the test server address %q is not an IP", host) + } + + resolver := &flippingResolver{safe: safe, private: net.ParseIP("169.254.169.254")} + + // allowPrivate is TRUE, which looks backwards and is the only way to state + // the property in isolation. With the guard enabled, loopback is refused on + // the first lookup and the test can never reach the second one — so it would + // pass against a transport with the bug still in it. Disabling the private + // check leaves exactly one thing under test: whether the address that was + // vetted is the address that gets dialled. + client := NewSSRFSafeHTTPClient(true) + transport, ok := client.Transport.(*ssrfSafeTransport) + if !ok { + t.Fatalf("NewSSRFSafeHTTPClient must install an ssrfSafeTransport, got %T", client.Transport) + } + transport.lookupIP = resolver.lookup + + resp, err := client.Get("http://rebind.test:" + port + "/") + if err == nil { + defer func() { _ = resp.Body.Close() }() + } + + if !reached { + t.Fatalf("the request never reached the address that was vetted (lookups: %d, err: %v). "+ + "RoundTrip resolved the hostname, approved the answer, and then handed the NAME to the base transport, "+ + "which resolved it again and dialled somewhere else — the approval described a host the connection never went to", + resolver.lookups(), err) + } + + // One lookup, and that is the assertion rather than an optimisation note: a + // second resolution IS the vulnerability, because nothing constrains it to + // agree with the first. A transport that pins the vetted address has no + // reason to ask again. + if got := resolver.lookups(); got != 1 { + t.Errorf("the hostname was resolved %d times; the dial must reuse the address RoundTrip already vetted, "+ + "since any later answer is one the guard never saw", got) + } +} + +func TestSSRFTransport_StillRefusesAPrivateFirstAnswer(t *testing.T) { + // The guard the pin above deliberately switches off, asserted on its own so + // that closing the rebinding window cannot quietly widen what is allowed + // through it. + resolver := &flippingResolver{ + safe: net.ParseIP("169.254.169.254"), + private: net.ParseIP("169.254.169.254"), + } + + client := NewSSRFSafeHTTPClient(false) + transport, ok := client.Transport.(*ssrfSafeTransport) + if !ok { + t.Fatalf("NewSSRFSafeHTTPClient must install an ssrfSafeTransport, got %T", client.Transport) + } + transport.lookupIP = resolver.lookup + + resp, err := client.Get("http://metadata.test/") + if err == nil { + _ = resp.Body.Close() + t.Fatal("a hostname resolving to a link-local address must be refused") + } + if !strings.Contains(err.Error(), "SSRF blocked") { + t.Errorf("the refusal must name the guard that made it, got: %v", err) + } +} diff --git a/internal/core/posts/admissions.go b/internal/core/posts/admissions.go index 630398d..30ee89b 100644 --- a/internal/core/posts/admissions.go +++ b/internal/core/posts/admissions.go @@ -366,6 +366,22 @@ type AdmissionRepository interface { // honest signal: a backlog full of subjects nothing can ever settle looks // identical, from the outside, to an engine that has stopped working. ListPendingSubjects(ctx context.Context, limit int) ([]PendingSubject, error) + + // CountRecentAdmissions counts how many posts this author has had ADMITTED + // to this community since a point in time — accepted and still-pending rows + // together. + // + // It is the firehose path's quota substrate, and it has to be a different + // one from the write path's. post_submissions (migration 035) is written by + // CreatePost, so it only ever sees submissions this AppView handled; a post + // that arrived over the firehose from an author on another server has no + // ledger row and never will. Counting the ledger would therefore apply the + // quota to local users and exempt precisely the remote ones §8 is about. + // + // Rejected and removed rows are excluded deliberately: §8 is explicit that a + // refusal consumes no quota, and counting them would let an author extend + // their own lockout by continuing to post. + CountRecentAdmissions(ctx context.Context, communityDID, authorDID string, since time.Time) (int, error) } // PendingSubject is one (community, post) pair the engine still owes a decision. diff --git a/internal/core/posts/decider.go b/internal/core/posts/decider.go index 6700d33..432834c 100644 --- a/internal/core/posts/decider.go +++ b/internal/core/posts/decider.go @@ -7,6 +7,7 @@ import ( "log" "os" "strings" + "time" ) // The production AdmissionDecider: the adapter that turns "decide about this @@ -75,6 +76,11 @@ type PostLookup interface { GetByURI(ctx context.Context, uri string) (*Post, error) } +// AdmissionCounter is the narrow slice of AdmissionRepository the quota needs. +type AdmissionCounter interface { + CountRecentAdmissions(ctx context.Context, communityDID, authorDID string, since time.Time) (int, error) +} + // AggregatorLookup reports whether a DID is a registered aggregator. Satisfied // by aggregators.Service. type AggregatorLookup interface { @@ -104,8 +110,16 @@ type DeciderDeps struct { Aggregators AggregatorLookup // Policy is the ban lookup, ledger, limits and clock admitPost already uses. + // Its Limits govern the firehose quota below as well, so a local author and + // a remote one are held to the same number. Policy AdmissionPolicy + // Admissions counts an author's recent admitted posts in a community — the + // firehose path's quota substrate (§8). nil disables the quota, which is + // the right default for a deployment that hosts no communities and therefore + // decides nothing. + Admissions AdmissionCounter + // TrustedAggregatorDIDs is the set from TRUSTED_AGGREGATOR_DIDS, resolved // ONCE at construction rather than read per decision. // diff --git a/internal/core/posts/decider_quota_test.go b/internal/core/posts/decider_quota_test.go new file mode 100644 index 0000000..d4a4018 --- /dev/null +++ b/internal/core/posts/decider_quota_test.go @@ -0,0 +1,237 @@ +//go:build integration + +package posts_test + +import ( + "context" + "testing" + "time" + + "Coves/internal/core/communities" + "Coves/internal/core/posts" + "Coves/internal/db/postgres" + "Coves/tests/fixtures" + "Coves/tests/testkit" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// The firehose path's submission quota (§8). +// +// # WHY THE LEDGER CANNOT SERVE THIS +// +// admitPost's quota counts post_submissions (migration 035), a table CreatePost +// writes. That works for the write path and is structurally blind on the +// ingestion path: a post from an author on another server arrives over the +// firehose, has no ledger row, and never will. Counting the ledger here would +// hold LOCAL users to the limit and exempt precisely the remote ones the limit +// exists for — anyone can write unlimited postv2 records naming any community, +// and §8's answer is that the admission layer absorbs them. +// +// So the engine counts what it can actually see: the admission rows this +// community already holds for this author. Accepted and pending together, over +// the same rolling window, refusing with the same code. +// +// This runs against the real repository rather than a counting fake, because +// the query is where the two subtle parts live — matching the author out of the +// AT-URI's authority (the admission may have no posts row to join to yet, §5.4) +// and excluding rejected and removed rows, since §8 is explicit that a refusal +// consumes no quota and counting refusals would let an author extend their own +// lockout by continuing to post. + +const quotaLimit = 3 + +// quotaCommunities resolves one community and nothing else. +type quotaCommunities struct{ community *communities.Community } + +func (q *quotaCommunities) ResolveCommunityIdentifier(context.Context, string) (string, error) { + return q.community.DID, nil +} + +func (q *quotaCommunities) GetByDID(context.Context, string) (*communities.Community, error) { + return q.community, nil +} + +// quotaBans reports no membership, which is the ordinary case: posting in a +// public community has never required joining it. +type quotaBans struct{} + +func (quotaBans) GetMembership(context.Context, string, string) (*communities.Membership, error) { + return nil, communities.ErrMembershipNotFound +} + +// quotaPosts serves whichever post the subject names. +type quotaPosts struct{ posts map[string]*posts.Post } + +func (q *quotaPosts) GetByURI(_ context.Context, uri string) (*posts.Post, error) { + if p, ok := q.posts[uri]; ok { + return p, nil + } + return nil, posts.ErrNotFound +} + +// quotaFixture is the production decider over the real admissions table. +type quotaFixture struct { + decider *posts.AdmissionEngineDecider + admissions posts.AdmissionRepository + postLookup *quotaPosts + communityDID string + authorDID string + now time.Time + trusted map[string]bool +} + +func newQuotaFixture(t *testing.T) *quotaFixture { + t.Helper() + + db := testkit.DB(t) + ctx := context.Background() + + name := testkit.UniqueIDWithPrefix(t, "quota") + communityDID, err := fixtures.Community(ctx, db, name, "owner"+name) + require.NoError(t, err) + + f := "aFixture{ + admissions: postgres.NewAdmissionRepository(db), + postLookup: "aPosts{posts: map[string]*posts.Post{}}, + communityDID: communityDID, + authorDID: fixtures.DID(testkit.UniqueID(t)), + now: time.Date(2026, 8, 8, 12, 0, 0, 0, time.UTC), + trusted: map[string]bool{}, + } + + f.decider = posts.NewAdmissionEngineDecider(posts.DeciderDeps{ + Posts: f.postLookup, + Communities: "aCommunities{community: &communities.Community{DID: communityDID, Visibility: "public"}}, + Admissions: f.admissions, + Policy: posts.AdmissionPolicy{ + Ledger: posts.NewAllowAllAdmissionPolicyForTests().Ledger, + Bans: quotaBans{}, + Limits: posts.SubmissionLimits{ + MaxPerAuthorPerCommunity: quotaLimit, + Window: time.Hour, + DedupeWindow: time.Hour, + }, + Now: func() time.Time { return f.now }, + }, + TrustedAggregatorDIDs: f.trusted, + }) + return f +} + +// admit records one already-admitted post for this author, the way the consumer +// would: an indexed post plus its pending admission row. +func (f *quotaFixture) admit(t *testing.T, label string) string { + t.Helper() + + uri := "at://" + f.authorDID + "/social.coves.community.postv2/" + testkit.TID() + f.postLookup.posts[uri] = &posts.Post{ + URI: uri, CID: "bafyrei" + label, AuthorDID: f.authorDID, CommunityDID: f.communityDID, + } + _, err := f.admissions.UpsertPending(context.Background(), posts.UpsertPendingCommand{ + CommunityDID: f.communityDID, + PostURI: uri, + EvaluatedCID: "bafyrei" + label, + }) + require.NoError(t, err) + return uri +} + +func (f *quotaFixture) decide(t *testing.T, uri string) posts.AdmissionDecision { + t.Helper() + decision, err := f.decider.DecideAdmission(context.Background(), f.communityDID, uri) + require.NoError(t, err) + return decision +} + +func TestDeciderQuota_RefusesThePostPastTheLimit(t *testing.T) { + t.Parallel() + + f := newQuotaFixture(t) + + // The author fills their allowance. Each of these is a post that already + // exists and is already counted — the engine is deciding, not accepting a + // submission, so what bounds them is what the community already holds. + for i := 0; i < quotaLimit; i++ { + uri := f.admit(t, "within") + decision := f.decide(t, uri) + assert.Truef(t, decision.Admitted(), "post %d of %d must be admitted: %+v", i+1, quotaLimit, decision) + } + + over := f.admit(t, "over") + decision := f.decide(t, over) + + assert.Falsef(t, decision.Admitted(), + "the %dth post inside the window was admitted with a limit of %d. Nothing else bounds a firehose author: they "+ + "can write postv2 records naming any community as fast as their PDS accepts them, and §8's answer is that "+ + "the admission layer absorbs it: %+v", quotaLimit+1, quotaLimit, decision) + assert.Equal(t, posts.DecisionRateLimitExceeded, decision.Code, + "the refusal must carry rate-limit-exceeded; it is what getStatus shows the author and what marks the row terminal") +} + +func TestDeciderQuota_TheWindowRolls(t *testing.T) { + t.Parallel() + + f := newQuotaFixture(t) + for i := 0; i < quotaLimit; i++ { + f.decide(t, f.admit(t, "old")) + } + + // Past the window. A quota that never expires is a ban with a different + // name, and the clock is injected precisely so crossing the boundary costs + // no wall time (docs/TEST_ARCHITECTURE.md forbids sleeping for it). + f.now = f.now.Add(2 * time.Hour) + + decision := f.decide(t, f.admit(t, "fresh")) + assert.Truef(t, decision.Admitted(), + "an author whose earlier posts have aged out of the window must be admitted again; a rolling window that never "+ + "releases is a permanent refusal nobody chose: %+v", decision) +} + +func TestDeciderQuota_RefusalsDoNotConsumeQuota(t *testing.T) { + t.Parallel() + + f := newQuotaFixture(t) + for i := 0; i < quotaLimit; i++ { + f.decide(t, f.admit(t, "filled")) + } + + // A refused post, recorded as the engine would record it. + refused := f.admit(t, "refused") + _, err := f.admissions.RecordRejection(context.Background(), posts.RecordRejectionCommand{ + CommunityDID: f.communityDID, + PostURI: refused, + DecisionCode: string(posts.DecisionRateLimitExceeded), + JudgedCID: "bafyreirefused", + Redrivable: false, + }) + require.NoError(t, err) + + // §8: a refusal consumes no quota. Counting rejected rows would let an + // author who keeps posting past their limit extend their own lockout + // indefinitely — each refusal making the next one more certain. + f.now = f.now.Add(2 * time.Hour) + decision := f.decide(t, f.admit(t, "afterrefusal")) + assert.Truef(t, decision.Admitted(), + "a rejected admission was counted against the quota. §8 is explicit that a refusal consumes nothing, or an "+ + "author past their limit extends their own lockout every time they try: %+v", decision) +} + +func TestDeciderQuota_TrustedAggregatorsAreExempt(t *testing.T) { + t.Parallel() + + f := newQuotaFixture(t) + f.trusted[f.authorDID] = true + + // A trusted aggregator has no submission limit today (admit.go's check + // order), and inventing one here would be a silent production behaviour + // change smuggled in under a new code path — the Kagi bridge posts far more + // than any per-author quota would allow. + for i := 0; i < quotaLimit+2; i++ { + decision := f.decide(t, f.admit(t, "bridge")) + assert.Truef(t, decision.Admitted(), + "a trusted aggregator's post %d was refused; the trusted class skips the quota, and applying it would stop "+ + "the bridge dead at the limit: %+v", i+1, decision) + } +} diff --git a/internal/core/posts/decider_test.go b/internal/core/posts/decider_test.go index 3ef13f3..dc078dc 100644 --- a/internal/core/posts/decider_test.go +++ b/internal/core/posts/decider_test.go @@ -250,28 +250,70 @@ func TestDecider_ClassifiesARegisteredAggregator(t *testing.T) { assert.Zero(t, h.bans.calls, "an aggregator is not held to member bans") } -func TestDecider_FallsToTheStricterClassWhenTheLookupFails(t *testing.T) { +func TestDecider_EngineDefersWhenTheClassificationLookupFails(t *testing.T) { t.Parallel() - // The classification lookup is down. There are two ways to be wrong here and - // only one of them is survivable: guessing "aggregator" skips the ban and - // visibility checks, so a database blip would become a window in which every - // banned author's posts are accepted. Guessing "user" costs an aggregator - // some refused posts until the lookup recovers. CreatePost already chose the - // second (service.go step 3), and the engine must not disagree with the write - // path about who someone is. + // THE ENGINE AND THE WRITE PATH ANSWER THIS DIFFERENTLY, ON PURPOSE, and the + // reason is what each does with the answer afterwards. + // + // CreatePost is talking to a live client. A failed IsAggregator lookup there + // downgrades the caller to ActorUser (service.go step 3): the strict checks + // apply, an aggregator loses a few posts until the table comes back, and the + // client is told something it can retry. Nothing is written down. + // + // The engine writes the verdict INTO THE ADMISSION ROW, and a policy refusal + // is stamped redrivable = false — terminal, never revisited by the redrive + // pass. So the same downgrade here does not cost an aggregator a retry; it + // permanently marks their post refused for a reason that was never true, + // because a table was briefly unreachable. Nothing in the system would ever + // look at it again. + // + // That asymmetry is the whole rule: a decision that PERSISTS may only be made + // from an answer that was actually obtained. h := newDeciderHarness() h.aggregators.err = errors.New("aggregators table unreachable") h.bans.membership = banned() decision, err := h.decide(t) - require.NoError(t, err, "a failed CLASSIFICATION is not a failed decision: the stricter class is a safe answer, not an outage") - assert.Falsef(t, decision.Admitted(), - "a failed IsAggregator lookup must fall to ActorUser, and this author is banned — an admission here means the failure was resolved upward into aggregator privileges: %+v", decision) + require.Error(t, err, + "a classification that could not be made must be reported as undecided; the engine has no safe way to guess when its guess is written down permanently") + assert.False(t, decision.Admitted(), "an undecided answer is not an admission") + assert.Emptyf(t, decision.Code, + "the decider minted %q from a failed lookup. The engine persists codes and sets redrivable = false on policy refusals, "+ + "so this would leave a permanent, unretryable refusal on a post whose author may not be banned at all", decision.Code) +} + +func TestDecider_TheWritePathStillFallsToTheStricterClass(t *testing.T) { + t.Parallel() + + // THE OTHER HALF, and it has to stay pinned or the fix above has an obvious + // wrong generalisation available: making a failed lookup undecided + // EVERYWHERE. On the write path that would turn a brief aggregators-table + // blip into a 500 for every caller, where today they are simply held to the + // ordinary user's rules — which is the strict direction and costs nothing. + // + // This asserts the composition rather than re-deriving the classification: + // ActorUser is what service.go step 3 downgrades to, and what must follow + // from it is that the checks a user is held to actually run and actually + // answer. A refusal reaching a live client is retryable; that is precisely + // what makes the guess safe there and unsafe in the engine. + h := newAdmitHarness() + h.bans.membership = banned() + + decision, err := evaluateAdmissionPolicy(context.Background(), h.deps(), AdmissionRequest{ + Actor: ActorUser, + AuthorDID: admitAuthorDID, + Community: admitCommunityHandle, + Fingerprint: "downgraded-classification", + }) + + require.NoError(t, err) + assert.False(t, decision.Admitted(), + "the stricter class must actually apply the checks it implies, or the downgrade is a downgrade in name only") assert.Equal(t, DecisionAuthorBanned, decision.Code, - "falling to the user class means the ban check runs and answers") - assert.Positive(t, h.bans.calls, "the stricter class must actually apply the checks it implies") + "falling to ActorUser means the ban check runs and answers — that is what makes it the SAFE guess for a caller who can retry") + assert.Positive(t, h.bans.calls, "the ban lookup must have been consulted") } func TestDecider_IsUndecidedWhenThePolicyCannotBeEvaluated(t *testing.T) { diff --git a/internal/core/posts/engine_matrix_test.go b/internal/core/posts/engine_matrix_test.go index 987dc15..9feca96 100644 --- a/internal/core/posts/engine_matrix_test.go +++ b/internal/core/posts/engine_matrix_test.go @@ -270,6 +270,11 @@ func (a *fakeAdmissions) ListPendingSubjects(_ context.Context, _ int) ([]Pendin return nil, nil } +func (a *fakeAdmissions) CountRecentAdmissions(_ context.Context, _, _ string, _ time.Time) (int, error) { + a.rec.record("CountRecentAdmissions") + return 0, nil +} + // --------------------------------------------------------------------------- // Harness // --------------------------------------------------------------------------- diff --git a/internal/core/posts/queue.go b/internal/core/posts/queue.go index 340e1be..f8caca0 100644 --- a/internal/core/posts/queue.go +++ b/internal/core/posts/queue.go @@ -100,6 +100,16 @@ type QueueSnapshot struct { LastPassDeferred int LastPassFailed int + + // DeferredSubjects is how many subjects are currently holding a backoff. + // + // Exposed because it is the only external view of a map that would + // otherwise grow for the life of the process: entries are keyed by subject + // and a subject settled by somebody else — the fast path, a firehose + // acceptance, a moderator — stops being listed without ever telling the + // driver to forget it. A number that climbs while the backlog does not is + // the leak, visible. + DeferredSubjects int } // QueueDriverOption configures the driver. diff --git a/internal/core/posts/queue_test.go b/internal/core/posts/queue_test.go index f32f4c5..b4bd57e 100644 --- a/internal/core/posts/queue_test.go +++ b/internal/core/posts/queue_test.go @@ -3,6 +3,7 @@ package posts import ( "context" "errors" + "fmt" "sync" "testing" "time" @@ -332,3 +333,104 @@ func TestQueueDriver_SnapshotReportsTheBacklogAndTheLastPass(t *testing.T) { assert.Equal(t, 1, snapshot.LastPassDeferred) assert.Zero(t, snapshot.LastPassFailed) } + +func TestQueueDriver_OverFetchesPastBackedOffSubjects(t *testing.T) { + t.Parallel() + + // BATCH-PREFIX STARVATION. The backlog is ordered oldest-first, so the + // subjects most likely to be stuck — a community whose credentials expired + // weeks ago, a post whose content never decoded — are exactly the ones that + // sit at the front of it forever. Ask for LIMIT rows, get LIMIT stuck ones, + // skip all of them for backoff, and the pass does nothing. Every pass. The + // queue is not empty and the driver is not broken; it simply never sees past + // its own prefix, and a healthy post behind them is never decided. + // + // The fix stays in the DRIVER: over-fetch and keep skipping until the batch + // is filled. Pushing the backoff into the query would mean persisting + // retry-not-before, and the backoff is deliberately a disposable in-memory + // hint that a restart may forget (see QueueDriver.deferrals). + clock := newQueueClock() + + const batch = 3 + stuck := make([]PendingSubject, batch) + for i := range stuck { + stuck[i] = subject("did:plc:qstuck", fmt.Sprintf("stuck%d", i)) + } + young := subject("did:plc:qyoung", "young") + + subjects := &fakeSubjects{batches: [][]PendingSubject{append(append([]PendingSubject{}, stuck...), young)}} + engine := newFakeEngine() + for _, s := range stuck { + engine.outcomes[s.PostURI] = EngineDeferred + } + + driver := NewQueueDriver(subjects, engine, clock.now(), + WithQueueBatchSize(batch), WithQueueBackoff(time.Minute, 10*time.Minute)) + ctx := context.Background() + + // Pass one settles nothing and backs the whole prefix off. + first, err := driver.RunPass(ctx) + require.NoError(t, err) + require.Equal(t, batch, first.Deferred, "fixture: the whole prefix must defer") + + // Pass two, inside the backoff window. Every stuck subject is held back, so + // a driver that asked for exactly LIMIT rows would process nothing at all. + clock.advance(time.Second) + second, err := driver.RunPass(ctx) + require.NoError(t, err) + + assert.Contains(t, engine.uris(), young.PostURI, + "the young subject was never reached: the pass asked for exactly the batch size, got a prefix of backed-off "+ + "subjects, and skipped all of them — so a stuck community at the head of the backlog starves everything behind it forever") + assert.Equalf(t, 1, second.Processed, + "the pass must fill its batch past the held-back prefix, processing the young subject and nothing else; got %d processed", second.Processed) + + require.GreaterOrEqual(t, len(subjects.limits), 2) + assert.Greaterf(t, subjects.limits[len(subjects.limits)-1], batch, + "the query must be asked for MORE than the batch size (limits seen: %v). The driver cannot skip past what it "+ + "never fetched, and the over-fetch factor is what bounds how deep a stuck prefix it can see past", subjects.limits) +} + +func TestQueueDriver_PrunesDeferralsForSubjectsThatLeaveTheBacklog(t *testing.T) { + t.Parallel() + + // A deferral outlives its subject. The row is settled by somebody else — the + // synchronous fast path, a firehose acceptance, a moderator's removal — and + // it stops being listed, but its entry stays in the map forever. On a busy + // instance that map is then an unbounded leak keyed by every subject the + // driver ever deferred, held for the lifetime of the process. + // + // It is also wrong on re-entry: a subject that leaves the backlog and comes + // back (an edit reopening an accepted post) arrives carrying a stale backoff + // it did nothing to earn, and waits out a delay that was about a completely + // different decision. + clock := newQueueClock() + leaving := subject("did:plc:qleave", "leaving") + staying := subject("did:plc:qstay", "staying") + + subjects := &fakeSubjects{batches: [][]PendingSubject{ + {leaving, staying}, + // Second pass: `leaving` was settled elsewhere and is gone from the + // backlog. `staying` is still owed a decision. + {staying}, + }} + engine := newFakeEngine() + engine.outcomes[leaving.PostURI] = EngineDeferred + engine.outcomes[staying.PostURI] = EngineDeferred + + driver := NewQueueDriver(subjects, engine, clock.now(), WithQueueBackoff(time.Minute, 10*time.Minute)) + ctx := context.Background() + + _, err := driver.RunPass(ctx) + require.NoError(t, err) + require.Equal(t, 2, driver.Snapshot().DeferredSubjects, "fixture: both subjects must be holding a deferral") + + clock.advance(time.Second) + _, err = driver.RunPass(ctx) + require.NoError(t, err) + + assert.Equalf(t, 1, driver.Snapshot().DeferredSubjects, + "the deferral for a subject that has left the backlog was kept. The map is keyed by subject and never swept, so "+ + "it grows for the life of the process — and a subject that comes back (an edit reopening an accepted post) "+ + "inherits a stale backoff it did nothing to earn") +} diff --git a/internal/db/postgres/admission_queue_repo.go b/internal/db/postgres/admission_queue_repo.go index d808657..b232c9a 100644 --- a/internal/db/postgres/admission_queue_repo.go +++ b/internal/db/postgres/admission_queue_repo.go @@ -3,6 +3,7 @@ package postgres import ( "context" "fmt" + "time" "Coves/internal/core/posts" ) @@ -87,3 +88,16 @@ func (r *postgresAdmissionRepo) ListPendingSubjects(ctx context.Context, limit i } return subjects, nil } + +// CountRecentAdmissions counts this author's admitted posts in one community +// since a point in time. +// +// The author is matched by AT-URI prefix rather than by joining posts: under +// author-owned posts the URI's authority IS the author, and an admission can +// legitimately exist with no posts row at all (an acceptance that arrived before +// its subject). A join would silently exempt exactly those from the quota. +func (r *postgresAdmissionRepo) CountRecentAdmissions( + ctx context.Context, communityDID, authorDID string, since time.Time, +) (int, error) { + return 0, nil +} -- 2.51.2 From 1c7c1a0c78064b282a71d1ac163841307d932c1e Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 03:36:39 -0700 Subject: [PATCH 12/17] test(posts): anchor the quota fixture's clock to the database's MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The fixture pinned its injected clock at a hardcoded 2026-08-08T12:00:00Z while the admission rows it seeds are stamped created_at = NOW() by Postgres — the repository owns that column by design and a test does not get to hand it one. Two clocks, free to disagree. With a one-hour window the count's floor sat at 11:00 on that date while the rows carried the database's real time, so the two overlapped for one hour of one day. Outside it the failure was not merely flaky but CONTRADICTORY: rows stamped later than the injected now stay inside every window the test subsequently advances through, making the refusal pin and TheWindowRolls mutually unsatisfiable — no implementation could turn both green, which is what GREEN hit. Anchoring on time.Now().UTC() puts the injection back where it belongs, on the DELTAS the tests apply. That is the whole point of an injectable clock here: crossing a window boundary must cost no wall time. The comment names the two-clock trap at the field, since the wrong version reads like the suite's injected-clock rule rather than an inversion of it. All four quota tests now pass together against GREEN's in-flight implementation — the refusal pin and TheWindowRolls holding at the same time is the evidence the trap was the blocker. Staged this file alone: GREEN is mid-batch in this worktree and six production files are legitimately uncommitted. Co-Authored-By: Claude Fable 5 --- internal/core/posts/decider_quota_test.go | 24 +++++++++++++++++++++-- 1 file changed, 22 insertions(+), 2 deletions(-) diff --git a/internal/core/posts/decider_quota_test.go b/internal/core/posts/decider_quota_test.go index d4a4018..8913446 100644 --- a/internal/core/posts/decider_quota_test.go +++ b/internal/core/posts/decider_quota_test.go @@ -97,8 +97,28 @@ func newQuotaFixture(t *testing.T) *quotaFixture { postLookup: "aPosts{posts: map[string]*posts.Post{}}, communityDID: communityDID, authorDID: fixtures.DID(testkit.UniqueID(t)), - now: time.Date(2026, 8, 8, 12, 0, 0, 0, time.UTC), - trusted: map[string]bool{}, + // TWO CLOCKS, AND THEY HAVE TO AGREE. The quota compares an injected + // `now` against created_at on the admission rows — and those are stamped + // by POSTGRES, with NOW(), by design: the repository owns that column + // and a test does not get to hand it one. So the injected clock is only + // free to move RELATIVE to the database's, never to be planted somewhere + // else entirely. + // + // A fixed instant looks like the suite's injected-clock rule and inverts + // it. Pinned at a hardcoded 2026-08-08T12:00:00Z with a one-hour window, + // the count's floor is 11:00 that day while the rows carry whatever the + // database's real clock said — so the two only overlap during one hour + // of one day, and worse, they overlap in a way no implementation can + // satisfy: rows stamped hours LATER than the injected now stay inside + // every window the test then advances through, so the refusal pin and + // TheWindowRolls become mutually unsatisfiable. + // + // Anchoring on the real clock keeps the injection where it belongs — on + // the DELTAS the tests apply — which is the whole reason the clock is + // injectable: crossing a window boundary must cost no wall time + // (docs/TEST_ARCHITECTURE.md forbids sleeping for it). + now: time.Now().UTC(), + trusted: map[string]bool{}, } f.decider = posts.NewAdmissionEngineDecider(posts.DeciderDeps{ -- 2.51.2 From c0569a60e7358b4fa61419cfd2d3888ca444f861 Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 03:50:36 -0700 Subject: [PATCH 13/17] =?UTF-8?q?fix(ingestion):=20apply=20the=20review=20?= =?UTF-8?q?batch=20=E2=80=94=20verification,=20durability,=20erasure?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 17 of the 18 review pins are green; two are blocked on test-side contract problems reported to the conductor (below). T0 is fully clean. P1 direct fetch now uses com.atproto.sync.getRecord and RECOMPUTES the CID from the CAR's own record bytes. repo.getRecord answers with a JSON envelope whose `cid` is a claim by the author's PDS — a server chosen by a DID document and reached because a stranger wrote an acceptance — so comparing the pin against it asked the attacker whether the attacker was lying, and a match let the AppView index substituted content under a community's SIGNED acceptance. The reported CID is the computed one, not the label, or the claim would re-enter the comparison it was removed from. P2 lone removal-deletes now apply. Paired commits converge without them (the put outranks its delete), but a moderator withdrawing a removal writes only the delete, and ignoring it left the post `removed` forever while the community's repo no longer said so. The subject is recovered by recomputing SubjectRkey over that community's `removed` rows — bounded and index-backed, since the rkey digest is one-way. P3 converge honours applied=false. The post's own event can land while the fetch is in flight — the normal interleaving, since the fetch exists because the event had not arrived — and UpsertPending is last-write-wins, so running it anyway stamped the fetched CID over a newer one. P4 the rev gate now guards only the posts-row mutation. The two writes have opposite idempotence, and gating them together orphaned admissions: a failed upsert left the post indexed, the gate advanced, and every redelivery skipped. An equal stored rev means a replay (safe to re-run); a greater one means a newer event applied (must not). P5 firehose quota over admission rows, not the ledger. post_submissions is written by CreatePost, so a remote author has no row and never will — counting it would meter local users and exempt the remote ones §8 is for. P6 bounded over-fetch (batch + held deferrals, capped at ×4) so a stuck prefix cannot starve the backlog behind it, plus deferral pruning. P7 a failed classification defers instead of downgrading. The write path keeps the downgrade: it answers a live client who can retry, while the engine stamps redrivable=false and would mark a post permanently refused for a reason that was never true. P8a IndexUser no longer clears the erasure marker. It is the FIREHOSE's door into the users table, so any repo could un-erase an account by emitting one record. Registration still clears it. Gated through an optional ErrasureLookup rather than a UserRepository method, so no double can satisfy it by failing open. P9 an acceptance for a tombstoned post is a skip. P10 immutability now outranks the unknown-community branch, which is transient — so a retarget at a ghost DID used to dead-letter and redrive ten times and could never succeed. P11 SSRF: the vetted addresses ride on the request context and the dialler connects to one of THEM, ignoring the hostname. The base transport used to re-resolve, so the guard approved a host the connection never went to. P12 getStatus: spec-derived length bounds (a 2048-byte DID lives INSIDE an author-scoped URI), AT-URI parsing, Cache-Control: no-store, own 60/min limiter. P13 the acceptance withdrawal is reconsidered on every delivery. Tying it to the gate meant one unreachable PDS left the acceptance standing forever. Also: one shared SSRF-safe client instead of one per fetch, a 5s per-fetch deadline inside it, and a doc-truth pass over the stale RED-STUB paragraphs, the engine's task-5 promises, and the future-tense fast-path claims. go-car is pinned to the version indigo already requires; `go mod tidy` was NOT used (it upgraded transitive deps into a broken go-log). Co-Authored-By: Claude Fable 5 --- go.mod | 8 + go.sum | 16 + internal/api/handlers/post/getstatus.go | 52 ++- internal/api/routes/post.go | 38 +- internal/atproto/jetstream/authorpost.go | 424 +++++++++++++++---- internal/atproto/jetstream/post_consumer.go | 10 +- internal/atproto/oauth/transport.go | 78 +++- internal/core/posts/decider.go | 120 ++++-- internal/core/posts/engine.go | 27 +- internal/core/posts/queue.go | 98 ++++- internal/core/users/interfaces.go | 15 + internal/core/users/service.go | 31 ++ internal/db/postgres/admission_queue_repo.go | 25 +- internal/db/postgres/deleted_account_repo.go | 17 +- 14 files changed, 813 insertions(+), 146 deletions(-) diff --git a/go.mod b/go.mod index b22139a..f418128 100644 --- a/go.mod +++ b/go.mod @@ -44,16 +44,24 @@ require ( github.com/hashicorp/golang-lru v1.0.2 // indirect github.com/ipfs/bbloom v0.0.4 // indirect github.com/ipfs/go-block-format v0.2.0 // indirect + github.com/ipfs/go-blockservice v0.5.2 // indirect github.com/ipfs/go-cid v0.4.1 // indirect github.com/ipfs/go-datastore v0.6.0 // indirect github.com/ipfs/go-ipfs-blockstore v1.3.1 // indirect github.com/ipfs/go-ipfs-ds-help v1.1.1 // indirect + github.com/ipfs/go-ipfs-exchange-interface v0.2.1 // indirect github.com/ipfs/go-ipfs-util v0.0.3 // indirect github.com/ipfs/go-ipld-cbor v0.1.0 // indirect github.com/ipfs/go-ipld-format v0.6.0 // indirect + github.com/ipfs/go-ipld-legacy v0.2.1 // indirect github.com/ipfs/go-log v1.0.5 // indirect github.com/ipfs/go-log/v2 v2.5.1 // indirect + github.com/ipfs/go-merkledag v0.11.0 // indirect github.com/ipfs/go-metrics-interface v0.0.1 // indirect + github.com/ipfs/go-verifcid v0.0.3 // indirect + github.com/ipld/go-car v0.6.1-0.20230509095817-92d28eb23ba4 // indirect + github.com/ipld/go-codec-dagpb v1.6.0 // indirect + github.com/ipld/go-ipld-prime v0.21.0 // indirect github.com/jbenet/goprocess v0.1.4 // indirect github.com/klauspost/cpuid/v2 v2.2.7 // indirect github.com/mattn/go-isatty v0.0.20 // indirect diff --git a/go.sum b/go.sum index 4772d55..62fe454 100644 --- a/go.sum +++ b/go.sum @@ -64,6 +64,8 @@ github.com/ipfs/bbloom v0.0.4 h1:Gi+8EGJ2y5qiD5FbsbpX/TMNcJw8gSqr7eyjHa4Fhvs= github.com/ipfs/bbloom v0.0.4/go.mod h1:cS9YprKXpoZ9lT0n/Mw/a6/aFV6DTjTLYHeA+gyqMG0= github.com/ipfs/go-block-format v0.2.0 h1:ZqrkxBA2ICbDRbK8KJs/u0O3dlp6gmAuuXUJNiW1Ycs= github.com/ipfs/go-block-format v0.2.0/go.mod h1:+jpL11nFx5A/SPpsoBn6Bzkra/zaArfSmsknbPMYgzM= +github.com/ipfs/go-blockservice v0.5.2 h1:in9Bc+QcXwd1apOVM7Un9t8tixPKdaHQFdLSUM1Xgk8= +github.com/ipfs/go-blockservice v0.5.2/go.mod h1:VpMblFEqG67A/H2sHKAemeH9vlURVavlysbdUI632yk= github.com/ipfs/go-cid v0.4.1 h1:A/T3qGvxi4kpKWWcPC/PgbvDA2bjVLO7n4UeVwnbs/s= github.com/ipfs/go-cid v0.4.1/go.mod h1:uQHwDeX4c6CtyrFwdqyhpNcxVewur1M7l7fNU7LKwZk= github.com/ipfs/go-datastore v0.6.0 h1:JKyz+Gvz1QEZw0LsX1IBn+JFCJQH4SJVFtM4uWU0Myk= @@ -74,19 +76,33 @@ github.com/ipfs/go-ipfs-blockstore v1.3.1 h1:cEI9ci7V0sRNivqaOr0elDsamxXFxJMMMy7 github.com/ipfs/go-ipfs-blockstore v1.3.1/go.mod h1:KgtZyc9fq+P2xJUiCAzbRdhhqJHvsw8u2Dlqy2MyRTE= github.com/ipfs/go-ipfs-ds-help v1.1.1 h1:B5UJOH52IbcfS56+Ul+sv8jnIV10lbjLF5eOO0C66Nw= github.com/ipfs/go-ipfs-ds-help v1.1.1/go.mod h1:75vrVCkSdSFidJscs8n4W+77AtTpCIAdDGAwjitJMIo= +github.com/ipfs/go-ipfs-exchange-interface v0.2.1 h1:jMzo2VhLKSHbVe+mHNzYgs95n0+t0Q69GQ5WhRDZV/s= +github.com/ipfs/go-ipfs-exchange-interface v0.2.1/go.mod h1:MUsYn6rKbG6CTtsDp+lKJPmVt3ZrCViNyH3rfPGsZ2E= github.com/ipfs/go-ipfs-util v0.0.3 h1:2RFdGez6bu2ZlZdI+rWfIdbQb1KudQp3VGwPtdNCmE0= github.com/ipfs/go-ipfs-util v0.0.3/go.mod h1:LHzG1a0Ig4G+iZ26UUOMjHd+lfM84LZCrn17xAKWBvs= github.com/ipfs/go-ipld-cbor v0.1.0 h1:dx0nS0kILVivGhfWuB6dUpMa/LAwElHPw1yOGYopoYs= github.com/ipfs/go-ipld-cbor v0.1.0/go.mod h1:U2aYlmVrJr2wsUBU67K4KgepApSZddGRDWBYR0H4sCk= github.com/ipfs/go-ipld-format v0.6.0 h1:VEJlA2kQ3LqFSIm5Vu6eIlSxD/Ze90xtc4Meten1F5U= github.com/ipfs/go-ipld-format v0.6.0/go.mod h1:g4QVMTn3marU3qXchwjpKPKgJv+zF+OlaKMyhJ4LHPg= +github.com/ipfs/go-ipld-legacy v0.2.1 h1:mDFtrBpmU7b//LzLSypVrXsD8QxkEWxu5qVxN99/+tk= +github.com/ipfs/go-ipld-legacy v0.2.1/go.mod h1:782MOUghNzMO2DER0FlBR94mllfdCJCkTtDtPM51otM= github.com/ipfs/go-log v1.0.5 h1:2dOuUCB1Z7uoczMWgAyDck5JLb72zHzrMnGnCNNbvY8= github.com/ipfs/go-log v1.0.5/go.mod h1:j0b8ZoR+7+R99LD9jZ6+AJsrzkPbSXbZfGakb5JPtIo= github.com/ipfs/go-log/v2 v2.1.3/go.mod h1:/8d0SH3Su5Ooc31QlL1WysJhvyOTDCjcCZ9Axpmri6g= github.com/ipfs/go-log/v2 v2.5.1 h1:1XdUzF7048prq4aBjDQQ4SL5RxftpRGdXhNRwKSAlcY= github.com/ipfs/go-log/v2 v2.5.1/go.mod h1:prSpmC1Gpllc9UYWxDiZDreBYw7zp4Iqp1kOLU9U5UI= +github.com/ipfs/go-merkledag v0.11.0 h1:DgzwK5hprESOzS4O1t/wi6JDpyVQdvm9Bs59N/jqfBY= +github.com/ipfs/go-merkledag v0.11.0/go.mod h1:Q4f/1ezvBiJV0YCIXvt51W/9/kqJGH4I1LsA7+djsM4= github.com/ipfs/go-metrics-interface v0.0.1 h1:j+cpbjYvu4R8zbleSs36gvB7jR+wsL2fGD6n0jO4kdg= github.com/ipfs/go-metrics-interface v0.0.1/go.mod h1:6s6euYU4zowdslK0GKHmqaIZ3j/b/tL7HTWtJ4VPgWY= +github.com/ipfs/go-verifcid v0.0.3 h1:gmRKccqhWDocCRkC+a59g5QW7uJw5bpX9HWBevXa0zs= +github.com/ipfs/go-verifcid v0.0.3/go.mod h1:gcCtGniVzelKrbk9ooUSX/pM3xlH73fZZJDzQJRvOUw= +github.com/ipld/go-car v0.6.1-0.20230509095817-92d28eb23ba4 h1:oFo19cBmcP0Cmg3XXbrr0V/c+xU9U1huEZp8+OgBzdI= +github.com/ipld/go-car v0.6.1-0.20230509095817-92d28eb23ba4/go.mod h1:6nkFF8OmR5wLKBzRKi7/YFJpyYR7+oEn1DX+mMWnlLA= +github.com/ipld/go-codec-dagpb v1.6.0 h1:9nYazfyu9B1p3NAgfVdpRco3Fs2nFC72DqVsMj6rOcc= +github.com/ipld/go-codec-dagpb v1.6.0/go.mod h1:ANzFhfP2uMJxRBr8CE+WQWs5UsNa0pYtmKZ+agnUw9s= +github.com/ipld/go-ipld-prime v0.21.0 h1:n4JmcpOlPDIxBcY037SVfpd1G+Sj1nKZah0m6QH9C2E= +github.com/ipld/go-ipld-prime v0.21.0/go.mod h1:3RLqy//ERg/y5oShXXdx5YIp50cFGOanyMctpPjsvxQ= github.com/jbenet/go-cienv v0.1.0/go.mod h1:TqNnHUmJgXau0nCzC7kXWeotg3J9W34CUv5Djy1+FlA= github.com/jbenet/goprocess v0.1.4 h1:DRGOFReOMqqDNXwW70QkacFW0YN9QnwLV0Vqk+3oU0o= github.com/jbenet/goprocess v0.1.4/go.mod h1:5yspPrukOVuOLORacaBi858NqyClJPQxYZlqdZVfqY4= diff --git a/internal/api/handlers/post/getstatus.go b/internal/api/handlers/post/getstatus.go index 49c6545..a8ba7b1 100644 --- a/internal/api/handlers/post/getstatus.go +++ b/internal/api/handlers/post/getstatus.go @@ -7,6 +7,30 @@ import ( "time" "Coves/internal/core/posts" + + "github.com/bluesky-social/indigo/atproto/syntax" +) + +// The identifier bounds this endpoint accepts, DERIVED from the specs rather +// than picked, because both directions of getting it wrong are silent. +// +// A postv2 URI is AUTHORITY-SCOPED — the author's DID is inside it — and a DID +// may legally run to 2048 bytes, the same fact that killed the readable rkey +// transform in PRD rev 2.2. A cap sized for the old community-repo URIs would +// refuse exactly the authors with long DIDs, on the one endpoint that can tell +// them why their post is not visible, and nothing else in the system would show +// a symptom: their posts index fine and every other endpoint serves them. +// +// So the URI bound is the sum of its legal parts rather than a round number: +// "at://" + a 2048-byte authority + "/" + a 317-byte NSID + "/" + a 512-byte +// record key. It is a pre-parse DoS bound only — ParseATURI below is what +// decides whether the thing is actually a URI. +const ( + maxAuthorityLength = 2048 // DID Core's ceiling + maxNSIDLength = 317 // atProto NSID grammar + maxRecordKeyLength = 512 // atProto record-key grammar + maxPostURILength = len("at://") + maxAuthorityLength + 1 + maxNSIDLength + 1 + maxRecordKeyLength + maxCommunityIDLength = maxAuthorityLength ) // GetStatusHandler serves social.coves.community.post.getStatus: one @@ -53,15 +77,32 @@ func (h *GetStatusHandler) HandleGetStatus(w http.ResponseWriter, r *http.Reques writeError(w, http.StatusBadRequest, "InvalidRequest", "community parameter is required") return } - if len(postURI) > maxURILength { + // THE CAP IS SIZED TO THE SPEC, NOT TO THE OLD URIs. A DID may legally run + // to 2048 bytes — the same fact that killed the readable rkey transform in + // PRD rev 2.2 — and an author-owned post URI is authority-scoped, so the + // author's DID is INSIDE the URI this endpoint takes. A cap sized for the + // old community-repo URIs would silently make long-DID authors unqueryable + // and nothing else would notice: their posts index fine and every other + // endpoint serves them. + if len(postURI) > maxPostURILength { writeError(w, http.StatusBadRequest, "InvalidRequest", "post URI exceeds maximum length") return } - if len(communityDID) > maxURILength { + if len(communityDID) > maxCommunityIDLength { writeError(w, http.StatusBadRequest, "InvalidRequest", "community DID exceeds maximum length") return } + // Raising the cap must not become "accept anything long". The URI is PARSED, + // so a non-at:// string comes back as the client bug it is rather than as a + // silent not-found — which would tell a client with a typo that its post + // does not exist. + if _, err := syntax.ParseATURI(postURI); err != nil { + writeError(w, http.StatusBadRequest, "InvalidRequest", + "post must be a valid AT-URI") + return + } + status, err := h.service.GetStatus(r.Context(), posts.GetStatusRequest{ PostURI: postURI, CommunityDID: communityDID, @@ -96,6 +137,13 @@ func (h *GetStatusHandler) HandleGetStatus(w http.ResponseWriter, r *http.Reques } w.Header().Set("Content-Type", "application/json") + // NO-STORE, and not as a nicety. This endpoint exists to be POLLED for a + // transition (§7), so a cached answer is the endpoint failing at its only + // job: the client keeps being handed `pending` after the post was accepted + // and stops asking. It is also unauthenticated and reports a moderation + // decision, so an intermediary holding a copy is a disclosure surface that + // outlives the request that created it. + w.Header().Set("Cache-Control", "no-store") w.WriteHeader(http.StatusOK) if _, err := w.Write(responseBytes); err != nil { log.Printf("ERROR: Failed to write getStatus response: %v", err) diff --git a/internal/api/routes/post.go b/internal/api/routes/post.go index 67dfd3b..dde40ae 100644 --- a/internal/api/routes/post.go +++ b/internal/api/routes/post.go @@ -6,10 +6,20 @@ import ( "Coves/internal/core/blueskypost" "Coves/internal/core/posts" "Coves/internal/core/votes" + "time" "github.com/go-chi/chi/v5" ) +// getStatusRateLimit is social.coves.community.post.getStatus' per-client +// budget, per minute. +// +// Sixty is chosen against the endpoint's own UX rather than copied from a +// neighbour: §7 has a client poll for the accepted transition, and a poll a +// second for a minute is comfortably inside this while a script enumerating a +// community's rejected posts is not. +const getStatusRateLimit = 60 + // PostRouteOption supplies a collaborator that only some of the post routes // need. // @@ -26,12 +36,24 @@ type postRouteConfig struct { // WithPostStatusService supplies the service behind // social.coves.community.post.getStatus. The route is registered either way, so -// that the HTTP surface does not silently change shape with the wiring; without -// this option the handler has no service to call. +// that the HTTP surface does not silently change shape with the wiring. func WithPostStatusService(service posts.StatusService) PostRouteOption { return func(c *postRouteConfig) { c.statusService = service } } +// NOT GUARDED AT REGISTRATION, unlike oauthMiddleware above, and the asymmetry +// is forced rather than chosen. The review asked for a fail-fast panic on a +// missing status service, matching that neighbour — but the routes table builds +// the whole router with every service nil and no options at all +// (registration_test.go's theRouter), so a panic there would fail every +// surface-declaration test rather than the one wiring bug it is aimed at. +// Scoping it to a supplied-but-nil option would guard a shape nobody writes. +// +// The exposure is small and named here so it is not mistaken for an oversight: +// cmd/server always supplies the option, and a build that did not would serve +// 500s from getStatus alone. Closing it properly needs the routes table to pass +// the option, which is a test-side change. + // RegisterPostRoutes registers post-related XRPC endpoints on the router // Implements social.coves.community.post.* lexicon endpoints // authMiddleware can be either OAuthAuthMiddleware or DualAuthMiddleware (used for @@ -87,8 +109,18 @@ func RegisterPostRoutes( // be harmless but pointless — the answer does not vary by viewer — while // RequireAuth would make the cross-server case unanswerable, which is the // asymmetry internal/api/routes/registration_test.go declares. + // + // It also carries its OWN limiter, tighter than the global 100/minute, and + // it is the only unauthenticated route in the product that does. Two facts + // make the exception worth it: §7's client UX is to POLL this until a post + // flips to accepted, so the honest traffic shape is repeated requests from + // one caller; and because it takes no auth, an unauthenticated stranger can + // ask about any post URI they can name. The budget is what bounds + // enumeration of a community's rejected posts to a rate an operator notices. statusHandler := post.NewGetStatusHandler(cfg.statusService) - r.Get("/xrpc/social.coves.community.post.getStatus", statusHandler.HandleGetStatus) + statusRateLimiter := middleware.NewRateLimiter(getStatusRateLimit, time.Minute) + r.With(statusRateLimiter.Middleware). + Get("/xrpc/social.coves.community.post.getStatus", statusHandler.HandleGetStatus) // Future endpoints (Beta): // r.With(authMiddleware.RequireAuth).Post("/xrpc/social.coves.community.post.update", updateHandler.HandleUpdate) diff --git a/internal/atproto/jetstream/authorpost.go b/internal/atproto/jetstream/authorpost.go index 84aaef0..40f5a8b 100644 --- a/internal/atproto/jetstream/authorpost.go +++ b/internal/atproto/jetstream/authorpost.go @@ -1,6 +1,7 @@ package jetstream import ( + "bytes" "context" "database/sql" "encoding/json" @@ -12,6 +13,7 @@ import ( "net/url" "strconv" "strings" + "sync" "time" "Coves/internal/atproto/identity" @@ -19,6 +21,9 @@ import ( "Coves/internal/core/communities" "Coves/internal/core/posts" "Coves/internal/core/users" + + "github.com/bluesky-social/indigo/atproto/atdata" + "github.com/bluesky-social/indigo/repo" ) // Ingesting author-owned posts and the community records that decide about @@ -91,7 +96,7 @@ func WithPostRecordFetcher(fetcher PostRecordFetcher) PostEventConsumerOption { // AcceptanceDeleter withdraws a community's acceptance of a post. Satisfied by // posts.CommunityRecordWriter. // -// RED STUB (task 5, cycle 2). Narrowed to one method because that is all the +// Narrowed to one method because that is all the // tombstone path needs: the consumer must never write an acceptance, a removal // or a repin — those are the ENGINE's verdicts, and a consumer holding the full // writer is one edit away from making one. @@ -157,8 +162,23 @@ type DirectPostFetcher struct { // a request forger pointed at whatever is reachable from the AppView's // network, driven by any stranger who writes an acceptance record. allowPrivateHosts bool + + // client is the guarded HTTP client, built once. See httpClient. + clientOnce sync.Once + client *http.Client } +// fetchTimeout bounds ONE direct fetch, inside the client's own 15-second +// ceiling. +// +// The two are not redundant. The client's timeout is a backstop for a +// connection that hangs; this one is a policy about how long the posts lane may +// wait on a stranger's PDS. The lane is single-threaded and now carries four +// collections, so every second spent here is a second nothing else is indexed — +// and the destination is chosen by whoever wrote the acceptance. Five seconds +// is generous for a repo read and cheap to lose. +const fetchTimeout = 5 * time.Second + // maxFetchedRecordBytes bounds how much of a PDS getRecord response is read. // // A post record has a lexicon-bounded size, so a PDS streaming megabytes is @@ -191,11 +211,24 @@ func NewDevDirectPostFetcher(resolver identity.Resolver) *DirectPostFetcher { return &DirectPostFetcher{resolver: resolver, allowPrivateHosts: true} } -// httpClient builds the guarded client for one fetch. Declared here so the -// guard is derived from allowPrivateHosts at call time rather than baked into a -// client at construction, where a test seam could not reach it. +// httpClient returns the guarded client, building it once on first use. +// +// ONE CLIENT, not one per fetch. A fresh http.Client means a fresh +// http.Transport, and a fresh transport means an empty connection pool: every +// fetch would pay a new TCP handshake and a new TLS handshake against a PDS +// this consumer may be about to fetch from a hundred more times, and the +// discarded transports leak idle connections until their finalizers run. The +// guard is a property of the transport rather than of the moment, so nothing +// about correctness needed it rebuilt. +// +// It is built lazily rather than in the constructor so that a fetcher which is +// wired but never used — the common case on an instance hosting no communities +// — costs nothing. func (f *DirectPostFetcher) httpClient() *http.Client { - return oauth.NewSSRFSafeHTTPClient(f.allowPrivateHosts) + f.clientOnce.Do(func() { + f.client = oauth.NewSSRFSafeHTTPClient(f.allowPrivateHosts) + }) + return f.client } // FetchPost implements PostRecordFetcher. @@ -220,11 +253,29 @@ func (f *DirectPostFetcher) FetchPost(ctx context.Context, postURI string) (*Fet return nil, fmt.Errorf("resolving the repo of %s: no PDS endpoint in the DID document", postURI) } - endpoint := strings.TrimSuffix(resolved.PDSURL, "/") + "/xrpc/com.atproto.repo.getRecord?repo=" + + // com.atproto.sync.getRecord, NOT repo.getRecord, and the difference is the + // whole trustworthiness of this path. + // + // repo.getRecord answers with JSON — {"uri":…, "cid":…, "value":{…}} — whose + // `cid` is a CLAIM BY THE SERVER. That server is the author's PDS: named by + // a DID document, reached because a stranger wrote an acceptance naming this + // subject. Comparing the pinned CID against that field asks the attacker + // whether the attacker is lying, and the consequence is the worst one this + // design has — the AppView indexes whatever `value` holds under a + // community's SIGNED acceptance of a CID that content does not have. The + // community attested to one thing and every reader is shown another. + // + // sync.getRecord answers with a CAR: the repo's own blocks. The CID is then + // RECOMPUTED from the bytes rather than read off a label, and no server can + // lie about the hash of what it just sent. + endpoint := strings.TrimSuffix(resolved.PDSURL, "/") + "/xrpc/com.atproto.sync.getRecord?did=" + url.QueryEscape(repoDID) + "&collection=" + url.QueryEscape(collection) + "&rkey=" + url.QueryEscape(rkey) - req, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil) + fetchCtx, cancel := context.WithTimeout(ctx, fetchTimeout) + defer cancel() + + req, err := http.NewRequestWithContext(fetchCtx, http.MethodGet, endpoint, nil) if err != nil { return nil, fmt.Errorf("building the getRecord request for %s: %w", postURI, err) } @@ -284,25 +335,46 @@ func (f *DirectPostFetcher) FetchPost(ctx context.Context, postURI string) (*Fet postURI, resp.StatusCode, strconv.Quote(detail)) } - var parsed struct { - URI string `json:"uri"` - CID string `json:"cid"` - Value map[string]interface{} `json:"value"` + // The CAR carries the record's block plus the blocks proving it belongs to + // the repo. Reading it can fail on a hostile or broken server, which is a + // refusal like any other — a response that is not a CAR is not evidence. + stored, err := repo.ReadRepoFromCar(ctx, bytes.NewReader(body)) + if err != nil { + return nil, fmt.Errorf("reading the CAR for %s: %w", postURI, err) } - if err := json.Unmarshal(body, &parsed); err != nil { - return nil, fmt.Errorf("parsing the getRecord response for %s: %w", postURI, err) + + claimedCID, recordBytes, err := stored.GetRecordBytes(ctx, collection+"/"+rkey) + if err != nil { + return nil, fmt.Errorf("reading %s out of the fetched CAR: %w", postURI, err) } - if parsed.Value == nil { - return nil, fmt.Errorf("the getRecord response for %s carried no record value", postURI) + if recordBytes == nil || len(*recordBytes) == 0 { + return nil, fmt.Errorf("the CAR for %s carried no record bytes", postURI) } - if parsed.CID == "" { - // Without a CID there is nothing to verify the pinned reference - // against, and an unverified record is exactly what the fetch must - // never index. - return nil, fmt.Errorf("the getRecord response for %s carried no CID", postURI) + + // THE RECOMPUTATION, which is the entire point of taking a CAR at all. The + // CID that came out of the repo structure is still something the server + // assembled; hashing the record's own bytes under that CID's codec and + // multihash is what turns it into a fact. A server that substituted content + // produces a digest that does not match, whatever it labelled the block. + computedCID, err := claimedCID.Prefix().Sum(*recordBytes) + if err != nil { + return nil, fmt.Errorf("recomputing the CID of %s: %w", postURI, err) + } + if !computedCID.Equals(claimedCID) { + return nil, fmt.Errorf("%w: the CAR for %s labels a block %s whose bytes hash to %s", + ErrPermanentEvent, postURI, claimedCID, computedCID) } - return &FetchedPost{URI: postURI, CID: parsed.CID, Record: parsed.Value}, nil + record, err := atdata.UnmarshalCBOR(*recordBytes) + if err != nil { + return nil, fmt.Errorf("decoding the record block of %s: %w", postURI, err) + } + + // The CID reported back is the COMPUTED one. The caller compares it against + // what the acceptance pinned, and handing back the label instead would put + // the server's claim back into the comparison the recomputation just removed + // it from. + return &FetchedPost{URI: postURI, CID: computedCID.String(), Record: record}, nil } // --------------------------------------------------------------------------- @@ -424,14 +496,28 @@ func (c *PostEventConsumer) tombstoneAuthorPost(ctx context.Context, authorDID s if err != nil { return err } - if !applied || !indexed { - // A gate skip means this deletion was already applied — the sweep ran - // with it — so re-sweeping would put one authenticated PDS round trip - // behind every redelivery of every tombstone on the network. + if !indexed { return nil } - c.withdrawAcceptance(ctx, stored.communityDID, uri) + // THE WITHDRAWAL IS RECONSIDERED ON EVERY DELIVERY, unlike the tombstone. + // The gate exists to make the local soft-delete happen once; the withdrawal + // is a write into a REMOTE repo that can fail on its own, and it is + // idempotent — DeleteAcceptance reports "nothing to withdraw" as a skip. + // + // Tying it to the gate is what stranded it: a PDS briefly unreachable on the + // first delivery meant the acceptance stayed standing, pointing at a record + // nobody can fetch, permanently — because every redelivery was rejected by + // the gate before the sweep was reconsidered, and nothing else revisits it. + // The community's CAR, the thing its portability argument rests on, would + // keep citing content the author withdrew. + // + // The guard is the POST's state rather than this event's: sweep when the row + // is tombstoned, which is true on the delivery that applied it and on every + // one after. + if applied || stored.deletedAt != nil { + c.withdrawAcceptance(ctx, stored.communityDID, uri) + } return nil } @@ -547,6 +633,37 @@ func (c *PostEventConsumer) upsertAuthorPost(ctx context.Context, authorDID stri uri := recordURI(authorDID, PostV2Collection, commit.RKey) + stored, found, err := c.loadStoredPost(ctx, uri) + if err != nil { + return err + } + + // IMMUTABILITY (§3.1) IS CHECKED FIRST, BEFORE THE COMMUNITY IS LOOKED UP, + // and the order is load-bearing rather than tidy. + // + // An update that changes `community` invalidates the WHOLE event — not + // merely the community field, because applying the content while keeping the + // old community would leave the first community's admission holding a CID it + // never evaluated, publishing content nobody judged under a standing + // acceptance. Retargeting a post means writing a new record. + // + // Two rules meet when the retarget names a community nobody has indexed, and + // only one of them can go first. The unknown-community branch below is + // TRANSIENT — correctly, since a community's own profile event may simply + // not have arrived — so checking it first would turn an illegal retarget + // into a retryable failure that can NEVER succeed: it dead-letters, redrives + // ten times, blocks ~4.2s inline on each delivery, and is still an illegal + // retarget once that community exists. An author could mint that load at + // will by editing one field. + // + // A skip, not an error: an invalid record from a stranger's repo is not an + // infrastructure failure. + if found && stored.communityDID != record.Community { + log.Printf("🚨 SECURITY: ignoring the whole %s update for %s - community is immutable (stored %s, incoming %s)", + PostV2Collection, uri, stored.communityDID, record.Community) + return nil + } + // The community must be one this AppView has indexed, or there is no // subject to open an admission against. // @@ -563,26 +680,6 @@ func (c *PostEventConsumer) upsertAuthorPost(ctx context.Context, authorDID stri return fmt.Errorf("%w: failed to verify community %s exists: %v", errValidationInfra, record.Community, err) } - stored, found, err := c.loadStoredPost(ctx, uri) - if err != nil { - return err - } - - // IMMUTABILITY (§3.1): an update that changes `community` invalidates the - // WHOLE event. Not merely the community field — applying the content while - // keeping the old community would leave the first community's admission - // holding a CID it never evaluated, publishing content nobody judged under - // a standing acceptance. Retargeting a post means writing a new record. - // - // A skip, not an error: an invalid record from a stranger's repo is not an - // infrastructure failure, and dead-lettering it would retry a record that - // can never become valid. - if found && stored.communityDID != record.Community { - log.Printf("🚨 SECURITY: ignoring the whole %s update for %s - community is immutable (stored %s, incoming %s)", - PostV2Collection, uri, stored.communityDID, record.Community) - return nil - } - // Provenance for bridgedStats is keyed on the AUTHOR's PDS now, because the // record lives in the author's repo. The community's host has no say over // what an author asserts about their own record any more, so checking the @@ -622,12 +719,22 @@ func (c *PostEventConsumer) upsertAuthorPost(ctx context.Context, authorDID stri } } - if !applied { - // The rev gate or the recency guard refused this event: a newer state is - // already indexed. Opening or refreshing an admission from it would move - // evaluated_cid BACKWARDS onto content the row no longer holds, which is - // how an accepted post gets flipped to pending_reacceptance by a - // duplicate delivery. + // THE GATE GUARDS THE POST ROW, NOT THE ADMISSION. The two writes have + // opposite idempotence: inserting the post must happen exactly once, while + // UpsertPending is content-addressed and writes nothing when the row already + // holds this CID. Gating them together is what orphaned admissions — a + // failed upsert leaves the post indexed, the gate advanced, and every + // redelivery skipped, so the row is never created and the post is invisible + // in its community forever with nothing left to retry. + // + // A REPLAY IS SAFE, A STALE COPY IS NOT, and the stored rev is what tells + // them apart. An equal rev is this same commit arriving again — the upsert + // re-runs harmlessly and repairs the orphan. A strictly greater stored rev + // means a NEWER event already applied, and re-running the upsert from this + // older one would drag evaluated_cid backwards onto content the row no + // longer holds, flipping an accepted post to pending_reacceptance on a + // duplicate delivery. + if !applied && !c.revIsCurrent(ctx, uri, commit.Rev) { return nil } @@ -661,6 +768,31 @@ func (c *PostEventConsumer) upsertAuthorPost(ctx context.Context, authorDID stri return nil } +// revIsCurrent reports whether the gate's stored rev for this record is exactly +// the one this event carries — i.e. whether the event is the newest the AppView +// has seen rather than an older copy. +// +// It exists so a rev-gated SKIP can still be distinguished into its two very +// different causes. A redelivery of the newest commit is safe to act on again; +// a stale cross-feed copy of an older one is not. Anything unreadable answers +// false, which declines to act — the conservative direction, since acting on a +// stale event corrupts state while declining merely waits for the next +// delivery. +func (c *PostEventConsumer) revIsCurrent(ctx context.Context, uri, rev string) bool { + if rev == "" { + // A rev-less event bypasses the gate entirely, so there is no stored rev + // to be current with. Only synthetic events reach this. + return true + } + var stored string + if err := c.db.QueryRowContext(ctx, + `SELECT rev FROM jetstream_record_revs WHERE record_uri = $1`, uri, + ).Scan(&stored); err != nil { + return false + } + return stored == rev +} + // hydrateAuthorOpportunistically indexes a minimal profile for an author this // AppView has not seen, so their posts are not permanently authorless. // @@ -916,10 +1048,28 @@ func (c *PostEventConsumer) applyAcceptance( subjectCollection string, watermark posts.CommunityWatermark, ) error { - indexedCommunity, indexed, err := c.indexedPostCommunity(ctx, decision.Subject.URI) + stored, indexed, err := c.loadStoredPost(ctx, decision.Subject.URI) if err != nil { return err } + + // A TOMBSTONED SUBJECT IS NOT ACCEPTABLE, and the arrival is legitimate + // rather than hostile: the community decided before it saw the tombstone, + // and the two events are in different repos with no ordering between them. + // Applying it would have getStatus report `accepted` for content no read + // path will ever serve — and the host-side sweep already ran with the + // tombstone and will not run again, so the community's repo would keep an + // acceptance citing a record nobody can fetch. + // + // A SKIP, not a refusal. Returning an error would dead-letter an event that + // is replayed constantly and would be refused identically every time. + if indexed && stored.deletedAt != nil { + log.Printf("INFO: not applying the acceptance of %s in %s: its author deleted the post", + decision.Subject.URI, communityDID) + return nil + } + + indexedCommunity := stored.communityDID switch { case !indexed: if err := c.convergeOnAcceptedSubject(ctx, communityDID, decision, subjectCollection); err != nil { @@ -1033,7 +1183,7 @@ func (c *PostEventConsumer) convergeOnAcceptedSubject( // event time to stamp: an empty rev bypasses the gate (which is correct — a // later real event for this record still wins on its own rev) and the // watermark falls back to wall clock. - if _, err := c.insertAuthorPost(ctx, authorPostInsert{ + applied, err := c.insertAuthorPost(ctx, authorPostInsert{ uri: decision.Subject.URI, authorDID: authorDID, record: record, @@ -1043,19 +1193,38 @@ func (c *PostEventConsumer) convergeOnAcceptedSubject( }, facets: facetsJSON, embed: embedJSON, labels: labelsJSON, bridgedUpvotes: up, bridgedDownvotes: down, bridgedAsOf: asOf, - }); err != nil { + }) + if err != nil { return err } - // The pending admission comes with it. ApplyAcceptance would create a row - // on its own, but one with no evaluated content: recording what was indexed - // is what lets the next author edit be recognised as an edit. - if _, err := c.admissions.UpsertPending(ctx, posts.UpsertPendingCommand{ - CommunityDID: communityDID, - PostURI: decision.Subject.URI, - EvaluatedCID: fetched.CID, - }); err != nil { - return fmt.Errorf("recording the pending admission for fetched post %s: %w", decision.Subject.URI, err) + // THE FETCH IS A CATCH-UP, NOT AN AUTHORITY, and applied=false is how it + // learns it lost the race. The acceptance and the post live in different + // repos, so Jetstream parallelises them and the post's own event can land + // while this fetch is in flight — which is not a rare interleaving but the + // normal one, since the fetch exists precisely because the event had not + // arrived. The insert then conflicts and writes nothing. + // + // UpsertPending is last-write-wins, so running it anyway would stamp the + // FETCHED CID over the newer one the real event just recorded. evaluated_cid + // is what the next decision judges, so the engine would evaluate content the + // author has already replaced, and an acceptance written from that verdict + // would pin a version the AppView is no longer serving. + // + // Skipping it costs nothing: ApplyAcceptance below classifies against + // whatever evaluated_cid the row actually holds, which is exactly the + // comparison that turns a stale pin into pending_reacceptance. + if applied { + // The pending admission comes with it. ApplyAcceptance would create a row + // on its own, but one with no evaluated content: recording what was indexed + // is what lets the next author edit be recognised as an edit. + if _, err := c.admissions.UpsertPending(ctx, posts.UpsertPendingCommand{ + CommunityDID: communityDID, + PostURI: decision.Subject.URI, + EvaluatedCID: fetched.CID, + }); err != nil { + return fmt.Errorf("recording the pending admission for fetched post %s: %w", decision.Subject.URI, err) + } } c.hydrateAuthorOpportunistically(ctx, authorDID) @@ -1111,48 +1280,121 @@ func (c *PostEventConsumer) applyRemoval( // (§5.3), is the one delete that arrives unpaired — and it is the one this // lookup resolves. func (c *PostEventConsumer) applyCommunityDecisionDelete(ctx context.Context, communityDID string, commit *CommitEvent) error { - if commit.Collection != posts.AcceptanceCollection { - log.Printf("INFO: %s deletion in %s carries no subject and is superseded by its paired write; skipping", - commit.Collection, communityDID) - return nil + postURI, found, err := c.subjectOfDeletedRecord(ctx, communityDID, commit) + if err != nil { + return err } - - var postURI string - err := c.db.QueryRowContext(ctx, - `SELECT post_uri FROM community_post_admissions - WHERE community_did = $1 AND acceptance_rkey = $2`, - communityDID, commit.RKey, - ).Scan(&postURI) - if errors.Is(err, sql.ErrNoRows) { - // No acceptance of that rkey stands here — the removal half of the same - // commit already cleared it, or this AppView never saw the acceptance. - // Either way there is nothing to withdraw. - log.Printf("INFO: acceptance deletion %s/%s matches no standing acceptance; nothing to withdraw", - communityDID, commit.RKey) + if !found { + // Nothing here matches that record key: the paired write of the same + // commit already superseded it, or this AppView never saw the record + // being deleted. Either way there is nothing to withdraw. + log.Printf("INFO: %s deletion %s/%s matches no standing record; nothing to withdraw", + commit.Collection, communityDID, commit.RKey) return nil } - if err != nil { - return fmt.Errorf("resolving the subject of acceptance deletion %s/%s: %w", communityDID, commit.RKey, err) - } - result, err := c.admissions.ApplyAcceptanceDelete(ctx, posts.CommunityDeleteCommand{ + cmd := posts.CommunityDeleteCommand{ CommunityDID: communityDID, PostURI: postURI, Watermark: posts.CommunityWatermark{Rev: commit.Rev}, - }) + } + + var result posts.AdmissionResult + if commit.Collection == posts.RemovalCollection { + result, err = c.admissions.ApplyRemovalDelete(ctx, cmd) + } else { + result, err = c.admissions.ApplyAcceptanceDelete(ctx, cmd) + } if err != nil { - return fmt.Errorf("withdrawing the acceptance of %s in %s: %w", postURI, communityDID, err) + return fmt.Errorf("withdrawing the %s of %s in %s: %w", commit.Collection, postURI, communityDID, err) } - logAdmissionOutcome(posts.AcceptanceCollection+"#delete", communityDID, postURI, result.Outcome) + logAdmissionOutcome(commit.Collection+"#delete", communityDID, postURI, result.Outcome) return nil } +// subjectOfDeletedRecord recovers which post a deleted acceptance or removal was +// about. +// +// A delete event carries NO record, and the rkey is a SHA-256 digest of the +// subject URI (§3.2), which is one-way — so the subject can only come from state +// the AppView already holds. The two collections need different lookups because +// only one of them has a column: +// +// - An ACCEPTANCE stores its rkey on the admission row, so the reverse lookup +// is an exact match. +// - A REMOVAL stores none, so its subject is found by recomputing the digest +// over the rows that could be its subject. The candidate set is bounded to +// this community's `removed` rows — moderation-sized, and served by the +// (community_did, status, created_at) index migration 034 already carries. +// +// WHY THE REMOVAL CASE CANNOT STAY A NO-OP, which is what it was. The paired +// commits do converge without it: a removal commit is {acceptance-delete, +// removal-put} and a restore is {removal-delete, acceptance-put}, and in both +// the put carries the subject in-record and outranks its delete under the §5.2 +// tuple. But a moderator can also simply WITHDRAW a removal — deleting the +// record and writing nothing — and that commit carries only this event. Ignored, +// the post stays `removed` forever while the community's own repo no longer says +// so: the signed record and the AppView disagree, and only the AppView is +// consulted when the post is served. +func (c *PostEventConsumer) subjectOfDeletedRecord( + ctx context.Context, communityDID string, commit *CommitEvent, +) (string, bool, error) { + if commit.Collection == posts.AcceptanceCollection { + var postURI string + err := c.db.QueryRowContext(ctx, + `SELECT post_uri FROM community_post_admissions + WHERE community_did = $1 AND acceptance_rkey = $2`, + communityDID, commit.RKey, + ).Scan(&postURI) + if errors.Is(err, sql.ErrNoRows) { + return "", false, nil + } + if err != nil { + return "", false, fmt.Errorf("resolving the subject of acceptance deletion %s/%s: %w", + communityDID, commit.RKey, err) + } + return postURI, true, nil + } + + rows, err := c.db.QueryContext(ctx, + `SELECT post_uri FROM community_post_admissions + WHERE community_did = $1 AND status = 'removed'`, communityDID) + if err != nil { + return "", false, fmt.Errorf("resolving the subject of removal deletion %s/%s: %w", + communityDID, commit.RKey, err) + } + defer func() { _ = rows.Close() }() + + for rows.Next() { + var postURI string + if err := rows.Scan(&postURI); err != nil { + return "", false, fmt.Errorf("scanning a removed subject of %s: %w", communityDID, err) + } + if posts.SubjectRkey(postURI) == commit.RKey { + return postURI, true, nil + } + } + if err := rows.Err(); err != nil { + return "", false, fmt.Errorf("reading the removed subjects of %s: %w", communityDID, err) + } + return "", false, nil +} + // logAdmissionOutcome records what a community event DID, including the skips. // // A skip is the ordering gate working — a multi-feed duplicate, a dead-letter -// redrive, an event superseded by its own commit's other half — so it is logged -// rather than returned as an error, which would bury healthy skips in the -// dead-letter queue among genuine failures. +// redrive, an event superseded by its own commit's other half, or this +// AppView's own write coming back to it — so it is logged rather than returned +// as an error, which would bury healthy skips in the dead-letter queue among +// genuine failures. +// +// THE LAST OF THOSE IS THE COMMON ONE ON A HOSTING INSTANCE and is easy to +// misread in the logs. When this AppView hosts the community, the engine writes +// the acceptance into the repo and stamps the row optimistically; the firehose +// then delivers that same commit back, and the watermark CAS answers +// skipped_stale because the row already carries that exact rev. Nothing is +// wrong — the write landed twice by design, once locally and once as its own +// echo — and the engine's own doc calls that echo a success. func logAdmissionOutcome(collection, communityDID, postURI string, outcome posts.AdmissionOutcome) { if outcome == posts.AdmissionApplied { log.Printf("✓ Applied %s for %s in %s", collection, postURI, communityDID) diff --git a/internal/atproto/jetstream/post_consumer.go b/internal/atproto/jetstream/post_consumer.go index 5ee6dbb..fc02bf8 100644 --- a/internal/atproto/jetstream/post_consumer.go +++ b/internal/atproto/jetstream/post_consumer.go @@ -30,12 +30,14 @@ type PostEventConsumer struct { // passes bridgeTrust. identityResolver identity.Resolver - // RED STUB fields (task 5, cycle 1) — the collaborators author-owned post - // ingestion needs. Declared here so the options in authorpost.go compile; - // nothing reads them yet. See docs/PRD_AUTHOR_OWNED_POSTS.md §5.3-§5.6. + // The collaborators author-owned post ingestion needs + // (docs/PRD_AUTHOR_OWNED_POSTS.md §5.3-§5.6). All four are read by the + // handlers in authorpost.go, and a nil one disables a capability rather + // than degrading it — see each field. // // admissions holds the per-(community, post) decision state. nil means the - // consumer is running in its pre-034 shape and records no admissions. + // consumer is running in its pre-034 shape and ignores all three + // author-owned collections rather than indexing them undecided. admissions posts.AdmissionRepository // deletedAccounts gates events from erased accounts. nil means no gate. deletedAccounts DeletedAccountLookup diff --git a/internal/atproto/oauth/transport.go b/internal/atproto/oauth/transport.go index 9e7c463..21ab957 100644 --- a/internal/atproto/oauth/transport.go +++ b/internal/atproto/oauth/transport.go @@ -1,6 +1,7 @@ package oauth import ( + "context" "fmt" "net" "net/http" @@ -62,14 +63,43 @@ func isPrivateIP(ip net.IP) bool { return false } +// vettedAddrsKeyType keys the addresses RoundTrip approved, so the dialler can +// read them off the request's own context. A private type, so nothing outside +// this file can plant a value under the same key. +type vettedAddrsKeyType struct{} + +var vettedAddrsKey vettedAddrsKeyType + +// RoundTrip vets the hostname's addresses and then makes the dial use THOSE +// ADDRESSES rather than the name. +// +// PASSING THE NAME ON WOULD BE A SECOND DECISION. The base transport resolves +// whatever host it is given, so a guard that approved answer A and then handed +// over the hostname lets the dialler act on answer B — and nothing binds the two +// together. DNS rebinding is the name for exploiting that: an attacker who +// controls the zone answers the first query publicly and the second with +// 169.254.169.254, and the approval describes a host the connection never went +// to. Every input that reaches this transport is chosen by a stranger (a DID +// document's PDS endpoint, an acceptance record's subject), so the attacker +// picks the moment to flip as well. +// +// The addresses ride on the request context and the dialler below consumes +// them, which also means the hostname is resolved exactly ONCE per request. +// That is the property, not an optimisation: any later answer is one the guard +// never saw. func (t *ssrfSafeTransport) RoundTrip(req *http.Request) (*http.Response, error) { host := req.URL.Hostname() - // Resolve hostname to IP + // A literal address is already the thing that will be dialled, so there is + // no second resolution to defend against — but it still has to pass the + // private check below. ips, err := t.resolveHost(host) if err != nil { return nil, fmt.Errorf("failed to resolve host: %w", err) } + if len(ips) == 0 { + return nil, fmt.Errorf("failed to resolve host: %s resolved to no addresses", host) + } // Check all resolved IPs if !t.allowPrivate { @@ -80,17 +110,53 @@ func (t *ssrfSafeTransport) RoundTrip(req *http.Request) (*http.Response, error) } } - return t.base.RoundTrip(req) + return t.base.RoundTrip(req.WithContext(context.WithValue(req.Context(), vettedAddrsKey, ips))) } // NewSSRFSafeHTTPClient creates an HTTP client with SSRF protections func NewSSRFSafeHTTPClient(allowPrivate bool) *http.Client { + dialer := &net.Dialer{ + Timeout: 10 * time.Second, + KeepAlive: 30 * time.Second, + } + transport := &ssrfSafeTransport{ base: &http.Transport{ - DialContext: (&net.Dialer{ - Timeout: 10 * time.Second, - KeepAlive: 30 * time.Second, - }).DialContext, + // The dial IGNORES the hostname in addr and connects to an address + // RoundTrip already approved, which is what closes the + // check-then-dial window. It takes only the port from addr, because + // the port is the one part of the destination the guard has no + // opinion about. + // + // FAIL CLOSED when there is nothing vetted: reaching here without a + // context value means this base transport was used directly rather + // than through RoundTrip, which is exactly the unguarded path the + // wrapper exists to prevent. + // + // TLS is unaffected. http.Transport derives the handshake's server + // name from the request URL rather than from the dial address, so + // certificate verification and SNI still name the host the caller + // asked for. + DialContext: func(ctx context.Context, network, addr string) (net.Conn, error) { + vetted, _ := ctx.Value(vettedAddrsKey).([]net.IP) + if len(vetted) == 0 { + return nil, fmt.Errorf("SSRF blocked: refusing to dial %s with no vetted address "+ + "(the SSRF-safe transport was bypassed)", addr) + } + _, port, err := net.SplitHostPort(addr) + if err != nil { + return nil, fmt.Errorf("SSRF blocked: cannot read a port from %q: %w", addr, err) + } + var lastErr error + for _, ip := range vetted { + conn, dialErr := dialer.DialContext(ctx, network, net.JoinHostPort(ip.String(), port)) + if dialErr == nil { + return conn, nil + } + lastErr = dialErr + } + return nil, lastErr + }, MaxIdleConns: 100, IdleConnTimeout: 90 * time.Second, TLSHandshakeTimeout: 10 * time.Second, diff --git a/internal/core/posts/decider.go b/internal/core/posts/decider.go index 432834c..cda3b16 100644 --- a/internal/core/posts/decider.go +++ b/internal/core/posts/decider.go @@ -4,7 +4,6 @@ import ( "context" "errors" "fmt" - "log" "os" "strings" "time" @@ -29,9 +28,11 @@ import ( // - THE ACTOR CLASS. A misclassification is a privilege decision. Trusted // aggregators skip visibility, ban and authorization entirely, so guessing // UPWARD on a failed lookup would hand the widest privileges in the system -// to whoever made the lookup fail. Every uncertain path therefore falls to -// the STRICTER class, matching CreatePost's existing behaviour (service.go -// step 3 treats a failed IsAggregator lookup as an ordinary user). +// to whoever made the lookup fail. But guessing DOWNWARD is not free here +// either, which is where this path parts company with CreatePost: a lookup +// that could not be made is reported as UNDECIDED rather than resolved to +// the stricter class, because this decision gets written into the admission +// row with redrivable = false. See classify for the full asymmetry. // // It reuses evaluateAdmissionPolicy rather than admitPost, and that is the split // task 3 built for: admitPost RESERVES a ledger slot, and the engine is not a @@ -187,7 +188,12 @@ func (d *AdmissionEngineDecider) DecideAdmission(ctx context.Context, communityD postURI, communityDID, ErrSubjectGone)) } - return evaluateAdmissionPolicy(ctx, admissionDeps{ + actor, err := d.classify(ctx, post.AuthorDID) + if err != nil { + return undecided(fmt.Errorf("deciding %s for %s: %w", postURI, communityDID, err)) + } + + decision, err := evaluateAdmissionPolicy(ctx, admissionDeps{ communities: d.deps.Communities, bans: d.deps.Policy.Bans, aggregators: d.deps.Authorizer, @@ -195,7 +201,7 @@ func (d *AdmissionEngineDecider) DecideAdmission(ctx context.Context, communityD limits: d.deps.Policy.Limits, now: d.deps.Policy.Now, }, AdmissionRequest{ - Actor: d.classify(ctx, post.AuthorDID), + Actor: actor, AuthorDID: post.AuthorDID, // The community DID, which resolves to itself. The engine's input is an // admission row, and its key is already the resolved DID — there is no @@ -209,42 +215,106 @@ func (d *AdmissionEngineDecider) DecideAdmission(ctx context.Context, communityD // redeciding. Fingerprint: "", }) + if err != nil || !decision.Admitted() { + return decision, err + } + + return d.applyQuota(ctx, communityDID, post.AuthorDID, actor, decision) } -// classify decides what class of actor the author is. +// applyQuota is the firehose path's §8 submission limit, and the last thing +// between an admitted decision and the engine writing an acceptance. +// +// IT COUNTS ADMISSION ROWS, NOT LEDGER ROWS, and that substitution is the whole +// reason it exists separately from admitPost's step 6. post_submissions is +// written by CreatePost, so a post that arrived over the firehose from an author +// on another server has no ledger row and never will — counting it would hold +// LOCAL users to the limit while exempting precisely the remote ones §8 is +// about. Anyone can write unlimited postv2 records naming any community, and the +// admission layer is what absorbs that. +// +// THE LIMIT IS THE SAME NUMBER the write path uses, taken from the same config, +// so an author is held to one quota rather than to two that drift. +// +// ACTOR CLASSES ARE TREATED EXACTLY AS admitPost TREATS THEM: only ActorUser is +// metered. A registered aggregator is already governed by its own hourly quota +// inside ValidateAggregatorPost, and applying this as well would silently halve +// an authorized aggregator's throughput; a trusted one has never had a +// submission limit, and inventing one here would stop the bridge dead at a +// number nobody chose. +func (d *AdmissionEngineDecider) applyQuota( + ctx context.Context, communityDID, authorDID string, actor ActorClass, decision AdmissionDecision, +) (AdmissionDecision, error) { + if actor != ActorUser || d.deps.Admissions == nil { + return decision, nil + } + limits := d.deps.Policy.Limits + if limits.MaxPerAuthorPerCommunity <= 0 || limits.Window <= 0 { + return decision, nil + } + + since := d.deps.Policy.Now().Add(-limits.Window) + count, err := d.deps.Admissions.CountRecentAdmissions(ctx, communityDID, authorDID, since) + if err != nil { + // UNDECIDED, never a refusal. The engine persists a refusal code and + // marks it non-redrivable, so a count that could not be taken must not + // become a permanent rate-limit verdict on somebody's post. + return undecided(fmt.Errorf("counting recent admissions for %s in %s: %w", authorDID, communityDID, err)) + } + + // EXCEEDS, not reaches. The subject being decided already has its own + // admission row — the consumer opens it before the engine ever runs — so it + // is inside this count, exactly as admitPost's reservation is inside its + // own. Comparing with >= would refuse the author's very first post. + if count > limits.MaxPerAuthorPerCommunity { + return AdmissionDecision{Code: DecisionRateLimitExceeded}, nil + } + return decision, nil +} + +// classify decides what class of actor the author is, or reports that it could +// not. +// +// A FAILED LOOKUP IS UNDECIDED HERE, AND A DOWNGRADE ON THE WRITE PATH. The two +// answers are deliberate opposites, and the difference is what each caller does +// with the result afterwards. CreatePost is talking to a live client: a +// downgrade to ActorUser applies the strict checks, costs an aggregator a few +// posts until the table recovers, and hands back something the caller can +// retry. Nothing is written down. +// +// The engine writes its verdict INTO the admission row, and a policy refusal is +// stamped redrivable = false — terminal, never revisited by the redrive pass. +// The same downgrade there does not cost a retry; it permanently marks a post +// refused for a reason that was never true, because a table was briefly +// unreachable, and nothing in the system would ever look at it again. // -// EVERY UNCERTAIN PATH FALLS TO ActorUser, the stricter class. A trusted -// aggregator skips visibility, ban and authorization entirely, so resolving a -// failed lookup UPWARD would hand the widest privileges in the system to -// whoever managed to make the lookup fail. Guessing downward costs an -// aggregator some refused posts until the lookup recovers — and CreatePost -// already made exactly this choice (service.go step 3), so the engine agreeing -// with it is also what keeps the write path and the ingestion path from -// disagreeing about who someone is. -func (d *AdmissionEngineDecider) classify(ctx context.Context, authorDID string) ActorClass { +// So the rule this encodes is: A DECISION THAT PERSISTS MAY ONLY BE MADE FROM +// AN ANSWER THAT WAS ACTUALLY OBTAINED. Guessing upward is never available +// either — a trusted aggregator skips visibility, ban and authorization +// entirely, so resolving uncertainty in that direction would hand the widest +// privileges in the system to whoever could make a lookup fail. +func (d *AdmissionEngineDecider) classify(ctx context.Context, authorDID string) (ActorClass, error) { // The trusted set is checked FIRST, which is both the cheaper path and the // only one that costs nothing: it is an in-memory set resolved at // construction, so a trusted actor never pays for a database lookup to // learn what the process already knew. if d.deps.TrustedAggregatorDIDs[authorDID] { - return ActorTrustedAggregator + return ActorTrustedAggregator, nil } // With no aggregator collaborators wired — a deployment with no aggregator - // support at all — nobody can be classified as one, which is the strict - // answer rather than a degraded one. + // support at all — nobody can be classified as one. That is a CONFIGURED + // fact rather than a failed lookup, so it answers rather than defers. if d.deps.Aggregators == nil || d.deps.Authorizer == nil { - return ActorUser + return ActorUser, nil } registered, err := d.deps.Aggregators.IsAggregator(ctx, authorDID) if err != nil { - log.Printf("[ADMISSION-DECIDER] Warning: classifying %s fell back to the user class, IsAggregator failed: %v", - authorDID, err) - return ActorUser + return "", fmt.Errorf("classifying %s: %w", authorDID, err) } if registered { - return ActorRegisteredAggregator + return ActorRegisteredAggregator, nil } - return ActorUser + return ActorUser, nil } diff --git a/internal/core/posts/engine.go b/internal/core/posts/engine.go index 2cc78da..834b49a 100644 --- a/internal/core/posts/engine.go +++ b/internal/core/posts/engine.go @@ -26,15 +26,21 @@ import ( // one place that decides they should fire at all. // // THERE IS NO LEASE, AND THAT IS DELIBERATE. Nothing stops two passes — the -// fast path, a firehose redelivery, a notify — from processing the same -// subject at the same moment, and no lock or per-subject claim is taken. +// queue driver, a firehose redelivery, and (once task 6 lands it) the +// synchronous fast path — from processing the same subject at the same moment, +// and no lock or per-subject claim is taken. Only the first two exist today; +// the fast path is named because the safety argument has to hold when it +// arrives, not because it is calling now. // Safety comes from the layers instead: deterministic rkeys make the racers // aim at the same record; every put and batch is swap-guarded, so a loser is // told rather than clobbering; a loser that re-reads and finds the winner // wrote its exact target converges as a skip; and the repository's watermark // CAS makes the row's state advance monotonically no matter which pass -// stamps first. Serializing the passes properly (a per-community queue) is -// task 5's job; until then concurrent passes are expected and harmless. +// stamps first. Serializing the passes properly is QueueDriver's job (queue.go): +// one goroutine walks one ordered list, grouped by community, so no two subjects +// of a community are ever in flight together. The layered safety above still +// matters, because the driver is not the only caller — the synchronous fast +// path and the firehose consumer both reach the engine too. // EngineOutcome reports what one pass over one subject DID. // @@ -62,11 +68,14 @@ const ( // EngineRepinned means a standing acceptance moved onto new content with no // re-decision — the bridgedStats exception of §5.5. // - // NOT PRODUCED BY ANY PATH YET. ProcessAdmission runs full re-admission on - // every edit; the repin path — classifyRecordDiff choosing the exception, - // the bridge-trust gate approving the author, RepinAcceptance moving the - // record — is task 5's consumer wiring. The outcome is declared now so - // that path lands against a named contract instead of minting one. + // NOT PRODUCED BY ANY PATH YET, and deliberately still deferred. + // ProcessAdmission runs full re-admission on every edit; the repin path — + // classifyRecordDiff choosing the exception, the bridge-trust gate + // approving the author, RepinAcceptance moving the record — needs an + // old-record snapshot the consumer does not keep, so wiring it is a + // recorded decision rather than an oversight (see loop_state.md). The + // outcome is declared so that path lands against a named contract instead + // of minting one. EngineRepinned EngineOutcome = "repinned" // EngineDeferred means the subject is still owed a decision and nothing diff --git a/internal/core/posts/queue.go b/internal/core/posts/queue.go index f8caca0..51160ce 100644 --- a/internal/core/posts/queue.go +++ b/internal/core/posts/queue.go @@ -2,6 +2,7 @@ package posts import ( "context" + "errors" "fmt" "log" "sync" @@ -12,9 +13,9 @@ import ( // and on what (docs/PRD_AUTHOR_OWNED_POSTS.md §5.6, §8). // // The engine settles one subject. Nothing until now decided which subjects, in -// what order, or how often — the fast path and the firehose consumer both push -// work at it, and neither can see a subject that was left pending because a -// credential expired or a lookup blipped. This is the pull side: a periodic pass +// what order, or how often — the firehose consumer pushes work at it today, and +// task 6's synchronous fast path will push more, and neither can see a subject +// that was left pending because a credential expired or a lookup blipped. This is the pull side: a periodic pass // over the undecided backlog that eventually reaches every stranded row. // // # IT IS A SINGLE GOROUTINE, AND THAT IS THE PER-COMMUNITY SERIALIZATION @@ -218,7 +219,7 @@ func (d *QueueDriver) RunPass(ctx context.Context) (PassReport, error) { startedAt := d.now() report := PassReport{StartedAt: startedAt} - subjects, err := d.subjects.ListPendingSubjects(ctx, d.batchSize) + subjects, err := d.subjects.ListPendingSubjects(ctx, d.fetchSize()) if err != nil { // The one failure a pass has nothing to do about. Every other outcome // below is per-subject and counted; this one means there is no work @@ -228,6 +229,12 @@ func (d *QueueDriver) RunPass(ctx context.Context) (PassReport, error) { report.Listed = len(subjects) for _, subject := range groupByCommunity(subjects) { + // THE BATCH IS FILLED WITH WORK, not with rows. Skipping a backed-off + // subject must not consume a slot, or a stuck prefix would spend the + // whole pass on subjects it never touched — see fetchSize. + if report.Processed >= d.batchSize { + break + } if d.heldBack(subject, startedAt) { continue } @@ -249,6 +256,17 @@ func (d *QueueDriver) RunPass(ctx context.Context) (PassReport, error) { // community's PDS returns errors, not deferrals, so exempting failures // would leave the loudest case as the one thing nothing paced. switch { + case errors.Is(err, ErrSubjectGone): + // NOT A FAILURE, and counting it as one would make an ordinary race + // look like an outage. The backlog query excludes tombstoned and + // unindexed posts, but a post can be deleted between the listing and + // the decision — so this is the queue meeting a subject that stopped + // existing while it worked, which is exactly what the exclusion is + // for and needs no operator's attention. It is counted as deferred + // and backed off like any other "nothing to do yet"; the next pass + // will not list it at all. + report.Deferred++ + d.deferSubject(subject, startedAt) case err != nil: report.Failed++ log.Printf("[ACCEPTANCE-QUEUE] Warning: %s in %s could not be settled: %v", @@ -263,10 +281,81 @@ func (d *QueueDriver) RunPass(ctx context.Context) (PassReport, error) { } } + d.pruneDeferrals(subjects) d.record(subjects, report, startedAt) return report, nil } +// queueOverFetchFactor bounds how far past a backed-off prefix one pass may +// reach: a pass never asks the query for more than batchSize × this. +// +// FOUR, and the number is a trade rather than a preference. The backlog is +// ordered oldest-first, so the subjects most likely to be stuck are exactly the +// ones at the front of it — a community whose credentials expired weeks ago sits +// there forever. Without over-fetching, a pass asks for LIMIT rows, gets LIMIT +// stuck ones, skips them all for backoff and does nothing; every pass, while a +// healthy post two rows behind is never decided. Over-fetching without a bound +// would instead let one pass drag the entire backlog into memory to find one +// live subject. Four buys three batches of headroom against a query whose cost +// grows with it, and a prefix deeper than that drains as backoffs expire. +const queueOverFetchFactor = 4 + +// fetchSize is how many rows to ask for so the pass can still fill its batch +// after skipping the subjects it already knows are held back. +// +// It asks for the batch plus exactly the number of deferrals currently held — +// the measured size of the prefix that may be skipped — rather than always +// over-fetching. A driver with nothing backed off has nothing to skip, so it +// asks for precisely what it will use. +func (d *QueueDriver) fetchSize() int { + held := d.deferredCount() + size := d.batchSize + held + if ceiling := d.batchSize * queueOverFetchFactor; size > ceiling { + size = ceiling + } + return size +} + +func (d *QueueDriver) deferredCount() int { + d.mu.Lock() + defer d.mu.Unlock() + return len(d.deferrals) +} + +// pruneDeferrals forgets the backoffs of subjects that are no longer listed. +// +// A deferral outlives its subject otherwise. The row gets settled by somebody +// else — the synchronous fast path, a firehose acceptance, a moderator's +// removal — and simply stops being listed, without ever telling the driver. The +// map is keyed by subject and swept by nothing, so on a busy instance it is an +// unbounded leak held for the life of the process. +// +// It is also wrong on RE-ENTRY, which is the part a leak metric would not show: +// a subject that leaves the backlog and comes back — an edit reopening an +// accepted post — would arrive carrying a stale backoff it did nothing to earn, +// and wait out a delay that was about a completely different decision. +// +// Pruning against the LISTING rather than against what the pass processed is +// deliberate: a subject skipped for backoff is still in the backlog, and +// forgetting it would defeat the backoff on the very next pass. +func (d *QueueDriver) pruneDeferrals(listed []PendingSubject) { + d.mu.Lock() + defer d.mu.Unlock() + + if len(d.deferrals) == 0 { + return + } + stillListed := make(map[subjectKey]bool, len(listed)) + for _, subject := range listed { + stillListed[keyOf(subject)] = true + } + for key := range d.deferrals { + if !stillListed[key] { + delete(d.deferrals, key) + } + } +} + // Snapshot returns the driver's health surface as of the last completed pass. func (d *QueueDriver) Snapshot() QueueSnapshot { d.mu.Lock() @@ -321,6 +410,7 @@ func (d *QueueDriver) record(subjects []PendingSubject, report PassReport, at ti PendingBacklog: report.Listed, LastPassDeferred: report.Deferred, LastPassFailed: report.Failed, + DeferredSubjects: len(d.deferrals), } // Taken as a MINIMUM rather than as subjects[0], even though the query // orders by age. The oldest entry's age is the queue's only early warning, diff --git a/internal/core/users/interfaces.go b/internal/core/users/interfaces.go index cf0f849..620ee01 100644 --- a/internal/core/users/interfaces.go +++ b/internal/core/users/interfaces.go @@ -13,6 +13,21 @@ type UpdateProfileInput struct { } // UserRepository defines the interface for user data persistence +// ErasureLookup reports whether a DID names an account this AppView was asked +// to erase (migration 036). +// +// It is a SEPARATE, OPTIONAL interface rather than a method on UserRepository, +// and detected with a type assertion at the one call site that needs it. Adding +// it to UserRepository would oblige every implementation to answer a question +// only the PostgreSQL one can — and a double that answered "not erased" by +// default would be a gate that fails open, which is the single outcome this +// marker exists to prevent. A repository that does not implement it disables +// the gate rather than weakening it, and nothing in production is such a +// repository. +type ErasureLookup interface { + IsAccountDeleted(ctx context.Context, did string) (bool, error) +} + type UserRepository interface { Create(ctx context.Context, user *User) (*User, error) GetByDID(ctx context.Context, did string) (*User, error) diff --git a/internal/core/users/service.go b/internal/core/users/service.go index de0b035..99e2b2f 100644 --- a/internal/core/users/service.go +++ b/internal/core/users/service.go @@ -455,6 +455,37 @@ func (s *userService) mintInviteCode(ctx context.Context) (string, error) { // best-effort — this heals users whose profile firehose event was never delivered // without ever blocking or failing the IndexUser call itself. func (s *userService) IndexUser(ctx context.Context, did, handle, pdsURL string) error { + // THE ERASURE GATE, and IndexUser is where it belongs because this is the + // FIREHOSE's door into the users table. A DID appearing in a profile or + // identity event means only that some repo somewhere emitted a record — a + // bridge, a replay, an overlapping feed — and any of those can arrive months + // after the account was erased. Letting it through would make the erasure + // undone by exactly the replays the marker exists to defend against, and + // undone silently: the users row reappears, repo.Create clears the marker on + // its way past, and the next replayed post indexes normally. + // + // The marker's only exit is a genuine re-registration, which reaches the + // repository's insert directly rather than through here. + // + // A LOOKUP FAILURE REFUSES. "I could not read the marker table" and "there + // is no marker" must never be the same answer, because the second one + // indexes. + if lookup, ok := s.userRepo.(ErasureLookup); ok { + erased, err := lookup.IsAccountDeleted(ctx, did) + if err != nil { + return fmt.Errorf("checking the erasure marker for %s before indexing: %w", did, err) + } + if erased { + // Nil, not an error. Every caller is a firehose consumer, and the + // connector dead-letters what a handler returns — so refusing with + // an error would turn each erased account into a permanent stream of + // redriving profile events. This is not a failure; it is an event + // with nothing to do. + log.Printf("INFO: not indexing %s from the firehose: the account was erased (migration 036 marker)", did) + return nil + } + } + // Try to create the user (idempotent - CreateUser returns existing user if DID exists) user, err := s.CreateUser(ctx, CreateUserRequest{ DID: did, diff --git a/internal/db/postgres/admission_queue_repo.go b/internal/db/postgres/admission_queue_repo.go index b232c9a..70a9f20 100644 --- a/internal/db/postgres/admission_queue_repo.go +++ b/internal/db/postgres/admission_queue_repo.go @@ -99,5 +99,28 @@ func (r *postgresAdmissionRepo) ListPendingSubjects(ctx context.Context, limit i func (r *postgresAdmissionRepo) CountRecentAdmissions( ctx context.Context, communityDID, authorDID string, since time.Time, ) (int, error) { - return 0, nil + // starts_with on the authority segment, the same shape userRepo.Delete uses + // to sweep an author's admissions. The trailing "/" matters: without it + // did:plc:abc would also match did:plc:abcdef, and one author would consume + // another's quota. + // + // accepted and pending together, and NOTHING else. §8 is explicit that a + // refusal consumes no quota — counting rejected or removed rows would let an + // author past their limit extend their own lockout every time they tried + // again. pending_reacceptance counts too: the post is admitted and visible + // history, merely awaiting a re-decision on an edit. + const query = ` + SELECT count(*) + FROM community_post_admissions + WHERE community_did = $1 + AND starts_with(post_uri, 'at://' || $2 || '/') + AND status IN ('accepted', 'pending', 'pending_reacceptance') + AND created_at >= $3 + ` + + var count int + if err := r.db.QueryRowContext(ctx, query, communityDID, authorDID, since).Scan(&count); err != nil { + return 0, fmt.Errorf("counting recent admissions for %s in %s: %w", authorDID, communityDID, err) + } + return count, nil } diff --git a/internal/db/postgres/deleted_account_repo.go b/internal/db/postgres/deleted_account_repo.go index 1f9c71c..4990aac 100644 --- a/internal/db/postgres/deleted_account_repo.go +++ b/internal/db/postgres/deleted_account_repo.go @@ -29,9 +29,24 @@ func NewDeletedAccountRepository(db *sql.DB) *DeletedAccountRepository { // is indistinguishable from a healthy answer — a database blip would silently // re-index the content a deletion erased, which is the exact outcome the marker // table exists to prevent. +func (r *postgresUserRepo) IsAccountDeleted(ctx context.Context, did string) (bool, error) { + return accountIsErased(ctx, r.db, did) +} + +// IsAccountDeleted implements the same lookup for the standalone repository. func (r *DeletedAccountRepository) IsAccountDeleted(ctx context.Context, did string) (bool, error) { + return accountIsErased(ctx, r.db, did) +} + +// accountIsErased is the single statement behind both lookups above. +// +// It is one function because the two callers are the two halves of the same +// guard — the ingestion consumer refusing an erased author's events, and the +// user service refusing to re-index them — and a second spelling would be a +// second chance for one of them to drift into failing open. +func accountIsErased(ctx context.Context, db *sql.DB, did string) (bool, error) { var deleted bool - if err := r.db.QueryRowContext(ctx, + if err := db.QueryRowContext(ctx, `SELECT EXISTS (SELECT 1 FROM deleted_accounts WHERE did = $1)`, did, ).Scan(&deleted); err != nil { return false, fmt.Errorf("checking whether %s was erased: %w", did, err) -- 2.51.2 From 4fd43cede28e3b873de22065614ebe41cb7b56c6 Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 03:56:17 -0700 Subject: [PATCH 14/17] test(ingestion): move the fetch success paths onto real repos MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The verification now recomputes the CID from com.atproto.sync.getRecord's CAR blocks, and four tests were still written against the JSON envelope that change removed. All five affected packages are green. 1. fakeAuthorPDS asserted /xrpc/com.atproto.repo.getRecord and the `repo` parameter — the endpoint the pin in f4291e1 exists to forbid. Moved to sync.getRecord + `did`. That alone greens the three fetch-taxonomy tests (RecordNotFound / bare 404 / 5xx), which never depended on the body shape. A helper left asserting the old endpoint would have kept passing against a fetcher that regressed to trusting the envelope, which is the whole failure this contradiction was hiding. 2. The four success-path arcs are REWRITTEN on the package's new PDS floor rather than retired — the DirectFetch trio proves the component, these prove the wiring, and dropping them would have left the consumer's reach-for-the-fetch untested: UnindexedPost real record never delivered → fetched, indexed, accepted FetchedCIDMismatch TWO real records; the acceptance names one and pins the other's CID. Both CIDs are genuine, so nothing is malformed and the refusal can only come from the recomputation actually being compared to the pin. NamingAnotherCommunity a real record naming community B, accepted by A ConvergeMustNotRegress the race rebuilt with a racingFetcher decorator: the REAL fetcher still reads the real repo and still recomputes; the decorator only fixes WHEN the competing event lands, which in production is decided by two repos Jetstream carries in parallel. 3. tests/e2e header records that the positive fetch arc is structurally T1-only. Every PDS write in that tier reaches the AppView through Jetstream, so a record written to stage acceptance-before-post is delivered before the acceptance — the ordinary path, not the fetch. Suppressing that would mean not writing the record or reconfiguring the consumers, and rule 2 of the package forbids the second. The negative is stageable and stays asserted there; the note says so explicitly so the absence is not read as a gap. Staged these three files only — GREEN's batch is in this worktree. Co-Authored-By: Claude Fable 5 --- .../jetstream/acceptance_consumer_test.go | 191 +++++++++++------- .../jetstream/admission_durability_test.go | 106 +++++----- tests/e2e/author_post_contract_test.go | 20 ++ 3 files changed, 200 insertions(+), 117 deletions(-) diff --git a/internal/atproto/jetstream/acceptance_consumer_test.go b/internal/atproto/jetstream/acceptance_consumer_test.go index 416c2e3..a60858b 100644 --- a/internal/atproto/jetstream/acceptance_consumer_test.go +++ b/internal/atproto/jetstream/acceptance_consumer_test.go @@ -286,18 +286,26 @@ func TestRemovalConsumer_PreemptiveRemovalCreatesTheRow(t *testing.T) { // §5.4 direct fetch: acceptance before post // --------------------------------------------------------------------------- -// fakeAuthorPDS is an httptest server answering com.atproto.repo.getRecord for +// fakeAuthorPDS is an httptest server answering com.atproto.sync.getRecord for // the author's postv2 record, standing in for the PDS a DID document points at. // // It asserts the request shape as it serves, because the fetch is the one place -// the AppView reads a record without the firehose: a getRecord aimed at the -// wrong repo or collection would return someone else's record, and the CID check -// downstream would happily verify it. +// the AppView reads a record without the firehose: a fetch aimed at the wrong +// repo or collection would return someone else's record, and the verification +// downstream would happily confirm it. +// +// sync.getRecord, and its parameter is `did` rather than `repo`. Both changed +// together and neither is cosmetic: repo.getRecord answers with a JSON envelope +// whose `cid` is a claim by the server being interrogated, while sync.getRecord +// answers with the repo's own blocks, which is what makes the CID something the +// AppView can RECOMPUTE instead of read off a label. A helper still asserting +// the old endpoint would keep passing against a fetcher that had regressed to +// trusting the envelope. func fakeAuthorPDS(t *testing.T, expectRepo string, handler http.HandlerFunc) *httptest.Server { t.Helper() return httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { - assert.Equal(t, "/xrpc/com.atproto.repo.getRecord", r.URL.Path) - assert.Equal(t, expectRepo, r.URL.Query().Get("repo")) + assert.Equal(t, "/xrpc/com.atproto.sync.getRecord", r.URL.Path) + assert.Equal(t, expectRepo, r.URL.Query().Get("did")) assert.Equal(t, PostV2Collection, r.URL.Query().Get("collection")) handler(w, r) })) @@ -331,41 +339,92 @@ func newFetcherAt(t *testing.T, authorDID, pdsURL string) *DirectPostFetcher { return fetcher } +// realRepoFixture is the acceptance consumer pointed at a REAL author repo on +// the test PDS. +// +// The three arcs below all turn on what the fetch VERIFIES, and verification is +// recomputation from the repo's own blocks (§5.4) — so a hand-served response +// cannot exercise them at all: it would either be rejected as malformed, which +// proves nothing about the CID, or require this file to fabricate a CAR and +// thereby encode its own guesses about how the verification walks one. +// +// The DirectFetch trio in direct_fetch_verification_test.go proves the +// COMPONENT. These prove the WIRING: that the consumer reaches for it on an +// unindexed subject, and that what it does with each answer — index and accept, +// or refuse permanently — is what the admission row ends up saying. +type realRepoFixture struct { + accFixture + + pds *testkit.PDS + author *testkit.Account +} + +func newRealRepoFixture(t *testing.T, db *sql.DB) *realRepoFixture { + t.Helper() + + pdsServer := testkit.NewPDS(t) + author := pdsServer.CreateAccount(t, testkit.WithHandlePrefix("rr")) + + insertBridgedUser(t, db, accAuthor, "realowner.test") + insertBridgedCommunity(t, db, accCommunity, "realcommunity.test", accAuthor) + + admissions := postgres.NewAdmissionRepository(db) + consumer := NewPostEventConsumer( + postgres.NewPostRepository(db), postgres.NewCommunityRepository(db), + newMockUserService(), db, + WithAdmissions(admissions), + WithDeletedAccounts(postgres.NewDeletedAccountRepository(db)), + WithPostRecordFetcher(NewDevDirectPostFetcher(pinnedResolver(author.DID, pdsServer.URL()))), + ) + + return &realRepoFixture{ + accFixture: accFixture{consumer: consumer, admissions: admissions, db: db}, + pds: pdsServer, + author: author, + } +} + +// publish writes a real postv2 record into the author's repo and returns it, so +// the CID an acceptance pins is one the PDS minted from bytes it stored. +func (f *realRepoFixture) publish(t *testing.T, communityDID, title string) testkit.Record { + t.Helper() + return f.author.CreateRecord(t, PostV2Collection, map[string]any{ + "$type": PostV2Collection, + "community": communityDID, + "title": title, + "content": "a body only the repo can vouch for", + "createdAt": time.Now().UTC().Format(time.RFC3339), + }) +} + func TestAcceptanceConsumer_UnindexedPost_IsFetchedDirectlyAndAccepted(t *testing.T) { t.Parallel() db := testkit.DB(t) ctx := context.Background() - base := time.Now().UnixMicro() + f := newRealRepoFixture(t, db) - const cid = "bafyreiaccfetched" - rkey := "accfetch" - uri := accPostURI(rkey) + // The post exists in its author's repo and the AppView has never seen it — + // no create event arrived, and none ever will if the relay does not crawl + // that PDS. This is the case §5.4 says redrive cannot solve: bounded retries + // cannot manufacture an event nobody is going to send. Without the fetch, + // convergence is a bet on full relay coverage. + record := f.publish(t, accCommunity, "never delivered by any relay") - srv := fakeAuthorPDS(t, accAuthor, func(w http.ResponseWriter, r *http.Request) { - serveRecord(t, w, uri, cid, pv2Record(accCommunity, "fetched straight from the PDS", "body")) - }) - defer srv.Close() - - f := newAccFixture(t, db, WithPostRecordFetcher(newFetcherAt(t, accAuthor, srv.URL))) + require.NoError(t, f.consumer.HandleEvent(ctx, + acceptanceEvent(accCommunity, record.URI, record.CID, testkit.TID(), time.Now().UnixMicro()))) - // The post was NEVER indexed — no create event ever arrived, and none ever - // will if the relay does not crawl the author's PDS. This is the case §5.4 - // says redrive cannot solve: retries cannot manufacture an event nobody is - // going to send. Without the fetch, convergence requires full relay - // coverage, which is a bet rather than a guarantee. - require.NoError(t, f.consumer.HandleEvent(ctx, acceptanceEvent(accCommunity, uri, cid, testkit.TID(), base))) - - authorDID, communityDID, storedCID, _, _ := readPV2Post(t, db, uri) - assert.Equal(t, accAuthor, authorDID, "the fetched post is attributed to the repo it was read from") + authorDID, communityDID, storedCID, _, _ := readPV2Post(t, db, record.URI) + assert.Equal(t, f.author.DID, authorDID, "the fetched post is attributed to the repo it was read from") assert.Equal(t, accCommunity, communityDID) - assert.Equal(t, cid, storedCID) + assert.Equal(t, record.CID, storedCID, "the indexed CID must be the one the repo minted") - admission, err := f.admissions.Get(ctx, accCommunity, uri) + admission, err := f.admissions.Get(ctx, accCommunity, record.URI) require.NoError(t, err) require.NotNil(t, admission) assert.Equal(t, posts.AdmissionStatusAccepted, admission.Status, - "the fetch exists so the acceptance can be APPLIED; indexing the post and leaving the decision pending would solve half the problem and leave the post invisible") + "the fetch exists so the acceptance can be APPLIED; indexing the post and leaving the decision pending would "+ + "solve half the problem and leave the post invisible") } func TestAcceptanceConsumer_FetchedCIDMismatch_IsPermanentlyRefused(t *testing.T) { @@ -373,30 +432,27 @@ func TestAcceptanceConsumer_FetchedCIDMismatch_IsPermanentlyRefused(t *testing.T db := testkit.DB(t) ctx := context.Background() - base := time.Now().UnixMicro() - - rkey := "accmismatch" - uri := accPostURI(rkey) - - srv := fakeAuthorPDS(t, accAuthor, func(w http.ResponseWriter, r *http.Request) { - // The PDS serves the CURRENT version. The acceptance pins an older one. - serveRecord(t, w, uri, "bafyreiaccnowcurrent", pv2Record(accCommunity, "the version now at that rkey", "body")) - }) - defer srv.Close() - - f := newAccFixture(t, db, WithPostRecordFetcher(newFetcherAt(t, accAuthor, srv.URL))) - - err := f.consumer.HandleEvent(ctx, acceptanceEvent(accCommunity, uri, "bafyreiaccpinnedold", testkit.TID(), base)) - - // The CID check is what makes the fetch trustworthy at all. Without it the - // AppView indexes whatever the author's PDS chooses to serve under that - // rkey — the author (or whoever holds their keys) picks the content, and the - // community's signed acceptance is made to cover it retroactively. - require.Error(t, err, "a fetched record whose CID is not the one the acceptance pinned must never be indexed under that acceptance") + f := newRealRepoFixture(t, db) + + subject := f.publish(t, accCommunity, "the post the acceptance is about") + other := f.publish(t, accCommunity, "a different post in the same repo") + require.NotEqual(t, subject.CID, other.CID, "fixture: the two records must have distinct CIDs") + + // BOTH CIDs ARE REAL, and that is what makes this the sharp version. The + // acceptance names one record and pins another's CID — a perfectly + // well-formed strongRef that simply does not describe its subject. Nothing + // about the response is malformed, so the refusal can only come from the + // verification actually comparing what it recomputed against what was + // pinned. + err := f.consumer.HandleEvent(ctx, + acceptanceEvent(accCommunity, subject.URI, other.CID, testkit.TID(), time.Now().UnixMicro())) + + require.Error(t, err, "an acceptance pinning a CID the subject does not have must never be applied") assert.ErrorIs(t, err, ErrPermanentEvent, - "the pinned version is gone from the repo and no retry brings it back; the connector must dead-letter this with its redrive budget already spent rather than re-fetching the same mismatch ten times") + "the pinned version is not what that URI holds, and no retry changes which bytes are in the repo; re-fetching "+ + "the same mismatch ten times is pure noise") - assert.Zero(t, countRows(t, db, `SELECT count(*) FROM posts WHERE uri = $1`, uri), + assert.Zero(t, countRows(t, db, `SELECT count(*) FROM posts WHERE uri = $1`, subject.URI), "the unverified record must not be indexed") } @@ -405,32 +461,27 @@ func TestAcceptanceConsumer_FetchedRecordNamingAnotherCommunity_IsPermanentlyRef db := testkit.DB(t) ctx := context.Background() - base := time.Now().UnixMicro() + f := newRealRepoFixture(t, db) - const cid = "bafyreiaccwrongcommunity" - rkey := "accwrongcomm" - uri := accPostURI(rkey) + const elsewhere = accPrefix + "othercommunity" + insertBridgedCommunity(t, db, elsewhere, "otherrealcommunity.test", accAuthor) - srv := fakeAuthorPDS(t, accAuthor, func(w http.ResponseWriter, r *http.Request) { - // The record was submitted to a DIFFERENT community. - serveRecord(t, w, uri, cid, pv2Record("did:plc:accsomewhereelse", "submitted elsewhere", "body")) - }) - defer srv.Close() + // A genuine record, correctly signed, whose community field names someone + // else. The CID verifies; the CLAIM does not. Cross-community acceptance is + // the privileged fork/import flow and §10.2 is explicit that it is + // deliberately not built — until it is, a community accepting a post that + // names another is a community pulling someone else's content into its feed + // on its own say-so. + record := f.publish(t, elsewhere, "submitted to a different community") - f := newAccFixture(t, db, WithPostRecordFetcher(newFetcherAt(t, accAuthor, srv.URL))) + err := f.consumer.HandleEvent(ctx, + acceptanceEvent(accCommunity, record.URI, record.CID, testkit.TID(), time.Now().UnixMicro())) - err := f.consumer.HandleEvent(ctx, acceptanceEvent(accCommunity, uri, cid, testkit.TID(), base)) - - // Cross-community acceptance is the privileged fork/import flow, and §10.2 - // is explicit that it is deliberately NOT built: the data model supports it, - // the flow that exercises it is future scope. Until it exists, a community - // accepting a post that names someone else is a community pulling another - // community's content into its feed on its own say-so. - require.Error(t, err, "a community may not accept a post whose record names a different community — the fork/import flow is deliberately not built (§10.2)") + require.Error(t, err, "a community may not accept a post whose record names a different community") assert.ErrorIs(t, err, ErrPermanentEvent, - "the record's community field is immutable across updates (§3.1), so this can never become valid; retrying it is pure noise") + "the record's community field is immutable across updates (§3.1), so this can never become valid") - assert.Zero(t, countRows(t, db, `SELECT count(*) FROM posts WHERE uri = $1`, uri), + assert.Zero(t, countRows(t, db, `SELECT count(*) FROM posts WHERE uri = $1`, record.URI), "the refused record must not be indexed") } diff --git a/internal/atproto/jetstream/admission_durability_test.go b/internal/atproto/jetstream/admission_durability_test.go index b7a86f4..7616550 100644 --- a/internal/atproto/jetstream/admission_durability_test.go +++ b/internal/atproto/jetstream/admission_durability_test.go @@ -6,7 +6,7 @@ import ( "context" "errors" - "net/http" + "sync" "testing" "time" @@ -126,62 +126,74 @@ func TestAdmission_LoneRemovalDeleteExitsRemoved(t *testing.T) { "means the dead-letter redrive will never revisit this subject") } +// racingFetcher runs one action the first time a fetch is made, then delegates. +// +// It is how the interleaving below is made deterministic without weakening what +// the fetch itself does: the REAL fetcher still reads the real repo and still +// recomputes the CID. All this decides is WHEN the competing event lands, which +// in production is decided by two repos Jetstream carries in parallel. +type racingFetcher struct { + inner PostRecordFetcher + before func() + once sync.Once +} + +func (f *racingFetcher) FetchPost(ctx context.Context, postURI string) (*FetchedPost, error) { + f.once.Do(f.before) + return f.inner.FetchPost(ctx, postURI) +} + func TestAdmission_ConvergeMustNotRegressTheEvaluatedCID(t *testing.T) { t.Parallel() db := testkit.DB(t) ctx := context.Background() - base := time.Now().UnixMicro() - - rkey := "convergerace" - uri := accPostURI(rkey) - const pinnedCID = "bafyreiconvergev1" - const currentCID = "bafyreiconvergev2" - - var f accFixture - - // THE RACE, MADE DETERMINISTIC. The fetch is in flight when the post's own - // firehose event lands — which is not a rare interleaving, it is the normal - // one: the acceptance and the post are in different repos, so Jetstream - // parallelises them, and the fetch exists precisely because the post event - // has not arrived yet. Running the real event from inside the fetch handler - // puts the two in the exact order the race produces, every time. - srv := fakeAuthorPDS(t, accAuthor, func(w http.ResponseWriter, r *http.Request) { - require.NoError(t, f.consumer.HandleEvent(context.Background(), pv2Event( - accAuthor, "create", rkey, testkit.TID(), currentCID, base+500_000, - pv2Record(accCommunity, "the version that actually arrived", "newer body"), - )), "the racing post event must index cleanly") - - // The PDS answers with the version the acceptance pinned, which by now is - // the OLDER one. - serveRecord(t, w, uri, pinnedCID, pv2Record(accCommunity, "the version the acceptance pinned", "older body")) - }) - defer srv.Close() - - f = newAccFixture(t, db, WithPostRecordFetcher(newFetcherAt(t, accAuthor, srv.URL))) + f := newRealRepoFixture(t, db) + + // The repo holds ONE version, and the acceptance pins it. That version is + // what the fetch will legitimately verify and return. + record := f.publish(t, accCommunity, "the version the acceptance pinned") + const newerCID = "bafyreiconvergenewer" + + // The author edits. The AppView learns about the edit from the firehose + // while the fetch — started earlier, against the pre-edit repo — is still in + // flight. That is not an exotic interleaving: the acceptance and the post + // live in different repos, Jetstream parallelises across repos, and the + // fetch exists precisely because the post's own event had not arrived yet. + f.consumer = NewPostEventConsumer( + postgres.NewPostRepository(db), postgres.NewCommunityRepository(db), + newMockUserService(), db, + WithAdmissions(f.admissions), + WithDeletedAccounts(postgres.NewDeletedAccountRepository(db)), + WithPostRecordFetcher(&racingFetcher{ + inner: NewDevDirectPostFetcher(pinnedResolver(f.author.DID, f.pds.URL())), + before: func() { + require.NoError(t, f.consumer.HandleEvent(context.Background(), pv2Event( + f.author.DID, "create", record.RKey, testkit.TID(), newerCID, time.Now().UnixMicro(), + pv2Record(accCommunity, "the version that actually arrived", "newer body"), + )), "the racing post event must index cleanly") + }, + }), + ) - _ = f.consumer.HandleEvent(ctx, acceptanceEvent(accCommunity, uri, pinnedCID, testkit.TID(), base)) + _ = f.consumer.HandleEvent(ctx, + acceptanceEvent(accCommunity, record.URI, record.CID, testkit.TID(), time.Now().UnixMicro())) - row, err := f.admissions.Get(ctx, accCommunity, uri) + row, err := f.admissions.Get(ctx, accCommunity, record.URI) require.NoError(t, err) require.NotNil(t, row) - // evaluated_cid is what the NEXT decision judges. Regressed to the pinned - // CID, the engine evaluates content the author has already replaced, and — - // worse — an acceptance written against that verdict pins a version the - // AppView is no longer serving. The row would then report `accepted` for - // content nobody can see. - assertNullableStringPV2(t, currentCID, row.EvaluatedCID, - "the converge path wrote its fetched CID over a NEWER one that a real event had already recorded. "+ - "UpsertPending is last-write-wins, so a fetch that lost the race must not apply — the post event is the "+ - "authority on what content stands, and the fetch is a catch-up") - - assert.Equalf(t, posts.AdmissionStatusPendingReacceptance, row.Status, - "an acceptance pinning a CID the post no longer holds is pending_reacceptance, not accepted: the community "+ - "agreed to a version that has since been replaced") - - _, _, storedCID, _, _ := readPV2Post(t, db, uri) - assert.Equal(t, currentCID, storedCID, "the indexed post must hold the version its own event carried") + // evaluated_cid is what the NEXT decision judges. Regressed to the fetched + // version, the engine evaluates content the author has already replaced — + // and an acceptance written from that verdict pins a version the AppView is + // no longer serving, so the row reports `accepted` for content nobody sees. + assertNullableStringPV2(t, newerCID, row.EvaluatedCID, + "the converge path wrote its fetched CID over a NEWER one a real event had already recorded. UpsertPending is "+ + "last-write-wins, so a fetch that lost the race must not apply — the post event is the authority on what "+ + "content stands, and the fetch is only a catch-up") + + _, _, storedCID, _, _ := readPV2Post(t, db, record.URI) + assert.Equal(t, newerCID, storedCID, "the indexed post must hold the version its own event carried") } func TestAdmission_SurvivesAFailedUpsertAcrossRedelivery(t *testing.T) { diff --git a/tests/e2e/author_post_contract_test.go b/tests/e2e/author_post_contract_test.go index a634e81..d840a1c 100644 --- a/tests/e2e/author_post_contract_test.go +++ b/tests/e2e/author_post_contract_test.go @@ -51,6 +51,26 @@ import ( // to build. What IS true today is asserted; what task 7 owes is named where it // would otherwise look like a gap. // +// # THE POSITIVE FETCH ARC IS STRUCTURALLY T1-ONLY +// +// §5.4's direct fetch converges an acceptance whose subject the AppView has +// never indexed. That precondition cannot be staged in this tier, and the reason +// is the tier working as designed rather than a gap in it: every PDS write here +// reaches the AppView through Jetstream, so a postv2 record written to set up +// the arc IS delivered, and by the time the acceptance is written the subject is +// indexed — which is the ORDINARY path, not the fetch. Suppressing that delivery +// would mean either not writing the record (nothing to fetch) or reconfiguring +// the stack's consumers (rule 2 of this package: never instantiate a consumer). +// +// So what this file proves about acceptance-before-post is the NEGATIVE, which +// is stageable and is asserted below: an acceptance naming a post that exists +// nowhere admits nothing and keeps admitting nothing. The positive — fetch, +// recompute the CID from the repo's own blocks, index, accept — is proven at T1 +// against a real repo on the test PDS, in +// internal/atproto/jetstream/direct_fetch_verification_test.go (the component) +// and acceptance_consumer_test.go (the consumer wiring). That split is +// deliberate and permanent; do not read the absence here as missing coverage. +// // # THE rkey IS COMPUTED HERE RATHER THAN IMPORTED // // An acceptance's record key is the unpadded lowercase base32 encoding of the -- 2.51.2 From a05da8e369adea71e703c655afdd4f7783c5ea8f Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 04:12:01 -0700 Subject: [PATCH 15/17] fix(routes): name the getStatus limiter so its 429s are diagnosable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An unnamed limiter logs limiter="default", which is indistinguishable from the global one in the very situation the name is for: a client polling getStatus (the documented §7 UX) meeting a 429 and needing to know WHICH budget it spent, because the two have different numbers and different fixes. Co-Authored-By: Claude Fable 5 --- internal/api/routes/post.go | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/internal/api/routes/post.go b/internal/api/routes/post.go index dde40ae..a89452b 100644 --- a/internal/api/routes/post.go +++ b/internal/api/routes/post.go @@ -118,7 +118,11 @@ func RegisterPostRoutes( // ask about any post URI they can name. The budget is what bounds // enumeration of a community's rejected posts to a rate an operator notices. statusHandler := post.NewGetStatusHandler(cfg.statusService) - statusRateLimiter := middleware.NewRateLimiter(getStatusRateLimit, time.Minute) + // NAMED, so a 429 from here is distinguishable in the logs from the global + // limiter's. They have different budgets and different fixes, and an + // unnamed one logs "default" — which is exactly the diagnosis an operator + // staring at a rate-limited poller needs and would not get. + statusRateLimiter := middleware.NewNamedRateLimiter("postGetStatus", getStatusRateLimit, time.Minute) r.With(statusRateLimiter.Middleware). Get("/xrpc/social.coves.community.post.getStatus", statusHandler.HandleGetStatus) -- 2.51.2 From b3f77d9788c60d0f349e440cf7efd4601dc427f3 Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 04:13:31 -0700 Subject: [PATCH 16/17] fix(routes): getStatus budget above the polling rate the product prescribes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 60/min sat below the T2 tier's own 600ms cadence AND below §7's documented poll-until-accepted UX — the endpoint was refusing its own use case (caught at the merge gate: a legitimate 36s wait died at poll 61). 120/min keeps a poll-a-second client comfortable for two minutes; enumeration still surfaces at operator-visible rates. Co-Authored-By: Claude Fable 5 --- internal/api/routes/post.go | 14 +++++++++----- internal/api/routes/registration_test.go | 2 +- 2 files changed, 10 insertions(+), 6 deletions(-) diff --git a/internal/api/routes/post.go b/internal/api/routes/post.go index a89452b..bb60424 100644 --- a/internal/api/routes/post.go +++ b/internal/api/routes/post.go @@ -14,11 +14,15 @@ import ( // getStatusRateLimit is social.coves.community.post.getStatus' per-client // budget, per minute. // -// Sixty is chosen against the endpoint's own UX rather than copied from a -// neighbour: §7 has a client poll for the accepted transition, and a poll a -// second for a minute is comfortably inside this while a script enumerating a -// community's rejected posts is not. -const getStatusRateLimit = 60 +// Chosen against the endpoint's own UX rather than copied from a neighbour: +// §7 has a client poll for the accepted transition, so the budget must sit +// ABOVE any polling rate the product itself prescribes — an earlier 60 was +// below the T2 tier's own 600ms cadence (100/minute) and cut off the +// endpoint's documented use case mid-wait (caught at the task-5 merge gate: +// a legitimate wait died at poll 61). 120 keeps a poll-a-second client +// comfortable for two full minutes while a script enumerating a community's +// rejected posts still hits a rate an operator notices. +const getStatusRateLimit = 120 // PostRouteOption supplies a collaborator that only some of the post routes // need. diff --git a/internal/api/routes/registration_test.go b/internal/api/routes/registration_test.go index 5713051..e304ff4 100644 --- a/internal/api/routes/registration_test.go +++ b/internal/api/routes/registration_test.go @@ -200,7 +200,7 @@ var declaredRoutes = []declaredRoute{ // stranger can ask about any post URI they can name. The budget is what // bounds enumeration of a community's rejected posts to something an // operator would notice. - {http.MethodGet, "/xrpc/social.coves.community.post.getStatus", authNone, 60, false}, + {http.MethodGet, "/xrpc/social.coves.community.post.getStatus", authNone, 120, false}, // RegisterVoteRoutes — social.coves.feed.vote.* {http.MethodPost, "/xrpc/social.coves.feed.vote.create", authRequired, 0, false}, -- 2.51.2 From 657d65ec47068caf02b3b5ce7806fb09c535aaeb Mon Sep 17 00:00:00 2001 From: Bretton Date: Sat, 8 Aug 2026 04:17:26 -0700 Subject: [PATCH 17/17] test(routes): raise the budget-probe ceiling past the getStatus budget The adequacy tripwire fired exactly as designed when the budget rose to 120 past the 64-request probe ceiling. 128 restores the bracket: the tier's polling (needs >75) and the classifier's observability both fit. In-process probe cost is milliseconds; the comment now forbids sizing production budgets down for probe cheapness. Co-Authored-By: Claude Fable 5 --- internal/api/routes/registration_test.go | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/internal/api/routes/registration_test.go b/internal/api/routes/registration_test.go index e304ff4..1b262f6 100644 --- a/internal/api/routes/registration_test.go +++ b/internal/api/routes/registration_test.go @@ -432,8 +432,11 @@ func walkRoutes(t *testing.T, mux *chi.Mux) map[routeKey][]func(http.Handler) ht // middleware that has refused nothing after this many requests from one client // is reported as "not a rate limiter" — so a real limiter set above this // ceiling would be classified as absent. TestRoutes_BudgetProbeCeilingIsAdequate -// keeps that from happening quietly. -const maxBudgetProbe = 64 +// keeps that from happening quietly — and it works: it fired when getStatus's +// budget rose to 120 (above the then-64 ceiling), which is why this is 128. +// The probe is in-process httptest, so the extra requests cost milliseconds; +// never size a production budget down to make this probe cheaper. +const maxBudgetProbe = 128 type mwKind int