//! Records and conventions of the `app.userinput.*` lexicon. //! //! [userinput.app](https://userinput.app) is a feedback board with no server //! of its own: a board is an `app.userinput.space` record in its owner's PDS, //! and every post, reply and vote is a record in the repo of whoever made it. //! atgc uses it for `atgc report` — filing feedback about atgc means writing //! an `app.userinput.discussion` into *your* PDS that points at atgc's board, //! which is the same shape as a Tangled pull request pointing at a repo, and //! needs the same nothing from the board's owner. //! //! The lexicon has no published client, so the conventions here were read out //! of the web app and must be kept matching it: //! //! - a vote's record key **is the subject's record key** ([`vote_rkey`]), //! which is what makes a second vote overwrite the first instead of //! double-counting; //! - the web client upvotes its own posts as it makes them, so a post //! arriving without one reads as having one vote fewer than its author //! meant; //! - the web client wraps bare URLs into markdown links at write time //! ([`autolink`]), though its renderer autolinks as well, so this changes //! what is stored and not what is seen. //! //! What the board renders is not markdown and should not be written as if it //! were: the whole formatter is `**strong**`, `__u__` (underline, not //! emphasis), `~~del~~`, `*em*` and links. No fences, headings or lists — //! bodies render inside `whitespace-pre-wrap`, so plain line breaks are the //! only structure available. //! //! This module is deliberately self-contained — jacquard types and nothing //! of atgc's — because it is intended to be lifted out into a reusable crate //! once the shape has settled. Keep atgc-specific concerns (accounts, //! diagnostics, output) in `crate::cmd::report`. #![allow(dead_code)] use anyhow::{Result, bail}; use jacquard::common::DefaultStr; use jacquard::common::types::value::{Data, to_data}; use jacquard::types::string::Datetime; use serde::{Deserialize, Serialize}; /// app.userinput.discussion — one post on a board, in the poster's PDS. /// Generated from `lexicons/app/userinput/discussion.json`; build one with /// [`new_discussion`], not a struct literal. pub use userinput_lexicon::app_userinput::discussion::Discussion; /// app.userinput.upvote — a vote, keyed by [`vote_rkey`] so it is one per /// subject per voter by construction. Written with putRecord, not /// createRecord: overwriting the previous vote *is* the semantics. Generated /// from `lexicons/app/userinput/upvote.json`; its `subject` is the untyped /// `Data` a ref resolves to unvendored — build it with /// [`StrongRef::to_data`], the same as [`Discussion`]'s `space`. pub use userinput_lexicon::app_userinput::upvote::Upvote; pub const SPACE_NSID: &str = "app.userinput.space"; pub const DISCUSSION_NSID: &str = "app.userinput.discussion"; pub const REPLY_NSID: &str = "app.userinput.reply"; pub const UPVOTE_NSID: &str = "app.userinput.upvote"; pub const DOWNVOTE_NSID: &str = "app.userinput.downvote"; pub const EDIT_NSID: &str = "app.userinput.edit"; /// The OAuth scope `atgc report` needs: userinput.app's own published /// permission set ("Start discussions, reply, vote, and edit your own /// posts"). One token instead of five `repo:` scopes, and — unlike asking /// collection by collection — it is the app's definition of "a participant", /// so it widens or narrows with the lexicon rather than with atgc releases. pub const AUTH_BASIC_SCOPE: &str = "include:app.userinput.authBasic"; /// What a grant of [`AUTH_BASIC_SCOPE`] covers, exactly as the permission /// set record at `at://did:plc:uyixj57k6nmxrdj7pjs2ss5s/com.atproto.lexicon.schema/app.userinput.authBasic` /// spells it. The moderation collections (status, pin, hide, ban, lock, /// member) are deliberately not in it, and so not writable by atgc. pub const AUTH_BASIC_COLLECTIONS: &[&str] = &[ DISCUSSION_NSID, REPLY_NSID, UPVOTE_NSID, DOWNVOTE_NSID, EDIT_NSID, ]; /// `{uri, cid}` as the lexicon embeds it — a com.atproto strong ref, except /// the web client omits `cid` entirely when it has none rather than sending /// null, and this must serialize the same way. /// /// [`Discussion`]'s `space` and [`Upvote`]'s `subject` are generated as /// untyped `Data` rather than a real `com.atproto.repo.strongRef` type, /// on purpose: that generated type marks `cid` required, which a vote's /// subject always has but a discussion's space sometimes does not, and /// jacquard-codegen resolves a ref the same way for every field that names /// it — there is no per-field override. Vendoring the strongRef schema /// would make both required at once, which would be wrong for `space`. See /// `scripts/vendor-userinput-lexicons.sh`'s note on the same tradeoff. #[derive(Debug, Clone, Serialize, Deserialize)] pub struct StrongRef { pub uri: String, #[serde(skip_serializing_if = "Option::is_none")] pub cid: Option, } impl StrongRef { /// As [`Discussion::space`]/[`Upvote::subject`] want it. pub fn to_data(&self) -> Result> { to_data(self).map_err(|e| anyhow::anyhow!("{e}")) } } /// A post, with empty `body`/`tags` omitted rather than stored empty, and /// checked against the lexicon's title/body/tag-count limits before /// anything is written. `space` names the board; `tags` is zero or more of /// whatever the board declares — the web composer is a row of checkboxes, /// and an untagged post is ordinary rather than degenerate (about one in /// six on the flagship board). A client may require its posters to pick /// exactly one; that is a policy for the client to hold, and [`Discussion`] /// itself deliberately cannot express it. Absent `body` and `tags` are /// omitted, not null — the web client writes `void 0` fields away, and a /// record that round-trips through it must not grow nulls. /// /// A free function rather than `Discussion::new` — [`Discussion`] is /// generated by `scripts/vendor-userinput-lexicons.sh`, and Rust's orphan /// rule keeps atgc from adding an inherent method to a foreign type (the /// same shape of workaround as `pull_target` on the sh.tangled.* side). /// /// Taking tags as a slice rather than a single value is the point — the /// lexicon permits none and permits several, so nothing here decides how /// many a given board's posters ought to pick. /// /// The body is stored as written. The web client runs bare URLs through /// [`autolink`] first, but the *renderer* autolinks too, so the display is /// identical either way and rewriting someone's text to match a client's /// habit buys nothing. pub fn new_discussion( space: StrongRef, title: String, body: Option, tags: &[String], ) -> Result { use userinput_lexicon::LexiconSchema as _; let discussion = Discussion { body: body.filter(|b| !b.trim().is_empty()).map(Into::into), created_at: Datetime::now(), images: None, space: space.to_data()?, tags: (!tags.is_empty()).then(|| tags.iter().cloned().map(Into::into).collect()), title: title.into(), extra_data: None, }; discussion.validate().map_err(|e| anyhow::anyhow!("{e}"))?; Ok(discussion) } /// The tag values a board declares, out of its `app.userinput.space` record. /// /// A board spells each tag as a `{label, value}` object while a discussion /// stores plain strings; the value is what a discussion carries and what /// filtering matches on, so the value is the tag. Bare strings are read too, /// since the lexicon is unpublished and has been seen in both shapes. /// Anything unreadable is skipped rather than raised — a strange board /// degrades to "declares nothing", which is a case every client must handle /// anyway. pub fn declared_tags(space: &serde_json::Value) -> Vec { space["tags"] .as_array() .map(|tags| { tags.iter() .filter_map(|t| t.as_str().or_else(|| t["value"].as_str())) .map(str::to_string) .collect() }) .unwrap_or_default() } /// An at:// URI taken apart, no more than this lexicon needs: authority, /// collection, rkey, all required. Not a general AT-URI parser — no query, /// no fragment, no bare-authority form. #[derive(Debug, Clone, PartialEq)] pub struct ParsedUri { pub did: String, pub collection: String, pub rkey: String, } pub fn parse_at_uri(uri: &str) -> Result { let Some(rest) = uri.strip_prefix("at://") else { bail!("{uri} is not an at:// URI"); }; let mut parts = rest.split('/'); let (Some(did), Some(collection), Some(rkey), None) = (parts.next(), parts.next(), parts.next(), parts.next()) else { bail!("{uri} is not of the form at://did/collection/rkey"); }; if did.is_empty() || collection.is_empty() || rkey.is_empty() { bail!("{uri} is not of the form at://did/collection/rkey"); } Ok(ParsedUri { did: did.to_string(), collection: collection.to_string(), rkey: rkey.to_string(), }) } /// The record key a vote on `subject_uri` must have: the subject's own. /// /// This is the web client's dedup scheme — voting is putRecord at a key /// derived from the subject, so a mind changed twice is still one record — /// and a vote written under a fresh TID instead would count double on /// every tally the app computes. pub fn vote_rkey(subject_uri: &str) -> Result { Ok(parse_at_uri(subject_uri)?.rkey) } /// The web page of a discussion. userinput.app routes by author and record /// key, not by any id of its own — there is no id of its own to have. pub fn discussion_url(author_did: &str, rkey: &str) -> String { format!("https://userinput.app/d/{author_did}/{rkey}") } /// The web page of a board. pub fn space_url(owner_did: &str, rkey: &str) -> String { format!("https://userinput.app/s/{owner_did}/{rkey}") } /// Wrap bare URLs into markdown links, the way the web client does before /// it stores a body. /// /// A faithful port of the web app's normalization, kept bug-for-bug: text /// already inside a `[text](dest)` link is left alone, a URL starts only at /// the beginning of the body or after whitespace or `(`, runs until /// whitespace, `<` or `)`, and then sheds trailing `.,;:!?]` so a sentence /// ending in a link does not swallow its period. Anything cleverer here — /// skipping code fences, say — would make atgc's records render differently /// from the web's, which is worse than the shared blind spot. pub fn autolink(body: &str) -> String { let protected = protected_ranges(body); let mut out = String::with_capacity(body.len()); let mut i = 0; while i < body.len() { if let Some(&(start, end)) = protected.iter().find(|&&(s, _)| s == i) { out.push_str(&body[start..end]); i = end; continue; } let rest = &body[i..]; let scheme = ["https://", "http://"] .iter() .find(|s| rest.starts_with(*s)) .copied(); let preceded_ok = i == 0 || body[..i] .chars() .next_back() .is_some_and(|c| c.is_whitespace() || c == '('); if let Some(scheme) = scheme && preceded_ok { let after = &rest[scheme.len()..]; let taken = after .char_indices() .find(|&(_, c)| c.is_whitespace() || c == '<' || c == ')') .map_or(after.len(), |(j, _)| j); let url = &rest[..scheme.len() + taken]; let url = url.trim_end_matches(['.', ',', ';', ':', '!', '?', ']']); // The web's pattern needs two characters after the scheme; a // shorter remainder is not a URL to it, and so not to us. if url.len() >= scheme.len() + 2 { out.push_str(&format!("[{url}]({url})")); i += url.len(); continue; } } let c = rest.chars().next().expect("i < body.len()"); out.push(c); i += c.len_utf8(); } out } /// Byte ranges of existing `[text](dest)` markdown links, matched exactly as /// loosely as the web client matches them: any run without `]`, then any run /// without `)`. What is protected from [`autolink`] must be what the web /// protects, or the two would double-wrap each other's output. fn protected_ranges(body: &str) -> Vec<(usize, usize)> { let bytes = body.as_bytes(); let mut ranges = Vec::new(); let mut i = 0; while i < bytes.len() { if bytes[i] == b'[' { let close = bytes[i + 1..].iter().position(|&b| b == b']'); if let Some(off) = close { let bracket_end = i + 1 + off; if bytes.get(bracket_end + 1) == Some(&b'(') { let paren = bytes[bracket_end + 2..].iter().position(|&b| b == b')'); if let Some(off) = paren { let end = bracket_end + 2 + off + 1; ranges.push((i, end)); i = end; continue; } } } } i += 1; } ranges } #[cfg(test)] mod tests { use super::*; /// The serialized discussion is what the PDS validates and the web app /// renders, so its shape is pinned: the `$type` the lexicon names, /// camelCase field names, and — the part that has broken clients before — /// absent optionals *omitted*, because the web client writes `undefined` /// fields away and a null where it expects absence is a different record. #[test] fn a_discussion_serializes_to_the_shape_the_web_app_writes() { let space = StrongRef { uri: "at://did:plc:owner/app.userinput.space/3aaa".into(), cid: Some("bafyexample".into()), }; let full = new_discussion( space.clone(), "a title".into(), Some("a body".into()), &["bug".into()], ) .unwrap(); let json = serde_json::to_value(&full).unwrap(); assert_eq!(json["$type"], "app.userinput.discussion"); assert_eq!(json["space"]["cid"], "bafyexample"); assert_eq!(json["tags"][0], "bug"); assert!(json.get("createdAt").is_some(), "camelCase, not snake"); let bare = new_discussion(space, "a title".into(), None, &[]).unwrap(); let json = serde_json::to_value(&bare).unwrap(); assert!(json.get("body").is_none(), "absent, not null"); assert!(json.get("tags").is_none(), "absent, not null"); } /// The limits the lexicon actually states (title 600 bytes/300 graphemes, /// body 20000/10000, tags max 8) — read out of generated code rather than /// out of the minified web client, and enforced before the write rather /// than left to the PDS's 400. A title one grapheme over is refused with /// the field named; one at the limit is accepted. #[test] fn a_discussion_over_the_lexicons_limits_is_refused() { let space = StrongRef { uri: "at://did:plc:owner/app.userinput.space/3aaa".into(), cid: None, }; let err = new_discussion(space.clone(), "x".repeat(301), None, &[]) .unwrap_err() .to_string(); assert!(err.contains("title"), "{err}"); assert!(new_discussion(space.clone(), "x".repeat(300), None, &[]).is_ok()); let err = new_discussion(space.clone(), "t".into(), Some("x".repeat(10001)), &[]) .unwrap_err() .to_string(); assert!(err.contains("body"), "{err}"); let nine_tags: Vec = (0..9).map(|i| format!("tag{i}")).collect(); let err = new_discussion(space.clone(), "t".into(), None, &nine_tags) .unwrap_err() .to_string(); assert!(err.contains("tags"), "{err}"); let eight_tags: Vec = (0..8).map(|i| format!("tag{i}")).collect(); assert!(new_discussion(space, "t".into(), None, &eight_tags).is_ok()); } /// The composer's range, which is wider than any one client's rules: no /// tags, one, or several. A client that requires exactly one holds that /// requirement itself — if this constructor could not express an /// untagged post it could not describe records the app already stores, /// and one in six on the flagship board has no tags. #[test] fn a_post_carries_no_tags_one_tag_or_several() { let space = StrongRef { uri: "at://did:plc:owner/app.userinput.space/3aaa".into(), cid: None, }; let made = |tags: &[String]| { serde_json::to_value(new_discussion(space.clone(), "t".into(), None, tags).unwrap()) .unwrap() }; assert!(made(&[]).get("tags").is_none(), "absent, not an empty list"); assert_eq!( made(&["bug".to_string()])["tags"], serde_json::json!(["bug"]) ); assert_eq!( made(&["bug".to_string(), "feature".to_string()])["tags"], serde_json::json!(["bug", "feature"]) ); } /// A body is stored as written. The web client autolinks bare URLs on /// the way in, but its renderer autolinks too, so rewriting the text /// would change the record without changing what anyone sees — and a /// client that silently edits what someone typed had better be buying /// something with it. A body that is only whitespace is no body. #[test] fn the_constructor_stores_the_body_as_written() { let space = StrongRef { uri: "at://did:plc:owner/app.userinput.space/3aaa".into(), cid: None, }; let with = new_discussion( space.clone(), "t".into(), Some("see https://x.yz".into()), &[], ) .unwrap(); assert_eq!(with.body.as_deref(), Some("see https://x.yz")); assert!( new_discussion(space, "t".into(), Some(" \n ".into()), &[]) .unwrap() .body .is_none() ); } /// Both tag shapes seen in the wild: a board writes `{label, value}` /// objects, while a discussion's own tags are bare strings. The value is /// what a discussion stores, so the value is the tag. #[test] fn a_boards_declared_tags_are_read_from_either_shape() { let space = serde_json::json!({ "name": "b", "tags": [ {"label": "Bug", "value": "bug"}, "feature", {"label": "no value here"}, 7, ], }); assert_eq!(declared_tags(&space), vec!["bug", "feature"]); assert!(declared_tags(&serde_json::json!({"name": "b"})).is_empty()); } /// A vote's record key is the subject's — the whole dedup scheme, and /// the convention least recoverable from the lexicon itself, so it gets /// its own pin. #[test] fn a_vote_is_keyed_by_its_subject() { assert_eq!( vote_rkey("at://did:plc:x/app.userinput.discussion/3xyz").unwrap(), "3xyz" ); assert!(vote_rkey("https://userinput.app/d/x/y").is_err()); } #[test] fn at_uris_come_apart_and_bad_ones_are_refused() { assert_eq!( parse_at_uri("at://did:plc:x/app.userinput.space/3aaa").unwrap(), ParsedUri { did: "did:plc:x".into(), collection: "app.userinput.space".into(), rkey: "3aaa".into(), } ); for bad in [ "at://did:plc:x/app.userinput.space", "at://did:plc:x/a/b/c", "at://", "did:plc:x", ] { assert!(parse_at_uri(bad).is_err(), "{bad} should be refused"); } } /// The write-time normalization: what the web client would store, atgc /// stores, because the renderer never autolinks — a divergence here is /// visible on every report forever. #[test] fn bare_urls_are_wrapped_the_way_the_web_client_wraps_them() { assert_eq!( autolink("see https://example.com/x for more"), "see [https://example.com/x](https://example.com/x) for more" ); // Sentence punctuation stays outside the link. assert_eq!( autolink("read https://example.com/x."), "read [https://example.com/x](https://example.com/x)." ); // A URL opening the body counts as preceded correctly. assert_eq!(autolink("https://a.bc"), "[https://a.bc](https://a.bc)"); // Parenthesized: the `(` precedes, the `)` terminates. assert_eq!(autolink("(https://a.bc)"), "([https://a.bc](https://a.bc))"); } #[test] fn existing_links_and_non_urls_are_left_alone() { // Already a markdown link: wrapping again would nest brackets. let linked = "see [the site](https://example.com/x) for more"; assert_eq!(autolink(linked), linked); // Mid-word, no preceding boundary: not a URL start. assert_eq!(autolink("xhttps://a.bc"), "xhttps://a.bc"); // Too short after the scheme to be a URL to the web's pattern. assert_eq!(autolink("https:// and https://x"), "https:// and https://x"); // Nothing resembling a URL at all. assert_eq!(autolink("plain text"), "plain text"); assert_eq!(autolink(""), ""); } }