From 84650e2f4ff904f4b66987dbe546d472c3326112 Mon Sep 17 00:00:00 2001 From: Cameron Pfiffer Date: Tue, 24 Mar 2026 16:12:58 -0700 Subject: [PATCH] feat: add centralized ATProto annotation skill Copy the ATProtocol web annotation workflow into the central skills directory so it can be reused outside the agent-local .letta path. --- skills/atproto-annotations/SKILL.md | 110 ++++++ .../atproto-annotations/references/lexicon.md | 74 ++++ .../atproto-annotations/scripts/annotate.py | 330 ++++++++++++++++++ 3 files changed, 514 insertions(+) create mode 100644 skills/atproto-annotations/SKILL.md create mode 100644 skills/atproto-annotations/references/lexicon.md create mode 100644 skills/atproto-annotations/scripts/annotate.py diff --git a/skills/atproto-annotations/SKILL.md b/skills/atproto-annotations/SKILL.md new file mode 100644 index 0000000..6888456 --- /dev/null +++ b/skills/atproto-annotations/SKILL.md @@ -0,0 +1,110 @@ +--- +name: atproto-annotations +description: Write and read W3C Web Annotations on ATProtocol using the at.margin.annotation lexicon. Use when annotating URLs, writing reading notes on web pages, or building a public research trail. Supports single writes, batch writes via applyWrites, and reading annotations from any ATProtocol user. Triggers on annotate, annotation, margin, reading notes, or web annotation. +--- + +# ATProtocol Annotations + +Write and read W3C Web Annotations using the `at.margin.annotation` lexicon on ATProtocol. Annotations are public, portable, and tied to your ATProtocol identity. + +## Prerequisites + +- ATProtocol account (Bluesky) with an app password +- Set env var: `export BLUESKY_CO_APP_PASSWORD=your-app-password` +- **Critical**: `source .env` does NOT export. Use `export $(grep BLUESKY_CO_APP_PASSWORD ~/lettabot/.env)` before running. + +## Quick Operations + +### Write a single annotation + +```bash +python scripts/annotate.py write "https://example.com/article" "My observation about this article" +``` + +### Write with a text quote (anchored to specific passage) + +```bash +python scripts/annotate.py write "https://example.com/article" "This is significant because..." --quote "exact text from the page" +``` + +### Write with a motivation (default: commenting) + +```bash +python scripts/annotate.py write "https://example.com/article" "Key passage" --motivation highlighting +``` + +### Batch annotate (recommended for multiple annotations on one URL) + +Prepare a JSONL file with one annotation per line: + +```json +{"text": "First observation about this page"} +{"text": "Second observation", "quote": "anchored to this text"} +{"text": "A highlight", "motivation": "highlighting"} +``` + +Then run: + +```bash +python scripts/annotate.py batch "https://example.com/article" annotations.jsonl +``` + +Or pipe from stdin: + +```bash +echo '{"text": "Quick note"}' | python scripts/annotate.py batch "https://example.com/article" - +``` + +**Why batch**: Single annotations make 3 HTTP requests each (auth + title fetch + create). Batch makes 3 total regardless of count (1 auth + 1 title + 1 applyWrites). For 10 annotations: 3 requests vs 30. + +### List your annotations + +```bash +python scripts/annotate.py list --limit 20 +``` + +### Read another user's annotations + +```bash +python scripts/annotate.py read "handle.bsky.social" --limit 20 +python scripts/annotate.py read "did:plc:abc123" --limit 20 +``` + +Cross-PDS resolution is automatic. Works with handles on any PDS (bsky.social, custom PDS, etc). + +## Annotation Workflow + +When annotating a document or webpage: + +1. Fetch and read the full content first +2. Identify key observations, patterns, critiques +3. Write annotations as a JSONL file (one per line) +4. Batch-write all annotations in one call + +Each annotation should be a standalone observation. Use quotes to anchor annotations to specific passages when relevant. + +## Motivations + +W3C Web Annotation motivations supported: +- `commenting` (default) - general observations +- `highlighting` - marking important passages +- `describing` - describing what something is +- `classifying` - categorizing content +- `questioning` - raising questions about content + +## Record Format + +Annotations are stored as `at.margin.annotation` records in your ATProtocol PDS. Each record contains: +- `target.source` - the annotated URL +- `target.sourceHash` - SHA256 of the URL +- `target.title` - page title (auto-fetched) +- `target.selector` - optional TextQuoteSelector for anchoring +- `body.value` - annotation text +- `motivation` - W3C motivation type + +## Configuration + +Edit `scripts/annotate.py` constants to change account: +- `PDS` - PDS endpoint (default: bsky.social) +- `HANDLE` - your ATProtocol handle +- `PASSWORD` - reads from `BLUESKY_CO_APP_PASSWORD` env var diff --git a/skills/atproto-annotations/references/lexicon.md b/skills/atproto-annotations/references/lexicon.md new file mode 100644 index 0000000..ebd83e1 --- /dev/null +++ b/skills/atproto-annotations/references/lexicon.md @@ -0,0 +1,74 @@ +# at.margin.annotation Lexicon Reference + +The `at.margin.annotation` lexicon follows the W3C Web Annotation Data Model, adapted for ATProtocol. + +## Record Schema + +```json +{ + "$type": "at.margin.annotation", + "body": { + "format": "text/plain", + "value": "The annotation text content" + }, + "motivation": "commenting", + "target": { + "source": "https://example.com/page", + "sourceHash": "sha256-of-url", + "title": "Page Title (auto-fetched)", + "selector": { + "type": "TextQuoteSelector", + "exact": "quoted text from the page" + } + }, + "createdAt": "2026-02-10T23:00:00Z" +} +``` + +## ATProtocol Operations + +### Write (single record) +``` +POST /xrpc/com.atproto.repo.createRecord +{ + "repo": "did:plc:...", + "collection": "at.margin.annotation", + "record": { ... } +} +``` + +### Write (batch via applyWrites) +``` +POST /xrpc/com.atproto.repo.applyWrites +{ + "repo": "did:plc:...", + "writes": [ + { + "$type": "com.atproto.repo.applyWrites#create", + "collection": "at.margin.annotation", + "value": { ... } + }, + ... + ] +} +``` + +### Read (list records) +``` +GET /xrpc/com.atproto.repo.listRecords?repo=DID&collection=at.margin.annotation&limit=N +``` + +### Cross-PDS Resolution + +To read annotations from users on different PDS instances: +1. Resolve handle to DID via `com.atproto.identity.resolveHandle` +2. Resolve DID to PDS endpoint via `https://plc.directory/{did}` +3. Query records from the resolved PDS endpoint + +## Ecosystem + +- **margin.at** - Browser extension for creating annotations via UI +- **Semble** - AppView that aggregates annotations alongside bookmarks +- **Cosmik** - Network layer connecting ATProtocol knowledge tools + +Annotations created via this tool appear in margin.at and Semble alongside annotations created through their UIs. diff --git a/skills/atproto-annotations/scripts/annotate.py b/skills/atproto-annotations/scripts/annotate.py new file mode 100644 index 0000000..01575a1 --- /dev/null +++ b/skills/atproto-annotations/scripts/annotate.py @@ -0,0 +1,330 @@ +""" +Co's ATProtocol annotation tool. + +Write and read W3C Web Annotations using the at.margin.annotation lexicon. +Adapted from Central's tools/annotate.py for Co's account. + +Usage: + python annotate.py write "https://example.com" "My observation" + python annotate.py write "https://example.com" "Note" --quote "exact text" + python annotate.py write "https://example.com" "Note" --motivation highlighting + python annotate.py batch "https://example.com" annotations.jsonl + python annotate.py list [--limit 20] + python annotate.py read "https://handle-or-did" [--limit 20] + +Batch format (JSONL, one annotation per line): + {"text": "My observation"} + {"text": "Another note", "quote": "exact text from page"} + {"text": "A highlight", "motivation": "highlighting"} +""" + +import hashlib +import json +import os +import sys +from datetime import datetime, timezone +from urllib.request import Request, urlopen +from urllib.error import HTTPError + +PDS = "https://bsky.social" +HANDLE = "co.cameron.stream" +PASSWORD = os.environ.get("BLUESKY_CO_APP_PASSWORD", "") + + +def authenticate(handle=HANDLE, password=PASSWORD): + body = json.dumps({"identifier": handle, "password": password}).encode() + req = Request(f"{PDS}/xrpc/com.atproto.server.createSession", + data=body, headers={"Content-Type": "application/json"}) + with urlopen(req) as resp: + data = json.loads(resp.read()) + return data["did"], data["accessJwt"] + + +def resolve_handle(handle_or_did: str) -> str: + """Resolve a handle to a DID, or return DID as-is.""" + if handle_or_did.startswith("did:"): + return handle_or_did + req = Request( + f"{PDS}/xrpc/com.atproto.identity.resolveHandle?handle={handle_or_did}") + with urlopen(req) as resp: + return json.loads(resp.read())["did"] + + +def resolve_pds(did: str) -> str: + """Resolve a DID to its PDS endpoint via plc.directory.""" + try: + req = Request(f"https://plc.directory/{did}") + with urlopen(req, timeout=10) as resp: + doc = json.loads(resp.read()) + for svc in doc.get("service", []): + if svc.get("type") == "AtprotoPersonalDataServer": + return svc["serviceEndpoint"] + except Exception: + pass + return PDS # fallback to bsky.social + + +def fetch_page_title(url: str) -> str | None: + """Fetch page title for annotation target.""" + try: + req = Request(url, headers={"User-Agent": "Co/1.0 (ATProto agent)"}) + with urlopen(req, timeout=10) as resp: + text = resp.read().decode("utf-8", errors="replace") + start = text.find("") + end = text.find("") + if start != -1 and end != -1: + return text[start + 7:end].strip() + except Exception: + pass + return None + + +def build_annotation_record(url: str, body: str, source_hash: str, + title: str | None = None, + quote: str | None = None, + motivation: str = "commenting") -> dict: + """Build an annotation record dict (without writing it).""" + selector = None + if quote: + selector = { + "type": "TextQuoteSelector", + "exact": quote, + } + + target = { + "source": url, + "sourceHash": source_hash, + } + if title: + target["title"] = title + if selector: + target["selector"] = selector + + return { + "$type": "at.margin.annotation", + "body": { + "format": "text/plain", + "value": body, + }, + "motivation": motivation, + "target": target, + "createdAt": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), + } + + +def write_annotation(url: str, body: str, quote: str | None = None, + motivation: str = "commenting") -> str: + """Write a single annotation record to ATProtocol.""" + did, token = authenticate() + source_hash = hashlib.sha256(url.encode()).hexdigest() + title = fetch_page_title(url) + + record = build_annotation_record(url, body, source_hash, title, quote, motivation) + + req_body = json.dumps({ + "repo": did, "collection": "at.margin.annotation", "record": record + }).encode() + req = Request(f"{PDS}/xrpc/com.atproto.repo.createRecord", + data=req_body, headers={ + "Content-Type": "application/json", + "Authorization": f"Bearer {token}" + }) + with urlopen(req) as resp: + result = json.loads(resp.read()) + return result["uri"] + + +def batch_annotate(url: str, annotations: list[dict]) -> list[str]: + """Write multiple annotations to one URL in a single applyWrites call. + + annotations: list of dicts with keys: text, quote (optional), motivation (optional) + Returns list of rkeys created. + """ + did, token = authenticate() + source_hash = hashlib.sha256(url.encode()).hexdigest() + title = fetch_page_title(url) + + writes = [] + for ann in annotations: + record = build_annotation_record( + url=url, + body=ann["text"], + source_hash=source_hash, + title=title, + quote=ann.get("quote"), + motivation=ann.get("motivation", "commenting"), + ) + writes.append({ + "$type": "com.atproto.repo.applyWrites#create", + "collection": "at.margin.annotation", + "value": record, + }) + + req_body = json.dumps({ + "repo": did, + "writes": writes, + }).encode() + req = Request(f"{PDS}/xrpc/com.atproto.repo.applyWrites", + data=req_body, headers={ + "Content-Type": "application/json", + "Authorization": f"Bearer {token}" + }) + with urlopen(req) as resp: + result = json.loads(resp.read()) + + # applyWrites returns results with uri for each created record + uris = [] + for r in result.get("results", []): + uris.append(r.get("uri", "")) + return uris + + +def list_annotations(repo_did: str | None = None, limit: int = 10): + """List annotations from a repo (default: Co's own).""" + if repo_did is None: + did, _ = authenticate() + pds_endpoint = PDS + else: + did = resolve_handle(repo_did) + pds_endpoint = resolve_pds(did) + + req = Request( + f"{pds_endpoint}/xrpc/com.atproto.repo.listRecords?" + f"repo={did}&collection=at.margin.annotation&limit={limit}") + with urlopen(req) as resp: + records = json.loads(resp.read()).get("records", []) + + if not records: + print("No annotations found.") + return [] + + for rec in records: + val = rec.get("value", {}) + target = val.get("target", {}) + body_val = val.get("body", {}).get("value", "") + source = target.get("source", "?") + title = target.get("title", "") + selector = target.get("selector") + exact = selector.get("exact", "") if selector else "" + created = val.get("createdAt", "") + motivation = val.get("motivation", "") + + print(f"[{created}] {source}") + if title: + print(f" Title: {title}") + if exact: + print(f" Quote: \"{exact[:200]}\"") + print(f" Note: {body_val}") + if motivation and motivation != "commenting": + print(f" Motivation: {motivation}") + print(f" URI: {rec['uri']}") + print() + + return records + + +def main(): + args = sys.argv[1:] + + if not args or args[0] == "--help": + print(__doc__) + return + + cmd = args[0] + + if cmd == "write": + if len(args) < 3: + print("Usage: annotate.py write [--quote ] [--motivation ]") + return + + url = args[1] + body = args[2] + quote = None + motivation = "commenting" + + i = 3 + while i < len(args): + if args[i] == "--quote" and i + 1 < len(args): + quote = args[i + 1] + i += 2 + elif args[i] == "--motivation" and i + 1 < len(args): + motivation = args[i + 1] + i += 2 + else: + i += 1 + + uri = write_annotation(url, body, quote=quote, motivation=motivation) + print(f"Annotated: {uri}") + print(f" URL: {url}") + print(f" Body: {body}") + if quote: + print(f" Quote: \"{quote}\"") + + elif cmd == "batch": + if len(args) < 3: + print("Usage: annotate.py batch ") + print(" JSONL format: {\"text\": \"...\", \"quote\": \"...\", \"motivation\": \"...\"}") + print(" Use - for stdin") + return + + url = args[1] + source = args[2] + + annotations = [] + if source == "-": + for line in sys.stdin: + line = line.strip() + if line: + annotations.append(json.loads(line)) + else: + with open(source) as f: + for line in f: + line = line.strip() + if line: + annotations.append(json.loads(line)) + + if not annotations: + print("No annotations found in input.") + return + + print(f"Writing {len(annotations)} annotations to {url}...") + uris = batch_annotate(url, annotations) + print(f"Created {len(uris)} annotations:") + for i, uri in enumerate(uris): + text_preview = annotations[i]["text"][:80] + print(f" {i+1}. {uri}") + print(f" {text_preview}...") + + elif cmd == "list": + limit = 10 + i = 1 + while i < len(args): + if args[i] == "--limit" and i + 1 < len(args): + limit = int(args[i + 1]) + i += 2 + else: + i += 1 + list_annotations(limit=limit) + + elif cmd == "read": + if len(args) < 2: + print("Usage: annotate.py read [--limit N]") + return + handle = args[1] + limit = 10 + i = 2 + while i < len(args): + if args[i] == "--limit" and i + 1 < len(args): + limit = int(args[i + 1]) + i += 2 + else: + i += 1 + list_annotations(repo_did=handle, limit=limit) + + else: + print(f"Unknown command: {cmd}") + print(__doc__) + + +if __name__ == "__main__": + main() -- 2.51.2