diff --git a/.env.example b/.env.example
new file mode 100644
index 0000000..1cd29b3
--- /dev/null
+++ b/.env.example
@@ -0,0 +1,2 @@
+DISCORD_TOKEN=your_token_here
+DISCORD_GUILD_ID=426513584428679168
diff --git a/.gitignore b/.gitignore
new file mode 100644
index 0000000..d8f53ab
--- /dev/null
+++ b/.gitignore
@@ -0,0 +1,28 @@
+# Dependencies
+node_modules/
+.venv/
+
+# Bun
+bun.lock
+
+# Environment / secrets
+.env
+.env.*
+!.env.example
+
+# Export output (large, reproducible)
+discord-export/
+
+# OS
+.DS_Store
+Thumbs.db
+
+# Editor
+*.swp
+*.swo
+*~
+
+# Personal notes
+holy_guacamole_braindump.md
+
+crush.md
diff --git a/README.md b/README.md
index 313b8d2..2304aca 100644
--- a/README.md
+++ b/README.md
@@ -1,8 +1,73 @@
# melty
-mcp server with fts5 full text search over sqlite
+MCP server with FTS5 full-text search over a SQLite export of a discord server.
-The canonical repo for this is hosted on tangled over at [`dunkirk.sh/melty`](https://tangled.org/dunkirk.sh/melty)
+The canonical repo is hosted on Tangled at [`dunkirk.sh/melty`](https://tangled.org/dunkirk.sh/melty).
+
+## Setup
+
+```bash
+bun install
+cp .env.example .env
+# Edit .env with your Discord token and guild ID
+```
+
+### Export Discord data
+
+```bash
+bun run export # Exports messages to discord-export/melty.db (resumable)
+```
+
+After export completes, build the thread index:
+
+```bash
+bun run src/build-threads.ts
+```
+
+### Run the MCP server
+
+```bash
+bun run mcp
+```
+
+Or configure it in your MCP client:
+
+```json
+{
+ "mcp": {
+ "melty-discord": {
+ "command": "bun",
+ "args": ["run", "./src/mcp-server.ts"],
+ "type": "stdio"
+ }
+ }
+}
+```
+
+## MCP Tools
+
+| Tool | Description |
+|------|-------------|
+| `search` | Full-text search across conversation threads (FTS5 with BM25 ranking). Start here. |
+| `get_thread` | Read full text of a thread by ID. Use after `search`. |
+| `channel_search` | Same as `search` but scoped to a specific channel. |
+| `get_messages` | Paginated raw message access within a channel, newest-first. |
+| `list_channels` | All channels with message counts, grouped by category. |
+| `stats` | Database statistics: total messages, threads, authors, date range. |
+| `query` | Raw read-only SQL against the database. Escape hatch. |
+
+### Search syntax
+
+Supports FTS5 query syntax:
+
+- `word` — matches any thread containing "word"
+- `"exact phrase"` — exact phrase match
+- `search*` — prefix match
+- `apple AND banana` — both terms required
+- `apple OR banana` — either term
+- `apple NOT banana` — exclude term
+
+Optional filters: `after`/`before` (ISO-8601 timestamps), `author` (name substring).
diff --git a/crush.json b/crush.json
new file mode 100644
index 0000000..ad0579c
--- /dev/null
+++ b/crush.json
@@ -0,0 +1,9 @@
+{
+ "mcp": {
+ "melty-discord": {
+ "command": "bun",
+ "args": ["run", "./src/mcp-server.ts"],
+ "type": "stdio"
+ }
+ }
+}
diff --git a/package.json b/package.json
new file mode 100644
index 0000000..0a58a64
--- /dev/null
+++ b/package.json
@@ -0,0 +1,16 @@
+{
+ "name": "melty-discord",
+ "private": true,
+ "type": "module",
+ "scripts": {
+ "export": "bun run src/export.ts",
+ "mcp": "bun run src/mcp-server.ts"
+ },
+ "dependencies": {
+ "@modelcontextprotocol/sdk": "^1.29.0",
+ "zod": "^4.4.3"
+ },
+ "devDependencies": {
+ "@types/bun": "latest"
+ }
+}
diff --git a/src/build-threads.ts b/src/build-threads.ts
new file mode 100644
index 0000000..ab5150c
--- /dev/null
+++ b/src/build-threads.ts
@@ -0,0 +1,233 @@
+#!/usr/bin/env bun
+/**
+ * Detect conversation threads from Discord messages.
+ *
+ * Threading strategy:
+ * 1. Explicit replies (reply_to_id) form the backbone
+ * 2. Messages without replies are grouped by channel + time proximity
+ * (new thread if gap > THRESHOLD between consecutive messages)
+ * 3. Reply chains are merged with their surrounding time-cluster
+ *
+ * Output: threads table in melty.db
+ */
+
+import { Database } from "bun:sqlite";
+import { join } from "node:path";
+
+const DB_PATH = process.env.MELTY_DB || join(import.meta.dir, "..", "discord-export", "melty.db");
+const TIME_GAP_THRESHOLD_MS = 30 * 60 * 1000; // 30 minutes = new thread
+
+const db = new Database(DB_PATH);
+
+// --- Schema ---
+
+db.exec(`
+ DROP TABLE IF EXISTS threads;
+ DROP TABLE IF EXISTS threads_fts;
+
+ CREATE TABLE threads (
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
+ channel_id TEXT NOT NULL,
+ channel_name TEXT,
+ category_name TEXT,
+ author_names TEXT, -- JSON array of unique authors
+ message_count INTEGER,
+ first_timestamp TEXT,
+ last_timestamp TEXT,
+ content TEXT, -- full concatenated text for FTS
+ messages_json TEXT -- structured JSON array of messages
+ );
+
+ CREATE INDEX idx_threads_channel ON threads(channel_id);
+`);
+
+// --- Load all messages ordered by channel + timestamp ---
+
+console.error("loading messages...");
+const messages = db.query(`
+ SELECT m.id, m.channel_id, m.author_name, m.timestamp, m.content, m.reply_to_id,
+ ch.name as channel_name, cat.name as category_name
+ FROM messages m
+ JOIN channels ch ON m.channel_id = ch.id
+ LEFT JOIN categories cat ON ch.category_id = cat.id
+ ORDER BY m.channel_id, m.timestamp
+`).all() as Array<{
+ id: string;
+ channel_id: string;
+ author_name: string | null;
+ timestamp: string;
+ content: string | null;
+ reply_to_id: string | null;
+ channel_name: string;
+ category_name: string | null;
+}>;
+
+console.error(`loaded ${messages.length} messages`);
+
+// --- Build reply chains ---
+
+// Map message ID -> root message ID of its reply chain
+const replyRoot = new Map();
+
+function findRoot(msgId: string): string {
+ if (!replyRoot.has(msgId)) return msgId;
+ let current = msgId;
+ const visited = new Set();
+ while (replyRoot.has(current) && !visited.has(current)) {
+ visited.add(current);
+ current = replyRoot.get(current)!;
+ }
+ return current;
+}
+
+for (const msg of messages) {
+ if (msg.reply_to_id) {
+ replyRoot.set(msg.id, findRoot(msg.reply_to_id));
+ }
+}
+
+// --- Thread detection ---
+
+interface Thread {
+ channel_id: string;
+ channel_name: string;
+ category_name: string | null;
+ authors: Set;
+ messages: typeof messages;
+ first_ts: string;
+ last_ts: string;
+}
+
+const threads: Thread[] = [];
+let currentThread: Thread | null = null;
+
+function flushThread() {
+ if (currentThread && currentThread.messages.length > 0) {
+ threads.push(currentThread);
+ }
+ currentThread = null;
+}
+
+function startThread(msg: typeof messages[0]) {
+ currentThread = {
+ channel_id: msg.channel_id,
+ channel_name: msg.channel_name,
+ category_name: msg.category_name,
+ authors: new Set(),
+ messages: [],
+ first_ts: msg.timestamp,
+ last_ts: msg.timestamp,
+ };
+}
+
+for (const msg of messages) {
+ const ts = new Date(msg.timestamp).getTime();
+ const prevTs = currentThread ? new Date(currentThread.last_ts).getTime() : 0;
+ const sameChannel = currentThread?.channel_id === msg.channel_id;
+ const withinGap = sameChannel && (ts - prevTs) < TIME_GAP_THRESHOLD_MS;
+
+ // Check if this message is a reply to something in the current thread
+ const isReplyToCurrentThread = msg.reply_to_id && currentThread?.messages.some(m => m.id === msg.reply_to_id);
+
+ // Check if this message's reply root is in the current thread
+ const root = findRoot(msg.id);
+ const rootInCurrentThread = currentThread?.messages.some(m => m.id === root || findRoot(m.id) === root);
+
+ if (!currentThread) {
+ startThread(msg);
+ } else if (!sameChannel) {
+ flushThread();
+ startThread(msg);
+ } else if (withinGap || isReplyToCurrentThread || rootInCurrentThread) {
+ // Continue current thread
+ } else {
+ // Time gap exceeded and no reply link → new thread
+ flushThread();
+ startThread(msg);
+ }
+
+ if (currentThread) {
+ currentThread.messages.push(msg);
+ currentThread.last_ts = msg.timestamp;
+ if (msg.author_name) currentThread.authors.add(msg.author_name);
+ }
+}
+flushThread();
+
+console.error(`detected ${threads.length} threads`);
+
+// --- Insert into DB ---
+
+const insertThread = db.query(`
+ INSERT INTO threads (channel_id, channel_name, category_name, author_names, message_count, first_timestamp, last_timestamp, content, messages_json)
+ VALUES ($chId, $chName, $catName, $authors, $count, $firstTs, $lastTs, $fullText, $msgsJson)
+`);
+
+db.transaction(() => {
+ for (const t of threads) {
+ // Build searchable content: concatenate all message text with author attribution
+ const fullText = t.messages
+ .map(m => `${m.author_name ?? "unknown"}: ${m.content ?? ""}`)
+ .join("\n");
+
+ // Structured messages for get_thread
+ const msgsJson = JSON.stringify(t.messages.map(m => ({
+ timestamp: m.timestamp,
+ author: m.author_name ?? "unknown",
+ content: m.content ?? "",
+ reply_to_id: m.reply_to_id ?? null,
+ })));
+
+ insertThread.run({
+ "$chId": t.channel_id,
+ "$chName": t.channel_name,
+ "$catName": t.category_name,
+ "$authors": JSON.stringify([...t.authors]),
+ "$count": t.messages.length,
+ "$firstTs": t.first_ts,
+ "$lastTs": t.last_ts,
+ "$fullText": fullText,
+ "$msgsJson": msgsJson,
+ });
+ }
+})();
+
+// --- Build FTS index on threads ---
+
+console.error("building FTS5 index on threads...");
+db.exec(`
+ CREATE VIRTUAL TABLE threads_fts USING fts5(
+ content,
+ channel_name,
+ category_name,
+ author_names,
+ content='threads',
+ content_rowid='id',
+ tokenize='porter unicode61'
+ );
+ INSERT INTO threads_fts(threads_fts) VALUES('rebuild');
+
+ CREATE TRIGGER IF NOT EXISTS threads_ai AFTER INSERT ON threads BEGIN
+ INSERT INTO threads_fts(rowid, content, channel_name, category_name, author_names)
+ VALUES (new.id, new.content, new.channel_name, new.category_name, new.author_names);
+ END;
+`);
+
+const ftsCount = (db.query("SELECT COUNT(*) as c FROM threads_fts").get() as { c: number }).c;
+console.error(`FTS5 index built: ${ftsCount} thread entries`);
+
+// --- Stats ---
+
+const stats = db.query(`
+ SELECT
+ COUNT(*) as total_threads,
+ AVG(message_count) as avg_messages,
+ MAX(message_count) as max_messages,
+ MIN(message_count) as min_messages
+ FROM threads
+`).get() as Record;
+
+console.error(`stats: ${JSON.stringify(stats)}`);
+
+db.close();
+console.error("done!");
diff --git a/src/export.ts b/src/export.ts
new file mode 100644
index 0000000..5bb6503
--- /dev/null
+++ b/src/export.ts
@@ -0,0 +1,356 @@
+#!/usr/bin/env bun
+/**
+ * Export all text channels from a Discord guild into SQLite.
+ * Supports resume, rate limiting, logging, and FTS5 indexing.
+ *
+ * Usage: DISCORD_TOKEN=xxx DISCORD_GUILD_ID=xxx bun run src/export.ts
+ */
+
+import { Database } from "bun:sqlite";
+import { mkdirSync, existsSync } from "node:fs";
+import { join } from "node:path";
+
+// --- Config ---
+
+const TOKEN = process.env.DISCORD_TOKEN;
+const GUILD_ID = process.env.DISCORD_GUILD_ID;
+const OUT_DIR = process.env.EXPORT_DIR || join(import.meta.dir, "..", "discord-export");
+const DB_PATH = join(OUT_DIR, "melty.db");
+const LOG_PATH = join(OUT_DIR, "export.log");
+const BASE = "https://discord.com/api/v10";
+const RATE_LIMIT_DELAY_MS = 100; // 10 req/s, well under 50 global cap
+
+if (!TOKEN) {
+ console.error("ERROR: DISCORD_TOKEN env var is required");
+ process.exit(1);
+}
+if (!GUILD_ID) {
+ console.error("ERROR: DISCORD_GUILD_ID env var is required");
+ process.exit(1);
+}
+
+// --- Logging ---
+
+mkdirSync(OUT_DIR, { recursive: true });
+const logFile = Bun.file(LOG_PATH);
+const logWriter = logFile.writer({ highWaterMark: 64 * 1024 });
+
+function log(level: string, msg: string) {
+ const line = `${new Date().toISOString()} [${level}] ${msg}\n`;
+ process.stdout.write(line);
+ logWriter.write(line);
+}
+
+// --- Rate Limiter ---
+
+class RateLimiter {
+ private lastRequestTime = 0;
+
+ async wait() {
+ const elapsed = Date.now() - this.lastRequestTime;
+ if (elapsed < RATE_LIMIT_DELAY_MS) {
+ await Bun.sleep(RATE_LIMIT_DELAY_MS - elapsed);
+ }
+ this.lastRequestTime = Date.now();
+ }
+}
+
+const limiter = new RateLimiter();
+
+// --- Discord API ---
+
+interface DiscordChannel {
+ id: string;
+ type: number;
+ name: string;
+ parent_id: string | null;
+ topic: string | null;
+ position: number;
+}
+
+interface DiscordMessage {
+ id: string;
+ timestamp: string;
+ content: string;
+ author: { id: string; username: string };
+ attachments: Array<{ filename: string; url: string }>;
+ embeds: Array<{ title?: string; description?: string; url?: string }>;
+ message_reference?: { message_id?: string };
+}
+
+async function apiGet(path: string): Promise {
+ const url = `${BASE}${path}`;
+
+ for (let attempt = 0; attempt < 5; attempt++) {
+ await limiter.wait();
+
+ const resp = await fetch(url, {
+ headers: { Authorization: TOKEN!, "User-Agent": "melty-export/1.0" },
+ });
+
+ if (resp.status === 429) {
+ const retryAfter = parseFloat(resp.headers.get("Retry-After") || "1");
+ log("WARN", `rate limited on ${path}, waiting ${retryAfter}s`);
+ await Bun.sleep(retryAfter * 1000);
+ continue;
+ }
+
+ if (!resp.ok) {
+ throw new Error(`HTTP ${resp.status}: ${resp.statusText}`);
+ }
+
+ // Proactive backoff from headers
+ const remaining = resp.headers.get("X-RateLimit-Remaining");
+ const reset = resp.headers.get("X-RateLimit-Reset");
+ if (remaining && parseInt(remaining) === 0 && reset) {
+ const wait = Math.max(0, parseFloat(reset) - Date.now() / 1000) + 0.1;
+ log("DEBUG", `rate limit bucket exhausted, preemptive wait ${wait.toFixed(1)}s`);
+ await Bun.sleep(wait * 1000);
+ }
+
+ return (await resp.json()) as T;
+ }
+
+ throw new Error(`failed after retries: ${path}`);
+}
+
+async function fetchAllMessages(channelId: string, channelName: string): Promise {
+ const messages: DiscordMessage[] = [];
+ let before: string | undefined;
+ let page = 0;
+
+ while (true) {
+ page++;
+ const params = new URLSearchParams({ limit: "100" });
+ if (before) params.set("before", before);
+
+ const batch = await apiGet(
+ `/channels/${channelId}/messages?${params}`,
+ );
+
+ if (!batch.length) break;
+ messages.push(...batch);
+ log("INFO", ` [${channelName}] page ${page}: ${batch.length} msgs (total: ${messages.length})`);
+
+ before = batch[batch.length - 1].id;
+ if (batch.length < 100) break;
+ }
+
+ return messages;
+}
+
+// --- Database ---
+
+function initDb(dbPath: string): Database {
+ const db = new Database(dbPath, { create: true });
+ db.exec(`
+ CREATE TABLE IF NOT EXISTS categories (
+ id TEXT PRIMARY KEY,
+ name TEXT NOT NULL,
+ position INTEGER
+ );
+ CREATE TABLE IF NOT EXISTS channels (
+ id TEXT PRIMARY KEY,
+ name TEXT NOT NULL,
+ category_id TEXT REFERENCES categories(id),
+ topic TEXT,
+ position INTEGER,
+ message_count INTEGER DEFAULT 0
+ );
+ CREATE TABLE IF NOT EXISTS messages (
+ id TEXT PRIMARY KEY,
+ channel_id TEXT NOT NULL REFERENCES channels(id),
+ author_id TEXT,
+ author_name TEXT,
+ timestamp TEXT NOT NULL,
+ content TEXT,
+ attachments TEXT,
+ embeds TEXT,
+ reply_to_id TEXT
+ );
+ CREATE TABLE IF NOT EXISTS export_state (
+ channel_id TEXT PRIMARY KEY,
+ status TEXT NOT NULL,
+ message_count INTEGER DEFAULT 0,
+ finished_at TEXT
+ );
+ CREATE INDEX IF NOT EXISTS idx_messages_channel ON messages(channel_id);
+ CREATE INDEX IF NOT EXISTS idx_messages_timestamp ON messages(timestamp);
+ CREATE INDEX IF NOT EXISTS idx_messages_author ON messages(author_id);
+ CREATE INDEX IF NOT EXISTS idx_channels_category ON channels(category_id);
+ `);
+ return db;
+}
+
+function getCompletedChannels(db: Database): Set {
+ const rows = db.query("SELECT channel_id FROM export_state WHERE status = 'done'").all() as Array<{ channel_id: string }>;
+ return new Set(rows.map((r) => r.channel_id));
+}
+
+function markChannelDone(db: Database, channelId: string, count: number) {
+ db.query(
+ "INSERT OR REPLACE INTO export_state (channel_id, status, message_count, finished_at) VALUES ($id, 'done', $count, datetime('now'))",
+ ).run({ $id: channelId, $count: count });
+}
+
+// --- Main ---
+
+async function main() {
+ log("INFO", "starting export");
+
+ log("INFO", "fetching channel list...");
+ const channels = await apiGet(`/guilds/${GUILD_ID}/channels`);
+
+ const categories = new Map();
+ for (const ch of channels) {
+ if (ch.type === 4) categories.set(ch.id, ch);
+ }
+
+ const textChannels = channels
+ .filter((ch) => ch.type === 0)
+ .sort((a, b) => {
+ const catA = categories.get(a.parent_id ?? "")?.name ?? "";
+ const catB = categories.get(b.parent_id ?? "")?.name ?? "";
+ if (catA !== catB) return catA.localeCompare(catB);
+ return (a.position ?? 999) - (b.position ?? 999);
+ });
+
+ const db = initDb(DB_PATH);
+
+ // Upsert categories
+ const upsertCat = db.query(
+ "INSERT OR REPLACE INTO categories (id, name, position) VALUES ($id, $name, $pos)",
+ );
+ for (const cat of categories.values()) {
+ upsertCat.run({ $id: cat.id, $name: cat.name, $pos: cat.position });
+ }
+
+ // Resume check
+ const completed = getCompletedChannels(db);
+ const pending = textChannels.filter((ch) => !completed.has(ch.id));
+ const skipped = textChannels.length - pending.length;
+
+ if (skipped > 0) {
+ log("INFO", `resuming: ${skipped} channels already done, ${pending.length} remaining`);
+ } else {
+ log("INFO", `fresh export: ${pending.length} channels`);
+ }
+
+ // Prepared statements for message insertion
+ const upsertChannel = db.query(
+ "INSERT OR REPLACE INTO channels (id, name, category_id, topic, position, message_count) VALUES ($id, $name, $catId, $topic, $pos, $count)",
+ );
+ const insertMessage = db.query(
+ "INSERT OR IGNORE INTO messages (id, channel_id, author_id, author_name, timestamp, content, attachments, embeds, reply_to_id) VALUES ($id, $chId, $authorId, $authorName, $ts, $content, $attachments, $embeds, $replyTo)",
+ );
+
+ let totalMsgs = 0;
+ let errors = 0;
+
+ for (let i = 0; i < pending.length; i++) {
+ const ch = pending[i];
+ const catName = categories.get(ch.parent_id ?? "")?.name ?? "uncategorized";
+
+ log("INFO", `[${i + 1}/${pending.length}] ${catName}/${ch.name}`);
+
+ try {
+ const messages = await fetchAllMessages(ch.id, ch.name);
+ const count = messages.length;
+ totalMsgs += count;
+ log("INFO", ` -> ${count} messages`);
+
+ upsertChannel.run({
+ $id: ch.id,
+ $name: ch.name,
+ $catId: ch.parent_id,
+ $topic: ch.topic,
+ $pos: ch.position,
+ $count,
+ });
+
+ // Batch insert messages using transaction
+ db.transaction(() => {
+ for (const msg of messages) {
+ const attachments = msg.attachments?.length
+ ? JSON.stringify(msg.attachments.map((a) => ({ filename: a.filename, url: a.url })))
+ : null;
+ const embeds = msg.embeds?.length
+ ? JSON.stringify(
+ msg.embeds.map((e) => ({
+ title: e.title,
+ description: (e.description ?? "").slice(0, 500),
+ url: e.url,
+ })),
+ )
+ : null;
+ const replyTo = msg.message_reference?.message_id ?? null;
+
+ insertMessage.run({
+ $id: msg.id,
+ $chId: ch.id,
+ $authorId: msg.author?.id,
+ $authorName: msg.author?.username,
+ $ts: msg.timestamp,
+ $content: msg.content ?? "",
+ $attachments,
+ $embeds,
+ $replyTo,
+ });
+ }
+ })();
+
+ markChannelDone(db, ch.id, count);
+ } catch (err) {
+ const errMsg = err instanceof Error ? err.message : String(err);
+ if (errMsg.includes("HTTP 403")) {
+ log("WARN", ` SKIP ${catName}/${ch.name}: no access (403)`);
+ markChannelDone(db, ch.id, 0);
+ } else {
+ log("ERROR", ` ERROR exporting ${catName}/${ch.name}: ${errMsg}`);
+ errors++;
+ }
+ }
+ }
+
+ // Build FTS5 index
+ log("INFO", "building FTS5 index...");
+ db.exec(`
+ DROP TABLE IF EXISTS messages_fts;
+ CREATE VIRTUAL TABLE messages_fts USING fts5(
+ content, author_name, channel_id,
+ content='messages', content_rowid='rowid',
+ tokenize='porter unicode61'
+ );
+ INSERT INTO messages_fts(messages_fts) VALUES('rebuild');
+
+ CREATE TRIGGER IF NOT EXISTS messages_ai AFTER INSERT ON messages BEGIN
+ INSERT INTO messages_fts(rowid, content, author_name, channel_id)
+ VALUES (new.rowid, new.content, new.author_name, new.channel_id);
+ END;
+ CREATE TRIGGER IF NOT EXISTS messages_ad AFTER DELETE ON messages BEGIN
+ INSERT INTO messages_fts(messages_fts, rowid, content, author_name, channel_id)
+ VALUES ('delete', old.rowid, old.content, old.author_name, old.channel_id);
+ END;
+ CREATE TRIGGER IF NOT EXISTS messages_au AFTER UPDATE ON messages BEGIN
+ INSERT INTO messages_fts(messages_fts, rowid, content, author_name, channel_id)
+ VALUES ('delete', old.rowid, old.content, old.author_name, old.channel_id);
+ INSERT INTO messages_fts(rowid, content, author_name, channel_id)
+ VALUES (new.rowid, new.content, new.author_name, new.channel_id);
+ END;
+ `);
+
+ const ftsCount = (db.query("SELECT COUNT(*) as c FROM messages_fts").get() as { c: number }).c;
+ log("INFO", `FTS5 index built: ${ftsCount} entries`);
+
+ db.close();
+ logWriter.flush();
+ logWriter.end();
+
+ log("INFO", `export complete: ${totalMsgs} messages across ${textChannels.length} channels (${errors} errors)`);
+ log("INFO", `database: ${DB_PATH}`);
+}
+
+main().catch((err) => {
+ log("ERROR", `fatal: ${err}`);
+ process.exit(1);
+});
diff --git a/src/mcp-server.ts b/src/mcp-server.ts
new file mode 100644
index 0000000..9669e6a
--- /dev/null
+++ b/src/mcp-server.ts
@@ -0,0 +1,422 @@
+import { Database } from "bun:sqlite";
+import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
+import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
+import { z } from "zod";
+
+const DB_PATH =
+ process.env.MELTY_DB ||
+ "/Users/kierank/code/personal/melty/discord-export/melty.db";
+
+let db: Database;
+try {
+ db = new Database(DB_PATH, { readonly: true });
+ db.exec("PRAGMA query_only = ON;");
+} catch (err) {
+ console.error(`Failed to open database at ${DB_PATH}: ${err}`);
+ process.exit(1);
+}
+
+// --- FTS5 syntax validation ---
+
+function validateFts5Query(query: string): string | null {
+ // Try a harmless MATCH against the FTS table to validate syntax
+ try {
+ db.query(
+ "SELECT 1 FROM threads_fts WHERE threads_fts MATCH $q LIMIT 0",
+ ).get({ $q: query });
+ return null; // valid
+ } catch (e) {
+ const msg = e instanceof Error ? e.message : String(e);
+ return `Invalid FTS5 query: ${msg}\n\nSyntax guide:\n word - matches any thread containing "word"\n "exact phrase" - matches the exact phrase\n search* - prefix match\n apple AND banana - both terms required\n apple OR banana - either term\n apple NOT banana - exclude term\n\nAvoid unbalanced quotes, trailing operators, or empty parentheses.`;
+ }
+}
+
+// --- Prepared Statements ---
+
+const stmtSearchThreads = db.query(`
+ SELECT
+ t.id,
+ t.channel_name,
+ t.category_name,
+ t.author_names,
+ t.message_count,
+ t.first_timestamp,
+ t.last_timestamp,
+ snippet(threads_fts, 0, '**', '**', '...', 64) as snippet
+ FROM threads_fts
+ JOIN threads t ON t.id = threads_fts.rowid
+ WHERE threads_fts MATCH $query
+ AND ($after IS NULL OR t.first_timestamp >= $after)
+ AND ($before IS NULL OR t.last_timestamp <= $before)
+ AND ($author IS NULL OR LOWER(t.author_names) LIKE '%' || LOWER($author) || '%')
+ ORDER BY bm25(threads_fts)
+ LIMIT $limit OFFSET $offset
+`);
+
+const stmtChannelSearch = db.query(`
+ SELECT
+ t.id,
+ t.channel_name,
+ t.category_name,
+ t.author_names,
+ t.message_count,
+ t.first_timestamp,
+ t.last_timestamp,
+ snippet(threads_fts, 0, '**', '**', '...', 64) as snippet
+ FROM threads_fts
+ JOIN threads t ON t.id = threads_fts.rowid
+ WHERE threads_fts MATCH $query
+ AND LOWER(t.channel_name) = LOWER($channel)
+ AND ($after IS NULL OR t.first_timestamp >= $after)
+ AND ($before IS NULL OR t.last_timestamp <= $before)
+ ORDER BY bm25(threads_fts)
+ LIMIT $limit OFFSET $offset
+`);
+
+const stmtGetThread = db.query(`
+ SELECT
+ t.id,
+ t.channel_name,
+ t.category_name,
+ t.author_names,
+ t.message_count,
+ t.first_timestamp,
+ t.last_timestamp,
+ t.messages_json
+ FROM threads t
+ WHERE t.id = $id
+`);
+
+const stmtListChannels = db.query(`
+ SELECT
+ ch.name,
+ COALESCE(cat.name, 'uncategorized') as category,
+ ch.message_count,
+ ch.topic
+ FROM channels ch
+ LEFT JOIN categories cat ON ch.category_id = cat.id
+ WHERE ch.message_count > 0
+ ORDER BY cat.name, ch.position
+`);
+
+const stmtGetMessages = db.query(`
+ SELECT
+ m.id,
+ m.timestamp,
+ m.author_name,
+ m.content,
+ m.reply_to_id,
+ m.attachments,
+ m.embeds
+ FROM messages m
+ WHERE m.channel_id = (SELECT id FROM channels WHERE LOWER(name) = LOWER($channel))
+ AND ($before IS NULL OR m.timestamp < $before)
+ ORDER BY m.timestamp DESC
+ LIMIT $limit
+`);
+
+const stmtStats = db.query(`
+ SELECT
+ (SELECT COUNT(*) FROM messages) as total_messages,
+ (SELECT COUNT(*) FROM threads) as total_threads,
+ (SELECT COUNT(*) FROM channels WHERE message_count > 0) as active_channels,
+ (SELECT COUNT(DISTINCT author_name) FROM messages WHERE author_name IS NOT NULL) as unique_authors,
+ (SELECT MIN(timestamp) FROM messages) as earliest,
+ (SELECT MAX(timestamp) FROM messages) as latest
+`);
+
+// --- Helpers ---
+
+function ok(data: unknown) {
+ return {
+ content: [{ type: "text" as const, text: JSON.stringify(data, null, 2) }],
+ };
+}
+
+function err(message: string) {
+ return { content: [{ type: "text" as const, text: message }], isError: true };
+}
+
+function safeQuery(fn: () => T, context: string): T | null {
+ try {
+ return fn();
+ } catch (e) {
+ console.error(`Error in ${context}: ${e}`);
+ return null;
+ }
+}
+
+// --- Server ---
+
+const server = new McpServer({
+ name: "melty-discord",
+ version: "3.0.0",
+});
+
+server.registerTool(
+ "search",
+ {
+ description: `Full-text search across conversation threads using FTS5 with BM25 ranking and porter stemming. Returns ranked threads with snippets.
+
+Query syntax:
+ word - matches any thread containing "word"
+ "exact phrase" - matches the exact phrase
+ search* - prefix match
+ apple AND banana - both terms required
+ apple OR banana - either term
+ apple NOT banana - exclude term
+
+Use get_thread to read the full conversation of a result.`,
+ inputSchema: {
+ query: z.string().max(200).describe("FTS5 search query"),
+ limit: z
+ .number()
+ .int()
+ .min(1)
+ .max(50)
+ .default(15)
+ .describe("Max results"),
+ offset: z.number().int().min(0).default(0).describe("Pagination offset"),
+ after: z
+ .string()
+ .optional()
+ .describe(
+ "Only threads starting after this ISO-8601 timestamp (e.g. 2025-01-01)",
+ ),
+ before: z
+ .string()
+ .optional()
+ .describe("Only threads ending before this ISO-8601 timestamp"),
+ author: z
+ .string()
+ .optional()
+ .describe("Filter to threads containing this author name"),
+ },
+ },
+ async ({ query, limit, offset, after, before, author }) => {
+ const syntaxError = validateFts5Query(query);
+ if (syntaxError) return err(syntaxError);
+
+ const results = safeQuery(
+ () =>
+ stmtSearchThreads.all({
+ $query: query,
+ $limit: limit,
+ $offset: offset,
+ $after: after ?? null,
+ $before: before ?? null,
+ $author: author ?? null,
+ }),
+ "search",
+ );
+ if (!results) return err("Search failed unexpectedly.");
+ return ok(results);
+ },
+);
+
+server.registerTool(
+ "channel_search",
+ {
+ description:
+ "Search threads within a specific channel. Same FTS5 syntax as search. Useful for drilling into a known channel.",
+ inputSchema: {
+ channel: z.string().describe("Channel name (case-insensitive)"),
+ query: z.string().max(200).describe("FTS5 search query"),
+ limit: z
+ .number()
+ .int()
+ .min(1)
+ .max(50)
+ .default(15)
+ .describe("Max results"),
+ offset: z.number().int().min(0).default(0).describe("Pagination offset"),
+ after: z
+ .string()
+ .optional()
+ .describe("Only threads starting after this ISO-8601 timestamp"),
+ before: z
+ .string()
+ .optional()
+ .describe("Only threads ending before this ISO-8601 timestamp"),
+ },
+ },
+ async ({ channel, query, limit, offset, after, before }) => {
+ const syntaxError = validateFts5Query(query);
+ if (syntaxError) return err(syntaxError);
+
+ const results = safeQuery(
+ () =>
+ stmtChannelSearch.all({
+ $query: query,
+ $channel: channel,
+ $limit: limit,
+ $offset: offset,
+ $after: after ?? null,
+ $before: before ?? null,
+ }),
+ "channel_search",
+ );
+ if (!results)
+ return err("Search failed. Check channel name and query syntax.");
+ return ok(results);
+ },
+);
+
+server.registerTool(
+ "get_thread",
+ {
+ description:
+ "Get the full text of a conversation thread by ID. Use after search to read complete discussions. Returns structured message array.",
+ inputSchema: {
+ id: z.number().int().describe("Thread ID from search results"),
+ },
+ },
+ async ({ id }) => {
+ const result = safeQuery(
+ () => stmtGetThread.get({ $id: id }),
+ "get_thread",
+ ) as Record | null;
+ if (!result) return err(`Thread ${id} not found`);
+
+ // Parse messages_json into structured array
+ let messages: unknown[] = [];
+ try {
+ messages = JSON.parse(result.messages_json as string);
+ } catch {
+ messages = [];
+ }
+
+ return ok({
+ id: result.id,
+ channel_name: result.channel_name,
+ category_name: result.category_name,
+ author_names: result.author_names,
+ message_count: result.message_count,
+ first_timestamp: result.first_timestamp,
+ last_timestamp: result.last_timestamp,
+ messages,
+ });
+ },
+);
+
+server.registerTool(
+ "list_channels",
+ {
+ description:
+ "List all channels with message counts, grouped by category. Useful for discovering what topics exist.",
+ inputSchema: {},
+ },
+ async () => {
+ const rows = safeQuery(
+ () => stmtListChannels.all(),
+ "list_channels",
+ ) as Array<{
+ name: string;
+ category: string;
+ message_count: number;
+ topic: string | null;
+ }> | null;
+ if (!rows) return err("Failed to list channels");
+
+ // Group by category
+ const grouped: Record<
+ string,
+ Array<{ name: string; message_count: number; topic: string | null }>
+ > = {};
+ for (const r of rows) {
+ if (!grouped[r.category]) grouped[r.category] = [];
+ grouped[r.category].push({
+ name: r.name,
+ message_count: r.message_count,
+ topic: r.topic,
+ });
+ }
+ return ok(grouped);
+ },
+);
+
+server.registerTool(
+ "get_messages",
+ {
+ description:
+ "Get raw messages from a specific channel, ordered newest-first. Use when thread-level search isn't granular enough. Supports pagination via 'before' timestamp.",
+ inputSchema: {
+ channel: z.string().describe("Channel name (case-insensitive)"),
+ limit: z
+ .number()
+ .int()
+ .min(1)
+ .max(100)
+ .default(50)
+ .describe("Max messages to return"),
+ before: z
+ .string()
+ .optional()
+ .describe(
+ "Return messages before this ISO-8601 timestamp (for pagination)",
+ ),
+ },
+ },
+ async ({ channel, limit, before }) => {
+ const results = safeQuery(
+ () =>
+ stmtGetMessages.all({
+ $channel: channel,
+ $limit: limit,
+ $before: before ?? null,
+ }),
+ "get_messages",
+ );
+ if (!results) return err("Failed to get messages. Check channel name.");
+ if ((results as unknown[]).length === 0)
+ return err(
+ `No messages found in channel "${channel}". Check spelling with list_channels.`,
+ );
+ return ok(results);
+ },
+);
+
+server.registerTool(
+ "stats",
+ {
+ description:
+ "Get overall database statistics: total messages, threads, channels, authors, date range.",
+ inputSchema: {},
+ },
+ async () => {
+ const result = safeQuery(() => stmtStats.get(), "stats");
+ if (!result) return err("Failed to get stats");
+ return ok(result);
+ },
+);
+
+server.registerTool(
+ "query",
+ {
+ description:
+ "Run a raw read-only SQL query against the database. Escape hatch for advanced analysis. Tables: messages, channels, categories, threads, threads_fts, messages_fts, export_state.",
+ inputSchema: {
+ sql: z.string().max(2000).describe("SQL SELECT query (read-only)"),
+ },
+ },
+ async ({ sql }) => {
+ const trimmed = sql.trim().toUpperCase();
+ if (!trimmed.startsWith("SELECT") && !trimmed.startsWith("WITH")) {
+ return err("Only SELECT and WITH queries are allowed");
+ }
+ const results = safeQuery(() => db.query(sql).all(), "raw_query");
+ if (!results) return err("Query failed. Check syntax.");
+ return ok(results);
+ },
+);
+
+async function main() {
+ const transport = new StdioServerTransport();
+ await server.connect(transport);
+ console.error("melty-discord MCP server v3 running on stdio");
+}
+
+main().catch((error) => {
+ console.error("Fatal error:", error);
+ process.exit(1);
+});
diff --git a/tsconfig.json b/tsconfig.json
new file mode 100644
index 0000000..801e41e
--- /dev/null
+++ b/tsconfig.json
@@ -0,0 +1,31 @@
+{
+ "compilerOptions": {
+ // Environment setup & latest features
+ "lib": ["ESNext"],
+ "target": "ESNext",
+ "module": "Preserve",
+ "moduleDetection": "force",
+ "jsx": "react-jsx",
+ "allowJs": true,
+ "types": ["bun"],
+
+ // Bundler mode
+ "moduleResolution": "bundler",
+ "allowImportingTsExtensions": true,
+ "verbatimModuleSyntax": true,
+ "noEmit": true,
+
+ // Best practices
+ "strict": true,
+ "skipLibCheck": true,
+ "noFallthroughCasesInSwitch": true,
+ "noUncheckedIndexedAccess": true,
+ "noImplicitOverride": true,
+
+ // Some stricter flags (disabled by default)
+ "noUnusedLocals": false,
+ "noUnusedParameters": false,
+ "noPropertyAccessFromIndexSignature": false
+ },
+ "include": ["src"]
+}