diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..1cd29b3 --- /dev/null +++ b/.env.example @@ -0,0 +1,2 @@ +DISCORD_TOKEN=your_token_here +DISCORD_GUILD_ID=426513584428679168 diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..d8f53ab --- /dev/null +++ b/.gitignore @@ -0,0 +1,28 @@ +# Dependencies +node_modules/ +.venv/ + +# Bun +bun.lock + +# Environment / secrets +.env +.env.* +!.env.example + +# Export output (large, reproducible) +discord-export/ + +# OS +.DS_Store +Thumbs.db + +# Editor +*.swp +*.swo +*~ + +# Personal notes +holy_guacamole_braindump.md + +crush.md diff --git a/README.md b/README.md index 313b8d2..2304aca 100644 --- a/README.md +++ b/README.md @@ -1,8 +1,73 @@ # melty -mcp server with fts5 full text search over sqlite +MCP server with FTS5 full-text search over a SQLite export of a discord server. -The canonical repo for this is hosted on tangled over at [`dunkirk.sh/melty`](https://tangled.org/dunkirk.sh/melty) +The canonical repo is hosted on Tangled at [`dunkirk.sh/melty`](https://tangled.org/dunkirk.sh/melty). + +## Setup + +```bash +bun install +cp .env.example .env +# Edit .env with your Discord token and guild ID +``` + +### Export Discord data + +```bash +bun run export # Exports messages to discord-export/melty.db (resumable) +``` + +After export completes, build the thread index: + +```bash +bun run src/build-threads.ts +``` + +### Run the MCP server + +```bash +bun run mcp +``` + +Or configure it in your MCP client: + +```json +{ + "mcp": { + "melty-discord": { + "command": "bun", + "args": ["run", "./src/mcp-server.ts"], + "type": "stdio" + } + } +} +``` + +## MCP Tools + +| Tool | Description | +|------|-------------| +| `search` | Full-text search across conversation threads (FTS5 with BM25 ranking). Start here. | +| `get_thread` | Read full text of a thread by ID. Use after `search`. | +| `channel_search` | Same as `search` but scoped to a specific channel. | +| `get_messages` | Paginated raw message access within a channel, newest-first. | +| `list_channels` | All channels with message counts, grouped by category. | +| `stats` | Database statistics: total messages, threads, authors, date range. | +| `query` | Raw read-only SQL against the database. Escape hatch. | + +### Search syntax + +Supports FTS5 query syntax: + +- `word` — matches any thread containing "word" +- `"exact phrase"` — exact phrase match +- `search*` — prefix match +- `apple AND banana` — both terms required +- `apple OR banana` — either term +- `apple NOT banana` — exclude term + +Optional filters: `after`/`before` (ISO-8601 timestamps), `author` (name substring).

diff --git a/crush.json b/crush.json new file mode 100644 index 0000000..ad0579c --- /dev/null +++ b/crush.json @@ -0,0 +1,9 @@ +{ + "mcp": { + "melty-discord": { + "command": "bun", + "args": ["run", "./src/mcp-server.ts"], + "type": "stdio" + } + } +} diff --git a/package.json b/package.json new file mode 100644 index 0000000..0a58a64 --- /dev/null +++ b/package.json @@ -0,0 +1,16 @@ +{ + "name": "melty-discord", + "private": true, + "type": "module", + "scripts": { + "export": "bun run src/export.ts", + "mcp": "bun run src/mcp-server.ts" + }, + "dependencies": { + "@modelcontextprotocol/sdk": "^1.29.0", + "zod": "^4.4.3" + }, + "devDependencies": { + "@types/bun": "latest" + } +} diff --git a/src/build-threads.ts b/src/build-threads.ts new file mode 100644 index 0000000..ab5150c --- /dev/null +++ b/src/build-threads.ts @@ -0,0 +1,233 @@ +#!/usr/bin/env bun +/** + * Detect conversation threads from Discord messages. + * + * Threading strategy: + * 1. Explicit replies (reply_to_id) form the backbone + * 2. Messages without replies are grouped by channel + time proximity + * (new thread if gap > THRESHOLD between consecutive messages) + * 3. Reply chains are merged with their surrounding time-cluster + * + * Output: threads table in melty.db + */ + +import { Database } from "bun:sqlite"; +import { join } from "node:path"; + +const DB_PATH = process.env.MELTY_DB || join(import.meta.dir, "..", "discord-export", "melty.db"); +const TIME_GAP_THRESHOLD_MS = 30 * 60 * 1000; // 30 minutes = new thread + +const db = new Database(DB_PATH); + +// --- Schema --- + +db.exec(` + DROP TABLE IF EXISTS threads; + DROP TABLE IF EXISTS threads_fts; + + CREATE TABLE threads ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + channel_id TEXT NOT NULL, + channel_name TEXT, + category_name TEXT, + author_names TEXT, -- JSON array of unique authors + message_count INTEGER, + first_timestamp TEXT, + last_timestamp TEXT, + content TEXT, -- full concatenated text for FTS + messages_json TEXT -- structured JSON array of messages + ); + + CREATE INDEX idx_threads_channel ON threads(channel_id); +`); + +// --- Load all messages ordered by channel + timestamp --- + +console.error("loading messages..."); +const messages = db.query(` + SELECT m.id, m.channel_id, m.author_name, m.timestamp, m.content, m.reply_to_id, + ch.name as channel_name, cat.name as category_name + FROM messages m + JOIN channels ch ON m.channel_id = ch.id + LEFT JOIN categories cat ON ch.category_id = cat.id + ORDER BY m.channel_id, m.timestamp +`).all() as Array<{ + id: string; + channel_id: string; + author_name: string | null; + timestamp: string; + content: string | null; + reply_to_id: string | null; + channel_name: string; + category_name: string | null; +}>; + +console.error(`loaded ${messages.length} messages`); + +// --- Build reply chains --- + +// Map message ID -> root message ID of its reply chain +const replyRoot = new Map(); + +function findRoot(msgId: string): string { + if (!replyRoot.has(msgId)) return msgId; + let current = msgId; + const visited = new Set(); + while (replyRoot.has(current) && !visited.has(current)) { + visited.add(current); + current = replyRoot.get(current)!; + } + return current; +} + +for (const msg of messages) { + if (msg.reply_to_id) { + replyRoot.set(msg.id, findRoot(msg.reply_to_id)); + } +} + +// --- Thread detection --- + +interface Thread { + channel_id: string; + channel_name: string; + category_name: string | null; + authors: Set; + messages: typeof messages; + first_ts: string; + last_ts: string; +} + +const threads: Thread[] = []; +let currentThread: Thread | null = null; + +function flushThread() { + if (currentThread && currentThread.messages.length > 0) { + threads.push(currentThread); + } + currentThread = null; +} + +function startThread(msg: typeof messages[0]) { + currentThread = { + channel_id: msg.channel_id, + channel_name: msg.channel_name, + category_name: msg.category_name, + authors: new Set(), + messages: [], + first_ts: msg.timestamp, + last_ts: msg.timestamp, + }; +} + +for (const msg of messages) { + const ts = new Date(msg.timestamp).getTime(); + const prevTs = currentThread ? new Date(currentThread.last_ts).getTime() : 0; + const sameChannel = currentThread?.channel_id === msg.channel_id; + const withinGap = sameChannel && (ts - prevTs) < TIME_GAP_THRESHOLD_MS; + + // Check if this message is a reply to something in the current thread + const isReplyToCurrentThread = msg.reply_to_id && currentThread?.messages.some(m => m.id === msg.reply_to_id); + + // Check if this message's reply root is in the current thread + const root = findRoot(msg.id); + const rootInCurrentThread = currentThread?.messages.some(m => m.id === root || findRoot(m.id) === root); + + if (!currentThread) { + startThread(msg); + } else if (!sameChannel) { + flushThread(); + startThread(msg); + } else if (withinGap || isReplyToCurrentThread || rootInCurrentThread) { + // Continue current thread + } else { + // Time gap exceeded and no reply link → new thread + flushThread(); + startThread(msg); + } + + if (currentThread) { + currentThread.messages.push(msg); + currentThread.last_ts = msg.timestamp; + if (msg.author_name) currentThread.authors.add(msg.author_name); + } +} +flushThread(); + +console.error(`detected ${threads.length} threads`); + +// --- Insert into DB --- + +const insertThread = db.query(` + INSERT INTO threads (channel_id, channel_name, category_name, author_names, message_count, first_timestamp, last_timestamp, content, messages_json) + VALUES ($chId, $chName, $catName, $authors, $count, $firstTs, $lastTs, $fullText, $msgsJson) +`); + +db.transaction(() => { + for (const t of threads) { + // Build searchable content: concatenate all message text with author attribution + const fullText = t.messages + .map(m => `${m.author_name ?? "unknown"}: ${m.content ?? ""}`) + .join("\n"); + + // Structured messages for get_thread + const msgsJson = JSON.stringify(t.messages.map(m => ({ + timestamp: m.timestamp, + author: m.author_name ?? "unknown", + content: m.content ?? "", + reply_to_id: m.reply_to_id ?? null, + }))); + + insertThread.run({ + "$chId": t.channel_id, + "$chName": t.channel_name, + "$catName": t.category_name, + "$authors": JSON.stringify([...t.authors]), + "$count": t.messages.length, + "$firstTs": t.first_ts, + "$lastTs": t.last_ts, + "$fullText": fullText, + "$msgsJson": msgsJson, + }); + } +})(); + +// --- Build FTS index on threads --- + +console.error("building FTS5 index on threads..."); +db.exec(` + CREATE VIRTUAL TABLE threads_fts USING fts5( + content, + channel_name, + category_name, + author_names, + content='threads', + content_rowid='id', + tokenize='porter unicode61' + ); + INSERT INTO threads_fts(threads_fts) VALUES('rebuild'); + + CREATE TRIGGER IF NOT EXISTS threads_ai AFTER INSERT ON threads BEGIN + INSERT INTO threads_fts(rowid, content, channel_name, category_name, author_names) + VALUES (new.id, new.content, new.channel_name, new.category_name, new.author_names); + END; +`); + +const ftsCount = (db.query("SELECT COUNT(*) as c FROM threads_fts").get() as { c: number }).c; +console.error(`FTS5 index built: ${ftsCount} thread entries`); + +// --- Stats --- + +const stats = db.query(` + SELECT + COUNT(*) as total_threads, + AVG(message_count) as avg_messages, + MAX(message_count) as max_messages, + MIN(message_count) as min_messages + FROM threads +`).get() as Record; + +console.error(`stats: ${JSON.stringify(stats)}`); + +db.close(); +console.error("done!"); diff --git a/src/export.ts b/src/export.ts new file mode 100644 index 0000000..5bb6503 --- /dev/null +++ b/src/export.ts @@ -0,0 +1,356 @@ +#!/usr/bin/env bun +/** + * Export all text channels from a Discord guild into SQLite. + * Supports resume, rate limiting, logging, and FTS5 indexing. + * + * Usage: DISCORD_TOKEN=xxx DISCORD_GUILD_ID=xxx bun run src/export.ts + */ + +import { Database } from "bun:sqlite"; +import { mkdirSync, existsSync } from "node:fs"; +import { join } from "node:path"; + +// --- Config --- + +const TOKEN = process.env.DISCORD_TOKEN; +const GUILD_ID = process.env.DISCORD_GUILD_ID; +const OUT_DIR = process.env.EXPORT_DIR || join(import.meta.dir, "..", "discord-export"); +const DB_PATH = join(OUT_DIR, "melty.db"); +const LOG_PATH = join(OUT_DIR, "export.log"); +const BASE = "https://discord.com/api/v10"; +const RATE_LIMIT_DELAY_MS = 100; // 10 req/s, well under 50 global cap + +if (!TOKEN) { + console.error("ERROR: DISCORD_TOKEN env var is required"); + process.exit(1); +} +if (!GUILD_ID) { + console.error("ERROR: DISCORD_GUILD_ID env var is required"); + process.exit(1); +} + +// --- Logging --- + +mkdirSync(OUT_DIR, { recursive: true }); +const logFile = Bun.file(LOG_PATH); +const logWriter = logFile.writer({ highWaterMark: 64 * 1024 }); + +function log(level: string, msg: string) { + const line = `${new Date().toISOString()} [${level}] ${msg}\n`; + process.stdout.write(line); + logWriter.write(line); +} + +// --- Rate Limiter --- + +class RateLimiter { + private lastRequestTime = 0; + + async wait() { + const elapsed = Date.now() - this.lastRequestTime; + if (elapsed < RATE_LIMIT_DELAY_MS) { + await Bun.sleep(RATE_LIMIT_DELAY_MS - elapsed); + } + this.lastRequestTime = Date.now(); + } +} + +const limiter = new RateLimiter(); + +// --- Discord API --- + +interface DiscordChannel { + id: string; + type: number; + name: string; + parent_id: string | null; + topic: string | null; + position: number; +} + +interface DiscordMessage { + id: string; + timestamp: string; + content: string; + author: { id: string; username: string }; + attachments: Array<{ filename: string; url: string }>; + embeds: Array<{ title?: string; description?: string; url?: string }>; + message_reference?: { message_id?: string }; +} + +async function apiGet(path: string): Promise { + const url = `${BASE}${path}`; + + for (let attempt = 0; attempt < 5; attempt++) { + await limiter.wait(); + + const resp = await fetch(url, { + headers: { Authorization: TOKEN!, "User-Agent": "melty-export/1.0" }, + }); + + if (resp.status === 429) { + const retryAfter = parseFloat(resp.headers.get("Retry-After") || "1"); + log("WARN", `rate limited on ${path}, waiting ${retryAfter}s`); + await Bun.sleep(retryAfter * 1000); + continue; + } + + if (!resp.ok) { + throw new Error(`HTTP ${resp.status}: ${resp.statusText}`); + } + + // Proactive backoff from headers + const remaining = resp.headers.get("X-RateLimit-Remaining"); + const reset = resp.headers.get("X-RateLimit-Reset"); + if (remaining && parseInt(remaining) === 0 && reset) { + const wait = Math.max(0, parseFloat(reset) - Date.now() / 1000) + 0.1; + log("DEBUG", `rate limit bucket exhausted, preemptive wait ${wait.toFixed(1)}s`); + await Bun.sleep(wait * 1000); + } + + return (await resp.json()) as T; + } + + throw new Error(`failed after retries: ${path}`); +} + +async function fetchAllMessages(channelId: string, channelName: string): Promise { + const messages: DiscordMessage[] = []; + let before: string | undefined; + let page = 0; + + while (true) { + page++; + const params = new URLSearchParams({ limit: "100" }); + if (before) params.set("before", before); + + const batch = await apiGet( + `/channels/${channelId}/messages?${params}`, + ); + + if (!batch.length) break; + messages.push(...batch); + log("INFO", ` [${channelName}] page ${page}: ${batch.length} msgs (total: ${messages.length})`); + + before = batch[batch.length - 1].id; + if (batch.length < 100) break; + } + + return messages; +} + +// --- Database --- + +function initDb(dbPath: string): Database { + const db = new Database(dbPath, { create: true }); + db.exec(` + CREATE TABLE IF NOT EXISTS categories ( + id TEXT PRIMARY KEY, + name TEXT NOT NULL, + position INTEGER + ); + CREATE TABLE IF NOT EXISTS channels ( + id TEXT PRIMARY KEY, + name TEXT NOT NULL, + category_id TEXT REFERENCES categories(id), + topic TEXT, + position INTEGER, + message_count INTEGER DEFAULT 0 + ); + CREATE TABLE IF NOT EXISTS messages ( + id TEXT PRIMARY KEY, + channel_id TEXT NOT NULL REFERENCES channels(id), + author_id TEXT, + author_name TEXT, + timestamp TEXT NOT NULL, + content TEXT, + attachments TEXT, + embeds TEXT, + reply_to_id TEXT + ); + CREATE TABLE IF NOT EXISTS export_state ( + channel_id TEXT PRIMARY KEY, + status TEXT NOT NULL, + message_count INTEGER DEFAULT 0, + finished_at TEXT + ); + CREATE INDEX IF NOT EXISTS idx_messages_channel ON messages(channel_id); + CREATE INDEX IF NOT EXISTS idx_messages_timestamp ON messages(timestamp); + CREATE INDEX IF NOT EXISTS idx_messages_author ON messages(author_id); + CREATE INDEX IF NOT EXISTS idx_channels_category ON channels(category_id); + `); + return db; +} + +function getCompletedChannels(db: Database): Set { + const rows = db.query("SELECT channel_id FROM export_state WHERE status = 'done'").all() as Array<{ channel_id: string }>; + return new Set(rows.map((r) => r.channel_id)); +} + +function markChannelDone(db: Database, channelId: string, count: number) { + db.query( + "INSERT OR REPLACE INTO export_state (channel_id, status, message_count, finished_at) VALUES ($id, 'done', $count, datetime('now'))", + ).run({ $id: channelId, $count: count }); +} + +// --- Main --- + +async function main() { + log("INFO", "starting export"); + + log("INFO", "fetching channel list..."); + const channels = await apiGet(`/guilds/${GUILD_ID}/channels`); + + const categories = new Map(); + for (const ch of channels) { + if (ch.type === 4) categories.set(ch.id, ch); + } + + const textChannels = channels + .filter((ch) => ch.type === 0) + .sort((a, b) => { + const catA = categories.get(a.parent_id ?? "")?.name ?? ""; + const catB = categories.get(b.parent_id ?? "")?.name ?? ""; + if (catA !== catB) return catA.localeCompare(catB); + return (a.position ?? 999) - (b.position ?? 999); + }); + + const db = initDb(DB_PATH); + + // Upsert categories + const upsertCat = db.query( + "INSERT OR REPLACE INTO categories (id, name, position) VALUES ($id, $name, $pos)", + ); + for (const cat of categories.values()) { + upsertCat.run({ $id: cat.id, $name: cat.name, $pos: cat.position }); + } + + // Resume check + const completed = getCompletedChannels(db); + const pending = textChannels.filter((ch) => !completed.has(ch.id)); + const skipped = textChannels.length - pending.length; + + if (skipped > 0) { + log("INFO", `resuming: ${skipped} channels already done, ${pending.length} remaining`); + } else { + log("INFO", `fresh export: ${pending.length} channels`); + } + + // Prepared statements for message insertion + const upsertChannel = db.query( + "INSERT OR REPLACE INTO channels (id, name, category_id, topic, position, message_count) VALUES ($id, $name, $catId, $topic, $pos, $count)", + ); + const insertMessage = db.query( + "INSERT OR IGNORE INTO messages (id, channel_id, author_id, author_name, timestamp, content, attachments, embeds, reply_to_id) VALUES ($id, $chId, $authorId, $authorName, $ts, $content, $attachments, $embeds, $replyTo)", + ); + + let totalMsgs = 0; + let errors = 0; + + for (let i = 0; i < pending.length; i++) { + const ch = pending[i]; + const catName = categories.get(ch.parent_id ?? "")?.name ?? "uncategorized"; + + log("INFO", `[${i + 1}/${pending.length}] ${catName}/${ch.name}`); + + try { + const messages = await fetchAllMessages(ch.id, ch.name); + const count = messages.length; + totalMsgs += count; + log("INFO", ` -> ${count} messages`); + + upsertChannel.run({ + $id: ch.id, + $name: ch.name, + $catId: ch.parent_id, + $topic: ch.topic, + $pos: ch.position, + $count, + }); + + // Batch insert messages using transaction + db.transaction(() => { + for (const msg of messages) { + const attachments = msg.attachments?.length + ? JSON.stringify(msg.attachments.map((a) => ({ filename: a.filename, url: a.url }))) + : null; + const embeds = msg.embeds?.length + ? JSON.stringify( + msg.embeds.map((e) => ({ + title: e.title, + description: (e.description ?? "").slice(0, 500), + url: e.url, + })), + ) + : null; + const replyTo = msg.message_reference?.message_id ?? null; + + insertMessage.run({ + $id: msg.id, + $chId: ch.id, + $authorId: msg.author?.id, + $authorName: msg.author?.username, + $ts: msg.timestamp, + $content: msg.content ?? "", + $attachments, + $embeds, + $replyTo, + }); + } + })(); + + markChannelDone(db, ch.id, count); + } catch (err) { + const errMsg = err instanceof Error ? err.message : String(err); + if (errMsg.includes("HTTP 403")) { + log("WARN", ` SKIP ${catName}/${ch.name}: no access (403)`); + markChannelDone(db, ch.id, 0); + } else { + log("ERROR", ` ERROR exporting ${catName}/${ch.name}: ${errMsg}`); + errors++; + } + } + } + + // Build FTS5 index + log("INFO", "building FTS5 index..."); + db.exec(` + DROP TABLE IF EXISTS messages_fts; + CREATE VIRTUAL TABLE messages_fts USING fts5( + content, author_name, channel_id, + content='messages', content_rowid='rowid', + tokenize='porter unicode61' + ); + INSERT INTO messages_fts(messages_fts) VALUES('rebuild'); + + CREATE TRIGGER IF NOT EXISTS messages_ai AFTER INSERT ON messages BEGIN + INSERT INTO messages_fts(rowid, content, author_name, channel_id) + VALUES (new.rowid, new.content, new.author_name, new.channel_id); + END; + CREATE TRIGGER IF NOT EXISTS messages_ad AFTER DELETE ON messages BEGIN + INSERT INTO messages_fts(messages_fts, rowid, content, author_name, channel_id) + VALUES ('delete', old.rowid, old.content, old.author_name, old.channel_id); + END; + CREATE TRIGGER IF NOT EXISTS messages_au AFTER UPDATE ON messages BEGIN + INSERT INTO messages_fts(messages_fts, rowid, content, author_name, channel_id) + VALUES ('delete', old.rowid, old.content, old.author_name, old.channel_id); + INSERT INTO messages_fts(rowid, content, author_name, channel_id) + VALUES (new.rowid, new.content, new.author_name, new.channel_id); + END; + `); + + const ftsCount = (db.query("SELECT COUNT(*) as c FROM messages_fts").get() as { c: number }).c; + log("INFO", `FTS5 index built: ${ftsCount} entries`); + + db.close(); + logWriter.flush(); + logWriter.end(); + + log("INFO", `export complete: ${totalMsgs} messages across ${textChannels.length} channels (${errors} errors)`); + log("INFO", `database: ${DB_PATH}`); +} + +main().catch((err) => { + log("ERROR", `fatal: ${err}`); + process.exit(1); +}); diff --git a/src/mcp-server.ts b/src/mcp-server.ts new file mode 100644 index 0000000..9669e6a --- /dev/null +++ b/src/mcp-server.ts @@ -0,0 +1,422 @@ +import { Database } from "bun:sqlite"; +import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js"; +import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js"; +import { z } from "zod"; + +const DB_PATH = + process.env.MELTY_DB || + "/Users/kierank/code/personal/melty/discord-export/melty.db"; + +let db: Database; +try { + db = new Database(DB_PATH, { readonly: true }); + db.exec("PRAGMA query_only = ON;"); +} catch (err) { + console.error(`Failed to open database at ${DB_PATH}: ${err}`); + process.exit(1); +} + +// --- FTS5 syntax validation --- + +function validateFts5Query(query: string): string | null { + // Try a harmless MATCH against the FTS table to validate syntax + try { + db.query( + "SELECT 1 FROM threads_fts WHERE threads_fts MATCH $q LIMIT 0", + ).get({ $q: query }); + return null; // valid + } catch (e) { + const msg = e instanceof Error ? e.message : String(e); + return `Invalid FTS5 query: ${msg}\n\nSyntax guide:\n word - matches any thread containing "word"\n "exact phrase" - matches the exact phrase\n search* - prefix match\n apple AND banana - both terms required\n apple OR banana - either term\n apple NOT banana - exclude term\n\nAvoid unbalanced quotes, trailing operators, or empty parentheses.`; + } +} + +// --- Prepared Statements --- + +const stmtSearchThreads = db.query(` + SELECT + t.id, + t.channel_name, + t.category_name, + t.author_names, + t.message_count, + t.first_timestamp, + t.last_timestamp, + snippet(threads_fts, 0, '**', '**', '...', 64) as snippet + FROM threads_fts + JOIN threads t ON t.id = threads_fts.rowid + WHERE threads_fts MATCH $query + AND ($after IS NULL OR t.first_timestamp >= $after) + AND ($before IS NULL OR t.last_timestamp <= $before) + AND ($author IS NULL OR LOWER(t.author_names) LIKE '%' || LOWER($author) || '%') + ORDER BY bm25(threads_fts) + LIMIT $limit OFFSET $offset +`); + +const stmtChannelSearch = db.query(` + SELECT + t.id, + t.channel_name, + t.category_name, + t.author_names, + t.message_count, + t.first_timestamp, + t.last_timestamp, + snippet(threads_fts, 0, '**', '**', '...', 64) as snippet + FROM threads_fts + JOIN threads t ON t.id = threads_fts.rowid + WHERE threads_fts MATCH $query + AND LOWER(t.channel_name) = LOWER($channel) + AND ($after IS NULL OR t.first_timestamp >= $after) + AND ($before IS NULL OR t.last_timestamp <= $before) + ORDER BY bm25(threads_fts) + LIMIT $limit OFFSET $offset +`); + +const stmtGetThread = db.query(` + SELECT + t.id, + t.channel_name, + t.category_name, + t.author_names, + t.message_count, + t.first_timestamp, + t.last_timestamp, + t.messages_json + FROM threads t + WHERE t.id = $id +`); + +const stmtListChannels = db.query(` + SELECT + ch.name, + COALESCE(cat.name, 'uncategorized') as category, + ch.message_count, + ch.topic + FROM channels ch + LEFT JOIN categories cat ON ch.category_id = cat.id + WHERE ch.message_count > 0 + ORDER BY cat.name, ch.position +`); + +const stmtGetMessages = db.query(` + SELECT + m.id, + m.timestamp, + m.author_name, + m.content, + m.reply_to_id, + m.attachments, + m.embeds + FROM messages m + WHERE m.channel_id = (SELECT id FROM channels WHERE LOWER(name) = LOWER($channel)) + AND ($before IS NULL OR m.timestamp < $before) + ORDER BY m.timestamp DESC + LIMIT $limit +`); + +const stmtStats = db.query(` + SELECT + (SELECT COUNT(*) FROM messages) as total_messages, + (SELECT COUNT(*) FROM threads) as total_threads, + (SELECT COUNT(*) FROM channels WHERE message_count > 0) as active_channels, + (SELECT COUNT(DISTINCT author_name) FROM messages WHERE author_name IS NOT NULL) as unique_authors, + (SELECT MIN(timestamp) FROM messages) as earliest, + (SELECT MAX(timestamp) FROM messages) as latest +`); + +// --- Helpers --- + +function ok(data: unknown) { + return { + content: [{ type: "text" as const, text: JSON.stringify(data, null, 2) }], + }; +} + +function err(message: string) { + return { content: [{ type: "text" as const, text: message }], isError: true }; +} + +function safeQuery(fn: () => T, context: string): T | null { + try { + return fn(); + } catch (e) { + console.error(`Error in ${context}: ${e}`); + return null; + } +} + +// --- Server --- + +const server = new McpServer({ + name: "melty-discord", + version: "3.0.0", +}); + +server.registerTool( + "search", + { + description: `Full-text search across conversation threads using FTS5 with BM25 ranking and porter stemming. Returns ranked threads with snippets. + +Query syntax: + word - matches any thread containing "word" + "exact phrase" - matches the exact phrase + search* - prefix match + apple AND banana - both terms required + apple OR banana - either term + apple NOT banana - exclude term + +Use get_thread to read the full conversation of a result.`, + inputSchema: { + query: z.string().max(200).describe("FTS5 search query"), + limit: z + .number() + .int() + .min(1) + .max(50) + .default(15) + .describe("Max results"), + offset: z.number().int().min(0).default(0).describe("Pagination offset"), + after: z + .string() + .optional() + .describe( + "Only threads starting after this ISO-8601 timestamp (e.g. 2025-01-01)", + ), + before: z + .string() + .optional() + .describe("Only threads ending before this ISO-8601 timestamp"), + author: z + .string() + .optional() + .describe("Filter to threads containing this author name"), + }, + }, + async ({ query, limit, offset, after, before, author }) => { + const syntaxError = validateFts5Query(query); + if (syntaxError) return err(syntaxError); + + const results = safeQuery( + () => + stmtSearchThreads.all({ + $query: query, + $limit: limit, + $offset: offset, + $after: after ?? null, + $before: before ?? null, + $author: author ?? null, + }), + "search", + ); + if (!results) return err("Search failed unexpectedly."); + return ok(results); + }, +); + +server.registerTool( + "channel_search", + { + description: + "Search threads within a specific channel. Same FTS5 syntax as search. Useful for drilling into a known channel.", + inputSchema: { + channel: z.string().describe("Channel name (case-insensitive)"), + query: z.string().max(200).describe("FTS5 search query"), + limit: z + .number() + .int() + .min(1) + .max(50) + .default(15) + .describe("Max results"), + offset: z.number().int().min(0).default(0).describe("Pagination offset"), + after: z + .string() + .optional() + .describe("Only threads starting after this ISO-8601 timestamp"), + before: z + .string() + .optional() + .describe("Only threads ending before this ISO-8601 timestamp"), + }, + }, + async ({ channel, query, limit, offset, after, before }) => { + const syntaxError = validateFts5Query(query); + if (syntaxError) return err(syntaxError); + + const results = safeQuery( + () => + stmtChannelSearch.all({ + $query: query, + $channel: channel, + $limit: limit, + $offset: offset, + $after: after ?? null, + $before: before ?? null, + }), + "channel_search", + ); + if (!results) + return err("Search failed. Check channel name and query syntax."); + return ok(results); + }, +); + +server.registerTool( + "get_thread", + { + description: + "Get the full text of a conversation thread by ID. Use after search to read complete discussions. Returns structured message array.", + inputSchema: { + id: z.number().int().describe("Thread ID from search results"), + }, + }, + async ({ id }) => { + const result = safeQuery( + () => stmtGetThread.get({ $id: id }), + "get_thread", + ) as Record | null; + if (!result) return err(`Thread ${id} not found`); + + // Parse messages_json into structured array + let messages: unknown[] = []; + try { + messages = JSON.parse(result.messages_json as string); + } catch { + messages = []; + } + + return ok({ + id: result.id, + channel_name: result.channel_name, + category_name: result.category_name, + author_names: result.author_names, + message_count: result.message_count, + first_timestamp: result.first_timestamp, + last_timestamp: result.last_timestamp, + messages, + }); + }, +); + +server.registerTool( + "list_channels", + { + description: + "List all channels with message counts, grouped by category. Useful for discovering what topics exist.", + inputSchema: {}, + }, + async () => { + const rows = safeQuery( + () => stmtListChannels.all(), + "list_channels", + ) as Array<{ + name: string; + category: string; + message_count: number; + topic: string | null; + }> | null; + if (!rows) return err("Failed to list channels"); + + // Group by category + const grouped: Record< + string, + Array<{ name: string; message_count: number; topic: string | null }> + > = {}; + for (const r of rows) { + if (!grouped[r.category]) grouped[r.category] = []; + grouped[r.category].push({ + name: r.name, + message_count: r.message_count, + topic: r.topic, + }); + } + return ok(grouped); + }, +); + +server.registerTool( + "get_messages", + { + description: + "Get raw messages from a specific channel, ordered newest-first. Use when thread-level search isn't granular enough. Supports pagination via 'before' timestamp.", + inputSchema: { + channel: z.string().describe("Channel name (case-insensitive)"), + limit: z + .number() + .int() + .min(1) + .max(100) + .default(50) + .describe("Max messages to return"), + before: z + .string() + .optional() + .describe( + "Return messages before this ISO-8601 timestamp (for pagination)", + ), + }, + }, + async ({ channel, limit, before }) => { + const results = safeQuery( + () => + stmtGetMessages.all({ + $channel: channel, + $limit: limit, + $before: before ?? null, + }), + "get_messages", + ); + if (!results) return err("Failed to get messages. Check channel name."); + if ((results as unknown[]).length === 0) + return err( + `No messages found in channel "${channel}". Check spelling with list_channels.`, + ); + return ok(results); + }, +); + +server.registerTool( + "stats", + { + description: + "Get overall database statistics: total messages, threads, channels, authors, date range.", + inputSchema: {}, + }, + async () => { + const result = safeQuery(() => stmtStats.get(), "stats"); + if (!result) return err("Failed to get stats"); + return ok(result); + }, +); + +server.registerTool( + "query", + { + description: + "Run a raw read-only SQL query against the database. Escape hatch for advanced analysis. Tables: messages, channels, categories, threads, threads_fts, messages_fts, export_state.", + inputSchema: { + sql: z.string().max(2000).describe("SQL SELECT query (read-only)"), + }, + }, + async ({ sql }) => { + const trimmed = sql.trim().toUpperCase(); + if (!trimmed.startsWith("SELECT") && !trimmed.startsWith("WITH")) { + return err("Only SELECT and WITH queries are allowed"); + } + const results = safeQuery(() => db.query(sql).all(), "raw_query"); + if (!results) return err("Query failed. Check syntax."); + return ok(results); + }, +); + +async function main() { + const transport = new StdioServerTransport(); + await server.connect(transport); + console.error("melty-discord MCP server v3 running on stdio"); +} + +main().catch((error) => { + console.error("Fatal error:", error); + process.exit(1); +}); diff --git a/tsconfig.json b/tsconfig.json new file mode 100644 index 0000000..801e41e --- /dev/null +++ b/tsconfig.json @@ -0,0 +1,31 @@ +{ + "compilerOptions": { + // Environment setup & latest features + "lib": ["ESNext"], + "target": "ESNext", + "module": "Preserve", + "moduleDetection": "force", + "jsx": "react-jsx", + "allowJs": true, + "types": ["bun"], + + // Bundler mode + "moduleResolution": "bundler", + "allowImportingTsExtensions": true, + "verbatimModuleSyntax": true, + "noEmit": true, + + // Best practices + "strict": true, + "skipLibCheck": true, + "noFallthroughCasesInSwitch": true, + "noUncheckedIndexedAccess": true, + "noImplicitOverride": true, + + // Some stricter flags (disabled by default) + "noUnusedLocals": false, + "noUnusedParameters": false, + "noPropertyAccessFromIndexSignature": false + }, + "include": ["src"] +}