diff --git a/ts/bun.lock b/ts/bun.lock index d7cbadf..92011ab 100644 --- a/ts/bun.lock +++ b/ts/bun.lock @@ -3,6 +3,9 @@ "workspaces": { "": { "name": "ts", + "dependencies": { + "glob-to-regex.js": "^1.2.0", + }, "devDependencies": { "@types/bun": "latest", }, @@ -22,6 +25,10 @@ "csstype": ["csstype@3.1.3", "", {}, "sha512-M1uQkMl8rQK/szD0LNhtqxIPLpimGm8sOBwU7lLnCpSbTyY3yeU1Vc7l4KT5zT4s/yOxHH5O7tIuuLOCnLADRw=="], + "glob-to-regex.js": ["glob-to-regex.js@1.2.0", "", { "peerDependencies": { "tslib": "2" } }, "sha512-QMwlOQKU/IzqMUOAZWubUOT8Qft+Y0KQWnX9nK3ch0CJg0tTp4TvGZsTfudYKv2NzoQSyPcnA6TYeIQ3jGichQ=="], + + "tslib": ["tslib@2.8.1", "", {}, "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w=="], + "typescript": ["typescript@5.9.3", "", { "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" } }, "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw=="], "undici-types": ["undici-types@7.14.0", "", {}, "sha512-QQiYxHuyZ9gQUIrmPo3IA+hUl4KYk8uSA7cHrcKd/l3p1OTpZcM0Tbp9x7FAtXdAYhlasd60ncPpgu6ihG6TOA=="], diff --git a/ts/package.json b/ts/package.json index b647e65..6bfacb9 100644 --- a/ts/package.json +++ b/ts/package.json @@ -8,5 +8,8 @@ "typescript": "^5" }, "private": true, - "type": "module" + "type": "module", + "dependencies": { + "glob-to-regex.js": "^1.2.0" + } } diff --git a/ts/searchEngine/crawler.test.ts b/ts/searchEngine/crawler.test.ts new file mode 100644 index 0000000..490869b --- /dev/null +++ b/ts/searchEngine/crawler.test.ts @@ -0,0 +1,48 @@ +import { describe, it, beforeEach, expect } from "bun:test"; +import { SearchIndex } from "."; +import { Crawler, RobotsParser } from "./crawler"; +import { sleep } from "bun"; + +describe("Robots Parser", () => { + it("should parse robots.txt file", () => { + const robotsTxt = ` + User-agent: * + User-agent: crawl + Disallow: /admin + Allow: /public + + User-Agent: crawl + Disallow: /no-robots + `; + const robotsParser = new RobotsParser(robotsTxt); + const { allows, disallows } = robotsParser.getUrlsForUA("crawl"); + const urls = { + allows, + disallows, + }; + expect(allows.has("/public")).toBe(true); + expect(disallows.has("/admin")).toBe(true); + expect(RobotsParser.checkUserAgent(urls, "/admin")).toBe(false); + expect(RobotsParser.checkUserAgent(urls, "/public")).toBe(true); + }); +}); + +describe("Crawler", () => { + let crawler: Crawler; + beforeEach(() => { + crawler = new Crawler("SmartFridge", new SearchIndex()); + }); + + it("should crawl a page", () => { + const url = new URL("https://google.com"); + crawler.crawl(url); + crawler.on("storePage", (url) => { + console.log(`Page stored: ${url}`); + sleep(4000).then(() => { + crawler.emit("stop"); + expect(crawler.index.size()).toBe(1); + }); + }); + // expect(crawler.index).toBe(1); + }); +}); diff --git a/ts/searchEngine/crawler.ts b/ts/searchEngine/crawler.ts new file mode 100644 index 0000000..05bcdf0 --- /dev/null +++ b/ts/searchEngine/crawler.ts @@ -0,0 +1,211 @@ +import { SearchIndex } from "."; +import { toRegex } from "glob-to-regex.js"; +import { EventEmitter } from "node:events"; + +interface RobotUrls { + allows: Set; + disallows: Set; +} + +export class RobotsParser { + disallow: Map> = new Map(); + allow: Map> = new Map(); + + constructor(text: string) { + const lines = text + .split("\n") + .filter((l) => !/^\s*#.*$/.test(l)) // remove full-line comments + .map((l) => l.replace(/\s*#.*$/, "")); // remove end-of-line comments + lines.push(""); + + const blocks: Array> = []; + let current_block: Array = []; + lines.forEach((line) => { + if (line == "") { + if (current_block.length == 0) return; // ignore consecutive empty lines + blocks.push(current_block); + current_block = new Array(); + } else { + current_block.push(line); + } + }); + + blocks.forEach((block) => { + let uas: string[] = []; + let disallows: string[] = []; + let allows: string[] = []; + block.forEach((line) => { + line = line.trim().toLowerCase(); + const fields: Array = line.split(/\s*:\s*/); + if (fields.length < 2) return; + if (fields[0] == "user-agent") { + uas.push(fields[1]!); + } else if (fields[0] == "disallow") { + disallows.push(fields[1]!); + } else if (fields[0] == "allow") { + allows.push(fields[1]!); + } + }); + uas.forEach((ua) => { + ua = ua.toLowerCase(); + this.disallow.set( + ua, + new Set([...(this.disallow.get(ua) || []), ...disallows]), + ); + this.allow.set( + ua, + new Set([...(this.allow.get(ua) || []), ...allows]), + ); + }); + }); + } + + static checkUserAgent(urls: RobotUrls, url: string): boolean { + const { allows, disallows } = urls; + const allowed = allows + .values() + .map((allow) => { + const regex = toRegex(allow); + return regex.test(url); + }) + .reduce((acc, curr) => acc || curr, false); + if (allowed) { + return true; + } + const disallowed = disallows + .values() + .map((disallow) => { + const regex = toRegex(disallow); + return regex.test(url); + }) + .reduce((acc, curr) => acc || curr, false); + return !disallowed; + } + + getUrlsForUA(ua: string): RobotUrls { + ua = ua.toLowerCase(); + const allowUAs = this.allow + .keys() + .filter((key) => toRegex(key).test(ua)); + const disallowUAs = this.disallow + .keys() + .filter((key) => toRegex(key).test(ua)); + let allows = new Set(); + let disallows = new Set(); + + allowUAs.forEach((ua) => { + const allow = this.allow.get(ua); + if (allow) { + allows = allows.union(allow); + } + }); + disallowUAs.forEach((ua) => { + const disallow = this.disallow.get(ua); + if (disallow) { + disallows = disallows.union(disallow); + } + }); + return { + allows, + disallows, + }; + } +} + +const urlRegex = /https?:\/\/[^\s\"]+/g; +export class Crawler extends EventEmitter { + private robots: Map = new Map(); // hostname, robots allowed and disallowed for the sepcified UA + private visited: Set = new Set(); // URLS + + constructor( + private readonly UA: string, + public index: SearchIndex, + ) { + super(); + this.on("addURL", (url: URL) => { + console.log(`Adding URL: ${url}`); + void this.processPage(url); + }); + this.once("stop", () => { + this.removeAllListeners(); + }); + } + + private async checkDisallowed(url: URL): Promise { + const robots = + this.robots.get(url.hostname) || (await this.getRobotsTxt(url)); + return !RobotsParser.checkUserAgent(robots, url.toString()); + } + + private async getRobotsTxt(url: URL): Promise { + const robotsTxtUrl = new URL( + `${url.protocol}//${url.hostname}/robots.txt`, + ); + + const response = await fetch(robotsTxtUrl, { + headers: { + "User-Agent": this.UA, + }, + }); + if (response.status !== 200) + return { allows: new Set(), disallows: new Set() }; + if (!response.headers.get("content-type")?.startsWith("text/plain")) + return { allows: new Set(), disallows: new Set() }; + const robotsTxt = await response.text(); + const parsed = new RobotsParser(robotsTxt); + const forUA = parsed.getUrlsForUA(this.UA); + this.robots.set(url.hostname, forUA); + return forUA; + } + + private async addOutlinks(html: string): Promise { + const links = html.matchAll(urlRegex); + if (!links) return; + for (const [link, ..._] of links) { + console.log(link); + const url = new URL(link); + if (await this.checkDisallowed(url)) { + this.emit("addURL", url); + } + } + } + + // private getText(html: string): string { + // const parser = new DOMParser(); + // const doc = parser.parseFromString(html, "text/html"); + // return doc.body.textContent || ""; + // } + + private async getPage(url: URL) { + if (this.visited.has(url)) return; + if (await this.checkDisallowed(url)) return; + const page = await fetch(url); + this.visited.add(url); + if (!page.ok) return; + if (!page.headers.get("Content-Type")?.startsWith("text/html")) return; + + return await page.text(); + } + + private async processPage(url: URL) { + const page = await this.getPage(url); + if (!page) return; + await this.addOutlinks(page); + this.index.addPage(url.toString(), page); + this.emit("storePage", url); + } + + crawl(url_str: string | URL) { + this.emit("addURL", new URL(url_str)); + } +} + +let crawler = new Crawler("SmartFridge", new SearchIndex()); + +const url = new URL("https://example.com"); +crawler.crawl(url); +crawler.on("storePage", (url) => { + console.log(`Page stored: ${url}`); + console.log("entries:", crawler.index.size()); + crawler.emit("stop"); +}); diff --git a/ts/searchEngine/index.test.ts b/ts/searchEngine/index.test.ts index f79d7f2..6dfcaa0 100644 --- a/ts/searchEngine/index.test.ts +++ b/ts/searchEngine/index.test.ts @@ -1,5 +1,6 @@ import { describe, it, beforeEach, expect } from "bun:test"; import { SearchIndex } from "."; +import { Crawler } from "./crawler"; describe("Search Index", () => { let index: SearchIndex; @@ -77,6 +78,7 @@ describe("Search Algorithm", () => { "beans beans beans beans beans beans beans beans beans beans beans beans beans beans beans beans beans beans beans beans beans beans beans beans beans beans", ); index.addPage("https://www.beans-are-ok.com", "beans are ok I guess"); + index.addPage("https://testsite.com", "beans"); index.addPage( "https://www.example.com/cats", "This is a sample web page about cats", @@ -127,8 +129,10 @@ describe("Search Algorithm", () => { const results = index.search("beans"); expect(results.indexOf("https://www.beans.com")).toBe(0); expect(results.indexOf("https://www.beans-are-ok.com")).toBe(1); + expect(results.indexOf("https://testsite.com")).toBe(2); const results2 = index.search("beans beans"); expect(results2.indexOf("https://www.beans.com")).toBe(0); expect(results2.indexOf("https://www.beans-are-ok.com")).toBe(1); + expect(results.indexOf("https://testsite.com")).toBe(2); }); }); diff --git a/ts/searchEngine/index.ts b/ts/searchEngine/index.ts index f9c80e4..85ad92d 100644 --- a/ts/searchEngine/index.ts +++ b/ts/searchEngine/index.ts @@ -77,7 +77,7 @@ const stopWords = new Set([ ]); export class SearchIndex { - index: Map; + private index: Map; constructor() { this.index = new Map(); @@ -169,6 +169,21 @@ export class SearchIndex { }); } + checkPage(search: string): boolean { + for (const urls of this.index.values()) { + for (const [url, _] of urls) { + if (search === url) { + return true; + } + } + } + return false; + } + + size() { + return this.index.size; + } + getPagesForKeyword(keyword: string): string[] { const pages = this.index.get(keyword); if (!pages) { @@ -194,6 +209,11 @@ export class SearchIndex { ); } } + urls.forEach((value, key) => { + if (key.includes(query)) { + value += 10; + } + }); return Array.from(urls.entries()) .sort((a, b) => b[1] - a[1]) .map(([url, _]) => url); diff --git a/ts/searchEngine/mainLoop.plan b/ts/searchEngine/mainLoop.plan new file mode 100644 index 0000000..e69de29