import type { CrawledPage } from "./types"; export class Crawler { private maxPages: number; private timeoutMs: number; private visited: Set = new Set(); private pages: CrawledPage[] = []; private queue: string[] = []; constructor(options: { maxPages: number; timeoutMs: number }) { this.maxPages = options.maxPages; this.timeoutMs = options.timeoutMs; } async crawl(startUrl: string): Promise { this.visited.clear(); this.pages = []; this.queue = [startUrl]; while (this.queue.length > 0 && this.pages.length < this.maxPages) { const url = this.queue.shift()!; if (this.visited.has(url)) continue; this.visited.add(url); try { const page = await this.fetchPage(url); this.pages.push(page); for (const link of page.links) { if (!this.visited.has(link)) { this.queue.push(link); } } } catch { // Skip unreachable pages, continue crawl } } return this.pages; } private async fetchPage(url: string): Promise { const controller = new AbortController(); const timeout = setTimeout(() => controller.abort(), this.timeoutMs); try { const res = await fetch(url, { signal: controller.signal, headers: { "User-Agent": "c0py/0.1" }, redirect: "follow", }); const body = await res.text(); const links = this.extractLinks(body, url); return { url, path: new URL(url).pathname, title: this.extractTag(body, "title"), description: this.extractMetaDescription(body), statusCode: res.status, contentType: res.headers.get("content-type"), links, }; } finally { clearTimeout(timeout); } } private extractLinks(html: string, baseUrl: string): string[] { const links: string[] = []; const re = /href\s*=\s*["']([^"']+)["']/gi; let m; while ((m = re.exec(html)) !== null) { try { const abs = new URL(m[1], baseUrl).href; if (!links.includes(abs)) links.push(abs); } catch { // Skip invalid URLs } } return links; } private extractTag(html: string, tag: string): string | null { const re = new RegExp(`<${tag}[^>]*>([\\s\\S]*?)<\\/${tag}>`, "i"); const m = html.match(re); if (!m) return null; return m[1].replace(/<[^>]+>/g, "").trim().slice(0, 500) || null; } private extractMetaDescription(html: string): string | null { const re = /]*name=["']description["'][^>]*content=["']([^"']*)["']/i; const m = html.match(re); return m ? m[1].slice(0, 500) : null; } }