Repo: sax3l/c0py (created on Gitea, code pushed) Initial scaffolding includes: - apps/api: Fastify API server (typecheck, build, test pass) - apps/web: Next.js UI (build passes) - packages/c0py-core: Crawler, registry assembler, analyzer (6 tests pass) - packages/config: Zod-validated env loader - packages/types: Shared TypeScript types CI/CD: .gitea/workflows/ci.yml + deploy.yml Dockerfile for Coolify deployment mise.toml (toolchain contract) lefthook.yml, renovate.json docs/architecture/OVERVIEW.md, docs/PLATFORM.md Platform setup required (see docs/PLATFORM.md): 1. Coolify app provision (c0py on server5/6) 2. Postgres database (dedicated, not shared) 3. Infisical secrets (DATABASE_URL, CL0UD, ZITADEL, AUD0, ST0RE, INF0, N0D) 4. DNS c0py.siax.io A-record 5. Zitadel OIDC project (Simon-decision) Status: dev (repo live, code pushable, platform infra pending)
97 lines
2.6 KiB
TypeScript
97 lines
2.6 KiB
TypeScript
import type { CrawledPage } from "./types";
|
|
|
|
export class Crawler {
|
|
private maxPages: number;
|
|
private timeoutMs: number;
|
|
private visited: Set<string> = new Set();
|
|
private pages: CrawledPage[] = [];
|
|
private queue: string[] = [];
|
|
|
|
constructor(options: { maxPages: number; timeoutMs: number }) {
|
|
this.maxPages = options.maxPages;
|
|
this.timeoutMs = options.timeoutMs;
|
|
}
|
|
|
|
async crawl(startUrl: string): Promise<CrawledPage[]> {
|
|
this.visited.clear();
|
|
this.pages = [];
|
|
this.queue = [startUrl];
|
|
|
|
while (this.queue.length > 0 && this.pages.length < this.maxPages) {
|
|
const url = this.queue.shift()!;
|
|
if (this.visited.has(url)) continue;
|
|
this.visited.add(url);
|
|
|
|
try {
|
|
const page = await this.fetchPage(url);
|
|
this.pages.push(page);
|
|
for (const link of page.links) {
|
|
if (!this.visited.has(link)) {
|
|
this.queue.push(link);
|
|
}
|
|
}
|
|
} catch {
|
|
// Skip unreachable pages, continue crawl
|
|
}
|
|
}
|
|
|
|
return this.pages;
|
|
}
|
|
|
|
private async fetchPage(url: string): Promise<CrawledPage> {
|
|
const controller = new AbortController();
|
|
const timeout = setTimeout(() => controller.abort(), this.timeoutMs);
|
|
|
|
try {
|
|
const res = await fetch(url, {
|
|
signal: controller.signal,
|
|
headers: { "User-Agent": "c0py/0.1" },
|
|
redirect: "follow",
|
|
});
|
|
const body = await res.text();
|
|
|
|
const links = this.extractLinks(body, url);
|
|
|
|
return {
|
|
url,
|
|
path: new URL(url).pathname,
|
|
title: this.extractTag(body, "title"),
|
|
description: this.extractMetaDescription(body),
|
|
statusCode: res.status,
|
|
contentType: res.headers.get("content-type"),
|
|
links,
|
|
};
|
|
} finally {
|
|
clearTimeout(timeout);
|
|
}
|
|
}
|
|
|
|
private extractLinks(html: string, baseUrl: string): string[] {
|
|
const links: string[] = [];
|
|
const re = /href\s*=\s*["']([^"']+)["']/gi;
|
|
let m;
|
|
while ((m = re.exec(html)) !== null) {
|
|
try {
|
|
const abs = new URL(m[1], baseUrl).href;
|
|
if (!links.includes(abs)) links.push(abs);
|
|
} catch {
|
|
// Skip invalid URLs
|
|
}
|
|
}
|
|
return links;
|
|
}
|
|
|
|
private extractTag(html: string, tag: string): string | null {
|
|
const re = new RegExp(`<${tag}[^>]*>([\\s\\S]*?)<\\/${tag}>`, "i");
|
|
const m = html.match(re);
|
|
if (!m) return null;
|
|
return m[1].replace(/<[^>]+>/g, "").trim().slice(0, 500) || null;
|
|
}
|
|
|
|
private extractMetaDescription(html: string): string | null {
|
|
const re = /<meta[^>]*name=["']description["'][^>]*content=["']([^"']*)["']/i;
|
|
const m = html.match(re);
|
|
return m ? m[1].slice(0, 500) : null;
|
|
}
|
|
}
|