Files
c0py/packages/c0py-core/src/crawler/crawler.ts
T
admin e04cf8fbd3 c0py: initial scaffolding for website registry tool
Repo: sax3l/c0py (created on Gitea, code pushed)

Initial scaffolding includes:
- apps/api: Fastify API server (typecheck, build, test pass)
- apps/web: Next.js UI (build passes)
- packages/c0py-core: Crawler, registry assembler, analyzer (6 tests pass)
- packages/config: Zod-validated env loader
- packages/types: Shared TypeScript types

CI/CD: .gitea/workflows/ci.yml + deploy.yml
Dockerfile for Coolify deployment
mise.toml (toolchain contract)
lefthook.yml, renovate.json
docs/architecture/OVERVIEW.md, docs/PLATFORM.md

Platform setup required (see docs/PLATFORM.md):
1. Coolify app provision (c0py on server5/6)
2. Postgres database (dedicated, not shared)
3. Infisical secrets (DATABASE_URL, CL0UD, ZITADEL, AUD0, ST0RE, INF0, N0D)
4. DNS c0py.siax.io A-record
5. Zitadel OIDC project (Simon-decision)

Status: dev (repo live, code pushable, platform infra pending)
2026-09-16 13:33:45 +02:00

97 lines
2.6 KiB
TypeScript

import type { CrawledPage } from "./types";
export class Crawler {
private maxPages: number;
private timeoutMs: number;
private visited: Set<string> = new Set();
private pages: CrawledPage[] = [];
private queue: string[] = [];
constructor(options: { maxPages: number; timeoutMs: number }) {
this.maxPages = options.maxPages;
this.timeoutMs = options.timeoutMs;
}
async crawl(startUrl: string): Promise<CrawledPage[]> {
this.visited.clear();
this.pages = [];
this.queue = [startUrl];
while (this.queue.length > 0 && this.pages.length < this.maxPages) {
const url = this.queue.shift()!;
if (this.visited.has(url)) continue;
this.visited.add(url);
try {
const page = await this.fetchPage(url);
this.pages.push(page);
for (const link of page.links) {
if (!this.visited.has(link)) {
this.queue.push(link);
}
}
} catch {
// Skip unreachable pages, continue crawl
}
}
return this.pages;
}
private async fetchPage(url: string): Promise<CrawledPage> {
const controller = new AbortController();
const timeout = setTimeout(() => controller.abort(), this.timeoutMs);
try {
const res = await fetch(url, {
signal: controller.signal,
headers: { "User-Agent": "c0py/0.1" },
redirect: "follow",
});
const body = await res.text();
const links = this.extractLinks(body, url);
return {
url,
path: new URL(url).pathname,
title: this.extractTag(body, "title"),
description: this.extractMetaDescription(body),
statusCode: res.status,
contentType: res.headers.get("content-type"),
links,
};
} finally {
clearTimeout(timeout);
}
}
private extractLinks(html: string, baseUrl: string): string[] {
const links: string[] = [];
const re = /href\s*=\s*["']([^"']+)["']/gi;
let m;
while ((m = re.exec(html)) !== null) {
try {
const abs = new URL(m[1], baseUrl).href;
if (!links.includes(abs)) links.push(abs);
} catch {
// Skip invalid URLs
}
}
return links;
}
private extractTag(html: string, tag: string): string | null {
const re = new RegExp(`<${tag}[^>]*>([\\s\\S]*?)<\\/${tag}>`, "i");
const m = html.match(re);
if (!m) return null;
return m[1].replace(/<[^>]+>/g, "").trim().slice(0, 500) || null;
}
private extractMetaDescription(html: string): string | null {
const re = /<meta[^>]*name=["']description["'][^>]*content=["']([^"']*)["']/i;
const m = html.match(re);
return m ? m[1].slice(0, 500) : null;
}
}