Overhaul crawler: depth crawl, link graph, robots, sitemaps, authority
Replace homepage-only crawling (the reason the index had ~9k docs and useless results) with a real crawler: - SQLite frontier (node:sqlite) — durable queue, seen-set, per-domain caps - follows internal links to configurable depth, seeds from sitemap.xml - robots.txt compliance + per-host politeness delay - per-URL doc IDs (old id=domain overwrote every page of a site) - link graph: cross-domain in-degree recorded as an authority signal - authority score field + 'score:desc' ranking rule (relevance-first, authority tiebreak) - body trimmed for the disk-bound VM Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
7e9eca6b0f
commit
c8cb4cbbe4
9 changed files with 314 additions and 84 deletions
1
.gitignore
vendored
1
.gitignore
vendored
|
|
@ -4,6 +4,7 @@ node_modules/
|
||||||
# Crawler data
|
# Crawler data
|
||||||
crawler/domains.csv
|
crawler/domains.csv
|
||||||
crawler/*.csv
|
crawler/*.csv
|
||||||
|
crawler/frontier.db*
|
||||||
|
|
||||||
# Wrangler / Cloudflare
|
# Wrangler / Cloudflare
|
||||||
.wrangler/
|
.wrangler/
|
||||||
|
|
|
||||||
|
|
@ -3,7 +3,7 @@
|
||||||
"version": "1.0.0",
|
"version": "1.0.0",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"scripts": {
|
"scripts": {
|
||||||
"start": "node src/index.js",
|
"start": "node --experimental-sqlite src/index.js",
|
||||||
"tranco": "node src/tranco.js"
|
"tranco": "node src/tranco.js"
|
||||||
},
|
},
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
|
|
|
||||||
|
|
@ -9,24 +9,24 @@ const HEADERS = {
|
||||||
'Accept-Language': 'en',
|
'Accept-Language': 'en',
|
||||||
}
|
}
|
||||||
|
|
||||||
export async function fetchPage(domain) {
|
// Accepts a bare domain (tries https then http) or a full URL (fetched as-is).
|
||||||
const urls = [`https://${domain}`, `http://${domain}`]
|
function candidates(target) {
|
||||||
for (const url of urls) {
|
if (/^https?:\/\//i.test(target)) return [target]
|
||||||
|
return [`https://${target}`, `http://${target}`]
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function fetchPage(target) {
|
||||||
|
for (const url of candidates(target)) {
|
||||||
try {
|
try {
|
||||||
const controller = new AbortController()
|
const controller = new AbortController()
|
||||||
const timer = setTimeout(() => controller.abort(), TIMEOUT_MS)
|
const timer = setTimeout(() => controller.abort(), TIMEOUT_MS)
|
||||||
const res = await fetch(url, {
|
const res = await fetch(url, { headers: HEADERS, signal: controller.signal, maxRedirections: 5 })
|
||||||
headers: HEADERS,
|
|
||||||
signal: controller.signal,
|
|
||||||
maxRedirections: 3,
|
|
||||||
})
|
|
||||||
clearTimeout(timer)
|
clearTimeout(timer)
|
||||||
|
|
||||||
if (!res.ok) continue
|
if (!res.ok) continue
|
||||||
const contentType = res.headers.get('content-type') ?? ''
|
const contentType = res.headers.get('content-type') ?? ''
|
||||||
if (!contentType.includes('html')) continue
|
if (!contentType.includes('html')) { res.body?.cancel?.(); continue }
|
||||||
|
|
||||||
// Read up to MAX_BODY_BYTES
|
|
||||||
const reader = res.body.getReader()
|
const reader = res.body.getReader()
|
||||||
const chunks = []
|
const chunks = []
|
||||||
let total = 0
|
let total = 0
|
||||||
|
|
@ -38,9 +38,27 @@ export async function fetchPage(domain) {
|
||||||
if (total >= MAX_BODY_BYTES) { reader.cancel(); break }
|
if (total >= MAX_BODY_BYTES) { reader.cancel(); break }
|
||||||
}
|
}
|
||||||
const html = Buffer.concat(chunks).toString('utf8')
|
const html = Buffer.concat(chunks).toString('utf8')
|
||||||
return { url, html, status: res.status }
|
return { url: res.url || url, html, status: res.status }
|
||||||
} catch {
|
} catch {
|
||||||
// try next URL
|
// try next candidate
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return null
|
||||||
|
}
|
||||||
|
|
||||||
|
// Fetch plain text (robots.txt, sitemap.xml). Returns null on any failure.
|
||||||
|
export async function fetchText(target) {
|
||||||
|
for (const url of candidates(target)) {
|
||||||
|
try {
|
||||||
|
const controller = new AbortController()
|
||||||
|
const timer = setTimeout(() => controller.abort(), TIMEOUT_MS)
|
||||||
|
const res = await fetch(url, { headers: HEADERS, signal: controller.signal, maxRedirections: 5 })
|
||||||
|
clearTimeout(timer)
|
||||||
|
if (!res.ok) { res.body?.cancel?.(); continue }
|
||||||
|
const txt = await res.text()
|
||||||
|
return txt.slice(0, MAX_BODY_BYTES)
|
||||||
|
} catch {
|
||||||
|
// try next candidate
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return null
|
return null
|
||||||
|
|
|
||||||
54
crawler/src/frontier.js
Normal file
54
crawler/src/frontier.js
Normal file
|
|
@ -0,0 +1,54 @@
|
||||||
|
// SQLite-backed crawl frontier + link graph (Node built-in node:sqlite).
|
||||||
|
// Run node with --experimental-sqlite. Durable across restarts.
|
||||||
|
import { DatabaseSync } from 'node:sqlite'
|
||||||
|
|
||||||
|
const DB_PATH = process.env.FRONTIER_DB || './frontier.db'
|
||||||
|
const db = new DatabaseSync(DB_PATH)
|
||||||
|
|
||||||
|
db.exec(`
|
||||||
|
PRAGMA journal_mode = WAL;
|
||||||
|
PRAGMA synchronous = NORMAL;
|
||||||
|
CREATE TABLE IF NOT EXISTS frontier (
|
||||||
|
url TEXT PRIMARY KEY,
|
||||||
|
domain TEXT NOT NULL,
|
||||||
|
depth INTEGER NOT NULL,
|
||||||
|
rank INTEGER NOT NULL DEFAULT 999999999,
|
||||||
|
status INTEGER NOT NULL DEFAULT 0 -- 0 queued, 1 done, 2 failed, 3 in-progress
|
||||||
|
);
|
||||||
|
CREATE INDEX IF NOT EXISTS idx_frontier_status ON frontier(status);
|
||||||
|
CREATE TABLE IF NOT EXISTS domain_pages (domain TEXT PRIMARY KEY, n INTEGER NOT NULL DEFAULT 0);
|
||||||
|
CREATE TABLE IF NOT EXISTS indeg (domain TEXT PRIMARY KEY, n INTEGER NOT NULL DEFAULT 0);
|
||||||
|
`)
|
||||||
|
|
||||||
|
const q = {
|
||||||
|
add: db.prepare('INSERT OR IGNORE INTO frontier(url,domain,depth,rank) VALUES(?,?,?,?)'),
|
||||||
|
next: db.prepare('SELECT url,domain,depth,rank FROM frontier WHERE status=0 LIMIT ?'),
|
||||||
|
claim: db.prepare('UPDATE frontier SET status=3 WHERE url=?'),
|
||||||
|
done: db.prepare('UPDATE frontier SET status=1 WHERE url=?'),
|
||||||
|
fail: db.prepare('UPDATE frontier SET status=2 WHERE url=?'),
|
||||||
|
reset: db.prepare('UPDATE frontier SET status=0 WHERE status=3'),
|
||||||
|
queued: db.prepare('SELECT count(*) c FROM frontier WHERE status=0'),
|
||||||
|
finished: db.prepare('SELECT count(*) c FROM frontier WHERE status=1'),
|
||||||
|
pages: db.prepare('SELECT n FROM domain_pages WHERE domain=?'),
|
||||||
|
incPages: db.prepare('INSERT INTO domain_pages(domain,n) VALUES(?,1) ON CONFLICT(domain) DO UPDATE SET n=n+1'),
|
||||||
|
incIndeg: db.prepare('INSERT INTO indeg(domain,n) VALUES(?,1) ON CONFLICT(domain) DO UPDATE SET n=n+1'),
|
||||||
|
indeg: db.prepare('SELECT n FROM indeg WHERE domain=?'),
|
||||||
|
}
|
||||||
|
|
||||||
|
export function enqueue(url, domain, depth, rank = 999999999) { q.add.run(url, domain, depth, rank) }
|
||||||
|
export function resetInProgress() { q.reset.run() }
|
||||||
|
export function markDone(url) { q.done.run(url) }
|
||||||
|
export function markFail(url) { q.fail.run(url) }
|
||||||
|
export function queued() { return q.queued.get().c }
|
||||||
|
export function finished() { return q.finished.get().c }
|
||||||
|
export function domainPages(domain) { return q.pages.get(domain)?.n ?? 0 }
|
||||||
|
export function incDomainPages(domain) { q.incPages.run(domain) }
|
||||||
|
export function incIndeg(domain) { q.incIndeg.run(domain) }
|
||||||
|
export function indegOf(domain) { return q.indeg.get(domain)?.n ?? 0 }
|
||||||
|
|
||||||
|
// Atomically claim up to n queued URLs (single-threaded JS => select+update is atomic).
|
||||||
|
export function claimBatch(n) {
|
||||||
|
const rows = q.next.all(n)
|
||||||
|
for (const r of rows) q.claim.run(r.url)
|
||||||
|
return rows
|
||||||
|
}
|
||||||
|
|
@ -1,66 +1,116 @@
|
||||||
import pLimit from 'p-limit'
|
import pLimit from 'p-limit'
|
||||||
import { downloadTranco, readDomains } from './tranco.js'
|
import { downloadTranco, readDomains } from './tranco.js'
|
||||||
import { fetchPage } from './fetcher.js'
|
import { fetchPage } from './fetcher.js'
|
||||||
import { parsePage } from './parser.js'
|
import { parsePage, regDomain } from './parser.js'
|
||||||
import { setupIndex, indexDoc, flush } from './indexer.js'
|
import { setupIndex, clearIndex, indexDoc, flush } from './indexer.js'
|
||||||
|
import { allowed } from './robots.js'
|
||||||
|
import { sitemapUrls } from './sitemap.js'
|
||||||
|
import * as F from './frontier.js'
|
||||||
|
|
||||||
const CONCURRENCY = parseInt(process.env.CONCURRENCY || '8')
|
const CONCURRENCY = parseInt(process.env.CONCURRENCY || '16')
|
||||||
const LIMIT = parseInt(process.env.LIMIT || '100000')
|
const LIMIT = parseInt(process.env.LIMIT || '200000') // max pages to index this run
|
||||||
|
const SEED_DOMAINS = parseInt(process.env.SEED_DOMAINS || '20000') // top Majestic domains to seed
|
||||||
|
const MAX_DEPTH = parseInt(process.env.MAX_DEPTH || '2') // link-follow depth
|
||||||
|
const MAX_PAGES_PER_DOMAIN = parseInt(process.env.MAX_PAGES_PER_DOMAIN || '30')
|
||||||
|
const HOST_DELAY_MS = parseInt(process.env.HOST_DELAY_MS || '1000') // politeness per host
|
||||||
|
const FRESH = process.env.FRESH === '1'
|
||||||
|
|
||||||
let crawled = 0
|
const sleep = ms => new Promise(r => setTimeout(r, ms))
|
||||||
let failed = 0
|
const lastHit = new Map() // host -> last fetch ts (politeness)
|
||||||
let skipped = 0
|
let indexed = 0, active = 0, crawled = 0, failed = 0
|
||||||
|
|
||||||
function log() {
|
function stats() {
|
||||||
if (crawled % 50 === 0) {
|
process.stdout.write(`\r[indexed=${indexed} crawled=${crawled} failed=${failed} queued=${F.queued()} active=${active}] `)
|
||||||
process.stdout.write(`\r[crawled=${crawled} failed=${failed} skipped=${skipped}]`)
|
}
|
||||||
|
|
||||||
|
function seed() {
|
||||||
|
let n = 0
|
||||||
|
for (const { rank, domain } of readDomains(SEED_DOMAINS)) {
|
||||||
|
F.enqueue(`https://${domain}`, regDomain(domain), 0, rank)
|
||||||
|
n++
|
||||||
|
}
|
||||||
|
console.log(`Seeded ${n} domains`)
|
||||||
|
}
|
||||||
|
|
||||||
|
async function politeWait(host) {
|
||||||
|
const wait = (lastHit.get(host) || 0) + HOST_DELAY_MS - Date.now()
|
||||||
|
if (wait > 0) await sleep(wait)
|
||||||
|
lastHit.set(host, Date.now())
|
||||||
|
}
|
||||||
|
|
||||||
|
async function crawlOne(row) {
|
||||||
|
const host = new URL(row.url).host
|
||||||
|
await politeWait(host)
|
||||||
|
|
||||||
|
if (!(await allowed(row.url))) { F.markDone(row.url); return }
|
||||||
|
|
||||||
|
const result = await fetchPage(row.url)
|
||||||
|
if (!result) { failed++; F.markFail(row.url); return }
|
||||||
|
crawled++
|
||||||
|
|
||||||
|
const doc = parsePage(result.url, result.html)
|
||||||
|
const links = doc.links
|
||||||
|
|
||||||
|
if ((doc.title || doc.description) && F.domainPages(doc.domain) < MAX_PAGES_PER_DOMAIN) {
|
||||||
|
doc.rank = row.rank
|
||||||
|
doc.indeg = F.indegOf(doc.domain)
|
||||||
|
await indexDoc(doc)
|
||||||
|
indexed++
|
||||||
|
F.incDomainPages(doc.domain)
|
||||||
|
}
|
||||||
|
F.markDone(row.url)
|
||||||
|
|
||||||
|
// depth-0: also seed from the sitemap
|
||||||
|
if (row.depth === 0) {
|
||||||
|
for (const u of await sitemapUrls(new URL(result.url).origin)) {
|
||||||
|
try { F.enqueue(u, regDomain(new URL(u).hostname), 1, row.rank) } catch { /* skip */ }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// enqueue links / record authority
|
||||||
|
if (row.depth < MAX_DEPTH) {
|
||||||
|
for (const l of links) {
|
||||||
|
if (l.sameDomain) {
|
||||||
|
if (F.domainPages(l.domain) < MAX_PAGES_PER_DOMAIN) F.enqueue(l.url, l.domain, row.depth + 1, row.rank)
|
||||||
|
} else {
|
||||||
|
F.incIndeg(l.domain) // inbound cross-domain link = authority signal
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
async function crawlDomain({ rank, domain }) {
|
|
||||||
const result = await fetchPage(domain)
|
|
||||||
if (!result) { failed++; log(); return }
|
|
||||||
|
|
||||||
const doc = parsePage(result.url, result.html)
|
|
||||||
if (!doc.title && !doc.description) { skipped++; log(); return }
|
|
||||||
|
|
||||||
doc.rank = rank
|
|
||||||
await indexDoc(doc)
|
|
||||||
crawled++
|
|
||||||
log()
|
|
||||||
}
|
|
||||||
|
|
||||||
// Process domains in a bounded sliding window instead of creating all promises upfront
|
|
||||||
async function main() {
|
async function main() {
|
||||||
await downloadTranco()
|
await downloadTranco()
|
||||||
await setupIndex()
|
await setupIndex()
|
||||||
|
if (FRESH) await clearIndex()
|
||||||
|
|
||||||
console.log(`Starting crawl: concurrency=${CONCURRENCY} limit=${LIMIT}`)
|
F.resetInProgress()
|
||||||
|
if (F.queued() === 0 && F.finished() === 0) seed()
|
||||||
|
|
||||||
const queue = []
|
console.log(`Crawl: concurrency=${CONCURRENCY} limit=${LIMIT} depth=${MAX_DEPTH} pages/domain=${MAX_PAGES_PER_DOMAIN}`)
|
||||||
let active = 0
|
const limit = pLimit(CONCURRENCY)
|
||||||
|
const statTimer = setInterval(stats, 1000)
|
||||||
|
|
||||||
async function runNext(entry) {
|
while (indexed < LIMIT) {
|
||||||
active++
|
const rows = F.claimBatch(CONCURRENCY * 2)
|
||||||
try { await crawlDomain(entry) } catch {}
|
if (rows.length === 0) {
|
||||||
active--
|
if (active === 0) break
|
||||||
|
await sleep(200)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for (const row of rows) {
|
||||||
|
active++
|
||||||
|
limit(() => crawlOne(row).catch(() => F.markFail(row.url)).finally(() => { active-- }))
|
||||||
|
}
|
||||||
|
// don't outrun the worker pool
|
||||||
|
while (active >= CONCURRENCY * 2) await sleep(20)
|
||||||
}
|
}
|
||||||
|
|
||||||
for (const entry of readDomains(LIMIT)) {
|
while (active > 0) await sleep(200)
|
||||||
while (active >= CONCURRENCY) {
|
|
||||||
await new Promise(r => setTimeout(r, 10))
|
|
||||||
}
|
|
||||||
queue.push(runNext(entry))
|
|
||||||
// Periodically drain settled promises to free memory
|
|
||||||
if (queue.length >= 500) {
|
|
||||||
await Promise.allSettled(queue.splice(0, 200))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
await Promise.allSettled(queue)
|
|
||||||
await flush()
|
await flush()
|
||||||
|
clearInterval(statTimer)
|
||||||
console.log(`\nDone. crawled=${crawled} failed=${failed} skipped=${skipped}`)
|
stats()
|
||||||
|
console.log(`\nDone. indexed=${indexed} crawled=${crawled} failed=${failed}`)
|
||||||
}
|
}
|
||||||
|
|
||||||
main().catch(console.error)
|
main().then(() => process.exit(0)).catch(e => { console.error(e); process.exit(1) })
|
||||||
|
|
|
||||||
|
|
@ -1,17 +1,14 @@
|
||||||
const MEILI_URL = process.env.MEILI_URL || 'http://localhost:7700'
|
const MEILI_URL = process.env.MEILI_URL || 'http://localhost:7700'
|
||||||
const MEILI_KEY = process.env.MEILI_KEY || 'masterKey'
|
const MEILI_KEY = process.env.MEILI_KEY || 'masterKey'
|
||||||
const INDEX = 'pages'
|
const INDEX = 'pages'
|
||||||
const BATCH_SIZE = 100
|
const BATCH_SIZE = 200
|
||||||
|
|
||||||
let buffer = []
|
let buffer = []
|
||||||
|
|
||||||
async function meili(method, path, body) {
|
async function meili(method, path, body) {
|
||||||
const res = await fetch(`${MEILI_URL}${path}`, {
|
const res = await fetch(`${MEILI_URL}${path}`, {
|
||||||
method,
|
method,
|
||||||
headers: {
|
headers: { 'Authorization': `Bearer ${MEILI_KEY}`, 'Content-Type': 'application/json' },
|
||||||
'Authorization': `Bearer ${MEILI_KEY}`,
|
|
||||||
'Content-Type': 'application/json',
|
|
||||||
},
|
|
||||||
body: body ? JSON.stringify(body) : undefined,
|
body: body ? JSON.stringify(body) : undefined,
|
||||||
})
|
})
|
||||||
return res.json()
|
return res.json()
|
||||||
|
|
@ -20,15 +17,30 @@ async function meili(method, path, body) {
|
||||||
export async function setupIndex() {
|
export async function setupIndex() {
|
||||||
await meili('POST', '/indexes', { uid: INDEX, primaryKey: 'id' })
|
await meili('POST', '/indexes', { uid: INDEX, primaryKey: 'id' })
|
||||||
await meili('PATCH', `/indexes/${INDEX}/settings`, {
|
await meili('PATCH', `/indexes/${INDEX}/settings`, {
|
||||||
searchableAttributes: ['title', 'description', 'domain', 'body'],
|
searchableAttributes: ['title', 'description', 'body', 'domain'],
|
||||||
displayedAttributes: ['id', 'url', 'domain', 'title', 'description', 'crawledAt'],
|
displayedAttributes: ['id', 'url', 'domain', 'title', 'description', 'crawledAt', 'score'],
|
||||||
rankingRules: ['words', 'typo', 'proximity', 'attribute', 'sort', 'exactness'],
|
// 'score:desc' as the last custom rule: authority breaks ties between similarly-relevant hits.
|
||||||
|
rankingRules: ['words', 'typo', 'proximity', 'attribute', 'exactness', 'score:desc'],
|
||||||
filterableAttributes: ['lang', 'domain'],
|
filterableAttributes: ['lang', 'domain'],
|
||||||
|
sortableAttributes: ['score'],
|
||||||
})
|
})
|
||||||
console.log('Index ready')
|
console.log('Index ready')
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Wipe all documents for a clean rebuild (FRESH=1).
|
||||||
|
export async function clearIndex() {
|
||||||
|
await meili('DELETE', `/indexes/${INDEX}/documents`)
|
||||||
|
console.log('Index cleared')
|
||||||
|
}
|
||||||
|
|
||||||
|
// Base authority score from Majestic rank (rank 1 => ~10M). Folded with in-degree later.
|
||||||
|
function baseScore(rank) {
|
||||||
|
return Math.max(1, 10_000_000 - Math.min(rank ?? 9_999_999, 9_999_999))
|
||||||
|
}
|
||||||
|
|
||||||
export async function indexDoc(doc) {
|
export async function indexDoc(doc) {
|
||||||
|
doc.score = baseScore(doc.rank) + (doc.indeg || 0) * 500
|
||||||
|
delete doc.links
|
||||||
buffer.push(doc)
|
buffer.push(doc)
|
||||||
if (buffer.length >= BATCH_SIZE) await flush()
|
if (buffer.length >= BATCH_SIZE) await flush()
|
||||||
}
|
}
|
||||||
|
|
@ -38,3 +50,5 @@ export async function flush() {
|
||||||
const docs = buffer.splice(0)
|
const docs = buffer.splice(0)
|
||||||
await meili('POST', `/indexes/${INDEX}/documents`, docs)
|
await meili('POST', `/indexes/${INDEX}/documents`, docs)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export { meili, baseScore }
|
||||||
|
|
|
||||||
|
|
@ -1,43 +1,66 @@
|
||||||
import { load } from 'cheerio'
|
import { load } from 'cheerio'
|
||||||
|
|
||||||
|
const BODY_CHARS = parseInt(process.env.BODY_CHARS || '2500') // keep docs small (disk-bound VM)
|
||||||
|
|
||||||
|
// Approximate registrable domain (eTLD+1) without a full public-suffix list.
|
||||||
|
const MULTI_TLD = new Set([
|
||||||
|
'co.uk', 'org.uk', 'ac.uk', 'gov.uk', 'com.au', 'net.au', 'org.au', 'edu.au', 'gov.au',
|
||||||
|
'co.nz', 'co.jp', 'com.br', 'co.za', 'com.sg', 'com.hk', 'co.in', 'co.kr',
|
||||||
|
])
|
||||||
|
export function regDomain(hostname) {
|
||||||
|
const host = hostname.replace(/^www\./, '')
|
||||||
|
const p = host.split('.')
|
||||||
|
if (p.length <= 2) return host
|
||||||
|
const last2 = p.slice(-2).join('.')
|
||||||
|
return MULTI_TLD.has(last2) ? p.slice(-3).join('.') : last2
|
||||||
|
}
|
||||||
|
|
||||||
|
// Stable per-URL id (FNV-1a) — must be per URL, not per domain, or pages overwrite each other.
|
||||||
|
function hashId(s) {
|
||||||
|
let h = 2166136261 >>> 0
|
||||||
|
for (let i = 0; i < s.length; i++) { h ^= s.charCodeAt(i); h = Math.imul(h, 16777619) >>> 0 }
|
||||||
|
return 'u' + h.toString(36) + s.length.toString(36)
|
||||||
|
}
|
||||||
|
|
||||||
export function parsePage(url, html) {
|
export function parsePage(url, html) {
|
||||||
const $ = load(html)
|
const $ = load(html)
|
||||||
|
|
||||||
// Remove noise
|
|
||||||
$('script, style, noscript, nav, footer, header, aside, [aria-hidden=true]').remove()
|
$('script, style, noscript, nav, footer, header, aside, [aria-hidden=true]').remove()
|
||||||
|
|
||||||
const title = ($('title').text() || $('h1').first().text()).trim().slice(0, 200)
|
const title = ($('title').text() || $('h1').first().text()).trim().slice(0, 200)
|
||||||
|
|
||||||
const description = (
|
const description = (
|
||||||
$('meta[name=description]').attr('content') ||
|
$('meta[name=description]').attr('content') ||
|
||||||
$('meta[property="og:description"]').attr('content') ||
|
$('meta[property="og:description"]').attr('content') ||
|
||||||
$('p').first().text()
|
$('p').first().text()
|
||||||
).trim().slice(0, 500)
|
).trim().slice(0, 500)
|
||||||
|
const body = $('body').text().replace(/\s+/g, ' ').trim().slice(0, BODY_CHARS)
|
||||||
|
const lang = ($('html').attr('lang') || 'en').slice(0, 8)
|
||||||
|
|
||||||
const body = $('body').text().replace(/\s+/g, ' ').trim().slice(0, 5000)
|
const pageDomain = regDomain(new URL(url).hostname)
|
||||||
|
|
||||||
const lang = $('html').attr('lang') || 'en'
|
const links = []
|
||||||
|
const seen = new Set()
|
||||||
const outlinks = []
|
|
||||||
$('a[href]').each((_, el) => {
|
$('a[href]').each((_, el) => {
|
||||||
try {
|
try {
|
||||||
const href = new URL($(el).attr('href'), url).href
|
const href = new URL($(el).attr('href'), url)
|
||||||
if (href.startsWith('http')) outlinks.push(href)
|
if (!/^https?:$/.test(href.protocol)) return
|
||||||
} catch {}
|
href.hash = ''
|
||||||
|
const clean = href.href
|
||||||
|
if (seen.has(clean)) return
|
||||||
|
seen.add(clean)
|
||||||
|
const d = regDomain(href.hostname)
|
||||||
|
links.push({ url: clean, domain: d, sameDomain: d === pageDomain })
|
||||||
|
} catch { /* skip bad href */ }
|
||||||
})
|
})
|
||||||
|
|
||||||
const domain = new URL(url).hostname.replace(/^www\./, '')
|
|
||||||
const id = domain.replace(/[^a-zA-Z0-9_-]/g, '_')
|
|
||||||
|
|
||||||
return {
|
return {
|
||||||
id,
|
id: hashId(url),
|
||||||
url,
|
url,
|
||||||
domain,
|
domain: pageDomain,
|
||||||
title: title || domain,
|
title: title || pageDomain,
|
||||||
description,
|
description,
|
||||||
body,
|
body,
|
||||||
lang,
|
lang,
|
||||||
outlinks: [...new Set(outlinks)].slice(0, 50),
|
links: links.slice(0, 150),
|
||||||
crawledAt: new Date().toISOString(),
|
crawledAt: new Date().toISOString(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
40
crawler/src/robots.js
Normal file
40
crawler/src/robots.js
Normal file
|
|
@ -0,0 +1,40 @@
|
||||||
|
// Minimal, polite robots.txt handling. Caches one ruleset per host.
|
||||||
|
import { fetchText } from './fetcher.js'
|
||||||
|
|
||||||
|
const UA = 'searchbot'
|
||||||
|
const cache = new Map() // host -> { disallow: string[] }
|
||||||
|
|
||||||
|
function parse(txt) {
|
||||||
|
const disallow = []
|
||||||
|
let apply = false
|
||||||
|
for (let raw of txt.split('\n')) {
|
||||||
|
const line = raw.replace(/#.*/, '').trim()
|
||||||
|
if (!line) continue
|
||||||
|
const idx = line.indexOf(':')
|
||||||
|
if (idx === -1) continue
|
||||||
|
const key = line.slice(0, idx).trim().toLowerCase()
|
||||||
|
const val = line.slice(idx + 1).trim()
|
||||||
|
if (key === 'user-agent') apply = val === '*' || val.toLowerCase().includes(UA)
|
||||||
|
else if (key === 'disallow' && apply && val) disallow.push(val)
|
||||||
|
}
|
||||||
|
return { disallow }
|
||||||
|
}
|
||||||
|
|
||||||
|
async function load(origin) {
|
||||||
|
try {
|
||||||
|
const txt = await fetchText(`${origin}/robots.txt`)
|
||||||
|
return txt ? parse(txt) : { disallow: [] }
|
||||||
|
} catch {
|
||||||
|
return { disallow: [] }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function allowed(url) {
|
||||||
|
const u = new URL(url)
|
||||||
|
let rules = cache.get(u.host)
|
||||||
|
if (!rules) {
|
||||||
|
rules = await load(u.origin)
|
||||||
|
cache.set(u.host, rules)
|
||||||
|
}
|
||||||
|
return !rules.disallow.some(d => u.pathname.startsWith(d))
|
||||||
|
}
|
||||||
30
crawler/src/sitemap.js
Normal file
30
crawler/src/sitemap.js
Normal file
|
|
@ -0,0 +1,30 @@
|
||||||
|
// Best-effort sitemap discovery. Returns a capped list of content URLs.
|
||||||
|
import { fetchText } from './fetcher.js'
|
||||||
|
|
||||||
|
const MAX_URLS = parseInt(process.env.SITEMAP_URLS || '40')
|
||||||
|
|
||||||
|
function locs(xml) {
|
||||||
|
const out = []
|
||||||
|
const re = /<loc>\s*([^<\s]+)\s*<\/loc>/gi
|
||||||
|
let m
|
||||||
|
while ((m = re.exec(xml))) out.push(m[1])
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function sitemapUrls(origin) {
|
||||||
|
const xml = await fetchText(`${origin}/sitemap.xml`)
|
||||||
|
if (!xml) return []
|
||||||
|
const entries = locs(xml)
|
||||||
|
const pages = entries.filter(u => !/\.xml(\.gz)?$/i.test(u))
|
||||||
|
if (pages.length) return pages.slice(0, MAX_URLS)
|
||||||
|
|
||||||
|
// sitemap index -> pull a couple of child sitemaps
|
||||||
|
const children = entries.filter(u => /\.xml$/i.test(u)).slice(0, 2)
|
||||||
|
const collected = []
|
||||||
|
for (const c of children) {
|
||||||
|
const cx = await fetchText(c)
|
||||||
|
if (cx) collected.push(...locs(cx).filter(u => !/\.xml(\.gz)?$/i.test(u)))
|
||||||
|
if (collected.length >= MAX_URLS) break
|
||||||
|
}
|
||||||
|
return collected.slice(0, MAX_URLS)
|
||||||
|
}
|
||||||
Loading…
Add table
Reference in a new issue