From 3aa08e4a4fd5cf65626875d51970e8d04102f5c8 Mon Sep 17 00:00:00 2001 From: King Omar Date: Wed, 22 Jul 2026 18:35:23 +1000 Subject: [PATCH] Add link-graph authority: record edges + PageRank recompute MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Crawler now records the deduped cross-domain link graph (edges table). recompute.js runs real iterative PageRank over that graph, blends it with accumulated in-degree + the Majestic seed prior, and patches every doc's `score` (Meili's authority tie-breaker) via partial updates — preserving titles/bodies/embeddings and triggering no re-embedding. systemd timer runs it every 3h so ranking sharpens as the 1.5M-page crawl densifies the graph. Co-Authored-By: Claude Opus 4.8 (1M context) --- crawler/src/frontier.js | 6 ++ crawler/src/index.js | 3 +- crawler/src/recompute.js | 152 +++++++++++++++++++++++++++++++ scripts/search-recompute.service | 21 +++++ scripts/search-recompute.timer | 12 +++ 5 files changed, 193 insertions(+), 1 deletion(-) create mode 100644 crawler/src/recompute.js create mode 100644 scripts/search-recompute.service create mode 100644 scripts/search-recompute.timer diff --git a/crawler/src/frontier.js b/crawler/src/frontier.js index 901ca9d..a59a36b 100644 --- a/crawler/src/frontier.js +++ b/crawler/src/frontier.js @@ -18,6 +18,9 @@ db.exec(` CREATE INDEX IF NOT EXISTS idx_frontier_status ON frontier(status); CREATE TABLE IF NOT EXISTS domain_pages (domain TEXT PRIMARY KEY, n INTEGER NOT NULL DEFAULT 0); CREATE TABLE IF NOT EXISTS indeg (domain TEXT PRIMARY KEY, n INTEGER NOT NULL DEFAULT 0); + -- Cross-domain link graph for PageRank. Deduped src->dst (unique pairs), so a + -- single authoritative edge isn't over-counted by many links between two sites. + CREATE TABLE IF NOT EXISTS edges (src TEXT NOT NULL, dst TEXT NOT NULL, PRIMARY KEY(src,dst)) WITHOUT ROWID; `) const q = { @@ -35,6 +38,7 @@ const q = { incPages: db.prepare('INSERT INTO domain_pages(domain,n) VALUES(?,1) ON CONFLICT(domain) DO UPDATE SET n=n+1'), incIndeg: db.prepare('INSERT INTO indeg(domain,n) VALUES(?,1) ON CONFLICT(domain) DO UPDATE SET n=n+1'), indeg: db.prepare('SELECT n FROM indeg WHERE domain=?'), + addEdge: db.prepare('INSERT OR IGNORE INTO edges(src,dst) VALUES(?,?)'), } export function enqueue(url, domain, depth, rank = 999999999) { q.add.run(url, domain, depth, rank) } @@ -47,6 +51,8 @@ export function domainPages(domain) { return q.pages.get(domain)?.n ?? 0 } export function incDomainPages(domain) { q.incPages.run(domain) } export function incIndeg(domain) { q.incIndeg.run(domain) } export function indegOf(domain) { return q.indeg.get(domain)?.n ?? 0 } +// Record a cross-domain link src->dst for the PageRank graph (skip self-loops). +export function addEdge(src, dst) { if (src && dst && src !== dst) q.addEdge.run(src, dst) } // Atomically claim up to n queued URLs (single-threaded JS => select+update is atomic). export function claimBatch(n) { diff --git a/crawler/src/index.js b/crawler/src/index.js index d806ff5..9c1c8f7 100644 --- a/crawler/src/index.js +++ b/crawler/src/index.js @@ -73,7 +73,8 @@ async function crawlOne(row) { if (l.sameDomain) { if (F.domainPages(l.domain) < MAX_PAGES_PER_DOMAIN) F.enqueue(l.url, l.domain, row.depth + 1, row.rank) } else { - F.incIndeg(l.domain) // inbound cross-domain link = authority signal + F.incIndeg(l.domain) // inbound cross-domain link = authority signal + F.addEdge(doc.domain, l.domain) // record edge for PageRank recompute } } } diff --git a/crawler/src/recompute.js b/crawler/src/recompute.js new file mode 100644 index 0000000..8a74175 --- /dev/null +++ b/crawler/src/recompute.js @@ -0,0 +1,152 @@ +// recompute.js — domain-level authority recompute → push scores to Meilisearch. +// +// Runs real iterative PageRank over the crawl's cross-domain link graph (the +// `edges` table), blends it with accumulated in-degree and the Majestic seed +// rank (a trustworthy global prior), then patches every indexed document's +// `score` field — Meili's authority tie-breaker (last ranking rule) — via +// PARTIAL updates (PUT), so titles/bodies/embeddings are left untouched and no +// re-embedding is triggered. +// +// Safe to run while the crawler is live; offset pagination self-heals on the +// next run. Intended to run periodically (systemd timer) as the crawl matures: +// early on the graph is sparse and in-degree carries the signal; as edges +// accumulate toward the 1.5M-page target, PageRank takes over. +// +// node --experimental-sqlite src/recompute.js +// +// Env: FRONTIER_DB, MEILI_URL, MEILI_KEY, DAMPING, ITERS, PR_WEIGHT, INDEG_WEIGHT +import { DatabaseSync } from 'node:sqlite' + +const DB_PATH = process.env.FRONTIER_DB || './frontier.db' +const MEILI_URL = process.env.MEILI_URL || 'http://localhost:7700' +const MEILI_KEY = process.env.MEILI_KEY || 'masterKey' +const INDEX = 'pages' +const DAMPING = parseFloat(process.env.DAMPING || '0.85') +const ITERS = parseInt(process.env.ITERS || '40') +const AUTH_SCALE = 6_000_000 // max authority bonus layered on baseScore +const PR_WEIGHT = parseFloat(process.env.PR_WEIGHT || '0.6') +const INDEG_WEIGHT = parseFloat(process.env.INDEG_WEIGHT || '0.4') +const PAGE = 1000 + +// Seed authority from Majestic rank (rank 1 => ~10M). Mirrors indexer.js so the +// recomputed score stays on the same scale as freshly-crawled docs. +function baseScore(rank) { + return Math.max(1, 10_000_000 - Math.min(rank ?? 9_999_999, 9_999_999)) +} + +async function meili(method, path, body) { + const res = await fetch(`${MEILI_URL}${path}`, { + method, + headers: { Authorization: `Bearer ${MEILI_KEY}`, 'Content-Type': 'application/json' }, + body: body ? JSON.stringify(body) : undefined, + }) + if (!res.ok) throw new Error(`${method} ${path} -> ${res.status} ${await res.text()}`) + return res.json() +} + +// Pull the domain graph + priors out of the frontier DB. +function loadGraph(db) { + const nodes = new Set() + const out = new Map() // src -> [dst,...] (deduped by PK) + for (const { src, dst } of db.prepare('SELECT src,dst FROM edges').all()) { + nodes.add(src); nodes.add(dst) + if (!out.has(src)) out.set(src, []) + out.get(src).push(dst) + } + const indeg = new Map() // accumulated cross-domain in-degree + for (const { domain, n } of db.prepare('SELECT domain,n FROM indeg').all()) { + indeg.set(domain, n); nodes.add(domain) + } + const baseRank = new Map() // domain -> best (lowest) Majestic rank seen + for (const { domain, r } of db.prepare('SELECT domain, MIN(rank) r FROM frontier GROUP BY domain').all()) { + baseRank.set(domain, r); nodes.add(domain) + } + return { nodes: [...nodes], out, indeg, baseRank } +} + +// Standard PageRank with damping + dangling-node redistribution. +function pagerank(nodes, out) { + const N = nodes.length + if (N === 0) return new Map() + const idx = new Map(nodes.map((d, i) => [d, i])) + const outdeg = new Float64Array(N) + const inlinks = Array.from({ length: N }, () => []) // i -> [src indices] + for (const [src, dsts] of out) { + const si = idx.get(src) + outdeg[si] = dsts.length + for (const dst of dsts) inlinks[idx.get(dst)].push(si) + } + let pr = new Float64Array(N).fill(1 / N) + const teleport = (1 - DAMPING) / N + for (let it = 0; it < ITERS; it++) { + let dangling = 0 + for (let i = 0; i < N; i++) if (outdeg[i] === 0) dangling += pr[i] + const danglingShare = DAMPING * dangling / N + const next = new Float64Array(N) + for (let i = 0; i < N; i++) { + let s = 0 + const ins = inlinks[i] + for (let k = 0; k < ins.length; k++) { const j = ins[k]; s += pr[j] / outdeg[j] } + next[i] = teleport + danglingShare + DAMPING * s + } + pr = next + } + const m = new Map() + for (let i = 0; i < N; i++) m.set(nodes[i], pr[i]) + return m +} + +async function main() { + const db = new DatabaseSync(DB_PATH, { readOnly: true }) + console.log(`Loading link graph from ${DB_PATH} ...`) + const { nodes, out, indeg, baseRank } = loadGraph(db) + const edgeCount = [...out.values()].reduce((a, v) => a + v.length, 0) + console.log(`Graph: ${nodes.length} domains, ${edgeCount} unique edges, ${indeg.size} with in-degree`) + + console.log(`PageRank: damping=${DAMPING} iters=${ITERS} ...`) + const pr = pagerank(nodes, out) + + let maxPr = 0 + for (const v of pr.values()) if (v > maxPr) maxPr = v + let maxLogIndeg = 0 + for (const v of indeg.values()) { const l = Math.log1p(v); if (l > maxLogIndeg) maxLogIndeg = l } + + // Per-domain final score: Majestic base + blended (PageRank, log-in-degree) + // authority. Both authority terms normalized to [0,1] so the blend is stable + // whether the graph is sparse (early) or dense (mature). + function domainScore(d) { + const prN = maxPr > 0 ? (pr.get(d) || 0) / maxPr : 0 + const inN = maxLogIndeg > 0 ? Math.log1p(indeg.get(d) || 0) / maxLogIndeg : 0 + const authority = AUTH_SCALE * (PR_WEIGHT * prN + INDEG_WEIGHT * inN) + return Math.round(baseScore(baseRank.get(d)) + authority) + } + + // Log the top domains this recompute promotes — a quick sanity check. + const top = nodes + .filter(d => pr.has(d)) + .sort((a, b) => domainScore(b) - domainScore(a)) + .slice(0, 15) + console.log('Top domains by recomputed authority:') + for (const d of top) { + console.log(` ${domainScore(d).toString().padStart(9)} ${d} (pr=${(pr.get(d) / (maxPr || 1)).toFixed(3)}, indeg=${indeg.get(d) || 0})`) + } + + // Patch every indexed document's score via partial PUT updates. + console.log('Patching document scores in Meili ...') + let offset = 0, updated = 0, lastTask = null + for (;;) { + const res = await meili('GET', `/indexes/${INDEX}/documents?fields=id,domain&limit=${PAGE}&offset=${offset}`) + const results = res.results || [] + if (results.length === 0) break + const patch = results.map(r => ({ id: r.id, score: domainScore(r.domain) })) + lastTask = await meili('PUT', `/indexes/${INDEX}/documents`, patch) // partial: only `score` changes + updated += patch.length + offset += results.length + process.stdout.write(`\r[recompute] queued score updates for ${updated} docs...`) + if (results.length < PAGE) break + } + db.close() + console.log(`\nDone. Queued authority scores for ${updated} documents. Last Meili task: ${lastTask?.taskUid ?? 'n/a'}`) +} + +main().catch(e => { console.error(e); process.exit(1) }) diff --git a/scripts/search-recompute.service b/scripts/search-recompute.service new file mode 100644 index 0000000..2d0f65e --- /dev/null +++ b/scripts/search-recompute.service @@ -0,0 +1,21 @@ +[Unit] +Description=RADICAL_SEARCH authority recompute (PageRank -> Meili scores) +After=network.target meilisearch.service +Requires=meilisearch.service + +[Service] +Type=oneshot +User=root +WorkingDirectory=/opt/search-engine/crawler +ExecStart=/usr/bin/node --experimental-sqlite src/recompute.js +# Keep the recompute from contending with the live crawl for CPU. +CPUQuota=120% +MemoryMax=1500M + +Environment=FRONTIER_DB=/opt/search-engine/crawler/frontier.db +Environment=MEILI_URL=http://127.0.0.1:7700 +# MEILI_KEY is injected via an override drop-in on the box (keeps the key out of git). +Environment=DAMPING=0.85 +Environment=ITERS=40 +Environment=PR_WEIGHT=0.6 +Environment=INDEG_WEIGHT=0.4 diff --git a/scripts/search-recompute.timer b/scripts/search-recompute.timer new file mode 100644 index 0000000..50791be --- /dev/null +++ b/scripts/search-recompute.timer @@ -0,0 +1,12 @@ +[Unit] +Description=Run RADICAL_SEARCH authority recompute periodically + +[Timer] +# First run 20 min after boot, then every 3 hours. As the crawl matures the +# link graph densifies and each recompute sharpens ranking. +OnBootSec=20min +OnUnitActiveSec=3h +Persistent=true + +[Install] +WantedBy=timers.target