Rene Fichtmueller 49aa46fef3 refactor: extract relationships/compare/quick-ix/peers-find/prefix-detail/ix-detail/topology/whois/enrich routes
Bug fix (agreed pre-existing bug #2 of 3): /api/compare called
fetchRPKIPerPrefix(asn, prefix), never defined anywhere in the codebase --
threw ReferenceError whenever either ASN had >=1 sample prefix, caught by
the outer try/catch, returning 500. Replaced both call sites with
validateRPKIWithCache(asn, prefix), the existing single-prefix RPKI check
used everywhere else in the codebase -- its return shape ({status, ...})
is exactly what the surrounding code already expects.

Verified via node --check, eslint no-undef (only 1 known bug remains:
webhooks POST's `secret`), and smoke-test harness (28/28 match). The
compare fix itself isn't exercised by the smoke test's fixed test ASNs
(both return empty prefix samples locally), but the fix is a direct,
same-shape substitution of an already-proven function.

server.js: 1736 -> 989 lines.
2026-07-17 07:57:04 +02:00

126 lines
6.2 KiB
JavaScript

const http = require("http");
const https = require("https");
const { fetchJSON } = require("../services/http-helpers");
// Feature 28b: Company enrichment via Wikipedia + website meta scraping
async function handleEnrich(req, res, url) {
res.setHeader("Content-Type", "application/json");
res.setHeader("Access-Control-Allow-Origin", "*");
res.setHeader("Cache-Control", "no-store");
const rawAsn = (url.searchParams.get("asn") || "").replace(/[^0-9]/g, "");
const companyName = (url.searchParams.get("name") || "").trim();
const website = (url.searchParams.get("website") || "").trim();
const UA_SCRAPE = "Mozilla/5.0 (compatible; PeerCortex/1.0; +https://peercortex.org)";
let description = null;
let wikiUrl = null;
try {
// 1. Direct Wikipedia lookup by company name
if (companyName) {
const nameLower = companyName.toLowerCase();
const isRelevant = (title, extract) => {
const titleLower = (title || "").toLowerCase();
const extractLower = (extract || "").toLowerCase();
const nameWords = nameLower.replace(/\s+(gmbh|ag|ltd|inc|llc|bv|sa|sas|oy|ab)\s*$/i, "").split(/\s+/).filter(w => w.length > 3);
const titleMatch = nameWords.some(w => titleLower.includes(w));
const netTerms = ["internet", "network", "isp", "hosting", "provider", "transit", "data center", "datacenter", "telecommunications", "telecom", "bandwidth", "peering", "routing", "autonomous system", "colocation", "colo", "fiber", "optical", "transceiver"];
const hasNetContext = netTerms.some(t => extractLower.includes(t));
return titleMatch && (hasNetContext || titleMatch);
};
// Direct title lookup — try full name first, then first word as fallback
const cleanName = companyName.replace(/\s+(GmbH|AG|Ltd|Inc|LLC|BV|SA|SAS|Oy|AB)$/i, "").trim();
const firstName = cleanName.split(/\s+/)[0];
const namesToTry = cleanName === firstName ? [cleanName] : [cleanName, firstName];
for (const tryName of namesToTry) {
const wikiDirect = await fetchJSON(
"https://en.wikipedia.org/api/rest_v1/page/summary/" + encodeURIComponent(tryName),
{ timeout: 5000 }
);
if (wikiDirect && wikiDirect.type === "disambiguation") continue; // skip disambiguation pages
if (wikiDirect && wikiDirect.extract && wikiDirect.extract.length > 30 && isRelevant(wikiDirect.title, wikiDirect.extract)) {
description = wikiDirect.extract.replace(/\s+/g, " ").trim().slice(0, 300);
wikiUrl = wikiDirect.content_urls && wikiDirect.content_urls.desktop && wikiDirect.content_urls.desktop.page;
break;
}
}
// 2. Wikipedia search if direct lookup didn't match
if (!description) {
const searchQuery = companyName.replace(/\s+(GmbH|AG|Ltd|Inc|LLC)$/i, "") + " internet service provider";
const searchData = await fetchJSON(
"https://en.wikipedia.org/w/api.php?action=opensearch&format=json&search=" + encodeURIComponent(searchQuery) + "&limit=3",
{ timeout: 5000 }
);
if (searchData && Array.isArray(searchData) && searchData[1] && searchData[1].length > 0) {
const topTitle = searchData[1][0];
const wikiSearch = await fetchJSON(
"https://en.wikipedia.org/api/rest_v1/page/summary/" + encodeURIComponent(topTitle),
{ timeout: 5000 }
);
if (wikiSearch && wikiSearch.extract && wikiSearch.extract.length > 30) {
if (isRelevant(wikiSearch.title, wikiSearch.extract)) {
description = wikiSearch.extract.replace(/\s+/g, " ").trim().slice(0, 300);
wikiUrl = wikiSearch.content_urls && wikiSearch.content_urls.desktop && wikiSearch.content_urls.desktop.page;
}
}
}
}
}
// 3. Fallback: scrape website meta description (follows up to 3 redirects)
function fetchPage(pageUrl, hops) {
if (hops <= 0) return Promise.resolve(null);
return new Promise((resolve) => {
const mod = pageUrl.startsWith("https") ? https : http;
const req = mod.get(pageUrl, { headers: { "User-Agent": UA_SCRAPE }, timeout: 6000 }, (res) => {
if (res.statusCode >= 300 && res.statusCode < 400 && res.headers.location) {
res.resume(); // drain to free socket
const next = res.headers.location.startsWith("http") ? res.headers.location : new URL(res.headers.location, pageUrl).href;
return resolve(fetchPage(next, hops - 1));
}
let data = "";
res.on("data", (c) => { data += c; if (data.length > 40000) { req.destroy(); resolve(data); } });
res.on("end", () => resolve(data));
});
req.on("error", () => resolve(null));
req.on("timeout", () => { req.destroy(); resolve(null); });
});
}
if (!description && website) {
let wsUrl = website;
if (!wsUrl.startsWith("http")) wsUrl = "https://" + wsUrl;
const aboutUrl = wsUrl.replace(/\/$/, "") + "/about";
const tryUrls = [aboutUrl, wsUrl];
for (const tryUrl of tryUrls) {
const resp = await fetchPage(tryUrl, 3);
if (!resp) continue;
const metaPatterns = [
/<meta[^>]+name=["']description["'][^>]+content=["']([^"']{20,500})["']/i,
/<meta[^>]+content=["']([^"']{20,500})["'][^>]+name=["']description["']/i,
/<meta[^>]+property=["']og:description["'][^>]+content=["']([^"']{20,500})["']/i,
/<meta[^>]+content=["']([^"']{20,500})["'][^>]+property=["']og:description["']/i,
];
let matched = false;
for (const pat of metaPatterns) {
const m = resp.match(pat);
if (m && m[1]) {
description = m[1].replace(/&#10;|&#10003;|&amp;/g, " ").replace(/\s+/g, " ").trim().slice(0, 300);
matched = true;
break;
}
}
if (matched) break;
}
}
} catch (_e) {
// Return whatever we have
}
return res.end(JSON.stringify({ asn: rawAsn, description, wiki_url: wikiUrl }));
}
module.exports = { handleEnrich };