← back to Fantasea Audit

crawl.mjs

134 lines

#!/usr/bin/env node
// FantaSea Yachts deep-dive crawler — grounded audit, no guessing.
// Enumerates every URL from the Yoast sitemaps, times each page, harvests
// every link, checks each unique link for dead status, and maps where
// contact info (phone/email/address) appears.
import { writeFileSync } from 'node:fs';

const ORIGIN = 'https://www.fantaseayachts.com';
const UA = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124 Safari/537.36';
const H = { 'user-agent': UA, 'accept': 'text/html,application/xhtml+xml' };

const sleep = ms => new Promise(r => setTimeout(r, ms));

async function getText(url, timeout = 20000) {
  const ctl = new AbortController();
  const t = setTimeout(() => ctl.abort(), timeout);
  const t0 = Date.now();
  try {
    const res = await fetch(url, { headers: H, redirect: 'follow', signal: ctl.signal });
    const body = await res.text();
    return { ok: true, status: res.status, ms: Date.now() - t0, bytes: Buffer.byteLength(body), body, finalUrl: res.url };
  } catch (e) {
    return { ok: false, status: 0, ms: Date.now() - t0, bytes: 0, body: '', err: String(e.name || e) };
  } finally { clearTimeout(t); }
}

async function checkLink(url) {
  const ctl = new AbortController();
  const t = setTimeout(() => ctl.abort(), 15000);
  const t0 = Date.now();
  try {
    let res = await fetch(url, { method: 'HEAD', headers: { 'user-agent': UA }, redirect: 'follow', signal: ctl.signal });
    // Some servers reject HEAD (405) — retry GET.
    if (res.status === 405 || res.status === 501) {
      res = await fetch(url, { method: 'GET', headers: { 'user-agent': UA }, redirect: 'follow', signal: ctl.signal });
    }
    return { url, status: res.status, ms: Date.now() - t0, finalUrl: res.url };
  } catch (e) {
    return { url, status: 0, ms: Date.now() - t0, err: String(e.name || e) };
  } finally { clearTimeout(t); }
}

async function sitemapUrls(sm) {
  const r = await getText(sm, 15000);
  if (!r.ok) return [];
  return [...r.body.matchAll(/<loc>([^<]+)<\/loc>/gi)].map(m => m[1].trim());
}

function extractLinks(html, pageUrl) {
  const out = new Set();
  for (const m of html.matchAll(/<a\b[^>]*\bhref\s*=\s*["']([^"'#]+)["']/gi)) {
    let href = m[1].trim();
    if (!href || href.startsWith('javascript:') || href.startsWith('mailto:') || href.startsWith('tel:')) continue;
    try { out.add(new URL(href, pageUrl).href.replace(/\/$/, '')); } catch {}
  }
  return [...out];
}

// contact patterns
const PHONE = /(\+?1[\s.-]?)?\(?\d{3}\)?[\s.-]?\d{3}[\s.-]?\d{4}/g;
const EMAIL = /[a-z0-9._%+-]+@[a-z0-9.-]+\.[a-z]{2,}/gi;

function contactSignals(html) {
  const text = html.replace(/<script[\s\S]*?<\/script>/gi, '').replace(/<style[\s\S]*?<\/style>/gi, '');
  const phones = [...new Set((text.match(PHONE) || []).map(s => s.replace(/[^\d]/g, '')).filter(d => d.length >= 10))];
  const emails = [...new Set((text.match(EMAIL) || []).map(s => s.toLowerCase()).filter(e => !/\.(png|jpg|jpeg|gif|svg|webp)$/i.test(e)))];
  // where does it appear? crude header/footer/body split
  const lower = html.toLowerCase();
  const headEnd = lower.indexOf('</header>');
  const footStart = lower.lastIndexOf('<footer');
  return { phones, emails, hasFooter: footStart > -1, hasHeaderContact: headEnd > -1 && PHONE.test(html.slice(0, headEnd > -1 ? headEnd : 3000)) };
}

(async () => {
  console.log('Enumerating sitemaps…');
  const index = await sitemapUrls(`${ORIGIN}/sitemap.xml`);
  let pages = [];
  for (const sm of index) { pages.push(...await sitemapUrls(sm)); await sleep(150); }
  pages = [...new Set(pages.map(u => u.replace(/\/$/, '')))].filter(u => u.startsWith('http'));
  console.log(`Found ${pages.length} URLs across ${index.length} sitemaps.`);

  const pageResults = [];
  const allLinks = new Map(); // link -> Set(foundOnPages)
  let i = 0;
  for (const url of pages) {
    i++;
    const r = await getText(url);
    const links = r.ok ? extractLinks(r.body, url) : [];
    const contact = r.ok ? contactSignals(r.body) : { phones: [], emails: [] };
    const title = (r.body.match(/<title>([^<]*)<\/title>/i) || [])[1] || '';
    pageResults.push({ url, status: r.status, ms: r.ms, kb: +(r.bytes / 1024).toFixed(0), title: title.trim(), linkCount: links.length, phones: contact.phones, emails: contact.emails, err: r.err });
    for (const l of links) { if (!allLinks.has(l)) allLinks.set(l, new Set()); allLinks.get(l).add(url); }
    process.stdout.write(`  [${i}/${pages.length}] ${r.status} ${r.ms}ms ${(r.bytes/1024).toFixed(0)}kb  ${url.replace(ORIGIN,'')}\n`);
    await sleep(120);
  }

  // Check unique links for dead status (cap external volume politely).
  const uniqueLinks = [...allLinks.keys()];
  console.log(`\nChecking ${uniqueLinks.length} unique links for dead/slow…`);
  const linkResults = [];
  const CONC = 8;
  for (let j = 0; j < uniqueLinks.length; j += CONC) {
    const batch = uniqueLinks.slice(j, j + CONC);
    const res = await Promise.all(batch.map(checkLink));
    res.forEach(rr => { rr.foundOn = [...allLinks.get(rr.url)].slice(0, 5); rr.internal = rr.url.includes('fantaseayachts.com'); linkResults.push(rr); });
    process.stdout.write(`  links ${Math.min(j+CONC,uniqueLinks.length)}/${uniqueLinks.length}\n`);
  }

  const dead = linkResults.filter(l => l.status === 0 || l.status >= 400);
  const slowPages = pageResults.filter(p => p.ms >= 2500).sort((a,b)=>b.ms-a.ms);
  const heavyPages = pageResults.filter(p => p.kb >= 200).sort((a,b)=>b.kb-a.kb);

  const report = { crawledAt: new Date().toISOString(), origin: ORIGIN, pageCount: pageResults.length, linkCount: linkResults.length,
    deadCount: dead.length, pages: pageResults, links: linkResults, dead, slowPages, heavyPages };
  writeFileSync('/Users/macstudio3/Projects/fantasea-audit/audit.json', JSON.stringify(report, null, 2));

  console.log('\n================ SUMMARY ================');
  console.log(`Pages crawled: ${pageResults.length}`);
  console.log(`Unique links checked: ${linkResults.length}`);
  console.log(`DEAD links (0/4xx/5xx): ${dead.length}`);
  dead.slice(0, 40).forEach(d => console.log(`   ✗ ${d.status||d.err}  ${d.url}  (on ${d.foundOn[0]?.replace(ORIGIN,'')||'?'})`));
  console.log(`\nSLOW pages (>=2.5s): ${slowPages.length}`);
  slowPages.slice(0,15).forEach(p => console.log(`   ⏱ ${p.ms}ms ${p.kb}kb  ${p.url.replace(ORIGIN,'')}`));
  console.log(`\nHEAVY pages (>=200kb): ${heavyPages.length}`);
  heavyPages.slice(0,15).forEach(p => console.log(`   🐘 ${p.kb}kb ${p.ms}ms  ${p.url.replace(ORIGIN,'')}`));
  const withPhone = pageResults.filter(p=>p.phones.length).length;
  const withEmail = pageResults.filter(p=>p.emails.length).length;
  console.log(`\nCONTACT footprint: phone on ${withPhone}/${pageResults.length} pages, email on ${withEmail}/${pageResults.length} pages`);
  const allPhones=[...new Set(pageResults.flatMap(p=>p.phones))]; const allEmails=[...new Set(pageResults.flatMap(p=>p.emails))];
  console.log(`   distinct phones: ${allPhones.join(', ')||'none'}`);
  console.log(`   distinct emails: ${allEmails.join(', ')||'none'}`);
  console.log('\nWrote audit.json');
})();