diff --git a/scripts/build-pages.js b/scripts/build-pages.js index 7679fe12..d0c16bb9 100644 --- a/scripts/build-pages.js +++ b/scripts/build-pages.js @@ -152,6 +152,13 @@ function renderPrompt(command) { ); } +// Every mirror page is emitted noindex. root.vc should be the only result +// Google shows: the terminal is the front door, and sitelinks to /about/ and +// /team/ give the trick away. noindex is a search-INDEXING directive, not a +// fetch permission — crawlers may still read the page, so GPTBot, ClaudeBot and +// PerplexityBot keep getting the full content. That is the whole reason this is +// not a robots.txt Disallow, which would hide the mirror from them too. +// "follow" keeps link equity flowing back to the homepage. function layout({ title, description, pathname, command, body, graph }) { const canonical = absUrl(pathname); const jsonLd = jsonLdScript({ "@context": "https://schema.org", "@graph": graph }); @@ -163,6 +170,7 @@ function layout({ title, description, pathname, command, body, graph }) { ${esc(title)} + @@ -947,14 +955,14 @@ function buildPages(config = loadConfig()) { files.push(renderPerson(config, slug)); }); - // Sitemap covers the terminal, the GeoCities page, and every generated page. + // The sitemap lists ONLY the homepage. Every mirror page is noindex, and + // listing a noindex URL in a sitemap asks Google to index something the page + // itself forbids — that contradiction is what produced sitelinks to /about/, + // /team/, and the rest under the root.vc result. Crawlers still reach the + // mirror via the