diff --git a/.github/workflows/deploy.yml b/.github/workflows/deploy.yml index 06ecd08..20a4efe 100644 --- a/.github/workflows/deploy.yml +++ b/.github/workflows/deploy.yml @@ -75,6 +75,12 @@ jobs: || { echo "::error::404.html has the wrong pathSegmentsToKeep for an apex domain."; exit 1; } echo "build output OK" + # Its own step, because this failure mode is silent: if the per-route emit + # stops running the site still builds, deploys and works — every page just + # quietly goes back to sharing the homepage's title and link preview. + - name: Verify per-route SEO + run: npm run verify:seo + - name: Upload build artifact uses: actions/upload-artifact@v4 with: diff --git a/package.json b/package.json index d5dfbc1..75e4b04 100644 --- a/package.json +++ b/package.json @@ -8,6 +8,7 @@ "dev": "vite", "build": "vite build", "lint": "eslint .", + "verify:seo": "node scripts/verify-seo.mjs", "preview": "vite preview", "predeploy": "npm run build", "deploy": "gh-pages -d dist", diff --git a/public/robots.txt b/public/robots.txt deleted file mode 100644 index 60621b4..0000000 --- a/public/robots.txt +++ /dev/null @@ -1,4 +0,0 @@ -User-agent: * -Allow: / - -Sitemap: https://kalapritidesigns.com/sitemap.xml diff --git a/public/sitemap.xml b/public/sitemap.xml deleted file mode 100644 index 1371404..0000000 --- a/public/sitemap.xml +++ /dev/null @@ -1,14 +0,0 @@ - - - - https://kalapritidesigns.com/1.0 - https://kalapritidesigns.com/projects0.9 - https://kalapritidesigns.com/services0.9 - https://kalapritidesigns.com/process0.8 - https://kalapritidesigns.com/resources0.7 - https://kalapritidesigns.com/about0.8 - https://kalapritidesigns.com/contact0.9 - diff --git a/scripts/verify-seo.mjs b/scripts/verify-seo.mjs new file mode 100644 index 0000000..c397562 --- /dev/null +++ b/scripts/verify-seo.mjs @@ -0,0 +1,110 @@ +#!/usr/bin/env node +/** + * Build guard for the per-route SEO artifacts. + * + * Run after `npm run build`, locally or in CI: node scripts/verify-seo.mjs + * + * This exists because the failure mode is silent. If the per-route emit in + * vite.config.js stops running, the site still builds, still deploys, and still + * works — every page just quietly goes back to sharing the homepage's title, + * description and link preview. Nobody notices until someone checks Search + * Console weeks later. + */ +import { existsSync, readFileSync } from 'node:fs' +import { readdirSync } from 'node:fs' +import { join, resolve } from 'node:path' +import { ROUTES, canonicalFor } from '../src/data/seo.js' + +const DIST = resolve(import.meta.dirname, '..', 'dist') +const problems = [] +const fail = (msg) => problems.push(msg) + +if (!existsSync(DIST)) { + console.error('dist/ does not exist — run `npm run build` first.') + process.exit(1) +} + +/** Exactly one of each. Two titles is as broken as none: the browser and the + * crawler each pick one, and they do not have to pick the same one. */ +const SINGLETONS = [ + ['title', /([^<]*)<\/title>/g], + ['description', /name="description"\s+content="([^"]*)"/g], + ['og:title', /property="og:title"\s+content="([^"]*)"/g], + ['og:description', /property="og:description"\s+content="([^"]*)"/g], + ['canonical', /<link rel="canonical" href="([^"]*)"/g], + ['og:url', /property="og:url"\s+content="([^"]*)"/g], +] + +const read = (p) => readFileSync(p, 'utf8') +const all = (html, re) => [...html.matchAll(new RegExp(re.source, 'g'))].map((m) => m[1]) + +const seenTitles = new Map() + +for (const route of ROUTES) { + const slug = route.path === '/' ? null : route.path.replace(/^\//, '') + // Both URL shapes, because which one GitHub Pages prefers is not worth + // guessing at — and a page that 404s is worse than one indexed twice. + const files = slug ? [`${slug}.html`, join(slug, 'index.html')] : ['index.html'] + + for (const rel of files) { + const abs = join(DIST, rel) + if (!existsSync(abs)) { + fail(`${rel} was not emitted — per-route SEO did not run for ${route.path}`) + continue + } + const html = read(abs) + + for (const [label, re] of SINGLETONS) { + const found = all(html, re) + if (found.length !== 1) fail(`${rel}: ${found.length} ${label} tags, expected exactly 1`) + } + + const [title] = all(html, /<title>([^<]*)<\/title>/) + const [canonical] = all(html, /<link rel="canonical" href="([^"]*)"/) + const [desc] = all(html, /name="description"\s+content="([^"]*)"/) + const [ogDesc] = all(html, /property="og:description"\s+content="([^"]*)"/) + + const want = canonicalFor(route.path) + if (canonical !== want) fail(`${rel}: canonical is ${canonical}, expected ${want}`) + if (desc !== ogDesc) fail(`${rel}: description and og:description diverged`) + + // Decoded, because the shell escapes & as & before this ever runs. + const decoded = title.replace(/&/g, '&') + if (decoded !== route.title) fail(`${rel}: title is "${decoded}", expected "${route.title}"`) + + const robots = all(html, /name="robots"\s+content="([^"]*)"/) + if (route.noindex && robots.length !== 1) fail(`${rel}: placeholder route is missing noindex`) + if (!route.noindex && robots.length) fail(`${rel}: indexable route carries a robots tag`) + + const owner = seenTitles.get(title) + if (owner && owner !== route.path) fail(`${rel} shares its title with ${owner}`) + seenTitles.set(title, route.path) + } +} + +// Nothing should be left behind claiming a title that no route owns. +for (const entry of readdirSync(DIST, { withFileTypes: true })) { + if (!entry.isFile() || !entry.name.endsWith('.html') || entry.name === '404.html') continue + const slug = entry.name === 'index.html' ? '/' : `/${entry.name.replace(/\.html$/, '')}` + if (!ROUTES.some((r) => r.path === slug)) fail(`dist/${entry.name} has no route in src/data/seo.js`) +} + +const sitemap = existsSync(join(DIST, 'sitemap.xml')) ? read(join(DIST, 'sitemap.xml')) : '' +if (!sitemap) fail('sitemap.xml was not generated') +for (const route of ROUTES) { + const listed = sitemap.includes(`<loc>${canonicalFor(route.path)}</loc>`) + if (route.noindex && listed) fail(`sitemap lists ${route.path}, which is noindex`) + if (!route.noindex && !listed) fail(`sitemap is missing ${route.path}`) +} + +if (problems.length) { + for (const p of problems) console.error(`::error::${p}`) + console.error(`\nper-route SEO verification failed: ${problems.length} problem(s)`) + process.exit(1) +} + +const indexed = ROUTES.filter((r) => !r.noindex).length +console.log( + `per-route SEO OK — ${ROUTES.length} routes, ${seenTitles.size} distinct titles, ` + + `${indexed} in the sitemap`, +) diff --git a/src/components/Layout.jsx b/src/components/Layout.jsx index 282176c..11de0d0 100644 --- a/src/components/Layout.jsx +++ b/src/components/Layout.jsx @@ -3,6 +3,7 @@ import { useLocation } from 'wouter' import Lenis from 'lenis' import 'lenis/dist/lenis.css' import { gsap, ScrollTrigger, finePointer, reducedMotion } from '../lib/motion' +import { useRouteMeta } from '../lib/seo' import { FEATURES } from '../data/site' import Nav from './Nav' import Footer from './Footer' @@ -97,6 +98,7 @@ export default function Layout({ children }) { const lenisRef = useLenis() const mainRef = useRef(null) const ruleRef = useRef(null) + useRouteMeta() useMagnetic() usePageTransition(lenisRef, mainRef, ruleRef) diff --git a/src/data/seo.js b/src/data/seo.js new file mode 100644 index 0000000..f7d1ea0 --- /dev/null +++ b/src/data/seo.js @@ -0,0 +1,100 @@ +/** + * PER-ROUTE SEO — one table, two consumers. + * + * At build time, vite.config.js reads this to emit a real HTML file per route, + * each with its own title, description, canonical and Open Graph tags. That is + * what crawlers and link scrapers see, and it is the only version WhatsApp, + * Facebook and Twitter ever see, because none of them run JavaScript. + * + * At runtime, src/lib/seo.js reads the same table to update the live document + * when the router moves between pages client-side, so the tab title and the + * canonical stay correct without a reload. + * + * Keep this file free of imports. vite.config.js loads it in plain Node, where + * import.meta.env and anything Vite-specific does not exist. + * + * Writing rules, so these stay useful rather than decorative: + * title — under ~60 characters before Google truncates it. The brand + * suffix is added here, not by the consumer. + * description — 120-158 characters. It is not a ranking factor; it is the + * line that decides whether anyone clicks. + * noindex — for routes that exist and route correctly but have nothing + * worth indexing yet. They stay reachable; they stay out of + * the sitemap and out of the index. + */ + +export const SITE_URL = 'https://kalapritidesigns.com' + +/** Shared Open Graph values. Per-route entries override title and description. */ +export const OG_DEFAULTS = { + siteName: 'Kalapriti Designs', + image: `${SITE_URL}/assets/hero.png`, + type: 'website', +} + +export const ROUTES = [ + { + path: '/', + title: 'Kalapriti Designs | Architectural & Design Consultancy', + description: + 'Kalapriti Designs is an architecture and interior practice working across exterior and interior design — from first sketch to styled handover.', + priority: '1.0', + }, + { + path: '/services', + title: 'Services — Architecture, Interiors, Landscape | Kalapriti Designs', + description: + 'Architecture planning, landscape, interior design and renovation — offered as advisory or end-to-end turnkey delivery. What we take on, and how far we carry it.', + priority: '0.9', + }, + { + path: '/process', + title: 'How a Project Runs | Kalapriti Designs', + description: + 'The same six steps on every project — brief, concept, drawings, approvals, build and handover — so you always know what is settled and what comes next.', + priority: '0.8', + }, + { + path: '/about', + title: 'About the Practice | Kalapriti Designs', + description: + 'An online-first architecture and design consultancy led by Jitendrakumar Patel, built around brand value and buildable detail. Working across Gujarat and beyond.', + priority: '0.8', + }, + { + path: '/contact', + title: 'Start a Project | Kalapriti Designs', + description: + 'Tell us roughly what you have in mind. The first consultation is a conversation, not a commitment. Call, WhatsApp or send an enquiry.', + priority: '0.9', + }, + + // Routed and reachable, but placeholder content. Indexing them now would put + // thin pages in front of the pages that are finished. + { + path: '/projects', + title: 'Projects | Kalapriti Designs', + description: + 'Exterior and interior work across residential and commercial briefs.', + noindex: true, + }, + { + path: '/resources', + title: 'Resources | Kalapriti Designs', + description: + 'Practical guidance on planning, budgeting and running a design project — free to read, no form in the way.', + noindex: true, + }, +] + +export const ROUTE_SEO = Object.fromEntries(ROUTES.map((r) => [r.path, r])) + +/** Unknown paths render NotFound, which should never be indexed. */ +export const NOT_FOUND_SEO = { + path: null, + title: 'Page not found | Kalapriti Designs', + description: 'That page does not exist. Everything else is one click away.', + noindex: true, +} + +export const canonicalFor = (path) => `${SITE_URL}${path === '/' ? '/' : path}` diff --git a/src/lib/seo.js b/src/lib/seo.js new file mode 100644 index 0000000..ae7893d --- /dev/null +++ b/src/lib/seo.js @@ -0,0 +1,88 @@ +import { useEffect } from 'react' +import { useLocation } from 'wouter' +import { ROUTE_SEO, NOT_FOUND_SEO, OG_DEFAULTS, canonicalFor } from '../data/seo' + +/** + * Keeps the live document's metadata in step with the route. + * + * This *mutates* the tags the build already put in the head rather than + * rendering new ones. React 19 can hoist <title> and <meta> on its own, but it + * appends — it does not replace what the static HTML shipped — so rendering + * them would leave two titles and two descriptions on every page, and the + * browser would keep the first. + * + * Only client-side navigation needs this. On a cold load the file the server + * sent is already correct: vite.config.js emits one HTML file per route. That + * matters because link scrapers (WhatsApp, Facebook, Twitter) never run this + * code — they read the shipped HTML and stop. + */ + +const setMeta = (selector, attr, value) => { + const el = document.head.querySelector(selector) + if (el) el.setAttribute(attr, value) + return el +} + +/** + * Creates the tag if it is absent, updates it if present, removes it when + * `value` is null. Tags that only some routes carry — robots on placeholders, + * canonical on anything real — have to be addable and removable, not just + * settable: leaving a stale one behind is worse than never having had it. + */ +function syncTag(selector, make, attr, value) { + const existing = document.head.querySelector(selector) + if (value === null) { + existing?.remove() + return + } + const el = existing ?? document.head.appendChild(make()) + el.setAttribute(attr, value) +} + +const makeMeta = (name, key = 'name') => () => { + const el = document.createElement('meta') + el.setAttribute(key, name) + return el +} + +const makeCanonical = () => { + const el = document.createElement('link') + el.setAttribute('rel', 'canonical') + return el +} + +export function useRouteMeta() { + const [loc] = useLocation() + + useEffect(() => { + const seo = ROUTE_SEO[loc] ?? NOT_FOUND_SEO + const url = seo.path ? canonicalFor(seo.path) : null + + document.title = seo.title + setMeta('meta[name="description"]', 'content', seo.description) + setMeta('meta[property="og:title"]', 'content', seo.title) + setMeta('meta[property="og:description"]', 'content', seo.description) + setMeta('meta[property="og:site_name"]', 'content', OG_DEFAULTS.siteName) + + // An unknown path has no canonical URL of its own, so the tag is removed + // rather than left pointing at whichever page was open before — a stale + // canonical tells Google this URL is a duplicate of that one, which is a + // worse claim than making none. + syncTag('link[rel="canonical"]', makeCanonical, 'href', url) + // og:url still gets the real address: link previews should show where the + // reader actually is, and nothing indexes on the strength of it. + syncTag( + 'meta[property="og:url"]', + makeMeta('og:url', 'property'), + 'content', + url ?? window.location.href, + ) + + syncTag( + 'meta[name="robots"]', + makeMeta('robots'), + 'content', + seo.noindex ? 'noindex, follow' : null, + ) + }, [loc]) +} diff --git a/vite.config.js b/vite.config.js index 70c8e69..9d2404e 100644 --- a/vite.config.js +++ b/vite.config.js @@ -1,7 +1,8 @@ -import { writeFileSync } from 'node:fs' +import { mkdirSync, readFileSync, writeFileSync } from 'node:fs' import { resolve } from 'node:path' import { defineConfig } from 'vite' import react from '@vitejs/plugin-react' +import { ROUTES, SITE_URL, canonicalFor } from './src/data/seo.js' /** * DEPLOY TARGET — GitHub Pages on the custom domain kalapritidesigns.com. @@ -52,13 +53,107 @@ function ghPagesSpaFallback(base) { ` } +/** + * PER-ROUTE HTML. + * + * The app is a single-page build, so without this every URL serves one head: + * one title, one description, one Open Graph block. Google would see five + * pages wearing the same name, and a link shared on WhatsApp would preview as + * the homepage whichever page was shared. + * + * So after the bundle is written, each route gets its own copy of index.html + * with its own head. Nothing is rendered server-side — the body is the same + * empty root div, and React boots and takes over exactly as before. The only + * difference is the handful of tags a crawler or a link scraper reads before + * it stops, and those are the ones that were wrong. + * + * Two files per route, deliberately: + * dist/services.html — GitHub Pages serves this at /services + * dist/services/index.html — and this at /services/ + * Both carry the same canonical, so the duplicate collapses to one URL. Which + * form Pages prefers is not worth guessing at when both cost 4kB. + */ +function renderRouteHtml(shell, route) { + const url = canonicalFor(route.path) + const esc = (s) => s.replace(/&/g, '&').replace(/</g, '<').replace(/"/g, '"') + + let html = shell + const sub = (pattern, replacement, label) => { + if (!pattern.test(html)) { + throw new Error( + `per-route SEO: no ${label} tag in index.html to rewrite for ${route.path}. ` + + `The shell changed shape — update vite.config.js rather than shipping a wrong head.` + ) + } + html = html.replace(pattern, replacement) + } + + sub(/<title>[\s\S]*?<\/title>/, `<title>${esc(route.title)}`, 'title') + sub( + /(', ' \n ') + } + return html +} + +function renderSitemap(routes) { + const urls = routes + .filter((r) => !r.noindex) + .map((r) => ` ${canonicalFor(r.path)}${r.priority}`) + .join('\n') + return ` + + +${urls} + +` +} + function githubPagesArtifacts({ base }) { return { name: 'kalapriti-gh-pages-artifacts', apply: 'build', // closeBundle runs after publicDir has been copied, so this always wins. closeBundle() { - writeFileSync(resolve(import.meta.dirname, 'dist', '404.html'), ghPagesSpaFallback(base)) + const dist = resolve(import.meta.dirname, 'dist') + writeFileSync(resolve(dist, '404.html'), ghPagesSpaFallback(base)) + + const shell = readFileSync(resolve(dist, 'index.html'), 'utf8') + for (const route of ROUTES) { + const html = renderRouteHtml(shell, route) + if (route.path === '/') { + writeFileSync(resolve(dist, 'index.html'), html) + continue + } + const slug = route.path.replace(/^\//, '') + writeFileSync(resolve(dist, `${slug}.html`), html) + mkdirSync(resolve(dist, slug), { recursive: true }) + writeFileSync(resolve(dist, slug, 'index.html'), html) + } + + // Generated from the same table as the pages, so the two cannot drift — + // the old hand-maintained public/sitemap.xml had already gone stale. + writeFileSync(resolve(dist, 'sitemap.xml'), renderSitemap(ROUTES)) + writeFileSync( + resolve(dist, 'robots.txt'), + `User-agent: *\nAllow: /\n\nSitemap: ${SITE_URL}/sitemap.xml\n`, + ) }, } }