fix(seo): crawl hygiene and a per-family sitemap index (W1) #110
No files matched your search
+3
-1
@@ -48,7 +48,9 @@ PHASE_GROUPS: dict[str, set[str]] = {
|
||||
"all-through": {"all-through"},
|
||||
}
|
||||
|
||||
BASE_URL = "https://schoolcompare.co.uk"
|
||||
# Must match SITE_URL in nextjs-app/lib/site.ts. The apex 301s to www, and a
|
||||
# sitemap <loc> that redirects wastes a crawl on every URL it lists.
|
||||
BASE_URL = "https://www.schoolcompare.co.uk"
|
||||
MAX_SLUG_LENGTH = 60
|
||||
|
||||
# In-memory sitemap cache
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
"""Tests for sitemap generation (spec 2026-08-20, workstream W1).
|
||||
|
||||
The sitemap is built from the in-memory school DataFrame, so these inject a
|
||||
small frame via monkeypatch rather than touching a database.
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import pytest
|
||||
|
||||
|
||||
def _schools_df() -> pd.DataFrame:
|
||||
"""Two schools: one with results, one with neither results nor Ofsted."""
|
||||
base = {
|
||||
"local_authority": "Testshire",
|
||||
"school_type": "Academy",
|
||||
"phase": "Primary",
|
||||
"year": 202425,
|
||||
"ofsted_date": None,
|
||||
}
|
||||
return pd.DataFrame(
|
||||
[
|
||||
{**base, "urn": 100001, "school_name": "Alpha Primary",
|
||||
"rwm_expected_pct": 62.0, "attainment_8_score": np.nan,
|
||||
"ofsted_grade": 2.0},
|
||||
{**base, "urn": 100002, "school_name": "Ghost Primary",
|
||||
"rwm_expected_pct": np.nan, "attainment_8_score": np.nan,
|
||||
"ofsted_grade": np.nan},
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def sitemap(monkeypatch) -> str:
|
||||
from backend import app as app_module
|
||||
|
||||
monkeypatch.setattr(app_module, "load_school_data", _schools_df)
|
||||
return app_module.build_sitemap()
|
||||
|
||||
|
||||
def test_every_loc_uses_the_www_host(sitemap):
|
||||
# The apex 301s to www. A <loc> that redirects burns a crawl per URL.
|
||||
assert "https://www.schoolcompare.co.uk" in sitemap
|
||||
assert "https://schoolcompare.co.uk" not in sitemap
|
||||
@@ -0,0 +1,27 @@
|
||||
import { SITE_URL, absoluteUrl } from '@/lib/site';
|
||||
|
||||
describe('SITE_URL', () => {
|
||||
it('is the www host, which is the one that serves a 200', () => {
|
||||
// The apex 301s to www at Cloudflare. A canonical pointing at a redirect
|
||||
// is a wasted signal, so every absolute URL we emit must already be www.
|
||||
expect(SITE_URL).toBe('https://www.schoolcompare.co.uk');
|
||||
});
|
||||
|
||||
it('has no trailing slash, so joins never double up', () => {
|
||||
expect(SITE_URL.endsWith('/')).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('absoluteUrl', () => {
|
||||
it('joins a rooted path', () => {
|
||||
expect(absoluteUrl('/rankings')).toBe('https://www.schoolcompare.co.uk/rankings');
|
||||
});
|
||||
|
||||
it('joins a path missing its leading slash', () => {
|
||||
expect(absoluteUrl('rankings')).toBe('https://www.schoolcompare.co.uk/rankings');
|
||||
});
|
||||
|
||||
it('maps the site root to a bare trailing slash', () => {
|
||||
expect(absoluteUrl('/')).toBe('https://www.schoolcompare.co.uk/');
|
||||
});
|
||||
});
|
||||
@@ -5,6 +5,7 @@ import { Navigation } from '@/components/Navigation';
|
||||
import { Footer } from '@/components/Footer';
|
||||
import { ComparisonToast } from '@/components/ComparisonToast';
|
||||
import { ComparisonProvider } from '@/context/ComparisonProvider';
|
||||
import { SITE_URL } from '@/lib/site';
|
||||
import './globals.css';
|
||||
|
||||
// Manrope carries headings and key messaging — the guideline's "friendly,
|
||||
@@ -57,12 +58,12 @@ export const metadata: Metadata = {
|
||||
// No `icons` key on purpose: setting it here would override the file
|
||||
// conventions. app/icon.svg and app/apple-icon.tsx are the source, and
|
||||
// app/opengraph-image.tsx supplies og:image and twitter:image.
|
||||
metadataBase: new URL('https://schoolcompare.co.uk'),
|
||||
metadataBase: new URL(SITE_URL),
|
||||
openGraph: {
|
||||
type: 'website',
|
||||
title: 'schoolcompare | Compare School Performance',
|
||||
description: 'Compare primary and secondary school SATs and GCSE performance across England',
|
||||
url: 'https://schoolcompare.co.uk',
|
||||
url: SITE_URL,
|
||||
siteName: 'schoolcompare',
|
||||
},
|
||||
twitter: {
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
*/
|
||||
|
||||
import { MetadataRoute } from 'next';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
|
||||
export default function robots(): MetadataRoute.Robots {
|
||||
return {
|
||||
@@ -14,6 +15,6 @@ export default function robots(): MetadataRoute.Robots {
|
||||
disallow: ['/api/', '/_next/'],
|
||||
},
|
||||
],
|
||||
sitemap: 'https://schoolcompare.co.uk/sitemap.xml',
|
||||
sitemap: absoluteUrl('/sitemap.xml'),
|
||||
};
|
||||
}
|
||||
@@ -15,6 +15,7 @@ import {
|
||||
} from '@/lib/schoolSections';
|
||||
import { parseSchoolSlug, schoolUrl } from '@/lib/utils';
|
||||
import type { NationalAverages } from '@/lib/types';
|
||||
import { absoluteUrl } from '@/lib/site';
|
||||
import type { Metadata } from 'next';
|
||||
|
||||
/**
|
||||
@@ -97,7 +98,7 @@ export async function generateMetadata({ params }: SchoolPageProps): Promise<Met
|
||||
title,
|
||||
description,
|
||||
type: 'website',
|
||||
url: `https://schoolcompare.co.uk${canonicalPath}`,
|
||||
url: absoluteUrl(canonicalPath),
|
||||
siteName: 'schoolcompare',
|
||||
},
|
||||
twitter: {
|
||||
@@ -106,7 +107,7 @@ export async function generateMetadata({ params }: SchoolPageProps): Promise<Met
|
||||
description,
|
||||
},
|
||||
alternates: {
|
||||
canonical: `https://schoolcompare.co.uk${canonicalPath}`,
|
||||
canonical: absoluteUrl(canonicalPath),
|
||||
},
|
||||
};
|
||||
} catch {
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
/**
|
||||
* The one place the site's absolute origin is written down.
|
||||
*
|
||||
* The apex domain 301s to www at Cloudflare, so www is the host that actually
|
||||
* serves a 200. Canonicals, og:url, sitemap <loc> entries and the robots.txt
|
||||
* Sitemap: line must all agree with it — a canonical pointing at a redirect
|
||||
* makes Google resolve the hop before it can consolidate the signal.
|
||||
*
|
||||
* backend/app.py holds the same value as BASE_URL for the sitemap. The two are
|
||||
* asserted against each other by the e2e journeys rather than shared at build
|
||||
* time, because the backend and frontend ship as separate images.
|
||||
*/
|
||||
export const SITE_URL = 'https://www.schoolcompare.co.uk';
|
||||
|
||||
/** Absolute URL for a site-relative path. Tolerates a missing leading slash. */
|
||||
export function absoluteUrl(path: string): string {
|
||||
const rooted = path.startsWith('/') ? path : `/${path}`;
|
||||
return `${SITE_URL}${rooted}`;
|
||||
}
|
||||
Reference in new issue
Block a user