279 lines
12 KiB
JavaScript
279 lines
12 KiB
JavaScript
// What must be true of the content layer before a page is built from it.
|
||
//
|
||
// ## Why this exists
|
||
//
|
||
// `src/data/*.js` is prose in a data structure, and nothing checked it. A
|
||
// misspelt block type, a link to a route that does not exist, a section whose
|
||
// answer paragraph is missing, a missing `benefits` list that crashes the
|
||
// prerender with `Cannot read properties of undefined`: each was found by a
|
||
// person looking at a page, or not found at all.
|
||
//
|
||
// The long-form service pages made that worse in a specific way. Their copy is
|
||
// owner-approved and arrives as a markdown sheet that MIXES DIRECTIONS TO THE
|
||
// WEBSITE MANAGER INTO THE COPY: "Do not promise that every number is always
|
||
// portable", "Keep this factual:", "Place an official screenshot". Those lines
|
||
// look exactly like copy. Publishing one puts an internal instruction on a
|
||
// customer-facing page, and no build step would have noticed.
|
||
//
|
||
// So this refuses to build rather than publish. It runs inside prerender.js,
|
||
// before any page renders, which means every `npm run build` enforces it: the
|
||
// pre-commit hook, `npm run verify`, and the Docker image build.
|
||
//
|
||
// Exit codes for the runner: 0 clean, 1 findings, 2 nothing was checked.
|
||
import { existsSync } from 'fs'
|
||
import path from 'path'
|
||
import { fileURLToPath } from 'url'
|
||
import { ROUTES } from './routes.js'
|
||
|
||
const publicDir = path.join(path.dirname(fileURLToPath(import.meta.url)), '../../public')
|
||
|
||
// A block type, and the field it cannot be missing.
|
||
const BLOCK_FIELDS = {
|
||
p: 'text',
|
||
h3: 'text',
|
||
ul: 'items',
|
||
ol: 'items',
|
||
callout: 'text',
|
||
image: 'alt',
|
||
links: 'items',
|
||
}
|
||
|
||
const ID_SHAPE = /^[a-z][a-z0-9]*(-[a-z0-9]+)*$/
|
||
// Ids already used by the layout. A section id that collides with one of these
|
||
// makes `#contact-form` scroll to the wrong element.
|
||
const RESERVED_IDS = new Set(['contact-form', 'mobile-nav-content', 'support-portal', 'root'])
|
||
|
||
// Directions to the website manager, which the sheets mix into their copy.
|
||
const SHEET_DIRECTIONS = [
|
||
/website manager/i,
|
||
/not for the (public )?webpage/i,
|
||
/place an official/i,
|
||
/do not add faq/i,
|
||
/this section is (important|highly useful)/i,
|
||
/\bQ\d+:/,
|
||
/approved visual/i,
|
||
/suggested (alt text|caption)/i,
|
||
/before publication/i,
|
||
/avoid (unsupported|promising|implying)/i,
|
||
/link naturally to/i,
|
||
/keep this factual/i,
|
||
/factual differentiators/i,
|
||
/reliability wording/i,
|
||
/^direct[- ]answer/i,
|
||
]
|
||
|
||
// Markdown that would print literally on the page, and characters that must not
|
||
// reach it. React escapes text, so pasted markup shows up as itself.
|
||
const TEXT_FAULTS = [
|
||
[/—/, 'an em dash (U+2014), which Null asked for nowhere in site copy'],
|
||
[/<2F>/, 'a U+FFFD replacement character, so the source was decoded with the wrong encoding'],
|
||
[/\*\*/, 'markdown bold markers'],
|
||
[/`/, 'a backtick'],
|
||
[/\]\(/, 'a markdown link'],
|
||
[/<\/?[a-z][^>]*>/i, 'an HTML tag'],
|
||
]
|
||
|
||
export const validateContent = ({ services, industries }) => {
|
||
const errors = []
|
||
const warnings = []
|
||
let checked = 0
|
||
|
||
const fail = (where, detail) => errors.push(`${where}: ${detail}`)
|
||
|
||
// Every string that reaches a page, wherever it sits in the tree.
|
||
const checkText = (where, value) => {
|
||
if (typeof value !== 'string') return
|
||
checked += 1
|
||
for (const [pattern, detail] of TEXT_FAULTS) {
|
||
if (pattern.test(value)) fail(where, `${detail} in ${JSON.stringify(value.slice(0, 60))}`)
|
||
}
|
||
for (const pattern of SHEET_DIRECTIONS) {
|
||
if (pattern.test(value)) {
|
||
fail(where, `reads as a direction to the website manager, not page copy: ${JSON.stringify(value.slice(0, 80))}`)
|
||
break
|
||
}
|
||
}
|
||
}
|
||
|
||
const checkLink = (where, to, ownIds) => {
|
||
checked += 1
|
||
if (typeof to !== 'string' || !to) return fail(where, 'link has no destination')
|
||
const [route, fragment] = to.split('#')
|
||
if (route && !ROUTES.includes(route)) fail(where, `links to ${route}, which is not a route this site serves`)
|
||
if (fragment && !route && !ownIds.has(fragment) && !RESERVED_IDS.has(fragment)) {
|
||
fail(where, `links to #${fragment}, which is not a section on this page`)
|
||
}
|
||
}
|
||
|
||
const checkBlocks = (where, section, ownIds) => {
|
||
const blocks = section.blocks
|
||
if (!Array.isArray(blocks) || blocks.length === 0) return fail(where, 'has no blocks')
|
||
|
||
blocks.forEach((block, index) => {
|
||
const at = `${where} blocks[${index}]`
|
||
const field = BLOCK_FIELDS[block?.type]
|
||
if (!field) {
|
||
return fail(at, `unknown block type ${JSON.stringify(block?.type)}. Known types: ${Object.keys(BLOCK_FIELDS).join(', ')}`)
|
||
}
|
||
if (block[field] == null) return fail(at, `a ${block.type} block has no ${field}`)
|
||
checked += 1
|
||
|
||
if (block.type === 'p') {
|
||
if (Array.isArray(block.text)) {
|
||
block.text.forEach((part, i) => {
|
||
if (typeof part === 'string') checkText(`${at}.text[${i}]`, part)
|
||
else {
|
||
checkText(`${at}.text[${i}].text`, part?.text)
|
||
checkLink(`${at}.text[${i}]`, part?.to, ownIds)
|
||
}
|
||
})
|
||
} else checkText(`${at}.text`, block.text)
|
||
} else if (block.type === 'ul' || block.type === 'ol') {
|
||
if (!Array.isArray(block.items) || !block.items.length) fail(at, 'a list with no items')
|
||
block.items?.forEach((item, i) => {
|
||
if (typeof item === 'string') checkText(`${at}.items[${i}]`, item)
|
||
else {
|
||
checkText(`${at}.items[${i}].text`, item?.text)
|
||
checkText(`${at}.items[${i}].detail`, item?.detail)
|
||
if (item?.text == null) fail(`${at}.items[${i}]`, 'a step with no text')
|
||
}
|
||
})
|
||
} else if (block.type === 'links') {
|
||
block.items?.forEach((item, i) => {
|
||
checkText(`${at}.items[${i}].label`, item?.label)
|
||
checkLink(`${at}.items[${i}]`, item?.to, ownIds)
|
||
})
|
||
} else if (block.type === 'image') {
|
||
checkText(`${at}.alt`, block.alt)
|
||
checkText(`${at}.caption`, block.caption)
|
||
if (!block.alt) fail(at, 'an image with no alt text')
|
||
// A slot waiting for a file the owner has to supply carries src: null and
|
||
// renders nothing. A src that IS set must point at a file that exists,
|
||
// or the page ships a broken image nobody sees until a visitor does.
|
||
if (block.src) {
|
||
if (!block.src.startsWith('/') || block.src.includes('..')) {
|
||
fail(at, `image src ${JSON.stringify(block.src)} must be a site-absolute path under public/`)
|
||
} else if (!existsSync(path.join(publicDir, block.src.replace(/^\//, '')))) {
|
||
fail(at, `image src ${JSON.stringify(block.src)} does not exist under public/`)
|
||
}
|
||
if (!block.width || !block.height) fail(at, 'an image with a src needs width and height, or the page shifts as it loads')
|
||
}
|
||
} else {
|
||
checkText(`${at}.text`, block.text)
|
||
}
|
||
})
|
||
|
||
// The opening block decides whether an answer engine can quote the section.
|
||
const first = blocks[0]?.type
|
||
if (section.kind === 'faq') {
|
||
blocks.forEach((block, index) => {
|
||
if (block.type !== 'h3') return
|
||
if (blocks[index + 1]?.type !== 'p') fail(`${where} blocks[${index}]`, 'a FAQ question with no answer paragraph after it')
|
||
})
|
||
if (first !== 'h3') fail(where, 'a FAQ section must open with its first question')
|
||
} else if (section.kind === 'list') {
|
||
if (first !== 'ul' && first !== 'ol') fail(where, "kind 'list' says the section opens with a list, and it does not")
|
||
} else if (first !== 'p') {
|
||
fail(where, `opens with a ${first} block. A section opens with its direct answer, which is what a search result or an AI answer quotes. Use kind 'list' when the approved copy genuinely has no answer paragraph.`)
|
||
}
|
||
}
|
||
|
||
const checkPage = (where, page) => {
|
||
const ownIds = new Set((page.sections || []).map((section) => section.id))
|
||
|
||
if (!page.seo?.title) fail(`${where}.page.seo`, 'no title')
|
||
if (!page.seo?.description) fail(`${where}.page.seo`, 'no description')
|
||
checkText(`${where}.page.seo.title`, page.seo?.title)
|
||
checkText(`${where}.page.seo.description`, page.seo?.description)
|
||
if (page.seo?.description && page.seo.description.length > 160) {
|
||
warnings.push(
|
||
`${where}.page.seo.description is ${page.seo.description.length} characters. Search results show about 155 to 160, so the tail may not be seen. Owner-approved copy is published as written.`,
|
||
)
|
||
}
|
||
|
||
for (const field of ['h1', 'subheading']) {
|
||
if (!page.hero?.[field]) fail(`${where}.page.hero`, `no ${field}`)
|
||
checkText(`${where}.page.hero.${field}`, page.hero?.[field])
|
||
}
|
||
if (!Array.isArray(page.hero?.intro) || !page.hero.intro.length) fail(`${where}.page.hero`, 'no intro copy')
|
||
page.hero?.intro?.forEach((text, i) => checkText(`${where}.page.hero.intro[${i}]`, text))
|
||
|
||
for (const cta of ['primaryCta', 'secondaryCta']) {
|
||
const value = page.hero?.[cta]
|
||
if (!value) {
|
||
if (cta === 'primaryCta') fail(`${where}.page.hero`, 'no primaryCta')
|
||
continue
|
||
}
|
||
checkText(`${where}.page.hero.${cta}.label`, value.label)
|
||
checkLink(`${where}.page.hero.${cta}`, value.to, ownIds)
|
||
}
|
||
|
||
if (!Array.isArray(page.sections) || !page.sections.length) return fail(`${where}.page`, 'no sections')
|
||
|
||
const seen = new Set()
|
||
page.sections.forEach((section, index) => {
|
||
const at = `${where}.page.sections[${index}]${section?.id ? ` (${section.id})` : ''}`
|
||
if (!ID_SHAPE.test(section?.id || '')) {
|
||
fail(at, `id ${JSON.stringify(section?.id)} must be lowercase, start with a letter and join words with single hyphens. An id starting with a digit cannot be used as a CSS selector`)
|
||
}
|
||
if (RESERVED_IDS.has(section?.id)) fail(at, `id ${JSON.stringify(section.id)} is already used by the page layout`)
|
||
if (seen.has(section?.id)) fail(at, `id ${JSON.stringify(section.id)} appears twice on this page`)
|
||
seen.add(section?.id)
|
||
if (!section?.title) fail(at, 'no title')
|
||
checkText(`${at}.title`, section?.title)
|
||
checkBlocks(at, section, ownIds)
|
||
})
|
||
|
||
page.related?.forEach((link, i) => {
|
||
checkText(`${where}.page.related[${i}].label`, link?.label)
|
||
checkLink(`${where}.page.related[${i}]`, link?.to, ownIds)
|
||
})
|
||
}
|
||
|
||
for (const service of services || []) {
|
||
const where = `services/${service?.id}`
|
||
for (const field of ['id', 'name', 'shortDesc', 'homeDesc', 'icon']) {
|
||
if (!service?.[field]) fail(where, `no ${field}`)
|
||
}
|
||
checkText(`${where}.shortDesc`, service.shortDesc)
|
||
checkText(`${where}.homeDesc`, service.homeDesc)
|
||
|
||
if (service.page) checkPage(where, service.page)
|
||
else {
|
||
// The layout used by every service without long-form copy maps these with
|
||
// no guard, so a missing one is a build crash with no useful message.
|
||
if (!service.fullDesc) fail(where, 'no fullDesc, and no page copy either')
|
||
for (const field of ['benefits', 'idealFor']) {
|
||
if (!Array.isArray(service[field]) || !service[field].length) fail(where, `no ${field}, which the service layout maps without checking`)
|
||
}
|
||
checkText(`${where}.fullDesc`, service.fullDesc)
|
||
}
|
||
// Every service is a card on the Services index, which shows idealFor[0].
|
||
if (!Array.isArray(service.idealFor) || !service.idealFor.length) {
|
||
fail(where, 'no idealFor. The Services index prints idealFor[0] as "Best fit"')
|
||
}
|
||
service.related?.forEach((link, i) => {
|
||
checkText(`${where}.related[${i}].label`, link?.label)
|
||
checkLink(`${where}.related[${i}]`, link?.to, new Set())
|
||
})
|
||
}
|
||
|
||
for (const industry of industries || []) {
|
||
const where = `industries/${industry?.id}`
|
||
for (const field of ['id', 'name', 'shortDesc', 'fullDesc', 'icon']) {
|
||
if (!industry?.[field]) fail(where, `no ${field}`)
|
||
}
|
||
for (const field of ['painPoints', 'solutions']) {
|
||
if (!Array.isArray(industry?.[field]) || !industry[field].length) fail(where, `no ${field}, which the industry layout maps without checking`)
|
||
}
|
||
checkText(`${where}.shortDesc`, industry?.shortDesc)
|
||
checkText(`${where}.fullDesc`, industry?.fullDesc)
|
||
industry?.related?.forEach((link, i) => {
|
||
checkText(`${where}.related[${i}].label`, link?.label)
|
||
checkLink(`${where}.related[${i}]`, link?.to, new Set())
|
||
})
|
||
}
|
||
|
||
return { errors, warnings, checked }
|
||
}
|