Return

All tests / src/helpers fileWritter.ts

0% Statements 0/36
0% Branches 0/6
0% Functions 0/13
0% Lines 0/31

Press n or j to go to the next uncovered block, b, p or k for the previous block.

1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109                                                                                                                                                                                                                         
import { defaultLocale } from '@/constants/site';
import { removeTrailingSlash } from '@/helpers';
import fs from 'fs';
import matter from 'gray-matter';
import path from 'path';
import prettier from 'prettier';
 
const PUBLIC_DIR = path.join(process.cwd(), 'public');
 
type SitemapRoute = {
    locale: string;
    date?: string | null;
    params: {
        category: string;
        slug: string;
    };
};
 
// Pages that must never reach the sitemap: framework internals, route directories
// handled by their own entries above, error pages, and utility pages with no search value.
const NON_SITEMAP_PAGES = new Set([
    '_app.tsx',
    '_document.tsx',
    'blog',
    'api',
    'legal',
    '404.tsx',
    '500.tsx',
    'survey.tsx',
    'settings.tsx',
    'comments.tsx',
]);
 
const urlEntry = (loc: string, lastmod?: string | null) => `
    <url>
        <loc>${loc}</loc>
        ${lastmod ? `<lastmod>${lastmod}</lastmod>` : ''}
    </url>
`;
 
/**
 * @description - Function to create sitemap.xml file: blog posts (with lastmod
 * from their publication date), blog hub + category hubs, legal pages and the
 * top-level pages, for every locale (SDD-002 D2).
 *
 * @example
 *     createSiteMap(routes, locales);
 *
 * @param {Array} routes - Array of blog post routes (locale, optional date, category/slug params)
 * @param {Array} locales - Array of locales
 * @returns {void}
 */
export const createSiteMap = (routes: SitemapRoute[], locales: string[]) => {
    const domain = process.env.NEXT_PUBLIC_DOMAIN;
    const localePrefix = (locale: string) => (locale !== defaultLocale ? `/${locale}` : '');
 
    // Blog posts — lastmod comes from the <Date /> tag embedded in each post
    const postEntries = routes.map(({ locale, date, params: { category, slug } }) =>
        urlEntry(`${domain}${localePrefix(locale)}/blog/${category}/${slug}`, date)
    );
 
    /**
     * Legal pages, minus the ones that tell crawlers not to index them.
     *
     * SDD-L04: this used to submit every legal slug for every locale unconditionally, and all three
     * documents carry `noindex: true` in their frontmatter. That put 9 guaranteed-uncrawlable URLs
     * into the sitemap — Search Console duly reported them as "Discovered – currently not indexed" —
     * which is the signal that erodes trust in the whole file. Submitting a URL is a request to index
     * it, so it has to agree with the page's own robots directive.
     *
     * The filter stays rather than the block being deleted, so a future indexable legal document is
     * picked up automatically.
     */
    const legalDir = path.join(process.cwd(), 'data/legal');
    const legalSlugs = fs
        .readdirSync(legalDir)
        .filter((file) => file.endsWith('.mdx'))
        .filter((file) => matter(fs.readFileSync(path.join(legalDir, file), 'utf8')).data?.noindex !== true)
        .map((file) => file.replace('.mdx', ''));
    const legalEntries = locales.flatMap((locale) =>
        legalSlugs.map((slug) => urlEntry(`${domain}${localePrefix(locale)}/legal/${slug}`))
    );
 
    // Top-level pages, read from the pages directory; utility pages stay excluded. /about and
    // /contact dropped out here on their own when the files were deleted and folded into the home.
    // No <lastmod> for evergreen pages: stamping the build date on every URL misleads crawlers
    const pages = fs
        .readdirSync(path.join(process.cwd(), '/src/pages'))
        .filter((page) => !NON_SITEMAP_PAGES.has(page))
        .map((page) => {
            page = page.replace('.tsx', '');
            return page === 'index' ? '' : page;
        });
    const pageEntries = locales.flatMap((locale) =>
        pages.map((page) => urlEntry(removeTrailingSlash(`${domain}${localePrefix(locale)}/${page}`)))
    );
 
    const xml = `<?xml version="1.0" encoding="UTF-8"?>
        <urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
        ${[...pageEntries, ...postEntries, ...legalEntries].join('')}
        </urlset>
    `;
 
    const formattedXml = prettier.format(xml, { parser: 'html', printWidth: 120 });
    const filePath = path.join(PUBLIC_DIR, 'sitemap.xml');
 
    fs.writeFileSync(filePath, formattedXml);
};