Files
ipal-kit/src/modules/seo/buildLlmsTxt.ts
T

121 lines
3.8 KiB
TypeScript

import type { BasePayload } from 'payload'
import type { I18nConfig } from '../i18n/index.js'
import { buildLocalizedPath } from '../i18n/index.js'
type BuildLlmsTxtArgs = {
/** Absolute site URL (https://…). */
baseUrl: string
config: I18nConfig
/** Locale for names/descriptions. Defaults to config.defaultLocale. */
locale?: string
/** Pages collection slug. Defaults to 'pages'. */
pagesSlug?: string
payload: BasePayload
/** Site settings global slug. Defaults to 'site-settings'. */
settingsSlug?: string
}
type SettingsShape = {
siteDescription?: null | string
siteName?: null | string
}
type PageRow = {
_status?: string
id: number | string
meta?: { description?: unknown; noindex?: boolean } | null
slug?: unknown
title?: unknown
}
/**
* Builds the body of /llms.txt — a Markdown file describing the site for AI
* agents / LLM crawlers, per the llmstxt.org convention. Mirrors buildRobots /
* buildSitemapEntries: the plugin already knows the site's name, description,
* and pages, so it can generate this automatically.
*
* Structure (llmstxt.org): H1 site name, a blockquote/summary, then a list of
* key pages as Markdown links with short descriptions. Agents read this to
* understand the site quickly without crawling everything.
*
* Wire it as a route that returns text/plain:
*
* // app/llms.txt/route.ts
* import { buildLlmsTxt } from '@intecion/ipal-kit'
* import { getCachedPayload } from '@/lib/content'
* import { i18nConfig } from '@/i18n.config'
* export const dynamic = 'force-dynamic'
* export async function GET() {
* const body = await buildLlmsTxt({
* payload: await getCachedPayload(),
* config: i18nConfig,
* baseUrl: process.env.NEXT_PUBLIC_SERVER_URL!,
* })
* return new Response(body, { headers: { 'Content-Type': 'text/plain; charset=utf-8' } })
* }
*
* Data comes from the panel (siteName, siteDescription, pages) — nothing
* hardcoded. Skips drafts and noindex pages (same as the sitemap).
*/
export async function buildLlmsTxt({
baseUrl,
config,
locale,
pagesSlug = 'pages',
payload,
settingsSlug = 'site-settings',
}: BuildLlmsTxtArgs): Promise<string> {
const loc = locale ?? config.defaultLocale
const origin = baseUrl.replace(/\/$/, '')
const settings = (await payload.findGlobal({
slug: settingsSlug as never,
depth: 0,
locale: loc as never,
})) as SettingsShape
const name = settings.siteName?.trim() || 'Website'
const description = settings.siteDescription?.trim()
// NO where:{_status} filter — collections without drafts enabled don't
// register the _status field, and querying it throws
// "path cannot be queried: _status" (a real bug report). Draft filtering
// happens in memory below, which is safe for every collection. Same as
// buildSitemapEntries and generateStaticParams.
const result = await payload.find({
collection: pagesSlug as never,
depth: 0,
limit: 1000,
locale: loc as never,
})
const lines: string[] = [`# ${name}`, '']
if (description) {
lines.push(`> ${description}`, '')
}
const pageLinks: string[] = []
for (const raw of result.docs as PageRow[]) {
if (raw._status && raw._status !== 'published') {continue}
if (raw.meta?.noindex) {continue}
const title = typeof raw.title === 'string' ? raw.title : undefined
const slug = typeof raw.slug === 'string' ? raw.slug : undefined
if (!title || !slug) {continue}
const path = buildLocalizedPath({ config, locale: loc, slugs: { [loc]: slug } })
if (!path) {continue}
const pageDesc = typeof raw.meta?.description === 'string' ? ` — ${raw.meta.description}` : ''
pageLinks.push(`- [${title}](${origin}${path})${pageDesc}`)
}
if (pageLinks.length > 0) {
lines.push('## Strony', '', ...pageLinks, '')
}
return lines.join('\n')
}