apps/web/src/lib/pageCatalogue.ts

1import { CHUNK_OVERLAP, CHUNK_SIZE } from '@local/ai/components/api/chunking'
2import type { CatalogueRecord } from '@local/ai/components/api/pineconeUpload'
3import { ALL_RESOURCE_TYPES } from '@local/config'
4import { getPublishedSanityClient } from '@local/sanity/client'
5import { sanityFetch } from '@local/sanity/utils/fetch'
6
7const LOG_PREFIX = '[PageCatalogue]'
8
9function log(msg: string, data?: Record<string, unknown>) {
10	if (process.env.SYNC_DEBUG !== '1' && process.env.SYNC_DEBUG !== 'true') return
11	const payload = data ? ` ${JSON.stringify(data)}` : ''
12	console.log(`${LOG_PREFIX} ${msg}${payload}`)
13}
14
15export type SitePage = {
16	type: string
17	slug: string
18	title: string
19	description: string | null
20}
21
22/** Structural page types on top of the resource/editorial types. */
23const PAGE_TYPES = ['page', 'event', 'taxonomy-page', 'home'] as const
24const STRUCTURAL_TYPES = new Set<string>(PAGE_TYPES)
25
26/**
27 * Visitor-facing link label for a page. Sanity `title` doubles as the SEO
28 * title ("Plan Your Visit - Travel & Hotels", "Sponsor UNLEASH Paris 2026 |
29 * Exhibition & Partnership Opportunities"), and the concierge copies link
30 * text verbatim from the store — so strip the SEO tail here, once, unless an
31 * editor set an explicit agentTitle (passed through untouched by the caller).
32 */
33export function displayTitle(title: string): string {
34	const trimmed = title.trim()
35	const head = trimmed.split(/\s+[-–—|]\s+/)[0]?.trim() ?? ''
36	return head.length >= 3 ? head : trimmed
37}
38
39/**
40 * The event an /events/<event-slug>/... page belongs to ("/events/unleash-paris"),
41 * stored on every event page vector so event-scoped tools can filter the
42 * shared sitemap store to their own event. Null for non-event pages.
43 */
44export function eventSlugOf(slug: string): string | null {
45	const match = slug.match(/^\/events\/[^/]+/)
46	return match ? match[0] : null
47}
48
49/** Flat metadata for one page vector; `eventSlug` only on event pages. */
50export function pageVectorMetadata(
51	page: Pick<SitePage, 'type' | 'slug' | 'title'>,
52): CatalogueRecord['metadata'] {
53	const { type, slug, title } = page
54	const eventSlug = eventSlugOf(slug)
55	if (eventSlug) return { type, slug, title, eventSlug }
56	return { type, slug, title }
57}
58
59/**
60 * Build a short searchable text for one page (what it is + where it lives).
61 * Event pages add their outline — the section headings and labels — so a
62 * "where do I find X" query matches on what the page actually covers, not
63 * only on its title and SEO description.
64 */
65export function formatSitePageText(
66	page: SitePage & { outline?: string[] },
67): string {
68	const lines = [
69		`Title: ${page.title}`,
70		`Type: ${page.type}`,
71		`URL: ${page.slug}`,
72		page.description ? `Description: ${page.description}` : '',
73		page.outline?.length ? `Covers: ${page.outline.join(' · ')}` : '',
74	].filter(Boolean)
75	return lines.join('\n')
76}
77
78/** Stable ID for Pinecone (one vector per page). */
79export function sitePageId(page: SitePage): string {
80	const safe = page.slug.replace(/^\/+/, '').replace(/\//g, '-') || 'home'
81	return `page::${page.type}::${safe}`
82}
83
84/**
85 * Fetch every linkable page (structural pages, events, taxonomy pages, and all
86 * resource types) with title + slug + short description. Same document set the
87 * XML sitemaps are built from, so the agent can only suggest real URLs.
88 */
89export async function fetchSitePagesFromSanity(): Promise<SitePage[]> {
90	log('Fetching site pages from Sanity')
91
92	const types = [...new Set([...PAGE_TYPES, ...ALL_RESOURCE_TYPES])]
93	const wrappedTypes = types.map((type) => `"${type}"`).join(', ')
94
95	// agentTitle/agentDescription are editor-facing overrides ("AI assistant"
96	// group in Studio) so the agent links pages by their visitor-facing name
97	// instead of the SEO title. Resource types don't have them — coalesce
98	// falls through cleanly.
99	const query = `*[_type in [${wrappedTypes}] && defined(fullSlug)]{
100    _type,
101    "slug": fullSlug,
102    agentTitle,
103    title,
104    "description": coalesce(agentDescription, seo.metaDescription, seo.description, excerpt)
105  }`
106
107	const docs = await sanityFetch<
108		Array<{
109			_type: string
110			slug?: string
111			agentTitle?: string
112			title?: string
113			description?: string
114		}>
115	>(getPublishedSanityClient(), query, undefined, { tag: 'page-catalogue' })
116
117	const pages: SitePage[] = []
118	for (const doc of docs) {
119		const slug = typeof doc.slug === 'string' ? doc.slug.trim() : ''
120		if (!slug) continue
121		pages.push({
122			type: doc._type,
123			slug,
124			title: pageTitle(doc, slug),
125			description: doc.description?.trim() || null,
126		})
127	}
128
129	log('Site pages fetch done', { docs: docs.length, pages: pages.length })
130	return pages
131}
132
133/**
134 * Editor override wins verbatim. Structural pages (page / event / taxonomy /
135 * home) lose their SEO tail; articles, podcasts and other resources keep
136 * their full title — a dash there is part of the headline, not SEO padding.
137 */
138export function pageTitle(
139	doc: { _type?: string; agentTitle?: string; title?: string },
140	fallback: string,
141): string {
142	const agentTitle = doc.agentTitle?.trim()
143	if (agentTitle) return agentTitle
144	const title = doc.title?.trim()
145	if (!title) return fallback
146	return doc._type && STRUCTURAL_TYPES.has(doc._type)
147		? displayTitle(title)
148		: title
149}
150
151/**
152 * Slice fields whose string values are visitor-facing copy. Everything else
153 * (refs, style/layout knobs, asset ids) is noise for retrieval. Button
154 * destinations are preserved separately alongside their visitor-facing label.
155 */
156const CONTENT_KEYS = new Set([
157	'text',
158	'title',
159	'subtitle',
160	'eyebrow',
161	'quote',
162	'question',
163	'answer',
164	'description',
165	'caption',
166	'label',
167	'subtext',
168	// Table cells (pricing / comparison tables) — an array of strings.
169	'cells',
170])
171
172/** Table cells carry inline markup (">##Core €7,600 <br/>saving €2,885"). */
173function cleanCell(cell: string): string {
174	return cell
175		.replace(/<br\s*\/?>/gi, ' ')
176		.replace(/^[>#\s]+/, '')
177		.replace(/\s+/g, ' ')
178		.trim()
179}
180
181/**
182 * Harvest visitor-facing copy from a page's slice tree in document order.
183 * Generic walk instead of per-slice handling: slices keep their prose in a
184 * small stable set of field names (portable-text spans included via `text`).
185 */
186export function extractSlicesText(slices: unknown): string {
187	const parts: string[] = []
188	const walk = (node: unknown): void => {
189		if (Array.isArray(node)) {
190			for (const item of node) walk(item)
191			return
192		}
193		if (!node || typeof node !== 'object') return
194		// CMS buttons carry their booking destination in link.url. Keep the
195		// label and exact URL together; a provider name alone invites guessing.
196		if (
197			'_type' in node &&
198			node._type === 'button' &&
199			'label' in node &&
200			typeof node.label === 'string'
201		) {
202			const link = 'link' in node ? node.link : node
203			if (
204				link &&
205				typeof link === 'object' &&
206				'url' in link &&
207				typeof link.url === 'string' &&
208				(!('linkType' in link) || link.linkType === 'external') &&
209				/^https?:\/\/[^\s<>]+$/i.test(link.url) &&
210				node.label.trim()
211			) {
212				parts.push(`[${node.label.trim()}](<${link.url}>)`)
213				return
214			}
215		}
216		for (const [key, value] of Object.entries(node)) {
217			if (key.startsWith('_')) continue
218			if (typeof value === 'string') {
219				if (CONTENT_KEYS.has(key)) {
220					const text = value.trim()
221					if (text) parts.push(text)
222				}
223			} else if (key === 'cells' && Array.isArray(value)) {
224				// A table row: keep its cells on one line so a price stays next
225				// to the package it belongs to.
226				const cells = value
227					.filter((cell): cell is string => typeof cell === 'string')
228					.map(cleanCell)
229					.filter(Boolean)
230				if (cells.length > 0) parts.push(cells.join(' | '))
231			} else {
232				walk(value)
233			}
234		}
235	}
236	walk(slices)
237	return parts.join('\n')
238}
239
240/** Logo grids render referenced companies/people; their public names are copy too. */
241export function sponsorReferenceIds(slices: unknown): string[] {
242	const ids = new Set<string>()
243	const walk = (node: unknown, sponsor = false): void => {
244		if (Array.isArray(node)) {
245			for (const item of node) walk(item, sponsor)
246			return
247		}
248		if (!node || typeof node !== 'object') return
249		const visible = sponsor || ('_type' in node && node._type === 'sponsors')
250		if (visible && '_ref' in node && typeof node._ref === 'string') {
251			ids.add(node._ref)
252			return
253		}
254		for (const value of Object.values(node)) walk(value, visible)
255	}
256	walk(slices)
257	return [...ids]
258}
259
260/** Unresolved/draft-only references stay nameless; asset IDs never become content. */
261export function enrichSponsorReferences(
262	slices: unknown,
263	titles: ReadonlyMap<string, string>,
264): unknown {
265	const walk = (node: unknown, sponsor = false): unknown => {
266		if (Array.isArray(node)) return node.map((item) => walk(item, sponsor))
267		if (!node || typeof node !== 'object') return node
268		const visible = sponsor || ('_type' in node && node._type === 'sponsors')
269		if (visible && '_ref' in node && typeof node._ref === 'string') {
270			const title = titles.get(node._ref)
271			return title ? { ...node, title } : node
272		}
273		return Object.fromEntries(
274			Object.entries(node).map(([key, value]) => [key, walk(value, visible)]),
275		)
276	}
277	return walk(slices)
278}
279
280/** Shared by incremental publish sync and full refresh; one published, bounded read. */
281export async function loadSponsorReferenceTitles(
282	slices: unknown,
283): Promise<ReadonlyMap<string, string>> {
284	const ids = sponsorReferenceIds(slices)
285	if (!ids.length) return new Map()
286	const rows = await sanityFetch<Array<{ _id: string; title?: string }>>(
287		getPublishedSanityClient(),
288		'*[_id in $ids && _type in ["company", "profile", "sanity.imageAsset"]]{_id, "title": select(_type == "sanity.imageAsset" => altText, title)}',
289		{ ids },
290		{ tag: 'page-catalogue' },
291	)
292	return new Map(
293		rows.flatMap((row) =>
294			row.title?.trim() ? [[row._id, row.title.trim()]] : [],
295		),
296	)
297}
298
299/** Heading-level fields: what a page is about, without its body prose. */
300const OUTLINE_KEYS = new Set(['title', 'eyebrow', 'subtitle', 'question'])
301const OUTLINE_MAX_ITEMS = 14
302const OUTLINE_MAX_CHARS = 90
303/** Editor-facing slice labels ("Header", "Intro", "… Section") are not content. */
304const OUTLINE_NOISE = /^(?:header|hero|intro|footer|cta|section|content|body)$/i
305const OUTLINE_LABEL_TAIL = /\s+section$/i
306
307/**
308 * A page's outline: its section headings, eyebrows and FAQ questions in
309 * document order, deduplicated and bounded. Feeds the page's index record
310 * and the page directory tool so retrieval can pick the precise page.
311 */
312export function extractPageOutline(slices: unknown): string[] {
313	const seen = new Set<string>()
314	const outline: string[] = []
315	const walk = (node: unknown): void => {
316		if (outline.length >= OUTLINE_MAX_ITEMS) return
317		if (Array.isArray(node)) {
318			for (const item of node) walk(item)
319			return
320		}
321		if (!node || typeof node !== 'object') return
322		for (const [key, value] of Object.entries(node)) {
323			if (key.startsWith('_')) continue
324			if (typeof value === 'string') {
325				if (!OUTLINE_KEYS.has(key)) continue
326				const text = value
327					.replace(/\s+/g, ' ')
328					.trim()
329					.replace(OUTLINE_LABEL_TAIL, '')
330				if (text.length < 3 || text.length > OUTLINE_MAX_CHARS) continue
331				if (OUTLINE_NOISE.test(text)) continue
332				const fold = text.toLowerCase()
333				if (seen.has(fold)) continue
334				seen.add(fold)
335				outline.push(text)
336			} else {
337				walk(value)
338			}
339		}
340	}
341	walk(slices)
342	return outline
343}
344
345/** Event subpages get full-content chunks in the sitemap store; the media and
346 * generic structural pages stay meta-only. */
347export function isEventSitePage(
348	page: Pick<SitePage, 'type' | 'slug'>,
349): boolean {
350	return (
351		(page.type === 'page' || page.type === 'event') &&
352		page.slug.startsWith('/events/')
353	)
354}
355
356export type EventPageContent = SitePage & { content: string; outline: string[] }
357
358/** Event subpages (and the two event roots) with their extracted body copy. */
359export async function fetchEventPagesWithContent(): Promise<
360	EventPageContent[]
361> {
362	log('Fetching event page content from Sanity')
363
364	const query = `*[_type in ["page", "event"] && defined(fullSlug) && string::startsWith(fullSlug, "/events/")]{
365    _type,
366    "slug": fullSlug,
367    agentTitle,
368    title,
369    "description": coalesce(agentDescription, seo.metaDescription, seo.description, excerpt),
370    slices
371  }`
372
373	const docs = await sanityFetch<
374		Array<{
375			_type: string
376			slug?: string
377			agentTitle?: string
378			title?: string
379			description?: string
380			slices?: unknown
381		}>
382	>(getPublishedSanityClient(), query, undefined, { tag: 'page-catalogue' })
383
384	const sponsorTitles = await loadSponsorReferenceTitles(
385		docs.map((doc) => doc.slices),
386	)
387	const pages: EventPageContent[] = []
388	for (const doc of docs) {
389		const slug = typeof doc.slug === 'string' ? doc.slug.trim() : ''
390		if (!slug) continue
391		pages.push({
392			type: doc._type,
393			slug,
394			title: pageTitle(doc, slug),
395			description: doc.description?.trim() || null,
396			content: extractSlicesText(
397				enrichSponsorReferences(doc.slices, sponsorTitles),
398			),
399			outline: extractPageOutline(doc.slices),
400		})
401	}
402
403	log('Event page content fetch done', { pages: pages.length })
404	return pages
405}
406
407/** The event page's own index record: title, URL, description and outline. */
408export function eventPageMetaRecord(page: EventPageContent): CatalogueRecord {
409	return {
410		id: sitePageId(page),
411		text: formatSitePageText(page),
412		metadata: pageVectorMetadata(page),
413	}
414}
415
416/**
417 * One vector per content chunk, each carrying the page title + URL so any
418 * retrieved chunk lets the agent cite the exact page. Ids extend sitePageId
419 * as `<pageId>::c<n>`: still under the global `page::` prefix (full-refresh
420 * stale cleanup covers them), while the `<pageId>::` prefix scopes per-page
421 * cleanup — the trailing `::` matters because dashed slugs make one page's id
422 * a plain prefix of its deeper siblings' ids.
423 */
424export function eventPageContentRecords(
425	page: SitePage & { content: string },
426): CatalogueRecord[] {
427	// Never split a CTA's label from its URL, even across the window overlap.
428	const content = page.content.replace(/\r\n/g, '\n').trim()
429	const links = [...content.matchAll(/^\[[^\n]*\]\(<https?:\/\/[^\n]*>\)$/gm)]
430	const chunks: string[] = []
431	let start = 0
432	while (start < content.length) {
433		let end = Math.min(start + CHUNK_SIZE, content.length)
434		for (const link of links) {
435			if (link.index < end && link.index + link[0].length > end) {
436				end = link.index + link[0].length
437			}
438		}
439		const chunk = content.slice(start, end).trim()
440		if (chunk) chunks.push(chunk)
441		if (end === content.length) break
442		let next = end - CHUNK_OVERLAP
443		for (const link of links) {
444			if (link.index < next && link.index + link[0].length > next) {
445				next = link.index
446			}
447		}
448		start = next > start ? next : end
449	}
450	const pageId = sitePageId(page)
451	return chunks.map((chunk, index) => ({
452		id: `${pageId}::c${index}`,
453		text: [`Title: ${page.title}`, `URL: ${page.slug}`, 'Content:', chunk].join(
454			'\n',
455		),
456		metadata: { ...pageVectorMetadata(page), chunk: String(index) },
457	}))
458}
459