apps/web/src/lib/pageCatalogue.ts
1import { CHUNK_OVERLAP, CHUNK_SIZE } from '@local/ai/components/api/chunking'
2import type { CatalogueRecord } from '@local/ai/components/api/pineconeUpload'
3import { ALL_RESOURCE_TYPES } from '@local/config'
4import { getPublishedSanityClient } from '@local/sanity/client'
5import { sanityFetch } from '@local/sanity/utils/fetch'
6
7const LOG_PREFIX = '[PageCatalogue]'
8
9function log(msg: string, data?: Record<string, unknown>) {
10 if (process.env.SYNC_DEBUG !== '1' && process.env.SYNC_DEBUG !== 'true') return
11 const payload = data ? ` ${JSON.stringify(data)}` : ''
12 console.log(`${LOG_PREFIX} ${msg}${payload}`)
13}
14
15export type SitePage = {
16 type: string
17 slug: string
18 title: string
19 description: string | null
20}
21
22/** Structural page types on top of the resource/editorial types. */
23const PAGE_TYPES = ['page', 'event', 'taxonomy-page', 'home'] as const
24const STRUCTURAL_TYPES = new Set<string>(PAGE_TYPES)
25
26/**
27 * Visitor-facing link label for a page. Sanity `title` doubles as the SEO
28 * title ("Plan Your Visit - Travel & Hotels", "Sponsor UNLEASH Paris 2026 |
29 * Exhibition & Partnership Opportunities"), and the concierge copies link
30 * text verbatim from the store — so strip the SEO tail here, once, unless an
31 * editor set an explicit agentTitle (passed through untouched by the caller).
32 */
33export function displayTitle(title: string): string {
34 const trimmed = title.trim()
35 const head = trimmed.split(/\s+[-–—|]\s+/)[0]?.trim() ?? ''
36 return head.length >= 3 ? head : trimmed
37}
38
39/**
40 * The event an /events/<event-slug>/... page belongs to ("/events/unleash-paris"),
41 * stored on every event page vector so event-scoped tools can filter the
42 * shared sitemap store to their own event. Null for non-event pages.
43 */
44export function eventSlugOf(slug: string): string | null {
45 const match = slug.match(/^\/events\/[^/]+/)
46 return match ? match[0] : null
47}
48
49/** Flat metadata for one page vector; `eventSlug` only on event pages. */
50export function pageVectorMetadata(
51 page: Pick<SitePage, 'type' | 'slug' | 'title'>,
52): CatalogueRecord['metadata'] {
53 const { type, slug, title } = page
54 const eventSlug = eventSlugOf(slug)
55 if (eventSlug) return { type, slug, title, eventSlug }
56 return { type, slug, title }
57}
58
59/**
60 * Build a short searchable text for one page (what it is + where it lives).
61 * Event pages add their outline — the section headings and labels — so a
62 * "where do I find X" query matches on what the page actually covers, not
63 * only on its title and SEO description.
64 */
65export function formatSitePageText(
66 page: SitePage & { outline?: string[] },
67): string {
68 const lines = [
69 `Title: ${page.title}`,
70 `Type: ${page.type}`,
71 `URL: ${page.slug}`,
72 page.description ? `Description: ${page.description}` : '',
73 page.outline?.length ? `Covers: ${page.outline.join(' · ')}` : '',
74 ].filter(Boolean)
75 return lines.join('\n')
76}
77
78/** Stable ID for Pinecone (one vector per page). */
79export function sitePageId(page: SitePage): string {
80 const safe = page.slug.replace(/^\/+/, '').replace(/\//g, '-') || 'home'
81 return `page::${page.type}::${safe}`
82}
83
84/**
85 * Fetch every linkable page (structural pages, events, taxonomy pages, and all
86 * resource types) with title + slug + short description. Same document set the
87 * XML sitemaps are built from, so the agent can only suggest real URLs.
88 */
89export async function fetchSitePagesFromSanity(): Promise<SitePage[]> {
90 log('Fetching site pages from Sanity')
91
92 const types = [...new Set([...PAGE_TYPES, ...ALL_RESOURCE_TYPES])]
93 const wrappedTypes = types.map((type) => `"${type}"`).join(', ')
94
95 // agentTitle/agentDescription are editor-facing overrides ("AI assistant"
96 // group in Studio) so the agent links pages by their visitor-facing name
97 // instead of the SEO title. Resource types don't have them — coalesce
98 // falls through cleanly.
99 const query = `*[_type in [${wrappedTypes}] && defined(fullSlug)]{
100 _type,
101 "slug": fullSlug,
102 agentTitle,
103 title,
104 "description": coalesce(agentDescription, seo.metaDescription, seo.description, excerpt)
105 }`
106
107 const docs = await sanityFetch<
108 Array<{
109 _type: string
110 slug?: string
111 agentTitle?: string
112 title?: string
113 description?: string
114 }>
115 >(getPublishedSanityClient(), query, undefined, { tag: 'page-catalogue' })
116
117 const pages: SitePage[] = []
118 for (const doc of docs) {
119 const slug = typeof doc.slug === 'string' ? doc.slug.trim() : ''
120 if (!slug) continue
121 pages.push({
122 type: doc._type,
123 slug,
124 title: pageTitle(doc, slug),
125 description: doc.description?.trim() || null,
126 })
127 }
128
129 log('Site pages fetch done', { docs: docs.length, pages: pages.length })
130 return pages
131}
132
133/**
134 * Editor override wins verbatim. Structural pages (page / event / taxonomy /
135 * home) lose their SEO tail; articles, podcasts and other resources keep
136 * their full title — a dash there is part of the headline, not SEO padding.
137 */
138export function pageTitle(
139 doc: { _type?: string; agentTitle?: string; title?: string },
140 fallback: string,
141): string {
142 const agentTitle = doc.agentTitle?.trim()
143 if (agentTitle) return agentTitle
144 const title = doc.title?.trim()
145 if (!title) return fallback
146 return doc._type && STRUCTURAL_TYPES.has(doc._type)
147 ? displayTitle(title)
148 : title
149}
150
151/**
152 * Slice fields whose string values are visitor-facing copy. Everything else
153 * (refs, style/layout knobs, asset ids) is noise for retrieval. Button
154 * destinations are preserved separately alongside their visitor-facing label.
155 */
156const CONTENT_KEYS = new Set([
157 'text',
158 'title',
159 'subtitle',
160 'eyebrow',
161 'quote',
162 'question',
163 'answer',
164 'description',
165 'caption',
166 'label',
167 'subtext',
168 // Table cells (pricing / comparison tables) — an array of strings.
169 'cells',
170])
171
172/** Table cells carry inline markup (">##Core €7,600 <br/>saving €2,885"). */
173function cleanCell(cell: string): string {
174 return cell
175 .replace(/<br\s*\/?>/gi, ' ')
176 .replace(/^[>#\s]+/, '')
177 .replace(/\s+/g, ' ')
178 .trim()
179}
180
181/**
182 * Harvest visitor-facing copy from a page's slice tree in document order.
183 * Generic walk instead of per-slice handling: slices keep their prose in a
184 * small stable set of field names (portable-text spans included via `text`).
185 */
186export function extractSlicesText(slices: unknown): string {
187 const parts: string[] = []
188 const walk = (node: unknown): void => {
189 if (Array.isArray(node)) {
190 for (const item of node) walk(item)
191 return
192 }
193 if (!node || typeof node !== 'object') return
194 // CMS buttons carry their booking destination in link.url. Keep the
195 // label and exact URL together; a provider name alone invites guessing.
196 if (
197 '_type' in node &&
198 node._type === 'button' &&
199 'label' in node &&
200 typeof node.label === 'string'
201 ) {
202 const link = 'link' in node ? node.link : node
203 if (
204 link &&
205 typeof link === 'object' &&
206 'url' in link &&
207 typeof link.url === 'string' &&
208 (!('linkType' in link) || link.linkType === 'external') &&
209 /^https?:\/\/[^\s<>]+$/i.test(link.url) &&
210 node.label.trim()
211 ) {
212 parts.push(`[${node.label.trim()}](<${link.url}>)`)
213 return
214 }
215 }
216 for (const [key, value] of Object.entries(node)) {
217 if (key.startsWith('_')) continue
218 if (typeof value === 'string') {
219 if (CONTENT_KEYS.has(key)) {
220 const text = value.trim()
221 if (text) parts.push(text)
222 }
223 } else if (key === 'cells' && Array.isArray(value)) {
224 // A table row: keep its cells on one line so a price stays next
225 // to the package it belongs to.
226 const cells = value
227 .filter((cell): cell is string => typeof cell === 'string')
228 .map(cleanCell)
229 .filter(Boolean)
230 if (cells.length > 0) parts.push(cells.join(' | '))
231 } else {
232 walk(value)
233 }
234 }
235 }
236 walk(slices)
237 return parts.join('\n')
238}
239
240/** Logo grids render referenced companies/people; their public names are copy too. */
241export function sponsorReferenceIds(slices: unknown): string[] {
242 const ids = new Set<string>()
243 const walk = (node: unknown, sponsor = false): void => {
244 if (Array.isArray(node)) {
245 for (const item of node) walk(item, sponsor)
246 return
247 }
248 if (!node || typeof node !== 'object') return
249 const visible = sponsor || ('_type' in node && node._type === 'sponsors')
250 if (visible && '_ref' in node && typeof node._ref === 'string') {
251 ids.add(node._ref)
252 return
253 }
254 for (const value of Object.values(node)) walk(value, visible)
255 }
256 walk(slices)
257 return [...ids]
258}
259
260/** Unresolved/draft-only references stay nameless; asset IDs never become content. */
261export function enrichSponsorReferences(
262 slices: unknown,
263 titles: ReadonlyMap<string, string>,
264): unknown {
265 const walk = (node: unknown, sponsor = false): unknown => {
266 if (Array.isArray(node)) return node.map((item) => walk(item, sponsor))
267 if (!node || typeof node !== 'object') return node
268 const visible = sponsor || ('_type' in node && node._type === 'sponsors')
269 if (visible && '_ref' in node && typeof node._ref === 'string') {
270 const title = titles.get(node._ref)
271 return title ? { ...node, title } : node
272 }
273 return Object.fromEntries(
274 Object.entries(node).map(([key, value]) => [key, walk(value, visible)]),
275 )
276 }
277 return walk(slices)
278}
279
280/** Shared by incremental publish sync and full refresh; one published, bounded read. */
281export async function loadSponsorReferenceTitles(
282 slices: unknown,
283): Promise<ReadonlyMap<string, string>> {
284 const ids = sponsorReferenceIds(slices)
285 if (!ids.length) return new Map()
286 const rows = await sanityFetch<Array<{ _id: string; title?: string }>>(
287 getPublishedSanityClient(),
288 '*[_id in $ids && _type in ["company", "profile", "sanity.imageAsset"]]{_id, "title": select(_type == "sanity.imageAsset" => altText, title)}',
289 { ids },
290 { tag: 'page-catalogue' },
291 )
292 return new Map(
293 rows.flatMap((row) =>
294 row.title?.trim() ? [[row._id, row.title.trim()]] : [],
295 ),
296 )
297}
298
299/** Heading-level fields: what a page is about, without its body prose. */
300const OUTLINE_KEYS = new Set(['title', 'eyebrow', 'subtitle', 'question'])
301const OUTLINE_MAX_ITEMS = 14
302const OUTLINE_MAX_CHARS = 90
303/** Editor-facing slice labels ("Header", "Intro", "… Section") are not content. */
304const OUTLINE_NOISE = /^(?:header|hero|intro|footer|cta|section|content|body)$/i
305const OUTLINE_LABEL_TAIL = /\s+section$/i
306
307/**
308 * A page's outline: its section headings, eyebrows and FAQ questions in
309 * document order, deduplicated and bounded. Feeds the page's index record
310 * and the page directory tool so retrieval can pick the precise page.
311 */
312export function extractPageOutline(slices: unknown): string[] {
313 const seen = new Set<string>()
314 const outline: string[] = []
315 const walk = (node: unknown): void => {
316 if (outline.length >= OUTLINE_MAX_ITEMS) return
317 if (Array.isArray(node)) {
318 for (const item of node) walk(item)
319 return
320 }
321 if (!node || typeof node !== 'object') return
322 for (const [key, value] of Object.entries(node)) {
323 if (key.startsWith('_')) continue
324 if (typeof value === 'string') {
325 if (!OUTLINE_KEYS.has(key)) continue
326 const text = value
327 .replace(/\s+/g, ' ')
328 .trim()
329 .replace(OUTLINE_LABEL_TAIL, '')
330 if (text.length < 3 || text.length > OUTLINE_MAX_CHARS) continue
331 if (OUTLINE_NOISE.test(text)) continue
332 const fold = text.toLowerCase()
333 if (seen.has(fold)) continue
334 seen.add(fold)
335 outline.push(text)
336 } else {
337 walk(value)
338 }
339 }
340 }
341 walk(slices)
342 return outline
343}
344
345/** Event subpages get full-content chunks in the sitemap store; the media and
346 * generic structural pages stay meta-only. */
347export function isEventSitePage(
348 page: Pick<SitePage, 'type' | 'slug'>,
349): boolean {
350 return (
351 (page.type === 'page' || page.type === 'event') &&
352 page.slug.startsWith('/events/')
353 )
354}
355
356export type EventPageContent = SitePage & { content: string; outline: string[] }
357
358/** Event subpages (and the two event roots) with their extracted body copy. */
359export async function fetchEventPagesWithContent(): Promise<
360 EventPageContent[]
361> {
362 log('Fetching event page content from Sanity')
363
364 const query = `*[_type in ["page", "event"] && defined(fullSlug) && string::startsWith(fullSlug, "/events/")]{
365 _type,
366 "slug": fullSlug,
367 agentTitle,
368 title,
369 "description": coalesce(agentDescription, seo.metaDescription, seo.description, excerpt),
370 slices
371 }`
372
373 const docs = await sanityFetch<
374 Array<{
375 _type: string
376 slug?: string
377 agentTitle?: string
378 title?: string
379 description?: string
380 slices?: unknown
381 }>
382 >(getPublishedSanityClient(), query, undefined, { tag: 'page-catalogue' })
383
384 const sponsorTitles = await loadSponsorReferenceTitles(
385 docs.map((doc) => doc.slices),
386 )
387 const pages: EventPageContent[] = []
388 for (const doc of docs) {
389 const slug = typeof doc.slug === 'string' ? doc.slug.trim() : ''
390 if (!slug) continue
391 pages.push({
392 type: doc._type,
393 slug,
394 title: pageTitle(doc, slug),
395 description: doc.description?.trim() || null,
396 content: extractSlicesText(
397 enrichSponsorReferences(doc.slices, sponsorTitles),
398 ),
399 outline: extractPageOutline(doc.slices),
400 })
401 }
402
403 log('Event page content fetch done', { pages: pages.length })
404 return pages
405}
406
407/** The event page's own index record: title, URL, description and outline. */
408export function eventPageMetaRecord(page: EventPageContent): CatalogueRecord {
409 return {
410 id: sitePageId(page),
411 text: formatSitePageText(page),
412 metadata: pageVectorMetadata(page),
413 }
414}
415
416/**
417 * One vector per content chunk, each carrying the page title + URL so any
418 * retrieved chunk lets the agent cite the exact page. Ids extend sitePageId
419 * as `<pageId>::c<n>`: still under the global `page::` prefix (full-refresh
420 * stale cleanup covers them), while the `<pageId>::` prefix scopes per-page
421 * cleanup — the trailing `::` matters because dashed slugs make one page's id
422 * a plain prefix of its deeper siblings' ids.
423 */
424export function eventPageContentRecords(
425 page: SitePage & { content: string },
426): CatalogueRecord[] {
427 // Never split a CTA's label from its URL, even across the window overlap.
428 const content = page.content.replace(/\r\n/g, '\n').trim()
429 const links = [...content.matchAll(/^\[[^\n]*\]\(<https?:\/\/[^\n]*>\)$/gm)]
430 const chunks: string[] = []
431 let start = 0
432 while (start < content.length) {
433 let end = Math.min(start + CHUNK_SIZE, content.length)
434 for (const link of links) {
435 if (link.index < end && link.index + link[0].length > end) {
436 end = link.index + link[0].length
437 }
438 }
439 const chunk = content.slice(start, end).trim()
440 if (chunk) chunks.push(chunk)
441 if (end === content.length) break
442 let next = end - CHUNK_OVERLAP
443 for (const link of links) {
444 if (link.index < next && link.index + link[0].length > next) {
445 next = link.index
446 }
447 }
448 start = next > start ? next : end
449 }
450 const pageId = sitePageId(page)
451 return chunks.map((chunk, index) => ({
452 id: `${pageId}::c${index}`,
453 text: [`Title: ${page.title}`, `URL: ${page.slug}`, 'Content:', chunk].join(
454 '\n',
455 ),
456 metadata: { ...pageVectorMetadata(page), chunk: String(index) },
457 }))
458}
459