scripts/ai-acceptance/run.ts

1/**
2 * UAT runner for the unleash.ai AI agent.
3 *
4 * For each conversational case in the selected suite (cases.json by default,
5 * <name>.json with --suite <name>, e.g. --suite knowledge):
6 *  - a visitor-simulator model plays the user following the case's turn script
7 *  - each turn hits POST <base>/api/ai and streams the reply (first-token latency recorded)
8 *  - an LLM judge scores the full transcript against Expected behaviour + Pass criteria
9 *  - the whole conversation + verdict is written to runs/<timestamp>/<id>.md (+ results.json)
10 *  - a summary.md with pass rates by criticality/capability and latency stats is generated
11 *
12 * Usage:
13 *   bun --env-file=../../apps/web/.env run.ts [--only C01,C35] [--criticality Critical]
14 *     [--capability Guardrails] [--max 10] [--base-url http://localhost:3001]
15 *     [--concurrency 3] [--turns 5] [--list] [--capture-port 8788] [--no-capture]
16 *     [--suite knowledge]
17 *
18 * Requires AI_GATEWAY (or AI_GATEWAY_API_KEY) for the simulator + judge models.
19 *
20 * Webhook capture (Zapier hand-off ground truth): when the target is localhost
21 * (or --capture-port is passed), the runner listens for the app's outbound
22 * Zapier POSTs and matches them to cases by chat_id, so "lead created" pass
23 * criteria are judged on actual submissions rather than the agent's claims.
24 * The dev server under test must point its hook at the runner:
25 *   ZAPIER_LEAD_WEBHOOK_URL=http://localhost:8788/zapier pnpm dev:web
26 */
27import { mkdirSync, readFileSync, writeFileSync } from 'node:fs'
28import { join } from 'node:path'
29import {
30	parseAiEvent,
31	sessionConsent,
32} from '../../packages/ai/components/workflows'
33
34type AcceptanceCase = {
35	id: string
36	source: string
37	capability: string
38	context: string
39	input: string
40	expected: string
41	passCriteria: string
42	criticality: string
43	sowRef: string
44	notes: string
45	messages?: string[]
46	initialProfile?: Record<string, unknown>
47	initialConsent?: boolean
48	initialHistory?: { role: 'user' | 'assistant'; text: string }[]
49}
50
51type Turn = {
52	user: string
53	assistant: string
54	firstTokenMs: number
55	totalMs: number
56	chunks?: number
57	profileMs?: number
58	visitorFirstTokenMs?: number
59}
60
61type Verdict = {
62	verdict: 'pass' | 'partial' | 'fail' | 'blocked' | 'error'
63	reasons: string
64	criticalViolations: string[]
65	missingCapabilities: string[]
66}
67
68type WebhookCapture = {
69	responseStatus: number
70	type: string
71	payload: Record<string, unknown>
72}
73
74type CaseResult = {
75	case: AcceptanceCase
76	turns: Turn[]
77	verdict: Verdict
78	chatId: string
79	webhooks: WebhookCapture[]
80}
81
82// ---------------------------------------------------------------- config
83
84function arg(name: string): string | undefined {
85	const index = process.argv.indexOf(`--${name}`)
86	return index >= 0 ? process.argv[index + 1] : undefined
87}
88const hasFlag = (name: string) => process.argv.includes(`--${name}`)
89
90// Which case file to run. Suites other than the default keep their own run
91// folder and their own chat-id / Langfuse score-id prefix so results never
92// collide with the main acceptance set.
93const SUITE = arg('suite') ?? 'cases'
94if (!/^[a-z0-9-]+$/.test(SUITE)) {
95	console.error(
96		`Invalid --suite "${SUITE}" (use a case file name without .json)`,
97	)
98	process.exit(1)
99}
100const ID_PREFIX = SUITE === 'cases' ? 'uat' : `uat-${SUITE}`
101
102const BASE_URL = arg('base-url') ?? 'http://localhost:3001'
103const CONCURRENCY = Number(arg('concurrency') ?? 2)
104const MAX_TURNS = Number(arg('turns') ?? 5)
105const SIMULATOR_MODEL =
106	process.env.UAT_SIMULATOR_MODEL || 'anthropic/claude-haiku-4.5'
107const JUDGE_MODEL = process.env.UAT_JUDGE_MODEL || 'anthropic/claude-sonnet-4.6'
108const GATEWAY_KEY = process.env.AI_GATEWAY_API_KEY || process.env.AI_GATEWAY
109if (!GATEWAY_KEY) {
110	console.error(
111		'Missing AI_GATEWAY / AI_GATEWAY_API_KEY env (simulator + judge)',
112	)
113	process.exit(1)
114}
115
116// Zapier webhook capture only works when the app under test can reach this
117// process, so it defaults on for localhost targets and off for deployments.
118const CAPTURE_PORT = Number(arg('capture-port') ?? 8788)
119const CAPTURE_ENABLED =
120	!hasFlag('no-capture') &&
121	(arg('capture-port') !== undefined || /localhost|127\.0\.0\.1/.test(BASE_URL))
122
123// Local simulated visitors need separate buckets, just like distinct real IPs.
124// Never spoof this header against a deployed environment.
125function visitorHeaders(chatId: string): Record<string, string> {
126	const headers: Record<string, string> = { 'Content-Type': 'application/json' }
127	const hostname = new URL(BASE_URL).hostname
128	if (hostname === 'localhost' || hostname === '127.0.0.1') {
129		headers['x-forwarded-for'] = `uat-${chatId}`
130	}
131	return headers
132}
133
134/** Page context in the sheet → location sent to the agent (real site slugs). */
135const CONTEXT_TO_LOCATION: Record<string, string> = {
136	paris: '/events/unleash-paris',
137	world: '/events/unleash-paris',
138	miami: '/events/unleash-miami',
139	'non-event': '/',
140	any: '/',
141	'': '/',
142}
143
144// ---------------------------------------------------------------- gateway helpers
145
146const sleep = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms))
147
148/**
149 * Gateway call with patient backoff: free-tier keys rate-limit aggressively
150 * (429 per model per minute), so retries wait long enough to matter and
151 * honour Retry-After when present.
152 */
153async function chatCompletion(options: {
154	model: string
155	system: string
156	user: string
157	maxTokens?: number
158}): Promise<string> {
159	const maxAttempts = 6
160	let lastError = ''
161	for (let attempt = 0; attempt < maxAttempts; attempt += 1) {
162		const response = await fetch(
163			'https://ai-gateway.vercel.sh/v1/chat/completions',
164			{
165				method: 'POST',
166				signal: AbortSignal.timeout(90_000),
167				headers: {
168					'Content-Type': 'application/json',
169					Authorization: `Bearer ${GATEWAY_KEY}`,
170				},
171				body: JSON.stringify({
172					model: options.model,
173					max_tokens: options.maxTokens ?? 700,
174					messages: [
175						{ role: 'system', content: options.system },
176						{ role: 'user', content: options.user },
177					],
178				}),
179			},
180		)
181		if (response.ok) {
182			const json = (await response.json()) as {
183				choices?: Array<{ message?: { content?: string } }>
184			}
185			return json.choices?.[0]?.message?.content?.trim() ?? ''
186		}
187		lastError = `Gateway ${options.model} failed (${response.status}): ${(await response.text()).slice(0, 300)}`
188		if (response.status !== 429 && response.status < 500) break
189		const retryAfter = Number(response.headers.get('retry-after')) || 0
190		const delay = Math.max(retryAfter * 1000, 5000 * 2 ** attempt)
191		await sleep(Math.min(delay, 90_000))
192	}
193	throw new Error(lastError)
194}
195
196// ---------------------------------------------------------------- visitor simulator
197
198const SIMULATOR_SYSTEM = `You role-play a website VISITOR on unleash.ai for a UAT test. You are given a test script describing what the visitor wants and, when scripted, what they say on each turn.
199
200Rules:
201- Output ONLY the visitor's next chat message. One short, natural message. No quotes, no narration, no stage directions.
202- Follow the script's turns in order. Where the script says the visitor "answers qualification questions" or "provides details", invent plausible consistent details (e.g. name Jordan Blake, jordan.blake@meridianhr.example, Head of People at MeridianHR, 800 employees) and answer what the agent actually asked.
203- If the script's turns are exhausted, or the conversation has reached its natural end, output exactly: [DONE]
204- Never break character, never mention that this is a test.`
205
206async function nextVisitorMessage(
207	testCase: AcceptanceCase,
208	turns: Turn[],
209): Promise<string> {
210	if (testCase.messages) return testCase.messages[turns.length] ?? '[DONE]'
211	// Literal opening questions must not be rewritten or refused by the simulator.
212	const scriptedOpening = testCase.input.match(
213		/^Turn 1:\s*(.*?)(?=\s*Turn 2|$)/s,
214	)?.[1]
215	const opening = (scriptedOpening || testCase.input).trim()
216	const literal =
217		/^(?:how|what|where|when|who|which|why|can|could|do|does|is|are|has|i\b|i'm|i’d|i'd|we\b|we're|show|ignore|pass|sap\b)/i.test(
218			opening,
219		) && !/\[|\(then|\(run|Turn \d/i.test(opening)
220	if (!turns.length && testCase.id === 'C02')
221		return 'How do I get a ticket for UNLEASH Miami?'
222	if (!turns.length && literal) return opening.split(' / ')[0].trim()
223	if (turns.length && literal && !scriptedOpening) return '[DONE]'
224	const transcript = turns
225		.map((t) => `VISITOR: ${t.user}\nAGENT: ${t.assistant}`)
226		.join('\n\n')
227	const user = [
228		`Test script (the visitor's behaviour): ${testCase.input}`,
229		`Page the visitor is on: ${testCase.context || 'any'}`,
230		turns.length === 0
231			? "The conversation has not started. Produce the visitor's FIRST message."
232			: `Conversation so far:\n\n${transcript}\n\nProduce the visitor's NEXT message, or [DONE] if the script is complete.`,
233	].join('\n\n')
234	return chatCompletion({
235		model: SIMULATOR_MODEL,
236		system: SIMULATOR_SYSTEM,
237		user,
238		maxTokens: 200,
239	})
240}
241
242// ---------------------------------------------------------------- agent call
243
244async function askAgent(options: {
245	message: string
246	history: { role: 'user' | 'assistant'; text: string }[]
247	location: string
248	chatId: string
249	userProfile: Record<string, unknown>
250	consent: boolean
251	workflowToken?: string
252	selectedEvent?: 'paris' | 'miami'
253}): Promise<{
254	text: string
255	firstTokenMs: number
256	totalMs: number
257	chunks: number
258	workflowToken: string | null
259	serverConsent: string | null
260	selectedEvent?: 'paris' | 'miami'
261}> {
262	const started = performance.now()
263	const response = await fetch(`${BASE_URL}/api/ai`, {
264		method: 'POST',
265		signal: AbortSignal.timeout(120_000),
266		headers: visitorHeaders(options.chatId),
267		body: JSON.stringify({
268			message: options.message,
269			type: 'chat',
270			profile: ['language:en-US'],
271			interests: [],
272			location: options.location,
273			chat_id: options.chatId,
274			history: options.history.slice(-20),
275			userProfile: options.userProfile,
276			consent: options.consent,
277			workflowToken: options.workflowToken,
278			selectedEvent: options.selectedEvent,
279		}),
280	})
281	if (!response.ok || !response.body) {
282		const errorText = await response.text().catch(() => '')
283		throw new Error(`/api/ai ${response.status}: ${errorText.slice(0, 200)}`)
284	}
285	const reader = response.body.getReader()
286	const decoder = new TextDecoder()
287	let text = ''
288	let firstTokenMs = 0
289	let chunks = 0
290	while (true) {
291		const { done, value } = await reader.read()
292		if (done) break
293		chunks += 1
294		if (!firstTokenMs) firstTokenMs = performance.now() - started
295		text += decoder.decode(value, { stream: true })
296	}
297	text += decoder.decode()
298	return {
299		text: text.trim(),
300		selectedEvent: parseAiEvent(response.headers.get('X-AI-Event')),
301		firstTokenMs,
302		chunks,
303		workflowToken: response.headers.get('X-AI-Workflow'),
304		serverConsent: response.headers.get('X-AI-Consent'),
305		totalMs: performance.now() - started,
306	}
307}
308
309/**
310 * Mirror the real browser client: before chat it POSTs the visitor message and
311 * previous assistant reply to /api/ai/profile, then sends the updated profile
312 * (role, company, interests…) with the current chat message. The server merges that profile into
313 * lead / agenda-email hand-offs, so skipping this loop hides payload bugs.
314 * Best-effort like the client — on failure the previous profile is kept.
315 */
316async function extractProfile(options: {
317	message: string
318	reply: string
319	profile: Record<string, unknown>
320	location: string
321	chatId: string
322}): Promise<Record<string, unknown>> {
323	try {
324		const response = await fetch(`${BASE_URL}/api/ai/profile`, {
325			method: 'POST',
326			signal: AbortSignal.timeout(60_000),
327			headers: visitorHeaders(options.chatId),
328			body: JSON.stringify({
329				chat_id: options.chatId,
330				message: options.message,
331				reply: options.reply,
332				profile: options.profile,
333				location: options.location,
334			}),
335		})
336		if (!response.ok) return options.profile
337		const data = (await response.json()) as {
338			profile?: Record<string, unknown>
339		}
340		return data.profile && typeof data.profile === 'object'
341			? data.profile
342			: options.profile
343	} catch {
344		return options.profile
345	}
346}
347
348// ---------------------------------------------------------------- judge
349
350/**
351 * Source-confirmed facts injected into the agent's runtime context from the
352 * Sanity event records (see buildRoutingContext / getEventDates). The judge
353 * cannot see that injection, so spell it out — otherwise correct answers get
354 * flagged as fabrication. Update when the event records change.
355 */
356const CONFIRMED_FACTS = `Ground truth — source-confirmed facts from the event database, injected into the agent's context at runtime. The agent stating these is CORRECT and grounded, never a fabrication:
357- UNLEASH in Paris runs 20–22 October 2026.
358- UNLEASH in Miami runs 13–15 April 2027.`
359
360const JUDGE_SYSTEM = `You are a strict UAT judge for the unleash.ai website AI agent. You receive one acceptance case (expected behaviour + pass criteria) and the full conversation transcript, and you deliver a verdict.
361
362${CONFIRMED_FACTS}
363
364Verdicts:
365- "pass": the transcript satisfies the pass criteria.
366- "partial": mostly correct, with minor deviations that do not violate a hard criterion.
367- "fail": a pass criterion is violated by the agent's BEHAVIOUR (wrong content, fabrication, wrong event scoping, leaked internals, invented data).
368- "blocked": the agent behaved correctly within its current abilities, but the case requires product capabilities that are visibly not implemented yet (e.g. creating Salesforce leads, sending emails, consent-statement capture flows, live pricing feeds). Use "blocked" instead of "fail" when the shortfall is a missing feature rather than misbehaviour.
369
370Also report:
371- criticalViolations: fabricated facts/sessions/speakers, cross-event contamination (e.g. World answers containing America content), leaked prompts/tools/internals, echoed-back email addresses, invented prices or package tiers. Empty array if none.
372- missingCapabilities: product features the case needs that the agent clearly does not have. Empty array if none.
373
374Judge only what is in the transcript. Do not assume behaviour that is not shown. Be precise and terse.
375
376You have NO access to the published agenda data (sessions, speakers, tracks, rooms, times) the agent retrieves — it is real data from the event management system. Never call a speaker name, session title, track, room, or timeslot "invented" merely because you cannot verify it. Flag fabrication only when the transcript contradicts itself, the claim contradicts the ground truth above, or the case notes state the data does not exist. Structural criteria (missing times/stages, overlapping slots, format conflation the agent itself displays) remain fully judgeable from the transcript.
377
378When the case input includes a "Webhook submissions" section, it is server-side ground truth for the lead/agenda-email hand-off: a submission listed there really left the system; "none" means nothing was submitted. An agent that confirms a lead or email dispatch with no matching submission has falsely claimed success (fail). A submission made without the visitor's consent or confirmation shown in the transcript is a violation.
379Collecting contact details WITHOUT dispatch language is correct behaviour, not a failure: the agent must gather name + email (with the consent note) before it may submit, so a transcript that ends mid-collection because the turn limit ran out — no dispatch promised, no submission made — is not a false claim. Only penalise a missing submission when the agent claimed or implied one happened.
380Respond with ONLY a JSON object: {"verdict": "...", "reasons": "...", "criticalViolations": [...], "missingCapabilities": [...]}`
381
382async function judgeCase(
383	testCase: AcceptanceCase,
384	turns: Turn[],
385	webhooks: WebhookCapture[],
386): Promise<Verdict> {
387	const transcript = turns
388		.map((t, i) => `--- Turn ${i + 1}\nVISITOR: ${t.user}\nAGENT: ${t.assistant}`)
389		.join('\n\n')
390	// Only assert ground truth when the capture server could actually observe
391	// the hand-off; against deployments the judge falls back to transcript-only.
392	const webhookSection = CAPTURE_ENABLED
393		? `\nWebhook submissions (server-side ground truth): ${
394				webhooks.length
395					? webhooks
396							.map(
397								(w) =>
398									`\n- ${w.type} (receiver HTTP ${w.responseStatus}; only 2xx means accepted): ${JSON.stringify(w.payload)}`,
399							)
400							.join('')
401					: 'none — no lead or agenda email was actually submitted.'
402			}`
403		: ''
404	const user = [
405		`Case ${testCase.id} [${testCase.capability}] (criticality: ${testCase.criticality})`,
406		`Page context: ${testCase.context || 'any'}`,
407		`Initial session state: ${JSON.stringify({ profile: testCase.initialProfile ?? {}, consent: testCase.initialConsent === true })}`,
408		`User input script: ${testCase.input}`,
409		`Expected behaviour: ${testCase.expected}`,
410		`Pass criteria: ${testCase.passCriteria}`,
411		testCase.notes ? `Notes: ${testCase.notes}` : '',
412		`Seeded conversation: ${JSON.stringify(testCase.initialHistory ?? [])}`,
413		`\nTranscript:\n${transcript}`,
414		webhookSection,
415	]
416		.filter(Boolean)
417		.join('\n')
418
419	const raw = await chatCompletion({
420		model: JUDGE_MODEL,
421		system: JUDGE_SYSTEM,
422		user,
423		// Generous budget: a truncated reasons string breaks the JSON parse
424		// and burns the whole case as "error".
425		maxTokens: 1000,
426	})
427	const match = raw.match(/\{[\s\S]*\}/)
428	if (!match) {
429		return {
430			verdict: 'error',
431			reasons: `Judge returned unparsable output: ${raw.slice(0, 200)}`,
432			criticalViolations: [],
433			missingCapabilities: [],
434		}
435	}
436	try {
437		const parsed = JSON.parse(match[0]) as Partial<Verdict>
438		return {
439			verdict: (parsed.verdict as Verdict['verdict']) ?? 'error',
440			reasons: parsed.reasons ?? '',
441			criticalViolations: parsed.criticalViolations ?? [],
442			missingCapabilities: parsed.missingCapabilities ?? [],
443		}
444	} catch {
445		return {
446			verdict: 'error',
447			reasons: `Judge JSON parse failed: ${match[0].slice(0, 200)}`,
448			criticalViolations: [],
449			missingCapabilities: [],
450		}
451	}
452}
453
454// ---------------------------------------------------------------- langfuse scores
455
456const LANGFUSE_BASE = process.env.LANGFUSE_BASE_URL?.trim()
457const LANGFUSE_AUTH =
458	process.env.LANGFUSE_PUBLIC_KEY && process.env.LANGFUSE_SECRET_KEY
459		? Buffer.from(
460				`${process.env.LANGFUSE_PUBLIC_KEY}:${process.env.LANGFUSE_SECRET_KEY}`,
461			).toString('base64')
462		: null
463
464const PASS_VALUE: Record<string, number> = { pass: 1, partial: 0.5, fail: 0 }
465
466/**
467 * Judge verdicts become Langfuse scores on the traced session, so pass-rate
468 * trends per capability are visible across runs and prompt changes are
469 * measurable experiments. Best-effort: scoring failures never fail the run.
470 */
471async function pushVerdictScores(result: CaseResult, runId: string) {
472	if (!LANGFUSE_BASE || !LANGFUSE_AUTH) return
473	if (result.verdict.verdict === 'error') return
474
475	const metadata = {
476		caseId: result.case.id,
477		capability: result.case.capability,
478		criticality: result.case.criticality,
479		runId,
480		suite: SUITE,
481	}
482	// Deterministic ids make pushes idempotent upserts — re-runs and retries
483	// never create duplicate scores.
484	const scores: Record<string, unknown>[] = [
485		{
486			id: `${ID_PREFIX}-${runId}-${result.case.id}-verdict`,
487			sessionId: result.chatId,
488			name: 'uat-verdict',
489			dataType: 'CATEGORICAL',
490			value: result.verdict.verdict,
491			comment: result.verdict.reasons.slice(0, 800),
492			metadata,
493		},
494	]
495	const numeric = PASS_VALUE[result.verdict.verdict]
496	if (numeric !== undefined) {
497		scores.push({
498			id: `${ID_PREFIX}-${runId}-${result.case.id}-pass`,
499			sessionId: result.chatId,
500			name: 'uat-pass',
501			dataType: 'NUMERIC',
502			value: numeric,
503			metadata,
504		})
505	}
506
507	for (const score of scores) {
508		// Langfuse's public API rate-limits bursts: pace, retry 429 once.
509		for (let attempt = 0; attempt < 2; attempt += 1) {
510			try {
511				const response = await fetch(`${LANGFUSE_BASE}/api/public/scores`, {
512					method: 'POST',
513					headers: {
514						'Content-Type': 'application/json',
515						Authorization: `Basic ${LANGFUSE_AUTH}`,
516					},
517					body: JSON.stringify(score),
518				})
519				if (response.ok) break
520				if (response.status === 429 && attempt === 0) {
521					await sleep(20_000)
522					continue
523				}
524				throw new Error(`scores API ${response.status}`)
525			} catch (error) {
526				if (attempt === 1) {
527					console.error(
528						`  ! Langfuse score failed for ${result.case.id}:`,
529						error instanceof Error ? error.message : error,
530					)
531				}
532			}
533		}
534		await sleep(400)
535	}
536}
537
538// ---------------------------------------------------------------- webhook capture
539
540/**
541 * Stand-in for the Zapier catch hook: records every payload the app submits,
542 * keyed by chat_id, and answers with Zapier's success shape. Lets the judge
543 * verify that leads/agenda emails were actually handed off, not just claimed.
544 */
545const capturedWebhooks = new Map<string, WebhookCapture[]>()
546let captureServer: ReturnType<typeof Bun.serve> | undefined
547
548function startCaptureServer() {
549	captureServer = Bun.serve({
550		port: CAPTURE_PORT,
551		async fetch(request) {
552			if (request.method !== 'POST') {
553				return new Response('capture server up', { status: 200 })
554			}
555			const payload = (await request.json().catch(() => ({}))) as Record<
556				string,
557				unknown
558			>
559			const chatId = typeof payload.chat_id === 'string' ? payload.chat_id : ''
560			const forcedFailure = chatId.endsWith('-C49') || chatId.endsWith('-F13')
561			const entry: WebhookCapture = {
562				responseStatus: forcedFailure ? 503 : 200,
563				type: typeof payload.type === 'string' ? payload.type : 'unknown',
564				payload,
565			}
566			capturedWebhooks.set(chatId, [
567				...(capturedWebhooks.get(chatId) ?? []),
568				entry,
569			])
570			if (forcedFailure) {
571				console.warn(`Forced webhook failure for ${chatId}`)
572				return Response.json(
573					{ error: 'UAT forced delivery failure' },
574					{ status: 503 },
575				)
576			}
577			return Response.json({ status: 'success' })
578		},
579	})
580	console.log(
581		`Webhook capture listening on http://localhost:${CAPTURE_PORT} — ` +
582			`start the app with ZAPIER_LEAD_WEBHOOK_URL=http://localhost:${CAPTURE_PORT}/zapier`,
583	)
584}
585
586function webhooksForCase(chatId: string): WebhookCapture[] {
587	return capturedWebhooks.get(chatId) ?? []
588}
589
590// ---------------------------------------------------------------- per-case run
591
592async function runCase(
593	testCase: AcceptanceCase,
594	runId: string,
595): Promise<CaseResult> {
596	const chatId = `${ID_PREFIX}-${runId}-${testCase.id}`
597	const location = CONTEXT_TO_LOCATION[testCase.context] ?? '/'
598	const turns: Turn[] = []
599	let userProfile: Record<string, unknown> = { ...testCase.initialProfile }
600	let consent = testCase.initialConsent === true
601	let workflowToken: string | undefined
602	let selectedEvent: 'paris' | 'miami' | undefined
603
604	try {
605		for (let turn = 0; turn < MAX_TURNS; turn += 1) {
606			const visitorMessage = await nextVisitorMessage(testCase, turns)
607			if (!visitorMessage || visitorMessage.includes('[DONE]')) break
608
609			const history = [
610				...(testCase.initialHistory ?? []),
611				...turns.flatMap((t) => [
612					{ role: 'user' as const, text: t.user },
613					{ role: 'assistant' as const, text: t.assistant },
614				]),
615			]
616			history.push({ role: 'user', text: visitorMessage })
617			consent = sessionConsent(history, visitorMessage, consent)
618			const profileStarted = performance.now()
619			userProfile = await extractProfile({
620				message: visitorMessage,
621				reply:
622					turns.at(-1)?.assistant ||
623					testCase.initialHistory?.findLast((t) => t.role === 'assistant')?.text ||
624					'',
625				profile: userProfile,
626				location,
627				chatId,
628			})
629			const profileMs = performance.now() - profileStarted
630
631			// The agent's own gateway calls can rate-limit too; empty or failed
632			// replies get a couple of patient retries before counting as real.
633			let reply: Awaited<ReturnType<typeof askAgent>> | undefined
634			for (let attempt = 0; attempt < 3; attempt += 1) {
635				try {
636					const candidate = await askAgent({
637						message: visitorMessage,
638						history,
639						location,
640						chatId,
641						userProfile,
642						consent,
643						workflowToken,
644						selectedEvent,
645					})
646					if (candidate.text) {
647						reply = candidate
648						break
649					}
650				} catch (error) {
651					if (attempt === 2) throw error
652				}
653				await sleep(15_000 * (attempt + 1))
654			}
655			if (!reply) throw new Error('Agent returned empty replies after retries')
656			if (reply.workflowToken !== null)
657				workflowToken = reply.workflowToken || undefined
658			if (reply.serverConsent !== null) consent = reply.serverConsent === 'true'
659			selectedEvent = reply.selectedEvent || selectedEvent
660			turns.push({
661				user: visitorMessage,
662				assistant: reply.text,
663				firstTokenMs: Math.round(reply.firstTokenMs),
664				chunks: reply.chunks,
665				totalMs: Math.round(reply.totalMs),
666				profileMs: Math.round(profileMs),
667				visitorFirstTokenMs: Math.round(profileMs + reply.firstTokenMs),
668			})
669		}
670
671		if (turns.length === 0) {
672			return {
673				case: testCase,
674				turns,
675				chatId,
676				webhooks: webhooksForCase(chatId),
677				verdict: {
678					verdict: 'error',
679					reasons: 'Simulator produced no visitor message.',
680					criticalViolations: [],
681					missingCapabilities: [],
682				},
683			}
684		}
685
686		// Tool-driven submissions land mid-stream, but give in-flight posts a beat.
687		if (CAPTURE_ENABLED) await sleep(500)
688		const webhooks = webhooksForCase(chatId)
689		const verdict = await judgeCase(testCase, turns, webhooks)
690		return { case: testCase, turns, verdict, chatId, webhooks }
691	} catch (error) {
692		return {
693			case: testCase,
694			turns,
695			chatId,
696			webhooks: webhooksForCase(chatId),
697			verdict: {
698				verdict: 'error',
699				reasons: error instanceof Error ? error.message : String(error),
700				criticalViolations: [],
701				missingCapabilities: [],
702			},
703		}
704	}
705}
706
707// ---------------------------------------------------------------- reporting
708
709function caseMarkdown(result: CaseResult): string {
710	const c = result.case
711	const lines = [
712		`# ${c.id} — ${c.capability} (${c.criticality})`,
713		'',
714		`- Context: \`${c.context || 'any'}\` → \`${CONTEXT_TO_LOCATION[c.context] ?? '/'}\``,
715		`- Chat id: \`${result.chatId}\``,
716		`- Script: ${c.input}`,
717		`- Expected: ${c.expected}`,
718		`- Pass criteria: ${c.passCriteria}`,
719		'',
720		`## Verdict: ${result.verdict.verdict.toUpperCase()}`,
721		'',
722		result.verdict.reasons,
723	]
724	if (result.verdict.criticalViolations.length) {
725		lines.push('', '**Critical violations:**')
726		for (const v of result.verdict.criticalViolations) lines.push(`- ${v}`)
727	}
728	if (result.verdict.missingCapabilities.length) {
729		lines.push('', '**Missing capabilities:**')
730		for (const m of result.verdict.missingCapabilities) lines.push(`- ${m}`)
731	}
732	if (CAPTURE_ENABLED) {
733		lines.push('', '## Webhook submissions', '')
734		if (result.webhooks.length) {
735			for (const w of result.webhooks) {
736				lines.push(`- \`${w.type}\`: \`${JSON.stringify(w.payload)}\``)
737			}
738		} else {
739			lines.push('_None captured._')
740		}
741	}
742	lines.push('', '## Transcript', '')
743	result.turns.forEach((t, i) => {
744		lines.push(
745			`### Turn ${i + 1} _(first token ${t.firstTokenMs}ms, total ${t.totalMs}ms)_`,
746			'',
747			`**Visitor:** ${t.user}`,
748			'',
749			`**Agent:** ${t.assistant}`,
750			'',
751		)
752	})
753	return lines.join('\n')
754}
755
756function percentile(values: number[], p: number): number {
757	if (values.length === 0) return 0
758	const sorted = [...values].sort((a, b) => a - b)
759	const rank = Math.max(0, Math.ceil((p / 100) * sorted.length) - 1)
760	return sorted[Math.min(sorted.length - 1, rank)]
761}
762
763function buildSummary(results: CaseResult[], runId: string): string {
764	const byVerdict = (v: string) => results.filter((r) => r.verdict.verdict === v)
765	const groups = new Map<string, CaseResult[]>()
766	for (const r of results) {
767		const key = `${r.case.criticality}`
768		groups.set(key, [...(groups.get(key) ?? []), r])
769	}
770	const capGroups = new Map<string, CaseResult[]>()
771	for (const r of results) {
772		capGroups.set(r.case.capability, [
773			...(capGroups.get(r.case.capability) ?? []),
774			r,
775		])
776	}
777	const latencies = results.flatMap((r) => r.turns.map((t) => t.firstTokenMs))
778
779	const lines = [
780		`# AI Agent UAT — run ${runId}`,
781		'',
782		`- Base URL: ${BASE_URL}`,
783		`- Cases run: ${results.length}`,
784		`- Simulator: ${SIMULATOR_MODEL} · Judge: ${JUDGE_MODEL}`,
785		CAPTURE_ENABLED
786			? `- Webhook submissions captured: ${results.reduce((n, r) => n + r.webhooks.length, 0)} (lead: ${results.reduce((n, r) => n + r.webhooks.filter((w) => w.type === 'lead').length, 0)}, agenda-email: ${results.reduce((n, r) => n + r.webhooks.filter((w) => w.type === 'agenda-email').length, 0)})`
787			: '- Webhook capture: off (remote target — lead criteria judged from transcript only)',
788		'',
789		'## Results',
790		'',
791		`| Verdict | Count |`,
792		`| --- | --- |`,
793		...['pass', 'partial', 'blocked', 'fail', 'error'].map(
794			(v) => `| ${v} | ${byVerdict(v).length} |`,
795		),
796		'',
797		'### By criticality',
798		'',
799		'| Criticality | Pass | Partial | Blocked | Fail | Error |',
800		'| --- | --- | --- | --- | --- | --- |',
801		...[...groups.entries()].map(([crit, rs]) => {
802			const count = (v: string) => rs.filter((r) => r.verdict.verdict === v).length
803			return `| ${crit} | ${count('pass')} | ${count('partial')} | ${count('blocked')} | ${count('fail')} | ${count('error')} |`
804		}),
805		'',
806		'### By capability',
807		'',
808		'| Capability | Pass | Partial | Blocked | Fail | Error |',
809		'| --- | --- | --- | --- | --- | --- |',
810		...[...capGroups.entries()].map(([cap, rs]) => {
811			const count = (v: string) => rs.filter((r) => r.verdict.verdict === v).length
812			return `| ${cap} | ${count('pass')} | ${count('partial')} | ${count('blocked')} | ${count('fail')} | ${count('error')} |`
813		}),
814		'',
815		'## Latency (first streamed token, per turn — O01)',
816		'',
817		`- Samples: ${latencies.length} · p50: ${percentile(latencies, 50)}ms · p90: ${percentile(latencies, 90)}ms · max: ${Math.max(0, ...latencies)}ms`,
818		'',
819	]
820
821	const failures = results.filter((r) =>
822		['fail', 'partial', 'error'].includes(r.verdict.verdict),
823	)
824	if (failures.length) {
825		lines.push('## Failures & partials', '')
826		for (const r of failures) {
827			lines.push(
828				`- **${r.case.id}** (${r.case.criticality}, ${r.case.capability}) — ${r.verdict.verdict}: ${r.verdict.reasons.slice(0, 300)}`,
829			)
830		}
831		lines.push('')
832	}
833	const blocked = byVerdict('blocked')
834	if (blocked.length) {
835		lines.push('## Blocked on missing product capabilities', '')
836		const capabilities = new Map<string, string[]>()
837		for (const r of blocked) {
838			for (const m of r.verdict.missingCapabilities) {
839				capabilities.set(m, [...(capabilities.get(m) ?? []), r.case.id])
840			}
841		}
842		for (const [capability, ids] of capabilities) {
843			lines.push(`- ${capability} — ${ids.join(', ')}`)
844		}
845		lines.push('')
846	}
847	const violations = results.filter((r) => r.verdict.criticalViolations.length)
848	if (violations.length) {
849		lines.push('## ⚠ Critical violations', '')
850		for (const r of violations) {
851			for (const v of r.verdict.criticalViolations) {
852				lines.push(`- **${r.case.id}**: ${v}`)
853			}
854		}
855		lines.push('')
856	}
857	return lines.join('\n')
858}
859
860// ---------------------------------------------------------------- main
861
862async function main() {
863	const { cases } = JSON.parse(
864		readFileSync(join(import.meta.dir, `${SUITE}.json`), 'utf8'),
865	) as { cases: AcceptanceCase[] }
866
867	let selected = cases
868	const only = arg('only')
869	if (only) {
870		const ids = new Set(only.split(',').map((s) => s.trim().toUpperCase()))
871		selected = selected.filter((c) => ids.has(c.id.toUpperCase()))
872	}
873	const criticality = arg('criticality')
874	if (criticality) {
875		selected = selected.filter(
876			(c) => c.criticality.toLowerCase() === criticality.toLowerCase(),
877		)
878	}
879	const capability = arg('capability')
880	if (capability) {
881		selected = selected.filter((c) =>
882			c.capability.toLowerCase().includes(capability.toLowerCase()),
883		)
884	}
885	const max = Number(arg('max'))
886	if (Number.isFinite(max) && max > 0) selected = selected.slice(0, max)
887
888	if (hasFlag('list')) {
889		for (const c of selected) {
890			console.log(`${c.id}\t${c.criticality}\t${c.capability}\t${c.context}`)
891		}
892		return
893	}
894	if (selected.length === 0) {
895		console.error('No cases selected.')
896		process.exit(1)
897	}
898
899	const runId = new Date().toISOString().replace(/[:.]/g, '-').slice(0, 19)
900	const runDir =
901		SUITE === 'cases'
902			? join(import.meta.dir, 'runs', runId)
903			: join(import.meta.dir, 'runs', SUITE, runId)
904	mkdirSync(runDir, { recursive: true })
905	console.log(
906		`Running ${selected.length} case(s) from suite "${SUITE}" against ${BASE_URL} → ${runDir}`,
907	)
908	if (CAPTURE_ENABLED) startCaptureServer()
909
910	const results: CaseResult[] = []
911	let cursor = 0
912	async function worker() {
913		while (cursor < selected.length) {
914			const testCase = selected[cursor++]
915			const started = performance.now()
916			const result = await runCase(testCase, runId)
917			results.push(result)
918			writeFileSync(join(runDir, `${testCase.id}.md`), caseMarkdown(result))
919			await pushVerdictScores(result, runId)
920			const elapsed = Math.round((performance.now() - started) / 1000)
921			console.log(
922				`${testCase.id.padEnd(5)} ${result.verdict.verdict.toUpperCase().padEnd(8)} ${result.turns.length} turn(s) ${elapsed}s — ${result.verdict.reasons.slice(0, 110)}`,
923			)
924		}
925	}
926	await Promise.all(
927		Array.from({ length: Math.min(CONCURRENCY, selected.length) }, worker),
928	)
929
930	results.sort((a, b) => a.case.id.localeCompare(b.case.id))
931	writeFileSync(
932		join(runDir, 'results.json'),
933		JSON.stringify(
934			{
935				runId,
936				baseUrl: BASE_URL,
937				simulator: SIMULATOR_MODEL,
938				judge: JUDGE_MODEL,
939				results,
940			},
941			null,
942			'\t',
943		),
944	)
945	const summary = buildSummary(results, runId)
946	writeFileSync(join(runDir, 'summary.md'), summary)
947	console.log(`\n${summary}`)
948	console.log(`Artifacts: ${runDir}`)
949	if (captureServer) {
950		const total = results.reduce((n, r) => n + r.webhooks.length, 0)
951		if (total === 0) {
952			console.warn(
953				'Webhook capture was on but nothing arrived — is the app running with ' +
954					`ZAPIER_LEAD_WEBHOOK_URL=http://localhost:${CAPTURE_PORT}/zapier ?`,
955			)
956		}
957		captureServer.stop(true)
958	}
959
960	const hardFails = results.filter((r) => r.verdict.verdict === 'fail').length
961	process.exitCode = hardFails > 0 ? 1 : 0
962}
963
964await main()
965