scripts/ai-acceptance/run.ts
1/**
2 * UAT runner for the unleash.ai AI agent.
3 *
4 * For each conversational case in the selected suite (cases.json by default,
5 * <name>.json with --suite <name>, e.g. --suite knowledge):
6 * - a visitor-simulator model plays the user following the case's turn script
7 * - each turn hits POST <base>/api/ai and streams the reply (first-token latency recorded)
8 * - an LLM judge scores the full transcript against Expected behaviour + Pass criteria
9 * - the whole conversation + verdict is written to runs/<timestamp>/<id>.md (+ results.json)
10 * - a summary.md with pass rates by criticality/capability and latency stats is generated
11 *
12 * Usage:
13 * bun --env-file=../../apps/web/.env run.ts [--only C01,C35] [--criticality Critical]
14 * [--capability Guardrails] [--max 10] [--base-url http://localhost:3001]
15 * [--concurrency 3] [--turns 5] [--list] [--capture-port 8788] [--no-capture]
16 * [--suite knowledge]
17 *
18 * Requires AI_GATEWAY (or AI_GATEWAY_API_KEY) for the simulator + judge models.
19 *
20 * Webhook capture (Zapier hand-off ground truth): when the target is localhost
21 * (or --capture-port is passed), the runner listens for the app's outbound
22 * Zapier POSTs and matches them to cases by chat_id, so "lead created" pass
23 * criteria are judged on actual submissions rather than the agent's claims.
24 * The dev server under test must point its hook at the runner:
25 * ZAPIER_LEAD_WEBHOOK_URL=http://localhost:8788/zapier pnpm dev:web
26 */
27import { mkdirSync, readFileSync, writeFileSync } from 'node:fs'
28import { join } from 'node:path'
29import {
30 parseAiEvent,
31 sessionConsent,
32} from '../../packages/ai/components/workflows'
33
34type AcceptanceCase = {
35 id: string
36 source: string
37 capability: string
38 context: string
39 input: string
40 expected: string
41 passCriteria: string
42 criticality: string
43 sowRef: string
44 notes: string
45 messages?: string[]
46 initialProfile?: Record<string, unknown>
47 initialConsent?: boolean
48 initialHistory?: { role: 'user' | 'assistant'; text: string }[]
49}
50
51type Turn = {
52 user: string
53 assistant: string
54 firstTokenMs: number
55 totalMs: number
56 chunks?: number
57 profileMs?: number
58 visitorFirstTokenMs?: number
59}
60
61type Verdict = {
62 verdict: 'pass' | 'partial' | 'fail' | 'blocked' | 'error'
63 reasons: string
64 criticalViolations: string[]
65 missingCapabilities: string[]
66}
67
68type WebhookCapture = {
69 responseStatus: number
70 type: string
71 payload: Record<string, unknown>
72}
73
74type CaseResult = {
75 case: AcceptanceCase
76 turns: Turn[]
77 verdict: Verdict
78 chatId: string
79 webhooks: WebhookCapture[]
80}
81
82// ---------------------------------------------------------------- config
83
84function arg(name: string): string | undefined {
85 const index = process.argv.indexOf(`--${name}`)
86 return index >= 0 ? process.argv[index + 1] : undefined
87}
88const hasFlag = (name: string) => process.argv.includes(`--${name}`)
89
90// Which case file to run. Suites other than the default keep their own run
91// folder and their own chat-id / Langfuse score-id prefix so results never
92// collide with the main acceptance set.
93const SUITE = arg('suite') ?? 'cases'
94if (!/^[a-z0-9-]+$/.test(SUITE)) {
95 console.error(
96 `Invalid --suite "${SUITE}" (use a case file name without .json)`,
97 )
98 process.exit(1)
99}
100const ID_PREFIX = SUITE === 'cases' ? 'uat' : `uat-${SUITE}`
101
102const BASE_URL = arg('base-url') ?? 'http://localhost:3001'
103const CONCURRENCY = Number(arg('concurrency') ?? 2)
104const MAX_TURNS = Number(arg('turns') ?? 5)
105const SIMULATOR_MODEL =
106 process.env.UAT_SIMULATOR_MODEL || 'anthropic/claude-haiku-4.5'
107const JUDGE_MODEL = process.env.UAT_JUDGE_MODEL || 'anthropic/claude-sonnet-4.6'
108const GATEWAY_KEY = process.env.AI_GATEWAY_API_KEY || process.env.AI_GATEWAY
109if (!GATEWAY_KEY) {
110 console.error(
111 'Missing AI_GATEWAY / AI_GATEWAY_API_KEY env (simulator + judge)',
112 )
113 process.exit(1)
114}
115
116// Zapier webhook capture only works when the app under test can reach this
117// process, so it defaults on for localhost targets and off for deployments.
118const CAPTURE_PORT = Number(arg('capture-port') ?? 8788)
119const CAPTURE_ENABLED =
120 !hasFlag('no-capture') &&
121 (arg('capture-port') !== undefined || /localhost|127\.0\.0\.1/.test(BASE_URL))
122
123// Local simulated visitors need separate buckets, just like distinct real IPs.
124// Never spoof this header against a deployed environment.
125function visitorHeaders(chatId: string): Record<string, string> {
126 const headers: Record<string, string> = { 'Content-Type': 'application/json' }
127 const hostname = new URL(BASE_URL).hostname
128 if (hostname === 'localhost' || hostname === '127.0.0.1') {
129 headers['x-forwarded-for'] = `uat-${chatId}`
130 }
131 return headers
132}
133
134/** Page context in the sheet → location sent to the agent (real site slugs). */
135const CONTEXT_TO_LOCATION: Record<string, string> = {
136 paris: '/events/unleash-paris',
137 world: '/events/unleash-paris',
138 miami: '/events/unleash-miami',
139 'non-event': '/',
140 any: '/',
141 '': '/',
142}
143
144// ---------------------------------------------------------------- gateway helpers
145
146const sleep = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms))
147
148/**
149 * Gateway call with patient backoff: free-tier keys rate-limit aggressively
150 * (429 per model per minute), so retries wait long enough to matter and
151 * honour Retry-After when present.
152 */
153async function chatCompletion(options: {
154 model: string
155 system: string
156 user: string
157 maxTokens?: number
158}): Promise<string> {
159 const maxAttempts = 6
160 let lastError = ''
161 for (let attempt = 0; attempt < maxAttempts; attempt += 1) {
162 const response = await fetch(
163 'https://ai-gateway.vercel.sh/v1/chat/completions',
164 {
165 method: 'POST',
166 signal: AbortSignal.timeout(90_000),
167 headers: {
168 'Content-Type': 'application/json',
169 Authorization: `Bearer ${GATEWAY_KEY}`,
170 },
171 body: JSON.stringify({
172 model: options.model,
173 max_tokens: options.maxTokens ?? 700,
174 messages: [
175 { role: 'system', content: options.system },
176 { role: 'user', content: options.user },
177 ],
178 }),
179 },
180 )
181 if (response.ok) {
182 const json = (await response.json()) as {
183 choices?: Array<{ message?: { content?: string } }>
184 }
185 return json.choices?.[0]?.message?.content?.trim() ?? ''
186 }
187 lastError = `Gateway ${options.model} failed (${response.status}): ${(await response.text()).slice(0, 300)}`
188 if (response.status !== 429 && response.status < 500) break
189 const retryAfter = Number(response.headers.get('retry-after')) || 0
190 const delay = Math.max(retryAfter * 1000, 5000 * 2 ** attempt)
191 await sleep(Math.min(delay, 90_000))
192 }
193 throw new Error(lastError)
194}
195
196// ---------------------------------------------------------------- visitor simulator
197
198const SIMULATOR_SYSTEM = `You role-play a website VISITOR on unleash.ai for a UAT test. You are given a test script describing what the visitor wants and, when scripted, what they say on each turn.
199
200Rules:
201- Output ONLY the visitor's next chat message. One short, natural message. No quotes, no narration, no stage directions.
202- Follow the script's turns in order. Where the script says the visitor "answers qualification questions" or "provides details", invent plausible consistent details (e.g. name Jordan Blake, jordan.blake@meridianhr.example, Head of People at MeridianHR, 800 employees) and answer what the agent actually asked.
203- If the script's turns are exhausted, or the conversation has reached its natural end, output exactly: [DONE]
204- Never break character, never mention that this is a test.`
205
206async function nextVisitorMessage(
207 testCase: AcceptanceCase,
208 turns: Turn[],
209): Promise<string> {
210 if (testCase.messages) return testCase.messages[turns.length] ?? '[DONE]'
211 // Literal opening questions must not be rewritten or refused by the simulator.
212 const scriptedOpening = testCase.input.match(
213 /^Turn 1:\s*(.*?)(?=\s*Turn 2|$)/s,
214 )?.[1]
215 const opening = (scriptedOpening || testCase.input).trim()
216 const literal =
217 /^(?:how|what|where|when|who|which|why|can|could|do|does|is|are|has|i\b|i'm|i’d|i'd|we\b|we're|show|ignore|pass|sap\b)/i.test(
218 opening,
219 ) && !/\[|\(then|\(run|Turn \d/i.test(opening)
220 if (!turns.length && testCase.id === 'C02')
221 return 'How do I get a ticket for UNLEASH Miami?'
222 if (!turns.length && literal) return opening.split(' / ')[0].trim()
223 if (turns.length && literal && !scriptedOpening) return '[DONE]'
224 const transcript = turns
225 .map((t) => `VISITOR: ${t.user}\nAGENT: ${t.assistant}`)
226 .join('\n\n')
227 const user = [
228 `Test script (the visitor's behaviour): ${testCase.input}`,
229 `Page the visitor is on: ${testCase.context || 'any'}`,
230 turns.length === 0
231 ? "The conversation has not started. Produce the visitor's FIRST message."
232 : `Conversation so far:\n\n${transcript}\n\nProduce the visitor's NEXT message, or [DONE] if the script is complete.`,
233 ].join('\n\n')
234 return chatCompletion({
235 model: SIMULATOR_MODEL,
236 system: SIMULATOR_SYSTEM,
237 user,
238 maxTokens: 200,
239 })
240}
241
242// ---------------------------------------------------------------- agent call
243
244async function askAgent(options: {
245 message: string
246 history: { role: 'user' | 'assistant'; text: string }[]
247 location: string
248 chatId: string
249 userProfile: Record<string, unknown>
250 consent: boolean
251 workflowToken?: string
252 selectedEvent?: 'paris' | 'miami'
253}): Promise<{
254 text: string
255 firstTokenMs: number
256 totalMs: number
257 chunks: number
258 workflowToken: string | null
259 serverConsent: string | null
260 selectedEvent?: 'paris' | 'miami'
261}> {
262 const started = performance.now()
263 const response = await fetch(`${BASE_URL}/api/ai`, {
264 method: 'POST',
265 signal: AbortSignal.timeout(120_000),
266 headers: visitorHeaders(options.chatId),
267 body: JSON.stringify({
268 message: options.message,
269 type: 'chat',
270 profile: ['language:en-US'],
271 interests: [],
272 location: options.location,
273 chat_id: options.chatId,
274 history: options.history.slice(-20),
275 userProfile: options.userProfile,
276 consent: options.consent,
277 workflowToken: options.workflowToken,
278 selectedEvent: options.selectedEvent,
279 }),
280 })
281 if (!response.ok || !response.body) {
282 const errorText = await response.text().catch(() => '')
283 throw new Error(`/api/ai ${response.status}: ${errorText.slice(0, 200)}`)
284 }
285 const reader = response.body.getReader()
286 const decoder = new TextDecoder()
287 let text = ''
288 let firstTokenMs = 0
289 let chunks = 0
290 while (true) {
291 const { done, value } = await reader.read()
292 if (done) break
293 chunks += 1
294 if (!firstTokenMs) firstTokenMs = performance.now() - started
295 text += decoder.decode(value, { stream: true })
296 }
297 text += decoder.decode()
298 return {
299 text: text.trim(),
300 selectedEvent: parseAiEvent(response.headers.get('X-AI-Event')),
301 firstTokenMs,
302 chunks,
303 workflowToken: response.headers.get('X-AI-Workflow'),
304 serverConsent: response.headers.get('X-AI-Consent'),
305 totalMs: performance.now() - started,
306 }
307}
308
309/**
310 * Mirror the real browser client: before chat it POSTs the visitor message and
311 * previous assistant reply to /api/ai/profile, then sends the updated profile
312 * (role, company, interests…) with the current chat message. The server merges that profile into
313 * lead / agenda-email hand-offs, so skipping this loop hides payload bugs.
314 * Best-effort like the client — on failure the previous profile is kept.
315 */
316async function extractProfile(options: {
317 message: string
318 reply: string
319 profile: Record<string, unknown>
320 location: string
321 chatId: string
322}): Promise<Record<string, unknown>> {
323 try {
324 const response = await fetch(`${BASE_URL}/api/ai/profile`, {
325 method: 'POST',
326 signal: AbortSignal.timeout(60_000),
327 headers: visitorHeaders(options.chatId),
328 body: JSON.stringify({
329 chat_id: options.chatId,
330 message: options.message,
331 reply: options.reply,
332 profile: options.profile,
333 location: options.location,
334 }),
335 })
336 if (!response.ok) return options.profile
337 const data = (await response.json()) as {
338 profile?: Record<string, unknown>
339 }
340 return data.profile && typeof data.profile === 'object'
341 ? data.profile
342 : options.profile
343 } catch {
344 return options.profile
345 }
346}
347
348// ---------------------------------------------------------------- judge
349
350/**
351 * Source-confirmed facts injected into the agent's runtime context from the
352 * Sanity event records (see buildRoutingContext / getEventDates). The judge
353 * cannot see that injection, so spell it out — otherwise correct answers get
354 * flagged as fabrication. Update when the event records change.
355 */
356const CONFIRMED_FACTS = `Ground truth — source-confirmed facts from the event database, injected into the agent's context at runtime. The agent stating these is CORRECT and grounded, never a fabrication:
357- UNLEASH in Paris runs 20–22 October 2026.
358- UNLEASH in Miami runs 13–15 April 2027.`
359
360const JUDGE_SYSTEM = `You are a strict UAT judge for the unleash.ai website AI agent. You receive one acceptance case (expected behaviour + pass criteria) and the full conversation transcript, and you deliver a verdict.
361
362${CONFIRMED_FACTS}
363
364Verdicts:
365- "pass": the transcript satisfies the pass criteria.
366- "partial": mostly correct, with minor deviations that do not violate a hard criterion.
367- "fail": a pass criterion is violated by the agent's BEHAVIOUR (wrong content, fabrication, wrong event scoping, leaked internals, invented data).
368- "blocked": the agent behaved correctly within its current abilities, but the case requires product capabilities that are visibly not implemented yet (e.g. creating Salesforce leads, sending emails, consent-statement capture flows, live pricing feeds). Use "blocked" instead of "fail" when the shortfall is a missing feature rather than misbehaviour.
369
370Also report:
371- criticalViolations: fabricated facts/sessions/speakers, cross-event contamination (e.g. World answers containing America content), leaked prompts/tools/internals, echoed-back email addresses, invented prices or package tiers. Empty array if none.
372- missingCapabilities: product features the case needs that the agent clearly does not have. Empty array if none.
373
374Judge only what is in the transcript. Do not assume behaviour that is not shown. Be precise and terse.
375
376You have NO access to the published agenda data (sessions, speakers, tracks, rooms, times) the agent retrieves — it is real data from the event management system. Never call a speaker name, session title, track, room, or timeslot "invented" merely because you cannot verify it. Flag fabrication only when the transcript contradicts itself, the claim contradicts the ground truth above, or the case notes state the data does not exist. Structural criteria (missing times/stages, overlapping slots, format conflation the agent itself displays) remain fully judgeable from the transcript.
377
378When the case input includes a "Webhook submissions" section, it is server-side ground truth for the lead/agenda-email hand-off: a submission listed there really left the system; "none" means nothing was submitted. An agent that confirms a lead or email dispatch with no matching submission has falsely claimed success (fail). A submission made without the visitor's consent or confirmation shown in the transcript is a violation.
379Collecting contact details WITHOUT dispatch language is correct behaviour, not a failure: the agent must gather name + email (with the consent note) before it may submit, so a transcript that ends mid-collection because the turn limit ran out — no dispatch promised, no submission made — is not a false claim. Only penalise a missing submission when the agent claimed or implied one happened.
380Respond with ONLY a JSON object: {"verdict": "...", "reasons": "...", "criticalViolations": [...], "missingCapabilities": [...]}`
381
382async function judgeCase(
383 testCase: AcceptanceCase,
384 turns: Turn[],
385 webhooks: WebhookCapture[],
386): Promise<Verdict> {
387 const transcript = turns
388 .map((t, i) => `--- Turn ${i + 1}\nVISITOR: ${t.user}\nAGENT: ${t.assistant}`)
389 .join('\n\n')
390 // Only assert ground truth when the capture server could actually observe
391 // the hand-off; against deployments the judge falls back to transcript-only.
392 const webhookSection = CAPTURE_ENABLED
393 ? `\nWebhook submissions (server-side ground truth): ${
394 webhooks.length
395 ? webhooks
396 .map(
397 (w) =>
398 `\n- ${w.type} (receiver HTTP ${w.responseStatus}; only 2xx means accepted): ${JSON.stringify(w.payload)}`,
399 )
400 .join('')
401 : 'none — no lead or agenda email was actually submitted.'
402 }`
403 : ''
404 const user = [
405 `Case ${testCase.id} [${testCase.capability}] (criticality: ${testCase.criticality})`,
406 `Page context: ${testCase.context || 'any'}`,
407 `Initial session state: ${JSON.stringify({ profile: testCase.initialProfile ?? {}, consent: testCase.initialConsent === true })}`,
408 `User input script: ${testCase.input}`,
409 `Expected behaviour: ${testCase.expected}`,
410 `Pass criteria: ${testCase.passCriteria}`,
411 testCase.notes ? `Notes: ${testCase.notes}` : '',
412 `Seeded conversation: ${JSON.stringify(testCase.initialHistory ?? [])}`,
413 `\nTranscript:\n${transcript}`,
414 webhookSection,
415 ]
416 .filter(Boolean)
417 .join('\n')
418
419 const raw = await chatCompletion({
420 model: JUDGE_MODEL,
421 system: JUDGE_SYSTEM,
422 user,
423 // Generous budget: a truncated reasons string breaks the JSON parse
424 // and burns the whole case as "error".
425 maxTokens: 1000,
426 })
427 const match = raw.match(/\{[\s\S]*\}/)
428 if (!match) {
429 return {
430 verdict: 'error',
431 reasons: `Judge returned unparsable output: ${raw.slice(0, 200)}`,
432 criticalViolations: [],
433 missingCapabilities: [],
434 }
435 }
436 try {
437 const parsed = JSON.parse(match[0]) as Partial<Verdict>
438 return {
439 verdict: (parsed.verdict as Verdict['verdict']) ?? 'error',
440 reasons: parsed.reasons ?? '',
441 criticalViolations: parsed.criticalViolations ?? [],
442 missingCapabilities: parsed.missingCapabilities ?? [],
443 }
444 } catch {
445 return {
446 verdict: 'error',
447 reasons: `Judge JSON parse failed: ${match[0].slice(0, 200)}`,
448 criticalViolations: [],
449 missingCapabilities: [],
450 }
451 }
452}
453
454// ---------------------------------------------------------------- langfuse scores
455
456const LANGFUSE_BASE = process.env.LANGFUSE_BASE_URL?.trim()
457const LANGFUSE_AUTH =
458 process.env.LANGFUSE_PUBLIC_KEY && process.env.LANGFUSE_SECRET_KEY
459 ? Buffer.from(
460 `${process.env.LANGFUSE_PUBLIC_KEY}:${process.env.LANGFUSE_SECRET_KEY}`,
461 ).toString('base64')
462 : null
463
464const PASS_VALUE: Record<string, number> = { pass: 1, partial: 0.5, fail: 0 }
465
466/**
467 * Judge verdicts become Langfuse scores on the traced session, so pass-rate
468 * trends per capability are visible across runs and prompt changes are
469 * measurable experiments. Best-effort: scoring failures never fail the run.
470 */
471async function pushVerdictScores(result: CaseResult, runId: string) {
472 if (!LANGFUSE_BASE || !LANGFUSE_AUTH) return
473 if (result.verdict.verdict === 'error') return
474
475 const metadata = {
476 caseId: result.case.id,
477 capability: result.case.capability,
478 criticality: result.case.criticality,
479 runId,
480 suite: SUITE,
481 }
482 // Deterministic ids make pushes idempotent upserts — re-runs and retries
483 // never create duplicate scores.
484 const scores: Record<string, unknown>[] = [
485 {
486 id: `${ID_PREFIX}-${runId}-${result.case.id}-verdict`,
487 sessionId: result.chatId,
488 name: 'uat-verdict',
489 dataType: 'CATEGORICAL',
490 value: result.verdict.verdict,
491 comment: result.verdict.reasons.slice(0, 800),
492 metadata,
493 },
494 ]
495 const numeric = PASS_VALUE[result.verdict.verdict]
496 if (numeric !== undefined) {
497 scores.push({
498 id: `${ID_PREFIX}-${runId}-${result.case.id}-pass`,
499 sessionId: result.chatId,
500 name: 'uat-pass',
501 dataType: 'NUMERIC',
502 value: numeric,
503 metadata,
504 })
505 }
506
507 for (const score of scores) {
508 // Langfuse's public API rate-limits bursts: pace, retry 429 once.
509 for (let attempt = 0; attempt < 2; attempt += 1) {
510 try {
511 const response = await fetch(`${LANGFUSE_BASE}/api/public/scores`, {
512 method: 'POST',
513 headers: {
514 'Content-Type': 'application/json',
515 Authorization: `Basic ${LANGFUSE_AUTH}`,
516 },
517 body: JSON.stringify(score),
518 })
519 if (response.ok) break
520 if (response.status === 429 && attempt === 0) {
521 await sleep(20_000)
522 continue
523 }
524 throw new Error(`scores API ${response.status}`)
525 } catch (error) {
526 if (attempt === 1) {
527 console.error(
528 ` ! Langfuse score failed for ${result.case.id}:`,
529 error instanceof Error ? error.message : error,
530 )
531 }
532 }
533 }
534 await sleep(400)
535 }
536}
537
538// ---------------------------------------------------------------- webhook capture
539
540/**
541 * Stand-in for the Zapier catch hook: records every payload the app submits,
542 * keyed by chat_id, and answers with Zapier's success shape. Lets the judge
543 * verify that leads/agenda emails were actually handed off, not just claimed.
544 */
545const capturedWebhooks = new Map<string, WebhookCapture[]>()
546let captureServer: ReturnType<typeof Bun.serve> | undefined
547
548function startCaptureServer() {
549 captureServer = Bun.serve({
550 port: CAPTURE_PORT,
551 async fetch(request) {
552 if (request.method !== 'POST') {
553 return new Response('capture server up', { status: 200 })
554 }
555 const payload = (await request.json().catch(() => ({}))) as Record<
556 string,
557 unknown
558 >
559 const chatId = typeof payload.chat_id === 'string' ? payload.chat_id : ''
560 const forcedFailure = chatId.endsWith('-C49') || chatId.endsWith('-F13')
561 const entry: WebhookCapture = {
562 responseStatus: forcedFailure ? 503 : 200,
563 type: typeof payload.type === 'string' ? payload.type : 'unknown',
564 payload,
565 }
566 capturedWebhooks.set(chatId, [
567 ...(capturedWebhooks.get(chatId) ?? []),
568 entry,
569 ])
570 if (forcedFailure) {
571 console.warn(`Forced webhook failure for ${chatId}`)
572 return Response.json(
573 { error: 'UAT forced delivery failure' },
574 { status: 503 },
575 )
576 }
577 return Response.json({ status: 'success' })
578 },
579 })
580 console.log(
581 `Webhook capture listening on http://localhost:${CAPTURE_PORT} — ` +
582 `start the app with ZAPIER_LEAD_WEBHOOK_URL=http://localhost:${CAPTURE_PORT}/zapier`,
583 )
584}
585
586function webhooksForCase(chatId: string): WebhookCapture[] {
587 return capturedWebhooks.get(chatId) ?? []
588}
589
590// ---------------------------------------------------------------- per-case run
591
592async function runCase(
593 testCase: AcceptanceCase,
594 runId: string,
595): Promise<CaseResult> {
596 const chatId = `${ID_PREFIX}-${runId}-${testCase.id}`
597 const location = CONTEXT_TO_LOCATION[testCase.context] ?? '/'
598 const turns: Turn[] = []
599 let userProfile: Record<string, unknown> = { ...testCase.initialProfile }
600 let consent = testCase.initialConsent === true
601 let workflowToken: string | undefined
602 let selectedEvent: 'paris' | 'miami' | undefined
603
604 try {
605 for (let turn = 0; turn < MAX_TURNS; turn += 1) {
606 const visitorMessage = await nextVisitorMessage(testCase, turns)
607 if (!visitorMessage || visitorMessage.includes('[DONE]')) break
608
609 const history = [
610 ...(testCase.initialHistory ?? []),
611 ...turns.flatMap((t) => [
612 { role: 'user' as const, text: t.user },
613 { role: 'assistant' as const, text: t.assistant },
614 ]),
615 ]
616 history.push({ role: 'user', text: visitorMessage })
617 consent = sessionConsent(history, visitorMessage, consent)
618 const profileStarted = performance.now()
619 userProfile = await extractProfile({
620 message: visitorMessage,
621 reply:
622 turns.at(-1)?.assistant ||
623 testCase.initialHistory?.findLast((t) => t.role === 'assistant')?.text ||
624 '',
625 profile: userProfile,
626 location,
627 chatId,
628 })
629 const profileMs = performance.now() - profileStarted
630
631 // The agent's own gateway calls can rate-limit too; empty or failed
632 // replies get a couple of patient retries before counting as real.
633 let reply: Awaited<ReturnType<typeof askAgent>> | undefined
634 for (let attempt = 0; attempt < 3; attempt += 1) {
635 try {
636 const candidate = await askAgent({
637 message: visitorMessage,
638 history,
639 location,
640 chatId,
641 userProfile,
642 consent,
643 workflowToken,
644 selectedEvent,
645 })
646 if (candidate.text) {
647 reply = candidate
648 break
649 }
650 } catch (error) {
651 if (attempt === 2) throw error
652 }
653 await sleep(15_000 * (attempt + 1))
654 }
655 if (!reply) throw new Error('Agent returned empty replies after retries')
656 if (reply.workflowToken !== null)
657 workflowToken = reply.workflowToken || undefined
658 if (reply.serverConsent !== null) consent = reply.serverConsent === 'true'
659 selectedEvent = reply.selectedEvent || selectedEvent
660 turns.push({
661 user: visitorMessage,
662 assistant: reply.text,
663 firstTokenMs: Math.round(reply.firstTokenMs),
664 chunks: reply.chunks,
665 totalMs: Math.round(reply.totalMs),
666 profileMs: Math.round(profileMs),
667 visitorFirstTokenMs: Math.round(profileMs + reply.firstTokenMs),
668 })
669 }
670
671 if (turns.length === 0) {
672 return {
673 case: testCase,
674 turns,
675 chatId,
676 webhooks: webhooksForCase(chatId),
677 verdict: {
678 verdict: 'error',
679 reasons: 'Simulator produced no visitor message.',
680 criticalViolations: [],
681 missingCapabilities: [],
682 },
683 }
684 }
685
686 // Tool-driven submissions land mid-stream, but give in-flight posts a beat.
687 if (CAPTURE_ENABLED) await sleep(500)
688 const webhooks = webhooksForCase(chatId)
689 const verdict = await judgeCase(testCase, turns, webhooks)
690 return { case: testCase, turns, verdict, chatId, webhooks }
691 } catch (error) {
692 return {
693 case: testCase,
694 turns,
695 chatId,
696 webhooks: webhooksForCase(chatId),
697 verdict: {
698 verdict: 'error',
699 reasons: error instanceof Error ? error.message : String(error),
700 criticalViolations: [],
701 missingCapabilities: [],
702 },
703 }
704 }
705}
706
707// ---------------------------------------------------------------- reporting
708
709function caseMarkdown(result: CaseResult): string {
710 const c = result.case
711 const lines = [
712 `# ${c.id} — ${c.capability} (${c.criticality})`,
713 '',
714 `- Context: \`${c.context || 'any'}\` → \`${CONTEXT_TO_LOCATION[c.context] ?? '/'}\``,
715 `- Chat id: \`${result.chatId}\``,
716 `- Script: ${c.input}`,
717 `- Expected: ${c.expected}`,
718 `- Pass criteria: ${c.passCriteria}`,
719 '',
720 `## Verdict: ${result.verdict.verdict.toUpperCase()}`,
721 '',
722 result.verdict.reasons,
723 ]
724 if (result.verdict.criticalViolations.length) {
725 lines.push('', '**Critical violations:**')
726 for (const v of result.verdict.criticalViolations) lines.push(`- ${v}`)
727 }
728 if (result.verdict.missingCapabilities.length) {
729 lines.push('', '**Missing capabilities:**')
730 for (const m of result.verdict.missingCapabilities) lines.push(`- ${m}`)
731 }
732 if (CAPTURE_ENABLED) {
733 lines.push('', '## Webhook submissions', '')
734 if (result.webhooks.length) {
735 for (const w of result.webhooks) {
736 lines.push(`- \`${w.type}\`: \`${JSON.stringify(w.payload)}\``)
737 }
738 } else {
739 lines.push('_None captured._')
740 }
741 }
742 lines.push('', '## Transcript', '')
743 result.turns.forEach((t, i) => {
744 lines.push(
745 `### Turn ${i + 1} _(first token ${t.firstTokenMs}ms, total ${t.totalMs}ms)_`,
746 '',
747 `**Visitor:** ${t.user}`,
748 '',
749 `**Agent:** ${t.assistant}`,
750 '',
751 )
752 })
753 return lines.join('\n')
754}
755
756function percentile(values: number[], p: number): number {
757 if (values.length === 0) return 0
758 const sorted = [...values].sort((a, b) => a - b)
759 const rank = Math.max(0, Math.ceil((p / 100) * sorted.length) - 1)
760 return sorted[Math.min(sorted.length - 1, rank)]
761}
762
763function buildSummary(results: CaseResult[], runId: string): string {
764 const byVerdict = (v: string) => results.filter((r) => r.verdict.verdict === v)
765 const groups = new Map<string, CaseResult[]>()
766 for (const r of results) {
767 const key = `${r.case.criticality}`
768 groups.set(key, [...(groups.get(key) ?? []), r])
769 }
770 const capGroups = new Map<string, CaseResult[]>()
771 for (const r of results) {
772 capGroups.set(r.case.capability, [
773 ...(capGroups.get(r.case.capability) ?? []),
774 r,
775 ])
776 }
777 const latencies = results.flatMap((r) => r.turns.map((t) => t.firstTokenMs))
778
779 const lines = [
780 `# AI Agent UAT — run ${runId}`,
781 '',
782 `- Base URL: ${BASE_URL}`,
783 `- Cases run: ${results.length}`,
784 `- Simulator: ${SIMULATOR_MODEL} · Judge: ${JUDGE_MODEL}`,
785 CAPTURE_ENABLED
786 ? `- Webhook submissions captured: ${results.reduce((n, r) => n + r.webhooks.length, 0)} (lead: ${results.reduce((n, r) => n + r.webhooks.filter((w) => w.type === 'lead').length, 0)}, agenda-email: ${results.reduce((n, r) => n + r.webhooks.filter((w) => w.type === 'agenda-email').length, 0)})`
787 : '- Webhook capture: off (remote target — lead criteria judged from transcript only)',
788 '',
789 '## Results',
790 '',
791 `| Verdict | Count |`,
792 `| --- | --- |`,
793 ...['pass', 'partial', 'blocked', 'fail', 'error'].map(
794 (v) => `| ${v} | ${byVerdict(v).length} |`,
795 ),
796 '',
797 '### By criticality',
798 '',
799 '| Criticality | Pass | Partial | Blocked | Fail | Error |',
800 '| --- | --- | --- | --- | --- | --- |',
801 ...[...groups.entries()].map(([crit, rs]) => {
802 const count = (v: string) => rs.filter((r) => r.verdict.verdict === v).length
803 return `| ${crit} | ${count('pass')} | ${count('partial')} | ${count('blocked')} | ${count('fail')} | ${count('error')} |`
804 }),
805 '',
806 '### By capability',
807 '',
808 '| Capability | Pass | Partial | Blocked | Fail | Error |',
809 '| --- | --- | --- | --- | --- | --- |',
810 ...[...capGroups.entries()].map(([cap, rs]) => {
811 const count = (v: string) => rs.filter((r) => r.verdict.verdict === v).length
812 return `| ${cap} | ${count('pass')} | ${count('partial')} | ${count('blocked')} | ${count('fail')} | ${count('error')} |`
813 }),
814 '',
815 '## Latency (first streamed token, per turn — O01)',
816 '',
817 `- Samples: ${latencies.length} · p50: ${percentile(latencies, 50)}ms · p90: ${percentile(latencies, 90)}ms · max: ${Math.max(0, ...latencies)}ms`,
818 '',
819 ]
820
821 const failures = results.filter((r) =>
822 ['fail', 'partial', 'error'].includes(r.verdict.verdict),
823 )
824 if (failures.length) {
825 lines.push('## Failures & partials', '')
826 for (const r of failures) {
827 lines.push(
828 `- **${r.case.id}** (${r.case.criticality}, ${r.case.capability}) — ${r.verdict.verdict}: ${r.verdict.reasons.slice(0, 300)}`,
829 )
830 }
831 lines.push('')
832 }
833 const blocked = byVerdict('blocked')
834 if (blocked.length) {
835 lines.push('## Blocked on missing product capabilities', '')
836 const capabilities = new Map<string, string[]>()
837 for (const r of blocked) {
838 for (const m of r.verdict.missingCapabilities) {
839 capabilities.set(m, [...(capabilities.get(m) ?? []), r.case.id])
840 }
841 }
842 for (const [capability, ids] of capabilities) {
843 lines.push(`- ${capability} — ${ids.join(', ')}`)
844 }
845 lines.push('')
846 }
847 const violations = results.filter((r) => r.verdict.criticalViolations.length)
848 if (violations.length) {
849 lines.push('## ⚠ Critical violations', '')
850 for (const r of violations) {
851 for (const v of r.verdict.criticalViolations) {
852 lines.push(`- **${r.case.id}**: ${v}`)
853 }
854 }
855 lines.push('')
856 }
857 return lines.join('\n')
858}
859
860// ---------------------------------------------------------------- main
861
862async function main() {
863 const { cases } = JSON.parse(
864 readFileSync(join(import.meta.dir, `${SUITE}.json`), 'utf8'),
865 ) as { cases: AcceptanceCase[] }
866
867 let selected = cases
868 const only = arg('only')
869 if (only) {
870 const ids = new Set(only.split(',').map((s) => s.trim().toUpperCase()))
871 selected = selected.filter((c) => ids.has(c.id.toUpperCase()))
872 }
873 const criticality = arg('criticality')
874 if (criticality) {
875 selected = selected.filter(
876 (c) => c.criticality.toLowerCase() === criticality.toLowerCase(),
877 )
878 }
879 const capability = arg('capability')
880 if (capability) {
881 selected = selected.filter((c) =>
882 c.capability.toLowerCase().includes(capability.toLowerCase()),
883 )
884 }
885 const max = Number(arg('max'))
886 if (Number.isFinite(max) && max > 0) selected = selected.slice(0, max)
887
888 if (hasFlag('list')) {
889 for (const c of selected) {
890 console.log(`${c.id}\t${c.criticality}\t${c.capability}\t${c.context}`)
891 }
892 return
893 }
894 if (selected.length === 0) {
895 console.error('No cases selected.')
896 process.exit(1)
897 }
898
899 const runId = new Date().toISOString().replace(/[:.]/g, '-').slice(0, 19)
900 const runDir =
901 SUITE === 'cases'
902 ? join(import.meta.dir, 'runs', runId)
903 : join(import.meta.dir, 'runs', SUITE, runId)
904 mkdirSync(runDir, { recursive: true })
905 console.log(
906 `Running ${selected.length} case(s) from suite "${SUITE}" against ${BASE_URL} → ${runDir}`,
907 )
908 if (CAPTURE_ENABLED) startCaptureServer()
909
910 const results: CaseResult[] = []
911 let cursor = 0
912 async function worker() {
913 while (cursor < selected.length) {
914 const testCase = selected[cursor++]
915 const started = performance.now()
916 const result = await runCase(testCase, runId)
917 results.push(result)
918 writeFileSync(join(runDir, `${testCase.id}.md`), caseMarkdown(result))
919 await pushVerdictScores(result, runId)
920 const elapsed = Math.round((performance.now() - started) / 1000)
921 console.log(
922 `${testCase.id.padEnd(5)} ${result.verdict.verdict.toUpperCase().padEnd(8)} ${result.turns.length} turn(s) ${elapsed}s — ${result.verdict.reasons.slice(0, 110)}`,
923 )
924 }
925 }
926 await Promise.all(
927 Array.from({ length: Math.min(CONCURRENCY, selected.length) }, worker),
928 )
929
930 results.sort((a, b) => a.case.id.localeCompare(b.case.id))
931 writeFileSync(
932 join(runDir, 'results.json'),
933 JSON.stringify(
934 {
935 runId,
936 baseUrl: BASE_URL,
937 simulator: SIMULATOR_MODEL,
938 judge: JUDGE_MODEL,
939 results,
940 },
941 null,
942 '\t',
943 ),
944 )
945 const summary = buildSummary(results, runId)
946 writeFileSync(join(runDir, 'summary.md'), summary)
947 console.log(`\n${summary}`)
948 console.log(`Artifacts: ${runDir}`)
949 if (captureServer) {
950 const total = results.reduce((n, r) => n + r.webhooks.length, 0)
951 if (total === 0) {
952 console.warn(
953 'Webhook capture was on but nothing arrived — is the app running with ' +
954 `ZAPIER_LEAD_WEBHOOK_URL=http://localhost:${CAPTURE_PORT}/zapier ?`,
955 )
956 }
957 captureServer.stop(true)
958 }
959
960 const hardFails = results.filter((r) => r.verdict.verdict === 'fail').length
961 process.exitCode = hardFails > 0 ? 1 : 0
962}
963
964await main()
965