apps/web/src/lib/aiGuards.ts

1/**
2 * Pre-LLM guards for the chat endpoints (SoW §5.4/§5.5):
3 * - sliding-window rate limit per client (configurable via env)
4 * - repeated-identical-message soft block with a friendly reply
5 * - known prompt-injection patterns answered without calling the LLM
6 *
7 * The counters are per serverless instance (in-memory). That still blunts
8 * bursts and loops; put platform-level rate limiting (Vercel WAF) in front
9 * for hard guarantees across instances.
10 */
11
12const WINDOW_MS = Number(process.env.AI_RATE_WINDOW_MS) || 60_000
13const MAX_REQUESTS_PER_WINDOW = Number(process.env.AI_RATE_LIMIT) || 20
14const MAX_IDENTICAL_MESSAGES = 3
15const MAX_TRACKED_CLIENTS = 5_000
16
17type ClientBucket = {
18	timestamps: number[]
19	recentMessages: { normalized: string; at: number; chatId: string }[]
20}
21
22const buckets = new Map<string, ClientBucket>()
23
24function clientKey(request: Request): string {
25	const ip =
26		request.headers.get('x-forwarded-for')?.split(',')[0]?.trim() ||
27		request.headers.get('x-real-ip') ||
28		'unknown'
29	return ip
30}
31
32function getBucket(key: string): ClientBucket {
33	let bucket = buckets.get(key)
34	if (!bucket) {
35		// Opportunistic cap so the map can't grow unbounded on a warm instance.
36		if (buckets.size >= MAX_TRACKED_CLIENTS) {
37			const oldest = buckets.keys().next().value
38			if (oldest) buckets.delete(oldest)
39		}
40		bucket = { timestamps: [], recentMessages: [] }
41		buckets.set(key, bucket)
42	}
43	return bucket
44}
45
46export type GuardResult =
47	| { kind: 'ok' }
48	| { kind: 'rate-limited' }
49	| { kind: 'soft-block'; reply: string }
50	| { kind: 'injection'; reply: string }
51
52const INJECTION_PATTERNS = [
53	/ignore\s+(all\s+|any\s+)?(previous|prior|above|earlier)\s+(instructions?|prompts?|messages?)/i,
54	/disregard\s+(your|the|all)\s+(system\s+)?(prompt|instructions?)/i,
55	/you\s+are\s+now\s+(a|an|in|no\s+longer)/i,
56	/\bDAN\s+mode\b/i,
57	/\bjailbreak\b/i,
58	/pretend\s+(you\s+have\s+no|there\s+are\s+no)\s+(rules|restrictions|guardrails)/i,
59	/(override|bypass)\s+(your\s+)?(safety|guardrails?|restrictions?|instructions?)/i,
60]
61
62const INJECTION_REPLY =
63	"I'm sorry, I wasn't able to process that. Please ask me about Unleash events and I'll do my best to help."
64
65const DUPLICATE_REPLY =
66	"You've sent that same message a few times — I gave it my best answer above. Try rephrasing, or ask me something else about Unleash events, content, or partnerships."
67
68/** Short confirmations and thanks are normal workflow turns, not repeated queries. */
69function ordinaryWorkflowReply(message: string): boolean {
70	return /^(?:yes(?: please)?|no(?: thanks)?|thanks(?:,? that.s all)?|thank you|ok(?:ay)?|sure|help(?: me)?(?: please)?|please help(?: me)?)[.!\s]*$/i.test(
71		message.trim(),
72	)
73}
74
75/** Run all pre-LLM guards. Call once per chat message before any model work. */
76export function checkAiGuards(
77	request: Request,
78	chatId: string,
79	message: string,
80	history?: { role: 'user' | 'assistant'; text: string }[],
81): GuardResult {
82	// 1. Injection patterns: answered without spending an LLM call.
83	if (INJECTION_PATTERNS.some((pattern) => pattern.test(message))) {
84		return { kind: 'injection', reply: INJECTION_REPLY }
85	}
86
87	const now = Date.now()
88	const bucket = getBucket(clientKey(request))
89
90	// 2. Sliding-window rate limit.
91	bucket.timestamps = bucket.timestamps.filter((t) => now - t < WINDOW_MS)
92	if (bucket.timestamps.length >= MAX_REQUESTS_PER_WINDOW) {
93		return { kind: 'rate-limited' }
94	}
95	bucket.timestamps.push(now)
96
97	// 3. Repeated-identical-message soft block.
98	const normalized = message.trim().toLowerCase().replace(/\s+/g, ' ')
99	bucket.recentMessages = bucket.recentMessages.filter(
100		(entry) => now - entry.at < WINDOW_MS,
101	)
102	const normalize = (text: string) =>
103		text.trim().toLowerCase().replace(/\s+/g, ' ')
104	const identical = history
105		? history.filter(
106				(turn, i) =>
107					turn.role === 'user' &&
108					normalize(turn.text) === normalized &&
109					history[i + 1]?.role === 'assistant' &&
110					history[i + 1].text.trim(),
111			).length
112		: bucket.recentMessages.filter(
113				(entry) => entry.chatId === chatId && entry.normalized === normalized,
114			).length
115	bucket.recentMessages.push({ normalized, at: now, chatId })
116	if (
117		!ordinaryWorkflowReply(message) &&
118		identical >= MAX_IDENTICAL_MESSAGES - 1
119	) {
120		return { kind: 'soft-block', reply: DUPLICATE_REPLY }
121	}
122
123	return { kind: 'ok' }
124}
125
126/** Plain-text response in the same shape the client's stream reader expects. */
127export function guardTextResponse(reply: string): Response {
128	return new Response(reply, {
129		status: 200,
130		headers: {
131			'Content-Type': 'text/plain; charset=utf-8',
132			'X-AI-Profile': 'skip',
133		},
134	})
135}
136
137/** Failed/empty responses are retryable; only delivered replies count as duplicates. */
138export function checkRepeatedExchange(
139	history: { role: 'user' | 'assistant'; text: string }[],
140	message: string,
141): string | undefined {
142	if (ordinaryWorkflowReply(message)) return undefined
143	const normalize = (text: string) =>
144		text.trim().toLowerCase().replace(/\s+/g, ' ')
145	const count = history.filter(
146		(turn, i) =>
147			turn.role === 'user' &&
148			normalize(turn.text) === normalize(message) &&
149			history[i + 1]?.role === 'assistant' &&
150			history[i + 1].text.trim(),
151	).length
152	return count >= MAX_IDENTICAL_MESSAGES - 1 ? DUPLICATE_REPLY : undefined
153}
154