apps/web/src/lib/aiGuards.ts
1/**
2 * Pre-LLM guards for the chat endpoints (SoW §5.4/§5.5):
3 * - sliding-window rate limit per client (configurable via env)
4 * - repeated-identical-message soft block with a friendly reply
5 * - known prompt-injection patterns answered without calling the LLM
6 *
7 * The counters are per serverless instance (in-memory). That still blunts
8 * bursts and loops; put platform-level rate limiting (Vercel WAF) in front
9 * for hard guarantees across instances.
10 */
11
12const WINDOW_MS = Number(process.env.AI_RATE_WINDOW_MS) || 60_000
13const MAX_REQUESTS_PER_WINDOW = Number(process.env.AI_RATE_LIMIT) || 20
14const MAX_IDENTICAL_MESSAGES = 3
15const MAX_TRACKED_CLIENTS = 5_000
16
17type ClientBucket = {
18 timestamps: number[]
19 recentMessages: { normalized: string; at: number; chatId: string }[]
20}
21
22const buckets = new Map<string, ClientBucket>()
23
24function clientKey(request: Request): string {
25 const ip =
26 request.headers.get('x-forwarded-for')?.split(',')[0]?.trim() ||
27 request.headers.get('x-real-ip') ||
28 'unknown'
29 return ip
30}
31
32function getBucket(key: string): ClientBucket {
33 let bucket = buckets.get(key)
34 if (!bucket) {
35 // Opportunistic cap so the map can't grow unbounded on a warm instance.
36 if (buckets.size >= MAX_TRACKED_CLIENTS) {
37 const oldest = buckets.keys().next().value
38 if (oldest) buckets.delete(oldest)
39 }
40 bucket = { timestamps: [], recentMessages: [] }
41 buckets.set(key, bucket)
42 }
43 return bucket
44}
45
46export type GuardResult =
47 | { kind: 'ok' }
48 | { kind: 'rate-limited' }
49 | { kind: 'soft-block'; reply: string }
50 | { kind: 'injection'; reply: string }
51
52const INJECTION_PATTERNS = [
53 /ignore\s+(all\s+|any\s+)?(previous|prior|above|earlier)\s+(instructions?|prompts?|messages?)/i,
54 /disregard\s+(your|the|all)\s+(system\s+)?(prompt|instructions?)/i,
55 /you\s+are\s+now\s+(a|an|in|no\s+longer)/i,
56 /\bDAN\s+mode\b/i,
57 /\bjailbreak\b/i,
58 /pretend\s+(you\s+have\s+no|there\s+are\s+no)\s+(rules|restrictions|guardrails)/i,
59 /(override|bypass)\s+(your\s+)?(safety|guardrails?|restrictions?|instructions?)/i,
60]
61
62const INJECTION_REPLY =
63 "I'm sorry, I wasn't able to process that. Please ask me about Unleash events and I'll do my best to help."
64
65const DUPLICATE_REPLY =
66 "You've sent that same message a few times — I gave it my best answer above. Try rephrasing, or ask me something else about Unleash events, content, or partnerships."
67
68/** Short confirmations and thanks are normal workflow turns, not repeated queries. */
69function ordinaryWorkflowReply(message: string): boolean {
70 return /^(?:yes(?: please)?|no(?: thanks)?|thanks(?:,? that.s all)?|thank you|ok(?:ay)?|sure|help(?: me)?(?: please)?|please help(?: me)?)[.!\s]*$/i.test(
71 message.trim(),
72 )
73}
74
75/** Run all pre-LLM guards. Call once per chat message before any model work. */
76export function checkAiGuards(
77 request: Request,
78 chatId: string,
79 message: string,
80 history?: { role: 'user' | 'assistant'; text: string }[],
81): GuardResult {
82 // 1. Injection patterns: answered without spending an LLM call.
83 if (INJECTION_PATTERNS.some((pattern) => pattern.test(message))) {
84 return { kind: 'injection', reply: INJECTION_REPLY }
85 }
86
87 const now = Date.now()
88 const bucket = getBucket(clientKey(request))
89
90 // 2. Sliding-window rate limit.
91 bucket.timestamps = bucket.timestamps.filter((t) => now - t < WINDOW_MS)
92 if (bucket.timestamps.length >= MAX_REQUESTS_PER_WINDOW) {
93 return { kind: 'rate-limited' }
94 }
95 bucket.timestamps.push(now)
96
97 // 3. Repeated-identical-message soft block.
98 const normalized = message.trim().toLowerCase().replace(/\s+/g, ' ')
99 bucket.recentMessages = bucket.recentMessages.filter(
100 (entry) => now - entry.at < WINDOW_MS,
101 )
102 const normalize = (text: string) =>
103 text.trim().toLowerCase().replace(/\s+/g, ' ')
104 const identical = history
105 ? history.filter(
106 (turn, i) =>
107 turn.role === 'user' &&
108 normalize(turn.text) === normalized &&
109 history[i + 1]?.role === 'assistant' &&
110 history[i + 1].text.trim(),
111 ).length
112 : bucket.recentMessages.filter(
113 (entry) => entry.chatId === chatId && entry.normalized === normalized,
114 ).length
115 bucket.recentMessages.push({ normalized, at: now, chatId })
116 if (
117 !ordinaryWorkflowReply(message) &&
118 identical >= MAX_IDENTICAL_MESSAGES - 1
119 ) {
120 return { kind: 'soft-block', reply: DUPLICATE_REPLY }
121 }
122
123 return { kind: 'ok' }
124}
125
126/** Plain-text response in the same shape the client's stream reader expects. */
127export function guardTextResponse(reply: string): Response {
128 return new Response(reply, {
129 status: 200,
130 headers: {
131 'Content-Type': 'text/plain; charset=utf-8',
132 'X-AI-Profile': 'skip',
133 },
134 })
135}
136
137/** Failed/empty responses are retryable; only delivered replies count as duplicates. */
138export function checkRepeatedExchange(
139 history: { role: 'user' | 'assistant'; text: string }[],
140 message: string,
141): string | undefined {
142 if (ordinaryWorkflowReply(message)) return undefined
143 const normalize = (text: string) =>
144 text.trim().toLowerCase().replace(/\s+/g, ' ')
145 const count = history.filter(
146 (turn, i) =>
147 turn.role === 'user' &&
148 normalize(turn.text) === normalize(message) &&
149 history[i + 1]?.role === 'assistant' &&
150 history[i + 1].text.trim(),
151 ).length
152 return count >= MAX_IDENTICAL_MESSAGES - 1 ? DUPLICATE_REPLY : undefined
153}
154