Repository navigation
Expand file tree
/
Copy pathqa_agent.py
More file actions
executable file
·401 lines (324 loc) · 13.8 KB
/
Copy pathqa_agent.py
File metadata and controls
executable file
·401 lines (324 loc) · 13.8 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
#!/usr/bin/env python3
"""
Question Answering Agent with DeepEval Quality Controls
This example demonstrates:
1. Using agent-control SDK with @control() decorator
2. DeepEval GEval evaluators for quality enforcement
3. Handling ControlViolationError gracefully
The agent is protected by DeepEval-based controls that check:
- Response coherence (logical consistency)
- Answer relevance (stays on topic)
- Factual correctness (when expected outputs available)
Usage:
# Setup first (creates controls on server)
python setup_controls.py
# Then run the agent
python qa_agent.py
Requirements:
- agent-control server running
- OPENAI_API_KEY set (for DeepEval)
- Controls configured via setup_controls.py
"""
import asyncio
import logging
import os
import sys
from uuid import UUID
import agent_control
from agent_control import ControlViolationError, control
# Enable DEBUG logging for agent_control to see what's happening
logging.basicConfig(level=logging.DEBUG, format='%(name)s - %(levelname)s - %(message)s')
logging.getLogger('agent_control').setLevel(logging.DEBUG)
# =============================================================================
# SDK INITIALIZATION
# =============================================================================
agent_control.init(
agent_name="qa-agent-with-deepeval",
agent_description="Q&A Agent with DeepEval",
agent_version="1.0.0",
)
# Debug: Check if controls were loaded
controls = agent_control.get_server_controls()
print(f"DEBUG: Loaded {len(controls) if controls else 0} controls from server")
if controls:
for c in controls:
ctrl_def = c.get('control', {})
print(f" - {c['name']}:")
print(f" enabled: {ctrl_def.get('enabled', False)}")
print(f" execution: {ctrl_def.get('execution', 'NOT SET')}")
print(f" scope: {ctrl_def.get('scope', {})}")
# =============================================================================
# MOCK LLM (Simulates various quality scenarios)
# =============================================================================
class MockQASystem:
"""
Simulates a Q&A system with various response quality scenarios.
This mock helps demonstrate how DeepEval controls catch quality issues:
- Coherent responses pass
- Incoherent responses are blocked
- Irrelevant responses are blocked
"""
GOOD_RESPONSES = {
"python": (
"Python is a high-level, interpreted programming language known for its "
"simplicity and readability. It was created by Guido van Rossum and first "
"released in 1991. Python supports multiple programming paradigms including "
"procedural, object-oriented, and functional programming."
),
"capital": (
"Paris is the capital and largest city of France. It is located in the "
"north-central part of the country along the River Seine. Paris has been "
"a major center of culture, art, and politics for centuries."
),
"photosynthesis": (
"Photosynthesis is the process by which plants convert light energy into "
"chemical energy. Using chlorophyll, plants absorb sunlight and combine "
"carbon dioxide from the air with water from the soil to produce glucose "
"and oxygen. This process is essential for life on Earth."
),
"gravity": (
"Gravity is a fundamental force of nature that attracts objects with mass "
"toward each other. On Earth, gravity gives objects weight and causes them "
"to fall toward the ground. The force of gravity was described by Newton's "
"laws and later refined by Einstein's theory of general relativity."
),
}
# Incoherent responses (logical inconsistencies, contradictions)
INCOHERENT_RESPONSES = {
"trigger_incoherent": (
"Python is a snake. Also Python is not a snake. "
"It's both simultaneously. Yesterday is tomorrow. "
"The sky is made of cheese but also not cheese. "
"Numbers are letters and letters are numbers."
),
}
# Irrelevant responses (don't answer the question)
IRRELEVANT_RESPONSES = {
"trigger_irrelevant": (
"Bananas are yellow fruits that grow on trees. "
"The weather today is sunny. I like pizza. "
"Dogs are mammals. The year has 12 months."
),
}
@classmethod
def answer_question(cls, question: str) -> str:
"""Generate an answer to the question."""
question_lower = question.lower()
# Check for test triggers
if "trigger_incoherent" in question_lower or "incoherent" in question_lower:
return cls.INCOHERENT_RESPONSES["trigger_incoherent"]
if "trigger_irrelevant" in question_lower or "irrelevant" in question_lower:
return cls.IRRELEVANT_RESPONSES["trigger_irrelevant"]
# Match question to good responses
if "python" in question_lower:
return cls.GOOD_RESPONSES["python"]
elif "capital" in question_lower and "france" in question_lower:
return cls.GOOD_RESPONSES["capital"]
elif "photosynthesis" in question_lower:
return cls.GOOD_RESPONSES["photosynthesis"]
elif "gravity" in question_lower:
return cls.GOOD_RESPONSES["gravity"]
else:
# Default educational response
return (
f"That's an interesting question about '{question}'. "
"Based on general knowledge, I can provide information on this topic. "
"Would you like me to explain in more detail?"
)
# =============================================================================
# PROTECTED AGENT FUNCTION
# =============================================================================
@control()
async def answer_question(question: str) -> str:
"""
Answer a question with quality controls.
The @control() decorator:
- Checks 'pre' controls before generating (validates input)
- Checks 'post' controls after generating (validates output quality)
DeepEval controls check:
- Coherence: Is the response logically consistent?
- Relevance: Does it address the question?
- Correctness: Is it factually accurate? (if enabled)
If a control fails, ControlViolationError is raised.
"""
print(f"DEBUG: answer_question called with question: {question[:50]}...")
response = MockQASystem.answer_question(question)
print(f"DEBUG: Generated response: {response[:50]}...")
return response
# =============================================================================
# Q&A AGENT CLASS
# =============================================================================
class QAAgent:
"""
Question answering agent with DeepEval quality controls.
Demonstrates graceful error handling when quality controls fail.
"""
def __init__(self):
self.conversation_history: list[dict[str, str]] = []
async def ask(self, question: str) -> str:
"""
Ask a question and get an answer.
Handles ControlViolationError gracefully by returning
a helpful message instead of exposing internal errors.
"""
self.conversation_history.append({"role": "user", "content": question})
# Debug: Check if agent is still initialized
import agent_control
current_agent = agent_control._current_agent if hasattr(agent_control, '_current_agent') else None
print(f"DEBUG: Current agent before call: {current_agent.agent_name if current_agent else 'NONE'}")
try:
# Get answer - protected by DeepEval controls
print("DEBUG: About to call answer_question (with @control decorator)")
answer = await answer_question(question)
print("DEBUG: answer_question returned successfully")
self.conversation_history.append({"role": "assistant", "content": answer})
return answer
except ControlViolationError as e:
print(f"DEBUG: ControlViolationError caught: {e}")
# Control triggered - return helpful feedback
fallback = (
f"I apologize, but my response didn't meet quality standards. "
f"({e.control_name})\n\n"
f"Could you rephrase your question or ask something else?"
)
self.conversation_history.append({"role": "assistant", "content": fallback})
print(f"\n⚠️ Quality control triggered: {e.control_name}")
print(f" Reason: {e.message}")
return fallback
except Exception as e:
print(f"DEBUG: Unexpected exception: {type(e).__name__}: {e}")
raise
# =============================================================================
# INTERACTIVE MODE
# =============================================================================
def print_header():
"""Print the demo header."""
print()
print("=" * 70)
print(" Q&A Agent with DeepEval Quality Controls")
print("=" * 70)
print()
print("This agent uses DeepEval GEval to enforce response quality:")
print(" ✓ Coherence - Responses must be logically consistent")
print(" ✓ Relevance - Answers must address the question")
print(" ○ Correctness - Factual accuracy (disabled by default)")
print()
print("Commands:")
print(" /test-good Test with high-quality questions")
print(" /test-bad Test quality control triggers")
print(" /help Show this help")
print(" /quit Exit")
print()
print("Or just type a question!")
print("-" * 70)
print()
def print_help():
"""Print help information."""
print()
print("Available Commands:")
print(" /test-good Test with questions that produce quality answers")
print(" /test-bad Test questions that trigger quality controls")
print(" /help Show this help message")
print(" /quit or /exit Exit the program")
print()
print("Or ask any question and see how DeepEval evaluates quality!")
print()
async def run_good_tests(agent: QAAgent):
"""Run tests with good quality responses."""
print("\n" + "=" * 70)
print("Testing Good Quality Responses")
print("=" * 70)
print("\nThese should pass all quality controls.\n")
test_questions = [
"What is Python?",
"What is the capital of France?",
"How does photosynthesis work?",
"What is gravity?",
]
for question in test_questions:
print(f"Q: {question}")
answer = await agent.ask(question)
print(f"A: {answer[:150]}...")
print()
async def run_bad_tests(agent: QAAgent):
"""Run tests that should trigger quality controls."""
print("\n" + "=" * 70)
print("Testing Quality Control Triggers")
print("=" * 70)
print("\nThese should trigger DeepEval controls.\n")
test_questions = [
"Test trigger_incoherent response please", # Should fail coherence
"Tell me about something trigger_irrelevant", # Should fail relevance
]
for question in test_questions:
print(f"Q: {question}")
answer = await agent.ask(question)
print(f"A: {answer}")
print()
async def run_interactive(agent: QAAgent):
"""Run interactive mode."""
print_header()
while True:
try:
user_input = input("You: ").strip()
except (KeyboardInterrupt, EOFError):
print("\nGoodbye!")
break
if not user_input:
continue
# Handle commands
if user_input.startswith("/"):
command = user_input.lower().split()[0]
if command in ("/quit", "/exit"):
print("Goodbye!")
break
elif command == "/help":
print_help()
elif command == "/test-good":
await run_good_tests(agent)
elif command == "/test-bad":
await run_bad_tests(agent)
else:
print(f"Unknown command: {command}")
print("Type /help for available commands")
else:
# Regular question
answer = await agent.ask(user_input)
print(f"\nAgent: {answer}\n")
# =============================================================================
# MAIN
# =============================================================================
async def main():
"""Run the Q&A agent."""
# Check for OPENAI_API_KEY
if not os.getenv("OPENAI_API_KEY"):
print("\n⚠️ Warning: OPENAI_API_KEY not set!")
print(" DeepEval requires OpenAI API access for GEval.")
print(" Set it with: export OPENAI_API_KEY='your-key'")
print()
response = input("Continue anyway? (y/N): ").strip().lower()
if response != "y":
print("Exiting. Set OPENAI_API_KEY and try again.")
sys.exit(1)
# Check server connection
server_url = os.getenv("AGENT_CONTROL_URL", "http://localhost:8000")
print(f"\nConnecting to agent-control server at {server_url}...")
import httpx
try:
async with httpx.AsyncClient() as client:
resp = await client.get(f"{server_url}/health", timeout=5.0)
resp.raise_for_status()
print("✓ Connected to server")
except Exception as e:
print(f"\n❌ Cannot connect to server: {e}")
print(" Make sure the agent-control server is running.")
print(" Run setup_controls.py first to configure the agent.")
sys.exit(1)
# Create and run agent
agent = QAAgent()
await run_interactive(agent)
if __name__ == "__main__":
try:
asyncio.run(main())
except KeyboardInterrupt:
print("\n\nInterrupted. Goodbye!")