Repository navigation
Expand file tree
/
Copy pathtest_evaluation_creation.py
More file actions
194 lines (161 loc) Β· 6.6 KB
/
Copy pathtest_evaluation_creation.py
File metadata and controls
194 lines (161 loc) Β· 6.6 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
#!/usr/bin/env python3
"""
Test script to verify evaluation creation via the dataset_loader agent
"""
import json
import requests
import time
# Configuration
AGENTUITY_BASE_URL = "https://dev-6iaifxmoj.agentuity.run "
DATASET_LOADER_AGENT_ID = "agent_abcf9ad4245d2d89aed9eb38aef21fd6"
RESULTS_API_AGENT_ID = "agent_b46de37831f94d01b06b2ccfd183efa0"
def test_evaluation_creation():
"""Test creating a new evaluation"""
print("π§ͺ Testing Evaluation Creation")
print(f"π‘ Dataset Loader Endpoint: {AGENTUITY_BASE_URL}/{DATASET_LOADER_AGENT_ID}")
# Small test dataset with just 2 examples
test_dataset = [
{
"query": "What is Superman's main weakness?",
"response": "Kryptonite is Superman's main weakness."
},
{
"query": "Can Batman fly?",
"response": "No, Batman cannot fly naturally, but he uses various gadgets and vehicles."
}
]
# Test payload matching the frontend configuration
test_payload = {
"evaluation_id": f"test_eval_{int(time.time())}",
"dataset_json": test_dataset, # Use inline dataset instead of file
"format": "query_response_pairs",
"prompt_template": {
"template": "You are an expert assistant. Answer the following question: {{query}}",
"variables": ["query"]
},
"model_config": {
"model_name": "claude-3-5-sonnet-latest",
"max_tokens": 100,
"temperature": 0.1
},
"evaluation_settings": {
"similarity_threshold": 80,
"judge_model": "claude-3-5-haiku-latest"
}
}
try:
print(f"π€ Sending request to dataset_loader agent...")
print(f"π Evaluation ID: {test_payload['evaluation_id']}")
response = requests.post(
f"{AGENTUITY_BASE_URL}/{DATASET_LOADER_AGENT_ID}",
headers={"Content-Type": "application/json"},
json=test_payload,
timeout=30
)
print(f"π Response Status: {response.status_code}")
if response.status_code == 200:
result = response.json()
print(f"β
Success! Response: {json.dumps(result, indent=2)}")
# Test if we can retrieve the evaluation from results API
print(f"\nπ Testing Results API...")
test_results_api(test_payload['evaluation_id'])
else:
print(f"β Failed with status {response.status_code}")
try:
error_data = response.json()
print(f"Error details: {json.dumps(error_data, indent=2)}")
except:
print(f"Error text: {response.text}")
except requests.exceptions.RequestException as e:
print(f"β Request failed: {e}")
except Exception as e:
print(f"β Unexpected error: {e}")
def test_results_api(evaluation_id):
"""Test retrieving evaluation from results API"""
try:
# Test listing evaluations
list_payload = {"operation": "list_evaluations"}
response = requests.post(
f"{AGENTUITY_BASE_URL}/{RESULTS_API_AGENT_ID}",
headers={"Content-Type": "application/json"},
json=list_payload,
timeout=10
)
if response.status_code == 200:
result = response.json()
evaluations = result.get("evaluations", [])
print(f"π Found {len(evaluations)} evaluations in results API")
# Check if our evaluation is in the list
found = any(eval_item["id"] == evaluation_id for eval_item in evaluations)
if found:
print(f"β
Evaluation {evaluation_id} found in results API!")
else:
print(f"β οΈ Evaluation {evaluation_id} not yet visible in results API (may take a moment)")
else:
print(f"β Results API failed with status {response.status_code}")
except Exception as e:
print(f"β Results API test failed: {e}")
def test_with_inline_dataset():
"""Test creating evaluation with inline dataset"""
print("\nπ§ͺ Testing Evaluation Creation with Inline Dataset")
# Small inline dataset for testing
inline_dataset = [
{
"query": "What is Superman's main weakness?",
"response": "Kryptonite is Superman's main weakness."
},
{
"query": "Can Batman fly?",
"response": "No, Batman cannot fly naturally, but he uses various gadgets and vehicles."
}
]
test_payload = {
"evaluation_id": f"inline_test_eval_{int(time.time())}",
"dataset_json": inline_dataset,
"format": "query_response_pairs",
"prompt_template": {
"template": "Answer this superhero question: {{query}}",
"variables": ["query"]
},
"model_config": {
"model_name": "claude-3-5-haiku-latest",
"max_tokens": 50,
"temperature": 0.0
},
"evaluation_settings": {
"similarity_threshold": 75,
"judge_model": "claude-3-5-haiku-latest"
}
}
try:
print(f"π€ Sending inline dataset request...")
print(f"π Evaluation ID: {test_payload['evaluation_id']}")
print(f"π Dataset size: {len(inline_dataset)} items")
response = requests.post(
f"{AGENTUITY_BASE_URL}/{DATASET_LOADER_AGENT_ID}",
headers={"Content-Type": "application/json"},
json=test_payload,
timeout=30
)
print(f"π Response Status: {response.status_code}")
if response.status_code == 200:
result = response.json()
print(f"β
Inline dataset test successful!")
print(f"Response: {json.dumps(result, indent=2)}")
else:
print(f"β Inline dataset test failed with status {response.status_code}")
try:
error_data = response.json()
print(f"Error details: {json.dumps(error_data, indent=2)}")
except:
print(f"Error text: {response.text}")
except Exception as e:
print(f"β Inline dataset test failed: {e}")
if __name__ == "__main__":
print("π AI Evaluation System - API Test")
print("=" * 50)
# Test with existing dataset
test_evaluation_creation()
# Test with inline dataset
test_with_inline_dataset()
print("\n⨠Testing complete!")