Skip to content

Commit 6e42d29

Browse files
committed
fix(examples): remove fake evaluation oracle leakage
1 parent b91bf05 commit 6e42d29

8 files changed

Lines changed: 835 additions & 410 deletions

File tree

Lines changed: 135 additions & 50 deletions
Original file line numberDiff line numberDiff line change
@@ -1,59 +1,144 @@
11
{
2-
"evalset_id": "eval_optimize_loop_train_v1",
3-
"split": "train",
4-
"description": "Training cases used for failure attribution and fake prompt optimization.",
5-
"cases": [
2+
"eval_set_id": "eval_optimize_loop_train_v1",
3+
"name": "Evaluation optimization loop train set",
4+
"description": "Training cases used for failure attribution and prompt optimization.",
5+
"eval_cases": [
66
{
7-
"id": "train_json_refund",
8-
"input": "Return strict JSON for a refund escalation. Use keys intent and priority.",
9-
"expectation": {
10-
"type": "json",
11-
"required_keys": [
12-
"intent",
13-
"priority"
14-
],
15-
"expected_values": {
16-
"intent": "refund",
17-
"priority": "high"
18-
},
19-
"expected_failure_category": "format_violation"
20-
},
21-
"tags": [
22-
"json",
23-
"format"
24-
]
7+
"eval_id": "train_json_refund",
8+
"conversation": [
9+
{
10+
"invocation_id": "train_json_refund_turn_1",
11+
"user_content": {
12+
"parts": [
13+
{
14+
"text": "Return strict JSON with intent=refund and priority=high."
15+
}
16+
],
17+
"role": "user"
18+
},
19+
"final_response": {
20+
"parts": [
21+
{
22+
"text": "{\"intent\":\"refund\",\"priority\":\"high\"}"
23+
}
24+
],
25+
"role": "model"
26+
}
27+
}
28+
],
29+
"session_input": {
30+
"app_name": "eval_optimize_loop",
31+
"user_id": "trainer",
32+
"state": {
33+
"eval_optimize_expectation": {
34+
"type": "json",
35+
"required_keys": [
36+
"intent",
37+
"priority"
38+
],
39+
"expected_values": {
40+
"intent": "refund",
41+
"priority": "high"
42+
},
43+
"expected_failure_category": "format_violation"
44+
},
45+
"eval_optimize_tags": [
46+
"json",
47+
"format"
48+
]
49+
}
50+
}
2551
},
2652
{
27-
"id": "train_exact_order_status",
28-
"input": "Answer exactly READY when the order can ship.",
29-
"expectation": {
30-
"type": "exact",
31-
"expected": "READY",
32-
"expected_failure_category": "final_response_mismatch"
33-
},
34-
"tags": [
35-
"exact",
36-
"format"
37-
]
53+
"eval_id": "train_exact_order_status",
54+
"conversation": [
55+
{
56+
"invocation_id": "train_json_order_turn_1",
57+
"user_content": {
58+
"parts": [
59+
{
60+
"text": "Return strict JSON with status=READY and next_step=ship."
61+
}
62+
],
63+
"role": "user"
64+
},
65+
"final_response": {
66+
"parts": [
67+
{
68+
"text": "{\"next_step\":\"ship\",\"status\":\"READY\"}"
69+
}
70+
],
71+
"role": "model"
72+
}
73+
}
74+
],
75+
"session_input": {
76+
"app_name": "eval_optimize_loop",
77+
"user_id": "trainer",
78+
"state": {
79+
"eval_optimize_expectation": {
80+
"type": "json",
81+
"required_keys": [
82+
"status",
83+
"next_step"
84+
],
85+
"expected_values": {
86+
"status": "READY",
87+
"next_step": "ship"
88+
},
89+
"expected_failure_category": "format_violation"
90+
},
91+
"eval_optimize_tags": [
92+
"json",
93+
"format"
94+
]
95+
}
96+
}
3897
},
3998
{
40-
"id": "train_rubric_retry_summary",
41-
"input": "In 80 chars or less, mention latency and retries.",
42-
"expectation": {
43-
"type": "rubric",
44-
"must_include": [
45-
"latency",
46-
"retries"
47-
],
48-
"forbidden": [
49-
"database"
50-
],
51-
"max_chars": 80
52-
},
53-
"tags": [
54-
"rubric",
55-
"no_effect"
56-
]
99+
"eval_id": "train_rubric_retry_summary",
100+
"conversation": [
101+
{
102+
"invocation_id": "train_latency_retries_turn_1",
103+
"user_content": {
104+
"parts": [
105+
{
106+
"text": "Explain latency and retries naturally in under 80 characters."
107+
}
108+
],
109+
"role": "user"
110+
},
111+
"final_response": {
112+
"parts": [
113+
{
114+
"text": "Latency can trigger retries."
115+
}
116+
],
117+
"role": "model"
118+
}
119+
}
120+
],
121+
"session_input": {
122+
"app_name": "eval_optimize_loop",
123+
"user_id": "trainer",
124+
"state": {
125+
"eval_optimize_expectation": {
126+
"type": "rubric",
127+
"must_include": [
128+
"latency",
129+
"retries"
130+
],
131+
"forbidden": [
132+
"database"
133+
],
134+
"max_chars": 80
135+
},
136+
"eval_optimize_tags": [
137+
"rubric",
138+
"no_effect"
139+
]
140+
}
141+
}
57142
}
58143
]
59144
}
Lines changed: 131 additions & 53 deletions
Original file line numberDiff line numberDiff line change
@@ -1,63 +1,141 @@
11
{
2-
"evalset_id": "eval_optimize_loop_validation_v1",
3-
"split": "validation",
2+
"eval_set_id": "eval_optimize_loop_validation_v1",
3+
"name": "Evaluation optimization loop validation set",
44
"description": "Validation cases that separate safe optimization from train-set overfitting.",
5-
"cases": [
5+
"eval_cases": [
66
{
7-
"id": "val_json_invoice",
8-
"input": "Return strict JSON for an invoice approval. Use keys status and next_step.",
9-
"expectation": {
10-
"type": "json",
11-
"required_keys": [
12-
"status",
13-
"next_step"
14-
],
15-
"expected_values": {
16-
"next_step": "email_customer",
17-
"status": "approved"
18-
},
19-
"expected_failure_category": "format_violation"
20-
},
21-
"tags": [
22-
"json",
23-
"safe_improves"
24-
]
7+
"eval_id": "val_json_invoice",
8+
"conversation": [
9+
{
10+
"invocation_id": "val_json_approval_turn_1",
11+
"user_content": {
12+
"parts": [
13+
{
14+
"text": "Return strict JSON with status=approved and next_step=email_customer."
15+
}
16+
],
17+
"role": "user"
18+
},
19+
"final_response": {
20+
"parts": [
21+
{
22+
"text": "{\"next_step\":\"email_customer\",\"status\":\"approved\"}"
23+
}
24+
],
25+
"role": "model"
26+
}
27+
}
28+
],
29+
"session_input": {
30+
"app_name": "eval_optimize_loop",
31+
"user_id": "validator",
32+
"state": {
33+
"eval_optimize_expectation": {
34+
"type": "json",
35+
"required_keys": [
36+
"status",
37+
"next_step"
38+
],
39+
"expected_values": {
40+
"next_step": "email_customer",
41+
"status": "approved"
42+
},
43+
"expected_failure_category": "format_violation"
44+
},
45+
"eval_optimize_tags": [
46+
"json",
47+
"safe_improves"
48+
]
49+
}
50+
}
2551
},
2652
{
27-
"id": "val_explain_cache",
28-
"input": "Explain cache invalidation for a junior engineer in natural language.",
29-
"expectation": {
30-
"type": "rubric",
31-
"must_include": [
32-
"cache",
33-
"stale data"
34-
],
35-
"forbidden": [
36-
"{",
37-
"}",
38-
"json"
39-
],
40-
"max_chars": 180,
41-
"expected_failure_category": "format_violation"
42-
},
43-
"tags": [
44-
"rubric",
45-
"overfit_guard"
46-
]
53+
"eval_id": "val_explain_cache",
54+
"conversation": [
55+
{
56+
"invocation_id": "val_cache_stale_turn_1",
57+
"user_content": {
58+
"parts": [
59+
{
60+
"text": "Explain cache invalidation and stale data naturally."
61+
}
62+
],
63+
"role": "user"
64+
},
65+
"final_response": {
66+
"parts": [
67+
{
68+
"text": "Cache invalidation decides when cached data is stale data and must be refreshed."
69+
}
70+
],
71+
"role": "model"
72+
}
73+
}
74+
],
75+
"session_input": {
76+
"app_name": "eval_optimize_loop",
77+
"user_id": "validator",
78+
"state": {
79+
"eval_optimize_expectation": {
80+
"type": "rubric",
81+
"must_include": [
82+
"cache",
83+
"stale data"
84+
],
85+
"forbidden": [
86+
"{",
87+
"}",
88+
"json"
89+
],
90+
"max_chars": 180,
91+
"expected_failure_category": "format_violation"
92+
},
93+
"eval_optimize_tags": [
94+
"rubric",
95+
"overfit_guard"
96+
]
97+
}
98+
}
4799
},
48100
{
49-
"id": "val_protected_yes_no",
50-
"input": "Answer exactly YES if idempotent retries are safe for duplicate requests.",
51-
"expectation": {
52-
"type": "exact",
53-
"expected": "YES",
54-
"expected_failure_category": "final_response_mismatch"
55-
},
56-
"protected": true,
57-
"tags": [
58-
"protected",
59-
"exact"
60-
]
101+
"eval_id": "val_protected_yes_no",
102+
"conversation": [
103+
{
104+
"invocation_id": "val_return_yes_turn_1",
105+
"user_content": {
106+
"parts": [
107+
{
108+
"text": "Return only YES; do not use JSON."
109+
}
110+
],
111+
"role": "user"
112+
},
113+
"final_response": {
114+
"parts": [
115+
{
116+
"text": "YES"
117+
}
118+
],
119+
"role": "model"
120+
}
121+
}
122+
],
123+
"session_input": {
124+
"app_name": "eval_optimize_loop",
125+
"user_id": "validator",
126+
"state": {
127+
"eval_optimize_expectation": {
128+
"type": "exact",
129+
"expected": "YES",
130+
"expected_failure_category": "final_response_mismatch"
131+
},
132+
"eval_optimize_tags": [
133+
"protected",
134+
"exact"
135+
],
136+
"eval_optimize_protected": true
137+
}
138+
}
61139
}
62140
]
63141
}

0 commit comments

Comments
 (0)