Skip to content

Commit f8bdacb

Browse files
committed
feat(eval_optimize_loop): add sample evalsets, prompts, and pipeline configs
1 parent a3e4b59 commit f8bdacb

10 files changed

Lines changed: 598 additions & 0 deletions
Lines changed: 58 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,58 @@
1+
{
2+
"eval_set_id": "live_train",
3+
"name": "Live mode training set",
4+
"description": "3 training cases for live mode. Use with call_agent.",
5+
"eval_cases": [
6+
{
7+
"eval_id": "case_train_optimizable",
8+
"conversation": [
9+
{
10+
"invocation_id": "t1",
11+
"user_content": {
12+
"parts": [{"text": "小明有 5 个苹果,小红给了他 7 个苹果,小明现在一共有多少个苹果?"}],
13+
"role": "user"
14+
},
15+
"final_response": {
16+
"parts": [{"text": "答案:12 个"}],
17+
"role": "model"
18+
}
19+
}
20+
],
21+
"session_input": {"app_name": "eval_opt_loop", "user_id": "trainer", "state": {}}
22+
},
23+
{
24+
"eval_id": "case_train_always_fail",
25+
"conversation": [
26+
{
27+
"invocation_id": "t2",
28+
"user_content": {
29+
"parts": [{"text": "一辆车开了 60 公里每小时,开了 2 小时,一共开了多少公里?"}],
30+
"role": "user"
31+
},
32+
"final_response": {
33+
"parts": [{"text": "答案:120 公里"}],
34+
"role": "model"
35+
}
36+
}
37+
],
38+
"session_input": {"app_name": "eval_opt_loop", "user_id": "trainer", "state": {}}
39+
},
40+
{
41+
"eval_id": "case_train_regression",
42+
"conversation": [
43+
{
44+
"invocation_id": "t3",
45+
"user_content": {
46+
"parts": [{"text": "5 排座位,每排 8 个,一共多少个?"}],
47+
"role": "user"
48+
},
49+
"final_response": {
50+
"parts": [{"text": "答案:40 个"}],
51+
"role": "model"
52+
}
53+
}
54+
],
55+
"session_input": {"app_name": "eval_opt_loop", "user_id": "trainer", "state": {}}
56+
}
57+
]
58+
}
Lines changed: 58 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,58 @@
1+
{
2+
"eval_set_id": "live_val",
3+
"name": "Live mode validation set",
4+
"description": "3 validation cases for live mode. Use with call_agent.",
5+
"eval_cases": [
6+
{
7+
"eval_id": "case_val_improves",
8+
"conversation": [
9+
{
10+
"invocation_id": "v1",
11+
"user_content": {
12+
"parts": [{"text": "一批商品共 50 件,其中 30% 是次品,次品有多少件?"}],
13+
"role": "user"
14+
},
15+
"final_response": {
16+
"parts": [{"text": "答案:15 件"}],
17+
"role": "model"
18+
}
19+
}
20+
],
21+
"session_input": {"app_name": "eval_opt_loop", "user_id": "validator", "state": {}}
22+
},
23+
{
24+
"eval_id": "case_val_no_change",
25+
"conversation": [
26+
{
27+
"invocation_id": "v2",
28+
"user_content": {
29+
"parts": [{"text": "已知 1 升水重 1 千克,3.5 升水重多少千克?"}],
30+
"role": "user"
31+
},
32+
"final_response": {
33+
"parts": [{"text": "答案:3.5 千克"}],
34+
"role": "model"
35+
}
36+
}
37+
],
38+
"session_input": {"app_name": "eval_opt_loop", "user_id": "validator", "state": {}}
39+
},
40+
{
41+
"eval_id": "case_val_regression",
42+
"conversation": [
43+
{
44+
"invocation_id": "v3",
45+
"user_content": {
46+
"parts": [{"text": "班里一共有 30 人,其中 60% 是女生,请问有多少名女生?"}],
47+
"role": "user"
48+
},
49+
"final_response": {
50+
"parts": [{"text": "答案:18 人"}],
51+
"role": "model"
52+
}
53+
}
54+
],
55+
"session_input": {"app_name": "eval_opt_loop", "user_id": "validator", "state": {}}
56+
}
57+
]
58+
}
Lines changed: 100 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,100 @@
1+
{
2+
"eval_set_id": "train_baseline",
3+
"name": "Training set - baseline traces",
4+
"description": "3 training cases with baseline prompt traces. Contains: optimizable, always-fail, and regression cases.",
5+
"eval_cases": [
6+
{
7+
"eval_id": "case_train_optimizable",
8+
"eval_mode": "trace",
9+
"conversation": [
10+
{
11+
"invocation_id": "t1",
12+
"user_content": {
13+
"parts": [{"text": "小明有 5 个苹果,小红给了他 7 个苹果,小明现在一共有多少个苹果?"}],
14+
"role": "user"
15+
},
16+
"final_response": {
17+
"parts": [{"text": "答案:12 个"}],
18+
"role": "model"
19+
}
20+
}
21+
],
22+
"actual_conversation": [
23+
{
24+
"invocation_id": "t1",
25+
"user_content": {
26+
"parts": [{"text": "小明有 5 个苹果,小红给了他 7 个苹果,小明现在一共有多少个苹果?"}],
27+
"role": "user"
28+
},
29+
"final_response": {
30+
"parts": [{"text": "答案:13 个"}],
31+
"role": "model"
32+
}
33+
}
34+
],
35+
"session_input": {"app_name": "eval_opt_loop", "user_id": "trainer", "state": {}}
36+
},
37+
{
38+
"eval_id": "case_train_always_fail",
39+
"eval_mode": "trace",
40+
"conversation": [
41+
{
42+
"invocation_id": "t2",
43+
"user_content": {
44+
"parts": [{"text": "一辆车开了 60 公里每小时,开了 2 小时,一共开了多少公里?"}],
45+
"role": "user"
46+
},
47+
"final_response": {
48+
"parts": [{"text": "答案:120 公里"}],
49+
"role": "model"
50+
}
51+
}
52+
],
53+
"actual_conversation": [
54+
{
55+
"invocation_id": "t2",
56+
"user_content": {
57+
"parts": [{"text": "一辆车开了 60 公里每小时,开了 2 小时,一共开了多少公里?"}],
58+
"role": "user"
59+
},
60+
"final_response": {
61+
"parts": [{"text": "120 公里"}],
62+
"role": "model"
63+
}
64+
}
65+
],
66+
"session_input": {"app_name": "eval_opt_loop", "user_id": "trainer", "state": {}}
67+
},
68+
{
69+
"eval_id": "case_train_regression",
70+
"eval_mode": "trace",
71+
"conversation": [
72+
{
73+
"invocation_id": "t3",
74+
"user_content": {
75+
"parts": [{"text": "5 排座位,每排 8 个,一共多少个?"}],
76+
"role": "user"
77+
},
78+
"final_response": {
79+
"parts": [{"text": "答案:40 个"}],
80+
"role": "model"
81+
}
82+
}
83+
],
84+
"actual_conversation": [
85+
{
86+
"invocation_id": "t3",
87+
"user_content": {
88+
"parts": [{"text": "5 排座位,每排 8 个,一共多少个?"}],
89+
"role": "user"
90+
},
91+
"final_response": {
92+
"parts": [{"text": "答案:40 个"}],
93+
"role": "model"
94+
}
95+
}
96+
],
97+
"session_input": {"app_name": "eval_opt_loop", "user_id": "trainer", "state": {}}
98+
}
99+
]
100+
}
Lines changed: 100 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,100 @@
1+
{
2+
"eval_set_id": "train_candidate",
3+
"name": "Training set - candidate traces",
4+
"description": "3 training cases with optimized prompt traces.",
5+
"eval_cases": [
6+
{
7+
"eval_id": "case_train_optimizable",
8+
"eval_mode": "trace",
9+
"conversation": [
10+
{
11+
"invocation_id": "t1",
12+
"user_content": {
13+
"parts": [{"text": "小明有 5 个苹果,小红给了他 7 个苹果,小明现在一共有多少个苹果?"}],
14+
"role": "user"
15+
},
16+
"final_response": {
17+
"parts": [{"text": "答案:12 个"}],
18+
"role": "model"
19+
}
20+
}
21+
],
22+
"actual_conversation": [
23+
{
24+
"invocation_id": "t1",
25+
"user_content": {
26+
"parts": [{"text": "小明有 5 个苹果,小红给了他 7 个苹果,小明现在一共有多少个苹果?"}],
27+
"role": "user"
28+
},
29+
"final_response": {
30+
"parts": [{"text": "答案:12 个"}],
31+
"role": "model"
32+
}
33+
}
34+
],
35+
"session_input": {"app_name": "eval_opt_loop", "user_id": "trainer", "state": {}}
36+
},
37+
{
38+
"eval_id": "case_train_always_fail",
39+
"eval_mode": "trace",
40+
"conversation": [
41+
{
42+
"invocation_id": "t2",
43+
"user_content": {
44+
"parts": [{"text": "一辆车开了 60 公里每小时,开了 2 小时,一共开了多少公里?"}],
45+
"role": "user"
46+
},
47+
"final_response": {
48+
"parts": [{"text": "答案:120 公里"}],
49+
"role": "model"
50+
}
51+
}
52+
],
53+
"actual_conversation": [
54+
{
55+
"invocation_id": "t2",
56+
"user_content": {
57+
"parts": [{"text": "一辆车开了 60 公里每小时,开了 2 小时,一共开了多少公里?"}],
58+
"role": "user"
59+
},
60+
"final_response": {
61+
"parts": [{"text": "the answer is 120 公里"}],
62+
"role": "model"
63+
}
64+
}
65+
],
66+
"session_input": {"app_name": "eval_opt_loop", "user_id": "trainer", "state": {}}
67+
},
68+
{
69+
"eval_id": "case_train_regression",
70+
"eval_mode": "trace",
71+
"conversation": [
72+
{
73+
"invocation_id": "t3",
74+
"user_content": {
75+
"parts": [{"text": "5 排座位,每排 8 个,一共多少个?"}],
76+
"role": "user"
77+
},
78+
"final_response": {
79+
"parts": [{"text": "答案:40 个"}],
80+
"role": "model"
81+
}
82+
}
83+
],
84+
"actual_conversation": [
85+
{
86+
"invocation_id": "t3",
87+
"user_content": {
88+
"parts": [{"text": "5 排座位,每排 8 个,一共多少个?"}],
89+
"role": "user"
90+
},
91+
"final_response": {
92+
"parts": [{"text": "答案:35 个"}],
93+
"role": "model"
94+
}
95+
}
96+
],
97+
"session_input": {"app_name": "eval_opt_loop", "user_id": "trainer", "state": {}}
98+
}
99+
]
100+
}

0 commit comments

Comments
 (0)