Build latest book artifacts / build (push) Canceled after 0s
dependency resolution / resolve (3.11) (push) Canceled after 0s
dependency resolution / resolve (3.13) (push) Canceled after 0s
deploy-pages / build (push) Canceled after 0s
deploy-pages / deploy (push) Canceled after 0s
i18n consistency check / check (push) Canceled after 0s
provider adoption tests / test (chapter2/context-compression) (push) Canceled after 0s
provider adoption tests / test (chapter2/prompt-injection) (push) Canceled after 0s
provider adoption tests / test (chapter2/system-hint) (push) Canceled after 0s
provider adoption tests / test (chapter3/log-sanitization) (push) Canceled after 0s
web-search-agent tests / test (push) Canceled after 0s
web-search-agent tests / agentbook (push) Canceled after 0s
477 lines
9.8 KiB
JSON
477 lines
9.8 KiB
JSON
{
|
|
"experiment_id": "2-4",
|
|
"created_at": "2026-07-30T06:13:09.682117+08:00",
|
|
"protocol_sha256": "ec1ec36d5a6ee21b5185295a7a4447f4e0173f1681a9fe471b240dda1e7cf32a",
|
|
"protocol_copy": "/Users/boj/book/ai-agent-book/chapter2/prompt-engineering/runs/exp2-4-kimi-k3-20260730-v4/experiment_protocol.json",
|
|
"provider": "openai",
|
|
"model": "kimi-k3",
|
|
"user_model": "kimi-k3",
|
|
"objective_scoring": "vendored tau-bench environment reward",
|
|
"arms": {
|
|
"baseline": {
|
|
"artifact": {
|
|
"path": "/Users/boj/book/ai-agent-book/chapter2/prompt-engineering/runs/exp2-4-kimi-k3-20260730-v4/tool-calling-kimi-k3-baseline_0730051848.json",
|
|
"sha256": "0adf5fe03e21908792e4ae1218c1ba0500c6b7942c8ee3cc679f36426513727e"
|
|
},
|
|
"rewards": [
|
|
0.0,
|
|
1.0,
|
|
0.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
0.0
|
|
],
|
|
"success_rate": 70.0,
|
|
"total": 10,
|
|
"successes": 7,
|
|
"failures": 3,
|
|
"tasks_completed": 10,
|
|
"expected_tasks": 10,
|
|
"task_errors": [],
|
|
"agent_steps": [
|
|
14,
|
|
10,
|
|
30,
|
|
18,
|
|
15,
|
|
14,
|
|
11,
|
|
14,
|
|
16,
|
|
24
|
|
],
|
|
"tool_calls": [
|
|
6,
|
|
5,
|
|
28,
|
|
14,
|
|
12,
|
|
11,
|
|
7,
|
|
11,
|
|
12,
|
|
17
|
|
],
|
|
"tool_errors": [
|
|
1,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0
|
|
],
|
|
"real_api_calls": 221,
|
|
"response_ids_present": true,
|
|
"usage": {
|
|
"prompt_tokens": 1599454,
|
|
"completion_tokens": 88017,
|
|
"total_tokens": 1687471
|
|
},
|
|
"observed_litellm_cost_usd": 0,
|
|
"all_calls_priced": false,
|
|
"native_cost_cny": 40.79078,
|
|
"arm_complete": true
|
|
},
|
|
"tone_trump": {
|
|
"artifact": {
|
|
"path": "/Users/boj/book/ai-agent-book/chapter2/prompt-engineering/runs/exp2-4-kimi-k3-20260730-v4/tool-calling-kimi-k3-tone_trump_0730052734.json",
|
|
"sha256": "3c9766ed2801aa70003b24cc013dcbf6997581236f76700798a591c28b1a0293"
|
|
},
|
|
"rewards": [
|
|
1.0,
|
|
1.0,
|
|
0.0,
|
|
1.0,
|
|
0.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
0.0,
|
|
0.0
|
|
],
|
|
"success_rate": 60.0,
|
|
"total": 10,
|
|
"successes": 6,
|
|
"failures": 4,
|
|
"tasks_completed": 10,
|
|
"expected_tasks": 10,
|
|
"task_errors": [
|
|
{
|
|
"task_id": 9,
|
|
"error": "User simulator returned empty content on three accepted responses"
|
|
}
|
|
],
|
|
"agent_steps": [
|
|
13,
|
|
11,
|
|
16,
|
|
18,
|
|
12,
|
|
15,
|
|
10,
|
|
12,
|
|
14,
|
|
null
|
|
],
|
|
"tool_calls": [
|
|
7,
|
|
7,
|
|
14,
|
|
14,
|
|
9,
|
|
12,
|
|
6,
|
|
8,
|
|
10,
|
|
null
|
|
],
|
|
"tool_errors": [
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
null
|
|
],
|
|
"real_api_calls": 171,
|
|
"response_ids_present": true,
|
|
"usage": {
|
|
"prompt_tokens": 1108201,
|
|
"completion_tokens": 81717,
|
|
"total_tokens": 1189918
|
|
},
|
|
"observed_litellm_cost_usd": 0,
|
|
"all_calls_priced": false,
|
|
"native_cost_cny": 30.335720000000002,
|
|
"arm_complete": false
|
|
},
|
|
"tone_casual": {
|
|
"artifact": {
|
|
"path": "/Users/boj/book/ai-agent-book/chapter2/prompt-engineering/runs/exp2-4-kimi-k3-20260730-v4/tool-calling-kimi-k3-tone_casual_0730053331.json",
|
|
"sha256": "8a07bcd0041469d53bc7228deaf02eb54edf350dbe7460c37cf5fca6e3a6e654"
|
|
},
|
|
"rewards": [
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
0.0
|
|
],
|
|
"success_rate": 90.0,
|
|
"total": 10,
|
|
"successes": 9,
|
|
"failures": 1,
|
|
"tasks_completed": 10,
|
|
"expected_tasks": 10,
|
|
"task_errors": [
|
|
{
|
|
"task_id": 9,
|
|
"error": "litellm.BadRequestError: OpenAIException - Invalid request: the message at position 13 with role 'user' must not be empty"
|
|
}
|
|
],
|
|
"agent_steps": [
|
|
14,
|
|
10,
|
|
16,
|
|
15,
|
|
15,
|
|
14,
|
|
12,
|
|
15,
|
|
16,
|
|
null
|
|
],
|
|
"tool_calls": [
|
|
9,
|
|
5,
|
|
13,
|
|
11,
|
|
11,
|
|
11,
|
|
8,
|
|
9,
|
|
13,
|
|
null
|
|
],
|
|
"tool_errors": [
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
null
|
|
],
|
|
"real_api_calls": 180,
|
|
"response_ids_present": true,
|
|
"usage": {
|
|
"prompt_tokens": 1187766,
|
|
"completion_tokens": 77626,
|
|
"total_tokens": 1265392
|
|
},
|
|
"observed_litellm_cost_usd": 0,
|
|
"all_calls_priced": false,
|
|
"native_cost_cny": 31.51792,
|
|
"arm_complete": false
|
|
},
|
|
"wiki_random": {
|
|
"artifact": {
|
|
"path": "/Users/boj/book/ai-agent-book/chapter2/prompt-engineering/runs/exp2-4-kimi-k3-20260730-v4/tool-calling-kimi-k3-wiki_random_0730054009.json",
|
|
"sha256": "dc3c8d20d531e7d555d56325ac2b40989316f65f1cb875a499a6a5331c62a522"
|
|
},
|
|
"rewards": [
|
|
1.0,
|
|
1.0,
|
|
0.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
0.0
|
|
],
|
|
"success_rate": 80.0,
|
|
"total": 10,
|
|
"successes": 8,
|
|
"failures": 2,
|
|
"tasks_completed": 10,
|
|
"expected_tasks": 10,
|
|
"task_errors": [],
|
|
"agent_steps": [
|
|
14,
|
|
9,
|
|
14,
|
|
15,
|
|
13,
|
|
10,
|
|
14,
|
|
30,
|
|
15,
|
|
23
|
|
],
|
|
"tool_calls": [
|
|
7,
|
|
5,
|
|
10,
|
|
13,
|
|
10,
|
|
7,
|
|
10,
|
|
28,
|
|
11,
|
|
15
|
|
],
|
|
"tool_errors": [
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0
|
|
],
|
|
"real_api_calls": 209,
|
|
"response_ids_present": true,
|
|
"usage": {
|
|
"prompt_tokens": 1446738,
|
|
"completion_tokens": 89732,
|
|
"total_tokens": 1536470
|
|
},
|
|
"observed_litellm_cost_usd": 0,
|
|
"all_calls_priced": false,
|
|
"native_cost_cny": 37.90796,
|
|
"arm_complete": true
|
|
},
|
|
"no_tool_desc": {
|
|
"artifact": {
|
|
"path": "/Users/boj/book/ai-agent-book/chapter2/prompt-engineering/runs/exp2-4-kimi-k3-20260730-v4/tool-calling-kimi-k3-no_tool_desc_0730055428.json",
|
|
"sha256": "590af99d99429ab69df3a46b9e895ac654f1697534a685e9891794c9d8471c9a"
|
|
},
|
|
"rewards": [
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
0.0
|
|
],
|
|
"success_rate": 90.0,
|
|
"total": 10,
|
|
"successes": 9,
|
|
"failures": 1,
|
|
"tasks_completed": 10,
|
|
"expected_tasks": 10,
|
|
"task_errors": [
|
|
{
|
|
"task_id": 9,
|
|
"error": "User simulator returned empty content on three accepted responses"
|
|
}
|
|
],
|
|
"agent_steps": [
|
|
14,
|
|
9,
|
|
15,
|
|
12,
|
|
15,
|
|
11,
|
|
13,
|
|
13,
|
|
17,
|
|
null
|
|
],
|
|
"tool_calls": [
|
|
9,
|
|
5,
|
|
11,
|
|
9,
|
|
12,
|
|
7,
|
|
10,
|
|
8,
|
|
14,
|
|
null
|
|
],
|
|
"tool_errors": [
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
null
|
|
],
|
|
"real_api_calls": 172,
|
|
"response_ids_present": true,
|
|
"usage": {
|
|
"prompt_tokens": 967540,
|
|
"completion_tokens": 80968,
|
|
"total_tokens": 1048508
|
|
},
|
|
"observed_litellm_cost_usd": 0,
|
|
"all_calls_priced": false,
|
|
"native_cost_cny": 27.4476,
|
|
"arm_complete": false
|
|
},
|
|
"all_ablations": {
|
|
"artifact": {
|
|
"path": "/Users/boj/book/ai-agent-book/chapter2/prompt-engineering/runs/exp2-4-kimi-k3-20260730-v4/tool-calling-kimi-k3-tone_casual_wiki_random_no_tool_desc_0730060254.json",
|
|
"sha256": "9bc93f263316cba3ee94ebd55311aa297d67dfd2f359c926e012cde5b9d2000b"
|
|
},
|
|
"rewards": [
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
0.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
1.0,
|
|
0.0
|
|
],
|
|
"success_rate": 80.0,
|
|
"total": 10,
|
|
"successes": 8,
|
|
"failures": 2,
|
|
"tasks_completed": 10,
|
|
"expected_tasks": 10,
|
|
"task_errors": [],
|
|
"agent_steps": [
|
|
13,
|
|
10,
|
|
16,
|
|
16,
|
|
9,
|
|
12,
|
|
15,
|
|
12,
|
|
15,
|
|
21
|
|
],
|
|
"tool_calls": [
|
|
7,
|
|
5,
|
|
13,
|
|
12,
|
|
8,
|
|
6,
|
|
10,
|
|
8,
|
|
10,
|
|
14
|
|
],
|
|
"tool_errors": [
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0,
|
|
0
|
|
],
|
|
"real_api_calls": 196,
|
|
"response_ids_present": true,
|
|
"usage": {
|
|
"prompt_tokens": 1247279,
|
|
"completion_tokens": 97489,
|
|
"total_tokens": 1344768
|
|
},
|
|
"observed_litellm_cost_usd": 0,
|
|
"all_calls_priced": false,
|
|
"native_cost_cny": 34.69448,
|
|
"arm_complete": true
|
|
}
|
|
},
|
|
"credential_scan_passed": true,
|
|
"credential_findings": [],
|
|
"usage_and_cost": {
|
|
"total_real_api_calls": 1149,
|
|
"prompt_tokens": 7556978,
|
|
"completion_tokens": 515549,
|
|
"total_tokens": 8072527,
|
|
"observed_litellm_cost_usd": 0,
|
|
"all_calls_priced": false,
|
|
"native_cost_cny": 202.69446,
|
|
"native_cost_complete": true,
|
|
"qualification": "Prompt cache detail is unavailable in these responses, so every prompt token is conservatively priced as uncached."
|
|
},
|
|
"campaign_complete": false,
|
|
"hypothesis_results": {
|
|
"historical_percentages_reproduced": false,
|
|
"qualification": "Current fixed ten-task campaign only; compare arm metrics directly."
|
|
}
|
|
} |