ai-agent-book 精选快照(<2MB 代码与文档,来自 github.com/bojieli/ai-agent-book)
Build latest book artifacts / build (push) Canceled after 0s
dependency resolution / resolve (3.11) (push) Canceled after 0s
dependency resolution / resolve (3.13) (push) Canceled after 0s
deploy-pages / build (push) Canceled after 0s
deploy-pages / deploy (push) Canceled after 0s
i18n consistency check / check (push) Canceled after 0s
provider adoption tests / test (chapter2/context-compression) (push) Canceled after 0s
provider adoption tests / test (chapter2/prompt-injection) (push) Canceled after 0s
provider adoption tests / test (chapter2/system-hint) (push) Canceled after 0s
provider adoption tests / test (chapter3/log-sanitization) (push) Canceled after 0s
web-search-agent tests / test (push) Canceled after 0s
web-search-agent tests / agentbook (push) Canceled after 0s
Build latest book artifacts / build (push) Canceled after 0s
dependency resolution / resolve (3.11) (push) Canceled after 0s
dependency resolution / resolve (3.13) (push) Canceled after 0s
deploy-pages / build (push) Canceled after 0s
deploy-pages / deploy (push) Canceled after 0s
i18n consistency check / check (push) Canceled after 0s
provider adoption tests / test (chapter2/context-compression) (push) Canceled after 0s
provider adoption tests / test (chapter2/prompt-injection) (push) Canceled after 0s
provider adoption tests / test (chapter2/system-hint) (push) Canceled after 0s
provider adoption tests / test (chapter3/log-sanitization) (push) Canceled after 0s
web-search-agent tests / test (push) Canceled after 0s
web-search-agent tests / agentbook (push) Canceled after 0s
This commit is contained in:
@@ -0,0 +1,170 @@
|
||||
{
|
||||
"experiment_id": "2-8",
|
||||
"created_at": "2026-07-29T22:22:07.705455+00:00",
|
||||
"protocol_sha256": "75231fd4c4f1335e51ddd00395478d1955e5f1670cf61e135ee1d8dc49be4e7a",
|
||||
"provider": {
|
||||
"name": "moonshot",
|
||||
"base_url": "https://api.moonshot.cn/v1",
|
||||
"model": "kimi-k3",
|
||||
"api": "OpenAI-compatible chat.completions with tools/tool_calls",
|
||||
"temperature": 1,
|
||||
"max_completion_tokens": 4096
|
||||
},
|
||||
"unique_runs": 65,
|
||||
"expected_unique_runs": 65,
|
||||
"contrasts": [
|
||||
{
|
||||
"feature": "timestamps_raw",
|
||||
"enabled": "timestamps_raw",
|
||||
"control": "disabled",
|
||||
"suite": "timestamps",
|
||||
"enabled_passes": 5,
|
||||
"control_passes": 3,
|
||||
"pass_rate_delta": 0.4,
|
||||
"n": 5,
|
||||
"enabled_mean_turns": 2.0,
|
||||
"control_mean_turns": 3.8,
|
||||
"enabled_primary_probes": 0,
|
||||
"control_primary_probes": 0,
|
||||
"enabled_mean_component_score": 1.0,
|
||||
"control_mean_component_score": 0.6,
|
||||
"hypothesis_supported": null,
|
||||
"hypothesis_qualification": "nondirectional caveat; report the observed delta rather than a win/loss"
|
||||
},
|
||||
{
|
||||
"feature": "timestamps_guided",
|
||||
"enabled": "timestamps_guided",
|
||||
"control": "disabled",
|
||||
"suite": "timestamps",
|
||||
"enabled_passes": 5,
|
||||
"control_passes": 3,
|
||||
"pass_rate_delta": 0.4,
|
||||
"n": 5,
|
||||
"enabled_mean_turns": 2.0,
|
||||
"control_mean_turns": 3.8,
|
||||
"enabled_primary_probes": 0,
|
||||
"control_primary_probes": 0,
|
||||
"enabled_mean_component_score": 1.0,
|
||||
"control_mean_component_score": 0.6,
|
||||
"hypothesis_supported": true,
|
||||
"hypothesis_qualification": "higher objective pass count"
|
||||
},
|
||||
{
|
||||
"feature": "tool_counter",
|
||||
"enabled": "tool_counter",
|
||||
"control": "disabled",
|
||||
"suite": "tool_counter",
|
||||
"enabled_passes": 5,
|
||||
"control_passes": 5,
|
||||
"pass_rate_delta": 0.0,
|
||||
"n": 5,
|
||||
"enabled_mean_turns": 3.2,
|
||||
"control_mean_turns": 3.0,
|
||||
"enabled_primary_probes": 6,
|
||||
"control_primary_probes": 5,
|
||||
"enabled_mean_component_score": 1.0,
|
||||
"control_mean_component_score": 1.0,
|
||||
"hypothesis_supported": false,
|
||||
"hypothesis_qualification": "higher pass count or fewer primary retries"
|
||||
},
|
||||
{
|
||||
"feature": "todo_list",
|
||||
"enabled": "todo_list",
|
||||
"control": "disabled",
|
||||
"suite": "todo_list",
|
||||
"enabled_passes": 0,
|
||||
"control_passes": 1,
|
||||
"pass_rate_delta": -0.2,
|
||||
"n": 5,
|
||||
"enabled_mean_turns": 3.8,
|
||||
"control_mean_turns": 2.0,
|
||||
"enabled_primary_probes": 0,
|
||||
"control_primary_probes": 0,
|
||||
"enabled_mean_component_score": 0.0,
|
||||
"control_mean_component_score": 0.2,
|
||||
"hypothesis_supported": false,
|
||||
"hypothesis_qualification": "higher complete-artifact count and no greater mean LLM turns"
|
||||
},
|
||||
{
|
||||
"feature": "detailed_errors",
|
||||
"enabled": "detailed_errors",
|
||||
"control": "disabled",
|
||||
"suite": "detailed_errors",
|
||||
"enabled_passes": 4,
|
||||
"control_passes": 5,
|
||||
"pass_rate_delta": -0.2,
|
||||
"n": 5,
|
||||
"enabled_mean_turns": 3.2,
|
||||
"control_mean_turns": 4.0,
|
||||
"enabled_primary_probes": 0,
|
||||
"control_primary_probes": 0,
|
||||
"enabled_mean_component_score": 0.8,
|
||||
"control_mean_component_score": 1.0,
|
||||
"hypothesis_supported": false,
|
||||
"hypothesis_qualification": "higher objective pass count"
|
||||
},
|
||||
{
|
||||
"feature": "system_state",
|
||||
"enabled": "system_state",
|
||||
"control": "disabled",
|
||||
"suite": "system_state",
|
||||
"enabled_passes": 5,
|
||||
"control_passes": 5,
|
||||
"pass_rate_delta": 0.0,
|
||||
"n": 5,
|
||||
"enabled_mean_turns": 3.4,
|
||||
"control_mean_turns": 3.6,
|
||||
"enabled_primary_probes": 0,
|
||||
"control_primary_probes": 0,
|
||||
"enabled_mean_component_score": 1.0,
|
||||
"control_mean_component_score": 1.0,
|
||||
"hypothesis_supported": false,
|
||||
"hypothesis_qualification": "higher objective pass count"
|
||||
},
|
||||
{
|
||||
"feature": "combined",
|
||||
"enabled": "combined",
|
||||
"control": "disabled",
|
||||
"suite": "combined",
|
||||
"enabled_passes": 5,
|
||||
"control_passes": 3,
|
||||
"pass_rate_delta": 0.4,
|
||||
"n": 5,
|
||||
"enabled_mean_turns": 4.8,
|
||||
"control_mean_turns": 5.0,
|
||||
"enabled_primary_probes": 5,
|
||||
"control_primary_probes": 5,
|
||||
"enabled_mean_component_score": 1.0,
|
||||
"control_mean_component_score": 0.9199999999999999,
|
||||
"hypothesis_supported": true,
|
||||
"hypothesis_qualification": "higher overall pass count and mean component score"
|
||||
}
|
||||
],
|
||||
"usage": {
|
||||
"prompt_tokens": 218509,
|
||||
"completion_tokens": 57646,
|
||||
"total_tokens": 276155
|
||||
},
|
||||
"cost": {
|
||||
"amount": 10.13478,
|
||||
"currency": "CNY",
|
||||
"qualification": "all prompt tokens conservatively priced as uncached"
|
||||
},
|
||||
"historical_claim_policy": {
|
||||
"todo_15_vs_21_iterations": "not directly reproduced unless the original task distribution and iteration definition are available; report current-suite means separately",
|
||||
"error_recovery_60_vs_95_percent": "not directly reproduced with this five-pair suite; report current-suite exact rates separately",
|
||||
"time_sense_19_to_49_point_gain": "not directly reproduced because this is a targeted status-bar suite, not the cited six-model time-sense benchmark"
|
||||
},
|
||||
"acceptance": {
|
||||
"all_preregistered_runs_complete": true,
|
||||
"exact_model_every_response": true,
|
||||
"all_tool_protocols_valid": true,
|
||||
"all_provider_receipts_valid": true,
|
||||
"preregistered_arm_order_recorded": true,
|
||||
"interventions_visible_and_controls_clean": true,
|
||||
"detailed_error_feature_exercised": true,
|
||||
"credential_scan_passed": true
|
||||
},
|
||||
"credential_scan_findings": [],
|
||||
"campaign_complete": true
|
||||
}
|
||||
Reference in New Issue
Block a user