Build latest book artifacts / build (push) Canceled after 0s
dependency resolution / resolve (3.11) (push) Canceled after 0s
dependency resolution / resolve (3.13) (push) Canceled after 0s
deploy-pages / build (push) Canceled after 0s
deploy-pages / deploy (push) Canceled after 0s
i18n consistency check / check (push) Canceled after 0s
provider adoption tests / test (chapter2/context-compression) (push) Canceled after 0s
provider adoption tests / test (chapter2/prompt-injection) (push) Canceled after 0s
provider adoption tests / test (chapter2/system-hint) (push) Canceled after 0s
provider adoption tests / test (chapter3/log-sanitization) (push) Canceled after 0s
web-search-agent tests / test (push) Canceled after 0s
web-search-agent tests / agentbook (push) Canceled after 0s
170 lines
5.6 KiB
JSON
170 lines
5.6 KiB
JSON
{
|
|
"experiment_id": "2-8",
|
|
"created_at": "2026-07-29T22:22:07.705455+00:00",
|
|
"protocol_sha256": "75231fd4c4f1335e51ddd00395478d1955e5f1670cf61e135ee1d8dc49be4e7a",
|
|
"provider": {
|
|
"name": "moonshot",
|
|
"base_url": "https://api.moonshot.cn/v1",
|
|
"model": "kimi-k3",
|
|
"api": "OpenAI-compatible chat.completions with tools/tool_calls",
|
|
"temperature": 1,
|
|
"max_completion_tokens": 4096
|
|
},
|
|
"unique_runs": 65,
|
|
"expected_unique_runs": 65,
|
|
"contrasts": [
|
|
{
|
|
"feature": "timestamps_raw",
|
|
"enabled": "timestamps_raw",
|
|
"control": "disabled",
|
|
"suite": "timestamps",
|
|
"enabled_passes": 5,
|
|
"control_passes": 3,
|
|
"pass_rate_delta": 0.4,
|
|
"n": 5,
|
|
"enabled_mean_turns": 2.0,
|
|
"control_mean_turns": 3.8,
|
|
"enabled_primary_probes": 0,
|
|
"control_primary_probes": 0,
|
|
"enabled_mean_component_score": 1.0,
|
|
"control_mean_component_score": 0.6,
|
|
"hypothesis_supported": null,
|
|
"hypothesis_qualification": "nondirectional caveat; report the observed delta rather than a win/loss"
|
|
},
|
|
{
|
|
"feature": "timestamps_guided",
|
|
"enabled": "timestamps_guided",
|
|
"control": "disabled",
|
|
"suite": "timestamps",
|
|
"enabled_passes": 5,
|
|
"control_passes": 3,
|
|
"pass_rate_delta": 0.4,
|
|
"n": 5,
|
|
"enabled_mean_turns": 2.0,
|
|
"control_mean_turns": 3.8,
|
|
"enabled_primary_probes": 0,
|
|
"control_primary_probes": 0,
|
|
"enabled_mean_component_score": 1.0,
|
|
"control_mean_component_score": 0.6,
|
|
"hypothesis_supported": true,
|
|
"hypothesis_qualification": "higher objective pass count"
|
|
},
|
|
{
|
|
"feature": "tool_counter",
|
|
"enabled": "tool_counter",
|
|
"control": "disabled",
|
|
"suite": "tool_counter",
|
|
"enabled_passes": 5,
|
|
"control_passes": 5,
|
|
"pass_rate_delta": 0.0,
|
|
"n": 5,
|
|
"enabled_mean_turns": 3.2,
|
|
"control_mean_turns": 3.0,
|
|
"enabled_primary_probes": 6,
|
|
"control_primary_probes": 5,
|
|
"enabled_mean_component_score": 1.0,
|
|
"control_mean_component_score": 1.0,
|
|
"hypothesis_supported": false,
|
|
"hypothesis_qualification": "higher pass count or fewer primary retries"
|
|
},
|
|
{
|
|
"feature": "todo_list",
|
|
"enabled": "todo_list",
|
|
"control": "disabled",
|
|
"suite": "todo_list",
|
|
"enabled_passes": 0,
|
|
"control_passes": 1,
|
|
"pass_rate_delta": -0.2,
|
|
"n": 5,
|
|
"enabled_mean_turns": 3.8,
|
|
"control_mean_turns": 2.0,
|
|
"enabled_primary_probes": 0,
|
|
"control_primary_probes": 0,
|
|
"enabled_mean_component_score": 0.0,
|
|
"control_mean_component_score": 0.2,
|
|
"hypothesis_supported": false,
|
|
"hypothesis_qualification": "higher complete-artifact count and no greater mean LLM turns"
|
|
},
|
|
{
|
|
"feature": "detailed_errors",
|
|
"enabled": "detailed_errors",
|
|
"control": "disabled",
|
|
"suite": "detailed_errors",
|
|
"enabled_passes": 4,
|
|
"control_passes": 5,
|
|
"pass_rate_delta": -0.2,
|
|
"n": 5,
|
|
"enabled_mean_turns": 3.2,
|
|
"control_mean_turns": 4.0,
|
|
"enabled_primary_probes": 0,
|
|
"control_primary_probes": 0,
|
|
"enabled_mean_component_score": 0.8,
|
|
"control_mean_component_score": 1.0,
|
|
"hypothesis_supported": false,
|
|
"hypothesis_qualification": "higher objective pass count"
|
|
},
|
|
{
|
|
"feature": "system_state",
|
|
"enabled": "system_state",
|
|
"control": "disabled",
|
|
"suite": "system_state",
|
|
"enabled_passes": 5,
|
|
"control_passes": 5,
|
|
"pass_rate_delta": 0.0,
|
|
"n": 5,
|
|
"enabled_mean_turns": 3.4,
|
|
"control_mean_turns": 3.6,
|
|
"enabled_primary_probes": 0,
|
|
"control_primary_probes": 0,
|
|
"enabled_mean_component_score": 1.0,
|
|
"control_mean_component_score": 1.0,
|
|
"hypothesis_supported": false,
|
|
"hypothesis_qualification": "higher objective pass count"
|
|
},
|
|
{
|
|
"feature": "combined",
|
|
"enabled": "combined",
|
|
"control": "disabled",
|
|
"suite": "combined",
|
|
"enabled_passes": 5,
|
|
"control_passes": 3,
|
|
"pass_rate_delta": 0.4,
|
|
"n": 5,
|
|
"enabled_mean_turns": 4.8,
|
|
"control_mean_turns": 5.0,
|
|
"enabled_primary_probes": 5,
|
|
"control_primary_probes": 5,
|
|
"enabled_mean_component_score": 1.0,
|
|
"control_mean_component_score": 0.9199999999999999,
|
|
"hypothesis_supported": true,
|
|
"hypothesis_qualification": "higher overall pass count and mean component score"
|
|
}
|
|
],
|
|
"usage": {
|
|
"prompt_tokens": 218509,
|
|
"completion_tokens": 57646,
|
|
"total_tokens": 276155
|
|
},
|
|
"cost": {
|
|
"amount": 10.13478,
|
|
"currency": "CNY",
|
|
"qualification": "all prompt tokens conservatively priced as uncached"
|
|
},
|
|
"historical_claim_policy": {
|
|
"todo_15_vs_21_iterations": "not directly reproduced unless the original task distribution and iteration definition are available; report current-suite means separately",
|
|
"error_recovery_60_vs_95_percent": "not directly reproduced with this five-pair suite; report current-suite exact rates separately",
|
|
"time_sense_19_to_49_point_gain": "not directly reproduced because this is a targeted status-bar suite, not the cited six-model time-sense benchmark"
|
|
},
|
|
"acceptance": {
|
|
"all_preregistered_runs_complete": true,
|
|
"exact_model_every_response": true,
|
|
"all_tool_protocols_valid": true,
|
|
"all_provider_receipts_valid": true,
|
|
"preregistered_arm_order_recorded": true,
|
|
"interventions_visible_and_controls_clean": true,
|
|
"detailed_error_feature_exercised": true,
|
|
"credential_scan_passed": true
|
|
},
|
|
"credential_scan_findings": [],
|
|
"campaign_complete": true
|
|
} |