{ "experiment_id": "2-8", "created_at": "2026-07-29T22:22:07.705455+00:00", "protocol_sha256": "75231fd4c4f1335e51ddd00395478d1955e5f1670cf61e135ee1d8dc49be4e7a", "provider": { "name": "moonshot", "base_url": "https://api.moonshot.cn/v1", "model": "kimi-k3", "api": "OpenAI-compatible chat.completions with tools/tool_calls", "temperature": 1, "max_completion_tokens": 4096 }, "unique_runs": 65, "expected_unique_runs": 65, "contrasts": [ { "feature": "timestamps_raw", "enabled": "timestamps_raw", "control": "disabled", "suite": "timestamps", "enabled_passes": 5, "control_passes": 3, "pass_rate_delta": 0.4, "n": 5, "enabled_mean_turns": 2.0, "control_mean_turns": 3.8, "enabled_primary_probes": 0, "control_primary_probes": 0, "enabled_mean_component_score": 1.0, "control_mean_component_score": 0.6, "hypothesis_supported": null, "hypothesis_qualification": "nondirectional caveat; report the observed delta rather than a win/loss" }, { "feature": "timestamps_guided", "enabled": "timestamps_guided", "control": "disabled", "suite": "timestamps", "enabled_passes": 5, "control_passes": 3, "pass_rate_delta": 0.4, "n": 5, "enabled_mean_turns": 2.0, "control_mean_turns": 3.8, "enabled_primary_probes": 0, "control_primary_probes": 0, "enabled_mean_component_score": 1.0, "control_mean_component_score": 0.6, "hypothesis_supported": true, "hypothesis_qualification": "higher objective pass count" }, { "feature": "tool_counter", "enabled": "tool_counter", "control": "disabled", "suite": "tool_counter", "enabled_passes": 5, "control_passes": 5, "pass_rate_delta": 0.0, "n": 5, "enabled_mean_turns": 3.2, "control_mean_turns": 3.0, "enabled_primary_probes": 6, "control_primary_probes": 5, "enabled_mean_component_score": 1.0, "control_mean_component_score": 1.0, "hypothesis_supported": false, "hypothesis_qualification": "higher pass count or fewer primary retries" }, { "feature": "todo_list", "enabled": "todo_list", "control": "disabled", "suite": "todo_list", "enabled_passes": 0, "control_passes": 1, "pass_rate_delta": -0.2, "n": 5, "enabled_mean_turns": 3.8, "control_mean_turns": 2.0, "enabled_primary_probes": 0, "control_primary_probes": 0, "enabled_mean_component_score": 0.0, "control_mean_component_score": 0.2, "hypothesis_supported": false, "hypothesis_qualification": "higher complete-artifact count and no greater mean LLM turns" }, { "feature": "detailed_errors", "enabled": "detailed_errors", "control": "disabled", "suite": "detailed_errors", "enabled_passes": 4, "control_passes": 5, "pass_rate_delta": -0.2, "n": 5, "enabled_mean_turns": 3.2, "control_mean_turns": 4.0, "enabled_primary_probes": 0, "control_primary_probes": 0, "enabled_mean_component_score": 0.8, "control_mean_component_score": 1.0, "hypothesis_supported": false, "hypothesis_qualification": "higher objective pass count" }, { "feature": "system_state", "enabled": "system_state", "control": "disabled", "suite": "system_state", "enabled_passes": 5, "control_passes": 5, "pass_rate_delta": 0.0, "n": 5, "enabled_mean_turns": 3.4, "control_mean_turns": 3.6, "enabled_primary_probes": 0, "control_primary_probes": 0, "enabled_mean_component_score": 1.0, "control_mean_component_score": 1.0, "hypothesis_supported": false, "hypothesis_qualification": "higher objective pass count" }, { "feature": "combined", "enabled": "combined", "control": "disabled", "suite": "combined", "enabled_passes": 5, "control_passes": 3, "pass_rate_delta": 0.4, "n": 5, "enabled_mean_turns": 4.8, "control_mean_turns": 5.0, "enabled_primary_probes": 5, "control_primary_probes": 5, "enabled_mean_component_score": 1.0, "control_mean_component_score": 0.9199999999999999, "hypothesis_supported": true, "hypothesis_qualification": "higher overall pass count and mean component score" } ], "usage": { "prompt_tokens": 218509, "completion_tokens": 57646, "total_tokens": 276155 }, "cost": { "amount": 10.13478, "currency": "CNY", "qualification": "all prompt tokens conservatively priced as uncached" }, "historical_claim_policy": { "todo_15_vs_21_iterations": "not directly reproduced unless the original task distribution and iteration definition are available; report current-suite means separately", "error_recovery_60_vs_95_percent": "not directly reproduced with this five-pair suite; report current-suite exact rates separately", "time_sense_19_to_49_point_gain": "not directly reproduced because this is a targeted status-bar suite, not the cited six-model time-sense benchmark" }, "acceptance": { "all_preregistered_runs_complete": true, "exact_model_every_response": true, "all_tool_protocols_valid": true, "all_provider_receipts_valid": true, "preregistered_arm_order_recorded": true, "interventions_visible_and_controls_clean": true, "detailed_error_feature_exercised": true, "credential_scan_passed": true }, "credential_scan_findings": [], "campaign_complete": true }