{ "schema_version": "2.0", "experiment": "5-13", "frozen_at_utc": "2026-07-30T00:00:00Z", "manuscript_source": "book/chapter5.md#experiment-5-13", "requirements": "Create a release-readiness Agent. It must inspect structured deployment facts, identify failed quality gates, refuse release when any required gate fails, and produce an evidence-backed remediation checklist.", "backend_requirement": { "provider": "moonshot", "model": "kimi-k3", "api_style": "OpenAI-compatible chat.completions with tools/tool_calls", "documentation_url": "https://platform.kimi.com/docs/guide/start-using-kimi-api", "pricing": { "as_of": "2026-07-29", "currency": "CNY", "uncached_input_per_million": 20.0, "cached_input_per_million": 2.0, "output_per_million": 100.0, "source_url": "https://platform.kimi.com/docs/pricing/chat-k3.md", "legacy_or_missing_cache_split_policy": "Treat all prompt tokens without an observed cached-token split as uncached." } }, "comparison_design": { "strategies": [ "template", "scratch" ], "controlled_variables": [ "creator provider and model", "generated-Agent provider and model", "requirements", "live cases and histories", "deterministic validation code", "timeouts and maximum repair attempts" ], "quality_metric": "Sum of preregistered deterministic case checks. Quality non-inferiority means template score >= scratch score; strict advantage means template score > scratch score.", "efficiency_metric": "Creation only. Template must use fewer total creator tokens (prompt + completion) and less creator wall time than scratch. Live task cost is reported separately and is not used to choose the creation winner.", "joint_book_claim": "Supported only when template has a strict quality advantage and an efficiency advantage. A quality tie is reported as non-inferior, never as a strict quality advantage.", "no_post_hoc_rule": "This file and its SHA-256 are saved with the campaign. Changing any criterion requires a new protocol version and a new campaign." }, "completion_gates": [ "protocol hash recorded", "required current provider, model, and API style used", "both arms generated by the same real model", "both arms pass required-file, secret, compile, and generated-test gates", "both arms use standard assistant.tool_calls followed by matching role=tool messages", "both arms run every common real basic task", "both arms preserve supplied multi-turn history and use it in the final answer", "credential-free raw creator and live evidence saved", "provider usage saved with complete native-currency cost accounting", "quality and efficiency conclusions computed from this frozen protocol" ], "live_cases": [ { "id": "refuse_failed_and_skipped", "kind": "basic_task", "history": [], "task": "Evaluate this release candidate and produce the final evidence-backed decision and remediation checklist without asking for more information: {\"deployment\":\"payment-service:v2.4.1\",\"environment\":\"production\",\"gates\":[{\"id\":\"unit_tests\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"1842/1842 tests passed\"},{\"id\":\"integration_tests\",\"required\":true,\"outcome\":\"failed\",\"evidence\":\"checkout_webhook test failed\"},{\"id\":\"security_scan\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"0 critical or high findings\"},{\"id\":\"load_test\",\"required\":true,\"outcome\":\"skipped\",\"evidence\":\"no report uploaded\"},{\"id\":\"code_review\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"two approvals\"}]}. A required gate passes only when outcome is exactly passed.", "expected": { "decision": "REFUSED", "failed_ids": [ "integration_tests", "load_test" ], "evidence": [ "checkout_webhook test failed", "no report uploaded" ], "answer_substrings": [ "REFUSED", "integration_tests", "load_test" ], "forbidden_answer_substrings": [ "APPROVED" ], "context_markers": [] } }, { "id": "approve_required_optional_failure", "kind": "basic_task", "history": [], "task": "Evaluate this release candidate and give the final evidence-backed decision: {\"deployment\":\"catalog-service:v1.8.0\",\"environment\":\"staging\",\"gates\":[{\"id\":\"unit_tests\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"912/912 tests passed\"},{\"id\":\"security_scan\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"0 high findings\"},{\"id\":\"optional_benchmark\",\"required\":false,\"outcome\":\"failed\",\"evidence\":\"optional latency target missed\"}]}. Optional failures do not block release.", "expected": { "decision": "APPROVED", "failed_ids": [], "evidence": [], "answer_substrings": [ "APPROVED" ], "forbidden_answer_substrings": [ "REFUSED" ], "context_markers": [] } }, { "id": "multiturn_state_and_refusal", "kind": "multi_turn_state", "history": [ { "role": "user", "content": "For the next release decision, remember that the accountable release owner is Mei-Lin and the change ticket is CR-4821." }, { "role": "assistant", "content": "Understood. I will retain release owner Mei-Lin and change ticket CR-4821 for the next decision." } ], "task": "Using the prior conversation state, name the release owner and change ticket, then evaluate: {\"deployment\":\"identity-service:v3.0.0\",\"environment\":\"production\",\"gates\":[{\"id\":\"unit_tests\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"2201/2201 tests passed\"},{\"id\":\"rollback_drill\",\"required\":true,\"outcome\":\"failed\",\"evidence\":\"rollback exceeded the 10-minute objective\"}]}. Give a final evidence-backed decision and remediation.", "expected": { "decision": "REFUSED", "failed_ids": [ "rollback_drill" ], "evidence": [ "rollback exceeded the 10-minute objective" ], "answer_substrings": [ "REFUSED", "rollback_drill" ], "forbidden_answer_substrings": [ "APPROVED" ], "context_markers": [ "Mei-Lin", "CR-4821" ] } } ] }