Build latest book artifacts / build (push) Canceled after 0s
dependency resolution / resolve (3.11) (push) Canceled after 0s
dependency resolution / resolve (3.13) (push) Canceled after 0s
deploy-pages / build (push) Canceled after 0s
deploy-pages / deploy (push) Canceled after 0s
i18n consistency check / check (push) Canceled after 0s
provider adoption tests / test (chapter2/context-compression) (push) Canceled after 0s
provider adoption tests / test (chapter2/prompt-injection) (push) Canceled after 0s
provider adoption tests / test (chapter2/system-hint) (push) Canceled after 0s
provider adoption tests / test (chapter3/log-sanitization) (push) Canceled after 0s
web-search-agent tests / test (push) Canceled after 0s
web-search-agent tests / agentbook (push) Canceled after 0s
134 lines
6.5 KiB
JSON
134 lines
6.5 KiB
JSON
{
|
|
"schema_version": "2.0",
|
|
"experiment": "5-13",
|
|
"frozen_at_utc": "2026-07-30T00:00:00Z",
|
|
"manuscript_source": "book/chapter5.md#experiment-5-13",
|
|
"requirements": "Create a release-readiness Agent. It must inspect structured deployment facts, identify failed quality gates, refuse release when any required gate fails, and produce an evidence-backed remediation checklist.",
|
|
"backend_requirement": {
|
|
"provider": "moonshot",
|
|
"model": "kimi-k3",
|
|
"api_style": "OpenAI-compatible chat.completions with tools/tool_calls",
|
|
"documentation_url": "https://platform.kimi.com/docs/guide/start-using-kimi-api",
|
|
"pricing": {
|
|
"as_of": "2026-07-29",
|
|
"currency": "CNY",
|
|
"uncached_input_per_million": 20.0,
|
|
"cached_input_per_million": 2.0,
|
|
"output_per_million": 100.0,
|
|
"source_url": "https://platform.kimi.com/docs/pricing/chat-k3.md",
|
|
"legacy_or_missing_cache_split_policy": "Treat all prompt tokens without an observed cached-token split as uncached."
|
|
}
|
|
},
|
|
"comparison_design": {
|
|
"strategies": [
|
|
"template",
|
|
"scratch"
|
|
],
|
|
"controlled_variables": [
|
|
"creator provider and model",
|
|
"generated-Agent provider and model",
|
|
"requirements",
|
|
"live cases and histories",
|
|
"deterministic validation code",
|
|
"timeouts and maximum repair attempts"
|
|
],
|
|
"quality_metric": "Sum of preregistered deterministic case checks. Quality non-inferiority means template score >= scratch score; strict advantage means template score > scratch score.",
|
|
"efficiency_metric": "Creation only. Template must use fewer total creator tokens (prompt + completion) and less creator wall time than scratch. Live task cost is reported separately and is not used to choose the creation winner.",
|
|
"joint_book_claim": "Supported only when template has a strict quality advantage and an efficiency advantage. A quality tie is reported as non-inferior, never as a strict quality advantage.",
|
|
"no_post_hoc_rule": "This file and its SHA-256 are saved with the campaign. Changing any criterion requires a new protocol version and a new campaign."
|
|
},
|
|
"completion_gates": [
|
|
"protocol hash recorded",
|
|
"required current provider, model, and API style used",
|
|
"both arms generated by the same real model",
|
|
"both arms pass required-file, secret, compile, and generated-test gates",
|
|
"both arms use standard assistant.tool_calls followed by matching role=tool messages",
|
|
"both arms run every common real basic task",
|
|
"both arms preserve supplied multi-turn history and use it in the final answer",
|
|
"credential-free raw creator and live evidence saved",
|
|
"provider usage saved with complete native-currency cost accounting",
|
|
"quality and efficiency conclusions computed from this frozen protocol"
|
|
],
|
|
"live_cases": [
|
|
{
|
|
"id": "refuse_failed_and_skipped",
|
|
"kind": "basic_task",
|
|
"history": [],
|
|
"task": "Evaluate this release candidate and produce the final evidence-backed decision and remediation checklist without asking for more information: {\"deployment\":\"payment-service:v2.4.1\",\"environment\":\"production\",\"gates\":[{\"id\":\"unit_tests\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"1842/1842 tests passed\"},{\"id\":\"integration_tests\",\"required\":true,\"outcome\":\"failed\",\"evidence\":\"checkout_webhook test failed\"},{\"id\":\"security_scan\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"0 critical or high findings\"},{\"id\":\"load_test\",\"required\":true,\"outcome\":\"skipped\",\"evidence\":\"no report uploaded\"},{\"id\":\"code_review\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"two approvals\"}]}. A required gate passes only when outcome is exactly passed.",
|
|
"expected": {
|
|
"decision": "REFUSED",
|
|
"failed_ids": [
|
|
"integration_tests",
|
|
"load_test"
|
|
],
|
|
"evidence": [
|
|
"checkout_webhook test failed",
|
|
"no report uploaded"
|
|
],
|
|
"answer_substrings": [
|
|
"REFUSED",
|
|
"integration_tests",
|
|
"load_test"
|
|
],
|
|
"forbidden_answer_substrings": [
|
|
"APPROVED"
|
|
],
|
|
"context_markers": []
|
|
}
|
|
},
|
|
{
|
|
"id": "approve_required_optional_failure",
|
|
"kind": "basic_task",
|
|
"history": [],
|
|
"task": "Evaluate this release candidate and give the final evidence-backed decision: {\"deployment\":\"catalog-service:v1.8.0\",\"environment\":\"staging\",\"gates\":[{\"id\":\"unit_tests\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"912/912 tests passed\"},{\"id\":\"security_scan\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"0 high findings\"},{\"id\":\"optional_benchmark\",\"required\":false,\"outcome\":\"failed\",\"evidence\":\"optional latency target missed\"}]}. Optional failures do not block release.",
|
|
"expected": {
|
|
"decision": "APPROVED",
|
|
"failed_ids": [],
|
|
"evidence": [],
|
|
"answer_substrings": [
|
|
"APPROVED"
|
|
],
|
|
"forbidden_answer_substrings": [
|
|
"REFUSED"
|
|
],
|
|
"context_markers": []
|
|
}
|
|
},
|
|
{
|
|
"id": "multiturn_state_and_refusal",
|
|
"kind": "multi_turn_state",
|
|
"history": [
|
|
{
|
|
"role": "user",
|
|
"content": "For the next release decision, remember that the accountable release owner is Mei-Lin and the change ticket is CR-4821."
|
|
},
|
|
{
|
|
"role": "assistant",
|
|
"content": "Understood. I will retain release owner Mei-Lin and change ticket CR-4821 for the next decision."
|
|
}
|
|
],
|
|
"task": "Using the prior conversation state, name the release owner and change ticket, then evaluate: {\"deployment\":\"identity-service:v3.0.0\",\"environment\":\"production\",\"gates\":[{\"id\":\"unit_tests\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"2201/2201 tests passed\"},{\"id\":\"rollback_drill\",\"required\":true,\"outcome\":\"failed\",\"evidence\":\"rollback exceeded the 10-minute objective\"}]}. Give a final evidence-backed decision and remediation.",
|
|
"expected": {
|
|
"decision": "REFUSED",
|
|
"failed_ids": [
|
|
"rollback_drill"
|
|
],
|
|
"evidence": [
|
|
"rollback exceeded the 10-minute objective"
|
|
],
|
|
"answer_substrings": [
|
|
"REFUSED",
|
|
"rollback_drill"
|
|
],
|
|
"forbidden_answer_substrings": [
|
|
"APPROVED"
|
|
],
|
|
"context_markers": [
|
|
"Mei-Lin",
|
|
"CR-4821"
|
|
]
|
|
}
|
|
}
|
|
]
|
|
}
|