Files
ai-agent-book/chapter5/agent-creator/experiment_protocol.json
T
liqiang b119135836
Build latest book artifacts / build (push) Canceled after 0s
dependency resolution / resolve (3.11) (push) Canceled after 0s
dependency resolution / resolve (3.13) (push) Canceled after 0s
deploy-pages / build (push) Canceled after 0s
deploy-pages / deploy (push) Canceled after 0s
i18n consistency check / check (push) Canceled after 0s
provider adoption tests / test (chapter2/context-compression) (push) Canceled after 0s
provider adoption tests / test (chapter2/prompt-injection) (push) Canceled after 0s
provider adoption tests / test (chapter2/system-hint) (push) Canceled after 0s
provider adoption tests / test (chapter3/log-sanitization) (push) Canceled after 0s
web-search-agent tests / test (push) Canceled after 0s
web-search-agent tests / agentbook (push) Canceled after 0s
ai-agent-book 精选快照(<2MB 代码与文档,来自 github.com/bojieli/ai-agent-book)
2026-08-20 13:12:50 +00:00

134 lines
6.5 KiB
JSON

{
"schema_version": "2.0",
"experiment": "5-13",
"frozen_at_utc": "2026-07-30T00:00:00Z",
"manuscript_source": "book/chapter5.md#experiment-5-13",
"requirements": "Create a release-readiness Agent. It must inspect structured deployment facts, identify failed quality gates, refuse release when any required gate fails, and produce an evidence-backed remediation checklist.",
"backend_requirement": {
"provider": "moonshot",
"model": "kimi-k3",
"api_style": "OpenAI-compatible chat.completions with tools/tool_calls",
"documentation_url": "https://platform.kimi.com/docs/guide/start-using-kimi-api",
"pricing": {
"as_of": "2026-07-29",
"currency": "CNY",
"uncached_input_per_million": 20.0,
"cached_input_per_million": 2.0,
"output_per_million": 100.0,
"source_url": "https://platform.kimi.com/docs/pricing/chat-k3.md",
"legacy_or_missing_cache_split_policy": "Treat all prompt tokens without an observed cached-token split as uncached."
}
},
"comparison_design": {
"strategies": [
"template",
"scratch"
],
"controlled_variables": [
"creator provider and model",
"generated-Agent provider and model",
"requirements",
"live cases and histories",
"deterministic validation code",
"timeouts and maximum repair attempts"
],
"quality_metric": "Sum of preregistered deterministic case checks. Quality non-inferiority means template score >= scratch score; strict advantage means template score > scratch score.",
"efficiency_metric": "Creation only. Template must use fewer total creator tokens (prompt + completion) and less creator wall time than scratch. Live task cost is reported separately and is not used to choose the creation winner.",
"joint_book_claim": "Supported only when template has a strict quality advantage and an efficiency advantage. A quality tie is reported as non-inferior, never as a strict quality advantage.",
"no_post_hoc_rule": "This file and its SHA-256 are saved with the campaign. Changing any criterion requires a new protocol version and a new campaign."
},
"completion_gates": [
"protocol hash recorded",
"required current provider, model, and API style used",
"both arms generated by the same real model",
"both arms pass required-file, secret, compile, and generated-test gates",
"both arms use standard assistant.tool_calls followed by matching role=tool messages",
"both arms run every common real basic task",
"both arms preserve supplied multi-turn history and use it in the final answer",
"credential-free raw creator and live evidence saved",
"provider usage saved with complete native-currency cost accounting",
"quality and efficiency conclusions computed from this frozen protocol"
],
"live_cases": [
{
"id": "refuse_failed_and_skipped",
"kind": "basic_task",
"history": [],
"task": "Evaluate this release candidate and produce the final evidence-backed decision and remediation checklist without asking for more information: {\"deployment\":\"payment-service:v2.4.1\",\"environment\":\"production\",\"gates\":[{\"id\":\"unit_tests\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"1842/1842 tests passed\"},{\"id\":\"integration_tests\",\"required\":true,\"outcome\":\"failed\",\"evidence\":\"checkout_webhook test failed\"},{\"id\":\"security_scan\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"0 critical or high findings\"},{\"id\":\"load_test\",\"required\":true,\"outcome\":\"skipped\",\"evidence\":\"no report uploaded\"},{\"id\":\"code_review\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"two approvals\"}]}. A required gate passes only when outcome is exactly passed.",
"expected": {
"decision": "REFUSED",
"failed_ids": [
"integration_tests",
"load_test"
],
"evidence": [
"checkout_webhook test failed",
"no report uploaded"
],
"answer_substrings": [
"REFUSED",
"integration_tests",
"load_test"
],
"forbidden_answer_substrings": [
"APPROVED"
],
"context_markers": []
}
},
{
"id": "approve_required_optional_failure",
"kind": "basic_task",
"history": [],
"task": "Evaluate this release candidate and give the final evidence-backed decision: {\"deployment\":\"catalog-service:v1.8.0\",\"environment\":\"staging\",\"gates\":[{\"id\":\"unit_tests\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"912/912 tests passed\"},{\"id\":\"security_scan\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"0 high findings\"},{\"id\":\"optional_benchmark\",\"required\":false,\"outcome\":\"failed\",\"evidence\":\"optional latency target missed\"}]}. Optional failures do not block release.",
"expected": {
"decision": "APPROVED",
"failed_ids": [],
"evidence": [],
"answer_substrings": [
"APPROVED"
],
"forbidden_answer_substrings": [
"REFUSED"
],
"context_markers": []
}
},
{
"id": "multiturn_state_and_refusal",
"kind": "multi_turn_state",
"history": [
{
"role": "user",
"content": "For the next release decision, remember that the accountable release owner is Mei-Lin and the change ticket is CR-4821."
},
{
"role": "assistant",
"content": "Understood. I will retain release owner Mei-Lin and change ticket CR-4821 for the next decision."
}
],
"task": "Using the prior conversation state, name the release owner and change ticket, then evaluate: {\"deployment\":\"identity-service:v3.0.0\",\"environment\":\"production\",\"gates\":[{\"id\":\"unit_tests\",\"required\":true,\"outcome\":\"passed\",\"evidence\":\"2201/2201 tests passed\"},{\"id\":\"rollback_drill\",\"required\":true,\"outcome\":\"failed\",\"evidence\":\"rollback exceeded the 10-minute objective\"}]}. Give a final evidence-backed decision and remediation.",
"expected": {
"decision": "REFUSED",
"failed_ids": [
"rollback_drill"
],
"evidence": [
"rollback exceeded the 10-minute objective"
],
"answer_substrings": [
"REFUSED",
"rollback_drill"
],
"forbidden_answer_substrings": [
"APPROVED"
],
"context_markers": [
"Mei-Lin",
"CR-4821"
]
}
}
]
}