ai-agent-book 精选快照(<2MB 代码与文档,来自 github.com/bojieli/ai-agent-book)
Build latest book artifacts / build (push) Canceled after 0s
dependency resolution / resolve (3.11) (push) Canceled after 0s
dependency resolution / resolve (3.13) (push) Canceled after 0s
deploy-pages / build (push) Canceled after 0s
deploy-pages / deploy (push) Canceled after 0s
i18n consistency check / check (push) Canceled after 0s
provider adoption tests / test (chapter2/context-compression) (push) Canceled after 0s
provider adoption tests / test (chapter2/prompt-injection) (push) Canceled after 0s
provider adoption tests / test (chapter2/system-hint) (push) Canceled after 0s
provider adoption tests / test (chapter3/log-sanitization) (push) Canceled after 0s
web-search-agent tests / test (push) Canceled after 0s
web-search-agent tests / agentbook (push) Canceled after 0s
Build latest book artifacts / build (push) Canceled after 0s
dependency resolution / resolve (3.11) (push) Canceled after 0s
dependency resolution / resolve (3.13) (push) Canceled after 0s
deploy-pages / build (push) Canceled after 0s
deploy-pages / deploy (push) Canceled after 0s
i18n consistency check / check (push) Canceled after 0s
provider adoption tests / test (chapter2/context-compression) (push) Canceled after 0s
provider adoption tests / test (chapter2/prompt-injection) (push) Canceled after 0s
provider adoption tests / test (chapter2/system-hint) (push) Canceled after 0s
provider adoption tests / test (chapter3/log-sanitization) (push) Canceled after 0s
web-search-agent tests / test (push) Canceled after 0s
web-search-agent tests / agentbook (push) Canceled after 0s
This commit is contained in:
@@ -0,0 +1,42 @@
|
||||
from demo import campaign_completion, paired_statistics
|
||||
|
||||
|
||||
def test_paired_statistics_detects_clear_code_gain():
|
||||
pure = [{"id": str(i), "correct": i < 2} for i in range(20)]
|
||||
code = [
|
||||
{"id": str(i), "correct": i < 19, "used_python_constraint": True}
|
||||
for i in range(20)
|
||||
]
|
||||
result = paired_statistics(pure, code)
|
||||
assert result["code_accuracy"] == 0.95
|
||||
assert result["acceptance"]["code_accuracy_over_90_percent"] is True
|
||||
assert result["acceptance"]["code_significantly_higher_than_pure"] is True
|
||||
|
||||
|
||||
def test_campaign_completion_requires_both_full_real_arms():
|
||||
puzzles = [{"id": f"p-{i}"} for i in range(84)]
|
||||
pure = [
|
||||
{"id": p["id"], "provider_receipts": [{"response_id": "r"}],
|
||||
"provider_error": None}
|
||||
for p in puzzles
|
||||
]
|
||||
code = [
|
||||
{"id": p["id"], "provider_receipts": [{"response_id": "r"}],
|
||||
"provider_error": None, "used_python_constraint": True}
|
||||
for p in puzzles
|
||||
]
|
||||
manifest = {
|
||||
"dataset": "K-and-K/perturbed-knights-and-knaves",
|
||||
"revision": "bc7ee75a15ee8196ccbdb7df3ab46284340412e2",
|
||||
"sampling": {"total": 84, "cells": 42, "per_cell": 2},
|
||||
"source_files": [{} for _ in range(42)],
|
||||
"label_validation": "all rows independently solved with python-constraint",
|
||||
}
|
||||
result = campaign_completion(
|
||||
{"pure": pure, "code": code}, puzzles, manifest, "both"
|
||||
)
|
||||
assert result["status"] == "complete"
|
||||
code[-1]["used_python_constraint"] = False
|
||||
assert campaign_completion(
|
||||
{"pure": pure, "code": code}, puzzles, manifest, "both"
|
||||
)["status"] == "incomplete"
|
||||
Reference in New Issue
Block a user