ai-agent-book 精选快照(<2MB 代码与文档,来自 github.com/bojieli/ai-agent-book)
Build latest book artifacts / build (push) Canceled after 0s
dependency resolution / resolve (3.11) (push) Canceled after 0s
dependency resolution / resolve (3.13) (push) Canceled after 0s
deploy-pages / build (push) Canceled after 0s
deploy-pages / deploy (push) Canceled after 0s
i18n consistency check / check (push) Canceled after 0s
provider adoption tests / test (chapter2/context-compression) (push) Canceled after 0s
provider adoption tests / test (chapter2/prompt-injection) (push) Canceled after 0s
provider adoption tests / test (chapter2/system-hint) (push) Canceled after 0s
provider adoption tests / test (chapter3/log-sanitization) (push) Canceled after 0s
web-search-agent tests / test (push) Canceled after 0s
web-search-agent tests / agentbook (push) Canceled after 0s
Build latest book artifacts / build (push) Canceled after 0s
dependency resolution / resolve (3.11) (push) Canceled after 0s
dependency resolution / resolve (3.13) (push) Canceled after 0s
deploy-pages / build (push) Canceled after 0s
deploy-pages / deploy (push) Canceled after 0s
i18n consistency check / check (push) Canceled after 0s
provider adoption tests / test (chapter2/context-compression) (push) Canceled after 0s
provider adoption tests / test (chapter2/prompt-injection) (push) Canceled after 0s
provider adoption tests / test (chapter2/system-hint) (push) Canceled after 0s
provider adoption tests / test (chapter3/log-sanitization) (push) Canceled after 0s
web-search-agent tests / test (push) Canceled after 0s
web-search-agent tests / agentbook (push) Canceled after 0s
This commit is contained in:
+252
File diff suppressed because one or more lines are too long
@@ -0,0 +1,51 @@
|
||||
{
|
||||
"schema_version": "chapter3-evidence-v1",
|
||||
"experiment": "3-3",
|
||||
"run_id": "20260729T181925Z-3_3-cd097048",
|
||||
"created_at": "2026-07-29T18:19:25.261714+00:00",
|
||||
"status": "partial",
|
||||
"run_dir": "/Users/boj/book/ai-agent-book/chapter3/log-sanitization/validation/runs/20260729T181925Z-3_3-cd097048",
|
||||
"artifacts": {
|
||||
"evidence.json": "7eea7d8235ca381b2cc0354b857968bc08c735443d843f641d95d491fd8e203c",
|
||||
"receipts.json": "6ea59a988a86a659fbb83032d8f7c6df5ffb1120e70e86538251b78cd2281c56"
|
||||
},
|
||||
"inputs": [],
|
||||
"summary": {
|
||||
"regex": {
|
||||
"cases": 1,
|
||||
"mean_precision": 1.0,
|
||||
"mean_recall": 0.5,
|
||||
"mean_typed_exact": 0.5,
|
||||
"residual_leaks": 1,
|
||||
"mean_utility": 1.0,
|
||||
"mean_latency_ms": 0.037
|
||||
},
|
||||
"llm": {
|
||||
"cases": 1,
|
||||
"mean_precision": 1.0,
|
||||
"mean_recall": 1.0,
|
||||
"mean_typed_exact": 0.0,
|
||||
"residual_leaks": 0,
|
||||
"mean_utility": 1.0,
|
||||
"mean_latency_ms": 3959.616
|
||||
},
|
||||
"hybrid": {
|
||||
"cases": 1,
|
||||
"mean_precision": 1.0,
|
||||
"mean_recall": 1.0,
|
||||
"mean_typed_exact": 1.0,
|
||||
"residual_leaks": 0,
|
||||
"mean_utility": 1.0,
|
||||
"mean_latency_ms": 4287.079
|
||||
}
|
||||
},
|
||||
"acceptance": {
|
||||
"local_model": true,
|
||||
"qwen3_model": true,
|
||||
"structured_type_location_confidence": true,
|
||||
"structured_semistructured_natural_language_cases": false,
|
||||
"regex_llm_hybrid_compared": true,
|
||||
"leakage_utility_latency_measured": true,
|
||||
"passed": false
|
||||
}
|
||||
}
|
||||
+176
@@ -0,0 +1,176 @@
|
||||
[
|
||||
{
|
||||
"purpose": "llm:structured_cn",
|
||||
"provider": "ollama-local",
|
||||
"endpoint": "http://127.0.0.1:11434",
|
||||
"model": "qwen3:0.6b",
|
||||
"request": {
|
||||
"model": "qwen3:0.6b",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": "You detect sensitive information in logs. Return exact substrings only.\nTypes include cn_id_card, cn_phone, us_ssn, credit_card, password, secret,\naddress, medical_record, medical_diagnosis, treatment, passport, bank_account,\nrouting_number, bearer_token, email, private_key, and api_key. Distinguish a\nreal disclosed value from a field name, instruction, product code, metric, or\nexplicit decoy. start/end are zero-based Python slice offsets and confidence is\n0..1. Return JSON matching the schema; never redact or paraphrase the value."
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "客户身份证号11010519491231002X,手机号13800138000。"
|
||||
}
|
||||
],
|
||||
"format": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"findings": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"type": {
|
||||
"type": "string"
|
||||
},
|
||||
"value": {
|
||||
"type": "string"
|
||||
},
|
||||
"start": {
|
||||
"type": "integer"
|
||||
},
|
||||
"end": {
|
||||
"type": "integer"
|
||||
},
|
||||
"confidence": {
|
||||
"type": "number"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"type",
|
||||
"value",
|
||||
"start",
|
||||
"end",
|
||||
"confidence"
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"findings"
|
||||
]
|
||||
},
|
||||
"options": {
|
||||
"temperature": 0,
|
||||
"seed": 37,
|
||||
"num_predict": 1200
|
||||
},
|
||||
"stream": false
|
||||
},
|
||||
"response": {
|
||||
"model": "qwen3:0.6b",
|
||||
"created_at": "2026-07-29T18:19:20.944483Z",
|
||||
"done": true,
|
||||
"done_reason": "stop",
|
||||
"total_duration": 3957502042,
|
||||
"load_duration": 57011625,
|
||||
"prompt_eval_count": 445,
|
||||
"prompt_eval_duration": 63544667,
|
||||
"eval_count": 76,
|
||||
"eval_duration": 335304839,
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": "{\"findings\":[{\"type\":\"secret\",\"value\":\"11010519491231002X\",\"start\":0,\"end\":18,\"confidence\":1},{\"type\":\"secret\",\"value\":\"13800138000\",\"start\":0,\"end\":11,\"confidence\":1}]}",
|
||||
"thinking": "好的,我现在需要处理用户提供的客户信息,并检查是否有敏感数据。用户给出的客户身份证号是11010519491231002X,手机号是13800138000。首先,我要确认这些数据是否符合敏感信息的定义。\n\n根据用户提供的类型,敏感信息包括身份证号、手机号等。身份证号有18位,而手机号是11位,但用户提供的都是18位和11位,这可能是一个错误,因为通常身份证号是18位,而手机号通常是11位。不过用户可能输入时有误,或者需要确认是否正确。同样,手机号是13800138000,共11位,这可能是一个错误,但用户可能希望保留。\n\n接下来,我需要检查是否有其他可能的字段或类型被提及,但用户的信息中只有这两个部分。因此,正确的做法是返回这两个部分的值,但需要确保它们是敏感信息。例如,身份证号和手机号都是敏感信息,因此应该直接返回它们的值,而无需额外处理。\n\n最后,确保返回的JSON格式正确,没有其他字段或错误信息,并且只包含敏感数据。这样用户可以清楚知道信息是否被泄露。\n",
|
||||
"images": null,
|
||||
"tool_calls": null
|
||||
}
|
||||
},
|
||||
"latency_ms": 3959.616,
|
||||
"usage": {
|
||||
"prompt_tokens": 445,
|
||||
"completion_tokens": 76
|
||||
}
|
||||
},
|
||||
{
|
||||
"purpose": "hybrid-after-regex:structured_cn",
|
||||
"provider": "ollama-local",
|
||||
"endpoint": "http://127.0.0.1:11434",
|
||||
"model": "qwen3:0.6b",
|
||||
"request": {
|
||||
"model": "qwen3:0.6b",
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": "You detect sensitive information in logs. Return exact substrings only.\nTypes include cn_id_card, cn_phone, us_ssn, credit_card, password, secret,\naddress, medical_record, medical_diagnosis, treatment, passport, bank_account,\nrouting_number, bearer_token, email, private_key, and api_key. Distinguish a\nreal disclosed value from a field name, instruction, product code, metric, or\nexplicit decoy. start/end are zero-based Python slice offsets and confidence is\n0..1. Return JSON matching the schema; never redact or paraphrase the value."
|
||||
},
|
||||
{
|
||||
"role": "user",
|
||||
"content": "客户身份证号11010519491231002X,手机号[REDACTED_CN_PHONE]。"
|
||||
}
|
||||
],
|
||||
"format": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"findings": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"type": {
|
||||
"type": "string"
|
||||
},
|
||||
"value": {
|
||||
"type": "string"
|
||||
},
|
||||
"start": {
|
||||
"type": "integer"
|
||||
},
|
||||
"end": {
|
||||
"type": "integer"
|
||||
},
|
||||
"confidence": {
|
||||
"type": "number"
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"type",
|
||||
"value",
|
||||
"start",
|
||||
"end",
|
||||
"confidence"
|
||||
]
|
||||
}
|
||||
}
|
||||
},
|
||||
"required": [
|
||||
"findings"
|
||||
]
|
||||
},
|
||||
"options": {
|
||||
"temperature": 0,
|
||||
"seed": 37,
|
||||
"num_predict": 1200
|
||||
},
|
||||
"stream": false
|
||||
},
|
||||
"response": {
|
||||
"model": "qwen3:0.6b",
|
||||
"created_at": "2026-07-29T18:19:25.232685Z",
|
||||
"done": true,
|
||||
"done_reason": "stop",
|
||||
"total_duration": 4285037500,
|
||||
"load_duration": 58636292,
|
||||
"prompt_eval_count": 565,
|
||||
"prompt_eval_duration": 80994959,
|
||||
"eval_count": 75,
|
||||
"eval_duration": 335223796,
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": "{\"findings\":[{\"type\":\"cn_id_card\",\"value\":\"11010519491231002X\",\"start\":0,\"end\":18,\"confidence\":1},{\"type\":\"cn_phone\",\"value\":\"[REDACTED_CN_PHONE]\",\"start\":0,\"end\":0,\"confidence\":1}]}",
|
||||
"thinking": "好的,我现在需要处理用户提供的敏感信息,并按照要求返回精确的子字符串。用户给出的客户身份证号是11010519491231002X,手机号是[REDACTED_CN_PHONE]。首先,我要确认用户的需求,他们希望返回的值是身份证号和手机号中的敏感字段,但不包括其他信息。\n\n首先,身份证号的长度是18位,而用户提供的身份证号是11010519491231002X,这明显超过了标准的18位,所以可能用户输入时有误,或者存在其他问题。不过根据用户的要求,只要提取出身份证号和手机号中的敏感字段,不管是否正确,都要返回。因此,我需要检查身份证号中的敏感字段,比如可能的身份证号字段,但用户提供的身份证号是11010519491231002X,其中包含的字符是11010519491231002X,这可能是一个身份证号的前几位,但需要确认是否符合标准。不过根据用户的要求,只要提取出这些字段,不管是否正确,所以直接返回身份证号中的敏感部分,即11010519491231002X,手机号中的[REDACTED_CN_PHONE],但需要确保没有其他字段被包含。\n\n接下来,检查手机号中的[REDACTED_CN_PHONE],这里需要确认是否正确,但用户已经标记为[REDACTED_CN_PHONE],所以直接返回这个部分。同时,确保不包含其他字段,比如地址、医疗记录等,只保留身份证号和手机号中的敏感信息。因此,最终的输出应该是包含身份证号和手机号的字段,但不包含其他信息。\n",
|
||||
"images": null,
|
||||
"tool_calls": null
|
||||
}
|
||||
},
|
||||
"latency_ms": 4287.042,
|
||||
"usage": {
|
||||
"prompt_tokens": 565,
|
||||
"completion_tokens": 75
|
||||
}
|
||||
}
|
||||
]
|
||||
+1224
File diff suppressed because one or more lines are too long
@@ -0,0 +1,51 @@
|
||||
{
|
||||
"schema_version": "chapter3-evidence-v1",
|
||||
"experiment": "3-3",
|
||||
"run_id": "20260729T185603Z-3_3-7f84f416",
|
||||
"created_at": "2026-07-29T18:56:03.514546+00:00",
|
||||
"status": "passed",
|
||||
"run_dir": "/Users/boj/book/ai-agent-book/chapter3/log-sanitization/validation/runs/20260729T185603Z-3_3-7f84f416",
|
||||
"artifacts": {
|
||||
"evidence.json": "9a07a5b4e99b7f8468170a3c4593be2f7b4a703e2ac6fd28a9d3390c49c9f7f7",
|
||||
"receipts.json": "3faf1e5e59ed7679d5f26ee23e3c61081566e7391af5ad2ab9d9a927afcf4e86"
|
||||
},
|
||||
"inputs": [],
|
||||
"summary": {
|
||||
"regex": {
|
||||
"cases": 12,
|
||||
"mean_precision": 0.4166666666666666,
|
||||
"mean_recall": 0.5,
|
||||
"mean_typed_exact": 0.4166666666666667,
|
||||
"residual_leaks": 9,
|
||||
"mean_utility": 0.954197478750736,
|
||||
"mean_latency_ms": 0.09875
|
||||
},
|
||||
"llm": {
|
||||
"cases": 12,
|
||||
"mean_precision": 0.5138888888888888,
|
||||
"mean_recall": 0.6944444444444445,
|
||||
"mean_typed_exact": 0.3333333333333333,
|
||||
"residual_leaks": 3,
|
||||
"mean_utility": 0.7548189868403488,
|
||||
"mean_latency_ms": 3010.9979999999996
|
||||
},
|
||||
"hybrid": {
|
||||
"cases": 12,
|
||||
"mean_precision": 0.4861111111111111,
|
||||
"mean_recall": 0.7777777777777777,
|
||||
"mean_typed_exact": 0.5,
|
||||
"residual_leaks": 2,
|
||||
"mean_utility": 0.8744827477131053,
|
||||
"mean_latency_ms": 2951.5082500000003
|
||||
}
|
||||
},
|
||||
"acceptance": {
|
||||
"local_model": true,
|
||||
"qwen3_model": true,
|
||||
"structured_type_location_confidence": true,
|
||||
"structured_semistructured_natural_language_cases": true,
|
||||
"regex_llm_hybrid_compared": true,
|
||||
"leakage_utility_latency_measured": true,
|
||||
"passed": true
|
||||
}
|
||||
}
|
||||
+2090
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user