Files
cmbuyer/scripts/validate_agent_context.py
QiuSWandClaude Opus 5 3c458bcf5b fix(tasks): make status git-authoritative, not Vikunja
修掉一处会让并行保护失效的设计缺口:AGENTS.md 声明「状态」权威在 Vikunja,
但离线门禁的检查 1 读的是 frontmatter 的 status,两者之间没有任何同步。
有人在看板上拖了卡片,frontmatter 不变,门禁就用过期数据放行了。

status 的权威定为 git frontmatter,理由与 write_paths 相同:它是判定
「两个活跃任务不得写同一路径」的输入,而门禁必须离线可跑;status 变更
决定谁能碰哪些文件,本来就该产生 commit。

- AGENTS.md:把 status 从 Vikunja 权威列表移入 git 原生表,并说明 bucket
  与 done 仅为人类视图,不一致时以 git 为准
- agent-context.json:tracker.git_native_fields 增加 status
- validate_agent_context.py:强制 git_native_fields 必须含 status,
  已用反例验证缺失时报错
- vikunja_export.py:新增只读漂移检测,导出时比对 frontmatter status 与
  Vikunja done,不一致则提示;只提示不修正,不反向写回
- docs/tasks/README.md、T-008 方案第 1/2 节:同步口径

顺带解决了窄 token 时期的 bucket 401 遗留问题——status 权威在 git,
agent 不再依赖 bucket 移动来表达状态。

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-03 17:17:35 +08:00

421 lines
16 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Validate the agent context manifest with the Python standard library."""
from __future__ import annotations
import argparse
import fnmatch
import hashlib
import itertools
import json
import re
import sys
from pathlib import Path, PurePosixPath
from typing import Any
EXPECTED_SCHEMA = "docs/agent-context.schema.json"
REQUIRED_TOP_LEVEL = {
"schema",
"schema_version",
"authority",
"bootstrap",
"routes",
"tasks",
"shared_documents",
"tracker",
"refresh",
"degraded_mode",
}
TRACKER_KEYS = {"kind", "project", "export_script", "direction", "git_native_fields"}
ACTIVE_STATUS = "DOING"
# 导出区块标记。区块内是 Vikunja 投影,区块外是 git 原生内容。
# sha256 覆盖两行标记之间的正文,用于发现手工改动投影。
EXPORT_BEGIN = re.compile(
r"<!-- BEGIN VIKUNJA EXPORT id=(?P<id>\d+) synced=(?P<synced>\S+) "
r"sha256=(?P<sha>[0-9a-f]{64}) -->"
)
EXPORT_END = "<!-- END VIKUNJA EXPORT -->"
REQUIRED_BOOTSTRAP = {
"AGENTS.md",
"docs/00-ai-start-here.md",
"docs/05-coding-rules.md",
"docs/current-state.md",
}
TASK_PATH_KEYS = {"roadmap", "directory", "template"}
SENSITIVE_KEY = re.compile(r"(?:token|password|secret|credential)", re.IGNORECASE)
URI_SCHEME = re.compile(r"^[A-Za-z][A-Za-z0-9+.-]*:")
def load_json(path: Path, root: Path, errors: list[str]) -> Any:
try:
return json.loads(path.read_text(encoding="utf-8"))
except FileNotFoundError:
errors.append(f"文件不存在:{display_path(path, root)}")
except json.JSONDecodeError as exc:
errors.append(
f"JSON 语法错误:{display_path(path, root)}:{exc.lineno}:{exc.colno}"
)
return None
def display_path(path: Path, root: Path) -> str:
try:
return path.relative_to(root).as_posix()
except ValueError:
return path.as_posix()
def require_mapping(value: Any, name: str, errors: list[str]) -> dict[str, Any]:
if not isinstance(value, dict):
errors.append(f"{name} 必须是对象。")
return {}
return value
def require_string_list(value: Any, name: str, errors: list[str]) -> list[str]:
if not isinstance(value, list) or not value or not all(
isinstance(item, str) and item for item in value
):
errors.append(f"{name} 必须是非空字符串数组。")
return []
if len(value) != len(set(value)):
errors.append(f"{name} 不得包含重复路径。")
return value
def validate_repo_path(root: Path, value: str, name: str, errors: list[str]) -> None:
path = PurePosixPath(value)
if (
path.is_absolute()
or ".." in path.parts
or "\\" in value
or URI_SCHEME.match(value)
):
errors.append(f"{name} 必须是安全的仓库相对路径:{value}")
return
target = root.joinpath(*path.parts)
if not target.exists():
errors.append(f"{name} 引用路径不存在:{value}")
def find_sensitive_keys(value: Any, location: str, errors: list[str]) -> None:
if isinstance(value, dict):
for key, child in value.items():
child_location = f"{location}.{key}"
if SENSITIVE_KEY.search(key):
errors.append(f"清单不得保存敏感配置字段:{child_location}")
find_sensitive_keys(child, child_location, errors)
elif isinstance(value, list):
for index, child in enumerate(value):
find_sensitive_keys(child, f"{location}[{index}]", errors)
def parse_task_file(path: Path) -> dict[str, Any]:
"""抽取任务文件里门禁需要的字段。只认约定的扁平 frontmatter,不引入 YAML 依赖。"""
text = path.read_text(encoding="utf-8")
parts = text.split("---\n", 2)
front = parts[1] if len(parts) >= 3 else ""
status = re.search(r"^status:[ \t]*(\S+)", front, re.MULTILINE)
task_id = re.search(r"^vikunja_task_id:[ \t]*(\S+)", front, re.MULTILINE)
write_paths: list[str] = []
in_block = False
for line in front.splitlines():
if re.match(r"^write_paths:", line):
in_block = True
continue
if in_block:
item = re.match(r"^[ \t]+-[ \t]+(\S+)", line)
if item:
write_paths.append(item.group(1))
elif line.strip():
break
return {
"path": path,
"status": status.group(1) if status else None,
"vikunja_task_id": task_id.group(1) if task_id else None,
"write_paths": write_paths,
"text": text,
}
def find_export_markers(text: str) -> tuple[int | None, int | None, re.Match | None]:
"""按行定位导出标记,跳过围栏代码块内的同名文本。
任务文件会在代码块里举例说明区块格式,直接做子串匹配会把示例当成真标记。
"""
lines = text.splitlines(keepends=True)
in_fence = False
begin_idx = end_idx = None
begin_match = None
for index, line in enumerate(lines):
if line.lstrip().startswith("```"):
in_fence = not in_fence
continue
if in_fence:
continue
stripped = line.rstrip("\n")
if begin_idx is None:
match = EXPORT_BEGIN.fullmatch(stripped)
if match:
begin_idx, begin_match = index, match
continue
if stripped == EXPORT_END and end_idx is None:
end_idx = index
return begin_idx, end_idx, begin_match
def check_export_block(task: dict[str, Any], errors: list[str], root: Path) -> None:
"""校验导出区块未被手工改动,且与 frontmatter 的 vikunja_task_id 一致。"""
text = task["text"]
name = display_path(task["path"], root)
begin_idx, end_idx, begin = find_export_markers(text)
if begin is None:
if end_idx is not None:
errors.append(f"{name}:有 END VIKUNJA EXPORT 标记但缺少合法的 BEGIN 标记。")
# 尚未迁移的任务文件豁免,避免为让门禁转绿去改属于其他活跃任务的文件。
return
if end_idx is None:
errors.append(f"{name}:有 BEGIN VIKUNJA EXPORT 标记但缺少 END 标记。")
return
if end_idx < begin_idx:
errors.append(f"{name}:END VIKUNJA EXPORT 标记出现在 BEGIN 之前。")
return
lines = text.splitlines(keepends=True)
body = "".join(lines[begin_idx + 1 : end_idx])
actual = hashlib.sha256(body.encode("utf-8")).hexdigest()
if actual != begin.group("sha"):
errors.append(
f"{name}:导出区块内容与标记中的 sha256 不符,"
"说明投影被手工修改。请改 Vikunja 后重新导出,不要直接编辑区块内容。"
)
declared = task["vikunja_task_id"]
if declared is None:
errors.append(f"{name}:存在导出区块但 frontmatter 缺少 vikunja_task_id。")
elif declared != begin.group("id"):
errors.append(
f"{name}:frontmatter 的 vikunja_task_id={declared} "
f"与导出区块的 id={begin.group('id')} 不一致。"
)
def validate_task_files(root: Path, shared_documents: list[str]) -> list[str]:
"""任务文件层面的门禁,全部离线可跑,不访问 Vikunja。"""
errors: list[str] = []
task_dir = root / "docs" / "tasks"
if not task_dir.is_dir():
return [f"任务目录不存在:{display_path(task_dir, root)}"]
tasks = [parse_task_file(p) for p in sorted(task_dir.glob("T-*.md"))]
# 检查 1:同时活跃的任务不得写入同一路径。单写入者是并行的前提。
active = [t for t in tasks if t["status"] == ACTIVE_STATUS]
for left, right in itertools.combinations(active, 2):
overlap = sorted(set(left["write_paths"]) & set(right["write_paths"]))
if overlap:
errors.append(
f"{display_path(left['path'], root)} 与 "
f"{display_path(right['path'], root)} 同为 {ACTIVE_STATUS},"
"write_paths 重叠:" + "、".join(overlap)
)
# 检查 2:通配条目不得悄悄圈走共享文档,共享文档必须逐条显式声明。
for task in tasks:
for pattern in task["write_paths"]:
if "*" not in pattern:
continue
swallowed = sorted(
doc for doc in shared_documents if fnmatch.fnmatch(doc, pattern)
)
if swallowed:
errors.append(
f"{display_path(task['path'], root)}:write_paths 的通配条目 "
f"{pattern} 覆盖了共享文档 " + "、".join(swallowed) +
"。共享文档必须逐条显式列出,便于检查 1 判定排他。"
)
# 检查 3、4:导出区块完整性与连接键一致性。
for task in tasks:
check_export_block(task, errors, root)
return errors
def validate_manifest(root: Path) -> list[str]:
root = root.resolve()
manifest_path = root / "docs" / "agent-context.json"
errors: list[str] = []
manifest = load_json(manifest_path, root, errors)
schema = load_json(root / EXPECTED_SCHEMA, root, errors)
if manifest is None or schema is None:
return errors
if not isinstance(schema, dict) or schema.get("type") != "object":
errors.append("agent-context.schema.json 不是有效的对象 Schema。")
root_object = require_mapping(manifest, "manifest", errors)
actual_keys = set(root_object)
missing = sorted(REQUIRED_TOP_LEVEL - actual_keys)
unexpected = sorted(actual_keys - REQUIRED_TOP_LEVEL)
if missing:
errors.append("缺少顶层字段:" + ", ".join(missing))
if unexpected:
errors.append("存在未知顶层字段:" + ", ".join(unexpected))
if root_object.get("schema") != EXPECTED_SCHEMA:
errors.append(f"schema 必须是 {EXPECTED_SCHEMA}。")
if root_object.get("schema_version") != 2:
errors.append("schema_version 必须为 2。")
authority = require_mapping(root_object.get("authority"), "authority", errors)
for key in ("bootstrap", "framework_templates", "project_facts", "coordination"):
if not isinstance(authority.get(key), str) or not authority[key]:
errors.append(f"authority.{key} 必须是非空字符串。")
bootstrap = require_mapping(root_object.get("bootstrap"), "bootstrap", errors)
always_read = require_string_list(
bootstrap.get("always_read"), "bootstrap.always_read", errors
)
missing_bootstrap = sorted(REQUIRED_BOOTSTRAP - set(always_read))
if missing_bootstrap:
errors.append("bootstrap.always_read 缺少:" + ", ".join(missing_bootstrap))
routes = require_mapping(root_object.get("routes"), "routes", errors)
if not routes:
errors.append("routes 至少需要一个任务类型。")
path_values: list[tuple[str, str]] = [(EXPECTED_SCHEMA, "schema")]
path_values.extend((path, "bootstrap.always_read") for path in always_read)
for route, value in routes.items():
paths = require_string_list(value, f"routes.{route}", errors)
path_values.extend((path, f"routes.{route}") for path in paths)
tasks = require_mapping(root_object.get("tasks"), "tasks", errors)
if set(tasks) != TASK_PATH_KEYS:
errors.append("tasks 必须且只能包含 roadmap、directory、template。")
for key in sorted(TASK_PATH_KEYS):
value = tasks.get(key)
if isinstance(value, str) and value:
path_values.append((value, f"tasks.{key}"))
else:
errors.append(f"tasks.{key} 必须是非空字符串。")
shared_documents = require_string_list(
root_object.get("shared_documents"), "shared_documents", errors
)
if not shared_documents:
errors.append("shared_documents 至少需要登记一份跨任务共享文档。")
path_values.extend((path, "shared_documents") for path in shared_documents)
tracker = require_mapping(root_object.get("tracker"), "tracker", errors)
if set(tracker) != TRACKER_KEYS:
errors.append(
"tracker 必须且只能包含 "
+ "、".join(sorted(TRACKER_KEYS))
+ "。"
)
if tracker.get("kind") != "vikunja":
errors.append("tracker.kind 目前只支持 vikunja。")
if tracker.get("direction") != "tracker_to_git":
errors.append(
"tracker.direction 必须是 tracker_to_git:"
"反向同步会重新引入两个权威,本项目不实现。"
)
if not isinstance(tracker.get("project"), str) or not tracker.get("project"):
errors.append("tracker.project 必须是非空字符串。")
export_script = tracker.get("export_script")
if isinstance(export_script, str) and export_script:
path_values.append((export_script, "tracker.export_script"))
else:
errors.append("tracker.export_script 必须是非空字符串。")
git_native = require_string_list(
tracker.get("git_native_fields"), "tracker.git_native_fields", errors
)
# 这三项的权威必须留在 git:status 与 write_paths 是本脚本判定并行安全的输入,
# 权威搬到远端就等于让离线门禁依赖网络;边界章节搬走则切断安全边界的审计链。
for required in ("status", "write_paths", "boundaries_section"):
if required not in git_native:
errors.append(
f"tracker.git_native_fields 必须包含 {required}:"
"该项迁往 Vikunja 会让离线门禁失去判定依据或切断审计链。"
)
refresh = require_mapping(root_object.get("refresh"), "refresh", errors)
expected_refresh = {
"context_ref": "default_branch_head_sha",
"cache_key": "file_sha",
"unchanged_file": "reuse_within_current_session",
"changed_ref": "reread_manifest_and_routed_documents",
}
if refresh != expected_refresh:
errors.append("refresh 必须使用约定的提交 SHA 与文件 SHA 刷新策略。")
degraded = require_mapping(root_object.get("degraded_mode"), "degraded_mode", errors)
expected_degraded = {
"continue_claimed_task": True,
"claim_new_task": False,
"write_remote_state": False,
}
if degraded != expected_degraded:
errors.append("degraded_mode 必须禁止领取新任务和写入远端状态。")
for value, name in path_values:
validate_repo_path(root, value, name, errors)
find_sensitive_keys(root_object, "manifest", errors)
errors.extend(validate_task_files(root, shared_documents))
return errors
def manifest_summary(root: Path) -> tuple[int, int]:
manifest = json.loads(
(root / "docs" / "agent-context.json").read_text(encoding="utf-8")
)
paths = {manifest["schema"]}
paths.update(manifest["bootstrap"]["always_read"])
for values in manifest["routes"].values():
paths.update(values)
paths.update(manifest["tasks"].values())
paths.update(manifest["shared_documents"])
paths.add(manifest["tracker"]["export_script"])
return len(manifest["routes"]), len(paths)
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(description="校验 Agent 上下文清单。")
parser.add_argument(
"--root",
type=Path,
default=Path(__file__).resolve().parents[1],
help="仓库根目录;默认取脚本上一级。",
)
return parser.parse_args()
def main() -> int:
args = parse_args()
root = args.root.resolve()
if not root.is_dir():
print("ERROR: 仓库根目录不存在。", file=sys.stderr)
return 2
errors = validate_manifest(root)
if errors:
for error in errors:
print(f"ERROR: {error}", file=sys.stderr)
return 1
route_count, path_count = manifest_summary(root)
print(
"agent-context 校验通过:"
f"{route_count} 个任务路由,{path_count} 个有效仓库路径。"
)
return 0
if __name__ == "__main__":
raise SystemExit(main())