From 083a4ad7a47701294cc1e6ba0156cff8ecf10028 Mon Sep 17 00:00:00 2001 From: Pinghua Date: Tue, 9 Jun 2026 13:32:40 +0800 Subject: [PATCH] initial --- .claude/agents/code-review-agent.md | 65 +++++ .claude/settings.json | 25 ++ scripts/review_pr.py | 401 ++++++++++++++++++++++++++++ 3 files changed, 491 insertions(+) create mode 100644 .claude/agents/code-review-agent.md create mode 100644 .claude/settings.json create mode 100644 scripts/review_pr.py diff --git a/.claude/agents/code-review-agent.md b/.claude/agents/code-review-agent.md new file mode 100644 index 0000000..e123fc6 --- /dev/null +++ b/.claude/agents/code-review-agent.md @@ -0,0 +1,65 @@ +--- +name: code-review-agent +description: "Code Review Agent: 通过 Gitea API 获取 PR diff,分析代码变更,发布 review 到 Gitea PR。" +--- + +# Code-Review Agent + +**你是 Code-Review Agent。你的职责是审查 PR 代码变更,提供专业、建设性的反馈。** + +## 工作方式 + +1. 读取提供的 PR diff 文件(路径在 prompt 中) +2. 逐文件分析代码变更 +3. 识别问题,按严重程度分类 +4. 输出结构化 JSON 供 CI 脚本解析并发布到 Gitea + +## 审查标准 + +### 严重(必须修复) +- 安全漏洞:注入、XSS、密钥/Token 泄露、权限绕过、目录遍历 +- 逻辑错误:条件判断错误、空值解引用、类型不匹配 +- 数据一致性风险:事务缺失、竞态条件 +- 功能缺陷:明显与 PR 描述不符的实现 + +### 中等(建议修复) +- 错误处理不完整、异常被吞没 +- 性能问题:不必要的循环、N+1 查询 +- 测试覆盖不足(新增代码无对应测试) +- 配置硬编码或环境相关隐患 + +### 轻微(可选优化) +- 命名不清晰或不符合项目约定 +- 代码重复 +- 无用的 import 或死代码 +- 注释与实际逻辑不一致 + +## 输出格式 + +**必须**输出一个 JSON 对象。不要输出 JSON 之外的任何内容。 + +{ + "event": "COMMENT", + "body": "## Code Review 总结\n\n### 概述\n...\n\n### 发现的问题\n- [严重] ...\n- [中等] ...\n\n### 建议\n...\n\n---\n*🤖 由 Code-Review Agent 自动生成*", + "comments": [ + {"path": "src/example.py", "body": "建议在此处对 None 做防御性检查", "line": 42} + ] +} + +- `event`(参见): + - `"APPROVED"` — 无问题,建议合并 + - `"REQUEST_CHANGES"` — 存在严重问题,必须修复后才能合并 + - `"COMMENT"` — 有建议但非阻塞性 +- `body`:Markdown 格式的完整 review 总结(必填),末尾附带 Agent 签名 +- `comments`:逐行评论数组(可选,无行级评论时为空数组 `[]`) + - `path`:文件相对路径(与 diff 中的路径一致) + - `body`:评论内容 + - `line`:**新文件**中的行号(注意是 new_file 的行号,不是 old_file 的行号) + +## 关键原则 + +1. **Be specific** — 每条评论必须引用具体的文件路径和行号 +2. **Be constructive** — 不仅指出问题,还要给出改进建议 +3. **Don't nitpick blindly** — 遵循项目现有风格,不要强制个人偏好 +4. **Prioritize** — 安全性 > 正确性 > 可维护性 > 风格 +5. **Review the diff, not the whole file** — 只审查变更部分,不对整个文件发表意见 diff --git a/.claude/settings.json b/.claude/settings.json new file mode 100644 index 0000000..9b959a7 --- /dev/null +++ b/.claude/settings.json @@ -0,0 +1,25 @@ +{ + "permissionMode": "bypass", + "permissions": { + "allow": [ + "Bash(git *)", + "Bash(python *)", + "Bash(claude *)", + "Bash(ls *)", + "Bash(mkdir *)", + "Bash(rm *)", + "Bash(cp *)", + "Bash(mv *)", + "Bash(cat *)", + "Bash(echo *)", + "Bash(which *)" + ] + }, + "autoMode": { + "allow": [ + "$defaults", + "Reading and analyzing PR diff files for code review", + "Producing structured JSON review output" + ] + } +} diff --git a/scripts/review_pr.py b/scripts/review_pr.py new file mode 100644 index 0000000..c63b540 --- /dev/null +++ b/scripts/review_pr.py @@ -0,0 +1,401 @@ +"""CI orchestrator for code-review-agent. + +Fetches a PR diff from Gitea, runs Claude Code to analyze it, +and posts the review back to the Gitea PR. + +Usage: + python scripts/review_pr.py \ + --target-repo owner/repo \ + --pr 42 \ + --gitea-url https://gitea.example.com \ + --api-token +""" + +import argparse +import json +import os +import re +import shutil +import subprocess +import sys +import urllib.error +import urllib.request + +MAX_DIFF_CHARS = 200_000 # ~50k words, safe for Claude context + + +def _find_claude(): + """Resolve the claude CLI binary path. + + Checks (in order): + 1. CLAUDE_BIN env var (explicit override) + 2. ``claude`` on PATH + 3. Common install locations (npm global, homebrew, etc.) + Returns the path string, or None. + """ + # 1. Explicit override + env_bin = os.environ.get("CLAUDE_BIN", "") + if env_bin: + if os.path.isfile(env_bin) and os.access(env_bin, os.X_OK): + return env_bin + print(f"WARNING: CLAUDE_BIN={env_bin} is not executable, falling back.", + file=sys.stderr) + + # 2. On PATH + path_bin = shutil.which("claude") + if path_bin: + return path_bin + + # 3. Common install locations + candidates = [ + os.path.expanduser("~/.npm-global/bin/claude"), + "/usr/local/bin/claude", + "/usr/bin/claude", + "/home/linuxbrew/.linuxbrew/bin/claude", + ] + # Also try npm global prefix + try: + result = subprocess.run( + ["npm", "config", "get", "prefix"], + capture_output=True, text=True, timeout=5, + ) + npm_prefix = result.stdout.strip() + if npm_prefix: + candidates.append(os.path.join(npm_prefix, "bin", "claude")) + except Exception: + pass + + for candidate in candidates: + if os.path.isfile(candidate) and os.access(candidate, os.X_OK): + return candidate + + return None + + +def _check_claude(): + claude_bin = _find_claude() + if not claude_bin: + print( + "ERROR: claude CLI not found. Install Claude Code or set CLAUDE_BIN env var.", + file=sys.stderr, + ) + sys.exit(1) + + missing = [] + for var in ("ANTHROPIC_BASE_URL", "ANTHROPIC_AUTH_TOKEN"): + if not os.environ.get(var): + missing.append(var) + if missing: + print( + f"ERROR: Required env vars not set: {', '.join(missing)}. " + f"Set them in the CI workflow or shell environment.", + file=sys.stderr, + ) + sys.exit(1) + + return claude_bin + + +def _api_req(gitea_url, api_token, method, path): + """Send a Gitea API request, return parsed JSON or raw bytes.""" + url = f"{gitea_url}/api/v1/repos{path}" + req = urllib.request.Request(url, method=method) + req.add_header("Authorization", f"token {api_token}") + req.add_header("Content-Type", "application/json") + + try: + with urllib.request.urlopen(req) as resp: + raw = resp.read() + if not raw: + return {} if path.endswith("/merge") else None + content_type = resp.headers.get("Content-Type", "") + if "application/json" in content_type: + return json.loads(raw) + return raw # raw bytes for .diff endpoints + except urllib.error.HTTPError as e: + body = e.read().decode(errors="replace") + print(f"API Error {e.code} on {method} {path}: {body}", file=sys.stderr) + sys.exit(1) + + +def fetch_pr_metadata(gitea_url, api_token, target_repo, pr_num): + """Fetch PR details: title, description, head SHA.""" + pr = _api_req(gitea_url, api_token, "GET", f"/{target_repo}/pulls/{pr_num}") + return { + "number": pr["number"], + "title": pr["title"], + "body": pr.get("body", ""), + "head_sha": pr.get("head", {}).get("sha", ""), + "html_url": pr.get("html_url", ""), + } + + +def fetch_pr_diff(gitea_url, api_token, target_repo, pr_num): + """Fetch raw unified diff for a PR.""" + raw = _api_req(gitea_url, api_token, "GET", f"/{target_repo}/pulls/{pr_num}.diff") + return raw.decode("utf-8", errors="replace") if isinstance(raw, bytes) else "" + + +def fetch_pr_files(gitea_url, api_token, target_repo, pr_num): + """Fetch changed file list with stats.""" + files = _api_req(gitea_url, api_token, "GET", f"/{target_repo}/pulls/{pr_num}/files") + result = [] + for f in files: + result.append({ + "filename": f["filename"], + "status": f["status"], + "additions": f.get("additions", 0), + "deletions": f.get("deletions", 0), + }) + return result + + +def format_files_summary(files): + """Build a one-line-per-file summary for the Claude prompt.""" + lines = [] + for f in files: + lines.append( + f" {f['status']:7} {f['filename']} " + f"(+{f['additions']} -{f['deletions']})" + ) + return "\n".join(lines) + + +def build_prompt(meta, files, diff_path): + """Build the prompt string for Claude.""" + files_summary = format_files_summary(files) + total_additions = sum(f["additions"] for f in files) + total_deletions = sum(f["deletions"] for f in files) + + return ( + f"You are reviewing PR #{meta['number']} in repository.\n\n" + f"PR Title: {meta['title']}\n" + f"PR Description:\n{meta.get('body', '(no description)')}\n\n" + f"Changed Files ({len(files)} files, +{total_additions} -{total_deletions}):\n" + f"{files_summary}\n\n" + f"The complete diff has been saved to: {diff_path}\n" + f"Read that file to see every line changed.\n\n" + f"Analyze the diff thoroughly. Focus on: security, correctness, " + f"error handling, performance, and code quality — in that order.\n\n" + f"Output your review as a single JSON object inside a ```json code block. " + f"Do NOT output anything else." + ) + + +def run_claude(claude_bin, agent_path, project_root, prompt): + """Run claude -p with the agent definition, return stdout.""" + cmd = [ + claude_bin, "-p", + "--agent", agent_path, + "--permission-mode", "acceptEdits", + "--verbose", + prompt, + ] + print(f"[review_pr] Running: {' '.join(cmd[:5])} ...", file=sys.stderr) + proc = subprocess.Popen( + cmd, + cwd=project_root, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + encoding="utf-8", + ) + stdout_lines = [] + for line in proc.stdout: + line = line.rstrip("\n") + stdout_lines.append(line) + if line: + print(f"[claude] {line}", file=sys.stderr) + + proc.wait(timeout=600) + # Drain remaining stderr + stderr_output = proc.stderr.read() + if stderr_output: + print(stderr_output, file=sys.stderr) + + if proc.returncode != 0: + print(f"Claude exited with code {proc.returncode}", file=sys.stderr) + sys.exit(1) + return "\n".join(stdout_lines) + + +def extract_json_from_output(stdout): + """Extract the JSON block from Claude's output.""" + # Try ```json ... ``` block first + match = re.search(r"```json\s*([\s\S]*?)\s*```", stdout) + if match: + return json.loads(match.group(1)) + + # Try bare JSON object + match = re.search(r'\{[\s\S]*"event"[\s\S]*\}', stdout) + if match: + return json.loads(match.group(0)) + + print("ERROR: Could not find JSON in Claude output. Raw stdout:", file=sys.stderr) + print(stdout[:3000], file=sys.stderr) + sys.exit(1) + + +def validate_review(review): + """Validate the review JSON structure.""" + if not isinstance(review, dict): + print(f"ERROR: review is not a JSON object, got {type(review).__name__}", file=sys.stderr) + print(f"Raw: {json.dumps(review, ensure_ascii=False)[:2000]}", file=sys.stderr) + sys.exit(1) + + event = review.get("event", "") + if event not in ("APPROVED", "REQUEST_CHANGES", "COMMENT", ""): + print(f"ERROR: invalid event '{event}'", file=sys.stderr) + sys.exit(1) + + body = review.get("body", "") + if not body or not isinstance(body, str): + print(f"ERROR: review.body is required and must be a string.", file=sys.stderr) + print(f"Got type: {type(body).__name__}, value: {json.dumps(body, ensure_ascii=False)[:500]}", file=sys.stderr) + print(f"Full review keys: {list(review.keys())}", file=sys.stderr) + sys.exit(1) + + comments = review.get("comments", []) + if not isinstance(comments, list): + print("ERROR: review.comments must be an array", file=sys.stderr) + sys.exit(1) + + for i, c in enumerate(comments): + if not isinstance(c, dict): + print(f"ERROR: comment[{i}] is not an object", file=sys.stderr) + sys.exit(1) + if "path" not in c or "body" not in c: + print(f"ERROR: comment[{i}] missing 'path' or 'body'", file=sys.stderr) + sys.exit(1) + + return True + + +def post_review(gitea_url, api_token, target_repo, pr_num, review): + """Post a PR review to Gitea.""" + payload = { + "body": review["body"], + "event": review.get("event", "COMMENT"), + } + comments = review.get("comments", []) + if comments: + # Translate {line} -> {new_line} for Gitea API + gitea_comments = [] + for c in comments: + gc = { + "path": c["path"], + "body": c["body"], + } + if c.get("line"): + gc["new_line"] = c["line"] + if c.get("old_line"): + gc["old_line"] = c["old_line"] + gitea_comments.append(gc) + payload["comments"] = gitea_comments + + body = json.dumps(payload).encode("utf-8") + url = f"{gitea_url}/api/v1/repos/{target_repo}/pulls/{pr_num}/reviews" + req = urllib.request.Request(url, data=body, method="POST") + req.add_header("Authorization", f"token {api_token}") + req.add_header("Content-Type", "application/json") + + try: + with urllib.request.urlopen(req) as resp: + result = json.loads(resp.read()) + print(f"Review posted: {result.get('html_url', result.get('url', 'unknown'))}") + return result + except urllib.error.HTTPError as e: + err_body = e.read().decode(errors="replace") + print(f"Failed to post review: {e.code} - {err_body}", file=sys.stderr) + sys.exit(1) + + +def main(): + parser = argparse.ArgumentParser( + description="Code Review Agent — fetch PR, analyze with Claude, post review" + ) + parser.add_argument("--target-repo", required=True, + help="Target repository path (e.g. owner/repo)") + parser.add_argument("--pr", type=int, required=True, + help="PR number to review") + parser.add_argument("--gitea-url", required=True, + help="Gitea instance URL") + parser.add_argument("--api-token", required=True, + help="Gitea API token with read:repository + write:repository") + args = parser.parse_args() + + claude_bin = _check_claude() + + project_root = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) + agent_path = os.path.join(project_root, ".claude", "agents", "code-review-agent.md") + + if not os.path.exists(agent_path): + print(f"ERROR: agent definition not found at {agent_path}", file=sys.stderr) + sys.exit(1) + + # ── 1. Fetch PR metadata ───────────────────────────────────────────── + print(f"[review_pr] Fetching PR #{args.pr} metadata from {args.target_repo}...") + meta = fetch_pr_metadata(args.gitea_url, args.api_token, args.target_repo, args.pr) + print(f" Title: {meta['title']}") + + # ── 2. Fetch changed files ─────────────────────────────────────────── + print("[review_pr] Fetching changed files...") + files = fetch_pr_files(args.gitea_url, args.api_token, args.target_repo, args.pr) + print(f" {len(files)} file(s) changed") + + if not files: + print("No files changed — skipping review.") + return + + # ── 3. Fetch diff ──────────────────────────────────────────────────── + print("[review_pr] Fetching diff...") + diff_text = fetch_pr_diff(args.gitea_url, args.api_token, args.target_repo, args.pr) + + if not diff_text.strip(): + print("Empty diff — skipping review.") + return + + if len(diff_text) > MAX_DIFF_CHARS: + print( + f"Warning: diff is {len(diff_text)} chars " + f"(>{MAX_DIFF_CHARS}). Review may be incomplete.", + file=sys.stderr, + ) + diff_text = diff_text[:MAX_DIFF_CHARS] + + print(f" Diff: {len(diff_text)} chars, ~{diff_text.count(chr(10))} lines") + + # ── 4. Write diff inside project_root so claude CLI can read it ───── + diff_dir = os.path.join(project_root, ".reviews") + os.makedirs(diff_dir, exist_ok=True) + diff_path = os.path.join(diff_dir, f"pr{args.pr}.diff") + with open(diff_path, "w", encoding="utf-8") as f: + f.write(diff_text) + print(f" Diff saved to: {diff_path}") + + try: + # ── 5. Build prompt & run Claude ───────────────────────────────── + prompt = build_prompt(meta, files, diff_path) + print("[review_pr] Invoking Claude for analysis...") + stdout = run_claude(claude_bin, agent_path, project_root, prompt) + + # ── 6. Parse Claude output ─────────────────────────────────────── + review = extract_json_from_output(stdout) + validate_review(review) + print(f" Review event: {review.get('event', 'COMMENT')}") + print(f" Body length: {len(review['body'])} chars") + print(f" Inline comments: {len(review.get('comments', []))}") + + # ── 7. Post review to Gitea ────────────────────────────────────── + print("[review_pr] Posting review to Gitea...") + post_review(args.gitea_url, args.api_token, args.target_repo, args.pr, review) + + print("[review_pr] Done.") + finally: + # Clean up temp file + if os.path.exists(diff_path): + os.unlink(diff_path) + + +if __name__ == "__main__": + main()