#!/usr/bin/env python3 """pragent pilot — minimal AI PR reviewer. Runs as a Gitea Actions step OR is called by the central webhook server (`webhook_server.py`). Fetches a PR diff, asks glm-5.2:cloud (via the on-network headroom proxy, Anthropic /v1/messages format) to review it, and posts the findings back as `pragent-bot` — as a **review summary** plus **inline line comments** with a fenced suggested-fix block (tagged with the file's language so Gitea syntax-highlights it) where the model could produce one and the line anchors cleanly to the post-change file. Features (pilot v2): - **Dedupe / persistence:** Gitea itself is the source of truth. Before reviewing, fetch the PR's existing reviews and look for a hidden `` marker matching this commit. If present, skip (no duplicate review on label-toggle / re-fire). Prior review bodies are fed back to the model as "already said" context so a re-push synthesizes instead of repeating (light version of design §6.1). - **Repo-local focus:** if the repo has a `.pr-review.json` at the PR's head ref, its `focus` / `exclude_paths` / `instructions` / `languages` steer the review. Optional — defaults apply when absent. - **Inline comments + suggestions:** the model emits structured JSON findings with `path`/`line`. We parse the diff hunks to learn which `(path, new_line)` pairs are valid post-change anchors and post each anchored finding as a positional review comment; the `suggestion` field, if non-empty, is wrapped in a fenced code block tagged with the file's language (via `_lang_for_path`) so Gitea syntax-highlights it. Gitea 1.26.x has no GitHub-style "Apply suggestion" button, so a language-tagged block is used for highlighting instead of a ```suggestion fence. Findings that don't anchor (bad line, unchanged file, etc.) are folded into the summary body as plain bullets. Fail-open by design: any error becomes a short "review failed" review comment, and review_pr never raises. Stdlib only — no pip install. Env (CI run() path): GITEA_API base URL of the in-cluster Gitea GITEA_REPOSITORY "owner/repo" of the PR (github.repository) PR_INDEX PR number (github.event.pull_request.number) PR_TITLE PR title PR_BODY PR body (optional) PR_BASE_REF base branch (.pr-review.json is read from here, not the PR head); optional, defaults to the repo default branch PRAGENT_BOT_TOKEN bot access token (repo secret) PRAGENT_SHA head SHA to tag the review OLLAMA_URL headroom proxy URL, e.g. http://model-proxy.internal:8789 OLLAMA_MODEL model id, e.g. glm-5.2:cloud OLLAMA_MAX_TOKENS (optional) output cap, default 8000 DIFF_MAX_CHARS (optional) diff truncation cap, default 150000 """ import base64 import json import os import re import sys import urllib.error import urllib.parse import urllib.request REVIEW_HEADER = "🤖 **AI Review** · pragent pilot · {model} · `{sha}` · Merge confidence: {confidence}" # Hidden marker the dedupe pass scans for. Full sha so a re-push (new sha) is # never mistaken for an already-reviewed commit, and a label-toggle (same sha) # is correctly skipped. SHA_MARKER = "" _SHA_MARKER_RE = re.compile(r"") SEVERITIES = ("critical", "high", "medium", "low", "trivial", "info") # Severity rank — higher = more severe. Used by `apply_repo_config` to drop # findings below `severity_threshold`. critical=4, high=3, medium=2, low=1, # trivial=0, info=-1. SEVERITY_RANK = {"info": -1, "trivial": 0, "low": 1, "medium": 2, "high": 3, "critical": 4} REPO_CONFIG_FILE = ".pr-review.json" # Style → (default max_findings, default severity_threshold). Strict is # terse/high-signal; lenient shows everything; balanced is the default for # unconfigured repos. Repo `.pr-review.json` overrides per-field. STYLE_DEFAULTS: dict[str, tuple[int, str]] = { "strict": (5, "high"), "balanced": (12, "medium"), "lenient": (15, "low"), } # Default provider to compare against in the usage section. The pilot runs on # headroom/glm-5.2:cloud at $0/MTok, so the actual line shows $0.00 — but the # equivalent provider line lets a maintainer see what they would have paid on # Claude/GPT for the same measured tokens. Override with PRAGENT_PRICE_TARGET # (env) or `.pr-review.json:cost_target` (per repo). DEFAULT_PRICE_TARGET = "claude-sonnet-5" # Default roster of paid providers shown in the equivalent-cost table when # `.pr-review.json` does not pin `compare_against`. The pilot is free-tier only, # so this list is the operator's budgeting signal — it answers "what would this # have cost on a mainstream paid API?". Override per-repo via # `.pr-review.json:compare_against` (capped at 12 entries; unknown keys are # dropped with a stderr line at parse time). DEFAULT_COMPARE_AGAINST = ("claude-sonnet-5", "gpt-5", "gemini-2.5-pro", "grok-4.5") SYSTEM_PROMPT = """You are a senior, pragmatic code reviewer. Review the pull request diff below. Report ONLY real, actionable issues: correctness bugs, security problems, risky changes, missing tests for changed behaviour, and breaking API/contract changes. Honour any repo-specific focus / instructions given in the prompt; if focus is given, weight those areas higher, but do not ignore a critical issue outside them. Output STRICT JSON only — no prose, no markdown fences. Shape: { "findings": [ { "severity": "critical|high|medium|low|trivial|info", "path": "file path exactly as it appears in the diff (`+++ b/` side)", "line": , "problem": "one line: what is wrong", "fix": "one line: how to fix it", "suggestion": "" } ], "walkthrough": ["2-6 short bullets, file- or change-grouped, plain prose"], "risk_verdict": "Low|Medium|High|Critical risk: ", "test_coverage": "Tests added" | "Tests changed" | "No tests for behavioral change" | "No test files in repo" } Rules: - `line` MUST be a line number that exists in the post-change version of `path` (i.e. a context line or an added `+` line shown in the diff). Never a removed line. If you are unsure of the exact line, set `line` to the closest context line you CAN see in the diff. - `suggestion` is the literal new code that should replace the flagged line(s). Keep it minimal — just the changed lines, indented as they would appear in the file. Leave it empty ("") if a safe textual replacement is not possible (e.g. a missing test, an architectural note). - `walkthrough`: 2-6 short bullets, file- or change-grouped, plain prose. Default to `[]` when the diff is trivial. Backward compatible: parsers default to `[]` if absent. - `risk_verdict`: exactly one line. Lead with "Low|Medium|High|Critical risk:" followed by a concrete reason. Default to `""` when not applicable. Backward compatible: parsers default to `""` if absent. - `test_coverage`: short string. One of "Tests added" / "Tests changed" / "No tests for behavioral change" / "No test files in repo". Default to `""` when not applicable. Backward compatible: parsers default to `""` if absent. - Skip nitpicks, pure formatting, and praise. At most ~15 findings, highest severity first. - If the diff is clean, output: {"findings": []} - Do NOT repeat anything already covered in "PREVIOUS REVIEWS" — only surface new or still-unresolved issues.""" # --------------------------------------------------------------------------- # Shared render constant retained here for compatibility with the extracted # modules and existing callers. _CONFIDENCE_BADGE = {5: "🟢", 4: "🟢", 3: "🟡", 2: "🟠", 1: "🔴"} # Internal modules provide pure transforms and adapters; this file retains # the orchestration entry point and backwards-compatible symbols. from . import adapters as _adapters from . import analysis as _analysis from . import configuration as _configuration from . import output as _output for _module in (_analysis, _output, _configuration, _adapters): globals().update({ _name: _value for _name, _value in vars(_module).items() if not _name.startswith("__") }) def review_pr( api: str, repo: str, index: str, title: str, body: str, sha: str, token: str, ollama_url: str, model: str, max_tokens: int = 8000, max_chars: int = 150000, base_ref: str = "", ) -> bool: """Run one review and post it as `pragent-bot`. Dedupe: if a prior review already carries this commit's sha marker, skip (no duplicate). Otherwise: fetch repo config + prior-review context, call the model, parse JSON findings, anchor what we can to diff lines, post a review with inline comments + suggestions (unanchored findings → summary bullets). `base_ref`: the PR's base branch. `.pr-review.json` is read from there (not from the PR head) so a PR cannot ship its own reviewer instructions; empty means "the repo's default branch". The opencode engine's measured token/cost usage is always rendered as a `## 🔋 AI usage` section on the review body and an attributed `🪙 ~N tok` line on each inline comment when usage data is available (i.e. when the opencode subprocess returned a `usage` dict). No-op on the ollama fallback (no usage available — `usage` is None). Returns True on success (including a deliberate skip), False on failure (failure note posted when possible). Never raises — fail-open by design. Both the CI `run()` entry point and the central webhook server call this. """ try: # Pre-compute a *fallback* display name for the early-exit paths # (already-reviewed dedupe skip, no-diff-content). We re-resolve # properly after `.pr-review.json` is loaded further down — that # version honours `OPENCODE_MODEL` env > `.pr-review.json:model` > # this fallback. display_model = f"headroom/{model}" reviews = fetch_existing_reviews(api, repo, index, token) # Dedupe: already reviewed this exact commit -> nothing to do. if sha and sha in reviewed_shas(reviews): print(f"pragent: {repo}#{index} sha={sha[:8]} already reviewed, skipping", flush=True) return True raw_diff, _truncated, _orig = fetch_pr_diff(api, repo, index, token, max_chars) if not raw_diff.strip(): post_review(api, repo, index, token, format_review_body("No diff content to review.", display_model, sha)) return True config = fetch_repo_config(api, repo, token, ref=base_ref) prior = compact_prior_reviews(prior_review_bodies(reviews, sha)) # Re-resolve display_model now that .pr-review.json is available — # per-repo override (`.pr-review.json:model`) takes precedence over # the bare OLLAMA_MODEL fallback, with OPENCODE_MODEL env still # winning above both (see `_resolve_display_model`). display_model = _resolve_display_model(model, config) # Trim the diff to +/- hunks plus a narrow context window. The agent # resends the brief prefix every step, so a 25k-char diff becomes # 25k × 30-step × cached-after-step-1 = hundreds of thousands of input # tokens. Default context=1: enough for the reviewer to see what an # added line is replacing; the full file is on disk in the workdir # anyway, so anything more is reading the diff twice. Tunable via # PRAGENT_DIFF_CONTEXT (0 = +/- only; -1 = disable compression). from diff_compress import compress_diff ctx = _int_env("PRAGENT_DIFF_CONTEXT", 1) if ctx < 0: diff = raw_diff compression_note = "" else: diff, orig_chars, kept_chars = compress_diff(raw_diff, context=ctx) if kept_chars < orig_chars: compression_note = ( f"\n\n> _diff compressed: {orig_chars:,} → {kept_chars:,} chars " f"(context={ctx}; PRAGENT_DIFF_CONTEXT to tune)_" ) else: compression_note = "" engine = os.environ.get("PRAGENT_ENGINE", "opencode").strip().lower() review_summary = "" # Static repo-provided context (architecture summary, module map, …) # fetched once from `additional_context_urls` (env + .pr-review.json). # Cheap, cached, capped — see fetch_additional_context. additional_context = fetch_additional_context(_resolve_additional_context_urls(config)) if engine == "opencode": # The review "brain" runs on opencode: it gets the checked-out repo, # the brief, and the pragent agent factory; returns stdout with a # summary + findings JSON. We parse + anchor + post here. import opencode_review # local import keeps the ollama path dep-free # Reuse the display_model resolved above for the subprocess — same # provider-prefixed ref goes to the engine and into the review body. oc_model = display_model # Multi-lens fan-out: when the repo declared `reviewers[]` (or the # operator pinned PRAGENT_REVIEWERS=1), spawn one opencode subprocess # per lens in parallel and synthesize. Falls through to the legacy # single-primary path when neither is set. use_lenses = bool((config or {}).get("reviewers")) or bool( os.environ.get("PRAGENT_REVIEWERS") ) if use_lenses and hasattr(opencode_review, "run_lenses_review"): stdout, usage = opencode_review.run_lenses_review( api=api, repo=repo, index=index, sha=sha, token=token, title=title, body=body, diff=diff, config=config, prior_reviews=prior, model=oc_model, compression_note=compression_note, additional_context=additional_context, ) else: stdout, usage = opencode_review.run( api=api, repo=repo, index=index, sha=sha, token=token, title=title, body=body, diff=diff, config=config, prior_reviews=prior, model=oc_model, compression_note=compression_note, additional_context=additional_context, ) review_summary, findings, summary_changes, risks, _walkthrough, _risk_verdict, _test_coverage = parse_review_output(stdout) parse_dropped = last_parse_dropped() if not findings and not review_summary: # The findings JSON was missing or malformed. Don't discard the # run: salvage the prose, keep the usage report (the tokens were # spent either way), and log enough of the raw output to # diagnose why the agent went off-format. print( f"pragent: {repo}#{index} sha={sha[:8]} unparseable output " f"({len(stdout)} chars); tail: {stdout[-600:]!r}", file=sys.stderr, flush=True, ) salvaged = salvage_summary(stdout) usage_section = _render_collapsible_usage(usage, display_model, config=config) if usage else "" post_review(api, repo, index, token, format_review_body( salvaged or "AI review produced no parseable output.", display_model, sha, usage_section=usage_section, static_message=(config or {}).get("static_message", ""))) _emit_langfuse( repo=repo, index=index, sha=sha, title=title, model=display_model, usage=usage, findings=[], summary=salvaged, engine=engine, config=config, dropped_count=parse_dropped, ) return True else: user_prompt = build_user_prompt(title, body + compression_note, diff, config, prior, additional_context) raw_findings = call_model(ollama_url, model, SYSTEM_PROMPT, user_prompt, max_tokens) findings = parse_findings(raw_findings) parse_dropped = last_parse_dropped() usage = None # Filter / cap findings per `.pr-review.json` (style, threshold, max, # patterns, exclude_tests). Without this every config knob would be a # no-op — the agent has no view into the config beyond instructions. # The synthetic require_tests finding (if any) is appended here. try: changed_paths = sorted({ f.get("path", "") for f in findings if f.get("path") }) except Exception: changed_paths = [] # Capture cross-lens agreement BEFORE apply_repo_config — by the time # findings land in `review_pr` the `_multi_lens` marker has already # been scrubbed (once by `opencode_review.run_lenses_review`'s # `_`-prefix strip, again by `_normalize_finding`'s 7-key rebuild), # so `merge_confidence` cannot read it off the dict. We scan here as # the convergence point for both engine paths; in practice the kwarg # currently always passes False, but the structural plumbing is # correct for any future code path that preserves the flag. multi_lens = any(f.get("_multi_lens") for f in findings) kept, _dropped = apply_repo_config(findings, config, changed_paths=changed_paths) findings = kept if _dropped: print( f"pragent: {repo}#{index} sha={sha[:8]} filtered " f"{len(_dropped)} finding(s) per .pr-review.json " f"(style={(config or {}).get('style', 'balanced')}, " f"threshold={(config or {}).get('severity_threshold', '?')}, " f"max={len(findings)})", flush=True, ) # Compute attribution so inline comments + the table can show per-comment # estimates. Only meaningful when we have measured usage. if usage and usage.get("output"): compute_attribution(findings, usage["output"]) usage_section = _render_collapsible_usage(usage, display_model, config=config) if usage else "" # Anchor against the RAW diff, never the compressed one. Compression # drops context lines, so a finding on a line that survived in the file # but not in the prompt would be demoted to a bullet for no reason. # (compress_diff renumbers its hunks, so both are line-accurate; the # raw diff is simply the complete set.) anchors = parse_diff_anchors(raw_diff) anchored, unanchored = split_findings(findings, anchors) # Summary body: unanchored bullets fall through to a "Unanchored notes" # section; the structured Findings Overview table covers both anchored # + unanchored so reviewers see the full set even if inline comments # are collapsed. bullets = summary_bullets(unanchored) summary_parts = [] if bullets: summary_parts.append("### Unanchored Notes\n\n" + bullets) # 1-5 merge verdict for the header badge. Computed AFTER filtering + # anchoring so the verdict reflects what the operator sees (a critical # finding that fails to anchor is still a critical finding). The # default 5 keeps any failure path (e.g. empty findings) green. # Cross-lens agreement is passed in via kwarg (see multi_lens scan # above) because the `_multi_lens` flag is stripped before findings # reach this call. confidence = merge_confidence(findings, multi_lens_observed=multi_lens) summary_body = format_review_body( "\n\n".join(summary_parts), display_model, sha, summary=review_summary, usage_section=usage_section, summary_changes=summary_changes, risks=risks, findings_for_table=findings, inline_count=len(anchored), confidence=confidence, static_message=(config or {}).get("static_message", ""), ) post_inline_review(api, repo, index, token, summary_body, anchored) _emit_langfuse( repo=repo, index=index, sha=sha, title=title, model=display_model, usage=usage, findings=findings, summary=review_summary, engine=engine, config=config, dropped_count=parse_dropped, ) print( f"pragent: reviewed {repo}#{index} sha={sha[:8]} " f"engine={engine} findings={len(findings)} inline={len(anchored)}", flush=True, ) return True except Exception as e: # fail-open try: post_review(api, repo, index, token, format_review_body(f"⚠️ AI review failed: {e}", display_model, sha)) except Exception as e2: print(f"pragent: could not post failure note: {e2}", file=sys.stderr) print(f"pragent: review failed: {e}", file=sys.stderr) return False def run() -> int: review_pr( api=_need("GITEA_API"), repo=_need("GITEA_REPOSITORY"), index=_need("PR_INDEX"), title=os.environ.get("PR_TITLE", ""), body=os.environ.get("PR_BODY", ""), sha=os.environ.get("PRAGENT_SHA", ""), token=_need("PRAGENT_BOT_TOKEN"), ollama_url=_need("OLLAMA_URL"), model=_need("OLLAMA_MODEL"), max_tokens=_int_env("OLLAMA_MAX_TOKENS", 8000), max_chars=_int_env("DIFF_MAX_CHARS", 150000), base_ref=os.environ.get("PR_BASE_REF", ""), ) return 0 if __name__ == "__main__": sys.exit(run())