"""pragent pilot — feedback harvester. For each PR the webhook server is about to review, walk back through the Gitea-side state of every bot comment from every prior review on that PR and record: - reactions on the review body + on each inline comment - thread-resolved state (Gitea's `resolver` field; non-empty = resolved) - replies (issue-comments with `review_comment_id` matching ours) - the bot's own findings_count + inline_count per review (for the restraint metric) Everything is best-effort. A single 404 or 5xx is logged and skipped — we must never abort a review because the feedback DB had a hiccup. The harvester is intentionally separate from `review_pr` so it can be called independently (e.g. by the daily analyzer's "backfill" mode) and tested in isolation against a mocked Gitea client. """ from __future__ import annotations import json import logging import re import time import urllib.parse import urllib.request from typing import Optional import ai_review # used as ai_review.gitea_get(...) so test mocks land on the binding from feedback import ( init, record_inline_finding, record_reaction, record_reply, record_review, record_thread_state, posthash, ) log = logging.getLogger("pragent.feedback.harvest") # Reviewer identity — only collect feedback on comments authored by us. # Avoids harvesting reactions on human comments (which we never want to # count toward "bot usefulness"). REVIEWER_LOGIN = "pragent-bot" # Reactions content tokens Gitea uses. We track +1 / -1 explicitly; the # others are stored as-is so the analyzer can mine them (👀 eyes, # laugh, hooray, confused, heart, rocket, …) without hardcoding a list # that drifts across Gitea versions. POSITIVE_REACTIONS = {"+1", "heart", "hooray", "laugh", "rocket"} NEGATIVE_REACTIONS = {"-1", "confused"} # Note: Gitea's `eyes` reaction (👀) means "I'm watching" — not approval # or disapproval. Treated as neutral by the analyzer. # Phrases that, in a reply, indicate the author thinks the bot's finding # was wrong. Casing + punctuation ignored; substring match is good enough # (false positives in the analyzer cost a human minute; false negatives # hide regressions). FALSE_POSITIVE_PHRASES = ( "false positive", "not actually", "this is fine", "this is intentional", "not a bug", "intentional", "wrong here", "isn't actually", "is not actually", "don't think this is", "i disagree", "this isn't right", "this is correct", "this is expected", "by design", "this is by design", ) # Gitea review-comment payload includes a 'body' field that may carry our # sha marker + severity header. We extract severity + path/line from it # as a fallback when the finding wasn't already seeded at post-time (old # reviews before feedback.py existed). SEV_RE = re.compile(r"\*\*\[(CRITICAL|HIGH|MEDIUM|LOW|INFO)\]\*\*", re.IGNORECASE) PATH_LINE_RE = re.compile(r"`([^?:\n]+?):(\d+)`") SHA_MARKER_RE = re.compile(r"", re.IGNORECASE) # --------------------------------------------------------------------------- # Low-level HTTP — tolerant JSON parse (Gitea sometimes returns `null` where # we expect `[]`, e.g. reactions on a fresh comment) # --------------------------------------------------------------------------- def _gitea_get_json(api: str, repo: str, path: str, token: str) -> tuple[int, object]: status, raw = ai_review.gitea_get(api, repo, path, token) if status != 200: return status, None try: return status, json.loads(raw.decode("utf-8", errors="replace")) except (json.JSONDecodeError, ValueError): return status, None # --------------------------------------------------------------------------- # Parse helpers # --------------------------------------------------------------------------- def _parse_severity(body: str) -> str: m = SEV_RE.search(body or "") return m.group(1).upper() if m else "INFO" def _parse_path_line(body: str) -> tuple[Optional[str], Optional[int]]: m = PATH_LINE_RE.search(body or "") if not m: return None, None path = m.group(1).strip() try: return path, int(m.group(2)) except ValueError: return path, None def _parse_sha(body: str) -> Optional[str]: m = SHA_MARKER_RE.search(body or "") return m.group(1) if m else None def _is_negation_reply(body: str) -> bool: if not body: return False norm = body.lower() return any(p in norm for p in FALSE_POSITIVE_PHRASES) # --------------------------------------------------------------------------- # Reaction classification (cheap, used by the analyzer — not the harvester # itself) # --------------------------------------------------------------------------- def classify_reaction(content: str) -> str: """Bucket a reaction into 'positive', 'negative', or 'neutral'.""" c = (content or "").strip().lower() if c in POSITIVE_REACTIONS: return "positive" if c in NEGATIVE_REACTIONS: return "negative" return "neutral" # --------------------------------------------------------------------------- # Main harvest entry # --------------------------------------------------------------------------- def harvest_for_pr( *, api: str, token: str, repo: str, pr_index: int, db_path: str, page_size: int = 50, ) -> dict: """Walk every bot-authored review on the given PR and record reactions + thread state + replies. Returns a stats dict for logging. `db_path` is the SQLite file path (env: `PRAGENT_FEEDBACK_DB`, typically `/data/feedback.db` mounted via the `feedback-data` PVC). """ conn = init(db_path) stats = { "reviews_seen": 0, "findings_seen": 0, "reactions_recorded": 0, "thread_states_recorded": 0, "replies_recorded": 0, "errors": 0, } try: # 1. List every review on the PR (paginated, but PRs rarely have >page_size) status, payload = _gitea_get_json( api, repo, f"pulls/{pr_index}/reviews?per_page={page_size}", token, ) if status != 200 or not isinstance(payload, list): log.info("harvest: reviews list failed status=%d", status) stats["errors"] += 1 return stats for rev in payload: user = (rev.get("user") or {}).get("login", "") if user != REVIEWER_LOGIN: continue stats["reviews_seen"] += 1 review_id_gitea = rev.get("id") head_sha = rev.get("commit_id", "") review_body = rev.get("body", "") or "" body_sha = _parse_sha(review_body) # Trust the sha marker inside the body — Gitea's commit_id field is # for the LAST commit, not necessarily the reviewed head. If we # can't find a marker, fall back to commit_id. effective_sha = body_sha or head_sha created_at = _parse_iso_ts(rev.get("created_at", "")) db_review_id = record_review( conn, repo=repo, pr=pr_index, head_sha=effective_sha, review_id_gitea=review_id_gitea, posted_at=created_at, ) # 2. Inline comments for this review if review_id_gitea is None: continue rstatus, rpayload = _gitea_get_json( api, repo, f"pulls/{pr_index}/reviews/{review_id_gitea}/comments", token, ) if rstatus != 200 or not isinstance(rpayload, list): stats["errors"] += 1 continue for ic in rpayload: ic_id = ic.get("id") if ic_id is None: continue ic_body = ic.get("body", "") or "" ic_path = ic.get("path") ic_line = ic.get("position") or ic.get("line") ic_severity = _parse_severity(ic_body) # Fall back to body parse when Gitea didn't echo path/line if not ic_path or not ic_line: bp, bl = _parse_path_line(ic_body) ic_path = ic_path or bp ic_line = ic_line or bl if not ic_path or not ic_line: log.info( "harvest: inline %s missing path/line, skipping", ic_id, ) continue finding_id = record_inline_finding( conn, review_id=db_review_id, repo=repo, pr=pr_index, path=ic_path, line=ic_line, severity=ic_severity, problem=_strip_severity_header(ic_body), fix="", suggestion="", comment_id=ic_id, posted_at=created_at, ) stats["findings_seen"] += 1 if finding_id is None: continue # 3. Reactions on the inline comment react_status, react_payload = _gitea_get_json( api, repo, f"issues/comments/{ic_id}/reactions", token, ) if react_status == 200 and isinstance(react_payload, list): for r in react_payload: ruser = (r.get("user") or {}).get("login", "") or "?" # Gitea has occasionally returned `content` as a # dict on older versions; coerce to str defensively. rcontent = str(r.get("content") or "").strip() if not rcontent: continue if record_reaction( conn, comment_id=ic_id, user=ruser, content=rcontent, created_at=_parse_iso_ts(r.get("created_at", "")), ): stats["reactions_recorded"] += 1 # 4. Thread state (Gitea's `resolver` field on the inline # comment). Some Gitea versions serialize this as a user # object ({login, ...}) instead of a username string — # coerce defensively before calling .strip(). resolver_raw = ic.get("resolver") if isinstance(resolver_raw, dict): resolver = (resolver_raw.get("login") or "").strip() else: resolver = str(resolver_raw or "").strip() if resolver_raw is not None: # field present, even if "" record_thread_state( conn, finding_id=finding_id, resolved=bool(resolver), ) stats["thread_states_recorded"] += 1 # 5. Replies on this review (issue-comments whose # `review_comment_id` points at one of our inline comments). # Some Gitea versions don't expose `review_comment_id` on the # issue-comment endpoint — in that case `replies` stays # empty; we degrade gracefully. try: _harvest_replies( api=api, repo=repo, token=token, pr_index=pr_index, review_id=review_id_gitea, inline_comments=rpayload, conn=conn, stats=stats, ) except Exception as e: log.info("harvest: replies fetch failed: %s", e) stats["errors"] += 1 finally: conn.close() return stats def _harvest_replies( *, api: str, repo: str, token: str, pr_index: int, review_id: int, inline_comments: list, conn, stats: dict, ) -> None: """Fetch issue comments on this PR; record those whose `review_comment_id` matches one of our inline comment IDs. Gitea 1.26 doesn't include that field — we fall back to fetching each inline comment individually via `issues/comments/{id}` (does include the field) only if the bulk fetch is empty. """ inline_ids = {c.get("id") for c in inline_comments if c.get("id") is not None} if not inline_ids: return status, payload = _gitea_get_json( api, repo, f"issues/{pr_index}/comments?per_page=100", token, ) if status != 200 or not isinstance(payload, list): return # Build mapping inline_id -> finding_id (one SELECT instead of N) rows = conn.execute( "SELECT comment_id, id FROM inline_finding WHERE comment_id IN (" + ",".join("?" * len(inline_ids)) + ")", list(inline_ids), ).fetchall() inline_to_finding = {r[0]: r[1] for r in rows} for c in payload: rcid = c.get("review_comment_id") if not rcid or rcid not in inline_to_finding: continue author = (c.get("user") or {}).get("login", "") or "?" body = c.get("body", "") or "" ts = _parse_iso_ts(c.get("created_at", "")) if record_reply( conn, finding_id=inline_to_finding[rcid], author=author, body=body, created_at=ts, ): stats["replies_recorded"] += 1 # --------------------------------------------------------------------------- # Helpers # --------------------------------------------------------------------------- def _strip_severity_header(body: str) -> str: """Drop the leading `**[SEVERITY]**` so the posthash captures the substance, not the severity label.""" return SEV_RE.sub("", body or "", count=1).strip() def _parse_iso_ts(s: str) -> int: if not s: return int(time.time()) try: # Python 3.11+ fromisoformat tolerates the trailing 'Z'. return int(__import__("datetime").datetime.fromisoformat( s.replace("Z", "+00:00") ).timestamp()) except Exception: return int(time.time()) # --------------------------------------------------------------------------- # CLI for manual backfill / first-time seed # --------------------------------------------------------------------------- def main() -> int: import argparse, os p = argparse.ArgumentParser( description="Harvest reactions/threads/replies on bot PR comments.", ) p.add_argument("--api", default=os.environ.get( "GITEA_API", "http://gitea-http.gitea.svc.cluster.local:3000", )) p.add_argument("--token", default=os.environ.get("PRAGENT_BOT_TOKEN", "")) p.add_argument("--repo", required=True, help="owner/name") p.add_argument("--pr", type=int, required=True, help="PR index") p.add_argument("--db", default=os.environ.get( "PRAGENT_FEEDBACK_DB", "/data/feedback.db", )) args = p.parse_args() if not args.token: print("PRAGENT_BOT_TOKEN required", flush=True) return 2 logging.basicConfig(level=logging.INFO) stats = harvest_for_pr( api=args.api, token=args.token, repo=args.repo, pr_index=args.pr, db_path=args.db, ) print(json.dumps(stats), flush=True) return 0 if __name__ == "__main__": raise SystemExit(main())