#!/usr/bin/env python3 r"""pragent pilot — diff compression + prior-review compaction. Two pure helpers that shrink what lands in the model prompt without losing signal: * ``compress_diff(diff, *, context=2)`` — re-renders a unified diff so each hunk keeps only ``context`` unchanged lines on either side of its +/- lines. The default 2 matches what most reviewers see on GitHub/Gitea, and is enough to anchor every ``+``/``-`` line and give the reviewer the enclosing statement. Wider context = more reading; narrower = less. Set ``context=0`` for +/- only, ``context=-1`` to disable entirely. Elided context is not merely deleted: each surviving run of lines is re-emitted as its *own* ``@@ -a,b +c,d @@`` hunk with recomputed line numbers, so the output stays a valid unified diff whose line numbers still describe the post-change file. ``parse_diff_anchors`` (and the model) therefore read the same line numbers before and after compression. * ``extract_finding_bullets(review_body)`` — pulls the lines of a prior review that look like a pragent finding (``- 🔴 [HIGH] `path:line` — …``, or the older ``- **[HIGH]** …`` form) and drops everything else. The model already has the diff — repeating the prose ("this PR adds eval() — risky") is just token burn. Bullet-only priors cut ~75% off prior-review bytes on a typical 4-finding review. Stdlib only. No I/O. Tolerant of malformed input — never raises. """ from __future__ import annotations import re # A real hunk header: `@@ -old[,count] +new[,count] @@[ trailing section]`. # Captures both starts, both counts, and the trailing function-context text. # Matching the full shape (not just a `@@` prefix) matters: a *removed* line # whose content begins with `@@` is body, not a header. _HUNK_RE = re.compile( r"^@@\s+-(\d+)(?:,(\d+))?\s+\+(\d+)(?:,(\d+))?\s+@@(.*)$" ) # Match a pragent summary-bullet line, in any of the shapes the renderer has # emitted: `- 🔴 [HIGH] \`path:line\` — …` (current, `_severity_badge`), # `- **[HIGH]** …` (bold, pre-badge), `- [high] …` (plain, oldest). # Anything between the bullet marker and `[SEV]` (emoji, bold markers, # whitespace) is tolerated — it is decoration, not signal. _FINDING_BULLET_RE = re.compile( r"^\s*[-*]\s*[^\w\[]*\[(?Pcritical|high|medium|low)\]", re.IGNORECASE, ) def compress_diff(diff: str, *, context: int = 2) -> tuple[str, int, int]: """Re-render `diff` keeping at most `context` unchanged lines around +/-. Args: diff: unified-diff text (what `gitea .../pulls/{n}.diff` returns). context: max unchanged lines to keep on each side of a hunk. Use 0 for +/- only, -1 to disable compression (raw passthrough). Returns: `(text, original_chars, kept_chars)`. `original_chars` is the character length of `diff` as given; `kept_chars` is the character length of `text`. Every emitted hunk header is recomputed to match the lines under it, so the result is a valid unified diff. Lines that are not part of a hunk (`diff --git`, `index …`, `Binary files differ`, mode changes) pass through verbatim. """ if not diff: return diff or "", len(diff or ""), len(diff or "") if context < 0: return diff, len(diff), len(diff) orig = len(diff) lines = diff.splitlines() out: list[str] = [] i = 0 n = len(lines) while i < n: m = _HUNK_RE.match(lines[i]) if m is None: # File header, index line, binary marker, mode change, prose — # anything outside a hunk body. Copy verbatim. out.append(lines[i]) i += 1 continue i += 1 body_start = i while i < n and _is_body_line(lines[i]): i += 1 body = lines[body_start:i] out.extend( _render_hunk( body, old_start=int(m.group(1)), new_start=int(m.group(3)), section=m.group(5) or "", context=context, ) ) text = "\n".join(out) + ("\n" if diff.endswith("\n") else "") if not text.strip(): # Nothing survived (or the input was nothing but newlines); fall back # to the original so the worst case is no improvement, not data loss. return diff, orig, orig if len(text) >= orig: # Re-emitted hunk headers can outweigh the context they replace on a # small, densely-changed diff. Never hand back something longer than # what we were given. return diff, orig, orig return text, orig, len(text) def _is_body_line(line: str) -> bool: r"""True if `line` belongs to the current hunk body. Hunk bodies contain only ` `/`+`/`-` prefixed lines and `\ No newline at end of file`. An empty line is a context line whose trailing space was stripped (common in mail-formatted diffs), so it counts as body too. The check is prefix-based *and* header-aware: a removed line reading `---` or an added line reading `+++` (YAML document separators, setext underlines, `--` SQL comments) is body, not a file header — the previous implementation misread those and silently dropped the rest of the hunk. A new file section always opens with `diff --git`, which ends the body. """ if line == "": return True if line.startswith("diff --git ") or line.startswith("Index: "): return False if _HUNK_RE.match(line): return False return line[0] in " +-\\" def _render_hunk( body: list[str], *, old_start: int, new_start: int, section: str, context: int, ) -> list[str]: r"""Trim `body` to `context` unchanged lines around its +/- lines. Each surviving run of consecutive lines is emitted as a standalone hunk with a recomputed ``@@ -a,b +c,d @@`` header, so post-change line numbers stay truthful. A hunk with no +/- lines at all (pure context) is dropped entirely; ``\ No newline at end of file`` markers are dropped as noise. Returns the rendered lines (headers included), or [] if nothing survived. """ # Number every body line on both sides before anything is dropped. numbered: list[tuple[str, int, int]] = [] # (line, old_no, new_no) old_no, new_no = old_start, new_start for ln in body: if ln.startswith("\\"): continue # `\ No newline at end of file` — no signal, no numbering kind = ln[0] if ln else " " if kind == "+": numbered.append((ln, -1, new_no)) new_no += 1 elif kind == "-": numbered.append((ln, old_no, -1)) old_no += 1 else: numbered.append((ln, old_no, new_no)) old_no += 1 new_no += 1 changed = [j for j, (ln, _, _) in enumerate(numbered) if ln[:1] in ("+", "-")] if not changed: return [] keep: set[int] = set() for k in changed: for j in range(max(0, k - context), min(len(numbered) - 1, k + context) + 1): keep.add(j) out: list[str] = [] for run in _consecutive_runs(sorted(keep)): chunk = [numbered[j] for j in run] old_count = sum(1 for ln, _, _ in chunk if ln[:1] != "+") new_count = sum(1 for ln, _, _ in chunk if ln[:1] != "-") # A run's start is the first line that exists on that side. When a # side has no lines at all (pure addition / pure deletion), unified # diff convention is `start = line before, count = 0`. old_first = next((o for ln, o, _ in chunk if o >= 0), None) new_first = next((nw for ln, _, nw in chunk if nw >= 0), None) old_hdr = old_first if old_first is not None else max(chunk[0][1], 0) new_hdr = new_first if new_first is not None else max(chunk[0][2], 0) if old_count == 0: old_hdr = _side_start_before(numbered, run[0], side=1) if new_count == 0: new_hdr = _side_start_before(numbered, run[0], side=2) out.append( f"@@ -{old_hdr},{old_count} +{new_hdr},{new_count} @@{section}" ) out.extend(ln for ln, _, _ in chunk) return out def _side_start_before( numbered: list[tuple[str, int, int]], idx: int, *, side: int ) -> int: """Line number on `side` (1=old, 2=new) just before body index `idx`. Used for the zero-count header form (`@@ -7,0 +8,3 @@`), where unified diff names the line the change is inserted *after*. """ for j in range(idx - 1, -1, -1): no = numbered[j][side] if no >= 0: return no # Nothing before it: derive from the first numbered line on that side. for _, old_no, new_no in numbered: no = old_no if side == 1 else new_no if no >= 0: return max(no - 1, 0) return 0 def _consecutive_runs(indices: list[int]) -> list[list[int]]: """Group a sorted index list into runs of consecutive integers.""" runs: list[list[int]] = [] for j in indices: if runs and j == runs[-1][-1] + 1: runs[-1].append(j) else: runs.append([j]) return runs def extract_finding_bullets(review_body: str) -> list[str]: """Pull the finding-bullet lines out of a prior review body. Returns the matching lines stripped of surrounding whitespace, preserving the rendered ``[SEV] `path:line` — problem`` shape (badge emoji and bold markers included, whichever the renderer used). Lines that look like bullets but carry no severity tag are dropped — the reviewer synthesizes from the matched ones. Continuation lines (` - **Fix:** …`) are not finding lines and are dropped with the rest of the prose. """ if not review_body: return [] out = [] for line in review_body.splitlines(): if _FINDING_BULLET_RE.match(line): out.append(line.strip()) return out