7a510a926d
Group review, feedback, evaluation, observability, and entrypoint code into packages. Keep thin top-level compatibility shims for existing scripts and imports, and mirror the structure in the tests.
255 lines
9.8 KiB
Python
255 lines
9.8 KiB
Python
#!/usr/bin/env python3
|
|
r"""pragent pilot — diff compression + prior-review compaction.
|
|
|
|
Two pure helpers that shrink what lands in the model prompt without losing
|
|
signal:
|
|
|
|
* ``compress_diff(diff, *, context=2)`` — re-renders a unified diff so each
|
|
hunk keeps only ``context`` unchanged lines on either side of its +/- lines.
|
|
The default 2 matches what most reviewers see on GitHub/Gitea, and is
|
|
enough to anchor every ``+``/``-`` line and give the reviewer the enclosing
|
|
statement. Wider context = more reading; narrower = less. Set
|
|
``context=0`` for +/- only, ``context=-1`` to disable entirely.
|
|
|
|
Elided context is not merely deleted: each surviving run of lines is
|
|
re-emitted as its *own* ``@@ -a,b +c,d @@`` hunk with recomputed line
|
|
numbers, so the output stays a valid unified diff whose line numbers
|
|
still describe the post-change file. ``parse_diff_anchors`` (and the
|
|
model) therefore read the same line numbers before and after compression.
|
|
|
|
* ``extract_finding_bullets(review_body)`` — pulls the lines of a prior
|
|
review that look like a pragent finding (``- 🔴 [HIGH] `path:line` — …``,
|
|
or the older ``- **[HIGH]** …`` form) and drops everything else. The model
|
|
already has the diff — repeating the prose ("this PR adds eval() — risky")
|
|
is just token burn. Bullet-only priors cut ~75% off prior-review bytes on
|
|
a typical 4-finding review.
|
|
|
|
Stdlib only. No I/O. Tolerant of malformed input — never raises.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
|
|
# A real hunk header: `@@ -old[,count] +new[,count] @@[ trailing section]`.
|
|
# Captures both starts, both counts, and the trailing function-context text.
|
|
# Matching the full shape (not just a `@@` prefix) matters: a *removed* line
|
|
# whose content begins with `@@` is body, not a header.
|
|
_HUNK_RE = re.compile(
|
|
r"^@@\s+-(\d+)(?:,(\d+))?\s+\+(\d+)(?:,(\d+))?\s+@@(.*)$"
|
|
)
|
|
|
|
# Match a pragent summary-bullet line, in any of the shapes the renderer has
|
|
# emitted: `- 🔴 [HIGH] \`path:line\` — …` (current, `_severity_badge`),
|
|
# `- **[HIGH]** …` (bold, pre-badge), `- [high] …` (plain, oldest).
|
|
# Anything between the bullet marker and `[SEV]` (emoji, bold markers,
|
|
# whitespace) is tolerated — it is decoration, not signal.
|
|
_FINDING_BULLET_RE = re.compile(
|
|
r"^\s*[-*]\s*[^\w\[]*\[(?P<sev>critical|high|medium|low)\]",
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
|
|
def compress_diff(diff: str, *, context: int = 2) -> tuple[str, int, int]:
|
|
"""Re-render `diff` keeping at most `context` unchanged lines around +/-.
|
|
|
|
Args:
|
|
diff: unified-diff text (what `gitea .../pulls/{n}.diff` returns).
|
|
context: max unchanged lines to keep on each side of a hunk. Use 0
|
|
for +/- only, -1 to disable compression (raw passthrough).
|
|
|
|
Returns:
|
|
`(text, original_chars, kept_chars)`. `original_chars` is the character
|
|
length of `diff` as given; `kept_chars` is the character length of
|
|
`text`. Every emitted hunk header is recomputed to match the lines
|
|
under it, so the result is a valid unified diff. Lines that are not
|
|
part of a hunk (`diff --git`, `index …`, `Binary files differ`, mode
|
|
changes) pass through verbatim.
|
|
"""
|
|
if not diff:
|
|
return diff or "", len(diff or ""), len(diff or "")
|
|
if context < 0:
|
|
return diff, len(diff), len(diff)
|
|
|
|
orig = len(diff)
|
|
lines = diff.splitlines()
|
|
out: list[str] = []
|
|
|
|
i = 0
|
|
n = len(lines)
|
|
while i < n:
|
|
m = _HUNK_RE.match(lines[i])
|
|
if m is None:
|
|
# File header, index line, binary marker, mode change, prose —
|
|
# anything outside a hunk body. Copy verbatim.
|
|
out.append(lines[i])
|
|
i += 1
|
|
continue
|
|
|
|
i += 1
|
|
body_start = i
|
|
while i < n and _is_body_line(lines[i]):
|
|
i += 1
|
|
body = lines[body_start:i]
|
|
|
|
out.extend(
|
|
_render_hunk(
|
|
body,
|
|
old_start=int(m.group(1)),
|
|
new_start=int(m.group(3)),
|
|
section=m.group(5) or "",
|
|
context=context,
|
|
)
|
|
)
|
|
|
|
text = "\n".join(out) + ("\n" if diff.endswith("\n") else "")
|
|
if not text.strip():
|
|
# Nothing survived (or the input was nothing but newlines); fall back
|
|
# to the original so the worst case is no improvement, not data loss.
|
|
return diff, orig, orig
|
|
if len(text) >= orig:
|
|
# Re-emitted hunk headers can outweigh the context they replace on a
|
|
# small, densely-changed diff. Never hand back something longer than
|
|
# what we were given.
|
|
return diff, orig, orig
|
|
return text, orig, len(text)
|
|
|
|
|
|
def _is_body_line(line: str) -> bool:
|
|
r"""True if `line` belongs to the current hunk body.
|
|
|
|
Hunk bodies contain only ` `/`+`/`-` prefixed lines and `\ No newline at
|
|
end of file`. An empty line is a context line whose trailing space was
|
|
stripped (common in mail-formatted diffs), so it counts as body too.
|
|
|
|
The check is prefix-based *and* header-aware: a removed line reading
|
|
`---` or an added line reading `+++` (YAML document separators, setext
|
|
underlines, `--` SQL comments) is body, not a file header — the previous
|
|
implementation misread those and silently dropped the rest of the hunk.
|
|
A new file section always opens with `diff --git`, which ends the body.
|
|
"""
|
|
if line == "":
|
|
return True
|
|
if line.startswith("diff --git ") or line.startswith("Index: "):
|
|
return False
|
|
if _HUNK_RE.match(line):
|
|
return False
|
|
return line[0] in " +-\\"
|
|
|
|
|
|
def _render_hunk(
|
|
body: list[str],
|
|
*,
|
|
old_start: int,
|
|
new_start: int,
|
|
section: str,
|
|
context: int,
|
|
) -> list[str]:
|
|
r"""Trim `body` to `context` unchanged lines around its +/- lines.
|
|
|
|
Each surviving run of consecutive lines is emitted as a standalone hunk
|
|
with a recomputed ``@@ -a,b +c,d @@`` header, so post-change line numbers
|
|
stay truthful. A hunk with no +/- lines at all (pure context) is dropped
|
|
entirely; ``\ No newline at end of file`` markers are dropped as noise.
|
|
|
|
Returns the rendered lines (headers included), or [] if nothing survived.
|
|
"""
|
|
# Number every body line on both sides before anything is dropped.
|
|
numbered: list[tuple[str, int, int]] = [] # (line, old_no, new_no)
|
|
old_no, new_no = old_start, new_start
|
|
for ln in body:
|
|
if ln.startswith("\\"):
|
|
continue # `\ No newline at end of file` — no signal, no numbering
|
|
kind = ln[0] if ln else " "
|
|
if kind == "+":
|
|
numbered.append((ln, -1, new_no))
|
|
new_no += 1
|
|
elif kind == "-":
|
|
numbered.append((ln, old_no, -1))
|
|
old_no += 1
|
|
else:
|
|
numbered.append((ln, old_no, new_no))
|
|
old_no += 1
|
|
new_no += 1
|
|
|
|
changed = [j for j, (ln, _, _) in enumerate(numbered) if ln[:1] in ("+", "-")]
|
|
if not changed:
|
|
return []
|
|
|
|
keep: set[int] = set()
|
|
for k in changed:
|
|
for j in range(max(0, k - context), min(len(numbered) - 1, k + context) + 1):
|
|
keep.add(j)
|
|
|
|
out: list[str] = []
|
|
for run in _consecutive_runs(sorted(keep)):
|
|
chunk = [numbered[j] for j in run]
|
|
old_count = sum(1 for ln, _, _ in chunk if ln[:1] != "+")
|
|
new_count = sum(1 for ln, _, _ in chunk if ln[:1] != "-")
|
|
# A run's start is the first line that exists on that side. When a
|
|
# side has no lines at all (pure addition / pure deletion), unified
|
|
# diff convention is `start = line before, count = 0`.
|
|
old_first = next((o for ln, o, _ in chunk if o >= 0), None)
|
|
new_first = next((nw for ln, _, nw in chunk if nw >= 0), None)
|
|
old_hdr = old_first if old_first is not None else max(chunk[0][1], 0)
|
|
new_hdr = new_first if new_first is not None else max(chunk[0][2], 0)
|
|
if old_count == 0:
|
|
old_hdr = _side_start_before(numbered, run[0], side=1)
|
|
if new_count == 0:
|
|
new_hdr = _side_start_before(numbered, run[0], side=2)
|
|
out.append(
|
|
f"@@ -{old_hdr},{old_count} +{new_hdr},{new_count} @@{section}"
|
|
)
|
|
out.extend(ln for ln, _, _ in chunk)
|
|
return out
|
|
|
|
|
|
def _side_start_before(
|
|
numbered: list[tuple[str, int, int]], idx: int, *, side: int
|
|
) -> int:
|
|
"""Line number on `side` (1=old, 2=new) just before body index `idx`.
|
|
|
|
Used for the zero-count header form (`@@ -7,0 +8,3 @@`), where unified
|
|
diff names the line the change is inserted *after*.
|
|
"""
|
|
for j in range(idx - 1, -1, -1):
|
|
no = numbered[j][side]
|
|
if no >= 0:
|
|
return no
|
|
# Nothing before it: derive from the first numbered line on that side.
|
|
for _, old_no, new_no in numbered:
|
|
no = old_no if side == 1 else new_no
|
|
if no >= 0:
|
|
return max(no - 1, 0)
|
|
return 0
|
|
|
|
|
|
def _consecutive_runs(indices: list[int]) -> list[list[int]]:
|
|
"""Group a sorted index list into runs of consecutive integers."""
|
|
runs: list[list[int]] = []
|
|
for j in indices:
|
|
if runs and j == runs[-1][-1] + 1:
|
|
runs[-1].append(j)
|
|
else:
|
|
runs.append([j])
|
|
return runs
|
|
|
|
|
|
def extract_finding_bullets(review_body: str) -> list[str]:
|
|
"""Pull the finding-bullet lines out of a prior review body.
|
|
|
|
Returns the matching lines stripped of surrounding whitespace, preserving
|
|
the rendered ``[SEV] `path:line` — problem`` shape (badge emoji and bold
|
|
markers included, whichever the renderer used). Lines that look like
|
|
bullets but carry no severity tag are dropped — the reviewer synthesizes
|
|
from the matched ones. Continuation lines (` - **Fix:** …`) are not
|
|
finding lines and are dropped with the rest of the prose.
|
|
"""
|
|
if not review_body:
|
|
return []
|
|
out = []
|
|
for line in review_body.splitlines():
|
|
if _FINDING_BULLET_RE.match(line):
|
|
out.append(line.strip())
|
|
return out
|