feat(input): add diff_compress module + prior-review compaction helpers #9

Merged
gitea_admin merged 8 commits from feat/cost-display-compress-config into main 2026-08-20 23:05:32 +00:00
4 changed files with 280 additions and 27 deletions
Showing only changes of commit 5302e8dcd7 - Show all commits
+4 -1
View File
@@ -164,4 +164,7 @@ Rules:
- If the diff is clean, output `{"summary":"...","findings":[]}`. - If the diff is clean, output `{"summary":"...","findings":[]}`.
- Do NOT repeat anything in `prior_reviews`. - Do NOT repeat anything in `prior_reviews`.
- The JSON block must be the LAST thing in your message — the Python shell parses - The JSON block must be the LAST thing in your message — the Python shell parses
the last ```json fenced block from your output. the last ```json fenced block from your output. If you run out of context/steps
before emitting it, your analysis is wasted: ALWAYS reserve the final step for
writing the JSON. Stop exploring and write findings at the first sign you've
covered the diff (no new findings in the last 2 file reads = stop).
+170 -25
View File
@@ -509,14 +509,42 @@ def _normalize_finding(f: dict) -> dict | None:
def _last_json_block(text: str) -> str | None: def _last_json_block(text: str) -> str | None:
"""Return the substring of the last fenced ```json block in text, or None. r"""Return the substring of the last JSON object/array in text, or None.
Falls back to _extract_first_json_object when no fence is present."""
The pragent agent emits ```json fences around its final block, but real
outputs drift:
* the fence contains nested objects (regex ``\{.*?\}`` only matches the
first ``}``, truncating the JSON — the parser then sees
``json.JSONDecodeError``);
* the fence is missing or unterminated, but a balanced JSON object sits
in the prose tail;
* the agent emits a bare array (findings only, no summary wrapper).
Strategy:
1. Find each fenced block, take the last. Inside it, walk a balanced
``{...}``/``[...]`` scanner (not a regex) so nested structures survive.
2. Fall back to a balanced scanner over the whole text, picking the LAST
balanced object/array (the agent writes its conclusion last).
"""
s = text or "" s = text or ""
# Find all ```json ... ``` fenced blocks; take the last. if not s:
blocks = list(re.finditer(r"```(?:json)?\s*(\{.*?\})\s*```", s, re.DOTALL)) return None
if blocks: # 1. Fenced blocks: take the last ```json ... ``` or ``` ... ``` region.
return blocks[-1].group(1) fences = list(re.finditer(r"```(?:json)?\n", s))
return _extract_first_json_object(s) for m in reversed(fences):
start = m.end()
# Find the matching closing fence.
end = s.find("```", start)
if end < 0:
# Unterminated fence — try to salvage the balanced object inside.
end = len(s)
inner = s[start:end].strip()
obj = _balanced_json_substring(inner)
if obj is not None:
return obj
# 2. No (parseable) fence — scan the whole text for the LAST balanced
# object/array. The agent's conclusion is at the tail.
return _last_balanced_json(s)
def parse_findings(text: str) -> list[dict]: def parse_findings(text: str) -> list[dict]:
@@ -526,11 +554,17 @@ def parse_findings(text: str) -> list[dict]:
scans for the first balanced `{...}` and extracts its `findings` array. scans for the first balanced `{...}` and extracts its `findings` array.
Drops findings missing path/line or with an unknown severity (normalised). Drops findings missing path/line or with an unknown severity (normalised).
Never raises — returns [] on any parse failure. Never raises — returns [] on any parse failure.
Also accepts a bare JSON array as the outer value: ``[{...}, {...}]`` —
some agents skip the ``{"summary":..., "findings":[...]}`` wrapper.
""" """
data = _parse_json_tolerant(text) data = _parse_json_tolerant(text)
if not isinstance(data, dict): if isinstance(data, dict):
findings = data.get("findings")
elif isinstance(data, list):
findings = data
else:
return [] return []
findings = data.get("findings")
if not isinstance(findings, list): if not isinstance(findings, list):
return [] return []
out = [] out = []
@@ -577,10 +611,11 @@ def salvage_summary(text: str, max_chars: int = SALVAGE_MAX_CHARS) -> str:
def parse_review_output(text: str) -> tuple[str, list[dict]]: def parse_review_output(text: str) -> tuple[str, list[dict]]:
"""Parse the opengine's stdout into (summary, findings). """Parse the opengine's stdout into (summary, findings).
Accepts `{"summary": "...", "findings": [...]}` (the opencode pragent agent) Accepts `{"summary": "...", "findings": [...]}` (the opencode pragent agent),
or a bare `{"findings": [...]}`. `summary` defaults to "". Uses the LAST `{"findings": [...]}`, or a bare `[...]` of finding dicts. `summary` defaults
```json fenced block (the pragent agent emits JSON as the final block), with to "". Uses the LAST fenced block (the pragent agent emits JSON as the final
a tolerant fallback. Never raises. block), with a tolerant fallback that scans for the last balanced
object/array in the prose tail. Never raises.
""" """
blob = _last_json_block(text) blob = _last_json_block(text)
if blob is None: if blob is None:
@@ -589,10 +624,15 @@ def parse_review_output(text: str) -> tuple[str, list[dict]]:
data = json.loads(blob) data = json.loads(blob)
except json.JSONDecodeError: except json.JSONDecodeError:
return "", [] return "", []
if not isinstance(data, dict): if isinstance(data, dict):
summary = str(data.get("summary", "") or "").strip()
findings = data.get("findings")
elif isinstance(data, list):
# Bare array: each item is a finding; no summary.
summary = ""
findings = data
else:
return "", [] return "", []
summary = str(data.get("summary", "") or "").strip()
findings = data.get("findings")
out = [] out = []
if isinstance(findings, list): if isinstance(findings, list):
for f in findings: for f in findings:
@@ -602,16 +642,18 @@ def parse_review_output(text: str) -> tuple[str, list[dict]]:
return summary, out return summary, out
def _parse_json_tolerant(text: str) -> dict | None: def _parse_json_tolerant(text: str) -> dict | list | None:
"""Parse a JSON object from text: try the last fenced block, then a direct """Parse a JSON object/array from text: try the last fenced block, then a
parse, then the first balanced object. Returns None on any failure.""" direct parse, then the first balanced object. Returns None on any failure.
Accepts both ``{...}`` (the pragent schema) and bare ``[...]`` arrays
(agents that skip the wrapper)."""
if not text: if not text:
return None return None
blob = _last_json_block(text) blob = _last_json_block(text)
if blob is not None: if blob is not None:
try: try:
d = json.loads(blob) d = json.loads(blob)
if isinstance(d, dict): if isinstance(d, (dict, list)):
return d return d
except json.JSONDecodeError: except json.JSONDecodeError:
pass pass
@@ -621,7 +663,7 @@ def _parse_json_tolerant(text: str) -> dict | None:
s = re.sub(r"\n?```$", "", s).strip() s = re.sub(r"\n?```$", "", s).strip()
try: try:
d = json.loads(s) d = json.loads(s)
if isinstance(d, dict): if isinstance(d, (dict, list)):
return d return d
except json.JSONDecodeError: except json.JSONDecodeError:
pass pass
@@ -629,7 +671,17 @@ def _parse_json_tolerant(text: str) -> dict | None:
if obj is not None: if obj is not None:
try: try:
d = json.loads(obj) d = json.loads(obj)
if isinstance(d, dict): if isinstance(d, (dict, list)):
return d
except json.JSONDecodeError:
pass
# Last resort: the JSON lives at the tail of the prose with no fence.
# Walk the whole text for the last balanced object/array.
last = _last_balanced_json(text)
if last is not None:
try:
d = json.loads(last)
if isinstance(d, (dict, list)):
return d return d
except json.JSONDecodeError: except json.JSONDecodeError:
pass pass
@@ -641,6 +693,62 @@ def _extract_first_json_object(s: str) -> str | None:
start = s.find("{") start = s.find("{")
if start < 0: if start < 0:
return None return None
end = _scan_balanced(s, start, "{", "}")
if end is None:
return None
return s[start:end + 1]
def _last_balanced_json(s: str) -> str | None:
"""Return the substring of the LAST balanced ``{...}`` or ``[...]`` in s.
Used when the agent emits no fence: the JSON lives in the prose tail.
Picks whichever closer (object or array) appears latest in the text.
"""
if not s:
return None
last_obj = _find_last_close(s, "{", "}")
last_arr = _find_last_close(s, "[", "]")
candidates = []
if last_obj is not None:
candidates.append(last_obj)
if last_arr is not None:
candidates.append(last_arr)
if not candidates:
return None
end, opener, start = max(candidates, key=lambda t: t[0])
return s[start:end + 1]
def _balanced_json_substring(s: str) -> str | None:
"""Return the first balanced ``{...}`` or ``[...]`` substring in ``s``.
Skips past leading whitespace/non-JSON and returns the full balanced
extent (handles nested objects/arrays and string literals with braces).
"""
if not s:
return None
# Try object first; the pragent schema is an object on the outer level.
for i, c in enumerate(s):
if c == "{":
end = _scan_balanced(s, i, "{", "}")
if end is not None:
return s[i:end + 1]
break
if c == "[":
end = _scan_balanced(s, i, "[", "]")
if end is not None:
return s[i:end + 1]
break
return None
def _scan_balanced(s: str, start: int, opener: str, closer: str) -> int | None:
"""Return the index of the matching ``closer`` for ``s[start] == opener``.
Tracks string literals (with ``\\`` escapes) so braces inside strings don't
fool the depth counter. Returns None if no balance is reached.
"""
depth = 0 depth = 0
in_str = False in_str = False
esc = False esc = False
@@ -656,12 +764,49 @@ def _extract_first_json_object(s: str) -> str | None:
continue continue
if c == '"': if c == '"':
in_str = True in_str = True
elif c == "{": elif c == opener:
depth += 1 depth += 1
elif c == "}": elif c == closer:
depth -= 1 depth -= 1
if depth == 0: if depth == 0:
return s[start:i + 1] return i
return None
def _find_last_close(s: str, opener: str, closer: str) -> tuple[int, str, int] | None:
"""Walk ``s`` backwards from the last ``closer`` to find its matching opener.
Returns ``(close_idx, opener_char, open_idx)`` for the rightmost balanced
structure, or None if no pair exists.
"""
# Find the last `closer` candidate.
last = s.rfind(closer)
while last >= 0:
# Walk left, tracking depth from the perspective of the opener.
depth = 1
in_str = False
esc = False
for j in range(last - 1, -1, -1):
c = s[j]
if in_str:
if esc:
esc = False
elif c == "\\":
esc = True
elif c == '"':
in_str = False
continue
if c == '"':
# Approximation: we don't track quotes perfectly walking
# backwards, but strings in agent output are short and rare.
in_str = not in_str
elif c == closer:
depth += 1
elif c == opener:
depth -= 1
if depth == 0:
return (last, opener, j)
last = s.rfind(closer, 0, last)
return None return None
+1 -1
View File
@@ -103,7 +103,7 @@ def compress_diff(diff: str, *, context: int = 2) -> tuple[str, int, int]:
def _render_hunk_body(body: list[str], *, context: int) -> tuple[list[str], int]: def _render_hunk_body(body: list[str], *, context: int) -> tuple[list[str], int]:
"""Trim `body` to `context` unchanged lines around the +/- lines. r"""Trim `body` to `context` unchanged lines around the +/- lines.
Body lines are classified: Body lines are classified:
- `+` line → keep - `+` line → keep
+105
View File
@@ -11,6 +11,9 @@ sys.path.insert(0, os.path.join(ROOT, "pilot"))
import ai_review # noqa: E402 import ai_review # noqa: E402
from ai_review import ( # noqa: E402 from ai_review import ( # noqa: E402
_balanced_json_substring,
_extract_first_json_object,
_last_balanced_json,
build_user_prompt, build_user_prompt,
compute_attribution, compute_attribution,
format_review_body, format_review_body,
@@ -408,6 +411,108 @@ def test_parse_review_output_uses_last_json_block():
assert fs[0]["path"] == "y" assert fs[0]["path"] == "y"
def test_parse_findings_fenced_json_with_nested_object():
# Real-world regression: agent emits a fence whose inner JSON has nested
# objects. The old regex `\{.*?\}` matched only the first `}`, truncating
# the JSON. Now we balance braces inside the fence.
txt = (
"```json\n"
'{"summary":"x","findings":[{"severity":"high","path":"a.py","line":1,'
'"problem":"p","fix":"f","suggestion":"","reference":""}],"meta":{"engine":"opencode"}}\n'
"```"
)
fs = parse_findings(txt)
assert len(fs) == 1
assert fs[0]["path"] == "a.py"
def test_parse_findings_unfenced_at_tail():
# No fence at all. Agent wrote the JSON inline at the very end of its
# prose. The old first-balanced regex caught the FIRST `{`, not this one.
txt = (
"I considered the diff carefully. Two findings stand out:\n"
"First one is just text.\n"
'{"findings":[{"severity":"critical","path":"x","line":1,"problem":"p","fix":"f"}]}'
)
fs = parse_findings(txt)
assert len(fs) == 1
assert fs[0]["severity"] == "critical"
def test_parse_findings_bare_array():
# Some agents skip the `{"summary":..., "findings":[...]}` wrapper and
# emit just the array.
txt = (
"Here are my findings:\n"
"```json\n"
'[{"severity":"low","path":"a","line":1,"problem":"p","fix":"f","suggestion":"","reference":""}]\n'
"```"
)
fs = parse_findings(txt)
assert len(fs) == 1
assert fs[0]["path"] == "a"
def test_parse_review_output_unfenced_at_tail():
# The exact shape canalhandia produced: long prose, JSON at the very end,
# no fence. Old parser returned ([], salvage) — now we recover findings.
txt = (
"Let me refine the fix: should call a dedicated `setPermanent`.\n"
"Let me finalize. Let me also double-check the `find` thread-safety.\n"
'{"summary":"Adds void protection; one critical race.","findings":['
'{"severity":"high","path":"VoidProtection.java","line":162,'
'"problem":"drop duplication race","fix":"use ItemMeta","suggestion":"","reference":""}]}'
)
summary, fs = parse_review_output(txt)
assert "void protection" in summary.lower()
assert len(fs) == 1
assert fs[0]["path"] == "VoidProtection.java"
def test_parse_review_output_bare_array_at_tail():
txt = (
"All wrapped up.\n"
'[{"severity":"low","path":"a","line":1,"problem":"p","fix":"","suggestion":"","reference":""}]'
)
summary, fs = parse_review_output(txt)
assert summary == ""
assert len(fs) == 1
def test_scan_balanced_handles_braces_in_strings():
# The JSON scanner must not be fooled by `{` or `}` inside string literals.
s = '{"a":"contains { and }","b":1}'
obj = _extract_first_json_object(s)
assert obj == s
d = json.loads(obj)
assert d["a"] == "contains { and }"
def test_last_balanced_json_picks_latest():
s = '{"a":1} some text {"b":2,"nested":{"c":3}} trailing'
out = _last_balanced_json(s)
assert out is not None
d = json.loads(out)
assert d == {"b": 2, "nested": {"c": 3}}
def test_last_balanced_json_no_json():
assert _last_balanced_json("nothing here") is None
assert _last_balanced_json("") is None
def test_balanced_json_substring_skips_leading_prose():
s = 'preamble {"a":1} more prose {"b":2}'
out = _balanced_json_substring(s)
assert out == '{"a":1}'
def test_balanced_json_substring_handles_array():
s = '[{"a":1},{"b":2}]'
out = _balanced_json_substring(s)
assert out == s
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# reference rendering in inline_comment_body + summary_bullets + summary section # reference rendering in inline_comment_body + summary_bullets + summary section
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------