Files
pragent/tests/pilot/test_ai_review.py
T
Marcos 5302e8dcd7 fix(review): salvage findings from nested-object fences + bare arrays + unfenced tail JSON
The canalhandia PR review lost all findings because the agent ran out of
context before emitting the closing json fence. Three failure modes hit
the old regex \{.*?\}:
  * nested objects inside the fence truncated at the first }
  * bare arrays (no {summary, findings} wrapper) returned []
  * unfenced JSON in the prose tail was never reached (first not last)

Replace the regex with a balanced-brace scanner:
  * _last_json_block walks the fence contents with a depth counter so
    nested objects survive
  * _last_balanced_json + _balanced_json_substring handle bare arrays and
    prose-tail JSON when no fence is present
  * _parse_json_tolerant returns list as well as dict; parse_findings and
    parse_review_output accept a bare array as the outer value

Agent prompt tightened: reserve the final step for emitting the JSON
block so the analysis isn't lost when context runs out.

10 new tests in tests/pilot/test_ai_review.py cover the new shapes.
Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-20 16:29:44 +00:00

1128 lines
41 KiB
Python

"""Unit tests for pragent pilot pure helpers. No network."""
import base64
import json
import os
import sys
# Allow running without install: add repo root to path.
HERE = os.path.dirname(os.path.abspath(__file__))
ROOT = os.path.abspath(os.path.join(HERE, "..", ".."))
sys.path.insert(0, os.path.join(ROOT, "pilot"))
import ai_review # noqa: E402
from ai_review import ( # noqa: E402
_balanced_json_substring,
_extract_first_json_object,
_last_balanced_json,
build_user_prompt,
compute_attribution,
format_review_body,
format_usage_section,
inline_comment_body,
parse_diff_anchors,
parse_findings,
parse_repo_config,
parse_review_output,
parse_text_blocks,
prior_review_bodies,
reviewed_shas,
split_findings,
summary_bullets,
truncate_diff,
)
# ---------------------------------------------------------------------------
# truncate_diff
# ---------------------------------------------------------------------------
def test_truncate_diff_short():
text, truncated, n = truncate_diff("abc", 100)
assert text == "abc"
assert truncated is False
assert n == 3
def test_truncate_diff_exact_boundary():
text, truncated, n = truncate_diff("x" * 100, 100)
assert truncated is False
assert n == 100
assert text == "x" * 100
def test_truncate_diff_over_cap():
text, truncated, n = truncate_diff("x" * 250, 100)
assert truncated is True
assert n == 250
assert text.startswith("x" * 100)
assert "[diff truncated at 100 characters]" in text
def test_truncate_diff_none():
text, truncated, n = truncate_diff(None, 100) # type: ignore[arg-type]
assert text == ""
assert truncated is False
assert n == 0
# ---------------------------------------------------------------------------
# parse_text_blocks
# ---------------------------------------------------------------------------
def test_parse_text_blocks_text_only():
content = [{"type": "text", "text": "hello"}, {"type": "text", "text": "world"}]
assert parse_text_blocks(content) == "hello\nworld"
def test_parse_text_blocks_drops_thinking():
content = [
{"type": "thinking", "thinking": "reasoning here"},
{"type": "text", "text": "- [high] a.go:3 — bug. fix."},
]
assert parse_text_blocks(content) == "- [high] a.go:3 — bug. fix."
def test_parse_text_blocks_empty_and_malformed():
assert parse_text_blocks([]) == ""
assert parse_text_blocks(None) == "" # type: ignore[arg-type]
assert parse_text_blocks([{"type": "text"}, "garbage", 5]) == ""
def test_parse_text_blocks_real_glm_shape():
# Captured from glm-5.2:cloud via headroom 8789.
content = [
{"type": "thinking", "thinking": "Analyze the request..."},
{"type": "text", "text": "- [critical] auth.py:12 — token compared with `==`. Use hmac.compare_digest."},
]
assert "compare_digest" in parse_text_blocks(content)
# ---------------------------------------------------------------------------
# format_review_body
# ---------------------------------------------------------------------------
def test_format_review_body_findings():
body = format_review_body("- [high] x:1 — bug. fix.", "glm-5.2:cloud", "abcdef1234567890")
assert "pragent pilot" in body
assert "glm-5.2:cloud" in body
assert "`abcdef12`" in body # 8-char sha
assert "- [high] x:1" in body
def test_format_review_body_empty_findings():
body = format_review_body("", "glm-5.2:cloud", "abcdef1234567890")
assert "No issues found." in body
def test_format_review_body_whitespace_findings():
body = format_review_body(" \n ", "glm-5.2:cloud", "abcdef1234567890")
assert "No issues found." in body
def test_format_review_body_no_sha():
body = format_review_body("- [low] y:2 — nit", "glm-5.2:cloud", "")
assert "`unknown`" in body
# ---------------------------------------------------------------------------
# build_user_prompt
# ---------------------------------------------------------------------------
def test_build_user_prompt_includes_title_and_diff():
p = build_user_prompt("Fix login", "Closes #1", "diff --git a/x b/x")
assert "Fix login" in p
assert "Closes #1" in p
assert "diff --git a/x b/x" in p
def test_build_user_prompt_truncates_long_body():
long_body = "B" * 6000
p = build_user_prompt("t", long_body, "d")
assert "[PR body truncated]" in p
assert p.count("B") < 6000
def test_build_user_prompt_no_body():
p = build_user_prompt("t", "", "d")
assert "Description:" not in p
def test_build_user_prompt_with_config_and_prior():
cfg = {"focus": ["security"], "instructions": "Use Result<T,E>."}
prior = ["🤖 **AI Review** …\n- [high] x:1 — old."]
p = build_user_prompt("t", "b", "diff --git a/x b/x", config=cfg, prior_reviews=prior)
assert "## Repo review config" in p
assert "security" in p
assert "Result<T,E>" in p
assert "## PREVIOUS REVIEWS" in p
assert "old." in p
# ---------------------------------------------------------------------------
# parse_diff_anchors
# ---------------------------------------------------------------------------
_DIFF = """\
diff --git a/src/a.py b/src/a.py
index 1..2 100644
--- a/src/a.py
+++ b/src/a.py
@@ -1,4 +1,5 @@
context
-removed
+added
context2
@@ -10,3 +10,4 @@
keep
+new
last
diff --git a/binary.bin b/binary.bin
new file mode 100644
index 0..1
Binary files differ
"""
def test_parse_diff_anchors_context_and_added():
a = parse_diff_anchors(_DIFF)
# context(1), +added(2), context2(3) | keep(10), +new(11), last(12)
assert a["src/a.py"] == {1, 2, 3, 10, 11, 12}
# removed line (-removed, old line 2) has no new-line anchor
assert 2 in a["src/a.py"] # 2 here is the +added line, not the removed one
def test_parse_diff_anchors_binary_file_present_no_lines():
a = parse_diff_anchors(_DIFF)
assert "binary.bin" in a
assert a["binary.bin"] == set()
def test_parse_diff_anchors_empty():
assert parse_diff_anchors("") == {}
assert parse_diff_anchors(None) == {} # type: ignore[arg-type]
def test_parse_diff_anchors_new_file():
diff = "diff --git a/new.ts b/new.ts\nnew file mode 100644\n--- /dev/null\n+++ b/new.ts\n@@ -0,0 +1,3 @@\n+a\n+b\n+c\n"
a = parse_diff_anchors(diff)
assert a["new.ts"] == {1, 2, 3}
# ---------------------------------------------------------------------------
# parse_findings
# ---------------------------------------------------------------------------
def test_parse_findings_clean_json():
txt = '{"findings":[{"severity":"high","path":"a.py","line":3,"problem":"x","fix":"y","suggestion":"z"}]}'
fs = parse_findings(txt)
assert len(fs) == 1
assert fs[0]["severity"] == "high"
assert fs[0]["path"] == "a.py"
assert fs[0]["line"] == 3
def test_parse_findings_fenced_json():
txt = '```json\n{"findings":[{"severity":"low","path":"b.go","line":1,"problem":"p","fix":"","suggestion":""}]}\n```'
fs = parse_findings(txt)
assert len(fs) == 1
assert fs[0]["path"] == "b.go"
def test_parse_findings_json_in_prose():
txt = 'Here is my review: {"findings":[{"severity":"critical","path":"c","line":9,"problem":"q"}]} thanks!'
fs = parse_findings(txt)
assert len(fs) == 1
assert fs[0]["severity"] == "critical"
def test_parse_findings_empty():
assert parse_findings('{"findings":[]}') == []
assert parse_findings("") == []
assert parse_findings("not json at all") == []
def test_parse_findings_drops_bad_entries():
# missing path, bad line, unknown severity (normalised)
txt = '{"findings":[{"line":1},{"path":"x","line":-1},{"path":"x","line":2,"severity":"bogus","problem":"p"}]}'
fs = parse_findings(txt)
assert len(fs) == 1
assert fs[0]["severity"] == "medium"
# ---------------------------------------------------------------------------
# split_findings + inline_comment_body + summary_bullets
# ---------------------------------------------------------------------------
def test_split_findings_by_anchor():
anchors = {"a.py": {1, 3, 4}}
fs = [
{"severity": "high", "path": "a.py", "line": 3, "problem": "p", "fix": "f", "suggestion": ""},
{"severity": "low", "path": "a.py", "line": 99, "problem": "off", "fix": "", "suggestion": ""},
{"severity": "medium", "path": "other.go", "line": 1, "problem": "x", "fix": "", "suggestion": ""},
]
anchored, unanchored = split_findings(fs, anchors)
assert [f["line"] for f in anchored] == [3]
assert len(unanchored) == 2
def test_inline_comment_body_with_suggestion():
f = {"severity": "high", "path": "a", "line": 1, "problem": "bad", "fix": "swap", "suggestion": "good()"}
body = inline_comment_body(f)
assert "**[HIGH]**" in body
assert "bad" in body
# no extension → bare fence (Gitea 1.26.x has no apply-suggestion; we tag
# with the file language for highlighting instead of ```suggestion)
assert "```\ngood()\n```" in body
assert "good()" in body
def test_inline_comment_body_suggestion_lang_tagged():
f = {"severity": "high", "path": "src/Foo.java", "line": 1,
"problem": "bad", "fix": "swap", "suggestion": "good();"}
body = inline_comment_body(f)
assert "```java\ngood();\n```" in body
def test_inline_comment_body_no_suggestion():
f = {"severity": "low", "path": "a", "line": 1, "problem": "p", "fix": "f", "suggestion": ""}
body = inline_comment_body(f)
assert "```" not in body
assert "Fix: f" in body
def test_summary_bullets_format():
fs = [{"severity": "high", "path": "a.py", "line": 7, "problem": "p", "fix": "f", "suggestion": ""}]
b = summary_bullets(fs)
assert "- **[HIGH]**" in b
assert "`a.py:7`" in b
# ---------------------------------------------------------------------------
# repo config parsing
# ---------------------------------------------------------------------------
def test_parse_repo_config_full():
raw = '{"focus":["security","perf"],"exclude_paths":["vendor/**"],"languages":["go"],"instructions":"be strict"}'
c = parse_repo_config(raw)
assert c["focus"] == ["security", "perf"]
assert c["exclude_paths"] == ["vendor/**"]
assert c["instructions"] == "be strict"
def test_parse_repo_config_partial_and_bad():
assert parse_repo_config('{"focus":"not-a-list"}') == {}
assert parse_repo_config('{"focus":["ok"]}') == {"focus": ["ok"]}
assert parse_repo_config("") == {}
assert parse_repo_config("not json") == {}
assert parse_repo_config('{"instructions":" "}') == {}
# ---------------------------------------------------------------------------
# dedupe / prior-context parsing
# ---------------------------------------------------------------------------
def test_reviewed_shas_extracts_marker():
reviews = [
{"body": "🤖 AI Review · glm · `abcdef12`\n\n<!-- pragent:sha=abcdef1234567890 -->"},
{"body": "human comment, no marker"},
{"body": "<!-- pragent:sha=0987654 -->"},
]
shas = reviewed_shas(reviews)
assert "abcdef1234567890" in shas
assert "0987654" in shas
def test_reviewed_shas_empty():
assert reviewed_shas([]) == set()
assert reviewed_shas([{"body": "no marker"}]) == set()
def test_prior_review_bodies_skips_current_sha():
reviews = [
{"body": "r1\n<!-- pragent:sha=1111111 -->"},
{"body": "r2\n<!-- pragent:sha=2222222 -->"},
{"body": "no marker here"},
]
prior = prior_review_bodies(reviews, current_sha="2222222")
assert len(prior) == 1
assert "r1" in prior[0]
def test_format_review_body_has_sha_marker():
body = format_review_body("- [high] x:1 — b", "glm-5.2:cloud", "abcdef1234567890")
assert "<!-- pragent:sha=abcdef1234567890 -->" in body
# ---------------------------------------------------------------------------
# parse_review_output (opencode engine: {summary, findings} + reference)
# ---------------------------------------------------------------------------
def test_parse_review_output_summary_and_findings():
txt = (
"This PR adds an eval helper — risky. See findings.\n\n"
"```json\n"
'{"summary":"Adds eval() — security risk.","findings":['
'{"severity":"critical","path":"src/x.ts","line":4,"problem":"eval on user input",'
'"fix":"parse explicitly","suggestion":"const n = Number(s)","reference":"https://owasp.org/x"}'
"]}",
"\n```",
)
summary, fs = parse_review_output("".join(txt))
assert "eval()" in summary
assert len(fs) == 1
assert fs[0]["severity"] == "critical"
assert fs[0]["reference"] == "https://owasp.org/x"
assert fs[0]["suggestion"] == "const n = Number(s)"
def test_parse_review_output_bare_findings_no_summary():
txt = '```json\n{"findings":[{"severity":"low","path":"a","line":1,"problem":"p"}]}\n```'
summary, fs = parse_review_output(txt)
assert summary == ""
assert len(fs) == 1
assert fs[0]["reference"] == "" # default
def test_parse_review_output_empty_and_bogus():
assert parse_review_output("") == ("", [])
assert parse_review_output("no json here") == ("", [])
assert parse_review_output('{"findings":[]}') == ("", [])
def test_parse_review_output_uses_last_json_block():
# Agent emits a stray json-ish block first, then the real one last.
txt = (
"```json\n{\"findings\":[{\"path\":\"x\",\"line\":1,\"severity\":\"low\"}]}\n```\n"
"more prose\n"
"```json\n{\"summary\":\"real\",\"findings\":[{\"path\":\"y\",\"line\":2,\"severity\":\"high\"}]}\n```"
)
summary, fs = parse_review_output(txt)
assert summary == "real"
assert len(fs) == 1
assert fs[0]["path"] == "y"
def test_parse_findings_fenced_json_with_nested_object():
# Real-world regression: agent emits a fence whose inner JSON has nested
# objects. The old regex `\{.*?\}` matched only the first `}`, truncating
# the JSON. Now we balance braces inside the fence.
txt = (
"```json\n"
'{"summary":"x","findings":[{"severity":"high","path":"a.py","line":1,'
'"problem":"p","fix":"f","suggestion":"","reference":""}],"meta":{"engine":"opencode"}}\n'
"```"
)
fs = parse_findings(txt)
assert len(fs) == 1
assert fs[0]["path"] == "a.py"
def test_parse_findings_unfenced_at_tail():
# No fence at all. Agent wrote the JSON inline at the very end of its
# prose. The old first-balanced regex caught the FIRST `{`, not this one.
txt = (
"I considered the diff carefully. Two findings stand out:\n"
"First one is just text.\n"
'{"findings":[{"severity":"critical","path":"x","line":1,"problem":"p","fix":"f"}]}'
)
fs = parse_findings(txt)
assert len(fs) == 1
assert fs[0]["severity"] == "critical"
def test_parse_findings_bare_array():
# Some agents skip the `{"summary":..., "findings":[...]}` wrapper and
# emit just the array.
txt = (
"Here are my findings:\n"
"```json\n"
'[{"severity":"low","path":"a","line":1,"problem":"p","fix":"f","suggestion":"","reference":""}]\n'
"```"
)
fs = parse_findings(txt)
assert len(fs) == 1
assert fs[0]["path"] == "a"
def test_parse_review_output_unfenced_at_tail():
# The exact shape canalhandia produced: long prose, JSON at the very end,
# no fence. Old parser returned ([], salvage) — now we recover findings.
txt = (
"Let me refine the fix: should call a dedicated `setPermanent`.\n"
"Let me finalize. Let me also double-check the `find` thread-safety.\n"
'{"summary":"Adds void protection; one critical race.","findings":['
'{"severity":"high","path":"VoidProtection.java","line":162,'
'"problem":"drop duplication race","fix":"use ItemMeta","suggestion":"","reference":""}]}'
)
summary, fs = parse_review_output(txt)
assert "void protection" in summary.lower()
assert len(fs) == 1
assert fs[0]["path"] == "VoidProtection.java"
def test_parse_review_output_bare_array_at_tail():
txt = (
"All wrapped up.\n"
'[{"severity":"low","path":"a","line":1,"problem":"p","fix":"","suggestion":"","reference":""}]'
)
summary, fs = parse_review_output(txt)
assert summary == ""
assert len(fs) == 1
def test_scan_balanced_handles_braces_in_strings():
# The JSON scanner must not be fooled by `{` or `}` inside string literals.
s = '{"a":"contains { and }","b":1}'
obj = _extract_first_json_object(s)
assert obj == s
d = json.loads(obj)
assert d["a"] == "contains { and }"
def test_last_balanced_json_picks_latest():
s = '{"a":1} some text {"b":2,"nested":{"c":3}} trailing'
out = _last_balanced_json(s)
assert out is not None
d = json.loads(out)
assert d == {"b": 2, "nested": {"c": 3}}
def test_last_balanced_json_no_json():
assert _last_balanced_json("nothing here") is None
assert _last_balanced_json("") is None
def test_balanced_json_substring_skips_leading_prose():
s = 'preamble {"a":1} more prose {"b":2}'
out = _balanced_json_substring(s)
assert out == '{"a":1}'
def test_balanced_json_substring_handles_array():
s = '[{"a":1},{"b":2}]'
out = _balanced_json_substring(s)
assert out == s
# ---------------------------------------------------------------------------
# reference rendering in inline_comment_body + summary_bullets + summary section
# ---------------------------------------------------------------------------
def test_inline_comment_body_renders_reference():
f = {"severity": "high", "path": "a", "line": 1, "problem": "p", "fix": "f",
"suggestion": "", "reference": "https://cve.example/X"}
body = inline_comment_body(f)
assert "📎 ref: https://cve.example/X" in body
def test_inline_comment_body_no_reference_no_ref_line():
f = {"severity": "low", "path": "a", "line": 1, "problem": "p", "fix": "",
"suggestion": "", "reference": ""}
assert "📎 ref" not in inline_comment_body(f)
def test_summary_bullets_renders_reference():
fs = [{"severity": "high", "path": "a.py", "line": 7, "problem": "p", "fix": "f",
"suggestion": "", "reference": "https://r.example"}]
b = summary_bullets(fs)
assert "https://r.example" in b
assert "`a.py:7`" in b
def test_format_review_body_with_summary_section():
body = format_review_body("- [high] x:1 — b", "glm-5.2:cloud", "abcdef1234567890",
summary="This PR adds a risky helper.")
assert "This PR adds a risky helper." in body
assert "- [high] x:1" in body
assert "<!-- pragent:sha=abcdef1234567890 -->" in body
# summary appears before the findings bullets
assert body.index("risky helper") < body.index("[high]")
# ---------------------------------------------------------------------------
# AI-USAGE: compute_attribution + format_usage_section + inline 🪙 line
# ---------------------------------------------------------------------------
def test_compute_attribution_weighted_split():
# weights 1 (problem="a") and 3 (problem="aaa"), output 100 → 25 / 75
fs = [
{"severity": "high", "path": "x", "line": 1, "problem": "a", "fix": "", "suggestion": ""},
{"severity": "low", "path": "x", "line": 2, "problem": "aaa", "fix": "", "suggestion": ""},
]
compute_attribution(fs, 100)
assert fs[0]["_tok_attrib"] == 25
assert fs[1]["_tok_attrib"] == 75
assert abs(fs[0]["_tok_pct"] - 0.25) < 1e-9
assert abs(fs[1]["_tok_pct"] - 0.75) < 1e-9
def test_compute_attribution_zero_weights_splits_evenly():
fs = [
{"severity": "high", "path": "x", "line": 1, "problem": "", "fix": "", "suggestion": ""},
{"severity": "low", "path": "x", "line": 2, "problem": "", "fix": "", "suggestion": ""},
]
compute_attribution(fs, 80)
assert fs[0]["_tok_attrib"] == 40
assert fs[1]["_tok_attrib"] == 40
assert abs(fs[0]["_tok_pct"] - 0.5) < 1e-9
def test_compute_attribution_noop_on_empty_or_zero_budget():
fs = [{"severity": "high", "path": "x", "line": 1, "problem": "a", "fix": "", "suggestion": ""}]
compute_attribution([], 100)
compute_attribution(fs, 0)
assert "_tok_attrib" not in fs[0]
def test_inline_comment_body_with_attribution_line():
f = {"severity": "high", "path": "a", "line": 1, "problem": "bad", "fix": "swap",
"suggestion": "", "_tok_attrib": 180, "_tok_pct": 0.29}
body = inline_comment_body(f)
assert "🪙 ~180 tok" in body
assert "29%" in body
assert "attributed output" in body
def test_inline_comment_body_no_attribution_no_coin_line():
f = {"severity": "high", "path": "a", "line": 1, "problem": "bad", "fix": "swap",
"suggestion": ""}
assert "🪙" not in inline_comment_body(f)
def test_format_usage_section_renders_totals_and_table():
fs = [
{"severity": "critical", "path": "src/Foo.java", "line": 98,
"problem": "p"*10, "fix": "f", "suggestion": "", "_tok_attrib": 180, "_tok_pct": 0.29},
]
usage = {"input": 18420, "output": 612, "reasoning": 0, "cache_read": 15210,
"cache_write": 0, "total": 19032, "cost": 0.0, "steps": 7, "duration_s": 142.0}
sec = format_usage_section(usage, fs, "glm-5.2:cloud")
assert "## 🔋 AI usage" in sec
assert "`glm-5.2:cloud`" in sec
assert "agent steps: 7" in sec
assert "duration: 142.0s" in sec
assert "18420 in" in sec and "612 out" in sec and "19032 total" in sec
assert "$0.00" in sec
assert "whole-repo checkout" in sec
assert "attributed" in sec
# table
assert "| severity | location | ≈out tok | % |" in sec
assert "CRITICAL" in sec
assert "`src/Foo.java:98`" in sec
assert "180" in sec and "29%" in sec
def test_format_usage_section_omits_table_when_no_attributed_rows():
usage = {"input": 10, "output": 0, "reasoning": 0, "cache_read": 0,
"cache_write": 0, "total": 10, "cost": 0.0, "steps": 1, "duration_s": 1.0}
sec = format_usage_section(usage, [], "glm-5.2:cloud")
assert "## 🔋 AI usage" in sec
assert "severity | location" not in sec # no rows → no table
def test_format_usage_section_none_returns_empty():
assert format_usage_section(None, [], "m") == ""
def test_format_usage_section_cost_nonzero():
usage = {"input": 10, "output": 0, "reasoning": 0, "cache_read": 0,
"cache_write": 0, "total": 10, "cost": 0.0123, "steps": 1, "duration_s": 1.0}
sec = format_usage_section(usage, [], "m")
assert "$0.0123" in sec
assert "billed by provider" in sec
def test_format_review_body_usage_section_between_summary_and_findings():
usage_sec = "## 🔋 AI usage\n\n- model: `m`"
body = format_review_body("- [high] x:1 — b", "glm-5.2:cloud", "abcdef1234567890",
summary="This PR is risky.", usage_section=usage_sec)
# order: header < summary < usage < findings < marker
assert body.index("risky.") < body.index("AI usage")
assert body.index("AI usage") < body.index("[high]")
assert body.index("[high]") < body.index("<!-- pragent:sha=")
assert "## 🔋 AI usage" in body
def test_format_review_body_no_usage_section_omitted():
body = format_review_body("- [high] x:1 — b", "glm-5.2:cloud", "abcdef1234567890")
assert "AI usage" not in body
# ---------------------------------------------------------------------------
# parse_diff_anchors — empty context lines
# ---------------------------------------------------------------------------
def test_parse_diff_anchors_counts_empty_context_line():
# A context line that is *blank* arrives as "" when trailing whitespace was
# stripped somewhere upstream. If it isn't counted, every later line in the
# hunk is off by one.
diff = (
"diff --git a/x.py b/x.py\n"
"--- a/x.py\n"
"+++ b/x.py\n"
"@@ -1,4 +1,5 @@\n"
" import os\n"
"\n" # blank context line, whitespace stripped
" def f():\n"
"+ return 1\n"
" # tail\n"
)
anchors = parse_diff_anchors(diff)
assert anchors["x.py"] == {1, 2, 3, 4, 5}
def test_parse_diff_anchors_space_prefixed_blank_line_still_counts():
diff = (
"+++ b/y.py\n"
"@@ -1,3 +1,4 @@\n"
" a\n"
" \n" # properly space-prefixed blank context line
"+b\n"
" c\n"
)
assert parse_diff_anchors(diff)["y.py"] == {1, 2, 3, 4}
# ---------------------------------------------------------------------------
# parse_repo_config — caps
# ---------------------------------------------------------------------------
def test_parse_repo_config_caps_instructions_length():
cfg = parse_repo_config(json.dumps({"instructions": "x" * 99999}))
assert len(cfg["instructions"]) == ai_review.CONFIG_MAX_INSTRUCTIONS_CHARS
def test_parse_repo_config_caps_list_length_and_items():
cfg = parse_repo_config(json.dumps({
"focus": ["a" * 9999] * 500,
"exclude_paths": ["vendor/**"],
}))
assert len(cfg["focus"]) == ai_review.CONFIG_MAX_LIST_ITEMS
assert all(len(x) == ai_review.CONFIG_MAX_ITEM_CHARS for x in cfg["focus"])
assert cfg["exclude_paths"] == ["vendor/**"]
def test_parse_repo_config_still_accepts_normal_config():
cfg = parse_repo_config(json.dumps({
"focus": ["security"], "languages": ["go"], "instructions": "No bare throw.",
}))
assert cfg == {"focus": ["security"], "languages": ["go"], "instructions": "No bare throw."}
# ---------------------------------------------------------------------------
# fetch_repo_config — reads the BASE ref, never the PR head
# ---------------------------------------------------------------------------
def _stub_config_response(payload: dict):
blob = base64.b64encode(json.dumps(payload).encode()).decode()
return 200, json.dumps({"content": blob}).encode()
def test_fetch_repo_config_uses_given_base_ref(monkeypatch):
seen = {}
def fake_get(api, repo, path, token, accept="application/json"):
seen["path"] = path
return _stub_config_response({"focus": ["security"]})
monkeypatch.setattr(ai_review, "gitea_get", fake_get)
cfg = ai_review.fetch_repo_config("http://g", "o/r", "tok", ref="main")
assert cfg == {"focus": ["security"]}
assert seen["path"] == "contents/.pr-review.json?ref=main"
def test_fetch_repo_config_without_ref_omits_ref_param(monkeypatch):
seen = {}
def fake_get(api, repo, path, token, accept="application/json"):
seen["path"] = path
return _stub_config_response({})
monkeypatch.setattr(ai_review, "gitea_get", fake_get)
ai_review.fetch_repo_config("http://g", "o/r", "tok")
assert "?ref=" not in seen["path"]
def test_fetch_repo_config_quotes_ref_with_slashes(monkeypatch):
seen = {}
def fake_get(api, repo, path, token, accept="application/json"):
seen["path"] = path
return _stub_config_response({})
monkeypatch.setattr(ai_review, "gitea_get", fake_get)
ai_review.fetch_repo_config("http://g", "o/r", "tok", ref="release/v1 x")
assert "release%2Fv1%20x" in seen["path"]
# ---------------------------------------------------------------------------
# post_inline_review — degraded fallback must not lose findings
# ---------------------------------------------------------------------------
def test_post_inline_review_fallback_keeps_anchored_findings(monkeypatch):
posted = []
def fake_post(api, repo, path, token, body):
posted.append((path, body))
# Reject the inline review, accept the plain one.
if "reviews" in path and body.get("comments"):
return 422, b"bad line"
return 201, b"{}"
monkeypatch.setattr(ai_review, "gitea_post", fake_post)
anchored = [{
"severity": "high", "path": "a.py", "line": 7,
"problem": "off-by-one", "fix": "use <=", "suggestion": "", "reference": "",
}]
ai_review.post_inline_review("http://g", "o/r", "1", "tok", "SUMMARY", anchored)
final_body = posted[-1][1]["body"]
assert "off-by-one" in final_body
assert "a.py:7" in final_body
assert "SUMMARY" in final_body
def test_post_inline_review_success_posts_no_fallback(monkeypatch):
posted = []
def fake_post(api, repo, path, token, body):
posted.append(path)
return 201, b"{}"
monkeypatch.setattr(ai_review, "gitea_post", fake_post)
ai_review.post_inline_review("http://g", "o/r", "1", "tok", "S", [])
assert len(posted) == 1
# ---------------------------------------------------------------------------
# fetch_pr_diff — files-endpoint fallback
# ---------------------------------------------------------------------------
def test_fetch_pr_diff_fallback_emits_git_style_prefixes(monkeypatch):
def fake_get(api, repo, path, token, accept="application/json"):
if path.endswith(".diff"):
return 404, b"nope"
return 200, json.dumps([
{"filename": "src/a.py", "patch": "@@ -1 +1,2 @@\n a\n+b"},
]).encode()
monkeypatch.setattr(ai_review, "gitea_get", fake_get)
diff, truncated, _ = ai_review.fetch_pr_diff("http://g", "o/r", "1", "tok", 10000)
assert "--- a/src/a.py" in diff
assert "+++ b/src/a.py" in diff
assert truncated is False
# And the synthesized diff must actually anchor.
assert parse_diff_anchors(diff)["src/a.py"] == {1, 2}
def test_fetch_pr_diff_error_reports_both_statuses(monkeypatch):
def fake_get(api, repo, path, token, accept="application/json"):
return (404, b"") if path.endswith(".diff") else (500, b"")
monkeypatch.setattr(ai_review, "gitea_get", fake_get)
try:
ai_review.fetch_pr_diff("http://g", "o/r", "1", "tok", 10000)
except RuntimeError as e:
assert ".diff=404" in str(e)
assert "files=500" in str(e)
else:
raise AssertionError("expected RuntimeError")
# ---------------------------------------------------------------------------
# salvage_summary — don't discard an expensive run over a missing JSON block
# ---------------------------------------------------------------------------
def test_salvage_summary_keeps_the_prose():
text = "I reviewed the diff. The retry loop in worker.py never terminates."
out = ai_review.salvage_summary(text)
assert "never terminates" in out
assert "unverified" in out
def test_salvage_summary_drops_fenced_blocks():
text = 'Analysis here.\n\n```json\n{"findings": [ truncated...\n'
out = ai_review.salvage_summary(text)
assert "Analysis here." in out
# The half-written JSON blob is gone (the banner legitimately says
# "findings", so assert on the blob's own content instead).
assert "truncated..." not in out
assert "[" not in out.split("_\n\n", 1)[1]
def test_salvage_summary_drops_complete_fences_too():
text = "Before.\n```python\nprint(1)\n```\nAfter."
out = ai_review.salvage_summary(text)
assert "Before." in out and "After." in out
assert "print(1)" not in out
def test_salvage_summary_keeps_the_tail_when_long():
text = "x" * 9000 + " FINAL CONCLUSION"
out = ai_review.salvage_summary(text, max_chars=1000)
assert "FINAL CONCLUSION" in out # the conclusion is written last
assert len(out) < 1600
def test_salvage_summary_empty_when_nothing_to_salvage():
assert ai_review.salvage_summary("") == ""
assert ai_review.salvage_summary(" \n ") == ""
assert ai_review.salvage_summary("```json\n{}\n```") == ""
# ---------------------------------------------------------------------------
# equivalent_cost + format_usage_section equivalent-provider line
# ---------------------------------------------------------------------------
def test_equivalent_cost_matches_cost_model():
usage = {"input": 1_000_000, "output": 0, "cache_read": 0, "cache_write": 0}
eq = ai_review.equivalent_cost(usage, "claude-sonnet-5")
# Sonnet 5 input is $2/MTok, so 1M input = $2.00 exactly.
assert abs(eq - 2.0) < 1e-9
def test_equivalent_cost_unknown_key_returns_zero():
assert ai_review.equivalent_cost({"input": 100}, "bogus") == 0.0
def test_format_usage_section_shows_equivalent_provider_cost():
usage = {"input": 200000, "output": 4000, "reasoning": 0,
"cache_read": 0, "cache_write": 0, "total": 204000,
"cost": 0.0, "steps": 6, "duration_s": 100.0}
sec = ai_review.format_usage_section(usage, [], "glm-5.2:cloud")
# Two cost lines now: an equivalent (default Sonnet 5) AND the $0 actual.
assert "## 🔋 AI usage" in sec
assert "est. cost on **Claude Sonnet 5**" in sec
assert "actual: $0.00" in sec
assert "free tier" in sec
# Equivalent should be > 0 for non-trivial token counts.
assert "$0.00" in sec # the actual line
# And a non-zero one for the equivalent.
import re
cost_lines = [ln for ln in sec.splitlines() if "cost on" in ln]
assert len(cost_lines) == 1
assert re.search(r"\$\d", cost_lines[0]) is not None
assert "$0.00" not in cost_lines[0]
def test_format_usage_section_honors_cost_target(monkeypatch):
monkeypatch.setenv("PRAGENT_PRICE_TARGET", "claude-haiku-4-5")
usage = {"input": 1000, "output": 100, "reasoning": 0,
"cache_read": 0, "cache_write": 0, "total": 1100,
"cost": 0.0, "steps": 1, "duration_s": 5.0}
sec = ai_review.format_usage_section(usage, [], "glm-5.2:cloud")
assert "Claude Haiku 4.5" in sec
# 1k * $1/MTok + 100 * $5/MTok = 0.001 + 0.0005 = $0.0015
assert "$0.0015" in sec
def test_format_usage_section_respects_repo_config_cost_target(monkeypatch):
monkeypatch.delenv("PRAGENT_PRICE_TARGET", raising=False)
usage = {"input": 1000, "output": 100, "reasoning": 0,
"cache_read": 0, "cache_write": 0, "total": 1100,
"cost": 0.0, "steps": 1, "duration_s": 5.0}
sec = ai_review.format_usage_section(
usage, [], "glm-5.2:cloud", config={"cost_target": "claude-opus-5"}
)
assert "Claude Opus 5" in sec
# Opus 5 = $5/MTok input + $25/MTok output → 1000*5e-6 + 100*25e-6 = 0.0075
assert "$0.0075" in sec
def test_format_usage_section_reports_unknown_price_target():
usage = {"input": 100, "output": 100, "reasoning": 0,
"cache_read": 0, "cache_write": 0, "total": 200,
"cost": 0.0, "steps": 1, "duration_s": 1.0}
sec = ai_review.format_usage_section(
usage, [], "glm-5.2:cloud", config={"cost_target": "bogus-model"}
)
# Falls back to default + surfaces the error in the line.
assert "Claude Sonnet 5" in sec
assert "unknown price target" in sec
assert "bogus-model" in sec
# ---------------------------------------------------------------------------
# parse_repo_config — extended schema
# ---------------------------------------------------------------------------
def test_parse_repo_config_new_fields_all_valid():
raw = json.dumps({
"focus": ["security"],
"style": "strict",
"severity_threshold": "high",
"max_findings": 5,
"exclude_tests": True,
"require_tests": True,
"patterns": {"allow": ["src/**"], "deny": ["**/*.test.ts"]},
"cost_target": "claude-opus-5",
})
c = ai_review.parse_repo_config(raw)
assert c["style"] == "strict"
assert c["severity_threshold"] == "high"
assert c["max_findings"] == 5
assert c["exclude_tests"] is True
assert c["require_tests"] is True
assert c["patterns"]["allow"] == ["src/**"]
assert c["patterns"]["deny"] == ["**/*.test.ts"]
assert c["cost_target"] == "claude-opus-5"
def test_parse_repo_config_rejects_bad_style_and_threshold():
c = ai_review.parse_repo_config(json.dumps({"style": "wild", "severity_threshold": "meh"}))
assert "style" not in c
assert "severity_threshold" not in c
def test_parse_repo_config_caps_max_findings():
c1 = ai_review.parse_repo_config(json.dumps({"max_findings": 0}))
c2 = ai_review.parse_repo_config(json.dumps({"max_findings": 999}))
c3 = ai_review.parse_repo_config(json.dumps({"max_findings": "12"}))
assert "max_findings" not in c1 # 0 invalid
assert "max_findings" not in c2 # > 30 invalid
assert c3["max_findings"] == 12 # numeric string accepted
def test_parse_repo_config_caps_patterns():
raw = json.dumps({
"patterns": {"allow": [f"a{i}" for i in range(20)], "deny": [f"d{i}" for i in range(20)]}
})
c = ai_review.parse_repo_config(raw)
assert len(c["patterns"]["allow"]) == ai_review.CONFIG_MAX_PATTERNS_ITEMS
assert len(c["patterns"]["deny"]) == ai_review.CONFIG_MAX_PATTERNS_ITEMS
def test_effective_config_applies_style_defaults():
eff = ai_review.effective_config({"focus": ["security"]})
assert eff["style"] == "balanced"
assert eff["max_findings"] == 12
assert eff["severity_threshold"] == "medium"
assert eff["focus"] == ["security"]
def test_effective_config_style_overrides_fields():
eff = ai_review.effective_config({"style": "strict"})
assert eff["max_findings"] == 5
assert eff["severity_threshold"] == "high"
# ---------------------------------------------------------------------------
# apply_repo_config — filter findings
# ---------------------------------------------------------------------------
_FINDINGS = [
{"severity": "critical", "path": "src/main.py", "line": 1, "problem": "p", "fix": "f", "suggestion": ""},
{"severity": "high", "path": "src/main.py", "line": 5, "problem": "p", "fix": "f", "suggestion": ""},
{"severity": "medium", "path": "src/main.py", "line": 9, "problem": "p", "fix": "f", "suggestion": ""},
{"severity": "low", "path": "src/main.py", "line": 13, "problem": "p", "fix": "f", "suggestion": ""},
{"severity": "high", "path": "src/FooTest.java", "line": 22, "problem": "p", "fix": "f", "suggestion": ""},
{"severity": "medium", "path": "src/app.test.ts", "line": 7, "problem": "p", "fix": "f", "suggestion": ""},
]
def test_apply_repo_config_severity_threshold():
kept, dropped = ai_review.apply_repo_config(_FINDINGS, {"severity_threshold": "high"})
assert len(kept) == 3 # critical + 2 highs (main.py + FooTest.java)
assert all(f["severity"] in ("critical", "high") for f in kept)
assert len(dropped) == 3
def test_apply_repo_config_exclude_tests_drops_test_files():
kept, dropped = ai_review.apply_repo_config(_FINDINGS, {"exclude_tests": True})
paths = {f["path"] for f in kept}
assert "src/FooTest.java" not in paths
assert "src/app.test.ts" not in paths
def test_apply_repo_config_patterns_deny_drops_matching():
cfg = {"patterns": {"deny": ["src/main.py"]}}
kept, dropped = ai_review.apply_repo_config(_FINDINGS, cfg)
paths = {f["path"] for f in kept}
assert "src/main.py" not in paths
def test_apply_repo_config_patterns_allow_keeps_only_matching():
cfg = {"patterns": {"allow": ["src/main.py"]}}
kept, dropped = ai_review.apply_repo_config(_FINDINGS, cfg)
paths = {f["path"] for f in kept}
assert paths == {"src/main.py"}
def test_apply_repo_config_max_findings_caps():
kept, dropped = ai_review.apply_repo_config(_FINDINGS, {"max_findings": 2})
assert len(kept) == 2
# Highest-severity first (critical, then high)
assert kept[0]["severity"] == "critical"
assert kept[1]["severity"] == "high"
def test_apply_repo_config_exclude_paths_glob():
cfg = {"exclude_paths": ["src/main.py"]}
kept, dropped = ai_review.apply_repo_config(_FINDINGS, cfg)
assert "src/main.py" not in {f["path"] for f in kept}
def test_apply_repo_config_require_tests_synthetic_finding():
cfg = {"require_tests": True}
changed = ["src/main.py", "src/lib.ts"]
kept, dropped = ai_review.apply_repo_config([], cfg, changed_paths=changed)
assert any(f.get("_config_synthetic") for f in kept)
def test_apply_repo_config_require_tests_no_synthetic_when_tests_present():
cfg = {"require_tests": True}
changed = ["src/main.py", "src/main_test.py"]
kept, dropped = ai_review.apply_repo_config(_FINDINGS, cfg, changed_paths=changed)
assert not any(f.get("_config_synthetic") for f in kept)
def test_is_test_path_recognises_common_patterns():
assert ai_review.is_test_path("src/FooTest.java")
assert ai_review.is_test_path("src/foo.test.ts")
assert ai_review.is_test_path("tests/foo_test.py")
assert ai_review.is_test_path("test_foo.py")
assert ai_review.is_test_path("packages/app/__tests__/foo.js")
assert not ai_review.is_test_path("src/main.py")
assert not ai_review.is_test_path("src/testing.py") # "testing" ≠ "test_"
# ---------------------------------------------------------------------------
# compact_prior_reviews
# ---------------------------------------------------------------------------
def test_compact_prior_reviews_drops_prose_keeps_bullets():
bodies = [
"🤖 AI Review · m · `abc`\n\nLong prose.\n\n- **[HIGH]** `a.py:1` — bug.\n- **[LOW]** `b.go:2` — nit.\n\n_2 inline comments posted._\n<!-- pragent:sha=abc -->",
"Just chatter, no findings.",
]
out = ai_review.compact_prior_reviews(bodies)
assert len(out) == 1
assert "HIGH" in out[0] and "a.py:1" in out[0]
assert "Long prose." not in out[0]
assert "inline comments posted" not in out[0]
def test_compact_prior_reviews_empty_and_none():
assert ai_review.compact_prior_reviews([]) == []
assert ai_review.compact_prior_reviews(None) == []