Files
pragent/tests/pilot/test_opencode_review.py
T
Marcos 087834565d feat(pilot): token-usage reporting gated by AI-USAGE label
Add per-review + per-comment token accounting, surfaced only when a PR carries
the new AI-USAGE label (on top of the existing AI-REVIEW trigger).

opencode_review:
- run_opencode now uses `--format json`; parse_opencode_events reconstructs the
  assistant text from `text` events and sums tokens/cost/steps from every
  `step_finish` event (tolerant of noise / missing fields).
- run() measures duration_s around the opencode call and returns (text, usage).
- changed_files(diff) extracts the `+++ b/` paths; the brief now lists them
  under a "Changed files" focus block so the agent grounds findings in the
  diff's neighbourhood instead of unbounded whole-repo walks.

ai_review:
- format_usage_section renders a `## AI usage` block: measured totals
  (in/out/reasoning/cache/cost/steps/duration), the whole-repo scope note, and
  an attributed per-finding table. Per-comment counts are output tokens split by
  each finding's body weight — labelled "attributed" since one model pass
  produces all findings.
- inline_comment_body appends `🪙 ~N tok (X% · attributed output)` when
  attribution is present.
- review_pr gains report_usage; compute_attribution stashes _tok_attrib/_tok_pct.
- format_review_body inserts the usage section between summary and findings.

webhook_server:
- Fire on every pull_request action except `closed` (denylist, was an allowlist)
  — the AI-REVIEW gate + sha dedupe keep this safe.
- AI-USAGE label detection + PRAGENT_USAGE_ALWAYS env drive report_usage.

.opencode factory + review-methodology skill: new "Ground findings in context"
step — read callers/imports/sibling functions per changed file (1-3 files per
finding), no unbounded walks.

Tests: parse_opencode_events (text+usage sum, malformed tolerance, none-usage),
changed_files, compute_attribution math, inline 🪙 line, format_usage_section
totals/table/cost, format_review_body ordering. 68 passing.

Co-Authored-By: Claude <noreply@anthropic.com>
2026-08-18 04:15:11 +00:00

227 lines
8.3 KiB
Python

"""Unit tests for the opencode engine glue (no network, no opencode run)."""
import io
import os
import sys
import tarfile
HERE = os.path.dirname(os.path.abspath(__file__))
ROOT = os.path.abspath(os.path.join(HERE, "..", ".."))
sys.path.insert(0, os.path.join(ROOT, "pilot"))
import opencode_review as oc # noqa: E402
# ---------------------------------------------------------------------------
# write_brief
# ---------------------------------------------------------------------------
def test_write_brief_contains_key_sections(tmp_path):
brief = oc.write_brief(
str(tmp_path),
repo="masi/portfolio", index="3", sha="abcdef1234567890",
title="Add eval helper", description="Closes #1",
diff="diff --git a/x b/x\n+++ b/x\n@@ -1 +1,2 @@\n+eval(input())",
config={"focus": ["security"], "instructions": "Flag eval()."},
prior_reviews=["🤖 AI Review …\n- [high] old finding"],
)
assert brief.endswith(".pragent/brief.md")
text = open(brief, encoding="utf-8").read()
assert "masi/portfolio" in text
assert "#3" in text
assert "abcdef1234567890" in text
assert "Add eval helper" in text
assert "Closes #1" in text
assert "eval(input())" in text
assert "security" in text
assert "Flag eval()" in text
assert "old finding" in text
assert "POST-CHANGE" in text # anchor hint
def test_write_brief_none_config_and_prior(tmp_path):
brief = oc.write_brief(
str(tmp_path), repo="o/r", index="1", sha="sha1234567",
title="t", description="", diff="d", config=None, prior_reviews=None,
)
text = open(brief, encoding="utf-8").read()
assert "_(none)_" in text # both config and prior fall back to none
assert "diff" in text
# ---------------------------------------------------------------------------
# _extract_tar_strip_one — strips the single top-level dir
# ---------------------------------------------------------------------------
def _make_tar(top: str) -> bytes:
"""Build a tar.gz in memory with one top-level dir `top` containing files."""
buf = io.BytesIO()
with tarfile.open(fileobj=buf, mode="w:gz") as tar:
# dir
ti = tarfile.TarInfo(name=f"{top}/")
ti.type = tarfile.DIRTYPE
tar.addfile(ti)
# file src/a.py
data = b"print('a')\n"
ti = tarfile.TarInfo(name=f"{top}/src/a.py")
ti.size = len(data)
tar.addfile(ti, io.BytesIO(data))
# file README.md
data = b"# hi\n"
ti = tarfile.TarInfo(name=f"{top}/README.md")
ti.size = len(data)
tar.addfile(ti, io.BytesIO(data))
return buf.getvalue()
def test_extract_tar_strips_top_level_dir(tmp_path):
blob = _make_tar("repo-deadbeef")
oc._extract_tar_strip_one(blob, str(tmp_path))
# files sit directly at dest root (prefix stripped)
assert os.path.isfile(tmp_path / "README.md")
assert os.path.isfile(tmp_path / "src" / "a.py")
assert not os.path.isdir(tmp_path / "repo-deadbeef") # top dir gone
def test_extract_tar_no_common_prefix_extracts_as_is(tmp_path):
# Two different top-level entries -> no strip.
buf = io.BytesIO()
with tarfile.open(fileobj=buf, mode="w:gz") as tar:
for name, data in (("a.txt", b"A"), ("b.txt", b"B")):
ti = tarfile.TarInfo(name=name)
ti.size = len(data)
tar.addfile(ti, io.BytesIO(data))
oc._extract_tar_strip_one(buf.getvalue(), str(tmp_path))
assert os.path.isfile(tmp_path / "a.txt")
assert os.path.isfile(tmp_path / "b.txt")
def test_extract_tar_skips_parent_traversal(tmp_path):
buf = io.BytesIO()
with tarfile.open(fileobj=buf, mode="w:gz") as tar:
ti = tarfile.TarInfo(name="top/../../escape.txt")
data = b"evil"
ti.size = len(data)
tar.addfile(ti, io.BytesIO(data))
ti = tarfile.TarInfo(name="top/ok.txt")
data = b"ok"
ti.size = len(data)
tar.addfile(ti, io.BytesIO(data))
oc._extract_tar_strip_one(buf.getvalue(), str(tmp_path))
assert os.path.isfile(tmp_path / "ok.txt")
assert not os.path.isfile(tmp_path / "escape.txt")
assert not os.path.isfile(os.path.join(str(tmp_path), "..", "escape.txt"))
# ---------------------------------------------------------------------------
# drop_factory — copies opencode.json + .opencode/ from the repo
# ---------------------------------------------------------------------------
def test_drop_factory_copies_config_and_agents(tmp_path):
oc.drop_factory(str(tmp_path))
assert os.path.isfile(tmp_path / "opencode.json")
assert os.path.isfile(tmp_path / ".opencode" / "agents" / "pragent.md")
assert os.path.isfile(tmp_path / ".opencode" / "skills" / "findings-schema" / "SKILL.md")
# ---------------------------------------------------------------------------
# changed_files — extract changed paths from a unified diff
# ---------------------------------------------------------------------------
def test_changed_files_extracts_new_side_paths():
diff = (
"diff --git a/src/a.py b/src/a.py\n+++ b/src/a.py\n@@ -1 +1 @@\n-x\n+y\n"
"diff --git a/README.md b/README.md\n+++ b/README.md\n@@ -1 +1 @@\n+z\n"
)
assert oc.changed_files(diff) == ["README.md", "src/a.py"]
def test_changed_files_skips_deletions_and_dedups():
diff = (
"diff --git a/gone.txt b/gone.txt\n+++ /dev/null\n@@ -1 +0,0 @@\n-old\n"
"diff --git a/dup.go b/dup.go\n+++ b/dup.go\n@@ -1 +1 @@\n+a\n"
"diff --git a/dup.go b/dup.go\n+++ b/dup.go\n@@ -1 +1 @@\n+b\n"
)
assert oc.changed_files(diff) == ["dup.go"]
def test_changed_files_empty():
assert oc.changed_files("") == []
assert oc.changed_files("no diff headers here") == []
def test_write_brief_lists_changed_files(tmp_path):
brief = oc.write_brief(
str(tmp_path), repo="o/r", index="1", sha="abcdef1234567890",
title="t", description="d",
diff="diff --git a/src/x.ts b/src/x.ts\n+++ b/src/x.ts\n@@ -1 +1 @@\n+x",
config=None, prior_reviews=None,
)
text = open(brief, encoding="utf-8").read()
assert "Changed files (focus your context research here)" in text
assert "`src/x.ts`" in text
# ---------------------------------------------------------------------------
# parse_opencode_events — NDJSON → (text, usage)
# ---------------------------------------------------------------------------
def _ev(obj):
import json
return json.dumps(obj)
def test_parse_events_text_and_usage_summed():
stdout = "\n".join([
_ev({"type": "step_start", "part": {}}),
_ev({"type": "text", "part": {"text": "Hello "}}),
_ev({"type": "text", "part": {"text": "world"}}),
_ev({"type": "step_finish", "part": {
"tokens": {"total": 100, "input": 90, "output": 10,
"reasoning": 0, "cache": {"write": 0, "read": 5}},
"cost": 0.0}}),
_ev({"type": "text", "part": {"text": " more"}}),
_ev({"type": "step_finish", "part": {
"tokens": {"total": 50, "input": 40, "output": 10,
"reasoning": 2, "cache": {"write": 1, "read": 0}},
"cost": 0.01}}),
])
text, usage = oc.parse_opencode_events(stdout)
assert text == "Hello world more"
assert usage is not None
assert usage["steps"] == 2
assert usage["input"] == 130
assert usage["output"] == 20
assert usage["reasoning"] == 2
assert usage["cache_read"] == 5
assert usage["cache_write"] == 1
assert usage["total"] == 150
assert abs(usage["cost"] - 0.01) < 1e-9
def test_parse_events_no_step_finish_returns_none_usage():
stdout = _ev({"type": "text", "part": {"text": "only text"}})
text, usage = oc.parse_opencode_events(stdout)
assert text == "only text"
assert usage is None
def test_parse_events_tolerates_noise_and_malformed():
stdout = "\n".join([
"not json at all",
_ev({"type": "text", "part": {"text": "ok"}}),
"{ broken json",
_ev({"type": "step_finish", "part": {}}), # no tokens field -> counted, zero
_ev({"type": "tool_start", "part": {"text": "ignored"}}),
" ",
])
text, usage = oc.parse_opencode_events(stdout)
assert text == "ok"
# step_finish with no tokens still counts as a step; usage dict returned
assert usage is not None
assert usage["steps"] == 1
assert usage["input"] == 0 and usage["output"] == 0