refactor: organize pilot packages

Group review, feedback, evaluation, observability, and entrypoint code into packages. Keep thin top-level compatibility shims for existing scripts and imports, and mirror the structure in the tests.
This commit is contained in:
Claude
2026-09-01 00:59:51 +00:00
parent 3a110ab52c
commit 7a510a926d
66 changed files with 8174 additions and 8039 deletions
+5 -210
View File
@@ -1,212 +1,7 @@
#!/usr/bin/env python3
"""pragent pilot — populate the Experiments tab from reviews already traced.
An "experiment" in Langfuse is a dataset run: a set of (dataset item, trace)
links under one run name. The Experiments tab then shows one row per item with
its scores, and lets two runs be diffed side by side.
Nothing here re-runs the reviewer. Every PR in `pragent-reviews` has already
been reviewed, and each of those reviews left a trace carrying its findings,
cost and scores. This links what exists, which is what makes the tab useful on
day one instead of after the next N pushes.
Runs are grouped by **model** by default, because that is the comparison the
pilot actually needs to make: the same PRs reviewed by MiniMax vs whatever
replaces it, with `finding_rate` and `cost_per_finding` side by side. Group by
`none` for a single "all traces" run.
One trace per (run, item) — the most recent. A PR re-reviewed on every push has
many traces, and a dataset run is defined as one output per input; feeding it
the other five would make the per-run averages meaningless.
Note on the endpoint: `POST /api/public/dataset-run-items` is deprecated in
favour of the SDK experiment runner / OTel ingestion, and disappears in
Langfuse v4. This instance is self-hosted v3, which the deprecation notice
explicitly exempts from the cutoff date, and the pilot is stdlib-only by
design. Revisit when this deployment moves to v4.
Usage:
LANGFUSE_HOST=... LANGFUSE_PUBLIC_KEY=... LANGFUSE_SECRET_KEY=... \\
python3 eval_experiment.py --dry-run
"""
from __future__ import annotations
import argparse
import json
import os
"""Compatibility import for evaluation experiments."""
import importlib
import sys
import urllib.parse
from collections import defaultdict
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import eval_bootstrap as eb # noqa: E402
TRACE_NAME = "pr-review"
# ---------------------------------------------------------------------------
# Reading what already exists
# ---------------------------------------------------------------------------
def fetch_traces(name: str = TRACE_NAME, limit: int = 100, max_pages: int = 50) -> list[dict]:
"""Every review trace, newest first."""
out: list[dict] = []
for page in range(1, max_pages + 1):
q = urllib.parse.urlencode({"name": name, "limit": limit, "page": page})
st, body = eb._call("GET", f"/api/public/traces?{q}")
if st != 200 or not isinstance(body, dict):
raise SystemExit(f"listing traces failed: {st} {body}")
data = body.get("data") or []
out.extend(data)
meta = body.get("meta") or {}
if page * meta.get("limit", limit) >= meta.get("totalItems", 0):
break
return out
def fetch_item_ids(dataset: str) -> set[str]:
"""Ids present in the dataset, so runs never reference a missing item."""
ids: set[str] = set()
for page in range(1, 51):
q = urllib.parse.urlencode({"datasetName": dataset, "limit": 100, "page": page})
st, body = eb._call("GET", f"/api/public/dataset-items?{q}")
if st != 200 or not isinstance(body, dict):
raise SystemExit(f"listing dataset items failed: {st} {body}")
ids.update(i["id"] for i in body.get("data") or [])
meta = body.get("meta") or {}
if page * meta.get("limit", 100) >= meta.get("totalItems", 0):
break
return ids
# ---------------------------------------------------------------------------
# Grouping traces into runs
# ---------------------------------------------------------------------------
def trace_model(trace: dict) -> str:
"""The model that produced a review, from its `model:` tag."""
for tag in trace.get("tags") or []:
if tag.startswith("model:"):
return tag[len("model:"):] or "unknown"
return "unknown"
def trace_item_id(trace: dict) -> str | None:
"""The dataset item a trace belongs to, or None if it is not a PR review."""
md = trace.get("metadata") or {}
repo, pr = md.get("repo"), md.get("pr")
if not repo or pr in (None, ""):
return None
return eb.item_id(str(repo), pr)
def _sort_key(trace: dict):
return (trace.get("timestamp") or "", trace.get("id") or "")
def plan_runs(traces: list[dict], known_items: set[str], group_by: str = "model") -> dict:
"""Map run name -> {item id: trace}, keeping only the newest trace per item.
Traces whose PR is not in the dataset are dropped: `feedback.db` is the
source for both, but a review can be traced without its row landing (the
posting step can fail after the model ran), and a run item pointing at a
non-existent dataset item is rejected.
"""
runs: dict[str, dict[str, dict]] = defaultdict(dict)
skipped_no_item, skipped_unknown = 0, 0
for tr in traces:
iid = trace_item_id(tr)
if iid is None:
skipped_unknown += 1
continue
if iid not in known_items:
skipped_no_item += 1
continue
run = "all-traces" if group_by == "none" else trace_model(tr)
prev = runs[run].get(iid)
if prev is None or _sort_key(tr) > _sort_key(prev):
runs[run][iid] = tr
return {
"runs": dict(runs),
"skipped_not_in_dataset": skipped_no_item,
"skipped_not_a_review": skipped_unknown,
}
def run_name(prefix: str, key: str) -> str:
return f"{prefix}-{key}" if prefix else key
# ---------------------------------------------------------------------------
# Writing the runs
# ---------------------------------------------------------------------------
def create_run(name: str, items: dict[str, dict], description: str = "") -> dict:
"""Link each (item, trace) pair into the named run. Idempotent per pair."""
created, failed = 0, []
for iid, tr in sorted(items.items()):
md = tr.get("metadata") or {}
body = {
"runName": name,
"runDescription": description,
"datasetItemId": iid,
"traceId": tr["id"],
"metadata": {
"model": trace_model(tr),
"engine": md.get("engine"),
"findings": md.get("findings"),
"duration_s": md.get("duration_s"),
"cost_basis": md.get("cost_basis"),
"linked_by": "eval_experiment.py",
},
}
st, resp = eb._call("POST", "/api/public/dataset-run-items", body)
if st in (200, 201):
created += 1
else:
failed.append({"item": iid, "status": st, "error": resp})
return {"run": name, "items_linked": created, "failed": failed}
def main(argv: list[str] | None = None) -> int:
ap = argparse.ArgumentParser(description=__doc__)
ap.add_argument("--dataset", default=eb.DATASET_NAME)
ap.add_argument("--group-by", choices=("model", "none"), default="model")
ap.add_argument("--prefix", default="baseline",
help="run name prefix; '' for the bare group key")
ap.add_argument("--dry-run", action="store_true")
args = ap.parse_args(argv)
traces = fetch_traces()
items = fetch_item_ids(args.dataset)
plan = plan_runs(traces, items, group_by=args.group_by)
report = {
"traces_read": len(traces),
"dataset_items": len(items),
"skipped_not_in_dataset": plan["skipped_not_in_dataset"],
"skipped_not_a_review": plan["skipped_not_a_review"],
"runs": {},
}
for key, mapping in sorted(plan["runs"].items()):
name = run_name(args.prefix, key)
if args.dry_run:
report["runs"][name] = {"items_would_link": len(mapping)}
continue
report["runs"][name] = create_run(
name,
mapping,
description=(
"Reviews already run by the pilot, linked after the fact. "
"Scores come from the traces; expectedOutput is the reviewer's "
"own prior output, not human-verified ground truth."
),
)
report["dry_run"] = args.dry_run
print(json.dumps(report, indent=2))
return 0
_module = importlib.import_module("evaluation.experiment")
sys.modules[__name__] = _module
if __name__ == "__main__":
raise SystemExit(main())
raise SystemExit(_module.main())