7a510a926d
Group review, feedback, evaluation, observability, and entrypoint code into packages. Keep thin top-level compatibility shims for existing scripts and imports, and mirror the structure in the tests.
213 lines
8.0 KiB
Python
213 lines
8.0 KiB
Python
#!/usr/bin/env python3
|
|
"""pragent pilot — populate the Experiments tab from reviews already traced.
|
|
|
|
An "experiment" in Langfuse is a dataset run: a set of (dataset item, trace)
|
|
links under one run name. The Experiments tab then shows one row per item with
|
|
its scores, and lets two runs be diffed side by side.
|
|
|
|
Nothing here re-runs the reviewer. Every PR in `pragent-reviews` has already
|
|
been reviewed, and each of those reviews left a trace carrying its findings,
|
|
cost and scores. This links what exists, which is what makes the tab useful on
|
|
day one instead of after the next N pushes.
|
|
|
|
Runs are grouped by **model** by default, because that is the comparison the
|
|
pilot actually needs to make: the same PRs reviewed by MiniMax vs whatever
|
|
replaces it, with `finding_rate` and `cost_per_finding` side by side. Group by
|
|
`none` for a single "all traces" run.
|
|
|
|
One trace per (run, item) — the most recent. A PR re-reviewed on every push has
|
|
many traces, and a dataset run is defined as one output per input; feeding it
|
|
the other five would make the per-run averages meaningless.
|
|
|
|
Note on the endpoint: `POST /api/public/dataset-run-items` is deprecated in
|
|
favour of the SDK experiment runner / OTel ingestion, and disappears in
|
|
Langfuse v4. This instance is self-hosted v3, which the deprecation notice
|
|
explicitly exempts from the cutoff date, and the pilot is stdlib-only by
|
|
design. Revisit when this deployment moves to v4.
|
|
|
|
Usage:
|
|
LANGFUSE_HOST=... LANGFUSE_PUBLIC_KEY=... LANGFUSE_SECRET_KEY=... \\
|
|
python3 eval_experiment.py --dry-run
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import sys
|
|
import urllib.parse
|
|
from collections import defaultdict
|
|
|
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
|
|
import eval_bootstrap as eb # noqa: E402
|
|
|
|
TRACE_NAME = "pr-review"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Reading what already exists
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def fetch_traces(name: str = TRACE_NAME, limit: int = 100, max_pages: int = 50) -> list[dict]:
|
|
"""Every review trace, newest first."""
|
|
out: list[dict] = []
|
|
for page in range(1, max_pages + 1):
|
|
q = urllib.parse.urlencode({"name": name, "limit": limit, "page": page})
|
|
st, body = eb._call("GET", f"/api/public/traces?{q}")
|
|
if st != 200 or not isinstance(body, dict):
|
|
raise SystemExit(f"listing traces failed: {st} {body}")
|
|
data = body.get("data") or []
|
|
out.extend(data)
|
|
meta = body.get("meta") or {}
|
|
if page * meta.get("limit", limit) >= meta.get("totalItems", 0):
|
|
break
|
|
return out
|
|
|
|
|
|
def fetch_item_ids(dataset: str) -> set[str]:
|
|
"""Ids present in the dataset, so runs never reference a missing item."""
|
|
ids: set[str] = set()
|
|
for page in range(1, 51):
|
|
q = urllib.parse.urlencode({"datasetName": dataset, "limit": 100, "page": page})
|
|
st, body = eb._call("GET", f"/api/public/dataset-items?{q}")
|
|
if st != 200 or not isinstance(body, dict):
|
|
raise SystemExit(f"listing dataset items failed: {st} {body}")
|
|
ids.update(i["id"] for i in body.get("data") or [])
|
|
meta = body.get("meta") or {}
|
|
if page * meta.get("limit", 100) >= meta.get("totalItems", 0):
|
|
break
|
|
return ids
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Grouping traces into runs
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def trace_model(trace: dict) -> str:
|
|
"""The model that produced a review, from its `model:` tag."""
|
|
for tag in trace.get("tags") or []:
|
|
if tag.startswith("model:"):
|
|
return tag[len("model:"):] or "unknown"
|
|
return "unknown"
|
|
|
|
|
|
def trace_item_id(trace: dict) -> str | None:
|
|
"""The dataset item a trace belongs to, or None if it is not a PR review."""
|
|
md = trace.get("metadata") or {}
|
|
repo, pr = md.get("repo"), md.get("pr")
|
|
if not repo or pr in (None, ""):
|
|
return None
|
|
return eb.item_id(str(repo), pr)
|
|
|
|
|
|
def _sort_key(trace: dict):
|
|
return (trace.get("timestamp") or "", trace.get("id") or "")
|
|
|
|
|
|
def plan_runs(traces: list[dict], known_items: set[str], group_by: str = "model") -> dict:
|
|
"""Map run name -> {item id: trace}, keeping only the newest trace per item.
|
|
|
|
Traces whose PR is not in the dataset are dropped: `feedback.db` is the
|
|
source for both, but a review can be traced without its row landing (the
|
|
posting step can fail after the model ran), and a run item pointing at a
|
|
non-existent dataset item is rejected.
|
|
"""
|
|
runs: dict[str, dict[str, dict]] = defaultdict(dict)
|
|
skipped_no_item, skipped_unknown = 0, 0
|
|
for tr in traces:
|
|
iid = trace_item_id(tr)
|
|
if iid is None:
|
|
skipped_unknown += 1
|
|
continue
|
|
if iid not in known_items:
|
|
skipped_no_item += 1
|
|
continue
|
|
run = "all-traces" if group_by == "none" else trace_model(tr)
|
|
prev = runs[run].get(iid)
|
|
if prev is None or _sort_key(tr) > _sort_key(prev):
|
|
runs[run][iid] = tr
|
|
return {
|
|
"runs": dict(runs),
|
|
"skipped_not_in_dataset": skipped_no_item,
|
|
"skipped_not_a_review": skipped_unknown,
|
|
}
|
|
|
|
|
|
def run_name(prefix: str, key: str) -> str:
|
|
return f"{prefix}-{key}" if prefix else key
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Writing the runs
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def create_run(name: str, items: dict[str, dict], description: str = "") -> dict:
|
|
"""Link each (item, trace) pair into the named run. Idempotent per pair."""
|
|
created, failed = 0, []
|
|
for iid, tr in sorted(items.items()):
|
|
md = tr.get("metadata") or {}
|
|
body = {
|
|
"runName": name,
|
|
"runDescription": description,
|
|
"datasetItemId": iid,
|
|
"traceId": tr["id"],
|
|
"metadata": {
|
|
"model": trace_model(tr),
|
|
"engine": md.get("engine"),
|
|
"findings": md.get("findings"),
|
|
"duration_s": md.get("duration_s"),
|
|
"cost_basis": md.get("cost_basis"),
|
|
"linked_by": "eval_experiment.py",
|
|
},
|
|
}
|
|
st, resp = eb._call("POST", "/api/public/dataset-run-items", body)
|
|
if st in (200, 201):
|
|
created += 1
|
|
else:
|
|
failed.append({"item": iid, "status": st, "error": resp})
|
|
return {"run": name, "items_linked": created, "failed": failed}
|
|
|
|
|
|
def main(argv: list[str] | None = None) -> int:
|
|
ap = argparse.ArgumentParser(description=__doc__)
|
|
ap.add_argument("--dataset", default=eb.DATASET_NAME)
|
|
ap.add_argument("--group-by", choices=("model", "none"), default="model")
|
|
ap.add_argument("--prefix", default="baseline",
|
|
help="run name prefix; '' for the bare group key")
|
|
ap.add_argument("--dry-run", action="store_true")
|
|
args = ap.parse_args(argv)
|
|
|
|
traces = fetch_traces()
|
|
items = fetch_item_ids(args.dataset)
|
|
plan = plan_runs(traces, items, group_by=args.group_by)
|
|
|
|
report = {
|
|
"traces_read": len(traces),
|
|
"dataset_items": len(items),
|
|
"skipped_not_in_dataset": plan["skipped_not_in_dataset"],
|
|
"skipped_not_a_review": plan["skipped_not_a_review"],
|
|
"runs": {},
|
|
}
|
|
for key, mapping in sorted(plan["runs"].items()):
|
|
name = run_name(args.prefix, key)
|
|
if args.dry_run:
|
|
report["runs"][name] = {"items_would_link": len(mapping)}
|
|
continue
|
|
report["runs"][name] = create_run(
|
|
name,
|
|
mapping,
|
|
description=(
|
|
"Reviews already run by the pilot, linked after the fact. "
|
|
"Scores come from the traces; expectedOutput is the reviewer's "
|
|
"own prior output, not human-verified ground truth."
|
|
),
|
|
)
|
|
report["dry_run"] = args.dry_run
|
|
print(json.dumps(report, indent=2))
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|