#!/usr/bin/env python3 """pragent pilot — populate the Experiments tab from reviews already traced. An "experiment" in Langfuse is a dataset run: a set of (dataset item, trace) links under one run name. The Experiments tab then shows one row per item with its scores, and lets two runs be diffed side by side. Nothing here re-runs the reviewer. Every PR in `pragent-reviews` has already been reviewed, and each of those reviews left a trace carrying its findings, cost and scores. This links what exists, which is what makes the tab useful on day one instead of after the next N pushes. Runs are grouped by **model** by default, because that is the comparison the pilot actually needs to make: the same PRs reviewed by MiniMax vs whatever replaces it, with `finding_rate` and `cost_per_finding` side by side. Group by `none` for a single "all traces" run. One trace per (run, item) — the most recent. A PR re-reviewed on every push has many traces, and a dataset run is defined as one output per input; feeding it the other five would make the per-run averages meaningless. Note on the endpoint: `POST /api/public/dataset-run-items` is deprecated in favour of the SDK experiment runner / OTel ingestion, and disappears in Langfuse v4. This instance is self-hosted v3, which the deprecation notice explicitly exempts from the cutoff date, and the pilot is stdlib-only by design. Revisit when this deployment moves to v4. Usage: LANGFUSE_HOST=... LANGFUSE_PUBLIC_KEY=... LANGFUSE_SECRET_KEY=... \\ python3 eval_experiment.py --dry-run """ from __future__ import annotations import argparse import json import os import sys import urllib.parse from collections import defaultdict sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import eval_bootstrap as eb # noqa: E402 TRACE_NAME = "pr-review" # --------------------------------------------------------------------------- # Reading what already exists # --------------------------------------------------------------------------- def fetch_traces(name: str = TRACE_NAME, limit: int = 100, max_pages: int = 50) -> list[dict]: """Every review trace, newest first.""" out: list[dict] = [] for page in range(1, max_pages + 1): q = urllib.parse.urlencode({"name": name, "limit": limit, "page": page}) st, body = eb._call("GET", f"/api/public/traces?{q}") if st != 200 or not isinstance(body, dict): raise SystemExit(f"listing traces failed: {st} {body}") data = body.get("data") or [] out.extend(data) meta = body.get("meta") or {} if page * meta.get("limit", limit) >= meta.get("totalItems", 0): break return out def fetch_item_ids(dataset: str) -> set[str]: """Ids present in the dataset, so runs never reference a missing item.""" ids: set[str] = set() for page in range(1, 51): q = urllib.parse.urlencode({"datasetName": dataset, "limit": 100, "page": page}) st, body = eb._call("GET", f"/api/public/dataset-items?{q}") if st != 200 or not isinstance(body, dict): raise SystemExit(f"listing dataset items failed: {st} {body}") ids.update(i["id"] for i in body.get("data") or []) meta = body.get("meta") or {} if page * meta.get("limit", 100) >= meta.get("totalItems", 0): break return ids # --------------------------------------------------------------------------- # Grouping traces into runs # --------------------------------------------------------------------------- def trace_model(trace: dict) -> str: """The model that produced a review, from its `model:` tag.""" for tag in trace.get("tags") or []: if tag.startswith("model:"): return tag[len("model:"):] or "unknown" return "unknown" def trace_item_id(trace: dict) -> str | None: """The dataset item a trace belongs to, or None if it is not a PR review.""" md = trace.get("metadata") or {} repo, pr = md.get("repo"), md.get("pr") if not repo or pr in (None, ""): return None return eb.item_id(str(repo), pr) def _sort_key(trace: dict): return (trace.get("timestamp") or "", trace.get("id") or "") def plan_runs(traces: list[dict], known_items: set[str], group_by: str = "model") -> dict: """Map run name -> {item id: trace}, keeping only the newest trace per item. Traces whose PR is not in the dataset are dropped: `feedback.db` is the source for both, but a review can be traced without its row landing (the posting step can fail after the model ran), and a run item pointing at a non-existent dataset item is rejected. """ runs: dict[str, dict[str, dict]] = defaultdict(dict) skipped_no_item, skipped_unknown = 0, 0 for tr in traces: iid = trace_item_id(tr) if iid is None: skipped_unknown += 1 continue if iid not in known_items: skipped_no_item += 1 continue run = "all-traces" if group_by == "none" else trace_model(tr) prev = runs[run].get(iid) if prev is None or _sort_key(tr) > _sort_key(prev): runs[run][iid] = tr return { "runs": dict(runs), "skipped_not_in_dataset": skipped_no_item, "skipped_not_a_review": skipped_unknown, } def run_name(prefix: str, key: str) -> str: return f"{prefix}-{key}" if prefix else key # --------------------------------------------------------------------------- # Writing the runs # --------------------------------------------------------------------------- def create_run(name: str, items: dict[str, dict], description: str = "") -> dict: """Link each (item, trace) pair into the named run. Idempotent per pair.""" created, failed = 0, [] for iid, tr in sorted(items.items()): md = tr.get("metadata") or {} body = { "runName": name, "runDescription": description, "datasetItemId": iid, "traceId": tr["id"], "metadata": { "model": trace_model(tr), "engine": md.get("engine"), "findings": md.get("findings"), "duration_s": md.get("duration_s"), "cost_basis": md.get("cost_basis"), "linked_by": "eval_experiment.py", }, } st, resp = eb._call("POST", "/api/public/dataset-run-items", body) if st in (200, 201): created += 1 else: failed.append({"item": iid, "status": st, "error": resp}) return {"run": name, "items_linked": created, "failed": failed} def main(argv: list[str] | None = None) -> int: ap = argparse.ArgumentParser(description=__doc__) ap.add_argument("--dataset", default=eb.DATASET_NAME) ap.add_argument("--group-by", choices=("model", "none"), default="model") ap.add_argument("--prefix", default="baseline", help="run name prefix; '' for the bare group key") ap.add_argument("--dry-run", action="store_true") args = ap.parse_args(argv) traces = fetch_traces() items = fetch_item_ids(args.dataset) plan = plan_runs(traces, items, group_by=args.group_by) report = { "traces_read": len(traces), "dataset_items": len(items), "skipped_not_in_dataset": plan["skipped_not_in_dataset"], "skipped_not_a_review": plan["skipped_not_a_review"], "runs": {}, } for key, mapping in sorted(plan["runs"].items()): name = run_name(args.prefix, key) if args.dry_run: report["runs"][name] = {"items_would_link": len(mapping)} continue report["runs"][name] = create_run( name, mapping, description=( "Reviews already run by the pilot, linked after the fact. " "Scores come from the traces; expectedOutput is the reviewer's " "own prior output, not human-verified ground truth." ), ) report["dry_run"] = args.dry_run print(json.dumps(report, indent=2)) return 0 if __name__ == "__main__": raise SystemExit(main())