refactor: organize pilot packages
Group review, feedback, evaluation, observability, and entrypoint code into packages. Keep thin top-level compatibility shims for existing scripts and imports, and mirror the structure in the tests.
This commit is contained in:
@@ -0,0 +1 @@
|
||||
"""Langfuse evaluation bootstrap, experiments, and scoring."""
|
||||
@@ -0,0 +1,350 @@
|
||||
#!/usr/bin/env python3
|
||||
"""pragent pilot — one-time Langfuse project setup for evaluation.
|
||||
|
||||
Three jobs, each idempotent so it can be re-run after any change:
|
||||
|
||||
1. **Score configs.** Registers the schema for every score pragent emits
|
||||
(`eval_scores.SCORE_CONFIGS` + `feedback_scores.SCORE_CONFIGS`). Without
|
||||
these the scores still ingest, but nothing stops a later scorer writing
|
||||
`severity_max="HIGH"` beside today's `"high"` and quietly splitting one
|
||||
series into two. Configs are immutable in Langfuse — a name that already
|
||||
exists is left alone rather than updated.
|
||||
|
||||
2. **Dataset.** Seeds `pragent-reviews` from `feedback.db`: one item per PR
|
||||
the reviewer has actually run on, carrying the repo/PR/sha as input and
|
||||
the findings it posted as `expectedOutput`.
|
||||
|
||||
Read `expectedOutput` here as "what the reviewer said last time", not "what
|
||||
is correct" — no human has labelled any of it. It is a regression baseline:
|
||||
re-run a candidate model over these PRs and the diff against this column is
|
||||
the behaviour change. Promoting an item to real ground truth means a human
|
||||
editing it after reviewing the PR, which is what the dataset view is for.
|
||||
|
||||
3. **Trace backfill** (`--backfill-traces`). Scores only ride along with new
|
||||
reviews, so without this the charts stay empty until the next PR lands.
|
||||
Every trace `langfuse_trace` has ever written already carries the finding
|
||||
count, the severity histogram and the cost in its metadata, which is
|
||||
everything four of the five scorers need. `dropped_findings` is absent from
|
||||
historical traces and is left unscored rather than backfilled as zero.
|
||||
|
||||
4. **Reports** what it found, so the gap between "reviews recorded" and
|
||||
"reviews with human feedback" is visible rather than assumed.
|
||||
|
||||
Usage:
|
||||
LANGFUSE_HOST=... LANGFUSE_PUBLIC_KEY=... LANGFUSE_SECRET_KEY=... \\
|
||||
python3 eval_bootstrap.py --db /data/feedback.db
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import sqlite3
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from datetime import datetime, timezone
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
import eval_scores # noqa: E402
|
||||
import feedback_scores # noqa: E402
|
||||
|
||||
DATASET_NAME = "pragent-reviews"
|
||||
|
||||
|
||||
def _conf() -> tuple[str, str, str]:
|
||||
host = (os.environ.get("LANGFUSE_HOST") or "").strip().rstrip("/")
|
||||
pk = (os.environ.get("LANGFUSE_PUBLIC_KEY") or "").strip()
|
||||
sk = (os.environ.get("LANGFUSE_SECRET_KEY") or "").strip()
|
||||
if not host or not pk or not sk:
|
||||
raise SystemExit("LANGFUSE_HOST / LANGFUSE_PUBLIC_KEY / LANGFUSE_SECRET_KEY must be set")
|
||||
return host, pk, sk
|
||||
|
||||
|
||||
def _call(method: str, path: str, body: dict | None = None, timeout: float = 20.0):
|
||||
host, pk, sk = _conf()
|
||||
auth = base64.b64encode(f"{pk}:{sk}".encode()).decode("ascii")
|
||||
data = json.dumps(body).encode() if body is not None else None
|
||||
req = urllib.request.Request(
|
||||
host + path,
|
||||
data=data,
|
||||
headers={
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Basic {auth}",
|
||||
"User-Agent": "pragent-pilot/1.0",
|
||||
},
|
||||
method=method,
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
raw = resp.read()
|
||||
return resp.status, (json.loads(raw) if raw else None)
|
||||
except urllib.error.HTTPError as e:
|
||||
return e.code, e.read()[:400].decode("utf-8", "replace")
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. Score configs
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def ensure_score_configs() -> dict:
|
||||
status, existing = _call("GET", "/api/public/score-configs?limit=100")
|
||||
have = set()
|
||||
if status == 200 and isinstance(existing, dict):
|
||||
have = {c.get("name") for c in existing.get("data", [])}
|
||||
|
||||
created, skipped, failed = [], [], []
|
||||
for cfg in list(eval_scores.SCORE_CONFIGS) + list(feedback_scores.SCORE_CONFIGS):
|
||||
if cfg["name"] in have:
|
||||
skipped.append(cfg["name"])
|
||||
continue
|
||||
st, resp = _call("POST", "/api/public/score-configs", cfg)
|
||||
if st in (200, 201):
|
||||
created.append(cfg["name"])
|
||||
else:
|
||||
failed.append({"name": cfg["name"], "status": st, "error": resp})
|
||||
return {"created": created, "already_present": skipped, "failed": failed}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Dataset from recorded reviews
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def item_id(repo: str, pr) -> str:
|
||||
"""A dataset-item id that survives being put in a URL path.
|
||||
|
||||
The obvious `{repo}#{pr}` is unusable: the UI routes items as
|
||||
`/datasets/{id}/items/{item_id}`, so the `/` in `owner/repo` splits into
|
||||
extra path segments and everything after the `#` is a fragment the browser
|
||||
never sends. The item is created fine and then 404s when opened.
|
||||
|
||||
Session ids elsewhere keep the `{repo}#{pr}` form — those are never path
|
||||
segments, and `feedback_scores` depends on that shape.
|
||||
"""
|
||||
return f"{repo.replace('/', '__')}__pr{pr}"
|
||||
|
||||
|
||||
def _item_metadata(*, repo, pr, head_sha, reviews_run, last_seen, findings) -> dict:
|
||||
"""Filterable facets for one dataset item.
|
||||
|
||||
Kept flat and primitive: the filter bar matches a metadata key against a
|
||||
literal, so a nested object or a list is not reachable from the UI.
|
||||
"""
|
||||
owner, _, repo_name = str(repo).partition("/")
|
||||
sevs = [str(f["severity"] or "").lower() for f in findings]
|
||||
ranked = [s for s in sevs if s in eval_scores.SEVERITY_RANK]
|
||||
return {
|
||||
"repo": repo,
|
||||
"owner": owner or repo,
|
||||
"repo_name": repo_name or repo,
|
||||
"pr": int(pr),
|
||||
"head_sha": head_sha,
|
||||
"reviews_run": reviews_run,
|
||||
"last_reviewed_at": last_seen,
|
||||
"last_reviewed_iso": datetime.fromtimestamp(last_seen, timezone.utc).isoformat(),
|
||||
"finding_count": len(findings),
|
||||
"has_findings": bool(findings),
|
||||
# "none" rather than omitting the key: a filter for silent reviews needs
|
||||
# something to match, and an absent key matches nothing.
|
||||
"max_severity": (
|
||||
max(ranked, key=lambda s: eval_scores.SEVERITY_RANK[s]) if ranked else "none"
|
||||
),
|
||||
# Flags that this row is the reviewer's own past output, not a human
|
||||
# judgement. Filter on it before anyone treats the dataset as truth.
|
||||
"labelled_by_human": False,
|
||||
}
|
||||
|
||||
|
||||
def read_review_items(db_path: str) -> list[dict]:
|
||||
"""One dataset item per (repo, pr) the reviewer has run on.
|
||||
|
||||
Keyed on the PR rather than on each individual review row: the same PR is
|
||||
re-reviewed on every push, and 113 rows over 26 PRs would make a benchmark
|
||||
that is 4x redundant and weighted towards whichever PR churned most.
|
||||
"""
|
||||
conn = sqlite3.connect(db_path)
|
||||
conn.row_factory = sqlite3.Row
|
||||
try:
|
||||
prs = conn.execute(
|
||||
"""
|
||||
SELECT repo, pr, MAX(posted_at) AS last_seen, COUNT(*) AS reviews,
|
||||
MAX(head_sha) AS head_sha
|
||||
FROM review GROUP BY repo, pr ORDER BY repo, pr
|
||||
"""
|
||||
).fetchall()
|
||||
items = []
|
||||
for row in prs:
|
||||
findings = conn.execute(
|
||||
"""
|
||||
SELECT path, line, severity, problem, fix
|
||||
FROM inline_finding WHERE repo = ? AND pr = ?
|
||||
ORDER BY path, line
|
||||
""",
|
||||
(row["repo"], row["pr"]),
|
||||
).fetchall()
|
||||
items.append(
|
||||
{
|
||||
"id": item_id(row["repo"], row["pr"]),
|
||||
"input": {
|
||||
"repo": row["repo"],
|
||||
"pr": int(row["pr"]),
|
||||
"head_sha": row["head_sha"],
|
||||
},
|
||||
"expectedOutput": {
|
||||
"findings": [dict(f) for f in findings],
|
||||
"finding_count": len(findings),
|
||||
},
|
||||
# The UI's filter bar reads metadata and nothing else, so
|
||||
# anything worth slicing on is a top-level key here even
|
||||
# where it duplicates `input`. `owner` and `repo_name` are
|
||||
# split out because a filter on the joined `repo` can only
|
||||
# match one repo at a time, never a whole org.
|
||||
"metadata": _item_metadata(
|
||||
repo=row["repo"],
|
||||
pr=row["pr"],
|
||||
head_sha=row["head_sha"],
|
||||
reviews_run=int(row["reviews"]),
|
||||
last_seen=int(row["last_seen"]),
|
||||
findings=findings,
|
||||
),
|
||||
}
|
||||
)
|
||||
return items
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def ensure_dataset(items: list[dict], name: str = DATASET_NAME) -> dict:
|
||||
st, _ = _call(
|
||||
"POST",
|
||||
"/api/public/datasets",
|
||||
{
|
||||
"name": name,
|
||||
"description": (
|
||||
"PRs the pragent pilot has reviewed, seeded from feedback.db. "
|
||||
"expectedOutput is the reviewer's own prior output — a regression "
|
||||
"baseline, not human-verified ground truth."
|
||||
),
|
||||
"metadata": {"source": "feedback.db", "seeded_by": "eval_bootstrap.py"},
|
||||
},
|
||||
)
|
||||
# A duplicate name is fine: the dataset already exists from an earlier run.
|
||||
dataset_ok = st in (200, 201, 409)
|
||||
|
||||
created, failed = 0, []
|
||||
for item in items:
|
||||
body = {
|
||||
"datasetName": name,
|
||||
"id": item["id"], # idempotent: same PR updates rather than duplicates
|
||||
"input": item["input"],
|
||||
"expectedOutput": item["expectedOutput"],
|
||||
"metadata": item["metadata"],
|
||||
}
|
||||
ist, resp = _call("POST", "/api/public/dataset-items", body)
|
||||
if ist in (200, 201):
|
||||
created += 1
|
||||
else:
|
||||
failed.append({"item": item["id"], "status": ist, "error": resp})
|
||||
return {"dataset": name, "dataset_created": dataset_ok, "items_upserted": created, "failed": failed}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Backfill scores onto traces that predate the scorers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _synth_findings(severities: dict) -> list[dict]:
|
||||
"""Rebuild a findings list from a trace's severity histogram.
|
||||
|
||||
Only severity matters to the scorers, and that is all the histogram kept.
|
||||
Reconstructing placeholders is honest here because every scorer being
|
||||
backfilled reads nothing else off a finding.
|
||||
"""
|
||||
out = []
|
||||
for sev, count in (severities or {}).items():
|
||||
out.extend({"severity": sev} for _ in range(int(count)))
|
||||
return out
|
||||
|
||||
|
||||
def backfill_traces(limit_pages: int = 20) -> dict:
|
||||
import eval_scores as es
|
||||
|
||||
scored, skipped, events = 0, 0, []
|
||||
page = 1
|
||||
while page <= limit_pages:
|
||||
st, resp = _call("GET", f"/api/public/traces?limit=50&page={page}&name=pr-review")
|
||||
if st != 200 or not isinstance(resp, dict):
|
||||
break
|
||||
rows = resp.get("data") or []
|
||||
if not rows:
|
||||
break
|
||||
for tr in rows:
|
||||
meta = tr.get("metadata") or {}
|
||||
severities = meta.get("severities") or {}
|
||||
count = meta.get("findings")
|
||||
if count is None:
|
||||
skipped += 1
|
||||
continue
|
||||
findings = _synth_findings(severities)
|
||||
# The histogram is authoritative when present; a trace that recorded
|
||||
# a count but no histogram still scores its rate.
|
||||
if not findings and count:
|
||||
findings = [{"severity": "medium"} for _ in range(int(count))]
|
||||
batch = es.build_scores(
|
||||
trace_id=tr["id"],
|
||||
findings=findings,
|
||||
environment=tr.get("environment") or "default",
|
||||
cost_usd=(tr.get("totalCost") or meta.get("provider_cost_usd")),
|
||||
timestamp=tr.get("timestamp"),
|
||||
comment="backfilled from trace metadata",
|
||||
)
|
||||
events.extend(batch)
|
||||
scored += 1
|
||||
page += 1
|
||||
|
||||
posted = False
|
||||
status = None
|
||||
if events:
|
||||
import langfuse_trace
|
||||
|
||||
host, pk, sk = _conf()
|
||||
# Chunked: one 2000-event POST is refused, and a partial backfill that
|
||||
# reports success is worse than a slow one.
|
||||
for i in range(0, len(events), 200):
|
||||
status = langfuse_trace._post(host, pk, sk, events[i:i + 200], 30.0)
|
||||
posted = status in (200, 201, 207)
|
||||
if not posted:
|
||||
break
|
||||
return {"traces_scored": scored, "traces_skipped": skipped, "scores": len(events),
|
||||
"posted": posted, "http_status": status}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser(description="Bootstrap Langfuse evaluation for the pragent pilot")
|
||||
ap.add_argument("--db", default=os.environ.get("PRAGENT_FEEDBACK_DB", "/data/feedback.db"))
|
||||
ap.add_argument("--skip-dataset", action="store_true")
|
||||
ap.add_argument("--skip-configs", action="store_true")
|
||||
ap.add_argument("--backfill-traces", action="store_true",
|
||||
help="score traces written before the scorers existed")
|
||||
args = ap.parse_args()
|
||||
|
||||
out: dict = {}
|
||||
if not args.skip_configs:
|
||||
out["score_configs"] = ensure_score_configs()
|
||||
if not args.skip_dataset:
|
||||
items = read_review_items(args.db)
|
||||
out["dataset"] = ensure_dataset(items)
|
||||
out["dataset"]["items_read"] = len(items)
|
||||
if args.backfill_traces:
|
||||
out["trace_backfill"] = backfill_traces()
|
||||
print(json.dumps(out, indent=2))
|
||||
|
||||
failed = (out.get("score_configs", {}).get("failed") or []) + (
|
||||
out.get("dataset", {}).get("failed") or []
|
||||
)
|
||||
return 1 if failed else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,212 @@
|
||||
#!/usr/bin/env python3
|
||||
"""pragent pilot — populate the Experiments tab from reviews already traced.
|
||||
|
||||
An "experiment" in Langfuse is a dataset run: a set of (dataset item, trace)
|
||||
links under one run name. The Experiments tab then shows one row per item with
|
||||
its scores, and lets two runs be diffed side by side.
|
||||
|
||||
Nothing here re-runs the reviewer. Every PR in `pragent-reviews` has already
|
||||
been reviewed, and each of those reviews left a trace carrying its findings,
|
||||
cost and scores. This links what exists, which is what makes the tab useful on
|
||||
day one instead of after the next N pushes.
|
||||
|
||||
Runs are grouped by **model** by default, because that is the comparison the
|
||||
pilot actually needs to make: the same PRs reviewed by MiniMax vs whatever
|
||||
replaces it, with `finding_rate` and `cost_per_finding` side by side. Group by
|
||||
`none` for a single "all traces" run.
|
||||
|
||||
One trace per (run, item) — the most recent. A PR re-reviewed on every push has
|
||||
many traces, and a dataset run is defined as one output per input; feeding it
|
||||
the other five would make the per-run averages meaningless.
|
||||
|
||||
Note on the endpoint: `POST /api/public/dataset-run-items` is deprecated in
|
||||
favour of the SDK experiment runner / OTel ingestion, and disappears in
|
||||
Langfuse v4. This instance is self-hosted v3, which the deprecation notice
|
||||
explicitly exempts from the cutoff date, and the pilot is stdlib-only by
|
||||
design. Revisit when this deployment moves to v4.
|
||||
|
||||
Usage:
|
||||
LANGFUSE_HOST=... LANGFUSE_PUBLIC_KEY=... LANGFUSE_SECRET_KEY=... \\
|
||||
python3 eval_experiment.py --dry-run
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import urllib.parse
|
||||
from collections import defaultdict
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
import eval_bootstrap as eb # noqa: E402
|
||||
|
||||
TRACE_NAME = "pr-review"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Reading what already exists
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def fetch_traces(name: str = TRACE_NAME, limit: int = 100, max_pages: int = 50) -> list[dict]:
|
||||
"""Every review trace, newest first."""
|
||||
out: list[dict] = []
|
||||
for page in range(1, max_pages + 1):
|
||||
q = urllib.parse.urlencode({"name": name, "limit": limit, "page": page})
|
||||
st, body = eb._call("GET", f"/api/public/traces?{q}")
|
||||
if st != 200 or not isinstance(body, dict):
|
||||
raise SystemExit(f"listing traces failed: {st} {body}")
|
||||
data = body.get("data") or []
|
||||
out.extend(data)
|
||||
meta = body.get("meta") or {}
|
||||
if page * meta.get("limit", limit) >= meta.get("totalItems", 0):
|
||||
break
|
||||
return out
|
||||
|
||||
|
||||
def fetch_item_ids(dataset: str) -> set[str]:
|
||||
"""Ids present in the dataset, so runs never reference a missing item."""
|
||||
ids: set[str] = set()
|
||||
for page in range(1, 51):
|
||||
q = urllib.parse.urlencode({"datasetName": dataset, "limit": 100, "page": page})
|
||||
st, body = eb._call("GET", f"/api/public/dataset-items?{q}")
|
||||
if st != 200 or not isinstance(body, dict):
|
||||
raise SystemExit(f"listing dataset items failed: {st} {body}")
|
||||
ids.update(i["id"] for i in body.get("data") or [])
|
||||
meta = body.get("meta") or {}
|
||||
if page * meta.get("limit", 100) >= meta.get("totalItems", 0):
|
||||
break
|
||||
return ids
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Grouping traces into runs
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def trace_model(trace: dict) -> str:
|
||||
"""The model that produced a review, from its `model:` tag."""
|
||||
for tag in trace.get("tags") or []:
|
||||
if tag.startswith("model:"):
|
||||
return tag[len("model:"):] or "unknown"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def trace_item_id(trace: dict) -> str | None:
|
||||
"""The dataset item a trace belongs to, or None if it is not a PR review."""
|
||||
md = trace.get("metadata") or {}
|
||||
repo, pr = md.get("repo"), md.get("pr")
|
||||
if not repo or pr in (None, ""):
|
||||
return None
|
||||
return eb.item_id(str(repo), pr)
|
||||
|
||||
|
||||
def _sort_key(trace: dict):
|
||||
return (trace.get("timestamp") or "", trace.get("id") or "")
|
||||
|
||||
|
||||
def plan_runs(traces: list[dict], known_items: set[str], group_by: str = "model") -> dict:
|
||||
"""Map run name -> {item id: trace}, keeping only the newest trace per item.
|
||||
|
||||
Traces whose PR is not in the dataset are dropped: `feedback.db` is the
|
||||
source for both, but a review can be traced without its row landing (the
|
||||
posting step can fail after the model ran), and a run item pointing at a
|
||||
non-existent dataset item is rejected.
|
||||
"""
|
||||
runs: dict[str, dict[str, dict]] = defaultdict(dict)
|
||||
skipped_no_item, skipped_unknown = 0, 0
|
||||
for tr in traces:
|
||||
iid = trace_item_id(tr)
|
||||
if iid is None:
|
||||
skipped_unknown += 1
|
||||
continue
|
||||
if iid not in known_items:
|
||||
skipped_no_item += 1
|
||||
continue
|
||||
run = "all-traces" if group_by == "none" else trace_model(tr)
|
||||
prev = runs[run].get(iid)
|
||||
if prev is None or _sort_key(tr) > _sort_key(prev):
|
||||
runs[run][iid] = tr
|
||||
return {
|
||||
"runs": dict(runs),
|
||||
"skipped_not_in_dataset": skipped_no_item,
|
||||
"skipped_not_a_review": skipped_unknown,
|
||||
}
|
||||
|
||||
|
||||
def run_name(prefix: str, key: str) -> str:
|
||||
return f"{prefix}-{key}" if prefix else key
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Writing the runs
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def create_run(name: str, items: dict[str, dict], description: str = "") -> dict:
|
||||
"""Link each (item, trace) pair into the named run. Idempotent per pair."""
|
||||
created, failed = 0, []
|
||||
for iid, tr in sorted(items.items()):
|
||||
md = tr.get("metadata") or {}
|
||||
body = {
|
||||
"runName": name,
|
||||
"runDescription": description,
|
||||
"datasetItemId": iid,
|
||||
"traceId": tr["id"],
|
||||
"metadata": {
|
||||
"model": trace_model(tr),
|
||||
"engine": md.get("engine"),
|
||||
"findings": md.get("findings"),
|
||||
"duration_s": md.get("duration_s"),
|
||||
"cost_basis": md.get("cost_basis"),
|
||||
"linked_by": "eval_experiment.py",
|
||||
},
|
||||
}
|
||||
st, resp = eb._call("POST", "/api/public/dataset-run-items", body)
|
||||
if st in (200, 201):
|
||||
created += 1
|
||||
else:
|
||||
failed.append({"item": iid, "status": st, "error": resp})
|
||||
return {"run": name, "items_linked": created, "failed": failed}
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
ap = argparse.ArgumentParser(description=__doc__)
|
||||
ap.add_argument("--dataset", default=eb.DATASET_NAME)
|
||||
ap.add_argument("--group-by", choices=("model", "none"), default="model")
|
||||
ap.add_argument("--prefix", default="baseline",
|
||||
help="run name prefix; '' for the bare group key")
|
||||
ap.add_argument("--dry-run", action="store_true")
|
||||
args = ap.parse_args(argv)
|
||||
|
||||
traces = fetch_traces()
|
||||
items = fetch_item_ids(args.dataset)
|
||||
plan = plan_runs(traces, items, group_by=args.group_by)
|
||||
|
||||
report = {
|
||||
"traces_read": len(traces),
|
||||
"dataset_items": len(items),
|
||||
"skipped_not_in_dataset": plan["skipped_not_in_dataset"],
|
||||
"skipped_not_a_review": plan["skipped_not_a_review"],
|
||||
"runs": {},
|
||||
}
|
||||
for key, mapping in sorted(plan["runs"].items()):
|
||||
name = run_name(args.prefix, key)
|
||||
if args.dry_run:
|
||||
report["runs"][name] = {"items_would_link": len(mapping)}
|
||||
continue
|
||||
report["runs"][name] = create_run(
|
||||
name,
|
||||
mapping,
|
||||
description=(
|
||||
"Reviews already run by the pilot, linked after the fact. "
|
||||
"Scores come from the traces; expectedOutput is the reviewer's "
|
||||
"own prior output, not human-verified ground truth."
|
||||
),
|
||||
)
|
||||
report["dry_run"] = args.dry_run
|
||||
print(json.dumps(report, indent=2))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,314 @@
|
||||
#!/usr/bin/env python3
|
||||
"""pragent pilot — LLM-as-a-judge evaluators for the reviewer.
|
||||
|
||||
The deterministic scorers in `eval_scores.py` measure *behaviour*: how many
|
||||
findings, how severe, how much they cost. None of them can say whether a
|
||||
finding was any good. With no human labels in `feedback.db`, a judge is the
|
||||
only thing that can — so these two ask the questions that need no ground truth,
|
||||
only the review itself:
|
||||
|
||||
`finding_actionability` — is each finding concrete enough to act on? A
|
||||
reviewer that says "consider improving error handling" at file level is
|
||||
indistinguishable from a useful one by finding count alone. This is the
|
||||
failure mode a cheap model degrades into first.
|
||||
|
||||
`review_self_consistency` — does the summary agree with the findings it
|
||||
posted? Claiming "no issues found" above a list of two criticals, or
|
||||
describing a problem in prose that never became a finding, is a defect the
|
||||
reviewer can commit entirely on its own.
|
||||
|
||||
Neither judge is asked whether a finding is *correct*. That needs the diff,
|
||||
which these traces do not carry, and a judge asked to rule on correctness from
|
||||
a summary alone will confabulate. Accuracy stays an open question until humans
|
||||
start labelling — which is what `feedback_scores.py` is there to capture.
|
||||
|
||||
**The judge is a different model from the reviewer.** The reviewer runs
|
||||
MiniMax-M2.7; the judge runs kimi-k2.7-code through the same headroom hub. A
|
||||
model grading its own output agrees with itself for reasons that have nothing
|
||||
to do with quality.
|
||||
|
||||
Evaluators score *observations*, and their variable mapping reads the
|
||||
observation's own input/output — which is why `langfuse_trace` now writes the
|
||||
review onto the generation and not just onto the trace.
|
||||
|
||||
Usage:
|
||||
LANGFUSE_HOST=... LANGFUSE_PUBLIC_KEY=... LANGFUSE_SECRET_KEY=... \\
|
||||
python3 eval_judges.py --dry-run
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
import eval_bootstrap as eb # noqa: E402
|
||||
|
||||
# The headroom hub in front of the local Ollama, plus a small pass-through
|
||||
# proxy (`judge-proxy` on 8802) that patches every `thinking` content block
|
||||
# to carry the `signature` field Langfuse's Anthropic adapter requires. The
|
||||
# underlying model is kimi-k2.7-code through the hub on 8790; the proxy fixes
|
||||
# the shape so Mastra's Zod parse stops failing.
|
||||
JUDGE_PROVIDER = "headroom-ollama"
|
||||
JUDGE_BASE_URL = os.environ.get("PRAGENT_JUDGE_BASE_URL", "http://100.74.17.70:8802")
|
||||
JUDGE_API_KEY = os.environ.get("PRAGENT_JUDGE_API_KEY", "ollama")
|
||||
JUDGE_MODEL = os.environ.get("PRAGENT_JUDGE_MODEL", "kimi-k2.7-code:cloud")
|
||||
|
||||
# The trace names this project emits (`pr-review` on the trace, `opencode-review`
|
||||
# on the generation). Filter on `traceName` rather than observation `name` — the
|
||||
# observation-rule schema only exposes `traceName` as a stringOptions column, and
|
||||
# every observation inside these traces is the review itself, so the narrowness
|
||||
# is the same.
|
||||
REVIEW_TRACE_NAMES = ["pr-review", "opencode-review"]
|
||||
|
||||
|
||||
def _model_config() -> dict:
|
||||
return {"provider": JUDGE_PROVIDER, "model": JUDGE_MODEL}
|
||||
|
||||
|
||||
JUDGES = [
|
||||
{
|
||||
"name": "finding_actionability",
|
||||
"prompt": (
|
||||
"You are auditing the output of an automated code reviewer.\n\n"
|
||||
"PR under review:\n{{input}}\n\n"
|
||||
"What the reviewer produced:\n{{output}}\n\n"
|
||||
"Rate how ACTIONABLE the findings are, from 0 to 1. A finding is "
|
||||
"actionable when a developer could act on it without asking a "
|
||||
"follow-up question: it points at a specific location, names a "
|
||||
"concrete problem, and proposes a fix that could be applied.\n\n"
|
||||
"Score 1.0 when every finding is specific and fixable. Score around "
|
||||
"0.5 when findings identify a real area but leave the developer to "
|
||||
"work out what to change. Score near 0.0 when findings are generic "
|
||||
"advice that would apply to almost any pull request.\n\n"
|
||||
"Judge only specificity and actionability. You cannot see the diff, "
|
||||
"so do NOT attempt to judge whether a finding is factually correct, "
|
||||
"and do not penalise a finding for being one you cannot verify.\n\n"
|
||||
"If the reviewer reported no findings at all, return 1.0 and say in "
|
||||
"your reasoning that there was nothing to judge — a silent review is "
|
||||
"measured by finding_rate, not here."
|
||||
),
|
||||
"outputDefinition": {
|
||||
"dataType": "NUMERIC",
|
||||
"minValue": 0,
|
||||
"maxValue": 1,
|
||||
"reasoning": {
|
||||
"description": (
|
||||
"Name the least actionable finding and say what it would "
|
||||
"need in order to be acted on."
|
||||
)
|
||||
},
|
||||
"score": {"description": "0 = generic advice, 1 = every finding is specific and fixable."},
|
||||
},
|
||||
},
|
||||
{
|
||||
"name": "review_self_consistency",
|
||||
"prompt": (
|
||||
"You are auditing the output of an automated code reviewer.\n\n"
|
||||
"PR under review:\n{{input}}\n\n"
|
||||
"What the reviewer produced:\n{{output}}\n\n"
|
||||
"The output contains a prose `summary` and a list of `findings`. "
|
||||
"Decide whether the summary is CONSISTENT with the findings.\n\n"
|
||||
"Inconsistent means, for example: the summary says no issues were "
|
||||
"found while findings are listed; the summary describes a problem "
|
||||
"that never became a finding; the summary characterises the severity "
|
||||
"of the findings in a way the findings themselves contradict; or the "
|
||||
"summary refers to files that appear in no finding and in no part of "
|
||||
"the PR description.\n\n"
|
||||
"A summary that adds context beyond the findings is NOT inconsistent "
|
||||
"as long as nothing in it contradicts them. A review that found "
|
||||
"nothing and says so is consistent.\n\n"
|
||||
"You cannot see the diff. Judge the summary against the findings and "
|
||||
"the PR title only — never against what you imagine the code does."
|
||||
),
|
||||
"outputDefinition": {
|
||||
"dataType": "BOOLEAN",
|
||||
"reasoning": {
|
||||
"description": "Quote the part of the summary that conflicts with the findings, if any."
|
||||
},
|
||||
"score": {"description": "true = summary agrees with the findings, false = it contradicts them."},
|
||||
},
|
||||
},
|
||||
]
|
||||
|
||||
# Both judges read the observation's own input/output.
|
||||
MAPPING = [
|
||||
{"variable": "input", "source": "input"},
|
||||
{"variable": "output", "source": "output"},
|
||||
]
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# LLM connection
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def ensure_llm_connection() -> dict:
|
||||
"""Point the project at the judge model. Upserted on `provider`."""
|
||||
body = {
|
||||
"provider": JUDGE_PROVIDER,
|
||||
"adapter": "anthropic",
|
||||
"baseURL": JUDGE_BASE_URL,
|
||||
"secretKey": JUDGE_API_KEY,
|
||||
"customModels": [JUDGE_MODEL],
|
||||
# The hub serves two local models and none of Anthropic's, so the
|
||||
# default catalogue would be a list of models that all fail on use.
|
||||
"withDefaultModels": False,
|
||||
}
|
||||
st, resp = eb._call("PUT", "/api/public/llm-connections", body)
|
||||
return {"status": st, "ok": st in (200, 201), "provider": JUDGE_PROVIDER,
|
||||
"error": None if st in (200, 201) else resp}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Evaluators
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def existing_evaluators() -> dict[str, str]:
|
||||
"""name -> id for evaluators already in the project."""
|
||||
out: dict[str, str] = {}
|
||||
st, body = eb._call("GET", "/api/public/unstable/evaluators?limit=100")
|
||||
if st == 200 and isinstance(body, dict):
|
||||
for ev in body.get("data") or []:
|
||||
out[ev.get("name")] = ev.get("id")
|
||||
return out
|
||||
|
||||
|
||||
def ensure_evaluators() -> dict:
|
||||
"""Create each judge if no version exists for the name yet.
|
||||
|
||||
POST /evaluators with a name that already exists creates a new version, not
|
||||
a no-op — re-running this script would pile up versions until the page
|
||||
listing them is unreadable. Skip when an evaluator of that name is present.
|
||||
"""
|
||||
created, skipped, failed = {}, [], []
|
||||
existing = set(existing_evaluators())
|
||||
for judge in JUDGES:
|
||||
if judge["name"] in existing:
|
||||
skipped.append(judge["name"])
|
||||
continue
|
||||
body = {
|
||||
"type": "llm_as_judge",
|
||||
"name": judge["name"],
|
||||
"prompt": judge["prompt"],
|
||||
"outputDefinition": judge["outputDefinition"],
|
||||
"modelConfig": _model_config(),
|
||||
}
|
||||
st, resp = eb._call("POST", "/api/public/unstable/evaluators", body, timeout=60.0)
|
||||
if st in (200, 201) and isinstance(resp, dict):
|
||||
created[judge["name"]] = resp.get("id")
|
||||
else:
|
||||
failed.append({"name": judge["name"], "status": st, "error": resp})
|
||||
return {"created": created, "skipped": skipped, "failed": failed}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Rules — what gets judged, and how often
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def rule_body(name: str, judge_name: str, sampling: float) -> dict:
|
||||
"""POST /evaluation-rules shape for an LLM-as-judge trace rule.
|
||||
|
||||
Target is `trace` rather than `observation` on purpose: the standard
|
||||
`/api/public/ingestion` path that ships review traces here feeds only
|
||||
the trace-upsert queue, and `evalService.createEvalJobs` only creates
|
||||
jobs for `targetObject ∈ {TRACE, DATASET}`. Observation rules are
|
||||
triggered exclusively from the OTel ingestion pipeline, which this
|
||||
pilot does not use. A trace rule reads the trace's own input/output —
|
||||
`langfuse_trace` already writes `_review_input`/`_review_output` onto
|
||||
the trace body for exactly this reason.
|
||||
|
||||
Mapping is required at both the rule root (server validates it there)
|
||||
and inside `evaluator` (the API echoes it back).
|
||||
"""
|
||||
return {
|
||||
"name": name,
|
||||
"enabled": True,
|
||||
"target": "trace",
|
||||
"sampling": sampling,
|
||||
"filter": [
|
||||
{"column": "traceName", "operator": "any of",
|
||||
"value": REVIEW_TRACE_NAMES, "type": "stringOptions"},
|
||||
],
|
||||
"evaluator": {
|
||||
"name": judge_name,
|
||||
"scope": "project",
|
||||
"variableMapping": MAPPING,
|
||||
},
|
||||
"mapping": MAPPING,
|
||||
}
|
||||
|
||||
|
||||
def ensure_rules(evaluator_ids: dict[str, str], sampling: float) -> dict:
|
||||
"""Idempotent: existing rules with the same name are skipped, not duplicated.
|
||||
|
||||
The API has no `name`-keyed upsert; the convention is to POST once and
|
||||
re-run the script to verify the response. A duplicate POST raises 409.
|
||||
"""
|
||||
created, failed, skipped = [], [], []
|
||||
existing = existing_rule_names()
|
||||
for name, eid in evaluator_ids.items():
|
||||
if not eid:
|
||||
continue
|
||||
rule_name = f"{name}-on-reviews"
|
||||
if rule_name in existing:
|
||||
skipped.append(name)
|
||||
continue
|
||||
st, resp = eb._call(
|
||||
"POST", "/api/public/unstable/evaluation-rules",
|
||||
rule_body(rule_name, name, sampling), timeout=60.0,
|
||||
)
|
||||
if st in (200, 201):
|
||||
created.append(name)
|
||||
else:
|
||||
failed.append({"rule": name, "status": st, "error": resp})
|
||||
return {"created": created, "failed": failed, "skipped": skipped}
|
||||
|
||||
|
||||
def existing_rule_names() -> set[str]:
|
||||
"""Names of observation-target rules already in the project."""
|
||||
out: set[str] = set()
|
||||
st, body = eb._call("GET", "/api/public/unstable/evaluation-rules?limit=100")
|
||||
if st == 200 and isinstance(body, dict):
|
||||
for r in body.get("data") or []:
|
||||
if r.get("target") == "observation":
|
||||
out.add(r.get("name"))
|
||||
return out
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
ap = argparse.ArgumentParser(description=__doc__)
|
||||
ap.add_argument("--sampling", type=float, default=1.0,
|
||||
help="fraction of matching observations to judge (default: all)")
|
||||
ap.add_argument("--skip-connection", action="store_true")
|
||||
ap.add_argument("--dry-run", action="store_true")
|
||||
args = ap.parse_args(argv)
|
||||
|
||||
if args.dry_run:
|
||||
print(json.dumps({
|
||||
"would_connect": {"provider": JUDGE_PROVIDER, "baseURL": JUDGE_BASE_URL,
|
||||
"model": JUDGE_MODEL},
|
||||
"would_create": [j["name"] for j in JUDGES],
|
||||
"existing_evaluators": sorted(existing_evaluators()),
|
||||
"sampling": args.sampling,
|
||||
}, indent=2))
|
||||
return 0
|
||||
|
||||
report = {}
|
||||
if not args.skip_connection:
|
||||
report["llm_connection"] = ensure_llm_connection()
|
||||
report["evaluators"] = ensure_evaluators()
|
||||
ids = dict(report["evaluators"]["created"])
|
||||
# Fall back to whatever is already registered, so a re-run still wires rules.
|
||||
for name, eid in existing_evaluators().items():
|
||||
ids.setdefault(name, eid)
|
||||
report["rules"] = ensure_rules(
|
||||
{j["name"]: ids.get(j["name"]) for j in JUDGES}, args.sampling
|
||||
)
|
||||
print(json.dumps(report, indent=2))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,233 @@
|
||||
#!/usr/bin/env python3
|
||||
"""pragent pilot — deterministic review scorers.
|
||||
|
||||
Four numbers computed from a review that already happened, shipped to Langfuse
|
||||
as scores on the review's trace. All are derived from data the reviewer already
|
||||
has in hand: no LLM judge, no ground truth, no extra token spend.
|
||||
|
||||
Why these four and not `helpfulness`/`quality`
|
||||
----------------------------------------------
|
||||
They come from what the recorded reviews actually did, not from a generic eval
|
||||
checklist:
|
||||
|
||||
* `severity_info_ratio` — of the findings ever posted to a PR, effectively all
|
||||
landed at `info`. Either the model will not commit to a severity or the
|
||||
per-repo `severity_threshold` is filtering the rest out. Trending the ratio
|
||||
per model says which.
|
||||
* `finding_rate` — most reviews post nothing at all. Silence on clean code is
|
||||
the goal; silence because the run degraded is a failure. Same output, two
|
||||
causes, and only the rate over time separates them.
|
||||
* `dropped_findings` — `ai_review.parse_findings` discards any finding whose
|
||||
`path`/`line` is unusable. That happens silently, so a model that emits ten
|
||||
findings at invalid locations is indistinguishable from one that found
|
||||
nothing. This is the only signal here that measures the *model's* output
|
||||
rather than the review's.
|
||||
* `cost_per_finding` — the equivalent-cost number is already trended per
|
||||
review; per finding is what actually compares two models, since a cheaper
|
||||
model that finds nothing is not cheaper.
|
||||
|
||||
None of these say whether a finding was *correct*. That needs labels, and the
|
||||
labels come from `feedback_scores.py` once maintainers start reacting to review
|
||||
comments. Read these as behavioural drift detectors, not as accuracy.
|
||||
|
||||
Fail-open, like every other telemetry path here: a scorer that raises returns no
|
||||
score rather than failing the review.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import uuid
|
||||
from datetime import datetime, timezone
|
||||
|
||||
# Mirrors ai_review.SEVERITY_RANK. Duplicated rather than imported because this
|
||||
# module is also run standalone (backfill) where ai_review's import side effects
|
||||
# are unwanted.
|
||||
SEVERITY_RANK = {"info": -1, "trivial": 0, "low": 1, "medium": 2, "high": 3, "critical": 4}
|
||||
|
||||
# Findings at or below this rank are "the model declined to commit". `trivial`
|
||||
# and `info` are advisory by the reviewer's own prompt contract.
|
||||
_ADVISORY_MAX_RANK = 0
|
||||
|
||||
# Score names. Named for what is measured, not for the mechanism producing it —
|
||||
# these land on every trace and become the axis of every chart.
|
||||
FINDING_RATE = "finding_rate"
|
||||
SEVERITY_INFO_RATIO = "severity_info_ratio"
|
||||
SEVERITY_MAX = "severity_max"
|
||||
DROPPED_FINDINGS = "dropped_findings"
|
||||
COST_PER_FINDING = "cost_per_finding"
|
||||
|
||||
|
||||
def _sev(f: dict) -> str:
|
||||
return str(f.get("severity") or "medium").strip().lower()
|
||||
|
||||
|
||||
def finding_rate(findings: list[dict] | None) -> float:
|
||||
"""How many findings this review posted. 0.0 is the restraint case."""
|
||||
return float(len(findings or []))
|
||||
|
||||
|
||||
def severity_info_ratio(findings: list[dict] | None) -> float | None:
|
||||
"""Share of findings the model rated advisory (`info`/`trivial`).
|
||||
|
||||
`None` for a review with no findings — a ratio over an empty set is not 0,
|
||||
it is undefined, and charting it as 0 would read as "perfectly calibrated".
|
||||
"""
|
||||
fs = findings or []
|
||||
if not fs:
|
||||
return None
|
||||
advisory = sum(1 for f in fs if SEVERITY_RANK.get(_sev(f), 2) <= _ADVISORY_MAX_RANK)
|
||||
return round(advisory / len(fs), 4)
|
||||
|
||||
|
||||
def severity_max(findings: list[dict] | None) -> str:
|
||||
"""Highest severity present, or `none` when the review was silent.
|
||||
|
||||
Categorical on purpose: the useful question is "did this review ever surface
|
||||
something serious", and an average of severity ranks answers nothing.
|
||||
"""
|
||||
fs = findings or []
|
||||
if not fs:
|
||||
return "none"
|
||||
top = max(fs, key=lambda f: SEVERITY_RANK.get(_sev(f), 2))
|
||||
sev = _sev(top)
|
||||
return sev if sev in SEVERITY_RANK else "medium"
|
||||
|
||||
|
||||
def dropped_findings(raw_count: int | None, kept_count: int | None) -> float | None:
|
||||
"""Findings the model emitted that the parser could not use.
|
||||
|
||||
`raw_count` is what came back in the JSON; `kept_count` is what survived
|
||||
`_normalize_finding`. `None` when the caller could not determine the raw
|
||||
count — better no score than a fabricated zero.
|
||||
"""
|
||||
if raw_count is None or kept_count is None:
|
||||
return None
|
||||
return float(max(0, int(raw_count) - int(kept_count)))
|
||||
|
||||
|
||||
def cost_per_finding(cost_usd: float | None, findings: list[dict] | None) -> float | None:
|
||||
"""Equivalent USD spent per finding posted.
|
||||
|
||||
`None` when nothing could be priced. A silent review divides by one, not by
|
||||
zero: the run still cost money, and attributing that whole cost to "found
|
||||
nothing" is the honest reading.
|
||||
"""
|
||||
if cost_usd is None:
|
||||
return None
|
||||
try:
|
||||
c = float(cost_usd)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
return round(c / max(1, len(findings or [])), 6)
|
||||
|
||||
|
||||
def build_scores(
|
||||
*,
|
||||
trace_id: str,
|
||||
findings: list[dict] | None,
|
||||
environment: str,
|
||||
cost_usd: float | None = None,
|
||||
dropped_count: float | None = None,
|
||||
timestamp: str | None = None,
|
||||
comment: str = "",
|
||||
) -> list[dict]:
|
||||
"""The `score-create` ingestion events for one review.
|
||||
|
||||
`dropped_count` must be measured at parse time, not here: by the time
|
||||
`findings` reaches this function the per-repo config has already filtered it
|
||||
by severity threshold and `max_findings`, and those drops are the config
|
||||
working as intended, not the model emitting garbage.
|
||||
|
||||
Returns [] rather than raising if something is unscoreable — scores are
|
||||
telemetry and must never cost a review.
|
||||
"""
|
||||
# The ingestion envelope requires a timestamp on every event; omitting it
|
||||
# gets the whole batch rejected with an HTTP 207 whose per-event 400s are
|
||||
# easy to mistake for success.
|
||||
ts = timestamp or datetime.now(timezone.utc).isoformat().replace("+00:00", "Z")
|
||||
out: list[dict] = []
|
||||
|
||||
def add(name: str, value, data_type: str) -> None:
|
||||
if value is None:
|
||||
return
|
||||
body = {
|
||||
"id": str(uuid.uuid4()),
|
||||
"traceId": trace_id,
|
||||
"name": name,
|
||||
"dataType": data_type,
|
||||
"environment": environment,
|
||||
}
|
||||
if data_type == "CATEGORICAL":
|
||||
body["value"] = str(value)
|
||||
else:
|
||||
body["value"] = float(value)
|
||||
if comment:
|
||||
body["comment"] = comment
|
||||
out.append(
|
||||
{
|
||||
"id": str(uuid.uuid4()),
|
||||
"type": "score-create",
|
||||
"timestamp": ts,
|
||||
"body": body,
|
||||
}
|
||||
)
|
||||
|
||||
try:
|
||||
add(FINDING_RATE, finding_rate(findings), "NUMERIC")
|
||||
add(SEVERITY_INFO_RATIO, severity_info_ratio(findings), "NUMERIC")
|
||||
add(SEVERITY_MAX, severity_max(findings), "CATEGORICAL")
|
||||
add(DROPPED_FINDINGS, dropped_count, "NUMERIC")
|
||||
add(COST_PER_FINDING, cost_per_finding(cost_usd, findings), "NUMERIC")
|
||||
except Exception: # pragma: no cover - defensive
|
||||
return out
|
||||
return out
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Score configs — the schema these scores must comply with
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Registered once per project via `eval_bootstrap.py`. Without configs the
|
||||
# scores still ingest, but nothing constrains a future scorer from writing
|
||||
# `severity_max="HIGH"` next to today's `"high"` and silently splitting the
|
||||
# series in two.
|
||||
SCORE_CONFIGS = [
|
||||
{
|
||||
"name": FINDING_RATE,
|
||||
"dataType": "NUMERIC",
|
||||
"minValue": 0,
|
||||
"description": "Findings posted by one review. 0 = the reviewer stayed silent.",
|
||||
},
|
||||
{
|
||||
"name": SEVERITY_INFO_RATIO,
|
||||
"dataType": "NUMERIC",
|
||||
"minValue": 0,
|
||||
"maxValue": 1,
|
||||
"description": "Share of a review's findings rated info/trivial. High = the model is not committing to a severity.",
|
||||
},
|
||||
{
|
||||
"name": SEVERITY_MAX,
|
||||
"dataType": "CATEGORICAL",
|
||||
"categories": [
|
||||
{"label": "none", "value": 0},
|
||||
{"label": "info", "value": 1},
|
||||
{"label": "trivial", "value": 2},
|
||||
{"label": "low", "value": 3},
|
||||
{"label": "medium", "value": 4},
|
||||
{"label": "high", "value": 5},
|
||||
{"label": "critical", "value": 6},
|
||||
],
|
||||
"description": "Highest severity surfaced by one review; 'none' when it posted nothing.",
|
||||
},
|
||||
{
|
||||
"name": DROPPED_FINDINGS,
|
||||
"dataType": "NUMERIC",
|
||||
"minValue": 0,
|
||||
"description": "Findings the model emitted that the parser rejected for an unusable path/line.",
|
||||
},
|
||||
{
|
||||
"name": COST_PER_FINDING,
|
||||
"dataType": "NUMERIC",
|
||||
"minValue": 0,
|
||||
"description": "Equivalent USD per finding posted. Silent reviews divide by 1, not 0.",
|
||||
},
|
||||
]
|
||||
Reference in New Issue
Block a user