feat(eval): filterable item metadata and dataset runs for the Experiments tab

The filter bar matches on metadata only — not on input and not on the item id —
so a dataset seeded with repo/pr in `input` alone could not be sliced by repo
at all. Every facet worth filtering on is now a flat primitive in `metadata`:
repo, owner, repo_name, pr, head_sha, finding_count, has_findings,
max_severity, reviews_run and the review timestamp both ways. `owner` is split
out because a filter on the joined repo matches one repo, never a whole org,
and `max_severity` is "none" rather than absent because an absent key matches
no filter.

`eval_experiment.py` links reviews that already ran into a dataset run, one run
per model, so the Experiments tab is populated without re-running the reviewer.
One trace per (run, item), the most recent: a PR re-reviewed on every push has
many traces and a run is one output per input.

It posts to the deprecated /api/public/dataset-run-items — the notice exempts
self-hosted v3 from the cutoff date and the pilot is stdlib-only by design.
Revisit at v4.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Marcos
2026-08-31 15:53:35 +00:00
parent 1534a99630
commit 2e1ad817e7
5 changed files with 535 additions and 8 deletions
+45 -8
View File
@@ -44,6 +44,7 @@ import sqlite3
import sys
import urllib.error
import urllib.request
from datetime import datetime, timezone
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
@@ -125,6 +126,37 @@ def item_id(repo: str, pr) -> str:
return f"{repo.replace('/', '__')}__pr{pr}"
def _item_metadata(*, repo, pr, head_sha, reviews_run, last_seen, findings) -> dict:
"""Filterable facets for one dataset item.
Kept flat and primitive: the filter bar matches a metadata key against a
literal, so a nested object or a list is not reachable from the UI.
"""
owner, _, repo_name = str(repo).partition("/")
sevs = [str(f["severity"] or "").lower() for f in findings]
ranked = [s for s in sevs if s in eval_scores.SEVERITY_RANK]
return {
"repo": repo,
"owner": owner or repo,
"repo_name": repo_name or repo,
"pr": int(pr),
"head_sha": head_sha,
"reviews_run": reviews_run,
"last_reviewed_at": last_seen,
"last_reviewed_iso": datetime.fromtimestamp(last_seen, timezone.utc).isoformat(),
"finding_count": len(findings),
"has_findings": bool(findings),
# "none" rather than omitting the key: a filter for silent reviews needs
# something to match, and an absent key matches nothing.
"max_severity": (
max(ranked, key=lambda s: eval_scores.SEVERITY_RANK[s]) if ranked else "none"
),
# Flags that this row is the reviewer's own past output, not a human
# judgement. Filter on it before anyone treats the dataset as truth.
"labelled_by_human": False,
}
def read_review_items(db_path: str) -> list[dict]:
"""One dataset item per (repo, pr) the reviewer has run on.
@@ -164,14 +196,19 @@ def read_review_items(db_path: str) -> list[dict]:
"findings": [dict(f) for f in findings],
"finding_count": len(findings),
},
"metadata": {
"reviews_run": int(row["reviews"]),
"last_reviewed_at": int(row["last_seen"]),
# Flags that this row is the reviewer's own past output,
# not a human judgement. Filter on it before anyone
# treats the dataset as ground truth.
"labelled_by_human": False,
},
# The UI's filter bar reads metadata and nothing else, so
# anything worth slicing on is a top-level key here even
# where it duplicates `input`. `owner` and `repo_name` are
# split out because a filter on the joined `repo` can only
# match one repo at a time, never a whole org.
"metadata": _item_metadata(
repo=row["repo"],
pr=row["pr"],
head_sha=row["head_sha"],
reviews_run=int(row["reviews"]),
last_seen=int(row["last_seen"]),
findings=findings,
),
}
)
return items