feat(pilot): behavioural scorers, feedback ground truth, and an eval dataset
Adds the evaluation layer on top of the review traces: five deterministic
scores describing how the reviewer behaved, a bridge that turns human reactions
into ground truth, and a dataset seeded from the reviews already run.
The two are kept apart on purpose. feedback.db has recorded 113 reviews and
zero reactions, resolutions or replies — nobody has ever responded to a bot
comment — so an accuracy metric cannot be built yet. The scorers therefore
measure behaviour, which is computable from data in hand, and feedback_scores
turns verdicts into scores the moment any arrive.
eval_scores.py emits finding_rate, severity_info_ratio, severity_max,
dropped_findings and cost_per_finding into the same ingestion batch as the
trace. Undefined values are omitted rather than reported as zero: an info ratio
over a silent review is undefined, and charting it as 0 would read as perfect
calibration.
dropped_findings needed a parser change. Both parsers silently discard findings
with an unusable path/line, which made a model emitting garbage locations
indistinguishable from one that found nothing. last_parse_dropped() exposes the
delta, read at parse time — after apply_repo_config the drops are the config
working as intended, not the model misbehaving.
feedback_scores.py scores the session ("{repo}#{pr}"), because feedback arrives
days later against a PR and nothing records which re-run produced which
comment. review_acceptance is absent rather than 0 when nothing was engaged.
eval_bootstrap.py registers the score configs, seeds the pragent-reviews
dataset, and can backfill scores onto traces that predate the scorers.
expectedOutput is the reviewer's own prior output, flagged
labelled_by_human: false — a regression baseline, not verified truth.
Also fixes a silent telemetry failure: the ingestion endpoint answers 207 when
only some events succeed, so a batch with every event rejected still looked
like success. Score events were missing the required per-event timestamp and
ingested nothing while reporting 207. _warn_on_rejected_events now logs the
per-event errors under LANGFUSE_DEBUG.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -145,15 +145,18 @@ def test_unknown_comparison_target_yields_no_cost_block_rather_than_a_wrong_one(
|
||||
def test_batch_has_a_trace_and_a_generation_linked_by_trace_id():
|
||||
batch = lt.build_batch(model="headroom/claude-sonnet-5", **BASE)
|
||||
types = [e["type"] for e in batch]
|
||||
assert types == ["trace-create", "generation-create"]
|
||||
trace, gen = batch
|
||||
# Scores ride in the same batch; the trace and generation lead it.
|
||||
assert types[:2] == ["trace-create", "generation-create"]
|
||||
trace, gen = batch[0], batch[1]
|
||||
assert gen["body"]["traceId"] == trace["body"]["id"]
|
||||
assert trace["body"]["environment"] == gen["body"]["environment"] == "claude"
|
||||
|
||||
|
||||
def test_batch_without_usage_is_trace_only():
|
||||
def test_batch_without_usage_has_no_generation():
|
||||
batch = lt.build_batch(model="headroom/glm-5.2:cloud", **{**BASE, "usage": None})
|
||||
assert [e["type"] for e in batch] == ["trace-create"]
|
||||
types = [e["type"] for e in batch]
|
||||
assert "generation-create" not in types
|
||||
assert types[0] == "trace-create"
|
||||
|
||||
|
||||
def test_trace_carries_repo_pr_session_and_severity_counts():
|
||||
@@ -226,7 +229,9 @@ def test_configured_emit_posts_to_the_ingestion_endpoint(monkeypatch):
|
||||
assert lt.emit_review_trace(model="headroom/claude-sonnet-5", **BASE) is True
|
||||
# Trailing slash stripped so the path is not doubled.
|
||||
assert seen["host"] == "http://langfuse.test:3000"
|
||||
assert len(seen["batch"]) == 2
|
||||
kinds = [e["type"] for e in seen["batch"]]
|
||||
assert kinds[:2] == ["trace-create", "generation-create"]
|
||||
assert "score-create" in kinds
|
||||
|
||||
|
||||
def test_transport_failure_is_swallowed(monkeypatch):
|
||||
@@ -243,3 +248,79 @@ def test_non_success_status_reports_failure_without_raising(monkeypatch):
|
||||
_configure(monkeypatch)
|
||||
monkeypatch.setattr(lt, "_post", lambda *a, **k: 401)
|
||||
assert lt.emit_review_trace(model="headroom/glm-5.2:cloud", **BASE) is False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Scores folded into the review batch (added with eval_scores)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _scores(events):
|
||||
return {e["body"]["name"]: e["body"] for e in events if e["type"] == "score-create"}
|
||||
|
||||
|
||||
def test_build_batch_appends_scores():
|
||||
events = lt.build_batch(
|
||||
repo="o/r", index="1", sha="abc", title="t",
|
||||
model="headroom/claude-sonnet-5",
|
||||
usage={"input": 100, "output": 10},
|
||||
findings=[{"severity": "high", "path": "a.py", "line": 1}],
|
||||
)
|
||||
names = set(_scores(events))
|
||||
assert "finding_rate" in names
|
||||
assert "severity_max" in names
|
||||
|
||||
|
||||
def test_scores_attach_to_the_same_trace():
|
||||
events = lt.build_batch(
|
||||
repo="o/r", index="1", sha="abc", title="t", model="m",
|
||||
usage={"input": 1, "output": 1}, findings=[], trace_id="fixed-id",
|
||||
)
|
||||
for body in _scores(events).values():
|
||||
assert body["traceId"] == "fixed-id"
|
||||
|
||||
|
||||
def test_scores_inherit_the_trace_environment():
|
||||
events = lt.build_batch(
|
||||
repo="o/r", index="1", sha="abc", title="t",
|
||||
model="headroom/glm-5.2:cloud",
|
||||
usage={"input": 1, "output": 1}, findings=[],
|
||||
)
|
||||
for body in _scores(events).values():
|
||||
assert body["environment"] == "ollama"
|
||||
|
||||
|
||||
def test_dropped_findings_scored_when_provided():
|
||||
events = lt.build_batch(
|
||||
repo="o/r", index="1", sha="abc", title="t", model="m",
|
||||
usage={"input": 1, "output": 1}, findings=[], dropped_count=3,
|
||||
)
|
||||
assert _scores(events)["dropped_findings"]["value"] == 3.0
|
||||
|
||||
|
||||
def test_dropped_findings_absent_when_not_measured():
|
||||
events = lt.build_batch(
|
||||
repo="o/r", index="1", sha="abc", title="t", model="m",
|
||||
usage={"input": 1, "output": 1}, findings=[],
|
||||
)
|
||||
assert "dropped_findings" not in _scores(events)
|
||||
|
||||
|
||||
def test_cost_score_carries_its_basis_in_the_comment():
|
||||
# An equivalent-cost $/finding must never be read as money spent.
|
||||
events = lt.build_batch(
|
||||
repo="o/r", index="1", sha="abc", title="t",
|
||||
model="headroom/glm-5.2:cloud",
|
||||
usage={"input": 1000, "output": 100}, findings=[{"severity": "low", "path": "a", "line": 1}],
|
||||
)
|
||||
cpf = _scores(events).get("cost_per_finding")
|
||||
if cpf is not None: # only when cost_model could price the comparison target
|
||||
assert "equivalent" in cpf["comment"]
|
||||
|
||||
|
||||
def test_batch_without_usage_still_scores_findings():
|
||||
# A run with no usage report still produced findings worth scoring.
|
||||
events = lt.build_batch(
|
||||
repo="o/r", index="1", sha="abc", title="t", model="m",
|
||||
usage=None, findings=[{"severity": "critical", "path": "a", "line": 2}],
|
||||
)
|
||||
assert _scores(events)["severity_max"]["value"] == "critical"
|
||||
|
||||
Reference in New Issue
Block a user