2f96e66aab
Adds the evaluation layer on top of the review traces: five deterministic
scores describing how the reviewer behaved, a bridge that turns human reactions
into ground truth, and a dataset seeded from the reviews already run.
The two are kept apart on purpose. feedback.db has recorded 113 reviews and
zero reactions, resolutions or replies — nobody has ever responded to a bot
comment — so an accuracy metric cannot be built yet. The scorers therefore
measure behaviour, which is computable from data in hand, and feedback_scores
turns verdicts into scores the moment any arrive.
eval_scores.py emits finding_rate, severity_info_ratio, severity_max,
dropped_findings and cost_per_finding into the same ingestion batch as the
trace. Undefined values are omitted rather than reported as zero: an info ratio
over a silent review is undefined, and charting it as 0 would read as perfect
calibration.
dropped_findings needed a parser change. Both parsers silently discard findings
with an unusable path/line, which made a model emitting garbage locations
indistinguishable from one that found nothing. last_parse_dropped() exposes the
delta, read at parse time — after apply_repo_config the drops are the config
working as intended, not the model misbehaving.
feedback_scores.py scores the session ("{repo}#{pr}"), because feedback arrives
days later against a PR and nothing records which re-run produced which
comment. review_acceptance is absent rather than 0 when nothing was engaged.
eval_bootstrap.py registers the score configs, seeds the pragent-reviews
dataset, and can backfill scores onto traces that predate the scorers.
expectedOutput is the reviewer's own prior output, flagged
labelled_by_human: false — a regression baseline, not verified truth.
Also fixes a silent telemetry failure: the ingestion endpoint answers 207 when
only some events succeed, so a batch with every event rejected still looked
like success. Score events were missing the required per-event timestamp and
ingested nothing while reporting 207. _warn_on_rejected_events now logs the
per-event errors under LANGFUSE_DEBUG.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
327 lines
12 KiB
Python
327 lines
12 KiB
Python
"""Unit tests for Langfuse trace emission. No network.
|
|
|
|
`_post` is monkeypatched everywhere a POST would happen; a test that reaches
|
|
the real network is a bug in the test, not a slow test.
|
|
"""
|
|
import json
|
|
import os
|
|
import sys
|
|
|
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
|
ROOT = os.path.abspath(os.path.join(HERE, "..", ".."))
|
|
sys.path.insert(0, os.path.join(ROOT, "pilot"))
|
|
|
|
import langfuse_trace as lt # noqa: E402
|
|
|
|
|
|
USAGE = {
|
|
"input": 2_000_000,
|
|
"output": 17_000,
|
|
"reasoning": 500,
|
|
"cache_read": 400_000,
|
|
"cache_write": 50_000,
|
|
"total": 2_017_000,
|
|
"cost": 0.0,
|
|
"steps": 28,
|
|
"duration_s": 348.3,
|
|
}
|
|
|
|
BASE = dict(
|
|
repo="techspark/pragent",
|
|
index="42",
|
|
sha="2613b3e1122334455",
|
|
title="Harden the review path",
|
|
usage=USAGE,
|
|
findings=[
|
|
{"severity": "critical", "path": "a.py"},
|
|
{"severity": "minor", "path": "b.py"},
|
|
{"severity": "minor", "path": "c.py"},
|
|
],
|
|
summary="Three findings.",
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# model -> environment split (the whole point of the integration)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_claude_models_land_in_the_claude_environment():
|
|
assert lt.resolve_environment("headroom/claude-sonnet-5") == "claude"
|
|
assert lt.resolve_environment("claude-opus-5") == "claude"
|
|
|
|
|
|
def test_everything_else_lands_in_the_ollama_environment():
|
|
for m in (
|
|
"headroom/glm-5.2:cloud",
|
|
"headroom/MiniMax-M2.7",
|
|
"vllm-qwen38/qwen3.8-27b",
|
|
"gpt-5",
|
|
):
|
|
assert lt.resolve_environment(m) == "ollama", m
|
|
|
|
|
|
def test_provider_and_bare_model_are_split_on_the_first_slash_only():
|
|
assert lt.provider_of("vllm-qwen38/qwen3.8-27b") == "vllm-qwen38"
|
|
assert lt.strip_provider("headroom/glm-5.2:cloud") == "glm-5.2:cloud"
|
|
# A bare name has no provider prefix; default to the pilot's proxy.
|
|
assert lt.provider_of("glm-5.2:cloud") == "headroom"
|
|
assert lt.strip_provider("glm-5.2:cloud") == "glm-5.2:cloud"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# usage accounting
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_cache_reads_are_subtracted_from_input_not_added():
|
|
# Langfuse sums usageDetails keys; opencode reports cache_read *inside*
|
|
# input, so reporting both raw would bill the prefix twice.
|
|
d = lt._usage_details(USAGE)
|
|
assert d["input"] == 2_000_000 - 400_000
|
|
assert d["cache_read_input_tokens"] == 400_000
|
|
assert d["cache_write_input_tokens"] == 50_000
|
|
assert d["output"] == 17_000
|
|
assert d["reasoning"] == 500
|
|
|
|
|
|
def test_zero_cache_fields_are_omitted_rather_than_sent_as_zero():
|
|
d = lt._usage_details({"input": 100, "output": 10})
|
|
assert d == {"input": 100, "output": 10}
|
|
|
|
|
|
def test_a_paid_model_is_priced_as_itself():
|
|
costs, basis = lt._cost_details(USAGE, "headroom/claude-sonnet-5")
|
|
assert costs["total"] > 0
|
|
assert basis == "actual"
|
|
|
|
|
|
def test_minimax_is_priced_against_the_comparison_target_not_zero():
|
|
# MiniMax-M2.7 is the model the webhook actually runs and it is absent from
|
|
# PRICES; charting it at $0 would make the whole dashboard a flat line.
|
|
costs, basis = lt._cost_details(USAGE, "headroom/MiniMax-M2.7")
|
|
assert costs["total"] > 0
|
|
assert basis == "equivalent:claude-sonnet-5"
|
|
|
|
|
|
def test_glm_is_priced_against_the_comparison_target():
|
|
costs, basis = lt._cost_details(USAGE, "headroom/glm-5.2:cloud")
|
|
assert costs["total"] > 0
|
|
assert basis.startswith("equivalent:")
|
|
|
|
|
|
def test_an_all_zero_price_entry_counts_as_free_not_as_priced():
|
|
# The self-hosted vLLM qwen IS in PRICES, at 0.00 across the board.
|
|
costs, basis = lt._cost_details(USAGE, "vllm-qwen38/qwen3.8-27b")
|
|
assert costs["total"] > 0
|
|
assert basis.startswith("equivalent:")
|
|
|
|
|
|
def test_explicit_price_target_wins_over_the_default():
|
|
costs, basis = lt._cost_details(USAGE, "headroom/MiniMax-M2.7", "claude-opus-5")
|
|
assert basis == "equivalent:claude-opus-5"
|
|
sonnet, _ = lt._cost_details(USAGE, "headroom/MiniMax-M2.7", "claude-sonnet-5")
|
|
assert costs["total"] > sonnet["total"]
|
|
|
|
|
|
def test_env_overrides_the_default_target(monkeypatch):
|
|
monkeypatch.setenv("PRAGENT_PRICE_TARGET", "claude-haiku-4-5")
|
|
assert lt.resolve_price_target() == "claude-haiku-4-5"
|
|
# An explicit argument still beats the env.
|
|
assert lt.resolve_price_target("gpt-5") == "gpt-5"
|
|
|
|
|
|
def test_unknown_comparison_target_yields_no_cost_block_rather_than_a_wrong_one():
|
|
costs, basis = lt._cost_details(USAGE, "headroom/MiniMax-M2.7", "not-a-real-model")
|
|
assert costs == {}
|
|
assert basis == ""
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# batch shape
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_batch_has_a_trace_and_a_generation_linked_by_trace_id():
|
|
batch = lt.build_batch(model="headroom/claude-sonnet-5", **BASE)
|
|
types = [e["type"] for e in batch]
|
|
# Scores ride in the same batch; the trace and generation lead it.
|
|
assert types[:2] == ["trace-create", "generation-create"]
|
|
trace, gen = batch[0], batch[1]
|
|
assert gen["body"]["traceId"] == trace["body"]["id"]
|
|
assert trace["body"]["environment"] == gen["body"]["environment"] == "claude"
|
|
|
|
|
|
def test_batch_without_usage_has_no_generation():
|
|
batch = lt.build_batch(model="headroom/glm-5.2:cloud", **{**BASE, "usage": None})
|
|
types = [e["type"] for e in batch]
|
|
assert "generation-create" not in types
|
|
assert types[0] == "trace-create"
|
|
|
|
|
|
def test_trace_carries_repo_pr_session_and_severity_counts():
|
|
batch = lt.build_batch(model="headroom/glm-5.2:cloud", **BASE)
|
|
body = batch[0]["body"]
|
|
assert body["sessionId"] == "techspark/pragent#42"
|
|
assert body["metadata"]["severities"] == {"critical": 1, "minor": 2}
|
|
assert body["metadata"]["findings"] == 3
|
|
assert "provider:headroom" in body["tags"]
|
|
assert "model:glm-5.2:cloud" in body["tags"]
|
|
|
|
|
|
def test_lens_names_become_tags():
|
|
batch = lt.build_batch(
|
|
model="headroom/glm-5.2:cloud", lenses=["security", "tests"], **BASE
|
|
)
|
|
assert "lens:security" in batch[0]["body"]["tags"]
|
|
assert "lens:tests" in batch[0]["body"]["tags"]
|
|
|
|
|
|
def test_cost_basis_is_tagged_so_equivalent_is_never_read_as_spend():
|
|
batch = lt.build_batch(model="headroom/MiniMax-M2.7", **BASE)
|
|
trace = batch[0]["body"]
|
|
assert "cost:equivalent:claude-sonnet-5" in trace["tags"]
|
|
assert trace["metadata"]["cost_basis"] == "equivalent:claude-sonnet-5"
|
|
|
|
paid = lt.build_batch(model="headroom/claude-sonnet-5", **BASE)
|
|
assert "cost:actual" in paid[0]["body"]["tags"]
|
|
|
|
|
|
def test_minimax_generation_carries_a_nonzero_cost():
|
|
batch = lt.build_batch(model="headroom/MiniMax-M2.7", **BASE)
|
|
assert batch[1]["body"]["costDetails"]["total"] > 0
|
|
|
|
|
|
def test_batch_is_json_serializable():
|
|
batch = lt.build_batch(model="headroom/claude-sonnet-5", **BASE)
|
|
json.dumps({"batch": batch})
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# emit_review_trace — config gate and fail-open
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _configure(monkeypatch):
|
|
monkeypatch.setenv("LANGFUSE_HOST", "http://langfuse.test:3000/")
|
|
monkeypatch.setenv("LANGFUSE_PUBLIC_KEY", "pk-lf-test")
|
|
monkeypatch.setenv("LANGFUSE_SECRET_KEY", "sk-lf-test")
|
|
|
|
|
|
def test_no_config_means_no_post_and_no_error(monkeypatch):
|
|
for k in ("LANGFUSE_HOST", "LANGFUSE_PUBLIC_KEY", "LANGFUSE_SECRET_KEY"):
|
|
monkeypatch.delenv(k, raising=False)
|
|
calls = []
|
|
monkeypatch.setattr(lt, "_post", lambda *a, **k: calls.append(a) or 200)
|
|
assert lt.emit_review_trace(model="headroom/glm-5.2:cloud", **BASE) is False
|
|
assert calls == []
|
|
|
|
|
|
def test_configured_emit_posts_to_the_ingestion_endpoint(monkeypatch):
|
|
_configure(monkeypatch)
|
|
seen = {}
|
|
|
|
def fake_post(host, pk, sk, batch, timeout):
|
|
seen.update(host=host, pk=pk, sk=sk, batch=batch, timeout=timeout)
|
|
return 207
|
|
|
|
monkeypatch.setattr(lt, "_post", fake_post)
|
|
assert lt.emit_review_trace(model="headroom/claude-sonnet-5", **BASE) is True
|
|
# Trailing slash stripped so the path is not doubled.
|
|
assert seen["host"] == "http://langfuse.test:3000"
|
|
kinds = [e["type"] for e in seen["batch"]]
|
|
assert kinds[:2] == ["trace-create", "generation-create"]
|
|
assert "score-create" in kinds
|
|
|
|
|
|
def test_transport_failure_is_swallowed(monkeypatch):
|
|
_configure(monkeypatch)
|
|
|
|
def boom(*a, **k):
|
|
raise OSError("connection refused")
|
|
|
|
monkeypatch.setattr(lt, "_post", boom)
|
|
assert lt.emit_review_trace(model="headroom/glm-5.2:cloud", **BASE) is False
|
|
|
|
|
|
def test_non_success_status_reports_failure_without_raising(monkeypatch):
|
|
_configure(monkeypatch)
|
|
monkeypatch.setattr(lt, "_post", lambda *a, **k: 401)
|
|
assert lt.emit_review_trace(model="headroom/glm-5.2:cloud", **BASE) is False
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Scores folded into the review batch (added with eval_scores)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
def _scores(events):
|
|
return {e["body"]["name"]: e["body"] for e in events if e["type"] == "score-create"}
|
|
|
|
|
|
def test_build_batch_appends_scores():
|
|
events = lt.build_batch(
|
|
repo="o/r", index="1", sha="abc", title="t",
|
|
model="headroom/claude-sonnet-5",
|
|
usage={"input": 100, "output": 10},
|
|
findings=[{"severity": "high", "path": "a.py", "line": 1}],
|
|
)
|
|
names = set(_scores(events))
|
|
assert "finding_rate" in names
|
|
assert "severity_max" in names
|
|
|
|
|
|
def test_scores_attach_to_the_same_trace():
|
|
events = lt.build_batch(
|
|
repo="o/r", index="1", sha="abc", title="t", model="m",
|
|
usage={"input": 1, "output": 1}, findings=[], trace_id="fixed-id",
|
|
)
|
|
for body in _scores(events).values():
|
|
assert body["traceId"] == "fixed-id"
|
|
|
|
|
|
def test_scores_inherit_the_trace_environment():
|
|
events = lt.build_batch(
|
|
repo="o/r", index="1", sha="abc", title="t",
|
|
model="headroom/glm-5.2:cloud",
|
|
usage={"input": 1, "output": 1}, findings=[],
|
|
)
|
|
for body in _scores(events).values():
|
|
assert body["environment"] == "ollama"
|
|
|
|
|
|
def test_dropped_findings_scored_when_provided():
|
|
events = lt.build_batch(
|
|
repo="o/r", index="1", sha="abc", title="t", model="m",
|
|
usage={"input": 1, "output": 1}, findings=[], dropped_count=3,
|
|
)
|
|
assert _scores(events)["dropped_findings"]["value"] == 3.0
|
|
|
|
|
|
def test_dropped_findings_absent_when_not_measured():
|
|
events = lt.build_batch(
|
|
repo="o/r", index="1", sha="abc", title="t", model="m",
|
|
usage={"input": 1, "output": 1}, findings=[],
|
|
)
|
|
assert "dropped_findings" not in _scores(events)
|
|
|
|
|
|
def test_cost_score_carries_its_basis_in_the_comment():
|
|
# An equivalent-cost $/finding must never be read as money spent.
|
|
events = lt.build_batch(
|
|
repo="o/r", index="1", sha="abc", title="t",
|
|
model="headroom/glm-5.2:cloud",
|
|
usage={"input": 1000, "output": 100}, findings=[{"severity": "low", "path": "a", "line": 1}],
|
|
)
|
|
cpf = _scores(events).get("cost_per_finding")
|
|
if cpf is not None: # only when cost_model could price the comparison target
|
|
assert "equivalent" in cpf["comment"]
|
|
|
|
|
|
def test_batch_without_usage_still_scores_findings():
|
|
# A run with no usage report still produced findings worth scoring.
|
|
events = lt.build_batch(
|
|
repo="o/r", index="1", sha="abc", title="t", model="m",
|
|
usage=None, findings=[{"severity": "critical", "path": "a", "line": 2}],
|
|
)
|
|
assert _scores(events)["severity_max"]["value"] == "critical"
|