refactor: organize pilot packages
Group review, feedback, evaluation, observability, and entrypoint code into packages. Keep thin top-level compatibility shims for existing scripts and imports, and mirror the structure in the tests.
This commit is contained in:
@@ -0,0 +1 @@
|
||||
"""Evaluation tests."""
|
||||
@@ -0,0 +1,158 @@
|
||||
"""Tests for the eval bootstrap's dataset-item construction."""
|
||||
import os
|
||||
import sqlite3
|
||||
import sys
|
||||
import urllib.parse
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "..", "..", "pilot"))
|
||||
|
||||
import eval_bootstrap as eb # noqa: E402
|
||||
|
||||
|
||||
# --- item_id --------------------------------------------------------------
|
||||
|
||||
def test_item_id_has_no_path_separator():
|
||||
"""A `/` would split the UI's item route into extra path segments."""
|
||||
assert "/" not in eb.item_id("netcracker/interview", 29)
|
||||
|
||||
|
||||
def test_item_id_has_no_fragment_marker():
|
||||
"""Everything after a `#` is a fragment the browser never sends."""
|
||||
assert "#" not in eb.item_id("netcracker/interview", 29)
|
||||
|
||||
|
||||
def test_item_id_survives_a_url_round_trip():
|
||||
"""The id must appear verbatim in a path, needing no percent-encoding."""
|
||||
ident = eb.item_id("netcracker/interview", 29)
|
||||
assert urllib.parse.quote(ident, safe="") == ident
|
||||
|
||||
|
||||
def test_item_id_keeps_repo_and_pr_readable():
|
||||
assert eb.item_id("netcracker/interview", 29) == "netcracker__interview__pr29"
|
||||
|
||||
|
||||
def test_item_id_is_unique_per_pr():
|
||||
assert eb.item_id("o/r", 1) != eb.item_id("o/r", 2)
|
||||
|
||||
|
||||
def test_item_id_is_unique_per_repo():
|
||||
assert eb.item_id("o/one", 1) != eb.item_id("o/two", 1)
|
||||
|
||||
|
||||
def test_item_id_accepts_a_string_pr():
|
||||
assert eb.item_id("o/r", "29") == eb.item_id("o/r", 29)
|
||||
|
||||
|
||||
# --- read_review_items ----------------------------------------------------
|
||||
|
||||
def _db(tmp_path, rows, findings=()):
|
||||
path = str(tmp_path / "feedback.db")
|
||||
conn = sqlite3.connect(path)
|
||||
conn.execute(
|
||||
"CREATE TABLE review (repo TEXT, pr INTEGER, posted_at INTEGER, head_sha TEXT)"
|
||||
)
|
||||
conn.execute(
|
||||
"CREATE TABLE inline_finding (repo TEXT, pr INTEGER, path TEXT, line INTEGER,"
|
||||
" severity TEXT, problem TEXT, fix TEXT)"
|
||||
)
|
||||
conn.executemany("INSERT INTO review VALUES (?,?,?,?)", rows)
|
||||
conn.executemany("INSERT INTO inline_finding VALUES (?,?,?,?,?,?,?)", findings)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
return path
|
||||
|
||||
|
||||
def test_items_use_url_safe_ids(tmp_path):
|
||||
path = _db(tmp_path, [("netcracker/interview", 29, 100, "abc")])
|
||||
items = eb.read_review_items(path)
|
||||
assert [i["id"] for i in items] == ["netcracker__interview__pr29"]
|
||||
|
||||
|
||||
def test_item_input_keeps_the_real_repo_name(tmp_path):
|
||||
"""The id is mangled for the URL; the payload must stay faithful."""
|
||||
path = _db(tmp_path, [("netcracker/interview", 29, 100, "abc")])
|
||||
item = eb.read_review_items(path)[0]
|
||||
assert item["input"]["repo"] == "netcracker/interview"
|
||||
assert item["input"]["pr"] == 29
|
||||
|
||||
|
||||
def test_one_item_per_pr_not_per_review(tmp_path):
|
||||
path = _db(
|
||||
tmp_path,
|
||||
[
|
||||
("o/r", 1, 100, "a"),
|
||||
("o/r", 1, 200, "b"),
|
||||
("o/r", 2, 300, "c"),
|
||||
],
|
||||
)
|
||||
items = eb.read_review_items(path)
|
||||
assert [i["id"] for i in items] == ["o__r__pr1", "o__r__pr2"]
|
||||
assert items[0]["metadata"]["reviews_run"] == 2
|
||||
|
||||
|
||||
def test_items_are_not_flagged_as_human_labelled(tmp_path):
|
||||
path = _db(tmp_path, [("o/r", 1, 100, "a")])
|
||||
assert eb.read_review_items(path)[0]["metadata"]["labelled_by_human"] is False
|
||||
|
||||
|
||||
# --- metadata facets ------------------------------------------------------
|
||||
|
||||
def _md(findings=(), repo="netcracker/interview", pr=29):
|
||||
return eb._item_metadata(
|
||||
repo=repo, pr=pr, head_sha="abc", reviews_run=2, last_seen=1788189422,
|
||||
findings=[{"severity": s} for s in findings],
|
||||
)
|
||||
|
||||
|
||||
def test_metadata_carries_the_repo_for_filtering():
|
||||
assert _md()["repo"] == "netcracker/interview"
|
||||
|
||||
|
||||
def test_metadata_splits_owner_from_repo_name():
|
||||
"""A filter on the joined repo can match one repo; owner matches an org."""
|
||||
md = _md()
|
||||
assert md["owner"] == "netcracker"
|
||||
assert md["repo_name"] == "interview"
|
||||
|
||||
|
||||
def test_owner_falls_back_when_the_repo_is_unqualified():
|
||||
md = _md(repo="standalone")
|
||||
assert md["owner"] == "standalone"
|
||||
assert md["repo_name"] == "standalone"
|
||||
|
||||
|
||||
def test_metadata_values_are_filterable_primitives():
|
||||
"""Nested objects and lists are not reachable from the filter bar."""
|
||||
for key, value in _md(["high"]).items():
|
||||
assert isinstance(value, (str, int, float, bool)), key
|
||||
|
||||
|
||||
def test_max_severity_is_the_worst_finding():
|
||||
assert _md(["low", "critical", "medium"])["max_severity"] == "critical"
|
||||
|
||||
|
||||
def test_max_severity_is_none_not_absent_for_a_silent_review():
|
||||
md = _md([])
|
||||
assert md["max_severity"] == "none"
|
||||
assert md["has_findings"] is False
|
||||
|
||||
|
||||
def test_unknown_severity_does_not_win_the_max():
|
||||
assert _md(["banana", "low"])["max_severity"] == "low"
|
||||
|
||||
|
||||
def test_severity_comparison_ignores_case():
|
||||
assert _md(["HIGH"])["max_severity"] == "high"
|
||||
|
||||
|
||||
def test_finding_count_matches_the_findings():
|
||||
md = _md(["low", "low"])
|
||||
assert md["finding_count"] == 2
|
||||
assert md["has_findings"] is True
|
||||
|
||||
|
||||
def test_last_reviewed_is_exposed_both_ways():
|
||||
"""The epoch sorts; the ISO string is what a human reads in a filter."""
|
||||
md = _md()
|
||||
assert md["last_reviewed_at"] == 1788189422
|
||||
assert md["last_reviewed_iso"].startswith("2026-08-31T")
|
||||
@@ -0,0 +1,159 @@
|
||||
"""Tests for linking existing review traces into dataset runs."""
|
||||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "..", "..", "pilot"))
|
||||
|
||||
import eval_experiment as ex # noqa: E402
|
||||
|
||||
|
||||
def trace(tid, repo="o/r", pr=1, model="M2", ts="2026-08-01T00:00:00Z", **md):
|
||||
meta = {"repo": repo, "pr": pr}
|
||||
meta.update(md)
|
||||
return {
|
||||
"id": tid,
|
||||
"timestamp": ts,
|
||||
"tags": [f"model:{model}", "engine:opencode"],
|
||||
"metadata": meta,
|
||||
}
|
||||
|
||||
|
||||
# --- trace_model ----------------------------------------------------------
|
||||
|
||||
def test_model_read_from_tag():
|
||||
assert ex.trace_model(trace("t1", model="MiniMax-M2.7")) == "MiniMax-M2.7"
|
||||
|
||||
|
||||
def test_model_falls_back_when_untagged():
|
||||
assert ex.trace_model({"tags": ["engine:opencode"]}) == "unknown"
|
||||
|
||||
|
||||
def test_model_falls_back_when_tags_absent():
|
||||
assert ex.trace_model({}) == "unknown"
|
||||
|
||||
|
||||
# --- trace_item_id --------------------------------------------------------
|
||||
|
||||
def test_item_id_matches_the_bootstrap_scheme():
|
||||
assert ex.trace_item_id(trace("t1", repo="netcracker/interview", pr=29)) == \
|
||||
"netcracker__interview__pr29"
|
||||
|
||||
|
||||
def test_trace_without_repo_is_not_an_item():
|
||||
assert ex.trace_item_id({"metadata": {"pr": 1}}) is None
|
||||
|
||||
|
||||
def test_trace_without_pr_is_not_an_item():
|
||||
assert ex.trace_item_id({"metadata": {"repo": "o/r"}}) is None
|
||||
|
||||
|
||||
def test_trace_without_metadata_is_not_an_item():
|
||||
assert ex.trace_item_id({}) is None
|
||||
|
||||
|
||||
# --- plan_runs ------------------------------------------------------------
|
||||
|
||||
ITEMS = {"o__r__pr1", "o__r__pr2"}
|
||||
|
||||
|
||||
def test_traces_group_by_model():
|
||||
plan = ex.plan_runs(
|
||||
[trace("a", pr=1, model="x"), trace("b", pr=2, model="y")], ITEMS
|
||||
)
|
||||
assert set(plan["runs"]) == {"x", "y"}
|
||||
|
||||
|
||||
def test_group_by_none_collapses_to_one_run():
|
||||
plan = ex.plan_runs(
|
||||
[trace("a", pr=1, model="x"), trace("b", pr=2, model="y")],
|
||||
ITEMS,
|
||||
group_by="none",
|
||||
)
|
||||
assert list(plan["runs"]) == ["all-traces"]
|
||||
|
||||
|
||||
def test_only_the_newest_trace_per_item_is_kept():
|
||||
"""A re-reviewed PR has many traces; a run takes one output per input."""
|
||||
plan = ex.plan_runs(
|
||||
[
|
||||
trace("old", pr=1, ts="2026-08-01T00:00:00Z"),
|
||||
trace("new", pr=1, ts="2026-08-09T00:00:00Z"),
|
||||
],
|
||||
ITEMS,
|
||||
)
|
||||
assert plan["runs"]["M2"]["o__r__pr1"]["id"] == "new"
|
||||
|
||||
|
||||
def test_newest_wins_regardless_of_input_order():
|
||||
older = trace("old", pr=1, ts="2026-08-01T00:00:00Z")
|
||||
newer = trace("new", pr=1, ts="2026-08-09T00:00:00Z")
|
||||
for order in ([older, newer], [newer, older]):
|
||||
plan = ex.plan_runs(order, ITEMS)
|
||||
assert plan["runs"]["M2"]["o__r__pr1"]["id"] == "new"
|
||||
|
||||
|
||||
def test_trace_for_a_pr_outside_the_dataset_is_skipped():
|
||||
plan = ex.plan_runs([trace("a", pr=99)], ITEMS)
|
||||
assert plan["runs"] == {}
|
||||
assert plan["skipped_not_in_dataset"] == 1
|
||||
|
||||
|
||||
def test_non_review_trace_is_counted_separately():
|
||||
plan = ex.plan_runs([{"id": "x", "metadata": {}}], ITEMS)
|
||||
assert plan["skipped_not_a_review"] == 1
|
||||
assert plan["skipped_not_in_dataset"] == 0
|
||||
|
||||
|
||||
def test_same_pr_different_models_lands_in_both_runs():
|
||||
plan = ex.plan_runs([trace("a", pr=1, model="x"), trace("b", pr=1, model="y")], ITEMS)
|
||||
assert plan["runs"]["x"]["o__r__pr1"]["id"] == "a"
|
||||
assert plan["runs"]["y"]["o__r__pr1"]["id"] == "b"
|
||||
|
||||
|
||||
# --- run_name -------------------------------------------------------------
|
||||
|
||||
def test_run_name_prefixed():
|
||||
assert ex.run_name("baseline", "MiniMax-M2.7") == "baseline-MiniMax-M2.7"
|
||||
|
||||
|
||||
def test_empty_prefix_leaves_the_key_bare():
|
||||
assert ex.run_name("", "MiniMax-M2.7") == "MiniMax-M2.7"
|
||||
|
||||
|
||||
# --- create_run -----------------------------------------------------------
|
||||
|
||||
def test_create_run_posts_one_item_per_pair(monkeypatch):
|
||||
calls = []
|
||||
|
||||
def fake_call(method, path, body=None, timeout=20.0):
|
||||
calls.append((method, path, body))
|
||||
return 201, {}
|
||||
|
||||
monkeypatch.setattr(ex.eb, "_call", fake_call)
|
||||
res = ex.create_run("run-1", {"o__r__pr1": trace("t1"), "o__r__pr2": trace("t2", pr=2)})
|
||||
assert res["items_linked"] == 2
|
||||
assert res["failed"] == []
|
||||
assert {c[1] for c in calls} == {"/api/public/dataset-run-items"}
|
||||
assert {c[2]["runName"] for c in calls} == {"run-1"}
|
||||
|
||||
|
||||
def test_create_run_links_the_trace_to_the_item(monkeypatch):
|
||||
seen = {}
|
||||
|
||||
def fake_call(method, path, body=None, timeout=20.0):
|
||||
seen.update(body)
|
||||
return 201, {}
|
||||
|
||||
monkeypatch.setattr(ex.eb, "_call", fake_call)
|
||||
ex.create_run("run-1", {"o__r__pr1": trace("t1")})
|
||||
assert seen["datasetItemId"] == "o__r__pr1"
|
||||
assert seen["traceId"] == "t1"
|
||||
assert seen["metadata"]["model"] == "M2"
|
||||
|
||||
|
||||
def test_create_run_reports_rejected_items(monkeypatch):
|
||||
monkeypatch.setattr(ex.eb, "_call", lambda *a, **k: (400, "nope"))
|
||||
res = ex.create_run("run-1", {"o__r__pr1": trace("t1")})
|
||||
assert res["items_linked"] == 0
|
||||
assert res["failed"][0]["item"] == "o__r__pr1"
|
||||
assert res["failed"][0]["status"] == 400
|
||||
@@ -0,0 +1,132 @@
|
||||
|
||||
|
||||
"""Tests for the LLM-as-judge evaluator bootstrap."""
|
||||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "..", "pilot"))
|
||||
|
||||
import eval_judges as ej # noqa: E402
|
||||
|
||||
|
||||
# --- rule_body ------------------------------------------------------------
|
||||
|
||||
def test_rule_body_targets_traces():
|
||||
"""Trace target matches the path `/api/public/ingestion` triggers.
|
||||
|
||||
Observation rules only fire from the OTel ingestion pipeline; this
|
||||
pilot uses standard ingestion, so its jobs only come from
|
||||
`evalService.createEvalJobs` and that dispatcher handles
|
||||
`targetObject ∈ {TRACE, DATASET}`.
|
||||
"""
|
||||
body = ej.rule_body("rule-x", "finding_actionability", 1.0)
|
||||
assert body["target"] == "trace"
|
||||
assert body["enabled"] is True
|
||||
|
||||
|
||||
def test_rule_body_filters_on_trace_name():
|
||||
"""`name` isn't a stringOptions column; only `traceName` is."""
|
||||
body = ej.rule_body("rule-x", "finding_actionability", 1.0)
|
||||
f = body["filter"][0]
|
||||
assert f["column"] == "traceName"
|
||||
assert f["operator"] == "any of"
|
||||
assert f["type"] == "stringOptions"
|
||||
assert "pr-review" in f["value"]
|
||||
|
||||
|
||||
def test_rule_body_references_evaluator_by_name():
|
||||
"""Ids are version-specific; rules must name the evaluator across versions."""
|
||||
body = ej.rule_body("rule-x", "finding_actionability", 1.0)
|
||||
assert body["evaluator"]["name"] == "finding_actionability"
|
||||
assert body["evaluator"]["scope"] == "project"
|
||||
|
||||
|
||||
def test_rule_body_maps_input_and_output():
|
||||
"""Both judges read the observation's own input/output."""
|
||||
body = ej.rule_body("rule-x", "any", 1.0)
|
||||
sources = {m["source"] for m in body["mapping"]}
|
||||
assert sources == {"input", "output"}
|
||||
|
||||
|
||||
def test_rule_body_carries_mapping_at_both_levels():
|
||||
"""The server validates `mapping` at the rule root and echoes it on the evaluator."""
|
||||
body = ej.rule_body("rule-x", "any", 1.0)
|
||||
assert body["mapping"]
|
||||
assert body["evaluator"]["variableMapping"] == body["mapping"]
|
||||
|
||||
|
||||
def test_rule_body_passes_sampling_through():
|
||||
assert ej.rule_body("r", "any", 0.25)["sampling"] == 0.25
|
||||
|
||||
|
||||
# --- ensure_evaluators idempotency ---------------------------------------
|
||||
|
||||
def test_ensure_evaluators_skips_existing(monkeypatch):
|
||||
seen = []
|
||||
|
||||
def fake_call(method, path, body=None, timeout=20.0):
|
||||
seen.append(path)
|
||||
return 200, {}
|
||||
|
||||
monkeypatch.setattr(ej.eb, "_call", fake_call)
|
||||
monkeypatch.setattr(ej, "existing_evaluators",
|
||||
lambda: {"finding_actionability": "id-1", "review_self_consistency": "id-2"})
|
||||
res = ej.ensure_evaluators()
|
||||
assert res["created"] == {}
|
||||
assert sorted(res["skipped"]) == ["finding_actionability", "review_self_consistency"]
|
||||
assert res["failed"] == []
|
||||
assert seen == []
|
||||
|
||||
|
||||
def test_ensure_evaluators_records_failures(monkeypatch):
|
||||
def fake_call(method, path, body=None, timeout=20.0):
|
||||
return 422, "boom"
|
||||
|
||||
monkeypatch.setattr(ej.eb, "_call", fake_call)
|
||||
monkeypatch.setattr(ej, "existing_evaluators", lambda: {})
|
||||
res = ej.ensure_evaluators()
|
||||
assert res["created"] == {}
|
||||
assert res["failed"][0]["status"] == 422
|
||||
|
||||
|
||||
# --- ensure_rules idempotency --------------------------------------------
|
||||
|
||||
def test_ensure_rules_skips_existing(monkeypatch):
|
||||
calls = []
|
||||
monkeypatch.setattr(ej.eb, "_call",
|
||||
lambda *a, **k: calls.append(a) or (200, {}))
|
||||
monkeypatch.setattr(ej, "existing_evaluators",
|
||||
lambda: {"finding_actionability": "id-1",
|
||||
"review_self_consistency": "id-2"})
|
||||
monkeypatch.setattr(ej, "existing_rule_names",
|
||||
lambda: {"finding_actionability-on-reviews",
|
||||
"review_self_consistency-on-reviews"})
|
||||
res = ej.ensure_rules({"finding_actionability": "id-1",
|
||||
"review_self_consistency": "id-2"}, 1.0)
|
||||
assert res["created"] == []
|
||||
assert sorted(res["skipped"]) == ["finding_actionability", "review_self_consistency"]
|
||||
assert calls == []
|
||||
|
||||
|
||||
def test_ensure_rules_creates_when_missing(monkeypatch):
|
||||
calls = []
|
||||
monkeypatch.setattr(ej.eb, "_call",
|
||||
lambda *a, **k: calls.append(a) or (201, {}))
|
||||
monkeypatch.setattr(ej, "existing_rule_names", lambda: set())
|
||||
res = ej.ensure_rules({"finding_actionability": "id-1"}, 1.0)
|
||||
assert res["created"] == ["finding_actionability"]
|
||||
assert calls[0][0] == "POST"
|
||||
assert calls[0][1] == "/api/public/unstable/evaluation-rules"
|
||||
|
||||
|
||||
# --- judge shape ----------------------------------------------------------
|
||||
|
||||
def test_judges_have_required_keys():
|
||||
for j in ej.JUDGES:
|
||||
assert j["prompt"]
|
||||
assert j["outputDefinition"]["dataType"] in ("NUMERIC", "BOOLEAN", "CATEGORICAL")
|
||||
|
||||
|
||||
def test_default_base_url_points_at_the_thinking_patch_proxy():
|
||||
"""`8802` is the judge-proxy that adds a `signature` to thinking blocks."""
|
||||
assert "8802" in ej.JUDGE_BASE_URL
|
||||
@@ -0,0 +1,200 @@
|
||||
"""Tests for the deterministic review scorers."""
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "..", "..", "pilot"))
|
||||
|
||||
import eval_scores as es # noqa: E402
|
||||
|
||||
|
||||
def f(sev, path="a.py", line=1):
|
||||
return {"severity": sev, "path": path, "line": line, "problem": "p", "fix": ""}
|
||||
|
||||
|
||||
# --- finding_rate ---------------------------------------------------------
|
||||
|
||||
def test_finding_rate_counts_findings():
|
||||
assert es.finding_rate([f("high"), f("low")]) == 2.0
|
||||
|
||||
|
||||
def test_finding_rate_zero_for_silent_review():
|
||||
assert es.finding_rate([]) == 0.0
|
||||
assert es.finding_rate(None) == 0.0
|
||||
|
||||
|
||||
# --- severity_info_ratio --------------------------------------------------
|
||||
|
||||
def test_info_ratio_all_advisory():
|
||||
assert es.severity_info_ratio([f("info"), f("trivial")]) == 1.0
|
||||
|
||||
|
||||
def test_info_ratio_mixed():
|
||||
assert es.severity_info_ratio([f("info"), f("high")]) == 0.5
|
||||
|
||||
|
||||
def test_info_ratio_none_when_no_findings():
|
||||
# Undefined, not zero — zero would read as perfectly calibrated.
|
||||
assert es.severity_info_ratio([]) is None
|
||||
|
||||
|
||||
def test_info_ratio_unknown_severity_treated_as_medium():
|
||||
# Matches _normalize_finding's fallback, so an odd severity is not
|
||||
# silently counted as advisory.
|
||||
assert es.severity_info_ratio([f("bogus")]) == 0.0
|
||||
|
||||
|
||||
# --- severity_max ---------------------------------------------------------
|
||||
|
||||
def test_severity_max_picks_highest():
|
||||
assert es.severity_max([f("info"), f("critical"), f("low")]) == "critical"
|
||||
|
||||
|
||||
def test_severity_max_none_when_silent():
|
||||
assert es.severity_max([]) == "none"
|
||||
|
||||
|
||||
def test_severity_max_case_insensitive():
|
||||
assert es.severity_max([f("HIGH")]) == "high"
|
||||
|
||||
|
||||
# --- dropped_findings -----------------------------------------------------
|
||||
|
||||
def test_dropped_findings_delta():
|
||||
assert es.dropped_findings(5, 2) == 3.0
|
||||
|
||||
|
||||
def test_dropped_findings_never_negative():
|
||||
assert es.dropped_findings(1, 3) == 0.0
|
||||
|
||||
|
||||
def test_dropped_findings_none_when_unknown():
|
||||
assert es.dropped_findings(None, 2) is None
|
||||
|
||||
|
||||
# --- cost_per_finding -----------------------------------------------------
|
||||
|
||||
def test_cost_per_finding_divides():
|
||||
assert es.cost_per_finding(1.0, [f("high"), f("low")]) == 0.5
|
||||
|
||||
|
||||
def test_cost_per_finding_silent_review_divides_by_one():
|
||||
# The run still cost money; attributing all of it to "found nothing" is
|
||||
# the honest reading, and it avoids a division by zero.
|
||||
assert es.cost_per_finding(0.25, []) == 0.25
|
||||
|
||||
|
||||
def test_cost_per_finding_none_when_unpriced():
|
||||
assert es.cost_per_finding(None, [f("high")]) is None
|
||||
|
||||
|
||||
def test_cost_per_finding_none_on_garbage():
|
||||
assert es.cost_per_finding("abc", [f("high")]) is None
|
||||
|
||||
|
||||
# --- build_scores ---------------------------------------------------------
|
||||
|
||||
def _by_name(events):
|
||||
return {e["body"]["name"]: e["body"] for e in events}
|
||||
|
||||
|
||||
def test_build_scores_emits_expected_set():
|
||||
events = es.build_scores(
|
||||
trace_id="t1", findings=[f("high"), f("info")], environment="claude",
|
||||
cost_usd=0.5, dropped_count=2, timestamp="2026-01-01T00:00:00Z",
|
||||
)
|
||||
names = _by_name(events)
|
||||
assert set(names) == {
|
||||
es.FINDING_RATE, es.SEVERITY_INFO_RATIO, es.SEVERITY_MAX,
|
||||
es.DROPPED_FINDINGS, es.COST_PER_FINDING,
|
||||
}
|
||||
assert names[es.FINDING_RATE]["value"] == 2.0
|
||||
assert names[es.SEVERITY_MAX]["value"] == "high"
|
||||
assert names[es.DROPPED_FINDINGS]["value"] == 2.0
|
||||
assert names[es.COST_PER_FINDING]["value"] == 0.25
|
||||
|
||||
|
||||
def test_build_scores_all_events_are_score_create_on_the_trace():
|
||||
events = es.build_scores(
|
||||
trace_id="t9", findings=[f("low")], environment="ollama", cost_usd=1.0,
|
||||
)
|
||||
assert all(e["type"] == "score-create" for e in events)
|
||||
assert all(e["body"]["traceId"] == "t9" for e in events)
|
||||
assert all(e["body"]["environment"] == "ollama" for e in events)
|
||||
|
||||
|
||||
def test_build_scores_omits_undefined_scores():
|
||||
# No cost and no drop count measured -> those scores are absent, not zero.
|
||||
events = es.build_scores(trace_id="t2", findings=[], environment="ollama")
|
||||
names = set(_by_name(events))
|
||||
assert es.COST_PER_FINDING not in names
|
||||
assert es.DROPPED_FINDINGS not in names
|
||||
assert es.SEVERITY_INFO_RATIO not in names
|
||||
assert names == {es.FINDING_RATE, es.SEVERITY_MAX}
|
||||
|
||||
|
||||
def test_build_scores_categorical_value_is_string():
|
||||
events = es.build_scores(trace_id="t3", findings=[f("high")], environment="claude")
|
||||
sev = _by_name(events)[es.SEVERITY_MAX]
|
||||
assert sev["dataType"] == "CATEGORICAL"
|
||||
assert isinstance(sev["value"], str)
|
||||
|
||||
|
||||
def test_build_scores_numeric_values_are_floats():
|
||||
events = es.build_scores(
|
||||
trace_id="t4", findings=[f("high")], environment="claude", cost_usd=1,
|
||||
)
|
||||
for name, body in _by_name(events).items():
|
||||
if body["dataType"] == "NUMERIC":
|
||||
assert isinstance(body["value"], float), name
|
||||
|
||||
|
||||
def test_build_scores_comment_propagates():
|
||||
events = es.build_scores(
|
||||
trace_id="t5", findings=[f("high")], environment="claude",
|
||||
cost_usd=1.0, comment="cost basis: equivalent:claude-sonnet-5",
|
||||
)
|
||||
assert all("equivalent" in e["body"]["comment"] for e in events)
|
||||
|
||||
|
||||
# --- score configs --------------------------------------------------------
|
||||
|
||||
def test_every_emitted_score_has_a_config():
|
||||
configured = {c["name"] for c in es.SCORE_CONFIGS}
|
||||
events = es.build_scores(
|
||||
trace_id="t6", findings=[f("high")], environment="claude",
|
||||
cost_usd=1.0, dropped_count=0,
|
||||
)
|
||||
assert set(_by_name(events)) <= configured
|
||||
|
||||
|
||||
def test_severity_max_config_covers_every_severity_it_can_emit():
|
||||
labels = {c["label"] for c in
|
||||
next(c for c in es.SCORE_CONFIGS if c["name"] == es.SEVERITY_MAX)["categories"]}
|
||||
assert set(es.SEVERITY_RANK) | {"none"} == labels
|
||||
|
||||
|
||||
# --- ingestion envelope ---------------------------------------------------
|
||||
|
||||
def test_every_event_carries_a_timestamp():
|
||||
# Ingestion rejects events without one, and reports the rejection as a
|
||||
# per-event 400 inside an HTTP 207 that reads as success.
|
||||
events = es.build_scores(
|
||||
trace_id="t7", findings=[f("high")], environment="claude", cost_usd=1.0,
|
||||
)
|
||||
assert events
|
||||
assert all(e.get("timestamp") for e in events)
|
||||
|
||||
|
||||
def test_timestamp_defaults_when_caller_omits_it():
|
||||
events = es.build_scores(trace_id="t8", findings=[f("low")], environment="claude")
|
||||
assert all(isinstance(e["timestamp"], str) and e["timestamp"].endswith("Z") for e in events)
|
||||
|
||||
|
||||
def test_explicit_timestamp_is_used():
|
||||
events = es.build_scores(
|
||||
trace_id="t9", findings=[f("low")], environment="claude",
|
||||
timestamp="2026-01-02T03:04:05Z",
|
||||
)
|
||||
assert all(e["timestamp"] == "2026-01-02T03:04:05Z" for e in events)
|
||||
Reference in New Issue
Block a user