#!/usr/bin/env python3 """pragent pilot — deterministic review scorers. Four numbers computed from a review that already happened, shipped to Langfuse as scores on the review's trace. All are derived from data the reviewer already has in hand: no LLM judge, no ground truth, no extra token spend. Why these four and not `helpfulness`/`quality` ---------------------------------------------- They come from what the recorded reviews actually did, not from a generic eval checklist: * `severity_info_ratio` — of the findings ever posted to a PR, effectively all landed at `info`. Either the model will not commit to a severity or the per-repo `severity_threshold` is filtering the rest out. Trending the ratio per model says which. * `finding_rate` — most reviews post nothing at all. Silence on clean code is the goal; silence because the run degraded is a failure. Same output, two causes, and only the rate over time separates them. * `dropped_findings` — `ai_review.parse_findings` discards any finding whose `path`/`line` is unusable. That happens silently, so a model that emits ten findings at invalid locations is indistinguishable from one that found nothing. This is the only signal here that measures the *model's* output rather than the review's. * `cost_per_finding` — the equivalent-cost number is already trended per review; per finding is what actually compares two models, since a cheaper model that finds nothing is not cheaper. None of these say whether a finding was *correct*. That needs labels, and the labels come from `feedback_scores.py` once maintainers start reacting to review comments. Read these as behavioural drift detectors, not as accuracy. Fail-open, like every other telemetry path here: a scorer that raises returns no score rather than failing the review. """ from __future__ import annotations import uuid from datetime import datetime, timezone # Mirrors ai_review.SEVERITY_RANK. Duplicated rather than imported because this # module is also run standalone (backfill) where ai_review's import side effects # are unwanted. SEVERITY_RANK = {"info": -1, "trivial": 0, "low": 1, "medium": 2, "high": 3, "critical": 4} # Findings at or below this rank are "the model declined to commit". `trivial` # and `info` are advisory by the reviewer's own prompt contract. _ADVISORY_MAX_RANK = 0 # Score names. Named for what is measured, not for the mechanism producing it — # these land on every trace and become the axis of every chart. FINDING_RATE = "finding_rate" SEVERITY_INFO_RATIO = "severity_info_ratio" SEVERITY_MAX = "severity_max" DROPPED_FINDINGS = "dropped_findings" COST_PER_FINDING = "cost_per_finding" def _sev(f: dict) -> str: return str(f.get("severity") or "medium").strip().lower() def finding_rate(findings: list[dict] | None) -> float: """How many findings this review posted. 0.0 is the restraint case.""" return float(len(findings or [])) def severity_info_ratio(findings: list[dict] | None) -> float | None: """Share of findings the model rated advisory (`info`/`trivial`). `None` for a review with no findings — a ratio over an empty set is not 0, it is undefined, and charting it as 0 would read as "perfectly calibrated". """ fs = findings or [] if not fs: return None advisory = sum(1 for f in fs if SEVERITY_RANK.get(_sev(f), 2) <= _ADVISORY_MAX_RANK) return round(advisory / len(fs), 4) def severity_max(findings: list[dict] | None) -> str: """Highest severity present, or `none` when the review was silent. Categorical on purpose: the useful question is "did this review ever surface something serious", and an average of severity ranks answers nothing. """ fs = findings or [] if not fs: return "none" top = max(fs, key=lambda f: SEVERITY_RANK.get(_sev(f), 2)) sev = _sev(top) return sev if sev in SEVERITY_RANK else "medium" def dropped_findings(raw_count: int | None, kept_count: int | None) -> float | None: """Findings the model emitted that the parser could not use. `raw_count` is what came back in the JSON; `kept_count` is what survived `_normalize_finding`. `None` when the caller could not determine the raw count — better no score than a fabricated zero. """ if raw_count is None or kept_count is None: return None return float(max(0, int(raw_count) - int(kept_count))) def cost_per_finding(cost_usd: float | None, findings: list[dict] | None) -> float | None: """Equivalent USD spent per finding posted. `None` when nothing could be priced. A silent review divides by one, not by zero: the run still cost money, and attributing that whole cost to "found nothing" is the honest reading. """ if cost_usd is None: return None try: c = float(cost_usd) except (TypeError, ValueError): return None return round(c / max(1, len(findings or [])), 6) def build_scores( *, trace_id: str, findings: list[dict] | None, environment: str, cost_usd: float | None = None, dropped_count: float | None = None, timestamp: str | None = None, comment: str = "", ) -> list[dict]: """The `score-create` ingestion events for one review. `dropped_count` must be measured at parse time, not here: by the time `findings` reaches this function the per-repo config has already filtered it by severity threshold and `max_findings`, and those drops are the config working as intended, not the model emitting garbage. Returns [] rather than raising if something is unscoreable — scores are telemetry and must never cost a review. """ # The ingestion envelope requires a timestamp on every event; omitting it # gets the whole batch rejected with an HTTP 207 whose per-event 400s are # easy to mistake for success. ts = timestamp or datetime.now(timezone.utc).isoformat().replace("+00:00", "Z") out: list[dict] = [] def add(name: str, value, data_type: str) -> None: if value is None: return body = { "id": str(uuid.uuid4()), "traceId": trace_id, "name": name, "dataType": data_type, "environment": environment, } if data_type == "CATEGORICAL": body["value"] = str(value) else: body["value"] = float(value) if comment: body["comment"] = comment out.append( { "id": str(uuid.uuid4()), "type": "score-create", "timestamp": ts, "body": body, } ) try: add(FINDING_RATE, finding_rate(findings), "NUMERIC") add(SEVERITY_INFO_RATIO, severity_info_ratio(findings), "NUMERIC") add(SEVERITY_MAX, severity_max(findings), "CATEGORICAL") add(DROPPED_FINDINGS, dropped_count, "NUMERIC") add(COST_PER_FINDING, cost_per_finding(cost_usd, findings), "NUMERIC") except Exception: # pragma: no cover - defensive return out return out # --------------------------------------------------------------------------- # Score configs — the schema these scores must comply with # --------------------------------------------------------------------------- # Registered once per project via `eval_bootstrap.py`. Without configs the # scores still ingest, but nothing constrains a future scorer from writing # `severity_max="HIGH"` next to today's `"high"` and silently splitting the # series in two. SCORE_CONFIGS = [ { "name": FINDING_RATE, "dataType": "NUMERIC", "minValue": 0, "description": "Findings posted by one review. 0 = the reviewer stayed silent.", }, { "name": SEVERITY_INFO_RATIO, "dataType": "NUMERIC", "minValue": 0, "maxValue": 1, "description": "Share of a review's findings rated info/trivial. High = the model is not committing to a severity.", }, { "name": SEVERITY_MAX, "dataType": "CATEGORICAL", "categories": [ {"label": "none", "value": 0}, {"label": "info", "value": 1}, {"label": "trivial", "value": 2}, {"label": "low", "value": 3}, {"label": "medium", "value": 4}, {"label": "high", "value": 5}, {"label": "critical", "value": 6}, ], "description": "Highest severity surfaced by one review; 'none' when it posted nothing.", }, { "name": DROPPED_FINDINGS, "dataType": "NUMERIC", "minValue": 0, "description": "Findings the model emitted that the parser rejected for an unusable path/line.", }, { "name": COST_PER_FINDING, "dataType": "NUMERIC", "minValue": 0, "description": "Equivalent USD per finding posted. Silent reviews divide by 1, not 0.", }, ]