Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions backend/app/evaluation/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -259,6 +259,14 @@ The predictions CSV includes `scope_score` (the LLM-assigned score) and `error_m
python3 app/evaluation/topic_relevance/run.py
```

Use `--backend` to run only one of the two validators (`topic_relevance` or `topic_relevance_llm`) instead of both:

```bash
python3 app/evaluation/topic_relevance/run.py --backend topic_relevance_llm
```

`topic_relevance` (the `LLMCritic`-based validator) is only imported when that backend actually runs, so `--backend topic_relevance_llm` works without installing the `llm_critic` hub validator.

> **Note:** Requires `OPENAI_API_KEY` to be set. Uses `gpt-4o-mini` by default (`DEFAULT_CONFIG` in the script).

---
Expand Down
4 changes: 3 additions & 1 deletion backend/app/evaluation/ban_list/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@

from app.evaluation.common.helper import (
build_evaluation_report,
build_validator_config,
Profiler,
compute_binary_metrics,
write_csv,
Expand Down Expand Up @@ -35,6 +36,7 @@ def run_evaluation(config: dict):
dataset = pd.read_csv(DATASET_PATH)

validator = BanList(banned_words=banned_words)
validator_config = build_validator_config(validator, banned_words=banned_words)

def run_ban_list(text: str) -> tuple[str, int]:
"""Validate a single text and return the (possibly redacted) text and a binary prediction label."""
Expand Down Expand Up @@ -81,7 +83,7 @@ def run_ban_list(text: str) -> tuple[str, int]:
guardrail="ban_list",
num_samples=len(dataset),
profiler=p,
banned_words=banned_words,
config=validator_config,
dataset=str(DATASET_PATH.name),
metrics=metrics,
),
Expand Down
91 changes: 82 additions & 9 deletions backend/app/evaluation/common/helper.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,4 @@
from enum import Enum
from pathlib import Path
from typing import Any
import json
Expand Down Expand Up @@ -52,12 +53,34 @@ def build_evaluation_report(
}


def compute_binary_metrics(y_true, y_pred):
tp = sum((yt == 1 and yp == 1) for yt, yp in zip(y_true, y_pred, strict=True))
tn = sum((yt == 0 and yp == 0) for yt, yp in zip(y_true, y_pred, strict=True))
fp = sum((yt == 0 and yp == 1) for yt, yp in zip(y_true, y_pred, strict=True))
fn = sum((yt == 1 and yp == 0) for yt, yp in zip(y_true, y_pred, strict=True))
def _to_jsonable(value: Any) -> Any:
"""Recursively normalize enums (and lists/dicts containing them) to JSON-safe values."""
if isinstance(value, Enum):
return value.value
if isinstance(value, (list, tuple)):
return [_to_jsonable(v) for v in value]
if isinstance(value, dict):
return {k: _to_jsonable(v) for k, v in value.items()}
return value


def build_validator_config(validator: Any, **fields: Any) -> dict[str, Any]:
"""
Build the `config` block for an evaluation's metrics.json.

`on_fail` is read from `validator.on_fail_descriptor`, which every
guardrails Validator subclass sets in its base __init__, so it's always
available without each evaluation script re-deriving it. Any
validator-specific constructor params (entity_types, threshold,
categories, ...) are passed in as keyword args and normalized the same
way (e.g. enums -> their .value).
"""
config = {"on_fail": _to_jsonable(validator.on_fail_descriptor)}
config.update({key: _to_jsonable(value) for key, value in fields.items()})
return config


def _confusion_rates(tp: int, tn: int, fp: int, fn: int) -> dict[str, float]:
precision = tp / (tp + fp) if tp + fp else 0.0
recall = tp / (tp + fn) if tp + fn else 0.0
f1 = 2 * precision * recall / (precision + recall) if precision + recall else 0.0
Expand All @@ -66,17 +89,67 @@ def compute_binary_metrics(y_true, y_pred):
accuracy = (tp + tn) / total if total else 0.0

return {
"true_positive": tp,
"true_negative": tn,
"false_positive": fp,
"false_negative": fn,
"accuracy": round(accuracy, 2),
"precision": round(precision, 2),
"recall": round(recall, 2),
"f1": round(f1, 2),
}


def compute_binary_metrics(y_true, y_pred):
tp = sum((yt == 1 and yp == 1) for yt, yp in zip(y_true, y_pred, strict=True))
tn = sum((yt == 0 and yp == 0) for yt, yp in zip(y_true, y_pred, strict=True))
fp = sum((yt == 0 and yp == 1) for yt, yp in zip(y_true, y_pred, strict=True))
fn = sum((yt == 1 and yp == 0) for yt, yp in zip(y_true, y_pred, strict=True))

return {
"true_positive": tp,
"true_negative": tn,
"false_positive": fp,
"false_negative": fn,
**_confusion_rates(tp, tn, fp, fn),
}


def combine_binary_metrics(
metrics_list: list[dict[str, Any]], labels: list[str] | None = None
) -> dict[str, Any]:
"""
Combine multiple compute_binary_metrics() results (e.g. one per dataset or
domain) into a single aggregate by summing their confusion-matrix counts
and re-deriving accuracy/precision/recall/f1 from the totals.

Pure arithmetic over already-computed metrics — does not re-run any
validator, so combining results costs no extra API/LLM calls.

`labels` (one per entry in metrics_list) namespaces category_metrics keys
(e.g. "education/Misc") so identically-named categories from different
sources don't collide; defaults to the entry's index if omitted.
"""
tp = sum(m["true_positive"] for m in metrics_list)
tn = sum(m["true_negative"] for m in metrics_list)
fp = sum(m["false_positive"] for m in metrics_list)
fn = sum(m["false_negative"] for m in metrics_list)

combined: dict[str, Any] = {
"true_positive": tp,
"true_negative": tn,
"false_positive": fp,
"false_negative": fn,
**_confusion_rates(tp, tn, fp, fn),
}

category_metrics = {}
for i, m in enumerate(metrics_list):
label = labels[i] if labels else str(i)
for category, category_metric in m.get("category_metrics", {}).items():
category_metrics[f"{label}/{category}"] = category_metric
if category_metrics:
combined["category_metrics"] = category_metrics

return combined


class Profiler:
def __enter__(self):
self.latencies = []
Expand Down
8 changes: 8 additions & 0 deletions backend/app/evaluation/gender_assumption_bias/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@
from app.core.validators.gender_assumption_bias import GenderAssumptionBias
from app.evaluation.common.helper import (
build_evaluation_report,
build_validator_config,
compute_binary_metrics,
Profiler,
write_csv,
Expand All @@ -18,6 +19,12 @@

validator = GenderAssumptionBias()

config = build_validator_config(
validator,
categories=validator.categories,
num_bias_words_loaded=len(validator.gender_bias_list),
)

with Profiler() as p:
df["biased_result"] = (
df["biased input"]
Expand Down Expand Up @@ -57,6 +64,7 @@
guardrail="gender_assumption_bias",
num_samples=len(df) * 2,
profiler=p,
config=config,
metrics=metrics,
),
OUT_DIR / "metrics.json",
Expand Down
9 changes: 9 additions & 0 deletions backend/app/evaluation/lexical_slur/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@
from app.core.validators.lexical_slur import LexicalSlur
from app.evaluation.common.helper import (
build_evaluation_report,
build_validator_config,
Profiler,
compute_binary_metrics,
write_csv,
Expand All @@ -18,6 +19,13 @@

validator = LexicalSlur()

config = build_validator_config(
validator,
severity=validator.severity,
languages=validator.languages,
num_slurs_loaded=len(validator.slur_list),
)

with Profiler() as p:
df["result"] = (
df["commentText"]
Expand All @@ -38,6 +46,7 @@
guardrail="lexical_slur",
num_samples=len(df),
profiler=p,
config=config,
metrics=metrics,
),
OUT_DIR / "metrics.json",
Expand Down
28 changes: 22 additions & 6 deletions backend/app/evaluation/pii/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,8 @@
from app.evaluation.common.helper import (
Profiler,
build_evaluation_report,
build_validator_config,
compute_binary_metrics,
write_csv,
write_json,
)
Expand All @@ -18,24 +20,36 @@

validator = PIIRemover()

config = build_validator_config(
validator,
entity_types=validator.entity_types,
threshold=validator.threshold,
nlp_engine_type=validator.nlp_engine_type,
model_name=validator.model_name,
language="en", # hardcoded in PIIRemover._validate; not a constructor param
)


def run_pii(text: str) -> str:
def run_pii(text: str) -> tuple[str, int]:
result = validator._validate(text)
if isinstance(result, FailResult):
return result.fix_value
return text
return result.fix_value, 1
return text, 0


with Profiler() as p:
df["anonymized"] = (
df["source_text"].astype(str).apply(lambda x: p.record(run_pii, x))
)
results = df["source_text"].astype(str).apply(lambda x: p.record(run_pii, x))
df["anonymized"] = results.apply(lambda r: r[0])
df["pii_detected"] = results.apply(lambda r: r[1])

entity_report = compute_entity_metrics(
df["target_text"],
df["anonymized"],
)

y_true = (df["label"] == "pii").astype(int)
combined_report = compute_binary_metrics(y_true, df["pii_detected"])

# ---- Save outputs ----
write_csv(df, OUT_DIR / "predictions.csv")

Expand All @@ -44,7 +58,9 @@ def run_pii(text: str) -> str:
guardrail="pii_remover",
num_samples=len(df),
profiler=p,
config=config,
entity_metrics=entity_report,
combined_metrics=combined_report,
),
OUT_DIR / "metrics.json",
)
Loading
Loading