Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion backend/app/evaluation/ban_list/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@

from app.evaluation.common.helper import (
build_evaluation_report,
build_validator_config,
Profiler,
compute_binary_metrics,
write_csv,
Expand Down Expand Up @@ -35,6 +36,7 @@ def run_evaluation(config: dict):
dataset = pd.read_csv(DATASET_PATH)

validator = BanList(banned_words=banned_words)
validator_config = build_validator_config(validator, banned_words=banned_words)

def run_ban_list(text: str) -> tuple[str, int]:
"""Validate a single text and return the (possibly redacted) text and a binary prediction label."""
Expand Down Expand Up @@ -81,7 +83,7 @@ def run_ban_list(text: str) -> tuple[str, int]:
guardrail="ban_list",
num_samples=len(dataset),
profiler=p,
banned_words=banned_words,
config=validator_config,
dataset=str(DATASET_PATH.name),
metrics=metrics,
),
Expand Down
28 changes: 28 additions & 0 deletions backend/app/evaluation/common/helper.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,4 @@
from enum import Enum
from pathlib import Path
from typing import Any
import json
Expand Down Expand Up @@ -52,6 +53,33 @@ def build_evaluation_report(
}


def _to_jsonable(value: Any) -> Any:
"""Recursively normalize enums (and lists/dicts containing them) to JSON-safe values."""
if isinstance(value, Enum):
return value.value
if isinstance(value, (list, tuple)):
return [_to_jsonable(v) for v in value]
if isinstance(value, dict):
return {k: _to_jsonable(v) for k, v in value.items()}
return value


def build_validator_config(validator: Any, **fields: Any) -> dict[str, Any]:
"""
Build the `config` block for an evaluation's metrics.json.

`on_fail` is read from `validator.on_fail_descriptor`, which every
guardrails Validator subclass sets in its base __init__, so it's always
available without each evaluation script re-deriving it. Any
validator-specific constructor params (entity_types, threshold,
categories, ...) are passed in as keyword args and normalized the same
way (e.g. enums -> their .value).
"""
config = {"on_fail": _to_jsonable(validator.on_fail_descriptor)}
config.update({key: _to_jsonable(value) for key, value in fields.items()})
return config


def compute_binary_metrics(y_true, y_pred):
tp = sum((yt == 1 and yp == 1) for yt, yp in zip(y_true, y_pred, strict=True))
tn = sum((yt == 0 and yp == 0) for yt, yp in zip(y_true, y_pred, strict=True))
Expand Down
8 changes: 8 additions & 0 deletions backend/app/evaluation/gender_assumption_bias/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@
from app.core.validators.gender_assumption_bias import GenderAssumptionBias
from app.evaluation.common.helper import (
build_evaluation_report,
build_validator_config,
compute_binary_metrics,
Profiler,
write_csv,
Expand All @@ -18,6 +19,12 @@

validator = GenderAssumptionBias()

config = build_validator_config(
validator,
categories=validator.categories,
num_bias_words_loaded=len(validator.gender_bias_list),
)

with Profiler() as p:
df["biased_result"] = (
df["biased input"]
Expand Down Expand Up @@ -57,6 +64,7 @@
guardrail="gender_assumption_bias",
num_samples=len(df) * 2,
profiler=p,
config=config,
metrics=metrics,
),
OUT_DIR / "metrics.json",
Expand Down
9 changes: 9 additions & 0 deletions backend/app/evaluation/lexical_slur/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@
from app.core.validators.lexical_slur import LexicalSlur
from app.evaluation.common.helper import (
build_evaluation_report,
build_validator_config,
Profiler,
compute_binary_metrics,
write_csv,
Expand All @@ -18,6 +19,13 @@

validator = LexicalSlur()

config = build_validator_config(
validator,
severity=validator.severity,
languages=validator.languages,
num_slurs_loaded=len(validator.slur_list),
)

with Profiler() as p:
df["result"] = (
df["commentText"]
Expand All @@ -38,6 +46,7 @@
guardrail="lexical_slur",
num_samples=len(df),
profiler=p,
config=config,
metrics=metrics,
),
OUT_DIR / "metrics.json",
Expand Down
28 changes: 22 additions & 6 deletions backend/app/evaluation/pii/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,8 @@
from app.evaluation.common.helper import (
Profiler,
build_evaluation_report,
build_validator_config,
compute_binary_metrics,
write_csv,
write_json,
)
Expand All @@ -18,24 +20,36 @@

validator = PIIRemover()

config = build_validator_config(
validator,
entity_types=validator.entity_types,
threshold=validator.threshold,
nlp_engine_type=validator.nlp_engine_type,
model_name=validator.model_name,
language="en", # hardcoded in PIIRemover._validate; not a constructor param
)


def run_pii(text: str) -> str:
def run_pii(text: str) -> tuple[str, int]:
result = validator._validate(text)
if isinstance(result, FailResult):
return result.fix_value
return text
return result.fix_value, 1
return text, 0


with Profiler() as p:
df["anonymized"] = (
df["source_text"].astype(str).apply(lambda x: p.record(run_pii, x))
)
results = df["source_text"].astype(str).apply(lambda x: p.record(run_pii, x))
df["anonymized"] = results.apply(lambda r: r[0])
df["pii_detected"] = results.apply(lambda r: r[1])

entity_report = compute_entity_metrics(
df["target_text"],
df["anonymized"],
)

y_true = (df["label"] == "pii").astype(int)
combined_report = compute_binary_metrics(y_true, df["pii_detected"])

# ---- Save outputs ----
write_csv(df, OUT_DIR / "predictions.csv")

Expand All @@ -44,7 +58,9 @@ def run_pii(text: str) -> str:
guardrail="pii_remover",
num_samples=len(df),
profiler=p,
config=config,
entity_metrics=entity_report,
combined_metrics=combined_report,
),
OUT_DIR / "metrics.json",
)
16 changes: 9 additions & 7 deletions backend/app/evaluation/topic_relevance/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
from app.evaluation.common.helper import (
Profiler,
build_evaluation_report,
build_validator_config,
compute_binary_metrics,
write_csv,
write_json,
Expand Down Expand Up @@ -42,9 +43,9 @@
prompt_schema_version=1,
llm_callable=settings.DEFAULT_LLM_CALLABLE,
),
"report_extra": {
"llm_callable": settings.DEFAULT_LLM_CALLABLE,
"prompt_schema_version": 1,
"config_fields": lambda v: {
"llm_callable": v.llm_callable,
"prompt_schema_version": v.prompt_schema_version,
},
},
{
Expand All @@ -55,9 +56,9 @@
llm_callable=settings.DEFAULT_LLM_CALLABLE,
threshold=settings.TOPIC_RELEVANCE_LLM_THRESHOLD,
),
"report_extra": {
"llm_callable": settings.DEFAULT_LLM_CALLABLE,
"threshold": settings.TOPIC_RELEVANCE_LLM_THRESHOLD,
"config_fields": lambda v: {
"llm_callable": v.llm_callable,
"threshold": v.threshold,
},
},
]
Expand All @@ -73,6 +74,7 @@ def run_evaluation(dataset: dict, backend: dict) -> None:

df = pd.read_csv(dataset_path)
validator = backend["build"](topic_config)
config = build_validator_config(validator, **backend["config_fields"](validator))

normalized_df = pd.DataFrame(
{
Expand Down Expand Up @@ -115,7 +117,7 @@ def run_evaluation(dataset: dict, backend: dict) -> None:
num_samples=len(normalized_df),
profiler=p,
dataset=str(dataset_path),
**backend["report_extra"],
config=config,
metrics=metrics,
),
out_dir / f"{domain}-metrics.json",
Expand Down
42 changes: 30 additions & 12 deletions backend/app/evaluation/toxicity/run.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@

from app.evaluation.common.helper import (
build_evaluation_report,
build_validator_config,
compute_binary_metrics,
Profiler,
write_csv,
Expand All @@ -32,16 +33,31 @@
}

VALIDATORS = {
"llamaguard_7b": lambda: LlamaGuard7B(on_fail="noop"),
"nsfw_text": lambda: NSFWText(
threshold=0.8,
validation_method="sentence",
device="cpu",
model_name="textdetox/xlmr-large-toxicity-classifier",
on_fail="noop",
use_local=True,
),
"profanity_free": lambda: ProfanityFree(on_fail="noop"),
"llamaguard_7b": {
"build": lambda: LlamaGuard7B(on_fail="noop"),
"config_fields": {},
},
"nsfw_text": {
"build": lambda: NSFWText(
threshold=0.8,
validation_method="sentence",
device="cpu",
model_name="textdetox/xlmr-large-toxicity-classifier",
on_fail="noop",
use_local=True,
),
"config_fields": {
"threshold": 0.8,
"validation_method": "sentence",
"device": "cpu",
"model_name": "textdetox/xlmr-large-toxicity-classifier",
"use_local": True,
},
},
"profanity_free": {
"build": lambda: ProfanityFree(on_fail="noop"),
"config_fields": {},
},
}


Expand All @@ -67,9 +83,10 @@ def run_dataset(dataset_name: str, dataset_cfg: dict):

all_metrics = {}

for validator_name, build_fn in VALIDATORS.items():
for validator_name, spec in VALIDATORS.items():
print(f" Running {validator_name} on {dataset_name}...")
validator = build_fn()
validator = spec["build"]()
config = build_validator_config(validator, **spec["config_fields"])

with Profiler() as p:
df[f"{validator_name}_result"] = df[text_col].apply(
Expand All @@ -89,6 +106,7 @@ def run_dataset(dataset_name: str, dataset_cfg: dict):
dataset=dataset_name,
num_samples=len(df),
profiler=p,
config=config,
metrics=metrics,
)

Expand Down
Loading