From ca9586ca4b80c95068403ff9c2b2d14d64682d5b Mon Sep 17 00:00:00 2001 From: PritamSGB <29685062+PritamSGB@users.noreply.github.com> Date: Thu, 10 Sep 2026 20:59:51 +0530 Subject: [PATCH 1/4] feat(eval): add combined binary metrics and validator config to PII eval output Surfaces overall precision/recall/f1/accuracy (not just per-entity) and the PIIRemover config (entity types, threshold, NLP engine, model, on_fail) used for the run, so metrics.json is self-describing for external readers. --- backend/app/evaluation/pii/run.py | 27 +++++++++++++++++++++------ 1 file changed, 21 insertions(+), 6 deletions(-) diff --git a/backend/app/evaluation/pii/run.py b/backend/app/evaluation/pii/run.py index ccd669c1..9e5c84e9 100644 --- a/backend/app/evaluation/pii/run.py +++ b/backend/app/evaluation/pii/run.py @@ -6,6 +6,7 @@ from app.evaluation.common.helper import ( Profiler, build_evaluation_report, + compute_binary_metrics, write_csv, write_json, ) @@ -18,24 +19,36 @@ validator = PIIRemover() +config = { + "entity_types": validator.entity_types, + "threshold": validator.threshold, + "nlp_engine_type": validator.nlp_engine_type, + "model_name": validator.model_name, + "on_fail": validator.on_fail, + "language": "en", # hardcoded in PIIRemover._validate; not a constructor param +} -def run_pii(text: str) -> str: + +def run_pii(text: str) -> tuple[str, int]: result = validator._validate(text) if isinstance(result, FailResult): - return result.fix_value - return text + return result.fix_value, 1 + return text, 0 with Profiler() as p: - df["anonymized"] = ( - df["source_text"].astype(str).apply(lambda x: p.record(run_pii, x)) - ) + results = df["source_text"].astype(str).apply(lambda x: p.record(run_pii, x)) + df["anonymized"] = results.apply(lambda r: r[0]) + df["pii_detected"] = results.apply(lambda r: r[1]) entity_report = compute_entity_metrics( df["target_text"], df["anonymized"], ) +y_true = (df["label"] == "pii").astype(int) +binary_report = compute_binary_metrics(y_true, df["pii_detected"]) + # ---- Save outputs ---- write_csv(df, OUT_DIR / "predictions.csv") @@ -44,7 +57,9 @@ def run_pii(text: str) -> str: guardrail="pii_remover", num_samples=len(df), profiler=p, + config=config, entity_metrics=entity_report, + binary_metrics=binary_report, ), OUT_DIR / "metrics.json", ) From d2a1effbb7cdcafc5de4268e98016b4b5c6b48d3 Mon Sep 17 00:00:00 2001 From: PritamSGB <29685062+PritamSGB@users.noreply.github.com> Date: Thu, 10 Sep 2026 21:05:25 +0530 Subject: [PATCH 2/4] refactor(eval): rename PII binary_metrics output key to combined_metrics Clarifies that this is PII's overall detection-level score alongside entity_metrics, distinct from the shared compute_binary_metrics() helper used as the sole metric in other eval scripts. --- backend/app/evaluation/pii/run.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/backend/app/evaluation/pii/run.py b/backend/app/evaluation/pii/run.py index 9e5c84e9..97678b76 100644 --- a/backend/app/evaluation/pii/run.py +++ b/backend/app/evaluation/pii/run.py @@ -47,7 +47,7 @@ def run_pii(text: str) -> tuple[str, int]: ) y_true = (df["label"] == "pii").astype(int) -binary_report = compute_binary_metrics(y_true, df["pii_detected"]) +combined_report = compute_binary_metrics(y_true, df["pii_detected"]) # ---- Save outputs ---- write_csv(df, OUT_DIR / "predictions.csv") @@ -59,7 +59,7 @@ def run_pii(text: str) -> tuple[str, int]: profiler=p, config=config, entity_metrics=entity_report, - binary_metrics=binary_report, + combined_metrics=combined_report, ), OUT_DIR / "metrics.json", ) From c91514debfee233e94acae14b27c5c67e0084d1f Mon Sep 17 00:00:00 2001 From: PritamSGB <29685062+PritamSGB@users.noreply.github.com> Date: Thu, 10 Sep 2026 21:23:20 +0530 Subject: [PATCH 3/4] feat(eval): include validator config in gender assumption bias metrics output Mirrors the pii_remover eval script, which already records the config used (entity types, threshold, etc.) alongside the metrics for reproducibility. Co-Authored-By: Claude Sonnet 5 --- backend/app/evaluation/gender_assumption_bias/run.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/backend/app/evaluation/gender_assumption_bias/run.py b/backend/app/evaluation/gender_assumption_bias/run.py index 301f4503..679a1345 100644 --- a/backend/app/evaluation/gender_assumption_bias/run.py +++ b/backend/app/evaluation/gender_assumption_bias/run.py @@ -18,6 +18,12 @@ validator = GenderAssumptionBias() +config = { + "categories": [category.value for category in validator.categories], + "on_fail": validator.on_fail_descriptor, + "num_bias_words_loaded": len(validator.gender_bias_list), +} + with Profiler() as p: df["biased_result"] = ( df["biased input"] @@ -57,6 +63,7 @@ guardrail="gender_assumption_bias", num_samples=len(df) * 2, profiler=p, + config=config, metrics=metrics, ), OUT_DIR / "metrics.json", From 350b81e2b68b394e2c7918b50a9e1eade3abcc6b Mon Sep 17 00:00:00 2001 From: PritamSGB <29685062+PritamSGB@users.noreply.github.com> Date: Thu, 10 Sep 2026 21:36:23 +0530 Subject: [PATCH 4/4] feat(eval): add shared validator config helper, apply config to all evaluations Introduces build_validator_config() in common/helper.py, which reads on_fail from validator.on_fail_descriptor (set by every guardrails Validator base class) and normalizes any extra constructor params passed in (enums -> their .value). PII and gender_assumption_bias now use it instead of hand-rolled dicts, and ban_list, lexical_slur, topic_relevance, and toxicity gain a config block in their metrics.json for the first time. --- backend/app/evaluation/ban_list/run.py | 4 +- backend/app/evaluation/common/helper.py | 28 +++++++++++++ .../evaluation/gender_assumption_bias/run.py | 11 ++--- backend/app/evaluation/lexical_slur/run.py | 9 ++++ backend/app/evaluation/pii/run.py | 17 ++++---- backend/app/evaluation/topic_relevance/run.py | 16 +++---- backend/app/evaluation/toxicity/run.py | 42 +++++++++++++------ 7 files changed, 94 insertions(+), 33 deletions(-) diff --git a/backend/app/evaluation/ban_list/run.py b/backend/app/evaluation/ban_list/run.py index 2d630991..03620da8 100644 --- a/backend/app/evaluation/ban_list/run.py +++ b/backend/app/evaluation/ban_list/run.py @@ -6,6 +6,7 @@ from app.evaluation.common.helper import ( build_evaluation_report, + build_validator_config, Profiler, compute_binary_metrics, write_csv, @@ -35,6 +36,7 @@ def run_evaluation(config: dict): dataset = pd.read_csv(DATASET_PATH) validator = BanList(banned_words=banned_words) + validator_config = build_validator_config(validator, banned_words=banned_words) def run_ban_list(text: str) -> tuple[str, int]: """Validate a single text and return the (possibly redacted) text and a binary prediction label.""" @@ -81,7 +83,7 @@ def run_ban_list(text: str) -> tuple[str, int]: guardrail="ban_list", num_samples=len(dataset), profiler=p, - banned_words=banned_words, + config=validator_config, dataset=str(DATASET_PATH.name), metrics=metrics, ), diff --git a/backend/app/evaluation/common/helper.py b/backend/app/evaluation/common/helper.py index a72c83c1..b24c6fa6 100644 --- a/backend/app/evaluation/common/helper.py +++ b/backend/app/evaluation/common/helper.py @@ -1,3 +1,4 @@ +from enum import Enum from pathlib import Path from typing import Any import json @@ -52,6 +53,33 @@ def build_evaluation_report( } +def _to_jsonable(value: Any) -> Any: + """Recursively normalize enums (and lists/dicts containing them) to JSON-safe values.""" + if isinstance(value, Enum): + return value.value + if isinstance(value, (list, tuple)): + return [_to_jsonable(v) for v in value] + if isinstance(value, dict): + return {k: _to_jsonable(v) for k, v in value.items()} + return value + + +def build_validator_config(validator: Any, **fields: Any) -> dict[str, Any]: + """ + Build the `config` block for an evaluation's metrics.json. + + `on_fail` is read from `validator.on_fail_descriptor`, which every + guardrails Validator subclass sets in its base __init__, so it's always + available without each evaluation script re-deriving it. Any + validator-specific constructor params (entity_types, threshold, + categories, ...) are passed in as keyword args and normalized the same + way (e.g. enums -> their .value). + """ + config = {"on_fail": _to_jsonable(validator.on_fail_descriptor)} + config.update({key: _to_jsonable(value) for key, value in fields.items()}) + return config + + def compute_binary_metrics(y_true, y_pred): tp = sum((yt == 1 and yp == 1) for yt, yp in zip(y_true, y_pred, strict=True)) tn = sum((yt == 0 and yp == 0) for yt, yp in zip(y_true, y_pred, strict=True)) diff --git a/backend/app/evaluation/gender_assumption_bias/run.py b/backend/app/evaluation/gender_assumption_bias/run.py index 679a1345..9d9ae11d 100644 --- a/backend/app/evaluation/gender_assumption_bias/run.py +++ b/backend/app/evaluation/gender_assumption_bias/run.py @@ -5,6 +5,7 @@ from app.core.validators.gender_assumption_bias import GenderAssumptionBias from app.evaluation.common.helper import ( build_evaluation_report, + build_validator_config, compute_binary_metrics, Profiler, write_csv, @@ -18,11 +19,11 @@ validator = GenderAssumptionBias() -config = { - "categories": [category.value for category in validator.categories], - "on_fail": validator.on_fail_descriptor, - "num_bias_words_loaded": len(validator.gender_bias_list), -} +config = build_validator_config( + validator, + categories=validator.categories, + num_bias_words_loaded=len(validator.gender_bias_list), +) with Profiler() as p: df["biased_result"] = ( diff --git a/backend/app/evaluation/lexical_slur/run.py b/backend/app/evaluation/lexical_slur/run.py index 9f1808d5..7cd74a01 100644 --- a/backend/app/evaluation/lexical_slur/run.py +++ b/backend/app/evaluation/lexical_slur/run.py @@ -5,6 +5,7 @@ from app.core.validators.lexical_slur import LexicalSlur from app.evaluation.common.helper import ( build_evaluation_report, + build_validator_config, Profiler, compute_binary_metrics, write_csv, @@ -18,6 +19,13 @@ validator = LexicalSlur() +config = build_validator_config( + validator, + severity=validator.severity, + languages=validator.languages, + num_slurs_loaded=len(validator.slur_list), +) + with Profiler() as p: df["result"] = ( df["commentText"] @@ -38,6 +46,7 @@ guardrail="lexical_slur", num_samples=len(df), profiler=p, + config=config, metrics=metrics, ), OUT_DIR / "metrics.json", diff --git a/backend/app/evaluation/pii/run.py b/backend/app/evaluation/pii/run.py index 97678b76..fd989038 100644 --- a/backend/app/evaluation/pii/run.py +++ b/backend/app/evaluation/pii/run.py @@ -6,6 +6,7 @@ from app.evaluation.common.helper import ( Profiler, build_evaluation_report, + build_validator_config, compute_binary_metrics, write_csv, write_json, @@ -19,14 +20,14 @@ validator = PIIRemover() -config = { - "entity_types": validator.entity_types, - "threshold": validator.threshold, - "nlp_engine_type": validator.nlp_engine_type, - "model_name": validator.model_name, - "on_fail": validator.on_fail, - "language": "en", # hardcoded in PIIRemover._validate; not a constructor param -} +config = build_validator_config( + validator, + entity_types=validator.entity_types, + threshold=validator.threshold, + nlp_engine_type=validator.nlp_engine_type, + model_name=validator.model_name, + language="en", # hardcoded in PIIRemover._validate; not a constructor param +) def run_pii(text: str) -> tuple[str, int]: diff --git a/backend/app/evaluation/topic_relevance/run.py b/backend/app/evaluation/topic_relevance/run.py index d450e117..52fa1b5f 100644 --- a/backend/app/evaluation/topic_relevance/run.py +++ b/backend/app/evaluation/topic_relevance/run.py @@ -11,6 +11,7 @@ from app.evaluation.common.helper import ( Profiler, build_evaluation_report, + build_validator_config, compute_binary_metrics, write_csv, write_json, @@ -42,9 +43,9 @@ prompt_schema_version=1, llm_callable=settings.DEFAULT_LLM_CALLABLE, ), - "report_extra": { - "llm_callable": settings.DEFAULT_LLM_CALLABLE, - "prompt_schema_version": 1, + "config_fields": lambda v: { + "llm_callable": v.llm_callable, + "prompt_schema_version": v.prompt_schema_version, }, }, { @@ -55,9 +56,9 @@ llm_callable=settings.DEFAULT_LLM_CALLABLE, threshold=settings.TOPIC_RELEVANCE_LLM_THRESHOLD, ), - "report_extra": { - "llm_callable": settings.DEFAULT_LLM_CALLABLE, - "threshold": settings.TOPIC_RELEVANCE_LLM_THRESHOLD, + "config_fields": lambda v: { + "llm_callable": v.llm_callable, + "threshold": v.threshold, }, }, ] @@ -73,6 +74,7 @@ def run_evaluation(dataset: dict, backend: dict) -> None: df = pd.read_csv(dataset_path) validator = backend["build"](topic_config) + config = build_validator_config(validator, **backend["config_fields"](validator)) normalized_df = pd.DataFrame( { @@ -115,7 +117,7 @@ def run_evaluation(dataset: dict, backend: dict) -> None: num_samples=len(normalized_df), profiler=p, dataset=str(dataset_path), - **backend["report_extra"], + config=config, metrics=metrics, ), out_dir / f"{domain}-metrics.json", diff --git a/backend/app/evaluation/toxicity/run.py b/backend/app/evaluation/toxicity/run.py index faf537cb..619c47a9 100644 --- a/backend/app/evaluation/toxicity/run.py +++ b/backend/app/evaluation/toxicity/run.py @@ -7,6 +7,7 @@ from app.evaluation.common.helper import ( build_evaluation_report, + build_validator_config, compute_binary_metrics, Profiler, write_csv, @@ -32,16 +33,31 @@ } VALIDATORS = { - "llamaguard_7b": lambda: LlamaGuard7B(on_fail="noop"), - "nsfw_text": lambda: NSFWText( - threshold=0.8, - validation_method="sentence", - device="cpu", - model_name="textdetox/xlmr-large-toxicity-classifier", - on_fail="noop", - use_local=True, - ), - "profanity_free": lambda: ProfanityFree(on_fail="noop"), + "llamaguard_7b": { + "build": lambda: LlamaGuard7B(on_fail="noop"), + "config_fields": {}, + }, + "nsfw_text": { + "build": lambda: NSFWText( + threshold=0.8, + validation_method="sentence", + device="cpu", + model_name="textdetox/xlmr-large-toxicity-classifier", + on_fail="noop", + use_local=True, + ), + "config_fields": { + "threshold": 0.8, + "validation_method": "sentence", + "device": "cpu", + "model_name": "textdetox/xlmr-large-toxicity-classifier", + "use_local": True, + }, + }, + "profanity_free": { + "build": lambda: ProfanityFree(on_fail="noop"), + "config_fields": {}, + }, } @@ -67,9 +83,10 @@ def run_dataset(dataset_name: str, dataset_cfg: dict): all_metrics = {} - for validator_name, build_fn in VALIDATORS.items(): + for validator_name, spec in VALIDATORS.items(): print(f" Running {validator_name} on {dataset_name}...") - validator = build_fn() + validator = spec["build"]() + config = build_validator_config(validator, **spec["config_fields"]) with Profiler() as p: df[f"{validator_name}_result"] = df[text_col].apply( @@ -89,6 +106,7 @@ def run_dataset(dataset_name: str, dataset_cfg: dict): dataset=dataset_name, num_samples=len(df), profiler=p, + config=config, metrics=metrics, )