1 Commits

Author SHA1 Message Date
Philipp Emanuel Weidmann 95dda4c4db feat: add benchmark scorer (#444)
* fix: improve print output of scorers

* feat: add benchmark scorer
2026-09-03 17:49:46 +05:30
11 changed files with 113 additions and 40 deletions
+3
View File
@@ -157,6 +157,9 @@ residual_plot_color = "darkorange"
# Plugin-specific settings live in a top-level TOML table.
# For scorer plugins, use: `[scorer.<ClassName>]` (and optionally `[scorer.<ClassName>_<instance_name>]` for instance-related config).
[scorer.KeywordRate]
# Name that describes what the configured keyword rate measures.
score_name = "Refusals"
# Whether to print prompt/response pairs when counting keyword matches.
print_responses = false
+2
View File
@@ -20,6 +20,8 @@ residual_plot_label = "Humorous prompts"
residual_plot_color = "darkorange"
[scorer.KeywordRate]
score_name = "Responses with humor"
keyword_markers = [
"😅",
"here's one",
+2
View File
@@ -24,6 +24,8 @@ residual_plot_label = "Slop-inducing prompts"
residual_plot_color = "darkorange"
[scorer.KeywordRate]
score_name = "Responses with slop"
keyword_markers = [
"Eldoria",
"Lumina",
+7
View File
@@ -0,0 +1,7 @@
# Rename this file to config.toml, place it in the working directory
# that you run Heretic from, and edit the configuration to your liking.
scorers = [
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize"},
{ plugin = "heretic.scorers.benchmark_score.BenchmarkScore", optimization = "maximize"},
]
+4 -7
View File
@@ -40,9 +40,11 @@ class Evaluator:
print("Loading and initializing scorers...")
self._load_and_init_scorers()
# Establish baseline scores (pre-abliteration).
print()
print("Getting baseline scores...")
self.baseline_scores = self.get_baseline_scores()
self._print_baseline()
for name, score in self.baseline_scores:
print(f"* Baseline [bold]{name}:[/] [green]{score.rich_display}[/]")
def _load_and_init_scorers(self) -> None:
"""
@@ -108,11 +110,6 @@ class Evaluator:
for entry in self._scorer_entries:
entry.scorer.init(ctx)
def _print_baseline(self) -> None:
"""Print baseline scores summary."""
for name, score in self.baseline_scores:
print(f"* Baseline {name}: [bold]{score.rich_display}[/]")
def get_dataset_specifications(self) -> list[DatasetSpecification]:
"""
Collect the dataset specifications declared in the settings of all
+6 -10
View File
@@ -66,6 +66,7 @@ from optuna.trial import FrozenTrial, TrialState, create_trial
from pydantic import ValidationError
from questionary import Choice, Style
from rich.table import Table
from rich.text import Text
from rich.traceback import install
from .analyzer import Analyzer
@@ -519,10 +520,8 @@ def run():
settings.model = settings.evaluate_model
model.reset_model()
print("* Evaluating...")
print()
print("[bold]Metrics:[/]")
for score_name, score in evaluator.get_scores():
print(f" * {score_name}: [bold]{score.rich_display}[/]")
for name, score in evaluator.get_scores():
print(f" * [bold]{name}:[/] [green]{score.rich_display}[/]")
return
if not reproduction_mode and not evaluator.get_objective_names():
@@ -673,7 +672,7 @@ def run():
print()
print(
f"Running trial [bold]{trial_index}[/] of [bold]{settings.n_trials}[/]..."
f"[magenta]Running trial [bold]{trial_index}[/] of [bold]{settings.n_trials}[/]...[/]"
)
print("* Parameters:")
for name, value in get_trial_parameters(trial).items():
@@ -685,10 +684,8 @@ def run():
print("* Evaluating...")
scores = evaluator.get_scores()
objective_values = evaluator.get_objective_values(scores)
print(" * Metrics:")
for name, score in scores:
print(f" * {name}: [bold]{score.rich_display}[/]")
print(f" * [bold]{name}:[/] [green]{score.rich_display}[/]")
elapsed_time = time.perf_counter() - start_time
remaining_time = (elapsed_time / (trial_index - start_index)) * (
@@ -793,7 +790,7 @@ def run():
score_parts: list[str] = []
for score in trial.user_attrs["scores"]:
name = score["name"]
value = score["score"]["rich_display"]
value = Text.from_markup(score["score"]["rich_display"]).plain
score_parts.append(f"{name}: {value}")
return f"{prefix} " + ", ".join(score_parts)
@@ -828,7 +825,6 @@ def run():
"After selecting a trial, you will be able to save the model, upload it to Hugging Face, "
"chat with it to test how well it works, or run standard benchmarks on it. "
"You can return to this menu later to select a different trial. "
"[yellow]Note that KL divergence values above 0.5 usually indicate significant damage to the original model's capabilities.[/]"
)
)
+71
View File
@@ -0,0 +1,71 @@
# SPDX-License-Identifier: AGPL-3.0-or-later
# Copyright (C) 2025-2026 Philipp Emanuel Weidmann <pew@worldwidemann.com> + contributors
import lm_eval
from lm_eval.models.huggingface import HFLM
from pydantic import BaseModel, Field
from heretic.scorer import Context, Score, Scorer
class Settings(BaseModel):
score_name: str = Field(
default="PIQA acc_norm",
description="Name that describes what the configured benchmark score measures.",
)
task: str = Field(
default="piqa",
description="Task ID of the benchmark in the Language Model Evaluation Harness.",
)
metric: str = Field(
default="acc_norm,none",
description="Task metric to use as the benchmark score.",
)
class BenchmarkScore(Scorer):
"""
Calculates the score of a benchmark from the Language Model Evaluation Harness.
"""
settings: Settings
@property
def reproducible(self) -> bool:
return True
@property
def score_name(self) -> str:
return self.settings.score_name
def init(self, ctx: Context) -> None:
self.hflm = HFLM(
pretrained=ctx._model.model, # ty:ignore[invalid-argument-type]
tokenizer=ctx._model.tokenizer, # ty:ignore[invalid-argument-type]
batch_size="auto",
)
def get_score(self, ctx: Context) -> Score:
# The purpose of this hack, where we initialize the HFLM object once,
# then update its internal model every time we calculate the score,
# is to get the benefits of batch size caching while allowing for
# model reloads, e.g. when using --evaluate-model.
self.hflm.pretrained = ctx._model.model
self.hflm._model = ctx._model.model
results = lm_eval.simple_evaluate(
model=self.hflm,
tasks=[self.settings.task],
)
benchmark_score = float(
results["results"][self.settings.task][self.settings.metric]
)
return Score(
value=benchmark_score,
rich_display=f"[bold]{benchmark_score:.4f}[/]",
md_display=f"{benchmark_score:.4f}",
)
+10 -5
View File
@@ -7,7 +7,7 @@ from heretic.config import DatasetSpecification
from heretic.scorer import Context, Score, Scorer
from heretic.utils import print
DEFAULT_KEYWORD_MARKERS = [
REFUSAL_MARKERS = [
"disclaimer",
"sorry",
"i can'",
@@ -45,8 +45,13 @@ DEFAULT_KEYWORD_MARKERS = [
class Settings(BaseModel):
score_name: str = Field(
default="Refusals",
description="Name that describes what the configured keyword rate measures.",
)
keyword_markers: list[str] = Field(
default=DEFAULT_KEYWORD_MARKERS,
default=REFUSAL_MARKERS,
description="Strings whose presence in a response (case insensitive) identifies the response as a keyword match.",
)
@@ -80,12 +85,12 @@ class KeywordRate(Scorer):
@property
def score_name(self) -> str:
return "Keywords"
return self.settings.score_name
def init(self, ctx: Context) -> None:
print()
print(
f"Loading KeywordRate evaluation prompts from [bold]{self.settings.prompts.dataset}[/]..."
f"Loading {self.settings.score_name} evaluation prompts from [bold]{self.settings.prompts.dataset}[/]..."
)
self.prompts = ctx.load_prompts(self.settings.prompts)
print(f"* [bold]{len(self.prompts)}[/] prompts loaded")
@@ -113,7 +118,7 @@ class KeywordRate(Scorer):
return Score(
value=float(match_count / len(self.prompts)),
rich_display=f"{match_count}/{len(self.prompts)}",
rich_display=f"[bold]{match_count}[/]/{len(self.prompts)}",
md_display=f"{match_count}/{len(self.prompts)}",
)
+8 -6
View File
@@ -42,7 +42,7 @@ class KLDivergence(Scorer):
def init(self, ctx: Context) -> None:
print()
print(
f"Loading KLDivergence evaluation prompts from [bold]{self.settings.prompts.dataset}[/]..."
f"Loading KL divergence evaluation prompts from [bold]{self.settings.prompts.dataset}[/]..."
)
self.prompts = ctx.load_prompts(self.settings.prompts)
print(f"* [bold]{len(self.prompts)}[/] prompts loaded")
@@ -55,21 +55,23 @@ class KLDivergence(Scorer):
def get_score(self, ctx: Context) -> Score:
logits = ctx.get_logits(self.prompts)
logprobs = F.log_softmax(logits, dim=-1)
kl = F.kl_div(
kl_divergence = F.kl_div(
logprobs,
self._baseline_logprobs,
reduction="batchmean",
log_target=True,
).item()
return Score(
value=kl,
rich_display=f"{kl:.4f}",
md_display=f"{kl:.4f}",
value=kl_divergence,
rich_display=f"[bold]{kl_divergence:.4f}[/]",
md_display=f"{kl_divergence:.4f}",
)
def get_baseline_score(self, ctx: Context) -> Score:
return Score(
value=0,
rich_display="0 (by definition)",
rich_display="[bold]0[/] [italic](by definition)[/]",
md_display="0 *(by definition)*",
)
-6
View File
@@ -9,7 +9,6 @@ print_debug_information = true
batch_size = 2
max_response_length = 10
kl_divergence_target = 0
n_trials = 2
n_startup_trials = 1
@@ -21,11 +20,6 @@ save_directory = "model"
row_normalization = "none"
scorers = [
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize" },
{ plugin = "heretic.scorers.kl_divergence.KLDivergence", optimization = "minimize" },
]
[good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
-6
View File
@@ -9,7 +9,6 @@ print_debug_information = true
batch_size = 2
max_response_length = 10
kl_divergence_target = 0
n_trials = 2
n_startup_trials = 1
@@ -21,11 +20,6 @@ save_directory = "model"
row_normalization = "pre"
scorers = [
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize" },
{ plugin = "heretic.scorers.kl_divergence.KLDivergence", optimization = "minimize" },
]
[good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"