feat: add benchmark scorer

This commit is contained in:
Philipp Emanuel Weidmann
2026-09-03 17:32:19 +05:30
parent ded115c669
commit 1c2688f756
4 changed files with 80 additions and 2 deletions
+1 -1
View File
@@ -157,7 +157,7 @@ residual_plot_color = "darkorange"
# Plugin-specific settings live in a top-level TOML table.
# For scorer plugins, use: `[scorer.<ClassName>]` (and optionally `[scorer.<ClassName>_<instance_name>]` for instance-related config).
[scorer.KeywordRate]
# Name that describes what the keyword rate measures as configured.
# Name that describes what the configured keyword rate measures.
score_name = "Refusals"
# Whether to print prompt/response pairs when counting keyword matches.
+7
View File
@@ -0,0 +1,7 @@
# Rename this file to config.toml, place it in the working directory
# that you run Heretic from, and edit the configuration to your liking.
scorers = [
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize"},
{ plugin = "heretic.scorers.benchmark_score.BenchmarkScore", optimization = "maximize"},
]
+71
View File
@@ -0,0 +1,71 @@
# SPDX-License-Identifier: AGPL-3.0-or-later
# Copyright (C) 2025-2026 Philipp Emanuel Weidmann <pew@worldwidemann.com> + contributors
import lm_eval
from lm_eval.models.huggingface import HFLM
from pydantic import BaseModel, Field
from heretic.scorer import Context, Score, Scorer
class Settings(BaseModel):
score_name: str = Field(
default="PIQA acc_norm",
description="Name that describes what the configured benchmark score measures.",
)
task: str = Field(
default="piqa",
description="Task ID of the benchmark in the Language Model Evaluation Harness.",
)
metric: str = Field(
default="acc_norm,none",
description="Task metric to use as the benchmark score.",
)
class BenchmarkScore(Scorer):
"""
Calculates the score of a benchmark from the Language Model Evaluation Harness.
"""
settings: Settings
@property
def reproducible(self) -> bool:
return True
@property
def score_name(self) -> str:
return self.settings.score_name
def init(self, ctx: Context) -> None:
self.hflm = HFLM(
pretrained=ctx._model.model, # ty:ignore[invalid-argument-type]
tokenizer=ctx._model.tokenizer, # ty:ignore[invalid-argument-type]
batch_size="auto",
)
def get_score(self, ctx: Context) -> Score:
# The purpose of this hack, where we initialize the HFLM object once,
# then update its internal model every time we calculate the score,
# is to get the benefits of batch size caching while allowing for
# model reloads, e.g. when using --evaluate-model.
self.hflm.pretrained = ctx._model.model
self.hflm._model = ctx._model.model
results = lm_eval.simple_evaluate(
model=self.hflm,
tasks=[self.settings.task],
)
benchmark_score = float(
results["results"][self.settings.task][self.settings.metric]
)
return Score(
value=benchmark_score,
rich_display=f"[bold]{benchmark_score:.4f}[/]",
md_display=f"{benchmark_score:.4f}",
)
+1 -1
View File
@@ -47,7 +47,7 @@ REFUSAL_MARKERS = [
class Settings(BaseModel):
score_name: str = Field(
default="Refusals",
description="Name that describes what the keyword rate measures as configured.",
description="Name that describes what the configured keyword rate measures.",
)
keyword_markers: list[str] = Field(