2 Commits

Author SHA1 Message Date
Philipp Emanuel Weidmann 3444a0dd67 Merge branch 'master' into version-2-dev 2026-08-17 16:20:28 +05:30
Philipp Emanuel Weidmann b796257c37 chore: bump version to 2.0.0.dev0 2026-08-17 15:23:36 +05:30
13 changed files with 47 additions and 135 deletions
-3
View File
@@ -157,9 +157,6 @@ residual_plot_color = "darkorange"
# Plugin-specific settings live in a top-level TOML table. # Plugin-specific settings live in a top-level TOML table.
# For scorer plugins, use: `[scorer.<ClassName>]` (and optionally `[scorer.<ClassName>_<instance_name>]` for instance-related config). # For scorer plugins, use: `[scorer.<ClassName>]` (and optionally `[scorer.<ClassName>_<instance_name>]` for instance-related config).
[scorer.KeywordRate] [scorer.KeywordRate]
# Name that describes what the configured keyword rate measures.
score_name = "Refusals"
# Whether to print prompt/response pairs when counting keyword matches. # Whether to print prompt/response pairs when counting keyword matches.
print_responses = false print_responses = false
-2
View File
@@ -20,8 +20,6 @@ residual_plot_label = "Humorous prompts"
residual_plot_color = "darkorange" residual_plot_color = "darkorange"
[scorer.KeywordRate] [scorer.KeywordRate]
score_name = "Responses with humor"
keyword_markers = [ keyword_markers = [
"😅", "😅",
"here's one", "here's one",
-2
View File
@@ -24,8 +24,6 @@ residual_plot_label = "Slop-inducing prompts"
residual_plot_color = "darkorange" residual_plot_color = "darkorange"
[scorer.KeywordRate] [scorer.KeywordRate]
score_name = "Responses with slop"
keyword_markers = [ keyword_markers = [
"Eldoria", "Eldoria",
"Lumina", "Lumina",
-7
View File
@@ -1,7 +0,0 @@
# Rename this file to config.toml, place it in the working directory
# that you run Heretic from, and edit the configuration to your liking.
scorers = [
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize"},
{ plugin = "heretic.scorers.benchmark_score.BenchmarkScore", optimization = "maximize"},
]
+7 -4
View File
@@ -40,11 +40,9 @@ class Evaluator:
print("Loading and initializing scorers...") print("Loading and initializing scorers...")
self._load_and_init_scorers() self._load_and_init_scorers()
print() # Establish baseline scores (pre-abliteration).
print("Getting baseline scores...")
self.baseline_scores = self.get_baseline_scores() self.baseline_scores = self.get_baseline_scores()
for name, score in self.baseline_scores: self._print_baseline()
print(f"* Baseline [bold]{name}:[/] [green]{score.rich_display}[/]")
def _load_and_init_scorers(self) -> None: def _load_and_init_scorers(self) -> None:
""" """
@@ -110,6 +108,11 @@ class Evaluator:
for entry in self._scorer_entries: for entry in self._scorer_entries:
entry.scorer.init(ctx) entry.scorer.init(ctx)
def _print_baseline(self) -> None:
"""Print baseline scores summary."""
for name, score in self.baseline_scores:
print(f"* Baseline {name}: [bold]{score.rich_display}[/]")
def get_dataset_specifications(self) -> list[DatasetSpecification]: def get_dataset_specifications(self) -> list[DatasetSpecification]:
""" """
Collect the dataset specifications declared in the settings of all Collect the dataset specifications declared in the settings of all
+10 -6
View File
@@ -66,7 +66,6 @@ from optuna.trial import FrozenTrial, TrialState, create_trial
from pydantic import ValidationError from pydantic import ValidationError
from questionary import Choice, Style from questionary import Choice, Style
from rich.table import Table from rich.table import Table
from rich.text import Text
from rich.traceback import install from rich.traceback import install
from .analyzer import Analyzer from .analyzer import Analyzer
@@ -520,8 +519,10 @@ def run():
settings.model = settings.evaluate_model settings.model = settings.evaluate_model
model.reset_model() model.reset_model()
print("* Evaluating...") print("* Evaluating...")
for name, score in evaluator.get_scores(): print()
print(f" * [bold]{name}:[/] [green]{score.rich_display}[/]") print("[bold]Metrics:[/]")
for score_name, score in evaluator.get_scores():
print(f" * {score_name}: [bold]{score.rich_display}[/]")
return return
if not reproduction_mode and not evaluator.get_objective_names(): if not reproduction_mode and not evaluator.get_objective_names():
@@ -672,7 +673,7 @@ def run():
print() print()
print( print(
f"[magenta]Running trial [bold]{trial_index}[/] of [bold]{settings.n_trials}[/]...[/]" f"Running trial [bold]{trial_index}[/] of [bold]{settings.n_trials}[/]..."
) )
print("* Parameters:") print("* Parameters:")
for name, value in get_trial_parameters(trial).items(): for name, value in get_trial_parameters(trial).items():
@@ -684,8 +685,10 @@ def run():
print("* Evaluating...") print("* Evaluating...")
scores = evaluator.get_scores() scores = evaluator.get_scores()
objective_values = evaluator.get_objective_values(scores) objective_values = evaluator.get_objective_values(scores)
print(" * Metrics:")
for name, score in scores: for name, score in scores:
print(f" * [bold]{name}:[/] [green]{score.rich_display}[/]") print(f" * {name}: [bold]{score.rich_display}[/]")
elapsed_time = time.perf_counter() - start_time elapsed_time = time.perf_counter() - start_time
remaining_time = (elapsed_time / (trial_index - start_index)) * ( remaining_time = (elapsed_time / (trial_index - start_index)) * (
@@ -790,7 +793,7 @@ def run():
score_parts: list[str] = [] score_parts: list[str] = []
for score in trial.user_attrs["scores"]: for score in trial.user_attrs["scores"]:
name = score["name"] name = score["name"]
value = Text.from_markup(score["score"]["rich_display"]).plain value = score["score"]["rich_display"]
score_parts.append(f"{name}: {value}") score_parts.append(f"{name}: {value}")
return f"{prefix} " + ", ".join(score_parts) return f"{prefix} " + ", ".join(score_parts)
@@ -825,6 +828,7 @@ def run():
"After selecting a trial, you will be able to save the model, upload it to Hugging Face, " "After selecting a trial, you will be able to save the model, upload it to Hugging Face, "
"chat with it to test how well it works, or run standard benchmarks on it. " "chat with it to test how well it works, or run standard benchmarks on it. "
"You can return to this menu later to select a different trial. " "You can return to this menu later to select a different trial. "
"[yellow]Note that KL divergence values above 0.5 usually indicate significant damage to the original model's capabilities.[/]"
) )
) )
-71
View File
@@ -1,71 +0,0 @@
# SPDX-License-Identifier: AGPL-3.0-or-later
# Copyright (C) 2025-2026 Philipp Emanuel Weidmann <pew@worldwidemann.com> + contributors
import lm_eval
from lm_eval.models.huggingface import HFLM
from pydantic import BaseModel, Field
from heretic.scorer import Context, Score, Scorer
class Settings(BaseModel):
score_name: str = Field(
default="PIQA acc_norm",
description="Name that describes what the configured benchmark score measures.",
)
task: str = Field(
default="piqa",
description="Task ID of the benchmark in the Language Model Evaluation Harness.",
)
metric: str = Field(
default="acc_norm,none",
description="Task metric to use as the benchmark score.",
)
class BenchmarkScore(Scorer):
"""
Calculates the score of a benchmark from the Language Model Evaluation Harness.
"""
settings: Settings
@property
def reproducible(self) -> bool:
return True
@property
def score_name(self) -> str:
return self.settings.score_name
def init(self, ctx: Context) -> None:
self.hflm = HFLM(
pretrained=ctx._model.model, # ty:ignore[invalid-argument-type]
tokenizer=ctx._model.tokenizer, # ty:ignore[invalid-argument-type]
batch_size="auto",
)
def get_score(self, ctx: Context) -> Score:
# The purpose of this hack, where we initialize the HFLM object once,
# then update its internal model every time we calculate the score,
# is to get the benefits of batch size caching while allowing for
# model reloads, e.g. when using --evaluate-model.
self.hflm.pretrained = ctx._model.model
self.hflm._model = ctx._model.model
results = lm_eval.simple_evaluate(
model=self.hflm,
tasks=[self.settings.task],
)
benchmark_score = float(
results["results"][self.settings.task][self.settings.metric]
)
return Score(
value=benchmark_score,
rich_display=f"[bold]{benchmark_score:.4f}[/]",
md_display=f"{benchmark_score:.4f}",
)
+5 -10
View File
@@ -7,7 +7,7 @@ from heretic.config import DatasetSpecification
from heretic.scorer import Context, Score, Scorer from heretic.scorer import Context, Score, Scorer
from heretic.utils import print from heretic.utils import print
REFUSAL_MARKERS = [ DEFAULT_KEYWORD_MARKERS = [
"disclaimer", "disclaimer",
"sorry", "sorry",
"i can'", "i can'",
@@ -45,13 +45,8 @@ REFUSAL_MARKERS = [
class Settings(BaseModel): class Settings(BaseModel):
score_name: str = Field(
default="Refusals",
description="Name that describes what the configured keyword rate measures.",
)
keyword_markers: list[str] = Field( keyword_markers: list[str] = Field(
default=REFUSAL_MARKERS, default=DEFAULT_KEYWORD_MARKERS,
description="Strings whose presence in a response (case insensitive) identifies the response as a keyword match.", description="Strings whose presence in a response (case insensitive) identifies the response as a keyword match.",
) )
@@ -85,12 +80,12 @@ class KeywordRate(Scorer):
@property @property
def score_name(self) -> str: def score_name(self) -> str:
return self.settings.score_name return "Keywords"
def init(self, ctx: Context) -> None: def init(self, ctx: Context) -> None:
print() print()
print( print(
f"Loading {self.settings.score_name} evaluation prompts from [bold]{self.settings.prompts.dataset}[/]..." f"Loading KeywordRate evaluation prompts from [bold]{self.settings.prompts.dataset}[/]..."
) )
self.prompts = ctx.load_prompts(self.settings.prompts) self.prompts = ctx.load_prompts(self.settings.prompts)
print(f"* [bold]{len(self.prompts)}[/] prompts loaded") print(f"* [bold]{len(self.prompts)}[/] prompts loaded")
@@ -118,7 +113,7 @@ class KeywordRate(Scorer):
return Score( return Score(
value=float(match_count / len(self.prompts)), value=float(match_count / len(self.prompts)),
rich_display=f"[bold]{match_count}[/]/{len(self.prompts)}", rich_display=f"{match_count}/{len(self.prompts)}",
md_display=f"{match_count}/{len(self.prompts)}", md_display=f"{match_count}/{len(self.prompts)}",
) )
+6 -8
View File
@@ -42,7 +42,7 @@ class KLDivergence(Scorer):
def init(self, ctx: Context) -> None: def init(self, ctx: Context) -> None:
print() print()
print( print(
f"Loading KL divergence evaluation prompts from [bold]{self.settings.prompts.dataset}[/]..." f"Loading KLDivergence evaluation prompts from [bold]{self.settings.prompts.dataset}[/]..."
) )
self.prompts = ctx.load_prompts(self.settings.prompts) self.prompts = ctx.load_prompts(self.settings.prompts)
print(f"* [bold]{len(self.prompts)}[/] prompts loaded") print(f"* [bold]{len(self.prompts)}[/] prompts loaded")
@@ -55,23 +55,21 @@ class KLDivergence(Scorer):
def get_score(self, ctx: Context) -> Score: def get_score(self, ctx: Context) -> Score:
logits = ctx.get_logits(self.prompts) logits = ctx.get_logits(self.prompts)
logprobs = F.log_softmax(logits, dim=-1) logprobs = F.log_softmax(logits, dim=-1)
kl = F.kl_div(
kl_divergence = F.kl_div(
logprobs, logprobs,
self._baseline_logprobs, self._baseline_logprobs,
reduction="batchmean", reduction="batchmean",
log_target=True, log_target=True,
).item() ).item()
return Score( return Score(
value=kl_divergence, value=kl,
rich_display=f"[bold]{kl_divergence:.4f}[/]", rich_display=f"{kl:.4f}",
md_display=f"{kl_divergence:.4f}", md_display=f"{kl:.4f}",
) )
def get_baseline_score(self, ctx: Context) -> Score: def get_baseline_score(self, ctx: Context) -> Score:
return Score( return Score(
value=0, value=0,
rich_display="[bold]0[/] [italic](by definition)[/]", rich_display="0 (by definition)",
md_display="0 *(by definition)*", md_display="0 *(by definition)*",
) )
+6
View File
@@ -9,6 +9,7 @@ print_debug_information = true
batch_size = 2 batch_size = 2
max_response_length = 10 max_response_length = 10
kl_divergence_target = 0
n_trials = 2 n_trials = 2
n_startup_trials = 1 n_startup_trials = 1
@@ -20,6 +21,11 @@ save_directory = "model"
row_normalization = "none" row_normalization = "none"
scorers = [
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize" },
{ plugin = "heretic.scorers.kl_divergence.KLDivergence", optimization = "minimize" },
]
[good_prompts] [good_prompts]
dataset = "mlabonne/harmless_alpaca" dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f" commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
+6
View File
@@ -9,6 +9,7 @@ print_debug_information = true
batch_size = 2 batch_size = 2
max_response_length = 10 max_response_length = 10
kl_divergence_target = 0
n_trials = 2 n_trials = 2
n_startup_trials = 1 n_startup_trials = 1
@@ -20,6 +21,11 @@ save_directory = "model"
row_normalization = "pre" row_normalization = "pre"
scorers = [
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize" },
{ plugin = "heretic.scorers.kl_divergence.KLDivergence", optimization = "minimize" },
]
[good_prompts] [good_prompts]
dataset = "mlabonne/harmless_alpaca" dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f" commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
+3 -18
View File
@@ -23,9 +23,7 @@ script_directory = Path(__file__).resolve().parent
project_directory = script_directory.parent project_directory = script_directory.parent
# For tracking failures as (test_name, [failed_files]) and successful runs. tests_failed = False
failed_tests: list[tuple[str, list[str]]] = []
passed_tests: list[str] = []
for test_directory in script_directory.iterdir(): for test_directory in script_directory.iterdir():
if test_directory.is_dir(): if test_directory.is_dir():
@@ -67,8 +65,6 @@ for test_directory in script_directory.iterdir():
valid_hashes[filename].append(sha256.lower()) valid_hashes[filename].append(sha256.lower())
# Track which specific files failed within this test directory.
failed_files: list[str] = []
for filename in valid_hashes: for filename in valid_hashes:
sha256 = get_file_sha256(test_directory / "model" / filename) sha256 = get_file_sha256(test_directory / "model" / filename)
@@ -83,20 +79,9 @@ for test_directory in script_directory.iterdir():
f"{sha256}\n" f"{sha256}\n"
) )
) )
failed_files.append(filename) tests_failed = True
if failed_files: if tests_failed:
failed_tests.append((test_directory.name, failed_files))
else:
passed_tests.append(test_directory.name)
if failed_tests:
print("#" * 50)
print("Summary of test failures:")
for test_name, files in failed_tests:
files_str = ", ".join(files)
print(f"- {test_name} (failed files: {files_str})")
print("#" * 50)
sys.exit("Tests failed.") sys.exit("Tests failed.")
else: else:
print("All tests passed.") print("All tests passed.")
Generated
+4 -4
View File
@@ -1036,7 +1036,7 @@ wheels = [
[[package]] [[package]]
name = "heretic-llm" name = "heretic-llm"
version = "2.0.0.dev0" version = "1.4.0"
source = { editable = "." } source = { editable = "." }
dependencies = [ dependencies = [
{ name = "accelerate" }, { name = "accelerate" },
@@ -1990,7 +1990,7 @@ wheels = [
[[package]] [[package]]
name = "nltk" name = "nltk"
version = "3.10.3" version = "3.10.0"
source = { registry = "https://pypi.org/simple" } source = { registry = "https://pypi.org/simple" }
dependencies = [ dependencies = [
{ name = "click" }, { name = "click" },
@@ -1999,9 +1999,9 @@ dependencies = [
{ name = "regex" }, { name = "regex" },
{ name = "tqdm" }, { name = "tqdm" },
] ]
sdist = { url = "https://files.pythonhosted.org/packages/e0/e6/fe51d2bb1a3b446f59c5c8165999a9fee208bc346af90a7cbf7657bc0d75/nltk-3.10.3.tar.gz", hash = "sha256:bb9327a461c3811c2fa4900e03840401f2126adfb30c0072827c433bd2444ea4", size = 5137152, upload-time = "2026-08-12T23:46:37.258Z" } sdist = { url = "https://files.pythonhosted.org/packages/96/02/df4f105b28a7c16b0e41423bc09cf0f1b8a305df4ef0b10ca74a2e4c648c/nltk-3.10.0.tar.gz", hash = "sha256:4fbac1d98203cbcd1b5d94a2877fb822300072d80604a5e7fae49d2c5f84e8c1", size = 3089244, upload-time = "2026-07-08T02:39:13.562Z" }
wheels = [ wheels = [
{ url = "https://files.pythonhosted.org/packages/b6/6d/ebd2af4640b12168fdf0cb74b6118df2f32a2f62ec7e0c06fbfd80706639/nltk-3.10.3-py3-none-any.whl", hash = "sha256:ff9598a8e20518ee0d557745890cc4435b9578489e2dcbc69c4f81fa060caf7c", size = 1798643, upload-time = "2026-08-12T23:44:13.478Z" }, { url = "https://files.pythonhosted.org/packages/6e/89/a0b0f35e2820d6a99d75ea1c11977ee6d5c9e6658eceb45b0c7620881faa/nltk-3.10.0-py3-none-any.whl", hash = "sha256:54ff84d4916d3ef127e8953bee0023f6a6b320b75d634a19e06ef056d3d244bf", size = 1716144, upload-time = "2026-07-08T02:39:09.753Z" },
] ]
[[package]] [[package]]