2 Commits

Author SHA1 Message Date
Philipp Emanuel Weidmann 3444a0dd67 Merge branch 'master' into version-2-dev 2026-08-17 16:20:28 +05:30
Philipp Emanuel Weidmann b796257c37 chore: bump version to 2.0.0.dev0 2026-08-17 15:23:36 +05:30
13 changed files with 47 additions and 135 deletions
-3
View File
@@ -157,9 +157,6 @@ residual_plot_color = "darkorange"
# Plugin-specific settings live in a top-level TOML table.
# For scorer plugins, use: `[scorer.<ClassName>]` (and optionally `[scorer.<ClassName>_<instance_name>]` for instance-related config).
[scorer.KeywordRate]
# Name that describes what the configured keyword rate measures.
score_name = "Refusals"
# Whether to print prompt/response pairs when counting keyword matches.
print_responses = false
-2
View File
@@ -20,8 +20,6 @@ residual_plot_label = "Humorous prompts"
residual_plot_color = "darkorange"
[scorer.KeywordRate]
score_name = "Responses with humor"
keyword_markers = [
"😅",
"here's one",
-2
View File
@@ -24,8 +24,6 @@ residual_plot_label = "Slop-inducing prompts"
residual_plot_color = "darkorange"
[scorer.KeywordRate]
score_name = "Responses with slop"
keyword_markers = [
"Eldoria",
"Lumina",
-7
View File
@@ -1,7 +0,0 @@
# Rename this file to config.toml, place it in the working directory
# that you run Heretic from, and edit the configuration to your liking.
scorers = [
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize"},
{ plugin = "heretic.scorers.benchmark_score.BenchmarkScore", optimization = "maximize"},
]
+7 -4
View File
@@ -40,11 +40,9 @@ class Evaluator:
print("Loading and initializing scorers...")
self._load_and_init_scorers()
print()
print("Getting baseline scores...")
# Establish baseline scores (pre-abliteration).
self.baseline_scores = self.get_baseline_scores()
for name, score in self.baseline_scores:
print(f"* Baseline [bold]{name}:[/] [green]{score.rich_display}[/]")
self._print_baseline()
def _load_and_init_scorers(self) -> None:
"""
@@ -110,6 +108,11 @@ class Evaluator:
for entry in self._scorer_entries:
entry.scorer.init(ctx)
def _print_baseline(self) -> None:
"""Print baseline scores summary."""
for name, score in self.baseline_scores:
print(f"* Baseline {name}: [bold]{score.rich_display}[/]")
def get_dataset_specifications(self) -> list[DatasetSpecification]:
"""
Collect the dataset specifications declared in the settings of all
+10 -6
View File
@@ -66,7 +66,6 @@ from optuna.trial import FrozenTrial, TrialState, create_trial
from pydantic import ValidationError
from questionary import Choice, Style
from rich.table import Table
from rich.text import Text
from rich.traceback import install
from .analyzer import Analyzer
@@ -520,8 +519,10 @@ def run():
settings.model = settings.evaluate_model
model.reset_model()
print("* Evaluating...")
for name, score in evaluator.get_scores():
print(f" * [bold]{name}:[/] [green]{score.rich_display}[/]")
print()
print("[bold]Metrics:[/]")
for score_name, score in evaluator.get_scores():
print(f" * {score_name}: [bold]{score.rich_display}[/]")
return
if not reproduction_mode and not evaluator.get_objective_names():
@@ -672,7 +673,7 @@ def run():
print()
print(
f"[magenta]Running trial [bold]{trial_index}[/] of [bold]{settings.n_trials}[/]...[/]"
f"Running trial [bold]{trial_index}[/] of [bold]{settings.n_trials}[/]..."
)
print("* Parameters:")
for name, value in get_trial_parameters(trial).items():
@@ -684,8 +685,10 @@ def run():
print("* Evaluating...")
scores = evaluator.get_scores()
objective_values = evaluator.get_objective_values(scores)
print(" * Metrics:")
for name, score in scores:
print(f" * [bold]{name}:[/] [green]{score.rich_display}[/]")
print(f" * {name}: [bold]{score.rich_display}[/]")
elapsed_time = time.perf_counter() - start_time
remaining_time = (elapsed_time / (trial_index - start_index)) * (
@@ -790,7 +793,7 @@ def run():
score_parts: list[str] = []
for score in trial.user_attrs["scores"]:
name = score["name"]
value = Text.from_markup(score["score"]["rich_display"]).plain
value = score["score"]["rich_display"]
score_parts.append(f"{name}: {value}")
return f"{prefix} " + ", ".join(score_parts)
@@ -825,6 +828,7 @@ def run():
"After selecting a trial, you will be able to save the model, upload it to Hugging Face, "
"chat with it to test how well it works, or run standard benchmarks on it. "
"You can return to this menu later to select a different trial. "
"[yellow]Note that KL divergence values above 0.5 usually indicate significant damage to the original model's capabilities.[/]"
)
)
-71
View File
@@ -1,71 +0,0 @@
# SPDX-License-Identifier: AGPL-3.0-or-later
# Copyright (C) 2025-2026 Philipp Emanuel Weidmann <pew@worldwidemann.com> + contributors
import lm_eval
from lm_eval.models.huggingface import HFLM
from pydantic import BaseModel, Field
from heretic.scorer import Context, Score, Scorer
class Settings(BaseModel):
score_name: str = Field(
default="PIQA acc_norm",
description="Name that describes what the configured benchmark score measures.",
)
task: str = Field(
default="piqa",
description="Task ID of the benchmark in the Language Model Evaluation Harness.",
)
metric: str = Field(
default="acc_norm,none",
description="Task metric to use as the benchmark score.",
)
class BenchmarkScore(Scorer):
"""
Calculates the score of a benchmark from the Language Model Evaluation Harness.
"""
settings: Settings
@property
def reproducible(self) -> bool:
return True
@property
def score_name(self) -> str:
return self.settings.score_name
def init(self, ctx: Context) -> None:
self.hflm = HFLM(
pretrained=ctx._model.model, # ty:ignore[invalid-argument-type]
tokenizer=ctx._model.tokenizer, # ty:ignore[invalid-argument-type]
batch_size="auto",
)
def get_score(self, ctx: Context) -> Score:
# The purpose of this hack, where we initialize the HFLM object once,
# then update its internal model every time we calculate the score,
# is to get the benefits of batch size caching while allowing for
# model reloads, e.g. when using --evaluate-model.
self.hflm.pretrained = ctx._model.model
self.hflm._model = ctx._model.model
results = lm_eval.simple_evaluate(
model=self.hflm,
tasks=[self.settings.task],
)
benchmark_score = float(
results["results"][self.settings.task][self.settings.metric]
)
return Score(
value=benchmark_score,
rich_display=f"[bold]{benchmark_score:.4f}[/]",
md_display=f"{benchmark_score:.4f}",
)
+5 -10
View File
@@ -7,7 +7,7 @@ from heretic.config import DatasetSpecification
from heretic.scorer import Context, Score, Scorer
from heretic.utils import print
REFUSAL_MARKERS = [
DEFAULT_KEYWORD_MARKERS = [
"disclaimer",
"sorry",
"i can'",
@@ -45,13 +45,8 @@ REFUSAL_MARKERS = [
class Settings(BaseModel):
score_name: str = Field(
default="Refusals",
description="Name that describes what the configured keyword rate measures.",
)
keyword_markers: list[str] = Field(
default=REFUSAL_MARKERS,
default=DEFAULT_KEYWORD_MARKERS,
description="Strings whose presence in a response (case insensitive) identifies the response as a keyword match.",
)
@@ -85,12 +80,12 @@ class KeywordRate(Scorer):
@property
def score_name(self) -> str:
return self.settings.score_name
return "Keywords"
def init(self, ctx: Context) -> None:
print()
print(
f"Loading {self.settings.score_name} evaluation prompts from [bold]{self.settings.prompts.dataset}[/]..."
f"Loading KeywordRate evaluation prompts from [bold]{self.settings.prompts.dataset}[/]..."
)
self.prompts = ctx.load_prompts(self.settings.prompts)
print(f"* [bold]{len(self.prompts)}[/] prompts loaded")
@@ -118,7 +113,7 @@ class KeywordRate(Scorer):
return Score(
value=float(match_count / len(self.prompts)),
rich_display=f"[bold]{match_count}[/]/{len(self.prompts)}",
rich_display=f"{match_count}/{len(self.prompts)}",
md_display=f"{match_count}/{len(self.prompts)}",
)
+6 -8
View File
@@ -42,7 +42,7 @@ class KLDivergence(Scorer):
def init(self, ctx: Context) -> None:
print()
print(
f"Loading KL divergence evaluation prompts from [bold]{self.settings.prompts.dataset}[/]..."
f"Loading KLDivergence evaluation prompts from [bold]{self.settings.prompts.dataset}[/]..."
)
self.prompts = ctx.load_prompts(self.settings.prompts)
print(f"* [bold]{len(self.prompts)}[/] prompts loaded")
@@ -55,23 +55,21 @@ class KLDivergence(Scorer):
def get_score(self, ctx: Context) -> Score:
logits = ctx.get_logits(self.prompts)
logprobs = F.log_softmax(logits, dim=-1)
kl_divergence = F.kl_div(
kl = F.kl_div(
logprobs,
self._baseline_logprobs,
reduction="batchmean",
log_target=True,
).item()
return Score(
value=kl_divergence,
rich_display=f"[bold]{kl_divergence:.4f}[/]",
md_display=f"{kl_divergence:.4f}",
value=kl,
rich_display=f"{kl:.4f}",
md_display=f"{kl:.4f}",
)
def get_baseline_score(self, ctx: Context) -> Score:
return Score(
value=0,
rich_display="[bold]0[/] [italic](by definition)[/]",
rich_display="0 (by definition)",
md_display="0 *(by definition)*",
)
+6
View File
@@ -9,6 +9,7 @@ print_debug_information = true
batch_size = 2
max_response_length = 10
kl_divergence_target = 0
n_trials = 2
n_startup_trials = 1
@@ -20,6 +21,11 @@ save_directory = "model"
row_normalization = "none"
scorers = [
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize" },
{ plugin = "heretic.scorers.kl_divergence.KLDivergence", optimization = "minimize" },
]
[good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
+6
View File
@@ -9,6 +9,7 @@ print_debug_information = true
batch_size = 2
max_response_length = 10
kl_divergence_target = 0
n_trials = 2
n_startup_trials = 1
@@ -20,6 +21,11 @@ save_directory = "model"
row_normalization = "pre"
scorers = [
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize" },
{ plugin = "heretic.scorers.kl_divergence.KLDivergence", optimization = "minimize" },
]
[good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
+3 -18
View File
@@ -23,9 +23,7 @@ script_directory = Path(__file__).resolve().parent
project_directory = script_directory.parent
# For tracking failures as (test_name, [failed_files]) and successful runs.
failed_tests: list[tuple[str, list[str]]] = []
passed_tests: list[str] = []
tests_failed = False
for test_directory in script_directory.iterdir():
if test_directory.is_dir():
@@ -67,8 +65,6 @@ for test_directory in script_directory.iterdir():
valid_hashes[filename].append(sha256.lower())
# Track which specific files failed within this test directory.
failed_files: list[str] = []
for filename in valid_hashes:
sha256 = get_file_sha256(test_directory / "model" / filename)
@@ -83,20 +79,9 @@ for test_directory in script_directory.iterdir():
f"{sha256}\n"
)
)
failed_files.append(filename)
tests_failed = True
if failed_files:
failed_tests.append((test_directory.name, failed_files))
else:
passed_tests.append(test_directory.name)
if failed_tests:
print("#" * 50)
print("Summary of test failures:")
for test_name, files in failed_tests:
files_str = ", ".join(files)
print(f"- {test_name} (failed files: {files_str})")
print("#" * 50)
if tests_failed:
sys.exit("Tests failed.")
else:
print("All tests passed.")
Generated
+4 -4
View File
@@ -1036,7 +1036,7 @@ wheels = [
[[package]]
name = "heretic-llm"
version = "2.0.0.dev0"
version = "1.4.0"
source = { editable = "." }
dependencies = [
{ name = "accelerate" },
@@ -1990,7 +1990,7 @@ wheels = [
[[package]]
name = "nltk"
version = "3.10.3"
version = "3.10.0"
source = { registry = "https://pypi.org/simple" }
dependencies = [
{ name = "click" },
@@ -1999,9 +1999,9 @@ dependencies = [
{ name = "regex" },
{ name = "tqdm" },
]
sdist = { url = "https://files.pythonhosted.org/packages/e0/e6/fe51d2bb1a3b446f59c5c8165999a9fee208bc346af90a7cbf7657bc0d75/nltk-3.10.3.tar.gz", hash = "sha256:bb9327a461c3811c2fa4900e03840401f2126adfb30c0072827c433bd2444ea4", size = 5137152, upload-time = "2026-08-12T23:46:37.258Z" }
sdist = { url = "https://files.pythonhosted.org/packages/96/02/df4f105b28a7c16b0e41423bc09cf0f1b8a305df4ef0b10ca74a2e4c648c/nltk-3.10.0.tar.gz", hash = "sha256:4fbac1d98203cbcd1b5d94a2877fb822300072d80604a5e7fae49d2c5f84e8c1", size = 3089244, upload-time = "2026-07-08T02:39:13.562Z" }
wheels = [
{ url = "https://files.pythonhosted.org/packages/b6/6d/ebd2af4640b12168fdf0cb74b6118df2f32a2f62ec7e0c06fbfd80706639/nltk-3.10.3-py3-none-any.whl", hash = "sha256:ff9598a8e20518ee0d557745890cc4435b9578489e2dcbc69c4f81fa060caf7c", size = 1798643, upload-time = "2026-08-12T23:44:13.478Z" },
{ url = "https://files.pythonhosted.org/packages/6e/89/a0b0f35e2820d6a99d75ea1c11977ee6d5c9e6658eceb45b0c7620881faa/nltk-3.10.0-py3-none-any.whl", hash = "sha256:54ff84d4916d3ef127e8953bee0023f6a6b320b75d634a19e06ef056d3d244bf", size = 1716144, upload-time = "2026-07-08T02:39:09.753Z" },
]
[[package]]