mirror of
https://github.com/p-e-w/heretic.git
synced 2026-10-01 16:31:28 -07:00
feat: modifier plugins + ARA (#446)
* feat: add modifier base class * feat: re-implement abliteration as a modifier plugin * fix: adjust good/bad prompts hack to match tests * feat: support dataset specifications containing multiple individual datasets * feat: re-implement ARA as a modifier plugin Arbitrary-Rank Ablation (ARA) (Weidmann 2026) was originally introduced by @p-e-w in #211 Co-authored-by: kabachuha <artemkhrapov2001@yandex.ru> Co-authored-by: joninco <joninco@bullpoint.org> Co-authored-by: Ashar <coder3101@users.noreply.github.com> * feat: add tests for ARA * feat: reduce default number of trials * ci: remove broken "semantic-pull-request" workflow * ci: add yet another alternative model hash --------- Co-authored-by: kabachuha <artemkhrapov2001@yandex.ru> Co-authored-by: joninco <joninco@bullpoint.org> Co-authored-by: Ashar <coder3101@users.noreply.github.com>
This commit is contained in:
co-authored by
kabachuha
joninco
Ashar
parent
71e6d5eb38
commit
662e4ba27e
+144
-80
@@ -71,58 +71,27 @@ chain_of_thought_skips = [
|
||||
# Whether to print additional information that can help with debugging.
|
||||
print_debug_information = false
|
||||
|
||||
# Whether to print detailed information about residuals and residual directions.
|
||||
print_residual_geometry = false
|
||||
|
||||
# Whether to generate plots showing PaCMAP projections of residual vectors.
|
||||
plot_residuals = false
|
||||
|
||||
# Base path to save plots of residual vectors to.
|
||||
residual_plot_path = "plots"
|
||||
|
||||
# Title placed above plots of residual vectors.
|
||||
residual_plot_title = 'PaCMAP Projection of Residual Vectors for "Harmless" and "Harmful" Prompts'
|
||||
|
||||
# Matplotlib style sheet to use for plots of residual vectors.
|
||||
residual_plot_style = "dark_background"
|
||||
|
||||
# List of scorers to evaluate.
|
||||
# Each entry is an object:
|
||||
# { plugin = <plugin>, optimization = <optimization>, instance_name = <optional> }
|
||||
# where <optimization> is one of "minimize", "maximize", "none" (do not optimize)
|
||||
# List of scorer plugin configs. Each entry is an object
|
||||
# { plugin = <plugin>, optimization = <optimization>, instance_name = <optional> }.
|
||||
# <optimization> is one of "minimize", "maximize", or "none" (do not optimize).
|
||||
scorers = [
|
||||
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize"},
|
||||
{ plugin = "heretic.scorers.kl_divergence.KLDivergence", optimization = "minimize"},
|
||||
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize" },
|
||||
{ plugin = "heretic.scorers.kl_divergence.KLDivergence", optimization = "minimize" },
|
||||
]
|
||||
|
||||
# Whether to adjust the residual directions so that only the component that is
|
||||
# orthogonal to the good direction is subtracted during abliteration.
|
||||
orthogonalize_direction = true
|
||||
|
||||
# How to apply row normalization of the weights. Options:
|
||||
# "none" (no normalization),
|
||||
# "pre" (compute LoRA adapter relative to row-normalized weights),
|
||||
# "full" (like "pre", but renormalizes to preserve original row magnitudes).
|
||||
row_normalization = "full"
|
||||
|
||||
# The rank of the LoRA adapter to use when "full" row normalization is used.
|
||||
# Row magnitude preservation is approximate due to non-linear effects,
|
||||
# and this determines the rank of that approximation. Higher ranks produce
|
||||
# larger output files and may slow down evaluation.
|
||||
full_normalization_lora_rank = 3
|
||||
|
||||
# The symmetric winsorization to apply to the per-prompt, per-layer residual vectors,
|
||||
# expressed as the quantile to clamp to (between 0 and 1). Disabled by default.
|
||||
# This can tame so-called "massive activations" that occur in some models.
|
||||
# Example: winsorization_quantile = 0.95 computes the 0.95-quantile of the absolute values
|
||||
# of the components, then clamps the magnitudes of all components to that quantile.
|
||||
winsorization_quantile = 1.0
|
||||
# List of modifier plugin configs. Each entry is an object
|
||||
# { plugin = <plugin>, instance_name = <optional> }.
|
||||
# Note that only a single modifier can currently be applied,
|
||||
# and this list must contain exactly one entry.
|
||||
modifiers = [
|
||||
{ plugin = "heretic.modifiers.ara.ARA" },
|
||||
]
|
||||
|
||||
# Number of abliteration trials to run during optimization.
|
||||
n_trials = 200
|
||||
n_trials = 100
|
||||
|
||||
# Number of trials that use random sampling for the purpose of exploration.
|
||||
n_startup_trials = 60
|
||||
n_startup_trials = 30
|
||||
|
||||
# Directory to save and load study progress to/from.
|
||||
study_checkpoint_dir = "checkpoints"
|
||||
@@ -133,6 +102,46 @@ max_shard_size = "5GB"
|
||||
# System prompt to use when prompting the model.
|
||||
system_prompt = "You are a helpful assistant."
|
||||
|
||||
# Dataset of prompts to use for automatically determining the optimal batch size.
|
||||
[batch_size_test_prompts]
|
||||
dataset = "mlabonne/harmless_alpaca"
|
||||
split = "train[:256]"
|
||||
column = "text"
|
||||
|
||||
# Dataset of prompts to use for automatically determining the response prefix.
|
||||
[[response_prefix_test_prompts]]
|
||||
dataset = "mlabonne/harmless_alpaca"
|
||||
split = "train[:100]"
|
||||
column = "text"
|
||||
|
||||
[[response_prefix_test_prompts]]
|
||||
dataset = "mlabonne/harmful_behaviors"
|
||||
split = "train[:100]"
|
||||
column = "text"
|
||||
|
||||
|
||||
# Plugin-specific settings live in top-level TOML tables.
|
||||
#
|
||||
# For scorer plugins, use: `[scorer.<ClassName>]` (and optionally `[scorer.<ClassName>_<instance_name>]` for instance-related config).
|
||||
# For modifier plugins, use: `[modifier.<ClassName>]` (and optionally `[modifier.<ClassName>_<instance_name>]` for instance-related config).
|
||||
#
|
||||
# You can load multiple instances of the same plugin class by setting `instance_name`
|
||||
# in the `scorers/modifiers = [...]` list. Each instance is still identified as `ClassName.instanceName`
|
||||
# internally, but its config overrides live under `[scorer/modifier.ClassName_<instance_name>]`.
|
||||
#
|
||||
# Example:
|
||||
# scorers = [
|
||||
# { plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize", instance_name = "small" },
|
||||
# { plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize", instance_name = "tiny" },
|
||||
# ]
|
||||
#
|
||||
# Shared defaults for all instances live under `[scorer.KeywordRate]` and can be overridden per
|
||||
# instance under `[scorer.KeywordRate_<instance_name>]`.
|
||||
#
|
||||
# Example instance override:
|
||||
# [scorer.KeywordRate_small.prompts]
|
||||
# split = "test[:10]"
|
||||
#
|
||||
# Each "dataset" below can be a Hugging Face dataset ID, a path to a dataset on disk,
|
||||
# or a path to a plain text file with one prompt per line (empty lines are ignored).
|
||||
# For text files, "column" is ignored and "split" is optional; when given, it selects
|
||||
@@ -140,24 +149,6 @@ system_prompt = "You are a helpful assistant."
|
||||
# "config" specifies a dataset's specific config/subset name (e.g. "english", "hindi").
|
||||
# Leave unset for datasets with a single configuration.
|
||||
|
||||
# Dataset of prompts that tend to not result in refusals (used for calculating residual directions).
|
||||
[good_prompts]
|
||||
dataset = "mlabonne/harmless_alpaca"
|
||||
split = "train[:400]"
|
||||
column = "text"
|
||||
residual_plot_label = '"Harmless" prompts'
|
||||
residual_plot_color = "royalblue"
|
||||
|
||||
# Dataset of prompts that tend to result in refusals (used for calculating residual directions).
|
||||
[bad_prompts]
|
||||
dataset = "mlabonne/harmful_behaviors"
|
||||
split = "train[:400]"
|
||||
column = "text"
|
||||
residual_plot_label = '"Harmful" prompts'
|
||||
residual_plot_color = "darkorange"
|
||||
|
||||
# Plugin-specific settings live in a top-level TOML table.
|
||||
# For scorer plugins, use: `[scorer.<ClassName>]` (and optionally `[scorer.<ClassName>_<instance_name>]` for instance-related config).
|
||||
[scorer.KeywordRate]
|
||||
# Name that describes what the configured keyword rate measures.
|
||||
score_name = "Refusals"
|
||||
@@ -202,30 +193,103 @@ keyword_markers = [
|
||||
"ethical boundaries",
|
||||
]
|
||||
|
||||
# Scorer-owned evaluation prompts
|
||||
# Dataset of prompts to evaluate the keyword match rate on.
|
||||
[scorer.KeywordRate.prompts]
|
||||
dataset = "mlabonne/harmful_behaviors"
|
||||
split = "test[:100]"
|
||||
column = "text"
|
||||
|
||||
# You can also load multiple instances of the same scorer class by setting `instance_name`
|
||||
# in the `scorers = [...]` list. Each instance is still identified as `ClassName.instanceName`
|
||||
# internally, but its config overrides live under `[scorer.ClassName_<instance_name>]`.
|
||||
#
|
||||
# Example:
|
||||
# scorers = [
|
||||
# { plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = 'minimize', instance_name = "small" },
|
||||
# { plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = 'minimize', instance_name = "tiny" },
|
||||
# ]
|
||||
#
|
||||
# Shared defaults for all instances live under `[scorer.KeywordRate]` and can be overridden per
|
||||
# instance under `[scorer.KeywordRate_<instance_name>]`.
|
||||
#
|
||||
# Example instance override:
|
||||
# [scorer.KeywordRate_small.prompts]
|
||||
# split = "test[:10]"
|
||||
|
||||
# Dataset of prompts used to measure KL divergence from original model.
|
||||
[scorer.KLDivergence.prompts]
|
||||
dataset = "mlabonne/harmless_alpaca"
|
||||
split = "test[:100]"
|
||||
column = "text"
|
||||
|
||||
|
||||
[scorer.BenchmarkScore]
|
||||
# Name that describes what the configured benchmark score measures.
|
||||
score_name = "PIQA acc_norm"
|
||||
|
||||
# Task ID of the benchmark in the Language Model Evaluation Harness.
|
||||
task = "piqa"
|
||||
|
||||
# Task metric to use as the benchmark score.
|
||||
metric = "acc_norm,none"
|
||||
|
||||
|
||||
[modifier.ARA]
|
||||
# Whether to renormalize the rows of the modified matrices to preserve
|
||||
# the original matrices' row magnitudes. This is believed to improve
|
||||
# intelligence retention (see Lai 2025, "Magnitude-Preserving Orthogonal Ablation").
|
||||
preserve_row_magnitudes = true
|
||||
|
||||
# The rank of the LoRA adapter to use.
|
||||
# While mathematically, ARA is of "arbitrary" rank, experiments have shown that
|
||||
# singular values tend to drop rapidly after a few dozen dimensions, and approximating
|
||||
# the full transformation with a LoRA has many practical advantages.
|
||||
lora_rank = 50
|
||||
|
||||
# Number of (outer) L-BFGS optimization steps to perform.
|
||||
n_optimization_steps = 5
|
||||
|
||||
# Learning rate to use in the L-BFGS optimizer.
|
||||
learning_rate = 1.0
|
||||
|
||||
# Maximum number of (inner) iterations to perform per (outer) L-BFGS optimization step.
|
||||
max_iter = 20
|
||||
|
||||
# Number of past updates to store for approximating the Hessian matrix in the L-BFGS optimizer.
|
||||
history_size = 10
|
||||
|
||||
# Whether to print the loss value for each L-BFGS optimization step.
|
||||
print_loss = false
|
||||
|
||||
# Dataset of prompts that tend to produce desirable responses.
|
||||
[modifier.ARA.good_prompts]
|
||||
dataset = "mlabonne/harmless_alpaca"
|
||||
split = "train[:400]"
|
||||
column = "text"
|
||||
|
||||
# Dataset of prompts that tend to produce undesirable responses.
|
||||
[modifier.ARA.bad_prompts]
|
||||
dataset = "mlabonne/harmful_behaviors"
|
||||
split = "train[:400]"
|
||||
column = "text"
|
||||
|
||||
|
||||
[modifier.Abliteration]
|
||||
# Whether to adjust the residual directions so that only the component that is
|
||||
# orthogonal to the good direction is subtracted during abliteration.
|
||||
orthogonalize_direction = true
|
||||
|
||||
# How to apply row normalization of the weights. Options:
|
||||
# "none" (no normalization),
|
||||
# "pre" (compute LoRA adapter relative to row-normalized weights),
|
||||
# "full" (like "pre", but renormalizes to preserve original row magnitudes).
|
||||
row_normalization = "full"
|
||||
|
||||
# The rank of the LoRA adapter to use when "full" row normalization is used.
|
||||
# Row magnitude preservation is approximate due to non-linear effects,
|
||||
# and this determines the rank of that approximation. Higher ranks produce
|
||||
# larger output files and may slow down evaluation.
|
||||
full_normalization_lora_rank = 3
|
||||
|
||||
# The symmetric winsorization to apply to the per-prompt, per-layer residual vectors,
|
||||
# expressed as the quantile to clamp to (between 0 and 1). Disabled by default.
|
||||
# This can tame so-called "massive activations" that occur in some models.
|
||||
# Example: winsorization_quantile = 0.95 computes the 0.95-quantile of the absolute values
|
||||
# of the components, then clamps the magnitudes of all components to that quantile.
|
||||
winsorization_quantile = 1.0
|
||||
|
||||
# Dataset of prompts that tend to produce desirable responses.
|
||||
[modifier.Abliteration.good_prompts]
|
||||
dataset = "mlabonne/harmless_alpaca"
|
||||
split = "train[:400]"
|
||||
column = "text"
|
||||
|
||||
# Dataset of prompts that tend to produce undesirable responses.
|
||||
[modifier.Abliteration.bad_prompts]
|
||||
dataset = "mlabonne/harmful_behaviors"
|
||||
split = "train[:400]"
|
||||
column = "text"
|
||||
|
||||
Reference in New Issue
Block a user