feat: modifier plugins + ARA (#446)

* feat: add modifier base class

* feat: re-implement abliteration as a modifier plugin

* fix: adjust good/bad prompts hack to match tests

* feat: support dataset specifications containing multiple individual datasets

* feat: re-implement ARA as a modifier plugin

Arbitrary-Rank Ablation (ARA) (Weidmann 2026) was originally introduced by @p-e-w in #211

Co-authored-by: kabachuha <artemkhrapov2001@yandex.ru>
Co-authored-by: joninco <joninco@bullpoint.org>
Co-authored-by: Ashar <coder3101@users.noreply.github.com>

* feat: add tests for ARA

* feat: reduce default number of trials

* ci: remove broken "semantic-pull-request" workflow

* ci: add yet another alternative model hash

---------

Co-authored-by: kabachuha <artemkhrapov2001@yandex.ru>
Co-authored-by: joninco <joninco@bullpoint.org>
Co-authored-by: Ashar <coder3101@users.noreply.github.com>
This commit is contained in:
Philipp Emanuel Weidmann
2026-10-01 14:30:07 +05:30
committed by GitHub
co-authored by kabachuha joninco Ashar
parent 71e6d5eb38
commit 662e4ba27e
33 changed files with 1944 additions and 1738 deletions
+7
View File
@@ -0,0 +1,7 @@
2f1b4d75d067bae3fe44e676721c7f077d243bc007156cb9c2f8b5836613d082 *chat_template.jinja
ca80080dfa4ec6ba87152fa2b9afe70b90c400e5c4b1d6bdc3aa3114467ca68f *config.json
70070bac883cf9c39b5992450d6b23cd160eaf33099e24c654e0359d2f87c760 *generation_config.json
9ff0593e3fbd0ba463bbc980ebb3ed34798e562b606f90cd93c2df5403732c7b *model.safetensors
32bdf45d2ad4cc29a0822ddd157a182de76644f0419a6228d151495256e9813c *processor_config.json
cc8d3a0ce36466ccc1278bf987df5f71db1719b9ca6b4118264f45cb627bfe0f *tokenizer.json
a1bab8c81ed15fa6ce912ec993c66cb49392e0487fb1ea5f5f11ea3618683627 *tokenizer_config.json
+20 -3
View File
@@ -1,4 +1,4 @@
# This test case is for Hybrid-Edge models.
# This test case is for hybrid models.
# After any change related to it, this test should PASS.
model = "tiny-random/gemma-4e"
@@ -9,6 +9,11 @@ print_debug_information = true
batch_size = 2
max_response_length = 10
modifiers = [
{ plugin = "heretic.modifiers.abliteration.Abliteration" },
]
n_trials = 2
n_startup_trials = 1
@@ -18,13 +23,13 @@ trial_index = 0
model_action = "save"
save_directory = "model"
[good_prompts]
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[bad_prompts]
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
@@ -41,3 +46,15 @@ dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "test[:5]"
column = "text"
[modifier.Abliteration.good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[modifier.Abliteration.bad_prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"
+6
View File
@@ -0,0 +1,6 @@
f8d9255777615591a7cc1a7c932f5a69e181128902295e1b81221d20d983cac7 *chat_template.jinja
91d2a5190c7ea0f74ed499428d4ad62b5208d63b36f7bbb562d15f4be25bd5c2 *config.json
dd6034a30113decdfaf8886622e20eb9e2e02d3f774918d474a4e26cfd7fbba8 *generation_config.json
aefe8b9c4b4969f6d13c5d778760f3dce4e25134324b33677934550d9df02a7c *model.safetensors
fce342a4642cb8afc42d8d89cfa21198b64a43458ded7f6ff28d1151a08c9cda *tokenizer.json
9ba5fa877168e24823cb583c55b4c2e4df0331f30084953c7cf07de294640384 *tokenizer_config.json
+56
View File
@@ -0,0 +1,56 @@
# This test case is for ARA.
# After any change related to it, this test should PASS.
model = "tiny-random/gpt-oss"
model_commit = "02ba5c61f879b5a38a8b1f7a8e0409b8e1bb8f38"
seed = 12345
print_debug_information = true
batch_size = 2
max_response_length = 10
n_trials = 2
n_startup_trials = 1
export_strategy = "merge"
checkpoint_action = "restart"
trial_index = 0
model_action = "save"
save_directory = "model"
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"
[scorer.KLDivergence.prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "test[:5]"
column = "text"
[scorer.KeywordRate.prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "test[:5]"
column = "text"
[modifier.ARA.good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[modifier.ARA.bad_prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"
+22 -4
View File
@@ -9,6 +9,11 @@ print_debug_information = true
batch_size = 2
max_response_length = 10
modifiers = [
{ plugin = "heretic.modifiers.abliteration.Abliteration" },
]
n_trials = 2
n_startup_trials = 1
@@ -18,15 +23,13 @@ trial_index = 0
model_action = "save"
save_directory = "model"
row_normalization = "none"
[good_prompts]
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[bad_prompts]
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
@@ -43,3 +46,18 @@ dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "test[:5]"
column = "text"
[modifier.Abliteration]
row_normalization = "none"
[modifier.Abliteration.good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[modifier.Abliteration.bad_prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"
+20 -3
View File
@@ -1,4 +1,4 @@
# This test case is for Dense models.
# This test case is for dense models.
# After any change related to it, this test should PASS.
model = "tiny-random/mistral-3"
@@ -9,6 +9,11 @@ print_debug_information = true
batch_size = 2
max_response_length = 10
modifiers = [
{ plugin = "heretic.modifiers.abliteration.Abliteration" },
]
n_trials = 2
n_startup_trials = 1
@@ -18,13 +23,13 @@ trial_index = 0
model_action = "save"
save_directory = "model"
[good_prompts]
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[bad_prompts]
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
@@ -41,3 +46,15 @@ dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "test[:5]"
column = "text"
[modifier.Abliteration.good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[modifier.Abliteration.bad_prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"
+22 -4
View File
@@ -9,6 +9,11 @@ print_debug_information = true
batch_size = 2
max_response_length = 10
modifiers = [
{ plugin = "heretic.modifiers.abliteration.Abliteration" },
]
n_trials = 2
n_startup_trials = 1
@@ -18,15 +23,13 @@ trial_index = 0
model_action = "save"
save_directory = "model"
row_normalization = "pre"
[good_prompts]
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[bad_prompts]
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
@@ -43,3 +46,18 @@ dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "test[:5]"
column = "text"
[modifier.Abliteration]
row_normalization = "pre"
[modifier.Abliteration.good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[modifier.Abliteration.bad_prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"
+19 -2
View File
@@ -9,6 +9,11 @@ print_debug_information = true
batch_size = 2
max_response_length = 10
modifiers = [
{ plugin = "heretic.modifiers.abliteration.Abliteration" },
]
n_trials = 2
n_startup_trials = 1
@@ -18,13 +23,13 @@ trial_index = 0
model_action = "save"
save_directory = "model"
[good_prompts]
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[bad_prompts]
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
@@ -41,3 +46,15 @@ dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "test[:5]"
column = "text"
[modifier.Abliteration.good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[modifier.Abliteration.bad_prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"
+6
View File
@@ -0,0 +1,6 @@
39a23a1a28f68acd747373fa688bea630021ec4db2bc8dec2fd3962240fb0418 *chat_template.jinja
35e3ad5aeaa78984629ceba6c22a4236b4c21d6959485900da2a5a98ea7e4062 *config.json
2ec7a17f55287d482ca956f5acff44d9e88f6c7fd5ec06c7c33d641afa9af330 *generation_config.json
0cdea9064dcbe6db666f9d42a283d664c133d4c67a2f6ea0fd863a54f522160c *model.safetensors
3cf3a6d9520f195638a36f0194239d817de7288710bca55f1f5753de226748f7 *tokenizer.json
388b47e61cb40f2fd51a89999053686ab4c45b40b43c0329d15645f5910d069e *tokenizer_config.json
+65
View File
@@ -0,0 +1,65 @@
# This test case is for ARA (non-standard settings).
# After any change related to it, this test should PASS.
model = "tiny-random/seed-oss"
model_commit = "6860befd78b678885f7a52bbf41d7fd0671af2db"
seed = 12345
print_debug_information = true
batch_size = 2
max_response_length = 10
scorers = [
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize" },
{ plugin = "heretic.scorers.kl_divergence.KLDivergence", optimization = "maximize" },
]
n_trials = 2
n_startup_trials = 1
export_strategy = "merge"
checkpoint_action = "restart"
trial_index = 0
model_action = "save"
save_directory = "model"
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"
[scorer.KLDivergence.prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "test[:5]"
column = "text"
[scorer.KeywordRate.prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "test[:5]"
column = "text"
[modifier.ARA]
preserve_row_magnitudes = false
lora_rank = 20
[modifier.ARA.good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[modifier.ARA.bad_prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"