feat: modifier plugins + ARA (#446)

* feat: add modifier base class

* feat: re-implement abliteration as a modifier plugin

* fix: adjust good/bad prompts hack to match tests

* feat: support dataset specifications containing multiple individual datasets

* feat: re-implement ARA as a modifier plugin

Arbitrary-Rank Ablation (ARA) (Weidmann 2026) was originally introduced by @p-e-w in #211

Co-authored-by: kabachuha <artemkhrapov2001@yandex.ru>
Co-authored-by: joninco <joninco@bullpoint.org>
Co-authored-by: Ashar <coder3101@users.noreply.github.com>

* feat: add tests for ARA

* feat: reduce default number of trials

* ci: remove broken "semantic-pull-request" workflow

* ci: add yet another alternative model hash

---------

Co-authored-by: kabachuha <artemkhrapov2001@yandex.ru>
Co-authored-by: joninco <joninco@bullpoint.org>
Co-authored-by: Ashar <coder3101@users.noreply.github.com>
This commit is contained in:
Philipp Emanuel Weidmann
2026-10-01 14:30:07 +05:30
committed by GitHub
co-authored by kabachuha joninco Ashar
parent 71e6d5eb38
commit 662e4ba27e
33 changed files with 1944 additions and 1738 deletions
+6
View File
@@ -0,0 +1,6 @@
f8d9255777615591a7cc1a7c932f5a69e181128902295e1b81221d20d983cac7 *chat_template.jinja
91d2a5190c7ea0f74ed499428d4ad62b5208d63b36f7bbb562d15f4be25bd5c2 *config.json
dd6034a30113decdfaf8886622e20eb9e2e02d3f774918d474a4e26cfd7fbba8 *generation_config.json
aefe8b9c4b4969f6d13c5d778760f3dce4e25134324b33677934550d9df02a7c *model.safetensors
fce342a4642cb8afc42d8d89cfa21198b64a43458ded7f6ff28d1151a08c9cda *tokenizer.json
9ba5fa877168e24823cb583c55b4c2e4df0331f30084953c7cf07de294640384 *tokenizer_config.json
+56
View File
@@ -0,0 +1,56 @@
# This test case is for ARA.
# After any change related to it, this test should PASS.
model = "tiny-random/gpt-oss"
model_commit = "02ba5c61f879b5a38a8b1f7a8e0409b8e1bb8f38"
seed = 12345
print_debug_information = true
batch_size = 2
max_response_length = 10
n_trials = 2
n_startup_trials = 1
export_strategy = "merge"
checkpoint_action = "restart"
trial_index = 0
model_action = "save"
save_directory = "model"
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"
[scorer.KLDivergence.prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "test[:5]"
column = "text"
[scorer.KeywordRate.prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "test[:5]"
column = "text"
[modifier.ARA.good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[modifier.ARA.bad_prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"