Files
heretic/tests/seed-oss/config.toml
T
662e4ba27e feat: modifier plugins + ARA (#446)
* feat: add modifier base class

* feat: re-implement abliteration as a modifier plugin

* fix: adjust good/bad prompts hack to match tests

* feat: support dataset specifications containing multiple individual datasets

* feat: re-implement ARA as a modifier plugin

Arbitrary-Rank Ablation (ARA) (Weidmann 2026) was originally introduced by @p-e-w in #211

Co-authored-by: kabachuha <artemkhrapov2001@yandex.ru>
Co-authored-by: joninco <joninco@bullpoint.org>
Co-authored-by: Ashar <coder3101@users.noreply.github.com>

* feat: add tests for ARA

* feat: reduce default number of trials

* ci: remove broken "semantic-pull-request" workflow

* ci: add yet another alternative model hash

---------

Co-authored-by: kabachuha <artemkhrapov2001@yandex.ru>
Co-authored-by: joninco <joninco@bullpoint.org>
Co-authored-by: Ashar <coder3101@users.noreply.github.com>
2026-10-01 14:30:07 +05:30

66 lines
1.6 KiB
TOML

# This test case is for ARA (non-standard settings).
# After any change related to it, this test should PASS.
model = "tiny-random/seed-oss"
model_commit = "6860befd78b678885f7a52bbf41d7fd0671af2db"
seed = 12345
print_debug_information = true
batch_size = 2
max_response_length = 10
scorers = [
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize" },
{ plugin = "heretic.scorers.kl_divergence.KLDivergence", optimization = "maximize" },
]
n_trials = 2
n_startup_trials = 1
export_strategy = "merge"
checkpoint_action = "restart"
trial_index = 0
model_action = "save"
save_directory = "model"
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[[response_prefix_test_prompts]]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"
[scorer.KLDivergence.prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "test[:5]"
column = "text"
[scorer.KeywordRate.prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "test[:5]"
column = "text"
[modifier.ARA]
preserve_row_magnitudes = false
lora_rank = 20
[modifier.ARA.good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[modifier.ARA.bad_prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"