mirror of
https://github.com/p-e-w/heretic.git
synced 2026-10-02 00:41:26 -07:00
feat: add tests for ARA
This commit is contained in:
+2
-2
@@ -55,12 +55,12 @@ dataset = "mlabonne/harmless_alpaca"
|
|||||||
split = "test[:100]"
|
split = "test[:100]"
|
||||||
column = "text"
|
column = "text"
|
||||||
|
|
||||||
[modifier.Abliteration.good_prompts]
|
[modifier.ARA.good_prompts]
|
||||||
dataset = "mlabonne/harmless_alpaca"
|
dataset = "mlabonne/harmless_alpaca"
|
||||||
split = "train[:400]"
|
split = "train[:400]"
|
||||||
column = "text"
|
column = "text"
|
||||||
|
|
||||||
[modifier.Abliteration.bad_prompts]
|
[modifier.ARA.bad_prompts]
|
||||||
dataset = "UnstableLlama/jokes"
|
dataset = "UnstableLlama/jokes"
|
||||||
split = "train[:200]"
|
split = "train[:200]"
|
||||||
column = "text"
|
column = "text"
|
||||||
|
|||||||
+2
-2
@@ -147,13 +147,13 @@ split = "train[1000:1100]"
|
|||||||
column = "prompt"
|
column = "prompt"
|
||||||
prefix = "Write a short story based on the writing prompt below. Avoid literary cliches, purple prose, and flowery language.\n\nWriting prompt:"
|
prefix = "Write a short story based on the writing prompt below. Avoid literary cliches, purple prose, and flowery language.\n\nWriting prompt:"
|
||||||
|
|
||||||
[modifier.Abliteration.good_prompts]
|
[modifier.ARA.good_prompts]
|
||||||
dataset = "llm-aes/writing-prompts"
|
dataset = "llm-aes/writing-prompts"
|
||||||
split = "train[:500]"
|
split = "train[:500]"
|
||||||
column = "prompt"
|
column = "prompt"
|
||||||
prefix = "Write a short story based on the writing prompt below. Avoid literary cliches, purple prose, and flowery language.\n\nWriting prompt:"
|
prefix = "Write a short story based on the writing prompt below. Avoid literary cliches, purple prose, and flowery language.\n\nWriting prompt:"
|
||||||
|
|
||||||
[modifier.Abliteration.bad_prompts]
|
[modifier.ARA.bad_prompts]
|
||||||
dataset = "llm-aes/writing-prompts"
|
dataset = "llm-aes/writing-prompts"
|
||||||
split = "train[:500]"
|
split = "train[:500]"
|
||||||
column = "prompt"
|
column = "prompt"
|
||||||
|
|||||||
@@ -0,0 +1,6 @@
|
|||||||
|
f8d9255777615591a7cc1a7c932f5a69e181128902295e1b81221d20d983cac7 *chat_template.jinja
|
||||||
|
91d2a5190c7ea0f74ed499428d4ad62b5208d63b36f7bbb562d15f4be25bd5c2 *config.json
|
||||||
|
dd6034a30113decdfaf8886622e20eb9e2e02d3f774918d474a4e26cfd7fbba8 *generation_config.json
|
||||||
|
aefe8b9c4b4969f6d13c5d778760f3dce4e25134324b33677934550d9df02a7c *model.safetensors
|
||||||
|
fce342a4642cb8afc42d8d89cfa21198b64a43458ded7f6ff28d1151a08c9cda *tokenizer.json
|
||||||
|
9ba5fa877168e24823cb583c55b4c2e4df0331f30084953c7cf07de294640384 *tokenizer_config.json
|
||||||
@@ -0,0 +1,56 @@
|
|||||||
|
# This test case is for ARA.
|
||||||
|
# After any change related to it, this test should PASS.
|
||||||
|
|
||||||
|
model = "tiny-random/gpt-oss"
|
||||||
|
model_commit = "02ba5c61f879b5a38a8b1f7a8e0409b8e1bb8f38"
|
||||||
|
|
||||||
|
seed = 12345
|
||||||
|
print_debug_information = true
|
||||||
|
|
||||||
|
batch_size = 2
|
||||||
|
max_response_length = 10
|
||||||
|
|
||||||
|
n_trials = 2
|
||||||
|
n_startup_trials = 1
|
||||||
|
|
||||||
|
export_strategy = "merge"
|
||||||
|
checkpoint_action = "restart"
|
||||||
|
trial_index = 0
|
||||||
|
model_action = "save"
|
||||||
|
save_directory = "model"
|
||||||
|
|
||||||
|
[[response_prefix_test_prompts]]
|
||||||
|
dataset = "mlabonne/harmless_alpaca"
|
||||||
|
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
|
||||||
|
split = "train[:5]"
|
||||||
|
column = "text"
|
||||||
|
|
||||||
|
[[response_prefix_test_prompts]]
|
||||||
|
dataset = "mlabonne/harmful_behaviors"
|
||||||
|
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
|
||||||
|
split = "train[:5]"
|
||||||
|
column = "text"
|
||||||
|
|
||||||
|
[scorer.KLDivergence.prompts]
|
||||||
|
dataset = "mlabonne/harmless_alpaca"
|
||||||
|
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
|
||||||
|
split = "test[:5]"
|
||||||
|
column = "text"
|
||||||
|
|
||||||
|
[scorer.KeywordRate.prompts]
|
||||||
|
dataset = "mlabonne/harmful_behaviors"
|
||||||
|
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
|
||||||
|
split = "test[:5]"
|
||||||
|
column = "text"
|
||||||
|
|
||||||
|
[modifier.ARA.good_prompts]
|
||||||
|
dataset = "mlabonne/harmless_alpaca"
|
||||||
|
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
|
||||||
|
split = "train[:5]"
|
||||||
|
column = "text"
|
||||||
|
|
||||||
|
[modifier.ARA.bad_prompts]
|
||||||
|
dataset = "mlabonne/harmful_behaviors"
|
||||||
|
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
|
||||||
|
split = "train[:5]"
|
||||||
|
column = "text"
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
39a23a1a28f68acd747373fa688bea630021ec4db2bc8dec2fd3962240fb0418 *chat_template.jinja
|
||||||
|
35e3ad5aeaa78984629ceba6c22a4236b4c21d6959485900da2a5a98ea7e4062 *config.json
|
||||||
|
2ec7a17f55287d482ca956f5acff44d9e88f6c7fd5ec06c7c33d641afa9af330 *generation_config.json
|
||||||
|
0cdea9064dcbe6db666f9d42a283d664c133d4c67a2f6ea0fd863a54f522160c *model.safetensors
|
||||||
|
3cf3a6d9520f195638a36f0194239d817de7288710bca55f1f5753de226748f7 *tokenizer.json
|
||||||
|
388b47e61cb40f2fd51a89999053686ab4c45b40b43c0329d15645f5910d069e *tokenizer_config.json
|
||||||
@@ -0,0 +1,65 @@
|
|||||||
|
# This test case is for ARA (non-standard settings).
|
||||||
|
# After any change related to it, this test should PASS.
|
||||||
|
|
||||||
|
model = "tiny-random/seed-oss"
|
||||||
|
model_commit = "6860befd78b678885f7a52bbf41d7fd0671af2db"
|
||||||
|
|
||||||
|
seed = 12345
|
||||||
|
print_debug_information = true
|
||||||
|
|
||||||
|
batch_size = 2
|
||||||
|
max_response_length = 10
|
||||||
|
|
||||||
|
scorers = [
|
||||||
|
{ plugin = "heretic.scorers.keyword_rate.KeywordRate", optimization = "minimize" },
|
||||||
|
{ plugin = "heretic.scorers.kl_divergence.KLDivergence", optimization = "maximize" },
|
||||||
|
]
|
||||||
|
|
||||||
|
n_trials = 2
|
||||||
|
n_startup_trials = 1
|
||||||
|
|
||||||
|
export_strategy = "merge"
|
||||||
|
checkpoint_action = "restart"
|
||||||
|
trial_index = 0
|
||||||
|
model_action = "save"
|
||||||
|
save_directory = "model"
|
||||||
|
|
||||||
|
[[response_prefix_test_prompts]]
|
||||||
|
dataset = "mlabonne/harmless_alpaca"
|
||||||
|
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
|
||||||
|
split = "train[:5]"
|
||||||
|
column = "text"
|
||||||
|
|
||||||
|
[[response_prefix_test_prompts]]
|
||||||
|
dataset = "mlabonne/harmful_behaviors"
|
||||||
|
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
|
||||||
|
split = "train[:5]"
|
||||||
|
column = "text"
|
||||||
|
|
||||||
|
[scorer.KLDivergence.prompts]
|
||||||
|
dataset = "mlabonne/harmless_alpaca"
|
||||||
|
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
|
||||||
|
split = "test[:5]"
|
||||||
|
column = "text"
|
||||||
|
|
||||||
|
[scorer.KeywordRate.prompts]
|
||||||
|
dataset = "mlabonne/harmful_behaviors"
|
||||||
|
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
|
||||||
|
split = "test[:5]"
|
||||||
|
column = "text"
|
||||||
|
|
||||||
|
[modifier.ARA]
|
||||||
|
preserve_row_magnitudes = false
|
||||||
|
lora_rank = 20
|
||||||
|
|
||||||
|
[modifier.ARA.good_prompts]
|
||||||
|
dataset = "mlabonne/harmless_alpaca"
|
||||||
|
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
|
||||||
|
split = "train[:5]"
|
||||||
|
column = "text"
|
||||||
|
|
||||||
|
[modifier.ARA.bad_prompts]
|
||||||
|
dataset = "mlabonne/harmful_behaviors"
|
||||||
|
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
|
||||||
|
split = "train[:5]"
|
||||||
|
column = "text"
|
||||||
Reference in New Issue
Block a user