Files
heretic/tests/qwen2.5/config.toml
Philipp Emanuel Weidmann 95dda4c4db feat: add benchmark scorer (#444)
* fix: improve print output of scorers

* feat: add benchmark scorer
2026-09-03 17:49:46 +05:30

46 lines
1.0 KiB
TOML

# This test case is for row_normalization="pre".
# After any change related to it, this test should PASS.
model = "tiny-random/qwen2.5"
model_commit = "7a6a3128ee4137a248d6d1582824592b87a81647"
seed = 12345
print_debug_information = true
batch_size = 2
max_response_length = 10
n_trials = 2
n_startup_trials = 1
export_strategy = "merge"
checkpoint_action = "restart"
trial_index = 0
model_action = "save"
save_directory = "model"
row_normalization = "pre"
[good_prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "train[:5]"
column = "text"
[bad_prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "train[:5]"
column = "text"
[scorer.KLDivergence.prompts]
dataset = "mlabonne/harmless_alpaca"
commit = "02c6a92cfcf11bb0c387334f8146d149d65b587f"
split = "test[:5]"
column = "text"
[scorer.KeywordRate.prompts]
dataset = "mlabonne/harmful_behaviors"
commit = "01cead01398926d81f7c52bdb790ee8cf77ebba7"
split = "test[:5]"
column = "text"