Files
heretic/config.nohumor.toml
T
662e4ba27e feat: modifier plugins + ARA (#446)
* feat: add modifier base class

* feat: re-implement abliteration as a modifier plugin

* fix: adjust good/bad prompts hack to match tests

* feat: support dataset specifications containing multiple individual datasets

* feat: re-implement ARA as a modifier plugin

Arbitrary-Rank Ablation (ARA) (Weidmann 2026) was originally introduced by @p-e-w in #211

Co-authored-by: kabachuha <artemkhrapov2001@yandex.ru>
Co-authored-by: joninco <joninco@bullpoint.org>
Co-authored-by: Ashar <coder3101@users.noreply.github.com>

* feat: add tests for ARA

* feat: reduce default number of trials

* ci: remove broken "semantic-pull-request" workflow

* ci: add yet another alternative model hash

---------

Co-authored-by: kabachuha <artemkhrapov2001@yandex.ru>
Co-authored-by: joninco <joninco@bullpoint.org>
Co-authored-by: Ashar <coder3101@users.noreply.github.com>
2026-10-01 14:30:07 +05:30

67 lines
1.2 KiB
TOML

# Rename this file to config.toml, place it in the working directory
# that you run Heretic from, and edit the configuration to your liking.
max_response_length = 300
[scorer.KeywordRate]
score_name = "Responses with humor"
keyword_markers = [
"😅",
"here's one",
"why did",
"walked into a bar",
"gets worse",
"the punchline",
"classic",
"dad joke",
"bad joke",
"pun intended",
"ba dum tss",
"rimshot",
"😂",
"😄",
"😆",
"haha",
"hehe",
"lol",
"funny",
"joke",
"humor",
"that's hilarious",
"you could say",
"one-liner",
"comedian",
"stand-up",
"unexpectedly",
"because apparently",
"to be fair",
"on the bright side",
"lmao",
"omg",
"rofl",
"silly",
"humorous",
"clever",
]
[scorer.KeywordRate.prompts]
dataset = "UnstableLlama/jokes"
split = "train[200:250]"
column = "text"
[scorer.KLDivergence.prompts]
dataset = "mlabonne/harmless_alpaca"
split = "test[:100]"
column = "text"
[modifier.ARA.good_prompts]
dataset = "mlabonne/harmless_alpaca"
split = "train[:400]"
column = "text"
[modifier.ARA.bad_prompts]
dataset = "UnstableLlama/jokes"
split = "train[:200]"
column = "text"