mirror of
https://github.com/p-e-w/heretic.git
synced 2026-09-29 15:31:25 -07:00
feat: support specifying a dataset's specific config/subset (#445)
* feat: Allow specifying a specific config/subset name for the datasets. This would be useful for using a single dataset that has harmful/harmless prompt pairs in different languages stored in different configs/subsets. * fix: setting config/subset value when loading the dataset. * fix: minor changes
This commit is contained in:
@@ -137,6 +137,8 @@ system_prompt = "You are a helpful assistant."
|
|||||||
# or a path to a plain text file with one prompt per line (empty lines are ignored).
|
# or a path to a plain text file with one prompt per line (empty lines are ignored).
|
||||||
# For text files, "column" is ignored and "split" is optional; when given, it selects
|
# For text files, "column" is ignored and "split" is optional; when given, it selects
|
||||||
# a subset of the lines using slice notation (e.g. "[:400]").
|
# a subset of the lines using slice notation (e.g. "[:400]").
|
||||||
|
# "config" specifies a dataset's specific config/subset name (e.g. "english", "hindi").
|
||||||
|
# Leave unset for datasets with a single configuration.
|
||||||
|
|
||||||
# Dataset of prompts that tend to not result in refusals (used for calculating residual directions).
|
# Dataset of prompts that tend to not result in refusals (used for calculating residual directions).
|
||||||
[good_prompts]
|
[good_prompts]
|
||||||
|
|||||||
@@ -54,6 +54,14 @@ class DatasetSpecification(BaseModel):
|
|||||||
description="Hugging Face commit hash of the dataset.",
|
description="Hugging Face commit hash of the dataset.",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
config: str | None = Field(
|
||||||
|
default=None,
|
||||||
|
description=(
|
||||||
|
"Dataset config/subset name. Each config can have its own split. "
|
||||||
|
"Used to load a specific config of a dataset that has multiple configurations."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
split: str | None = Field(
|
split: str | None = Field(
|
||||||
default=None,
|
default=None,
|
||||||
description="Portion of the dataset to use. Required for datasets, optional for plain text files.",
|
description="Portion of the dataset to use. Required for datasets, optional for plain text files.",
|
||||||
|
|||||||
@@ -208,6 +208,7 @@ def load_prompts(
|
|||||||
)
|
)
|
||||||
dataset = load_dataset(
|
dataset = load_dataset(
|
||||||
path,
|
path,
|
||||||
|
name=specification.config,
|
||||||
revision=specification.commit,
|
revision=specification.commit,
|
||||||
split=split_str,
|
split=split_str,
|
||||||
)
|
)
|
||||||
@@ -225,6 +226,7 @@ def load_prompts(
|
|||||||
# Path should be a local directory.
|
# Path should be a local directory.
|
||||||
dataset = load_dataset(
|
dataset = load_dataset(
|
||||||
path,
|
path,
|
||||||
|
name=specification.config,
|
||||||
split=split_str,
|
split=split_str,
|
||||||
# Don't require the number of examples (lines) per split to be pre-defined.
|
# Don't require the number of examples (lines) per split to be pre-defined.
|
||||||
verification_mode=VerificationMode.NO_CHECKS,
|
verification_mode=VerificationMode.NO_CHECKS,
|
||||||
|
|||||||
Reference in New Issue
Block a user