mirror of
https://github.com/p-e-w/heretic.git
synced 2026-10-02 08:51:27 -07:00
feat: support specifying a dataset's specific config/subset (#445)
* feat: Allow specifying a specific config/subset name for the datasets. This would be useful for using a single dataset that has harmful/harmless prompt pairs in different languages stored in different configs/subsets. * fix: setting config/subset value when loading the dataset. * fix: minor changes
This commit is contained in:
@@ -54,6 +54,14 @@ class DatasetSpecification(BaseModel):
|
||||
description="Hugging Face commit hash of the dataset.",
|
||||
)
|
||||
|
||||
config: str | None = Field(
|
||||
default=None,
|
||||
description=(
|
||||
"Dataset config/subset name. Each config can have its own split. "
|
||||
"Used to load a specific config of a dataset that has multiple configurations."
|
||||
),
|
||||
)
|
||||
|
||||
split: str | None = Field(
|
||||
default=None,
|
||||
description="Portion of the dataset to use. Required for datasets, optional for plain text files.",
|
||||
|
||||
@@ -208,6 +208,7 @@ def load_prompts(
|
||||
)
|
||||
dataset = load_dataset(
|
||||
path,
|
||||
name=specification.config,
|
||||
revision=specification.commit,
|
||||
split=split_str,
|
||||
)
|
||||
@@ -225,6 +226,7 @@ def load_prompts(
|
||||
# Path should be a local directory.
|
||||
dataset = load_dataset(
|
||||
path,
|
||||
name=specification.config,
|
||||
split=split_str,
|
||||
# Don't require the number of examples (lines) per split to be pre-defined.
|
||||
verification_mode=VerificationMode.NO_CHECKS,
|
||||
|
||||
Reference in New Issue
Block a user