From ab43ebda03c21e11dd6e298ac6ea65d0bff8d1f8 Mon Sep 17 00:00:00 2001 From: Hasan Demirkiran Date: Wed, 19 Aug 2026 10:53:15 +0200 Subject: [PATCH 1/3] feat: add Hugging Face dataset export --- README.md | 16 +++- huggingface/README.md | 118 ++++++++++++++++++++++++++ shellrisk_bench/export_huggingface.py | 115 +++++++++++++++++++++++++ tests/test_export_huggingface.py | 64 ++++++++++++++ 4 files changed, 312 insertions(+), 1 deletion(-) create mode 100644 huggingface/README.md create mode 100644 shellrisk_bench/export_huggingface.py create mode 100644 tests/test_export_huggingface.py diff --git a/README.md b/README.md index 146a1dc..d8b27b9 100644 --- a/README.md +++ b/README.md @@ -90,10 +90,24 @@ Prediction rows contain a stable command hash and a binary verdict: {"id":"sha256:…","prediction":"risky"} ``` +## Hugging Face export + +Create the files used by the Hugging Face Dataset Viewer and +`datasets.load_dataset()` from the verified local split: + +```bash +.venv/bin/python -m shellrisk_bench.export_huggingface +``` + +This writes `dist/huggingface/README.md`, one Parquet file per split, the +canonical split manifest, and an export manifest containing file hashes. The +command does not upload anything. The dataset card is maintained under +[`huggingface/README.md`](huggingface/README.md). + ## Safety This repository processes potentially destructive commands as inert text. Nothing in the build or evaluation path executes benchmark commands. Do not pipe dataset contents into a shell. ## License -The benchmark code is licensed under Apache-2.0. Upstream datasets retain their own licenses and terms; see [DATASETS.md](DATASETS.md). The generated dataset is intentionally git-ignored and is not redistributed here. +The benchmark code is licensed under Apache-2.0. Upstream datasets retain their own licenses and terms; see [DATASETS.md](DATASETS.md). Generated data and Hub export files are git-ignored; publishing them requires a separate source-by-source redistribution review. diff --git a/huggingface/README.md b/huggingface/README.md new file mode 100644 index 0000000..7470900 --- /dev/null +++ b/huggingface/README.md @@ -0,0 +1,118 @@ +--- +pretty_name: ShellRisk-Bench +language: +- en +license: other +task_categories: +- text-classification +tags: +- shell +- bash +- cybersecurity +- safety +- benchmark +size_categories: +- 10K str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _jsonl_rows(path: Path) -> list[dict]: + with path.open(encoding="utf-8") as handle: + return [json.loads(line) for line in handle if line.strip()] + + +def _verify_input(split_dir: Path, manifest: dict) -> None: + for split in SPLITS: + source_path = split_dir / f"{split}.jsonl" + if not source_path.exists(): + raise FileNotFoundError(f"missing {source_path}") + expected = manifest.get(split, {}).get("sha256") + actual = _sha256(source_path) + if expected != actual: + raise ValueError( + f"{split} checksum mismatch: expected {expected!r}, got {actual!r}; " + "rebuild and verify the frozen split before exporting" + ) + + +def export(split_dir: Path = DEFAULT_SPLIT_DIR, output_dir: Path = DEFAULT_OUTPUT_DIR) -> dict: + """Write a viewer-compatible dataset repository without uploading it.""" + manifest_path = split_dir / "manifest.json" + if not manifest_path.exists(): + raise FileNotFoundError(f"missing {manifest_path}") + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + _verify_input(split_dir, manifest) + + data_dir = output_dir / "data" + data_dir.mkdir(parents=True, exist_ok=True) + shutil.copyfile(CARD_PATH, output_dir / "README.md") + shutil.copyfile(manifest_path, output_dir / "split-manifest.json") + + exported: dict[str, dict] = {} + for split in SPLITS: + rows = _jsonl_rows(split_dir / f"{split}.jsonl") + expected_rows = manifest[split]["n"] + if len(rows) != expected_rows: + raise ValueError( + f"{split} row count mismatch: expected {expected_rows}, got {len(rows)}" + ) + output_path = data_dir / f"{split}-00000-of-00001.parquet" + temporary_path = output_path.with_suffix(".parquet.tmp") + try: + Dataset.from_list(rows).to_parquet(temporary_path) + temporary_path.replace(output_path) + finally: + temporary_path.unlink(missing_ok=True) + exported[split] = { + "path": str(output_path.relative_to(output_dir)), + "rows": len(rows), + "sha256": _sha256(output_path), + } + + export_manifest = { + "benchmark": manifest["benchmark"], + "version": manifest["version"], + "source_manifest_sha256": _sha256(manifest_path), + "files": exported, + } + (output_dir / "export-manifest.json").write_text( + json.dumps(export_manifest, indent=2, sort_keys=True) + "\n", + encoding="utf-8", + ) + return export_manifest + + +def verify_loadable(output_dir: Path = DEFAULT_OUTPUT_DIR) -> dict[str, int]: + """Load the exported Parquet files using the public consumer API.""" + files = { + split: str(output_dir / "data" / f"{split}-00000-of-00001.parquet") + for split in SPLITS + } + dataset = load_dataset("parquet", data_files=files) + return {split: dataset[split].num_rows for split in SPLITS} + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--split-dir", type=Path, default=DEFAULT_SPLIT_DIR) + parser.add_argument("--output-dir", type=Path, default=DEFAULT_OUTPUT_DIR) + args = parser.parse_args() + exported = export(args.split_dir, args.output_dir) + loaded = verify_loadable(args.output_dir) + print(json.dumps({"export": exported, "loaded_rows": loaded}, indent=2, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/tests/test_export_huggingface.py b/tests/test_export_huggingface.py new file mode 100644 index 0000000..68305a4 --- /dev/null +++ b/tests/test_export_huggingface.py @@ -0,0 +1,64 @@ +import hashlib +import json +from pathlib import Path + +import pytest + +from shellrisk_bench.export_huggingface import export, verify_loadable + + +def _write_jsonl(path: Path, rows: list[dict]) -> str: + path.write_text( + "".join(json.dumps(row, sort_keys=True) + "\n" for row in rows), + encoding="utf-8", + ) + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _row(index: int, label: str) -> dict: + return { + "id": f"sha256:{index:064x}", + "source": "fixture", + "upstream_id": f"fixture-{index:06d}", + "command": f"printf fixture-{index}", + "label": label, + } + + +def _split_fixture(tmp_path: Path) -> Path: + split_dir = tmp_path / "splits" + split_dir.mkdir() + train = [_row(1, "not_risky"), _row(2, "risky")] + test = [_row(3, "not_risky")] + train_sha = _write_jsonl(split_dir / "train.jsonl", train) + test_sha = _write_jsonl(split_dir / "test.jsonl", test) + manifest = { + "benchmark": "ShellRisk-Bench", + "version": "0.1.0", + "train": {"n": len(train), "sha256": train_sha}, + "test": {"n": len(test), "sha256": test_sha}, + } + (split_dir / "manifest.json").write_text(json.dumps(manifest), encoding="utf-8") + return split_dir + + +def test_export_is_loadable_by_datasets(tmp_path: Path) -> None: + split_dir = _split_fixture(tmp_path) + output_dir = tmp_path / "hub" + + manifest = export(split_dir, output_dir) + + assert manifest["files"]["train"]["rows"] == 2 + assert manifest["files"]["test"]["rows"] == 1 + assert verify_loadable(output_dir) == {"train": 2, "test": 1} + assert (output_dir / "README.md").exists() + assert (output_dir / "split-manifest.json").exists() + + +def test_export_rejects_checksum_mismatch(tmp_path: Path) -> None: + split_dir = _split_fixture(tmp_path) + with (split_dir / "test.jsonl").open("a", encoding="utf-8") as handle: + handle.write(json.dumps(_row(4, "risky")) + "\n") + + with pytest.raises(ValueError, match="test checksum mismatch"): + export(split_dir, tmp_path / "hub") From 02dad01447a090862765b1c0b1e31109bb8791de Mon Sep 17 00:00:00 2001 From: Hasan Demirkiran Date: Wed, 19 Aug 2026 11:18:14 +0200 Subject: [PATCH 2/3] docs: clarify Hugging Face data release status --- huggingface/README.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/huggingface/README.md b/huggingface/README.md index 7470900..7d7e6c9 100644 --- a/huggingface/README.md +++ b/huggingface/README.md @@ -29,6 +29,10 @@ classification of individual shell-command submissions. It asks whether a command poses meaningful cyber or system risk when evaluated without task, user, or session context. +**Release status:** The dataset card is public, but the Parquet splits are not +yet published while source-by-source redistribution terms are reviewed. The +`load_dataset()` example below becomes available with the data release. + The benchmark contains a deterministic train split of 16,772 rows and test split of 4,194 rows. The test set contains 193 `risky` and 4,001 `not_risky` examples. From 7249edce943e331948cac0252a9354ba93b8f3ff Mon Sep 17 00:00:00 2001 From: Hasan Demirkiran Date: Wed, 19 Aug 2026 11:41:37 +0200 Subject: [PATCH 3/3] docs: mark Hugging Face dataset released --- huggingface/README.md | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/huggingface/README.md b/huggingface/README.md index 7d7e6c9..e175197 100644 --- a/huggingface/README.md +++ b/huggingface/README.md @@ -29,9 +29,8 @@ classification of individual shell-command submissions. It asks whether a command poses meaningful cyber or system risk when evaluated without task, user, or session context. -**Release status:** The dataset card is public, but the Parquet splits are not -yet published while source-by-source redistribution terms are reviewed. The -`load_dataset()` example below becomes available with the data release. +**Release:** The v0.1 Parquet train and test splits are publicly available +through Dataset Viewer and `load_dataset()`. The benchmark contains a deterministic train split of 16,772 rows and test split of 4,194 rows. The test set contains 193 `risky` and 4,001 `not_risky`