From ab43ebda03c21e11dd6e298ac6ea65d0bff8d1f8 Mon Sep 17 00:00:00 2001
From: Hasan Demirkiran
Date: Wed, 19 Aug 2026 10:53:15 +0200
Subject: [PATCH 1/3] feat: add Hugging Face dataset export
---
README.md | 16 +++-
huggingface/README.md | 118 ++++++++++++++++++++++++++
shellrisk_bench/export_huggingface.py | 115 +++++++++++++++++++++++++
tests/test_export_huggingface.py | 64 ++++++++++++++
4 files changed, 312 insertions(+), 1 deletion(-)
create mode 100644 huggingface/README.md
create mode 100644 shellrisk_bench/export_huggingface.py
create mode 100644 tests/test_export_huggingface.py
diff --git a/README.md b/README.md
index 146a1dc..d8b27b9 100644
--- a/README.md
+++ b/README.md
@@ -90,10 +90,24 @@ Prediction rows contain a stable command hash and a binary verdict:
{"id":"sha256:…","prediction":"risky"}
```
+## Hugging Face export
+
+Create the files used by the Hugging Face Dataset Viewer and
+`datasets.load_dataset()` from the verified local split:
+
+```bash
+.venv/bin/python -m shellrisk_bench.export_huggingface
+```
+
+This writes `dist/huggingface/README.md`, one Parquet file per split, the
+canonical split manifest, and an export manifest containing file hashes. The
+command does not upload anything. The dataset card is maintained under
+[`huggingface/README.md`](huggingface/README.md).
+
## Safety
This repository processes potentially destructive commands as inert text. Nothing in the build or evaluation path executes benchmark commands. Do not pipe dataset contents into a shell.
## License
-The benchmark code is licensed under Apache-2.0. Upstream datasets retain their own licenses and terms; see [DATASETS.md](DATASETS.md). The generated dataset is intentionally git-ignored and is not redistributed here.
+The benchmark code is licensed under Apache-2.0. Upstream datasets retain their own licenses and terms; see [DATASETS.md](DATASETS.md). Generated data and Hub export files are git-ignored; publishing them requires a separate source-by-source redistribution review.
diff --git a/huggingface/README.md b/huggingface/README.md
new file mode 100644
index 0000000..7470900
--- /dev/null
+++ b/huggingface/README.md
@@ -0,0 +1,118 @@
+---
+pretty_name: ShellRisk-Bench
+language:
+- en
+license: other
+task_categories:
+- text-classification
+tags:
+- shell
+- bash
+- cybersecurity
+- safety
+- benchmark
+size_categories:
+- 10K str:
+ digest = hashlib.sha256()
+ with path.open("rb") as handle:
+ for chunk in iter(lambda: handle.read(1024 * 1024), b""):
+ digest.update(chunk)
+ return digest.hexdigest()
+
+
+def _jsonl_rows(path: Path) -> list[dict]:
+ with path.open(encoding="utf-8") as handle:
+ return [json.loads(line) for line in handle if line.strip()]
+
+
+def _verify_input(split_dir: Path, manifest: dict) -> None:
+ for split in SPLITS:
+ source_path = split_dir / f"{split}.jsonl"
+ if not source_path.exists():
+ raise FileNotFoundError(f"missing {source_path}")
+ expected = manifest.get(split, {}).get("sha256")
+ actual = _sha256(source_path)
+ if expected != actual:
+ raise ValueError(
+ f"{split} checksum mismatch: expected {expected!r}, got {actual!r}; "
+ "rebuild and verify the frozen split before exporting"
+ )
+
+
+def export(split_dir: Path = DEFAULT_SPLIT_DIR, output_dir: Path = DEFAULT_OUTPUT_DIR) -> dict:
+ """Write a viewer-compatible dataset repository without uploading it."""
+ manifest_path = split_dir / "manifest.json"
+ if not manifest_path.exists():
+ raise FileNotFoundError(f"missing {manifest_path}")
+ manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
+ _verify_input(split_dir, manifest)
+
+ data_dir = output_dir / "data"
+ data_dir.mkdir(parents=True, exist_ok=True)
+ shutil.copyfile(CARD_PATH, output_dir / "README.md")
+ shutil.copyfile(manifest_path, output_dir / "split-manifest.json")
+
+ exported: dict[str, dict] = {}
+ for split in SPLITS:
+ rows = _jsonl_rows(split_dir / f"{split}.jsonl")
+ expected_rows = manifest[split]["n"]
+ if len(rows) != expected_rows:
+ raise ValueError(
+ f"{split} row count mismatch: expected {expected_rows}, got {len(rows)}"
+ )
+ output_path = data_dir / f"{split}-00000-of-00001.parquet"
+ temporary_path = output_path.with_suffix(".parquet.tmp")
+ try:
+ Dataset.from_list(rows).to_parquet(temporary_path)
+ temporary_path.replace(output_path)
+ finally:
+ temporary_path.unlink(missing_ok=True)
+ exported[split] = {
+ "path": str(output_path.relative_to(output_dir)),
+ "rows": len(rows),
+ "sha256": _sha256(output_path),
+ }
+
+ export_manifest = {
+ "benchmark": manifest["benchmark"],
+ "version": manifest["version"],
+ "source_manifest_sha256": _sha256(manifest_path),
+ "files": exported,
+ }
+ (output_dir / "export-manifest.json").write_text(
+ json.dumps(export_manifest, indent=2, sort_keys=True) + "\n",
+ encoding="utf-8",
+ )
+ return export_manifest
+
+
+def verify_loadable(output_dir: Path = DEFAULT_OUTPUT_DIR) -> dict[str, int]:
+ """Load the exported Parquet files using the public consumer API."""
+ files = {
+ split: str(output_dir / "data" / f"{split}-00000-of-00001.parquet")
+ for split in SPLITS
+ }
+ dataset = load_dataset("parquet", data_files=files)
+ return {split: dataset[split].num_rows for split in SPLITS}
+
+
+def main() -> None:
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("--split-dir", type=Path, default=DEFAULT_SPLIT_DIR)
+ parser.add_argument("--output-dir", type=Path, default=DEFAULT_OUTPUT_DIR)
+ args = parser.parse_args()
+ exported = export(args.split_dir, args.output_dir)
+ loaded = verify_loadable(args.output_dir)
+ print(json.dumps({"export": exported, "loaded_rows": loaded}, indent=2, sort_keys=True))
+
+
+if __name__ == "__main__":
+ main()
diff --git a/tests/test_export_huggingface.py b/tests/test_export_huggingface.py
new file mode 100644
index 0000000..68305a4
--- /dev/null
+++ b/tests/test_export_huggingface.py
@@ -0,0 +1,64 @@
+import hashlib
+import json
+from pathlib import Path
+
+import pytest
+
+from shellrisk_bench.export_huggingface import export, verify_loadable
+
+
+def _write_jsonl(path: Path, rows: list[dict]) -> str:
+ path.write_text(
+ "".join(json.dumps(row, sort_keys=True) + "\n" for row in rows),
+ encoding="utf-8",
+ )
+ return hashlib.sha256(path.read_bytes()).hexdigest()
+
+
+def _row(index: int, label: str) -> dict:
+ return {
+ "id": f"sha256:{index:064x}",
+ "source": "fixture",
+ "upstream_id": f"fixture-{index:06d}",
+ "command": f"printf fixture-{index}",
+ "label": label,
+ }
+
+
+def _split_fixture(tmp_path: Path) -> Path:
+ split_dir = tmp_path / "splits"
+ split_dir.mkdir()
+ train = [_row(1, "not_risky"), _row(2, "risky")]
+ test = [_row(3, "not_risky")]
+ train_sha = _write_jsonl(split_dir / "train.jsonl", train)
+ test_sha = _write_jsonl(split_dir / "test.jsonl", test)
+ manifest = {
+ "benchmark": "ShellRisk-Bench",
+ "version": "0.1.0",
+ "train": {"n": len(train), "sha256": train_sha},
+ "test": {"n": len(test), "sha256": test_sha},
+ }
+ (split_dir / "manifest.json").write_text(json.dumps(manifest), encoding="utf-8")
+ return split_dir
+
+
+def test_export_is_loadable_by_datasets(tmp_path: Path) -> None:
+ split_dir = _split_fixture(tmp_path)
+ output_dir = tmp_path / "hub"
+
+ manifest = export(split_dir, output_dir)
+
+ assert manifest["files"]["train"]["rows"] == 2
+ assert manifest["files"]["test"]["rows"] == 1
+ assert verify_loadable(output_dir) == {"train": 2, "test": 1}
+ assert (output_dir / "README.md").exists()
+ assert (output_dir / "split-manifest.json").exists()
+
+
+def test_export_rejects_checksum_mismatch(tmp_path: Path) -> None:
+ split_dir = _split_fixture(tmp_path)
+ with (split_dir / "test.jsonl").open("a", encoding="utf-8") as handle:
+ handle.write(json.dumps(_row(4, "risky")) + "\n")
+
+ with pytest.raises(ValueError, match="test checksum mismatch"):
+ export(split_dir, tmp_path / "hub")
From 02dad01447a090862765b1c0b1e31109bb8791de Mon Sep 17 00:00:00 2001
From: Hasan Demirkiran
Date: Wed, 19 Aug 2026 11:18:14 +0200
Subject: [PATCH 2/3] docs: clarify Hugging Face data release status
---
huggingface/README.md | 4 ++++
1 file changed, 4 insertions(+)
diff --git a/huggingface/README.md b/huggingface/README.md
index 7470900..7d7e6c9 100644
--- a/huggingface/README.md
+++ b/huggingface/README.md
@@ -29,6 +29,10 @@ classification of individual shell-command submissions. It asks whether a
command poses meaningful cyber or system risk when evaluated without task,
user, or session context.
+**Release status:** The dataset card is public, but the Parquet splits are not
+yet published while source-by-source redistribution terms are reviewed. The
+`load_dataset()` example below becomes available with the data release.
+
The benchmark contains a deterministic train split of 16,772 rows and test
split of 4,194 rows. The test set contains 193 `risky` and 4,001 `not_risky`
examples.
From 7249edce943e331948cac0252a9354ba93b8f3ff Mon Sep 17 00:00:00 2001
From: Hasan Demirkiran
Date: Wed, 19 Aug 2026 11:41:37 +0200
Subject: [PATCH 3/3] docs: mark Hugging Face dataset released
---
huggingface/README.md | 5 ++---
1 file changed, 2 insertions(+), 3 deletions(-)
diff --git a/huggingface/README.md b/huggingface/README.md
index 7d7e6c9..e175197 100644
--- a/huggingface/README.md
+++ b/huggingface/README.md
@@ -29,9 +29,8 @@ classification of individual shell-command submissions. It asks whether a
command poses meaningful cyber or system risk when evaluated without task,
user, or session context.
-**Release status:** The dataset card is public, but the Parquet splits are not
-yet published while source-by-source redistribution terms are reviewed. The
-`load_dataset()` example below becomes available with the data release.
+**Release:** The v0.1 Parquet train and test splits are publicly available
+through Dataset Viewer and `load_dataset()`.
The benchmark contains a deterministic train split of 16,772 rows and test
split of 4,194 rows. The test set contains 193 `risky` and 4,001 `not_risky`