From 16beea7ebbdcf161c1d0f56c830efe5b5af636b6 Mon Sep 17 00:00:00 2001 From: yoshifuminakamura Date: Mon, 7 Sep 2026 17:54:12 +0900 Subject: [PATCH] Add input metadata helper Signed-off-by: yoshifuminakamura --- .github/workflows/result-server-tests.yml | 1 + docs/cx/BENCHKIT_SPEC.md | 4 +- docs/guides/add-app.md | 11 +-- .../tests/test_app_contracts_script.py | 33 +++++++++ scripts/bk_functions.sh | 57 +++++++++++++++ scripts/tests/check_app_contracts.py | 7 +- scripts/tests/test_bk_input_info.sh | 69 +++++++++++++++++++ 7 files changed, 175 insertions(+), 7 deletions(-) create mode 100644 result_server/tests/test_app_contracts_script.py create mode 100755 scripts/tests/test_bk_input_info.sh diff --git a/.github/workflows/result-server-tests.yml b/.github/workflows/result-server-tests.yml index efdb94d5..81bf1486 100644 --- a/.github/workflows/result-server-tests.yml +++ b/.github/workflows/result-server-tests.yml @@ -71,6 +71,7 @@ jobs: run: | bash scripts/tests/test_bk_profiler.sh bash scripts/tests/test_bk_fetch_source.sh + bash scripts/tests/test_bk_input_info.sh bash scripts/tests/test_build_cache.sh bash scripts/tests/test_build_environment_snapshot.sh bash scripts/tests/test_ncu_plan_generation.sh diff --git a/docs/cx/BENCHKIT_SPEC.md b/docs/cx/BENCHKIT_SPEC.md index 7a1dca61..12d0acd1 100644 --- a/docs/cx/BENCHKIT_SPEC.md +++ b/docs/cx/BENCHKIT_SPEC.md @@ -433,7 +433,7 @@ Benchkit は、pre-staged input、restart、学習済みモデル、公開 archi これは top-level app source を表す `source_info` とは別の補助情報であり、`source_info` に app source と benchmark input を混在させすぎない。 `input_info` は最初の段階では任意項目であり、存在しない result を ingest failure として扱わない。 -一方、`results/input_info.json` を app が生成する場合、Benchkit はそれを JSON object として検証し、Result JSON の top-level `input_info` に添付できる。 +一方、app が Benchkit の入力metadata helper または対応する受け渡し形式で input metadata を渡す場合、Benchkit はそれを JSON object として検証し、Result JSON の top-level `input_info` に添付できる。 `input_info` には少なくとも次のような情報を置けることが望ましい。 @@ -457,7 +457,7 @@ Benchkit should preferably be able to retain provenance for benchmark inputs suc This is auxiliary information separate from `source_info`, which represents the top-level application source; app source and benchmark input should not be mixed too heavily into one object. At the initial stage, `input_info` is optional, and results without it are not treated as ingest failures. -When an application produces `results/input_info.json`, Benchkit can validate it as a JSON object and attach it to the top-level `input_info` field in Result JSON. +When an application passes input metadata through the Benchkit input metadata helper or a supported handoff format, Benchkit can validate it as a JSON object and attach it to the top-level `input_info` field in Result JSON. `input_info` should preferably be able to include information such as: diff --git a/docs/guides/add-app.md b/docs/guides/add-app.md index 3427980e..3e360750 100644 --- a/docs/guides/add-app.md +++ b/docs/guides/add-app.md @@ -127,9 +127,10 @@ portal の `/results/usage` では、この source provenance が各 app / syste 入力ファイルが app repository 内に既にあり、そのまま使う場合は、`source_info.resolved_commit` が app source と repo 内 input の固定点になります。 この場合、別 manifest や input digest を必須にする必要はありません。 -Portal や review で dataset 名を見せたい場合だけ、任意の `input_info` で repo-relative path を補足できます。 +Portal や review で dataset 名を見せたい場合だけ、任意の入力metadataで repo-relative path を補足できます。 -```json +```bash +bk_record_input_info <<'EOF' { "schema_version": 1, "inputs": [ @@ -142,6 +143,7 @@ Portal や review で dataset 名を見せたい場合だけ、任意の `input_ } ] } +EOF ``` ### pre-staged input と site-local 情報の扱い @@ -161,12 +163,13 @@ site-local path や allocation / project ID は、それ自体を一律に secre pre-staged input を使う app では、「正しい場所にファイルがある」だけでは再現性の説明として不足します。 可能であれば input directory と同じ場所に manifest を置き、run 前に manifest / digest を検証して、Result metadata へ dataset identity を残してください。 -`run.sh` が `results/input_info.json` を生成すると、`scripts/result.sh` はそれを JSON object として検証し、Result JSON の top-level `input_info` に添付します。 +app から実行時の入力metadataを渡す場合は、`scripts/bk_functions.sh` を source して `bk_record_input_info` を使ってください。 +app 側は共通層の受け渡し file path を意識せず、入力metadataの中身だけを定義します。 最小例: ```bash -cat > results/input_info.json <<'EOF' +bk_record_input_info <<'EOF' { "schema_version": 1, "inputs": [ diff --git a/result_server/tests/test_app_contracts_script.py b/result_server/tests/test_app_contracts_script.py new file mode 100644 index 00000000..cca20c2a --- /dev/null +++ b/result_server/tests/test_app_contracts_script.py @@ -0,0 +1,33 @@ +"""Tests for app contract warning helpers.""" + +from __future__ import annotations + +import importlib.util +from pathlib import Path + + +REPO_ROOT = Path(__file__).resolve().parents[2] +SCRIPT_PATH = REPO_ROOT / "scripts" / "tests" / "check_app_contracts.py" + + +def _load_script_module(): + spec = importlib.util.spec_from_file_location("check_app_contracts", SCRIPT_PATH) + assert spec is not None + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + spec.loader.exec_module(module) + return module + + +def test_input_provenance_visible_accepts_common_helper(tmp_path, monkeypatch): + module = _load_script_module() + programs_dir = tmp_path / "programs" + app_dir = programs_dir / "demo" + app_dir.mkdir(parents=True) + (app_dir / "run.sh").write_text( + "source scripts/bk_functions.sh\nbk_record_input_info metadata.json\n", + encoding="utf-8", + ) + monkeypatch.setattr(module, "PROGRAMS_DIR", programs_dir) + + assert module.input_provenance_visible("demo") is True diff --git a/scripts/bk_functions.sh b/scripts/bk_functions.sh index cbeaa745..a0f84800 100644 --- a/scripts/bk_functions.sh +++ b/scripts/bk_functions.sh @@ -978,6 +978,63 @@ bk_json_string_array() { printf ']' } +# bk_record_input_info - Pass benchmark input metadata to Benchkit. +# +# Usage: +# bk_record_input_info path/to/input-info.json +# bk_record_input_info <<'EOF' +# {"schema_version": 1, "inputs": [...]} +# EOF +# +# The file path used by the common result packaging layer is intentionally kept +# behind this helper so app scripts can focus on dataset identity and evidence. +bk_record_input_info() { + if [ $# -gt 1 ]; then + echo "bk_record_input_info: accepts at most one JSON file path" >&2 + return 1 + fi + + _bk_input_info_file="${BK_INPUT_INFO_FILE:-results/input_info.json}" + _bk_input_info_dir=$(dirname "$_bk_input_info_file") + mkdir -p "$_bk_input_info_dir" || return 1 + _bk_input_info_tmp=$(mktemp "${_bk_input_info_file}.tmp.XXXXXX") || return 1 + + if [ $# -eq 1 ]; then + _bk_input_info_source="$1" + if [ ! -f "$_bk_input_info_source" ]; then + echo "bk_record_input_info: input info JSON file not found: $_bk_input_info_source" >&2 + rm -f "$_bk_input_info_tmp" + return 1 + fi + if ! cp "$_bk_input_info_source" "$_bk_input_info_tmp"; then + rm -f "$_bk_input_info_tmp" + return 1 + fi + elif ! cat > "$_bk_input_info_tmp"; then + rm -f "$_bk_input_info_tmp" + return 1 + fi + + if [ ! -s "$_bk_input_info_tmp" ]; then + echo "bk_record_input_info: input info JSON must not be empty" >&2 + rm -f "$_bk_input_info_tmp" + return 1 + fi + + if command -v jq >/dev/null 2>&1; then + if ! jq -e 'type == "object"' "$_bk_input_info_tmp" >/dev/null 2>&1; then + echo "bk_record_input_info: input info must be a JSON object" >&2 + rm -f "$_bk_input_info_tmp" + return 1 + fi + fi + + if ! mv "$_bk_input_info_tmp" "$_bk_input_info_file"; then + rm -f "$_bk_input_info_tmp" + return 1 + fi +} + # Write a compact, tool-neutral manifest for the profiler archive. Result JSON # generation reads this manifest to expose summary fields without opening every # raw profiler artifact. For fapp, run_events contains counter names; for ncu it diff --git a/scripts/tests/check_app_contracts.py b/scripts/tests/check_app_contracts.py index a33461ac..e7444ef1 100644 --- a/scripts/tests/check_app_contracts.py +++ b/scripts/tests/check_app_contracts.py @@ -100,7 +100,12 @@ def input_provenance_visible(app: str) -> bool: run_text = app_script_text(app, "run.sh") return any( marker in run_text - for marker in ("results/input_info.json", "input_info", "BK_INPUT_") + for marker in ( + "bk_record_input_info", + "results/input_info.json", + "input_info", + "BK_INPUT_", + ) ) diff --git a/scripts/tests/test_bk_input_info.sh b/scripts/tests/test_bk_input_info.sh new file mode 100755 index 00000000..86428be3 --- /dev/null +++ b/scripts/tests/test_bk_input_info.sh @@ -0,0 +1,69 @@ +#!/bin/bash +set -euo pipefail + +SCRIPT_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +REPO_DIR=$(cd "${SCRIPT_DIR}/../.." && pwd) + +TMP_DIR=$(mktemp -d) +trap 'rm -rf "${TMP_DIR}"' EXIT + +source "${REPO_DIR}/scripts/bk_functions.sh" + +pushd "${TMP_DIR}" >/dev/null + +bk_record_input_info <<'EOF' +{ + "schema_version": 1, + "inputs": [ + { + "dataset_id": "demo-case0", + "verification_status": "declared" + } + ] +} +EOF + +test -s results/input_info.json +if command -v jq >/dev/null 2>&1; then + jq -e '.inputs[0].dataset_id == "demo-case0"' results/input_info.json >/dev/null +fi + +mkdir -p metadata +cat > metadata/input_info.json <<'EOF' +{ + "schema_version": 1, + "inputs": [ + { + "dataset_id": "demo-case1", + "verification_status": "declared" + } + ] +} +EOF + +BK_INPUT_INFO_FILE=custom/input_info.json bk_record_input_info metadata/input_info.json +test -s custom/input_info.json +if command -v jq >/dev/null 2>&1; then + jq -e '.inputs[0].dataset_id == "demo-case1"' custom/input_info.json >/dev/null +fi + +if bk_record_input_info missing.json >/dev/null 2>&1; then + echo "bk_record_input_info accepted a missing file" >&2 + exit 1 +fi + +if printf '' | bk_record_input_info >/dev/null 2>&1; then + echo "bk_record_input_info accepted empty input" >&2 + exit 1 +fi + +if command -v jq >/dev/null 2>&1; then + if printf '[]' | bk_record_input_info >/dev/null 2>&1; then + echo "bk_record_input_info accepted a non-object JSON value" >&2 + exit 1 + fi +fi + +popd >/dev/null + +echo "bk_record_input_info test passed"