diff --git a/.dockerignore b/.dockerignore index 7e9ae1d..a274855 100644 --- a/.dockerignore +++ b/.dockerignore @@ -1,6 +1,7 @@ # Keep the build context small: llama.cpp is fetched inside the image, models # are mounted at runtime, and build/output dirs are host artifacts. .git/ +.claude/ models/ build/ build-*/ diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 6a96ce9..5f64f33 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -13,19 +13,6 @@ concurrency: cancel-in-progress: true jobs: - cpp-unit: - runs-on: ubuntu-24.04 - steps: - - uses: actions/checkout@v7 - # Compiled directly: pure, no llama.cpp or protobuf/zmq needed. - - name: pure unit tests - run: | - for t in test_vision_common test_rope_conventions; do - g++ -std=c++17 -Isrc -Wall -Wextra -fsanitize=address,undefined \ - -fno-omit-frame-pointer "tests/$t.cpp" -o "/tmp/$t" - "/tmp/$t" - done - py-tooling: runs-on: ubuntu-24.04 steps: @@ -56,17 +43,19 @@ jobs: - uses: actions/cache@v6 with: path: build/_deps - key: llama-${{ steps.pin.outputs.tag }}-${{ runner.os }} + key: llama-${{ steps.pin.outputs.tag }}-${{ runner.os }}-asan # Everything, not a target list: ctest registers tests this job must build, # and a named list goes stale the next time one is added. - - name: build + ctest (CPU, -Wall -Wextra) + - name: build + ctest (CPU, ASan+UBSan, -Werror) run: | - cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=OFF -DVLA_BUILD_TESTS=ON + san="-fsanitize=address,undefined -fno-sanitize-recover=undefined -fno-omit-frame-pointer" + cmake -B build -DCMAKE_BUILD_TYPE=RelWithDebInfo -DGGML_CUDA=OFF -DVLA_BUILD_TESTS=ON \ + -DVLA_WERROR=ON -DCMAKE_C_FLAGS="$san" -DCMAKE_CXX_FLAGS="$san" cmake --build build -j"$(nproc)" ctest --test-dir build --output-on-failure - # This job is the only one that fetches llama.cpp, and neither patch script - # runs on a CPU build, so their anchors would otherwise rot unnoticed until - # someone configures a CUDA or OpenVINO tree. Patch a copy: the real one is + # Neither patch script runs on a CPU build, and build-cuda patches only on a + # cache miss, so their anchors would otherwise rot unnoticed until someone + # configures a fresh CUDA or OpenVINO tree. Patch a copy: the real one is # cached. - name: patch anchors still apply run: | @@ -74,3 +63,63 @@ jobs: python3 scripts/patch_ggml_cuda_ext_hook.py /tmp/llama-patchtest python3 scripts/patch_ggml_openvino.py /tmp/llama-patchtest python3 scripts/patch_ggml_openvino.py /tmp/llama-patchtest # idempotent + + build-cuda: + runs-on: ubuntu-24.04 + container: nvidia/cuda:12.9.1-devel-ubuntu24.04 + steps: + - name: deps + run: | + apt-get update -qq + DEBIAN_FRONTEND=noninteractive apt-get install -y -qq --no-install-recommends \ + build-essential cmake git ca-certificates pkg-config python3 \ + libzmq3-dev cppzmq-dev libprotobuf-dev protobuf-compiler + - uses: actions/checkout@v7 + - name: read llama.cpp pin + id: pin + run: echo "tag=$(bash scripts/llama_tag.sh)" >> "$GITHUB_OUTPUT" + - uses: actions/cache@v6 + with: + path: build/_deps + key: llama-${{ steps.pin.outputs.tag }}-${{ runner.os }}-cuda-${{ hashFiles('scripts/patch_ggml_cuda_ext_hook.py') }} + - name: build (compile only, the runner has no GPU) + run: | + cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=89-real \ + -DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined -DVLA_BUILD_TESTS=ON -DVLA_WERROR=ON + cmake --build build -j"$(nproc)" + + build-backends: + runs-on: ${{ matrix.os }} + strategy: + fail-fast: false + matrix: + include: + - { name: opencl, os: ubuntu-24.04, cmake: -DGGML_OPENCL=ON -DVLA_WERROR=ON } + - { name: metal, os: macos-15, cmake: -DGGML_METAL=ON, test: true } + steps: + - uses: actions/checkout@v7 + - name: deps + if: runner.os == 'Linux' + run: | + sudo apt-get update -qq + sudo apt-get install -y -qq --no-install-recommends \ + build-essential cmake git ca-certificates pkg-config \ + libzmq3-dev cppzmq-dev libprotobuf-dev protobuf-compiler \ + ocl-icd-opencl-dev opencl-headers + - name: deps + if: runner.os == 'macOS' + run: brew install cmake zeromq cppzmq protobuf + - name: read llama.cpp pin + id: pin + run: echo "tag=$(bash scripts/llama_tag.sh)" >> "$GITHUB_OUTPUT" + - uses: actions/cache@v6 + with: + path: build/_deps + key: llama-${{ steps.pin.outputs.tag }}-${{ runner.os }}-${{ matrix.name }} + - name: build + run: | + cmake -B build -DCMAKE_BUILD_TYPE=Release ${{ matrix.cmake }} -DVLA_BUILD_TESTS=ON + cmake --build build -j"$(getconf _NPROCESSORS_ONLN)" + - name: ctest + if: matrix.test + run: ctest --test-dir build --output-on-failure diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 786f7f8..100c643 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -1,19 +1,14 @@ -# Tagged binaries and a container image. build.yml already compiles all of this -# on every push; this is the same work with the artifacts kept. +# Tagged binaries and a container image. A tag push publishes them; a manual +# run builds the same artifacts on the chosen ref and publishes nothing. name: release on: push: tags: ['v*'] workflow_dispatch: - inputs: - tag: - description: Tag to build (dry run, nothing is published) - required: true permissions: - contents: write - packages: write + contents: read env: BINARIES: vla-server vlm-server vla-cli vla-bench @@ -21,123 +16,191 @@ env: jobs: linux: runs-on: ${{ matrix.runner }} + container: ${{ matrix.image }} + defaults: + run: + shell: bash strategy: fail-fast: false matrix: include: - name: linux-x86_64-cpu + runner: ubuntu-24.04 + image: ubuntu:24.04 cmake: -DGGML_CUDA=OFF - cuda: false + - name: linux-x86_64-cuda-12.8 runner: ubuntu-24.04 - - name: linux-x86_64-cuda - cmake: -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES="75;86;89;120" + image: nvidia/cuda:12.8.1-devel-ubuntu24.04 + cmake: -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES="75-real;80-real;86-real;89-real;90;120-real" cuda: true + - name: linux-x86_64-cuda-13.4 runner: ubuntu-24.04 - # Jetson and other aarch64 boards. Native arm64 runner, CPU only: the - # hosted images carry no CUDA for arm64, so a Jetson GPU build still - # has to happen on the device. + image: nvidia/cuda:13.4.1-devel-ubuntu24.04 + cmake: -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES="75-real;80-real;86-real;89-real;90;120-real" + cuda: true - name: linux-aarch64-cpu - cmake: -DGGML_CUDA=OFF - cuda: false runner: ubuntu-24.04-arm + image: ubuntu:24.04 + cmake: -DGGML_CUDA=OFF -DGGML_CPU_ARM_ARCH=armv8.2-a+dotprod+fp16 + - name: linux-aarch64-cuda-13.4 + runner: ubuntu-24.04-arm + image: nvidia/cuda:13.4.1-devel-ubuntu24.04 + cmake: -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES="87-real;110-real;121-real" -DGGML_CPU_ARM_ARCH=armv8.2-a+dotprod+fp16 + cuda: true steps: - - uses: actions/checkout@v7 - - name: deps run: | - sudo apt-get update -qq - sudo apt-get install -y -qq --no-install-recommends \ - build-essential cmake git ca-certificates pkg-config \ + apt-get update -qq + DEBIAN_FRONTEND=noninteractive apt-get install -y -qq --no-install-recommends \ + build-essential cmake git ca-certificates pkg-config python3 \ libzmq3-dev cppzmq-dev libprotobuf-dev protobuf-compiler - - name: cuda toolkit - if: matrix.cuda - run: | - wget -q https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2404/x86_64/cuda-keyring_1.1-1_all.deb - sudo dpkg -i cuda-keyring_1.1-1_all.deb - sudo apt-get update -qq - sudo apt-get install -y -qq --no-install-recommends cuda-toolkit-12-8 - echo "/usr/local/cuda/bin" >> "$GITHUB_PATH" + - uses: actions/checkout@v7 - name: build run: | - cmake -B build -DCMAKE_BUILD_TYPE=Release ${{ matrix.cmake }} + cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_NATIVE=OFF \ + -DCMAKE_INSTALL_RPATH='$ORIGIN' -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \ + ${{ matrix.cuda && '-DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined -DGGML_CUDA_NCCL=OFF' || '' }} ${{ matrix.cmake }} cmake --build build -j"$(nproc)" --target $BINARIES vla - name: package run: | - out="vla.cpp-${{ github.ref_name }}-${{ matrix.name }}" - mkdir -p "$out" + out="vla.cpp-${GITHUB_REF_NAME//\//-}-${{ matrix.name }}" + mkdir -p "$out/scripts" for b in $BINARIES; do cp "build/$b" "$out/"; done - cp build/libvla.so "$out/" + cp -P build/*.so* build/bin/*.so* "$out/" + ldd "$out/vla-server" | awk '/libgomp|libprotobuf/ {print $3}' | xargs cp -L -t "$out/" + cp /usr/share/doc/libprotobuf32t64/copyright "$out/LICENSE.protobuf" + cp /usr/share/doc/libgomp1/copyright "$out/LICENSE.libgomp" + cp build/_deps/llama-src/LICENSE "$out/LICENSE.llama.cpp" + cp build/_deps/llama-src/licenses/LICENSE-jsonhpp "$out/LICENSE.jsonhpp" + cp build/_deps/sentencepiece-src/LICENSE "$out/LICENSE.sentencepiece" + cp build/_deps/sentencepiece-src/third_party/darts_clone/LICENSE "$out/LICENSE.darts_clone" cp include/vla.h LICENSE.md README.md "$out/" - # vla-cli --text runs this; VLA_TOKENIZE_SCRIPT points at it. - mkdir -p "$out/scripts" && cp scripts/tokenize_prompt.py "$out/scripts/" + # vla-cli --text finds this in scripts/ next to itself. + cp scripts/tokenize_prompt.py "$out/scripts/" tar -czf "$out.tar.gz" "$out" + - name: smoke + run: | + out="vla.cpp-${GITHUB_REF_NAME//\//-}-${{ matrix.name }}" + mv build build.moved + ldd "$out"/vl*-* "$out"/*.so* > ldd.txt + if grep -v libcuda.so.1 ldd.txt | grep 'not found'; then exit 1; fi + LD_LIBRARY_PATH=/usr/local/cuda/compat "$out/vla-cli" --help + + - name: cuda runtime + if: matrix.cuda + run: | + out="vla.cpp-${GITHUB_REF_NAME//\//-}-${{ matrix.name }}" + mkdir cudart + cp -L /usr/local/cuda/lib64/lib{cudart,cublas,cublasLt}.so."${CUDA_VERSION%%.*}" cudart/ + tar -czf "cudart-$out.tar.gz" --transform "s,^cudart,$out," cudart + - uses: actions/upload-artifact@v7 with: name: ${{ matrix.name }} path: '*.tar.gz' + if-no-files-found: error macos: - runs-on: macos-14 + runs-on: macos-15 + env: + BINARIES: vla-cli vla-bench steps: - uses: actions/checkout@v7 - name: deps - run: brew install cmake zeromq cppzmq protobuf + run: brew install cmake - name: build run: | - cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_METAL=ON + cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_METAL=ON -DGGML_NATIVE=OFF \ + -DVLA_BUILD_SERVER=OFF -DVLA_SPM=OFF \ + -DCMAKE_INSTALL_RPATH='@loader_path' -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON cmake --build build -j"$(sysctl -n hw.ncpu)" --target $BINARIES vla - name: package run: | - out="vla.cpp-${{ github.ref_name }}-macos-arm64-metal" - mkdir -p "$out" + out="vla.cpp-${GITHUB_REF_NAME//\//-}-macos-arm64-metal" + mkdir -p "$out/scripts" for b in $BINARIES; do cp "build/$b" "$out/"; done - cp build/libvla.dylib "$out/" + cp -a build/*.dylib build/bin/*.dylib "$out/" + cp build/_deps/llama-src/LICENSE "$out/LICENSE.llama.cpp" + cp build/_deps/llama-src/licenses/LICENSE-jsonhpp "$out/LICENSE.jsonhpp" cp include/vla.h LICENSE.md README.md "$out/" - mkdir -p "$out/scripts" && cp scripts/tokenize_prompt.py "$out/scripts/" + cp scripts/tokenize_prompt.py "$out/scripts/" # Metal needs the shader library next to the binary. find build -name 'default.metallib' -exec cp {} "$out/" \; tar -czf "$out.tar.gz" "$out" + - name: smoke + run: | + out="vla.cpp-${GITHUB_REF_NAME//\//-}-macos-arm64-metal" + mv build build.moved + otool -L "$out"/vl*-* "$out"/*.dylib + if otool -L "$out"/vl*-* "$out"/*.dylib | grep -E '/opt/homebrew|/usr/local/'; then exit 1; fi + for lib in $(otool -L "$out"/vl*-* "$out"/*.dylib | awk '/@rpath\//{print $1}' | sort -u); do + test -e "$out/${lib#@rpath/}" || { echo "missing $lib"; exit 1; } + done + "$out/vla-cli" --help + - uses: actions/upload-artifact@v7 with: name: macos-arm64-metal path: '*.tar.gz' + if-no-files-found: error docker: runs-on: ubuntu-24.04 + permissions: + contents: read + packages: write steps: + - name: free disk + run: sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache - uses: actions/checkout@v7 - uses: docker/setup-buildx-action@v4 - uses: docker/login-action@v4 - if: startsWith(github.ref, 'refs/tags/') + if: github.event_name == 'push' with: registry: ghcr.io username: ${{ github.actor }} password: ${{ secrets.GITHUB_TOKEN }} - name: image name - run: echo "IMAGE=ghcr.io/${GITHUB_REPOSITORY,,}" >> "$GITHUB_ENV" + run: | + echo "IMAGE=ghcr.io/${GITHUB_REPOSITORY,,}" >> "$GITHUB_ENV" + echo "REF=${GITHUB_REF_NAME//\//-}" >> "$GITHUB_ENV" - uses: docker/build-push-action@v7 with: context: . - push: ${{ startsWith(github.ref, 'refs/tags/') }} + push: ${{ github.event_name == 'push' }} + build-args: | + CUDA_ARCH=75-real;80-real;86-real;89-real;90;120-real + GGML_NATIVE=OFF tags: | - ${{ env.IMAGE }}:${{ github.ref_name }} + ${{ env.IMAGE }}:${{ env.REF }} ${{ env.IMAGE }}:latest cache-from: type=gha cache-to: type=gha,mode=max publish: needs: [linux, macos, docker] - if: startsWith(github.ref, 'refs/tags/') + if: github.event_name == 'push' runs-on: ubuntu-24.04 + permissions: + contents: write steps: + - uses: actions/checkout@v7 + # The release body is this tag's CHANGELOG.md section; a tag with none + # fails here rather than publishing an empty release. + - name: release notes + run: | + ver="${GITHUB_REF_NAME#v}" + awk -v h="## [$ver]" 'index($0, h) == 1 {on=1; next} on && /^## \[/ {exit} on' \ + CHANGELOG.md > notes.md + test -s notes.md || { echo "no CHANGELOG.md section for $ver"; exit 1; } # Tarballs only. The docker job also leaves a .dockerbuild build record # artifact behind, and pulling that one fails the whole download. - uses: actions/download-artifact@v8 @@ -146,4 +209,5 @@ jobs: with: files: dist/*.tar.gz fail_on_unmatched_files: true + body_path: notes.md generate_release_notes: true diff --git a/.github/workflows/vla-ci.yml b/.github/workflows/vla-ci.yml index 57c88bd..c17da61 100644 --- a/.github/workflows/vla-ci.yml +++ b/.github/workflows/vla-ci.yml @@ -1,7 +1,9 @@ # Cross-platform VLA regression CI. -# Fires when a PR from `dev` targets `main`. Runs on the self-hosted runner -# labelled `vla-ci-orchestrator` (register your orchestrator host with that -# label - the actual hostname stays out of the repo, in ci/config/hosts.env). +# Fires on a push to `dev`, or by hand. Runs on the self-hosted runner labelled +# `vla-ci-orchestrator` (register your orchestrator host with that label - the +# actual hostname stays out of the repo, in the hosts.env that the runner's +# VLA_CI_HOSTS_ENV points at). Keep that runner in a runner group restricted to +# VinRobotics/vla.cpp/.github/workflows/vla-ci.yml@refs/heads/dev. # Every platform is a remote server reached over the LAN via vla-ci-agent (no # SSH); the orchestrator drives one LIBERO client per platform. The three # platforms are swept CONCURRENTLY in a single job (a self-hosted runner runs one @@ -9,8 +11,9 @@ name: vla-ci on: - pull_request: - branches: [main] + push: + branches: [dev] + workflow_dispatch: concurrency: group: vla-ci-${{ github.ref }} @@ -18,13 +21,19 @@ concurrency: jobs: ci: - # Only PRs whose source branch is `dev`. - if: github.head_ref == 'dev' runs-on: [self-hosted, vla-ci-orchestrator] timeout-minutes: 240 + env: + VLA_CI_EXPECTED_COMMIT: ${{ github.sha }} + CI_OUTPUT_ROOT: ${{ github.workspace }}/outputs/ci steps: - uses: actions/checkout@v7 - - name: Sweep all platforms (parallel) + gate + - name: Host config + control tools + run: | + cp "${VLA_CI_HOSTS_ENV:?set VLA_CI_HOSTS_ENV in the runner .env}" ci/config/hosts.env + cmake -S ci/agent -B ci/agent/build -DCMAKE_BUILD_TYPE=Release + cmake --build ci/agent/build -j + - name: Move servers to the pushed commit, build, sweep all platforms (parallel) + gate run: bash ci/orchestrate.sh all - uses: actions/upload-artifact@v7 if: always() diff --git a/CHANGELOG.md b/CHANGELOG.md index f98d4c6..5164360 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,129 +4,58 @@ Notable changes to vla.cpp. Format loosely follows [Keep a Changelog](https://ke ## [Unreleased] +## [0.4.0] - 2026-09-30 + ### Added -- **Snapdragon X on Windows on Arm: Hexagon NPU, Adreno GPU and CPU.** - - `-DGGML_HEXAGON=ON` and `-DGGML_OPENCL=ON` build natively with Visual - Studio's Clang; `scripts/build_windows_snapdragon.ps1` drives the build, - including skel signing. - - Ops either accelerator rejects run on the CPU through a wrapper backend - (`src/backend_fallback.cpp`), with no change to any arch. - - SmolVLA runs in 1.23 s on the NPU (2.57 s CPU), within 1.5e-3 of the CPU - reference; eleven checkpoints run on both accelerators. - - Five ggml-hexagon kernels that give wrong answers for VLA shapes are routed - around. See `docs/backend/hexagon-windows.md`. -- `--weight-dtype f16`. It is the default on Hexagon and OpenCL, and 2.5-3.5x - faster than BF16 on CPUs without BF16 matmul. -- `VLA_BUILD_SERVER=OFF` builds `vla-cli` and `vla-bench` without protobuf or - ZeroMQ. -- **OpenVINO backend.** `-DGGML_OPENVINO=ON` runs the archs on Intel CPUs, iGPUs - and NPUs through ggml's OpenVINO backend. SmolVLA, π0.5, Evo-1 and VLA-Adapter - match an F32 CPU reference to 1e-3; on an Arc B390 iGPU that is 3.0x to 9.6x - the native CPU backend. Every arch that can reach this backend - ten of the - eleven - is inside the accuracy bar on the OpenVINO CPU plugin, and nine of the - ten on the iGPU. BitVLA is the eleventh and pins to the CPU backend by design. +- **New models: Octo-Small and TurboVLA**, both with LIBERO checkpoints + ([octo](https://hf.co/vrfai/octo-small-libero-gguf), + [turbovla](https://hf.co/vrfai/turbovla-libero-gguf)). vla.cpp now runs 13 + architectures. +- **OpenVINO backend** (`-DGGML_OPENVINO=ON`) for Intel CPUs, iGPUs and NPUs. + Every arch except BitVLA runs on it within 1e-3 of an F32 CPU reference on the + OpenVINO CPU plugin; 3.0x to 9.6x the native CPU backend on an Arc B390 iGPU. See `docs/backend/ov.md`. -- `scripts/install_ov.sh` installs the OpenVINO runtime and the Intel GPU/NPU - driver stack on Ubuntu 22.04 and 24.04, with the runtime archive checksummed - against a digest pinned in the script. -- `scripts/patch_ggml_openvino.py` applies thirteen fixes to the fetched - ggml-openvino sources at configure time. Each hunk is checked on its own, so a - `build/_deps` patched by an older checkout fails loudly instead of building - something quietly wrong. -- `tests/test_graph_names.cpp` pins `vla::graph_unique_names`. -- CI now checks that both llama.cpp patch scripts still apply, on a copy of - the fetched tree. Neither ran on a CPU build, so their anchors could rot - unnoticed until someone configured a CUDA or OpenVINO tree. -- `docs/UPSTREAMING.md` and `scripts/upstream_split.py` regroup the thirteen - ggml-openvino fixes into one llama.cpp branch per PR. They are generic - backend defects, not vla.cpp workarounds; landing them upstream removes the - configure-time patch step entirely. +- **Snapdragon X on Windows on Arm**: Hexagon NPU (`-DGGML_HEXAGON=ON`) and + Adreno GPU (`-DGGML_OPENCL=ON`), with unsupported ops falling back to the CPU. + See `docs/backend/hexagon-windows.md`. +- **Portable release tarballs** for Linux x86-64 (CPU, CUDA 12.8, CUDA 13.4), + Linux aarch64 (CPU, CUDA 13.4 for Orin, Thor and DGX Spark) and macOS Metal, + plus `cmake --install` rules and a self-contained Python wheel. +- **Tokenizer in the GGUF**: `vla-cli --text` builds each arch's real prompt + and tokenizes in-process for Octo, π0, π0.5 and OpenVLA-OFT, with no Python. +- `-hf` picks files and tags within a repo (`user/repo:Q8_0`), `--num-steps` + sets the flow-matching step count, and `--weight-dtype f16` is 2.5-3.5x faster + than BF16 on CPUs without BF16 matmul. +- GR00T N1.7 checkpoints trained with relative actions. +- Per-device latency and memory reports in `docs/benchmark/`, and a real-robot + rollout guide in the README. -### Fixed +### Changed -- A `scripts/quantize_gguf.py` file did not load for SmolVLA on any platform: - its loader read every weight as float and refused the packed connector. It now - keeps packed GEMM weights packed, like the other archs, and dequantizes the - rest. -- Two elementwise adds stacked on a GEMM came out wrong on the Intel iGPU. The - GPU plugin folds elementwise ops into the preceding GEMM as post-ops, and given - `ADD(ADD(residual, GEMM), graph_input)` it folds both and silently drops the - second operand - the result equals the inner add. A llama.cpp graph never builds - that chain; a VLA does, wherever a vision tower's features are added on top of an - FFN residual. VLA-JEPA (5.4e-1) and GR00T N1.7 (1.9e0) were wrong on the iGPU - while matching the CPU plugin to 1e-4. Re-associating the two adds so the GEMM - keeps one post-op puts both at 2.6e-3. Bisected with `GGML_OPENVINO_DEBUG_NODE`. -- π0's action dims drifted 4e-2 on the iGPU and its gripper flipped a step late, - because the GPU plugin computes in F16 and π0 unrolls its whole denoise loop - inside one graph. `GGML_OPENVINO_GPU_PRECISION` now exposes the plugin's - inference precision; `backend_init` defaults it to f32 for π0 alone, which costs - about 3x on that arch and puts it at 6.5e-5. -- `scripts/patch_ggml_openvino.py` now fails if `EDITS` has a duplicate key. Python - keeps the last one silently, and a duplicate briefly removed the whole Intel - OpenCL platform fix from the patch without any error. -- The position-input fix stopped running when llama.cpp moved to `b10729`. That - release relocated the naming out of `GgmlOvDecoder::get_graph_input_ov_name()`, - which the patch guards, into a new free `get_tensor_graph_input_ov_name()`, and - left the member behind with no callers. The hunk still applied cleanly, so - nothing failed loudly - SmolVLA and π0.5 simply stopped returning actions - ("Argument shapes are inconsistent", a 113-token prefix ROPE reading the - 50-token suffix's table). Both functions are guarded now, and the patch script - says to check for a live caller, not just a matching anchor, on every tag bump. -- `scripts/upstream_split.py` addressed hunks by position in the patch script's - edit list. Adding a hunk to the front of a file's list silently handed every - later hunk to the wrong branch, and its own coverage count still read 29/29 - because each index was still used exactly once. Two branches had been swapped - this way. Hunks are now addressed by a unique substring of their anchor, which - fails loudly instead. The PERMUTE `op_case` fix, which had no branch at all, - now has one. -- `graph_unique_names` renamed through `ggml_format_name`, which passes the - tensor's own name to `vsnprintf` as both destination and `%s` source. glibc - empties it, so every duplicate node became the bare string `#`. -- The OpenVINO naive-path compiled-model cache was keyed on node count plus the - first and last node name. Two graphs of the same size collided and the second - ran the first's compiled model. It now also keys on every node's op and shape, - and the map is bounded. -- `GGML_OPENVINO_NAIVE_GRAPH_SIZE` went through `atoi`, so junk parsed to 0 and - sent every graph down the decoder-only-LLM path with nothing said. Empty - environment values no longer count as a setting either. -- `GGML_OPENVINO_CACHE_DIR` is cleared rather than warned about: a warm cache - returns wrong actions, and stderr is not always read. `VLA_ALLOW_OV_CACHE=1` - keeps it. -- `scripts/print_versions.sh` printed `?` for the llama.cpp pin ever since the - tag moved behind `VLA_LLAMA_TAG`. -- The OpenVINO `find_package` failure message was unreachable, sitting after the - fetch whose own `find_package(REQUIRED)` fired first. -- BitVLA indexed its action slots as `seq-2-n_action+i` with no check that the - sequence is long enough. Neither `ggml_get_rows` nor the CUDA gather - bound-checks, so a short prompt read out of bounds and returned it as hidden - states. One guard now covers both LM paths. -- pi0 and pi0.5 fell back to identity normalisation stats on a dimension mismatch - or a short read, and said so on stdout. That returns un-denormalised actions - from a checkpoint that looked fine. Both now fail the load, and the message - goes to stderr - stdout is the action stream `predict_check` diffs. -- `scratch_ctx::reset` ignored an arena larger than the first call's, which would - abort in `ggml_new_tensor` if any call site ever sized one from the input. -- The safetensors arch probe would allocate up to 256 MB for a header it only - substring-searches. Capped at 16 MB. -- The two CUDA targets were the only first-party code built without - `-Wall -Wextra`. -- `tests/bitvla_gemm_check.cu` had no build target and a comment claiming it was - never committed. It builds now, under `GGML_CUDA`. -- Stale references to `vision_common.h` (now `modules/preprocess.h`) and to the - retired `VLA_EVO1_BF16_ACT` switch. +- **Faster predict**: 6-18% lower latency on an RTX 5090 for every arch except + BitVLA, with byte-identical actions. +- llama.cpp `b11223`, SentencePiece `v0.2.1`, OpenVINO 2026.4. +- The Docker image covers sm_75 to sm_120 instead of sm_89 only. +- The x86 CUDA 12.8 tarball is renamed `linux-x86_64-cuda-12.8`, and the macOS + tarball ships `vla-cli` and `vla-bench` only. +- `VLA_OCTO` is renamed `VLA_SPM`; the old name still works with a warning. +- The command line now overrides a `--config` file, and a missing file or a bad + value in it is an error. -### Changed +### Fixed -- llama.cpp pinned at `b10729`, up from `b10331`. Brings OpenVINO 2026.3.1, the - IM2COL+MatMul to native-convolution fusion, and the `RELU`/`NEG`/`SQR` - translators, which the local patch no longer has to add. Byte-identical on the - CPU backend for all eleven archs. The build.yml cache key now reads the tag out - of `CMakeLists.txt` instead of repeating it. -- `src/models/dit_common.h` is gone. It redefined six `vla::` functions that - `src/layers/` already had, with both copies linked into `vla_core`. Every - includer used only `sinusoidal_time_emb` or `build_causal_mask`, so they now - include `layers/embed.h`. Byte-identical across all 11 archs. +- **Numerics now match each model's reference** for π0, SmolVLA, GR00T N1.7, + VLA-JEPA, Evo-1 and BitVLA, and the eval client + normalizes VLA-Adapter and GR00T state like the reference. On 100 paired + LIBERO-Object episodes π0 goes from 83 to 90 successes, SmolVLA from 90 to 92 + and GR00T N1.7 from 97 to 99 (none statistically significant). +- `vla-server` and `vlm-server` return an error on a bad request instead of + crashing, and malformed GGUF metadata is rejected at load. +- BitVLA on CUDA: wrong actions from BF16/F16/Q8_0 weights, ignored + `VLA_DEVICE`, and kernel races. +- The C API and Python bindings are safe to call from several threads. +- Quantized SmolVLA, TurboVLA and Evo-1 files failed to load. ## [0.3.0] - 2026-08-14 @@ -240,6 +169,9 @@ expert + dataset stats), CPU or CUDA, no external mmproj and no patch to llama.c - llama.cpp is fetched + pinned via CMake `FetchContent` (tag `b9866`); bumping is a one-line `GIT_TAG` change. Removed the `patches/` fetch script. +[Unreleased]: https://github.com/VinRobotics/vla.cpp/compare/v0.4.0...HEAD +[0.4.0]: https://github.com/VinRobotics/vla.cpp/releases/tag/v0.4.0 +[0.3.0]: https://github.com/VinRobotics/vla.cpp/releases/tag/v0.3.0 [0.2.0]: https://github.com/VinRobotics/vla.cpp/releases/tag/v0.2.0 [0.1.1]: https://github.com/VinRobotics/vla.cpp/releases/tag/v0.1.1 [0.1.0]: https://github.com/VinRobotics/vla.cpp/releases/tag/v0.1.0 diff --git a/CMakeLists.txt b/CMakeLists.txt index 1066aea..ce078c1 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -11,6 +11,13 @@ if(NOT CMAKE_BUILD_TYPE) set(CMAKE_BUILD_TYPE Release CACHE STRING "" FORCE) endif() +include(GNUInstallDirs) +if(APPLE) + set(CMAKE_INSTALL_RPATH "@loader_path/../${CMAKE_INSTALL_LIBDIR}" CACHE STRING "") +elseif(UNIX) + set(CMAKE_INSTALL_RPATH "$ORIGIN/../${CMAKE_INSTALL_LIBDIR}" CACHE STRING "") +endif() + set(_vla_accel "") foreach(_flag GGML_CUDA GGML_SYCL GGML_METAL GGML_OPENVINO GGML_HEXAGON GGML_OPENCL) if(${_flag}) @@ -32,6 +39,11 @@ if(_vla_accel_n GREATER 0 AND GGML_BACKEND_DL) "GGML_BACKEND_DL=ON is not supported with ${_vla_accel}: the backend is " "built as a module and src/backend.h links its init directly.") endif() +foreach(_flag GGML_VULKAN GGML_HIP GGML_MUSA GGML_CANN GGML_WEBGPU) + if(${_flag}) + message(WARNING "${_flag}=ON reaches vlm-server through llama only; src/backend.h has no ${_flag} path, so the VLA archs run on CPU.") + endif() +endforeach() set(LLAMA_BUILD_COMMON ON CACHE BOOL "" FORCE) set(LLAMA_BUILD_TOOLS ON CACHE BOOL "" FORCE) @@ -43,6 +55,28 @@ if(GGML_CUDA) find_package(Python3 COMPONENTS Interpreter REQUIRED) set(_vla_llama_patch PATCH_COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/scripts/patch_ggml_cuda_ext_hook.py ) + + find_package(CUDAToolkit REQUIRED) + if(NOT DEFINED CMAKE_CUDA_ARCHITECTURES AND "$ENV{CUDAARCHS}" STREQUAL "") + # SASS per supported GPU, sm_90 PTX so future cards JIT. + set(_vla_cuda_archs 80-real 86-real 87-real 89-real 90-real) + if(CUDAToolkit_VERSION VERSION_GREATER_EQUAL 12.8) + list(APPEND _vla_cuda_archs 100-real 120-real) + else() + message(WARNING + "CUDA ${CUDAToolkit_VERSION} < 12.8: omitting Blackwell sm_100/sm_120. " + "Pass -DCMAKE_CUDA_ARCHITECTURES= explicitly to override.") + endif() + if(CUDAToolkit_VERSION VERSION_GREATER_EQUAL 12.9) + list(APPEND _vla_cuda_archs 121-real) + endif() + if(CUDAToolkit_VERSION VERSION_GREATER_EQUAL 13.0) + list(APPEND _vla_cuda_archs 110-real) + endif() + set(CMAKE_CUDA_ARCHITECTURES ${_vla_cuda_archs} 90-virtual + CACHE STRING "" FORCE) + endif() + set(GGML_CUDA_FA_QUANTS "f16-f16" CACHE STRING "") elseif(GGML_OPENVINO) find_package(Python3 COMPONENTS Interpreter REQUIRED) set(_vla_llama_patch PATCH_COMMAND ${Python3_EXECUTABLE} @@ -63,7 +97,7 @@ endif() # build dir (-DVLA_LLAMA_TAG=b10326) without editing this file. The patch # anchors in scripts/patch_ggml_cuda_ext_hook.py and scripts/patch_ggml_openvino.py # are checked against the default. -set(_vla_llama_tag_default "b10729") +set(_vla_llama_tag_default "b11223") set(VLA_LLAMA_TAG "${_vla_llama_tag_default}" CACHE STRING "llama.cpp tag to fetch") # A cache entry survives an edit to the line above, so an existing build dir keeps # the tag it was first configured with and quietly builds the wrong llama.cpp. @@ -100,17 +134,27 @@ function(vla_exclude_fetched_targets dir) endfunction() vla_exclude_fetched_targets(${llama_SOURCE_DIR}) -# Octo is the one arch that tokenizes in-process, so it is also the one that -# needs SentencePiece and, through it, a protobuf. Optional so a build that does -# not want Octo does not inherit that dependency. -option(VLA_OCTO "Build the Octo arch (fetches SentencePiece, needs system protobuf)" ON) -if(VLA_OCTO) +# SentencePiece, and through it a protobuf, is only needed to tokenize --text with +# a tokenizer stored in the GGUF. Optional so a build without it has no protobuf. +option(VLA_SPM "Link SentencePiece for in-GGUF tokenizers (needs system protobuf)" ON) +if(DEFINED VLA_OCTO) + message(DEPRECATION "VLA_OCTO is deprecated, use VLA_SPM") + set_property(CACHE VLA_SPM PROPERTY VALUE ${VLA_OCTO}) + unset(VLA_OCTO CACHE) +endif() +if(VLA_SPM) # The system protobuf, not sentencepiece's vendored protobuf-lite: vla-server # aborts at static-init if two protobuf runtimes reach the same binary. set(SPM_ENABLE_SHARED OFF CACHE BOOL "" FORCE) set(SPM_BUILD_TEST OFF CACHE BOOL "" FORCE) set(SPM_ENABLE_TCMALLOC OFF CACHE BOOL "" FORCE) set(SPM_PROTOBUF_PROVIDER "package" CACHE STRING "" FORCE) + if(NOT WIN32) + find_package(Protobuf QUIET) + if(Protobuf_VERSION VERSION_GREATER_EQUAL 4.22) + set(SPM_ABSL_PROVIDER "package" CACHE STRING "" FORCE) + endif() + endif() set(_vla_spm_patch "") if(WIN32 AND NOT MSVC) set(_vla_spm_patch PATCH_COMMAND ${CMAKE_COMMAND} -P @@ -118,17 +162,13 @@ if(VLA_OCTO) endif() FetchContent_Declare(sentencepiece GIT_REPOSITORY https://github.com/google/sentencepiece - GIT_TAG v0.2.0 + GIT_TAG v0.2.1 GIT_SHALLOW TRUE ${_vla_spm_patch} ) - # sentencepiece v0.2.0 still asks for cmake_minimum_required(VERSION 3.1), - # which CMake 4 refuses outright. - set(CMAKE_POLICY_VERSION_MINIMUM 3.5) FetchContent_MakeAvailable(sentencepiece) - unset(CMAKE_POLICY_VERSION_MINIMUM) # Only the static library is linked; its CLI tools are dead weight. - vla_exclude_fetched_targets(${sentencepiece_SOURCE_DIR}) + set_property(DIRECTORY ${sentencepiece_SOURCE_DIR} PROPERTY EXCLUDE_FROM_ALL TRUE) # vcpkg's FindProtobuf shim reports the DLL, not its import library, as # PROTOBUF_LITE_LIBRARY, and sentencepiece links that path verbatim. Link the # imported target instead, which carries the import library and protobuf's @@ -181,6 +221,8 @@ add_library(vla_core ${VLA_CORE_LIB_TYPE} src/models/openvla_oft.cpp src/models/vla_jepa.cpp src/models/turbovla.cpp + src/models/octo.cpp + src/tokenizer.cpp ) target_include_directories(vla_core PUBLIC @@ -189,33 +231,16 @@ target_include_directories(vla_core ) # The VLA archs call no llama_* API; only vlm_core needs llama. target_link_libraries(vla_core PUBLIC ggml) -if(VLA_OCTO) - target_sources(vla_core PRIVATE src/models/octo.cpp) +if(VLA_SPM) target_include_directories(vla_core PRIVATE ${sentencepiece_SOURCE_DIR}/src) target_link_libraries(vla_core PRIVATE sentencepiece-static ${VLA_SPM_PROTOBUF}) - target_compile_definitions(vla_core PUBLIC VLA_USE_OCTO) + target_compile_definitions(vla_core PRIVATE VLA_USE_SPM) endif() if(GGML_CUDA) target_compile_definitions(vla_core PUBLIC GGML_USE_CUDA) enable_language(CUDA) - find_package(CUDAToolkit REQUIRED) - if(NOT DEFINED CMAKE_CUDA_ARCHITECTURES) - # SASS per supported GPU, PTX only for the newest arch so future cards JIT. - set(_vla_cuda_archs 80-real 86-real 87-real 89-real 90-real) - set(_vla_cuda_ptx 90-virtual) - if(CUDAToolkit_VERSION VERSION_GREATER_EQUAL 12.8) - list(APPEND _vla_cuda_archs 100-real 120-real) - set(_vla_cuda_ptx 120-virtual) - else() - message(WARNING - "CUDA ${CUDAToolkit_VERSION} < 12.8: omitting Blackwell sm_100/sm_120. " - "Pass -DCMAKE_CUDA_ARCHITECTURES= explicitly to override.") - endif() - set(CMAKE_CUDA_ARCHITECTURES ${_vla_cuda_archs} ${_vla_cuda_ptx} - CACHE STRING "" FORCE) - endif() add_library(bitvla_cuda_kernels STATIC src/kernels/bitvla/bitnet_kernels.cu src/kernels/bitvla/bitvla_lm_cuda.cu @@ -289,7 +314,7 @@ endif() # headers for the backend vtable, hence the private include. if(GGML_HEXAGON OR GGML_OPENCL) target_sources(vla_core PRIVATE src/backend_fallback.cpp) - target_include_directories(vla_core PRIVATE ${llama_SOURCE_DIR}/ggml/src) + target_include_directories(vla_core SYSTEM PRIVATE ${llama_SOURCE_DIR}/ggml/src) if(GGML_HEXAGON) target_compile_definitions(vla_core PUBLIC GGML_USE_HEXAGON) else() @@ -300,13 +325,11 @@ endif() add_library(vlm_core ${VLA_CORE_LIB_TYPE} src/vlm/engine.cpp ) -target_include_directories(vlm_core - PUBLIC - ${CMAKE_CURRENT_SOURCE_DIR}/src - PRIVATE - ${llama_SOURCE_DIR}/common - ${llama_SOURCE_DIR}/tools/mtmd - ${llama_SOURCE_DIR}/vendor +target_include_directories(vlm_core PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/src) +target_include_directories(vlm_core SYSTEM PRIVATE + ${llama_SOURCE_DIR}/common + ${llama_SOURCE_DIR}/tools/mtmd + ${llama_SOURCE_DIR}/vendor ) target_link_libraries(vlm_core PUBLIC llama ggml mtmd llama-common) @@ -412,7 +435,7 @@ target_include_directories(vla-cli PRIVATE ${llama_SOURCE_DIR}/vendor/stb ) target_link_libraries(vla-cli PRIVATE vla_core) -# --text shells out to the tokenizer script; VLA_TOKENIZE_SCRIPT overrides it. +# --text shells out to the tokenizer script when the GGUF has no tokenizer. target_compile_definitions(vla-cli PRIVATE VLA_SOURCE_DIR="${CMAKE_CURRENT_SOURCE_DIR}") add_executable(vla-bench @@ -420,6 +443,17 @@ add_executable(vla-bench ) target_link_libraries(vla-bench PRIVATE vla_core) +install(TARGETS vla COMPONENT libvla) +install(FILES LICENSE.md ${llama_SOURCE_DIR}/licenses/LICENSE-jsonhpp + DESTINATION ${CMAKE_INSTALL_DOCDIR} COMPONENT libvla) +install(FILES ${llama_SOURCE_DIR}/LICENSE DESTINATION ${CMAKE_INSTALL_DOCDIR} + RENAME LICENSE.llama.cpp COMPONENT libvla) +install(TARGETS vla_core vla-cli vla-bench) +install(FILES scripts/tokenize_prompt.py DESTINATION ${CMAKE_INSTALL_DATADIR}/vla) +if(VLA_BUILD_SERVER) + install(TARGETS vlm_core vla-server vlm-server) +endif() + # Warnings and LTO for our own targets only, never the llama.cpp subtree. set(VLA_FIRST_PARTY_TARGETS vla_core vlm_core vla vla-cli vla-bench) if(VLA_BUILD_SERVER) @@ -430,8 +464,10 @@ if(GGML_CUDA) list(APPEND VLA_FIRST_PARTY_TARGETS bitvla_cuda_kernels vla_cuda_ops) endif() +option(VLA_WERROR "Treat warnings in first-party code as errors" OFF) foreach(tgt IN LISTS VLA_FIRST_PARTY_TARGETS) - target_compile_options(${tgt} PRIVATE $<$:-Wall -Wextra>) + target_compile_options(${tgt} PRIVATE + $<$:-Wall -Wextra $<$:-Werror>>) endforeach() # llama.cpp sends its DLLs to /bin, and Windows has no rpath: a binary diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index bc5915d..a2cdd71 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -28,10 +28,12 @@ Any difference is a bug unless the change is meant to alter numerics, in which case say so in the commit message and back it with a LIBERO sweep. `VLA_IMG_SIZE` must match the model or `predict` returns empty: 512 for -SmolVLA, 448 for Evo-1, 256 for GR00T N1.7 and VLA-JEPA, 224 for the rest. Other -knobs: `VLA_BENCH_ITERS` (timing), `VLA_TIMING=phase`, `VLA_EXTRA_TOKEN` / -`VLA_EXTRA_COUNT` (VLA-JEPA needs its `` tokens), `VLA_N_THREADS`, -`VLA_DEVICE`. +SmolVLA, 448 for Evo-1, 256 for GR00T N1.7 and VLA-JEPA, 256 with 2 views for +TurboVLA, 224 for the rest. Other knobs: `VLA_BENCH_ITERS` (timing), +`VLA_TIMING=phase`, `VLA_EXTRA_TOKEN` / `VLA_EXTRA_COUNT` (VLA-JEPA needs its +`` tokens; Octo needs exactly 16 tokens, so +`VLA_EXTRA_TOKEN=0 VLA_EXTRA_COUNT=10`), `VLA_OCTO_UNNORM_DATASET` (Octo, e.g. +`libero_object`), `VLA_N_THREADS`, `VLA_DEVICE`. Checkpoints are at [huggingface.co/vrfai](https://huggingface.co/vrfai), or let the binaries fetch them: @@ -42,23 +44,26 @@ the binaries fetch them: ## Adding an architecture -Six sites, all mechanical. `smolvla` is the reference for a two-file (mmproj + -ckpt) model, `bitvla` for a vision-baked one. +Seven sites, all mechanical. Every arch loads one GGUF with its vision tower +bundled; `mmproj_path` is accepted and ignored. 1. `src/arch.h` - add to `enum class Arch`. -2. `src/arch.h` - declare `_create(mmproj_path, ckpt_path, config_path)`. +2. `src/arch.h` - declare `_create(mmproj_path, ckpt_path, config_path, opts)`. 3. `src/model.cpp` - add `.architecture` to the `try_str` list in `detect_arch_gguf`. 4. `src/model.cpp` - map the string to the enum in the same function. 5. `src/model.cpp` - add a `case` to the `model_load` switch. -6. `CMakeLists.txt` - add `src/models/.cpp` to `vla_core`. - -Then write `src/models/.cpp`. Before adding a helper, check -`src/models/`: `gguf_reader.h` (tensor and KV reads), `modules/preprocess.h` -(preprocessing, pixel shuffle), `dual_tower.h` (DINOv2 + SigLIP), -`qwen3vl_vit.h` (Qwen3-VL tower), `layers/embed.h` (time embeddings, causal -mask), `scratch_ctx.h` (compute context reuse), `backend.h` (accelerator -selection). +6. `src/model.cpp` - if the arch has no flow-matching solver, add it to the + `opts.num_steps` refusal list in `model_load`; if it has one, read the step + count with `resolve_num_steps`. +7. `CMakeLists.txt` - add `src/models/.cpp` to `vla_core`. + +Then write `src/models/.cpp`. Before adding a helper, check `src/`: +`gguf_reader.h` (tensor and KV reads), `loader.h` (weight upload and fusion), +`modules/preprocess.h` (view checks, CHW preprocessing), `modules/dual_tower.h` +(DINOv2 + SigLIP), `modules/qwen3vl_vit.h` (Qwen3-VL tower), `layers/` (attention, +FFN, norms, RoPE, time embeddings, causal mask), `scratch_ctx.h` (compute context +and graph reuse), `backend.h` (accelerator selection). Your loader must fail rather than return a half-built model: check every tensor lookup, and check `real_*_dim <= max_*_dim` (`config_is_sane` in `src/model.cpp` diff --git a/Dockerfile b/Dockerfile index a31dcd4..c06c4b3 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,13 +1,21 @@ -# vla-server, CPU or CUDA. cmake fetches llama.cpp at build time. -# GPU sm_89 (default): docker build -t vla-cpp . -# GPU other arch: --build-arg CUDA_ARCH=120 (86=RTX30 90=H100 87=Orin 120=RTX50; sm_120 needs CUDA>=12.8) -# older card: --build-arg BASE_IMAGE=nvidia/cuda:12.4.1-devel-ubuntu24.04 --build-arg CUDA_ARCH=86 -# CPU: --build-arg BACKEND=cpu --build-arg BASE_IMAGE=ubuntu:24.04 -t vla-cpp-cpu +# vla-server, CPU or CUDA. cmake fetches llama.cpp at build time; the image keeps +# only the binaries and their libs on a -runtime (or plain ubuntu) base. +# GPU, CUDA 12.9, sm_75..sm_121: docker build -t vla-cpp . +# one arch, faster build: --build-arg CUDA_ARCH=120 (86=RTX30 89=RTX40 90=H100 87=Orin 120=RTX50) +# CUDA 13 (driver 580+): --build-arg CUDA_VERSION=13.4.1 +# CPU: --build-arg BACKEND=cpu -t vla-cpp-cpu # run: docker run --gpus all -p5555:5555 -v $PWD/models:/models vla-cpp --bind tcp://*:5555 /models/M.gguf # (CDI hosts use --device nvidia.com/gpu=all) -ARG BASE_IMAGE=nvidia/cuda:12.9.1-devel-ubuntu24.04 -FROM ${BASE_IMAGE} +ARG CUDA_VERSION=12.9.1 +ARG BACKEND=cuda + +FROM nvidia/cuda:${CUDA_VERSION}-devel-ubuntu24.04 AS cuda-build +FROM nvidia/cuda:${CUDA_VERSION}-runtime-ubuntu24.04 AS cuda-run +FROM ubuntu:24.04 AS cpu-build +FROM ubuntu:24.04 AS cpu-run + +FROM ${BACKEND}-build AS build RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \ build-essential cmake git ca-certificates pkg-config python3 \ @@ -17,8 +25,9 @@ RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-ins WORKDIR /src COPY . . -ARG BACKEND=cuda -ARG CUDA_ARCH=89 +ARG BACKEND +ARG CUDA_ARCH="75-real;80-real;86-real;89-real;90;120-real;121-real" +ARG GGML_NATIVE=ON # nvcc can segfault on the flash-attn kernels under high -j; lower JOBS if so. ARG JOBS= # CUDA: -devel ships only a libcuda stub (real driver injected at runtime), so @@ -26,14 +35,31 @@ ARG JOBS= RUN set -eux; \ if [ "$BACKEND" = "cuda" ]; then \ export LIBRARY_PATH=/usr/local/cuda/lib64/stubs; \ - cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=ON \ - -DCMAKE_CUDA_ARCHITECTURES="${CUDA_ARCH}" \ - -DCMAKE_SHARED_LINKER_FLAGS="-lcuda" -DCMAKE_EXE_LINKER_FLAGS="-lcuda"; \ + set -- -DGGML_CUDA=ON "-DCMAKE_CUDA_ARCHITECTURES=${CUDA_ARCH}" \ + -DCMAKE_SHARED_LINKER_FLAGS="-lcuda" -DCMAKE_EXE_LINKER_FLAGS="-lcuda"; \ else \ - cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=OFF; \ + set -- -DGGML_CUDA=OFF; \ fi; \ + cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_NATIVE=${GGML_NATIVE} \ + -DCMAKE_INSTALL_RPATH='$ORIGIN' -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON "$@"; \ cmake --build build -j"${JOBS:-$(nproc)}" --target vla-server vla-cli; \ - cp build/vla-server build/vla-cli /usr/local/bin/ + mkdir /app; \ + cp build/vla-server build/vla-cli /app/; \ + cp -P build/*.so* build/bin/*.so* /app/; \ + cp LICENSE.md /app/; \ + cp build/_deps/llama-src/LICENSE /app/LICENSE.llama.cpp; \ + cp build/_deps/llama-src/licenses/LICENSE-jsonhpp /app/LICENSE.jsonhpp; \ + cp build/_deps/sentencepiece-src/LICENSE /app/LICENSE.sentencepiece; \ + cp build/_deps/sentencepiece-src/third_party/darts_clone/LICENSE /app/LICENSE.darts_clone + +FROM ${BACKEND}-run + +RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \ + libgomp1 libprotobuf-lite32t64 libprotobuf32t64 libzmq5 \ + && rm -rf /var/lib/apt/lists/* + +COPY --from=build /app/ /app/ +ENV PATH=/app:$PATH EXPOSE 5555 ENTRYPOINT ["vla-server"] diff --git a/README.md b/README.md index 59a06af..9ea731d 100644 --- a/README.md +++ b/README.md @@ -20,13 +20,43 @@ via OpenCL and the Hexagon backend. --- -## Build the server +## Install -### Prerequisites +Download a prebuilt release, or build from source for any other platform or +backend. + +### Prebuilt binaries + +Each [release](https://github.com/VinRobotics/vla.cpp/releases) has a +`vla.cpp--.tar.gz` with `libvla`, `vla.h` and the binaries: + +| Platform | Needs | +|---|---| +| `linux-x86_64-cpu` | AVX2 (Haswell or newer) | +| `linux-x86_64-cuda-12.8` | AVX2; sm_75/80/86/89/90/120 | +| `linux-x86_64-cuda-13.4` | AVX2; sm_75/80/86/89/90/120, driver 580 or newer | +| `linux-aarch64-cpu` | ARMv8.2-A with dotprod and fp16 (Cortex-A76, Neoverse N1 or newer) | +| `linux-aarch64-cuda-13.4` | sm_87 (Orin), sm_110 (Thor), sm_121 (DGX Spark); a CUDA 13 driver | +| `macos-arm64-metal` | `vla-cli` and `vla-bench` only | + +```bash +TAG= +curl -LO https://github.com/VinRobotics/vla.cpp/releases/download/$TAG/vla.cpp-$TAG-linux-x86_64-cpu.tar.gz +tar -xzf vla.cpp-$TAG-linux-x86_64-cpu.tar.gz +``` + +The Linux tarballs need Ubuntu 24.04's glibc or newer, `libzmq5` for the +servers, and a CUDA runtime for the CUDA ones (shipped as a separate `cudart-` +tarball). Details, and the macOS and Windows notes, are in +[docs/PREBUILT.md](docs/PREBUILT.md). + +### Build from source + +#### Prerequisites - CMake ≥ 3.22 - A C++17 compiler (GCC 11+ or Clang 14+) -- CUDA 12.x (optional - required only for CUDA GPU builds) +- CUDA 12.x or 13.x (optional - required only for CUDA GPU builds) - Intel oneAPI 2025.x + GPU compute runtime (optional - only for Intel GPU builds, see [docs/backend/sycl.md](docs/backend/sycl.md)) - OpenVINO 2026.x runtime (optional - only for Intel CPU/GPU/NPU builds via @@ -37,7 +67,7 @@ via OpenCL and the Hexagon backend. sudo apt-get install -y libzmq3-dev cppzmq-dev libprotobuf-dev protobuf-compiler ``` -### From source +#### Configure and build Identify your machine CUDA architecture: @@ -49,6 +79,8 @@ Identify your machine CUDA architecture: | Hopper | H100, H200 | `90` | | Blackwell (consumer) | RTX 50-series | `120` | | Blackwell (datacenter) | B100, B200, GB200 | `100` | +| Blackwell (Jetson) | Jetson Thor | `110` | +| Blackwell (DGX Spark) | GB10 | `121` | Then configure and build. CMake fetches and pins `llama.cpp` automatically (no patch, no submodule): @@ -72,6 +104,15 @@ export PATH=/usr/local/cuda/bin:$PATH export LD_LIBRARY_PATH=/usr/local/cuda/lib64:$LD_LIBRARY_PATH ``` +`-DVLA_BUILD_SERVER=OFF -DVLA_SPM=OFF` builds `vla-cli`, `vla-bench` and +`libvla` without protobuf, ZeroMQ or SentencePiece, so none of the apt packages +above are needed. Without SentencePiece, pass Octo `--tokens` instead of `--text`. + +`cmake --install build --prefix ` copies the binaries, libraries and +`share/vla/tokenize_prompt.py` into ``; the result does not need the build +tree. `pip install ./bindings/python` builds the Python bindings, see +[bindings/python/README.md](bindings/python/README.md). + Check [docs/backend](docs/backend) for compiling `vla.cpp` on other platforms. WSL2, Apple Silicon, and Intel GPU are all tested. To build and run in containers instead, see [docs/DOCKER.md](docs/DOCKER.md). @@ -94,223 +135,17 @@ pip install -U "huggingface_hub[cli]" transformers --image assets/front.jpg --text "pick up the black bowl" --pretty ``` -`vla-cli` runs a single prediction without a server or simulator: give it a model, -an image, and an instruction, and it prints the action chunk. Handy for -smoke-testing a GGUF or scripting a quick inference. - -There is no tokenizer in the C++ core, so `--text` calls -`scripts/tokenize_prompt.py` with the tokenizer the architecture was trained on -(`VLA_PYTHON` picks the interpreter, `VLA_TOKENIZE_SCRIPT` the script). Pass -`--tokens 1,100,200,2` instead if you already have ids. -`--pretty` prints one action row per line; -`--state` sets proprioception (defaults to zeros). - -For the design overview see -[docs/ARCHITECTURE.md](docs/ARCHITECTURE.md), for the long-running path see -[Running the server](#running-the-server), and for the other checkpoints see [Roadmap](#roadmap). - -The rest of this README refers to a few shell variables: - -```bash -export VLA_GGUF=models/smolvla/smolvla-libero.gguf # the checkpoint to serve -export VLA_ARCH=smolvla # client-side arch preset, see --help -``` +With a prebuilt tarball, run `vla-cli` from the extracted +`vla.cpp--/` directory instead of `./build/`. +`--pretty` prints one action row per line and `--state` sets proprioception. +Prompt tokenization, `-hf` tags, `vla-server`, runtime flags and environment +variables are in [docs/USAGE.md](docs/USAGE.md). --- -## Install simulators - -The eval scaffold under [`eval/`](eval/) supports two simulators end-to-end. Each setup script bootstraps an isolated Python 3.10 `uv` venv next to itself and clones the upstream sim repo. Both require [`uv`](https://github.com/astral-sh/uv) on `PATH`. - -### LIBERO - -```bash -bash eval/sim/libero/setup_libero.sh -``` - -Clones LIBERO into [`eval/sim/libero/LIBERO/`](eval/sim/libero/LIBERO), creates `eval/sim/libero/libero_uv/.venv/`, and pins compatible versions of torch, lerobot, transformers, and gymnasium. - -### SimplerEnv - -```bash -bash eval/sim/simpler/setup_SimplerEnv.sh -``` - -Clones SimplerEnv (and its nested `ManiSkill2_real2sim`) into [`eval/sim/simpler/SimplerEnv/`](eval/sim/simpler/SimplerEnv), creates `eval/sim/simpler/simpler_uv/.venv/`. - ---- - -## Running the server - -`vla-server` loads the model once at startup and answers ZeroMQ REQ/REP requests synchronously. - -```bash -./build/vla-server "$VLA_GGUF" -``` - -When ready, the server prints: - -``` -vla-server: bound to tcp://*:5555. ready. -``` - -Use `--bind` to change the address and port. Stop the server with `Ctrl-C`. - -`vla-server` also takes `-hf user/repo[:file.gguf]` in place of a checkpoint path. - -Precision flags (`vla-server --help` for the full list). -The fastest configuration per model, with measured latency and success rate, is -in [`CHANGELOG.md`](CHANGELOG.md): - -- `--weight-dtype f32|bf16` - resident dtype for GEMM weights. -- `--act-dtype f32|bf16` - activation dtype; needs CUDA and bf16 weights. -- `--flash-attn` - faster on the larger towers, but changes numerics. -- `--mm-prec default|f32` - matmul accumulation precision. - - -Environment knobs that apply to every arch: - -- `VLA_N_THREADS` - CPU backend thread count, default core count capped at 16. -- `VLA_DEVICE` - GPU ordinal for CUDA and SYCL builds, default 0. -- `VLA_CACHE` - where `-hf` stores checkpoints, default `~/.cache/vla`. - ---- - -## Running the client - -[`eval/client/`](eval/client/) ships an end-to-end LIBERO benchmark runner that drives `vla-server` directly over the protobuf protocol. Make sure the LIBERO venv from [Install simulators](#install-simulators) is set up first. - -### LIBERO - -With `vla-server` already running: - -```bash -source eval/sim/libero/libero_uv/.venv/bin/activate -python eval/client/run_sim_client_direct.py \ - --task libero_object --task-id 0 --n-episodes 1 \ - --output-dir /tmp/libero_outputs \ - --arch "$VLA_ARCH" -``` +## Support matrix -The GR00T models need two extras: - -- client side: `--stats-json /path/to/dataset_statistics.json` -- server side: `VLA_GR00T_EMBODIMENT` (`new_embodiment` for N1.5, `libero_panda` for N1.6, `libero_sim` for N1.7). - -### SimplerEnv - -So far only **GR00T-N1.6** is wired (the `gr00t-n1d6-bridge` checkpoint with the `oxe_widowx` embodiment). Start `vla-server` on port 5566 with `oxe_widowx` embodiment: - -```bash -VLA_GR00T_EMBODIMENT=oxe_widowx \ - ./build/vla-server "$GR00T_N1D6_GGUF" -``` - -Then drive it from the SimplerEnv venv (set up via [Install simulators](#install-simulators)): - -```bash -source eval/sim/simpler/simpler_uv/.venv/bin/activate -python eval/client/run_simpler_client_direct.py \ - --arch gr00t_n1_6 \ - --task-id oxe_widowx/widowx_spoon_on_towel --n-episodes 1 \ - --embodiment oxe_widowx --image-size 252 \ - --stats-json "$VLA_STATS_JSON" -``` - ---- - -## Models - -### Conversion - -Each model ships as a single self-contained GGUF. To convert a HuggingFace safetensors -checkpoint yourself, [`scripts/`](scripts/) has a converter per arch. Set up its venv: - -```bash -python3 -m venv .venv-converter -source .venv-converter/bin/activate -pip install -e ".[convert]" -``` - -Then run any of the per-arch converters (`--help` for the full flag list): - -```bash -python scripts/convert_smolvla_to_gguf.py \ - --ckpt /path/to/smolvla-libero \ - --out /path/to/smolvla-libero-bf16.gguf -``` - -### Quantization - -The shipped GGUFs are bf16. `scripts/quantize_gguf.py` repacks the LM-backbone weight -matrices to a smaller type and copies everything else unchanged; the loader keeps the -packed weights and lets `ggml_mul_mat` dequantize at compute, so the file just loads and -runs like the bf16 one. - -```bash -python scripts/quantize_gguf.py --in model-bf16.gguf --out model-q8_0.gguf --type Q8_0 -``` - -`Q8_0` is near-lossless and roughly halves the LM. `Q4_0` is 4-bit for a bigger cut. -Embeddings, the output head, norms and the action expert stay float; pass `--vision` to -pack the vision tower too (smaller, but more accuracy loss). - ---- - -## Benchmarks - -`vla-bench` times `predict()` in-process on synthetic inputs: engine only, no -transport, no simulator, no claim about task success. - -```bash -./build/vla-bench -hf vrfai/smolvla-libero-gguf --images 2 --size 512 --markdown -``` - -RTX 5090, driver 595.84, CUDA 13.2, 24-core host, weights as shipped, 20 reps -after 3 warmups, best of three sweeps, each model at its native input size and -view count. - -| Model | Views | Input | min ms | p50 ms | p90 ms | vision ms | -|---|--:|--:|--:|--:|--:|--:| -| VLA-Adapter | 1 | 224 | 18.2 | 19.8 | 21.1 | 9.4 | -| VLA-JEPA | 1 | 256 | 19.9 | 21.5 | 22.9 | 6.3 | -| BitVLA | 1 | 224 | 23.6 | 25.3 | 26.4 | 5.4 | -| GR00T N1.5 | 1 | 224 | 28.2 | 29.4 | 30.5 | 5.9 | -| GR00T N1.7 | 1 | 256 | 31.0 | 33.4 | 34.6 | 6.2 | -| GR00T N1.6 | 1 | 224 | 33.4 | 35.7 | 37.3 | 6.3 | -| OpenVLA-OFT | 1 | 224 | 47.4 | 49.2 | 50.2 | 10.3 | -| SmolVLA | 2 | 512 | 47.8 | 49.6 | 54.0 | 16.1 | -| pi0 | 2 | 224 | 48.9 | 52.1 | 55.0 | 11.6 | -| Evo-1 | 1 | 448 | 52.2 | 55.2 | 57.3 | 17.8 | -| pi0.5 | 2 | 224 | 53.4 | 56.1 | 59.3 | 11.4 | - -### Task success - -Latency says nothing about whether a policy works. LIBERO-Object, 10 tasks and 20 -episodes per model, terminated episodes counted as failures: - -| Model | Chunk replay | Success rate | -|---|--:|--:| -| BitVLA | 8 | 100.0% | -| GR00T N1.7 | 16 | 98.0% | -| GR00T N1.5 | 16 | 96.0% | -| Evo-1 | 8 | 94.5% | -| SmolVLA | 4 | 90.5% | -| π0 | 32 | 87.5% | -| GR00T N1.6 | 16 | 86.5% | - -Success rate belongs to the checkpoint, not the engine; -`vla_predict_check` in [CONTRIBUTING.md](CONTRIBUTING.md) is how a -change is shown to leave it alone. - -Experimental results on other platforms can be found in -[eval/reports](eval/reports) or [docs/backend](docs/backend). - ---- - -## Roadmap - -Support matrix of models (rows) against platforms (columns). Legend: `Y` = +Models (rows) against platforms (columns). Legend: `Y` = supported (released and benchmarked), `~` = in progress, `-` = planned. | Model | CPU (x86-64 / ARM) | CUDA | [SYCL (Intel)](docs/backend/sycl.md) | [Metal](docs/backend/metal.md) | [OpenVINO](docs/backend/ov.md) | [Hexagon](docs/backend/hexagon-windows.md) | @@ -331,19 +166,76 @@ supported (released and benchmarked), `~` = in progress, `-` = planned. --- -## Contributing +## Rollout on a real robot -See [CONTRIBUTING.md](CONTRIBUTING.md) for how to prove a change is numerically -neutral, and the six sites you touch to add an architecture. +[khanhnd61-vr/lerobot](https://github.com/khanhnd61-vr/lerobot/tree/vla-simd) is a +LeRobot fork whose `lerobot-vla-cpp` client drives an SO-101 arm against a +`vla-server`. The client does the per-arch preprocessing (tokenize, resize, +normalize), so it covers SmolVLA, π0, π0.5 and GR00T N1.5/1.6/1.7. ---- +```bash +git clone -b vla-simd https://github.com/khanhnd61-vr/lerobot.git +cd lerobot && pip install -e ".[vla-cpp]" +``` + +Serve the policy, converted to GGUF as in [docs/MODELS.md](docs/MODELS.md), on a +GPU host. SmolVLA takes 38 ms per query on an RTX 5090 and 218 ms on a Jetson +AGX Orin, against the 1.67 s of motion a 50-step chunk buys at 30 fps; on a +desktop CPU it takes ~1.7 s and the action queue stalls. + +```bash +./build/vla-server smolvla-so101.gguf # binds tcp://*:5555; --bind moves it +``` + +Check the round trip with no arm attached, then run the arm. `ROBOT` holds the +arm's `--robot.*` flags; the fork's README sets it up: -## Contributors +```bash +# prints round-trip latency next to the recorded actions +lerobot-vla-cpp --server_address=tcp://127.0.0.1:5555 --arch=smolvla \ + --replay.repo_id=khanhnd61/so101-multi-task-clean --replay.steps=10 + +lerobot-vla-cpp --server_address=tcp://127.0.0.1:5555 --arch=smolvla "${ROBOT[@]}" \ + --task="Put the tape into the box" --n_action_steps=25 --fps=30 --duration=60 +``` + +- `--arch` selects the preprocessing (`smolvla`, `pi0`, `pi05`, `gr00t_n1_5/6/7`, + or `passthrough` for a server that preprocesses itself). The wrong arch loads, + runs and returns plausible, wrong actions. +- `--stats_json` is required for `pi05` and the GR00T archs; GR00T also takes + `--embodiment`, and `--rel_stats_json` for an N1.7 checkpoint trained with + relative actions. +- `--task` must match a trained instruction exactly, and the camera keys must + stay `front` and `wrist` in that order. +- `vla-server` answers one request at a time, so the loop is synchronous and + `--n_action_steps` is the feedback rate: 25 at 30 fps leaves ~0.83 s between + observations. + +The GR00T paths of the client have not been tested against real checkpoints yet. +Wiring, recording, training and queue sizing are in the +[fork's README](https://github.com/khanhnd61-vr/lerobot/tree/vla-simd#readme). + +--- -- [Khanh Dang Nguyen](https://github.com/khanhnd61-vr) -- [Hung Thinh Ho](https://github.com/hungho77) -- [Chinh Truong Nguyen](https://github.com/nguyentruongchinh04z) -- [An Thai Le](https://github.com/anindex) +## Documentation + +| Doc | Content | +|---|---| +| [docs/PREBUILT.md](docs/PREBUILT.md) | Release tarballs: layout, glibc and CUDA runtime requirements, macOS and Windows notes | +| [docs/USAGE.md](docs/USAGE.md) | `vla-cli`, `vla-server`, `vla-bench`: prompt tokenization, `-hf` tags, runtime flags, environment variables | +| [docs/EVAL.md](docs/EVAL.md) | Installing LIBERO and SimplerEnv, running the eval clients against `vla-server` | +| [docs/MODELS.md](docs/MODELS.md) | Converting safetensors checkpoints to GGUF, quantizing to Q8_0/Q4_0 | +| [docs/DOCKER.md](docs/DOCKER.md) | Building and running the eval in containers | +| [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) | Engine design: layers, the prediction path, backends, adding an architecture | +| [docs/backend/](docs/backend) | Per-backend build and run notes: [SYCL](docs/backend/sycl.md), [OpenVINO](docs/backend/ov.md), [Metal](docs/backend/metal.md), [Hexagon](docs/backend/hexagon.md), [Hexagon on Windows](docs/backend/hexagon-windows.md), [WSL2](docs/backend/wsl.md) | +| [docs/benchmark/](docs/benchmark) | Per-device latency and memory for every model, and the fastest flags per device | +| [docs/KNOWN_ISSUES.md](docs/KNOWN_ISSUES.md) | Known issues and their resolutions | +| [docs/ADOPTION.md](docs/ADOPTION.md) | C ABI, Python bindings, release packaging, and what is left | +| [docs/UPSTREAMING.md](docs/UPSTREAMING.md) | The ggml-openvino fixes to send upstream to llama.cpp | +| [CHANGELOG.md](CHANGELOG.md) | Per-release changes, with the fastest configuration and success rate per model | +| [CONTRIBUTING.md](CONTRIBUTING.md) | Proving a change is numerically neutral, adding an architecture | +| [bindings/python/](bindings/python/README.md) | Python bindings | +| [Learn vla.cpp](https://fai-modelopt-tech.github.io/learn-vla-cpp/) | Walkthrough of the engine design and each policy on ggml | --- diff --git a/bindings/python/README.md b/bindings/python/README.md index ebf406c..6ea354f 100644 --- a/bindings/python/README.md +++ b/bindings/python/README.md @@ -14,24 +14,41 @@ actions = model.predict(frame_hwc_uint8, tokens=[1, 100, 200, 2]) `model.config.real_action_dim` columns carry values. `model.config.denormalized` says whether they are already in world units. -## Finding the library +## Installing -Wheels bundle `libvla.so`. From a source checkout, build it and point at it: +From a checkout of the repo: + +```bash +pip install ./bindings/python +``` + +This builds `libvla.so` with CMake and puts it inside the package, so nothing +else needs to be on the library path. You need a C++17 compiler and network +access (CMake fetches llama.cpp). The build is CPU only (Metal on macOS). It +uses `GGML_NATIVE=OFF`, so the wheel runs on other machines (on x86 it needs +AVX2). +`pip wheel ./bindings/python -w dist` gives you the wheel file. With +`python -m build`, pass `--wheel`: the sdist does not include the C++ sources. + +## Using your own build + +Point `VLA_LIBRARY` at a `libvla.so` from a normal CMake build: ```bash cmake -B build -DCMAKE_BUILD_TYPE=Release cmake --build build -j"$(nproc)" --target vla -VLA_LIBRARY=build/libvla.so LD_LIBRARY_PATH=build:build/bin python your_script.py +VLA_LIBRARY=build/libvla.so python your_script.py ``` -`LD_LIBRARY_PATH` is needed because `libvla.so` links `libvla_core.so` and the -ggml libraries from the same build tree. +`cmake --install build --prefix ` gives a relocatable copy that does not +need the build tree. The libraries go to `/lib`, or `/lib64` on +Fedora and RHEL. ## API | | | |---|---| -| `load(ckpt, mmproj=None, config=None)` | `mmproj` only for SmolVLA, pi0, pi0.5 | +| `load(ckpt_path, mmproj_path=None, config_path=None)` | every arch ignores `mmproj_path`; the `runtime` block of `config_path` sets the precision options | | `Model.predict(images, tokens, state=None, noise=None, ...)` | `images` is one HWC array or a sequence | | `Model.config` | resolved hyper-parameters | | `Model.last_stats()` | per-phase timings, needs `timing=TIMING_PHASE` | diff --git a/bindings/python/pyproject.toml b/bindings/python/pyproject.toml index af666ad..aa23771 100644 --- a/bindings/python/pyproject.toml +++ b/bindings/python/pyproject.toml @@ -1,14 +1,14 @@ [build-system] -requires = ["setuptools>=68"] -build-backend = "setuptools.build_meta" +requires = ["scikit-build-core>=0.11"] +build-backend = "scikit_build_core.build" [project] name = "vla-cpp" -version = "0.2.0" +version = "0.4.0" description = "Python bindings for vla.cpp, a C++ inference engine for Vision-Language-Action models." readme = "README.md" -requires-python = ">=3.9" -license = { text = "Apache-2.0" } +requires-python = ">=3.10" +license = "Apache-2.0" dependencies = [] [project.optional-dependencies] @@ -18,10 +18,18 @@ numpy = ["numpy>=1.24"] [project.urls] Homepage = "https://github.com/VinRobotics/vla.cpp" -[tool.setuptools] -packages = ["vla_cpp"] +[tool.scikit-build] +cmake.source-dir = "../.." +build.targets = ["vla"] +install.components = ["libvla"] +wheel.install-dir = "vla_cpp" +wheel.py-api = "py3" -# libvla is built by cmake, not by setuptools. A wheel bundles it next to the -# package; a source checkout finds it through VLA_LIBRARY or the loader path. -[tool.setuptools.package-data] -vla_cpp = ["*.so", "*.dylib", "*.dll", "lib/*"] +[tool.scikit-build.cmake.define] +VLA_BUILD_SERVER = "OFF" +VLA_SPM = "OFF" +BUILD_SHARED_LIBS = "OFF" +GGML_NATIVE = "OFF" +GGML_OPENMP = "OFF" +CMAKE_INSTALL_LIBDIR = "lib" +CMAKE_INSTALL_BINDIR = "lib" diff --git a/bindings/python/vla_cpp/__init__.py b/bindings/python/vla_cpp/__init__.py index 5893866..10f02a3 100644 --- a/bindings/python/vla_cpp/__init__.py +++ b/bindings/python/vla_cpp/__init__.py @@ -11,6 +11,7 @@ from __future__ import annotations import ctypes +import threading from ctypes import POINTER, c_float, c_int32, c_int64 from typing import Sequence @@ -53,6 +54,7 @@ class Model: """A loaded checkpoint. Free it with ``close()`` or a ``with`` block.""" def __init__(self, handle, lib): + self._mu = threading.Lock() self._h = handle self._lib = lib cfg = _ffi.Config() @@ -69,9 +71,10 @@ def __exit__(self, *exc): return False def close(self): - if getattr(self, "_h", None): - self._lib.vla_model_free(self._h) - self._h = None + with self._mu: + if getattr(self, "_h", None): + self._lib.vla_model_free(self._h) + self._h = None def __del__(self): self.close() @@ -83,9 +86,6 @@ def predict(self, images, tokens: Sequence[int], state=None, noise=None, images: one HWC array, or a sequence of them for multi-view. uint8 RGB by default; pass pixel_format=PIXEL_F32_RGB_01 for float RGB in [0, 1]. """ - if self._h is None: - raise RuntimeError("model is closed") - views = images if isinstance(images, (list, tuple)) else [images] if not views: raise ValueError("at least one image is required") @@ -99,6 +99,9 @@ def predict(self, images, tokens: Sequence[int], state=None, noise=None, shape = mv.shape if len(shape) != 3 or shape[2] != 3: raise ValueError(f"image must be HxWx3, got {shape}") + want = "f" if pixel_format == PIXEL_F32_RGB_01 else "B" + if mv.format != want: + raise ValueError(f"image format {mv.format!r} does not match pixel_format (expected {want!r})") raw = (ctypes.c_char * mv.nbytes).from_buffer_copy(mv) keep.append(raw) img_array[i].data = ctypes.cast(raw, ctypes.c_void_p) @@ -133,13 +136,16 @@ def predict(self, images, tokens: Sequence[int], state=None, noise=None, out = POINTER(c_float)() n = c_int64() - rc = self._lib.vla_predict(self._h, ctypes.byref(cin), ctypes.byref(out), ctypes.byref(n)) - if rc != _ffi.OK: - raise RuntimeError(f"vla_predict failed ({rc})") - try: - flat = [out[i] for i in range(n.value)] - finally: - self._lib.vla_free_actions(out) + with self._mu: + if self._h is None: + raise RuntimeError("model is closed") + rc = self._lib.vla_predict(self._h, ctypes.byref(cin), ctypes.byref(out), ctypes.byref(n)) + if rc != _ffi.OK: + raise RuntimeError(f"vla_predict failed ({rc})") + try: + flat = [out[i] for i in range(n.value)] + finally: + self._lib.vla_free_actions(out) cols = int(self.config.max_action_dim) or 1 rows = len(flat) // cols if cols else len(flat) @@ -151,14 +157,15 @@ def predict(self, images, tokens: Sequence[int], state=None, noise=None, def last_stats(self) -> _ffi.Stats: st = _ffi.Stats() - rc = self._lib.vla_last_stats(self._h, ctypes.byref(st)) + with self._mu: + rc = self._lib.vla_last_stats(self._h, ctypes.byref(st)) if rc != _ffi.OK: raise RuntimeError(f"vla_last_stats failed ({rc})") return st def load(ckpt_path: str, mmproj_path: str | None = None, config_path: str | None = None) -> Model: - """Load a checkpoint. mmproj_path is only needed for SmolVLA, pi0 and pi0.5.""" + """Load a checkpoint. mmproj_path is accepted and ignored; every arch bundles its vision tower.""" lib = _lib_handle() handle = lib.vla_model_load( mmproj_path.encode() if mmproj_path else None, diff --git a/ci/README.md b/ci/README.md index ae36dfe..7696a3c 100644 --- a/ci/README.md +++ b/ci/README.md @@ -1,9 +1,9 @@ # vla.cpp cross-platform CI -Per-PR regression gate for `vla.cpp` across the three target platforms - RTX 3090 -(`rtx3090`), Jetson Orin Nano (`orin`), Apple M4 (`m4`). On a PR from `dev` into -`main`, each platform's `vla-server` is evaluated in LIBERO and gated on success -rate (client side) + latency / memory (server side). +Regression gate for `vla.cpp` across the three target platforms - RTX 3090 +(`rtx3090`), Jetson Orin Nano (`orin`), Apple M4 (`m4`). On every push to `dev`, +each platform's `vla-server` is evaluated in LIBERO and gated on success rate +(client side) + latency / memory (server side). Machines are referred to by **role** (`orchestrator`) and **platform key**; real hostnames / IPs live only in the gitignored `ci/config/hosts.env`. @@ -37,7 +37,9 @@ Every cell is **10 tasks × 1 episode**. ## Gating -- **SR** (client side) - must be **> 0**. +- **SR** (client side) - must be **≥ 0.5× `sr_reported`** of the baseline + (`sr_tolerance` in `ci/baselines/.json` overrides 0.5; **> 0** where a + model has no `sr_reported`). - **Server latency & memory** - must be **≤ 1.10× baseline** (`ci/baselines/.json`). M4 memory has no baseline (recorded, not gated). @@ -65,7 +67,8 @@ $EDITOR ci/config/hosts.env # LAN IPs, ctrl/data ports, repo paths, MODELS_R ### 3. Bring up servers + agents, check with ping On each server (its own git checkout, with `vla-server` already built), run the -agent as a service (systemd / launchd / nohup): +agent as a service (systemd / launchd / nohup). It binds 127.0.0.1 unless given +`--bind`: ```bash ci/agent/build/vla-ci-agent --bind 'tcp://*:5600' [--token "$VLA_CI_TOKEN"] @@ -113,5 +116,17 @@ python ci/check_thresholds.py --platform rtx3090 \ ## CI trigger `.github/workflows/vla-ci.yml` runs `ci/orchestrate.sh all` on a self-hosted -runner labelled `vla-ci-orchestrator` for PRs from `dev` into `main`, and uploads -`outputs/ci/` as an artifact. +runner labelled `vla-ci-orchestrator` on every push to `dev` (or by hand via +workflow_dispatch), and uploads `outputs/ci/` as an artifact. It sets +`VLA_CI_EXPECTED_COMMIT` to the pushed commit, so every server first fetches that +commit from its `origin`, checks it out and rebuilds (`ci/build_servers.sh`); the +gate fails if a server is on any other commit. The runner's `.env` must set +`VLA_CI_HOSTS_ENV` to a hosts.env outside the workspace (checkout wipes ignored +files). + +It does not run on `pull_request`: a PR run takes its workflow file from the PR's +merge commit, so a fork could drop any guard in it and reach the LAN agents. Put +the orchestrator runner in an org runner group whose workflow access is restricted +to `VinRobotics/vla.cpp/.github/workflows/vla-ci.yml@refs/heads/dev`. A path-only +restriction is not enough, because a fork edits the same path. With that group, +a manual dispatch gets a runner only when run from `dev`. diff --git a/ci/agent/agent.cpp b/ci/agent/agent.cpp index de8e101..67d164d 100644 --- a/ci/agent/agent.cpp +++ b/ci/agent/agent.cpp @@ -7,7 +7,7 @@ // // Single-threaded request loop (the orchestrator drives it serially). Spawned // servers are detached into their own session/group so `stop` can signal the -// whole group; dead detached children are reaped at the top of the loop. +// whole group; only `stop` or a respawn reaps them, so a dead one keeps its pgid. // // Security: with --token T (or env VLA_CI_TOKEN) every request must carry a // matching token. Bind to a LAN address only - this runs arbitrary commands by @@ -42,14 +42,6 @@ namespace { std::map g_spawned; // name -> session-leader pid -std::vector to_argv(const google::protobuf::RepeatedPtrField& a) { - std::vector v; - v.reserve(a.size() + 1); - for (const auto& s : a) v.push_back(const_cast(s.c_str())); - v.push_back(nullptr); - return v; -} - void apply_env(const google::protobuf::RepeatedPtrField& env) { for (const auto& kv : env) { auto p = kv.find('='); @@ -297,20 +289,19 @@ void handle(const Request& req, Reply& rep) { int main(int argc, char** argv) { GOOGLE_PROTOBUF_VERIFY_VERSION; - std::string bind = "tcp://*:5600"; + std::string bind = "tcp://127.0.0.1:5600"; std::string token = std::getenv("VLA_CI_TOKEN") ? std::getenv("VLA_CI_TOKEN") : ""; for (int i = 1; i < argc; ++i) { std::string a = argv[i]; if (a == "--bind" && i + 1 < argc) bind = argv[++i]; else if (a == "--token" && i + 1 < argc) token = argv[++i]; else if (a == "-h" || a == "--help") { - std::printf("usage: %s [--bind tcp://*:5600] [--token SECRET]\n", argv[0]); + std::printf("usage: %s [--bind tcp://127.0.0.1:5600] [--token SECRET]\n", argv[0]); return 0; } else { std::fprintf(stderr, "unknown arg: %s\n", a.c_str()); return 2; } } - // A dropped peer must not kill us with SIGPIPE; detached children are reaped - // at the top of the request loop instead. + // A dropped peer must not kill us with SIGPIPE. signal(SIGPIPE, SIG_IGN); zmq::context_t zctx(1); @@ -323,7 +314,6 @@ int main(int argc, char** argv) { zmq::pollitem_t poll[] = {{static_cast(sock), 0, ZMQ_POLLIN, 0}}; for (;;) { - while (waitpid(-1, nullptr, WNOHANG) > 0) {} // reap dead detached children try { zmq::poll(poll, 1, std::chrono::milliseconds(200)); } catch (const zmq::error_t&) { continue; } diff --git a/ci/agent/ctl.cpp b/ci/agent/ctl.cpp index b272543..6d972d1 100644 --- a/ci/agent/ctl.cpp +++ b/ci/agent/ctl.cpp @@ -8,7 +8,7 @@ // ping // put --src DIR --dst DIR [--prune] [--protect P]... (general; CI no longer deploys with it) // exec --cwd DIR -- ARGV... (general; CI no longer builds with it) -// build --cwd DIR [--flags ""] [--jobs N] [--no-patch] [--prelude ""] # build vla-server +// build --cwd DIR [--flags ""] [--jobs N] [--prelude ""] # build vla-server // spawn --name N --cwd DIR --log FILE -- ARGV... // stop --name N // get --remote PATH --local PATH @@ -158,8 +158,10 @@ int main(int argc, char** argv) { std::string jobs = rest("--jobs"); // optional -j value std::string jexpr = jobs.empty() ? "$(getconf _NPROCESSORS_ONLN)" : jobs; std::string prelude = rest("--prelude"); // shell run first, e.g. CUDA exports (Jetson) - std::string patch = has("--no-patch") ? "" : "bash patches/patch.sh; "; - std::string script = "set -e; " + (prelude.empty() ? std::string() : prelude + " ") + patch + + std::string script = "set -e; " + (prelude.empty() ? std::string() : prelude + " ") + + "t=$(bash scripts/llama_tag.sh); " + "grep -qx \"VLA_LLAMA_TAG:STRING=$t\" build/CMakeCache.txt 2>/dev/null || " + "rm -rf build/CMakeCache.txt build/_deps; " "cmake -B build -DCMAKE_BUILD_TYPE=Release " + flags + "; " "cmake --build build -j" + jexpr; auto* e = req.mutable_exec(); diff --git a/ci/build_servers.sh b/ci/build_servers.sh index cae9401..a0e684f 100755 --- a/ci/build_servers.sh +++ b/ci/build_servers.sh @@ -4,8 +4,9 @@ # Build vla-server on the platform servers via the control agent (vla-ci-ctl # build), in parallel. A convenience for the self-managed model: it builds each # server's OWN checkout with that platform's CMake flags (from hosts.env) and does -# NOT change the server's commit. Run it after the servers are at the target -# commit, before `ci/orchestrate.sh all`. +# NOT change the server's commit, unless VLA_CI_EXPECTED_COMMIT is set: then each +# server first fetches that commit from its `origin` and checks it out. Otherwise +# run it after the servers are at the target commit, before `ci/orchestrate.sh all`. # # bash ci/build_servers.sh # all ALL_PLATFORMS, in parallel # bash ci/build_servers.sh rtx3090 orin # a subset @@ -30,8 +31,13 @@ for p in ${PLATFORMS}; do resolve_platform "${p}" || exit 1 ep="tcp://${SRV_HOST}:${CTRL_PORT}" echo "[build] ${p} -> ${ep} cwd=${RROOT} flags=[${CMAKE_FLAGS:-}] prelude=[${BUILD_ENV:+set}] log=${CI_OUTPUT_ROOT}/${p}.build.log" - ( "${CTL}" --endpoint "${ep}" build --cwd "${RROOT}" --flags "${CMAKE_FLAGS}" --prelude "${BUILD_ENV:-}" ) \ - >"${CI_OUTPUT_ROOT}/${p}.build.log" 2>&1 & + ( + if [[ -n "${VLA_CI_EXPECTED_COMMIT:-}" ]]; then + "${CTL}" --endpoint "${ep}" exec --cwd "${RROOT}" -- git fetch --quiet origin "${VLA_CI_EXPECTED_COMMIT}" + "${CTL}" --endpoint "${ep}" exec --cwd "${RROOT}" -- git checkout --quiet --detach "${VLA_CI_EXPECTED_COMMIT}" + fi + "${CTL}" --endpoint "${ep}" build --cwd "${RROOT}" --flags "${CMAKE_FLAGS}" --prelude "${BUILD_ENV:-}" + ) >"${CI_OUTPUT_ROOT}/${p}.build.log" 2>&1 & PID[$p]=$! done diff --git a/ci/check_commits.sh b/ci/check_commits.sh index 7e6b5cb..d8b6ce2 100755 --- a/ci/check_commits.sh +++ b/ci/check_commits.sh @@ -4,10 +4,11 @@ # Verify git-commit consistency across the tested platforms BEFORE running any # sim. Each server self-manages its checkout; the agent's `rev` reports that # server's real `git rev-parse HEAD`. The orchestrator's own commit need NOT -# match the servers. Rules (over the TESTED machines = the platform servers): +# match the servers unless VLA_CI_EXPECTED_COMMIT is set, which then replaces it. +# Rules (over the TESTED machines = the platform servers): # # all four equal (orchestrator + servers) -> INFO -# servers equal, orchestrator differs -> WARNING (proceed) +# servers equal, orchestrator differs -> WARNING (proceed); ERROR, exit 2 with VLA_CI_EXPECTED_COMMIT # servers disagree (or a rev is unreadable) -> ERROR, exit 2 (stop the CI) # # bash ci/check_commits.sh # all ALL_PLATFORMS (the CI default) @@ -26,7 +27,7 @@ CTL="${VLA_CI_CTL:-${CI_DIR}/agent/build/vla-ci-ctl}" export VLA_CI_TOKEN="${VLA_CI_TOKEN:-}" PLATFORMS="${*:-${ALL_PLATFORMS}}" -ORCH_COMMIT="$(git -C "${REPO_ROOT}" rev-parse HEAD 2>/dev/null || echo unknown)" +ORCH_COMMIT="${VLA_CI_EXPECTED_COMMIT:-$(git -C "${REPO_ROOT}" rev-parse HEAD 2>/dev/null || echo unknown)}" mkdir -p "${CI_OUTPUT_ROOT}" declare -A COMMIT @@ -64,6 +65,9 @@ if [[ ${servers_same} -eq 0 ]]; then fi if [[ "${first}" == "${ORCH_COMMIT}" ]]; then echo "INFO: orchestrator and all tested machines are at the same commit (${first})." +elif [[ -n "${VLA_CI_EXPECTED_COMMIT:-}" ]]; then + echo "ERROR: tested machines are at ${first}, not the expected ${VLA_CI_EXPECTED_COMMIT}; stopping CI." >&2 + exit 2 else echo "WARNING: orchestrator (${ORCH_COMMIT}) differs from the tested machines (${first}); proceeding." >&2 fi diff --git a/ci/check_thresholds.py b/ci/check_thresholds.py index fe80414..1dc6f73 100755 --- a/ci/check_thresholds.py +++ b/ci/check_thresholds.py @@ -20,7 +20,9 @@ verdict. Gates (all must pass for exit 0): - * SR > 0 per (model, suite) - "must be a positive number" + * SR >= sr_tol*sr_reported per (model, suite) - sr_tolerance in the + baseline, default 0.5; + SR > 0 without sr_reported * server latency <= tol*baseline per model - mean `total` ms / call * server memory <= tol*baseline per model - platform mem metric (skipped where the @@ -117,7 +119,8 @@ def suite_sr(model_dir: Path, suite: str) -> tuple[int, int, list[int]]: def check_model(name: str, model_dir: Path, logs_dir: Path, base: dict, - latency_metric: str, mem_metric: str | None, tol: float) -> dict: + latency_metric: str, mem_metric: str | None, tol: float, + sr_tol: float) -> dict: res: dict = {"model": name, "suites": {}, "checks": [], "ok": True} def gate(ok: bool, label: str, detail: str): @@ -125,7 +128,8 @@ def gate(ok: bool, label: str, detail: str): if not ok: res["ok"] = False - # ---- SR per suite (gate: > 0) ----------------------------------------- + # ---- SR per suite (gate: >= sr_tol * sr_reported, else > 0) ---------- + base_sr = base.get("sr_reported") suites = discover_suites(model_dir) if not suites: gate(False, "outputs", f"no summary.txt found under {model_dir}") @@ -135,8 +139,14 @@ def gate(ok: bool, label: str, detail: str): sr = succ / eps if eps else 0.0 res["suites"][suite] = {"successes": succ, "episodes": eps, "tasks": len(seen), "sr": sr} - gate(succ > 0, f"SR>0 [{suite}]", - f"{succ}/{eps} success ({sr:.1%}) over {len(seen)} tasks") + if base_sr: + floor = sr_tol * base_sr + gate(sr >= floor - 1e-9, f"SR [{suite}]", + f"{succ}/{eps} success ({sr:.1%}) over {len(seen)} tasks vs " + f"{sr_tol:g}*{base_sr:.1%}={floor:.1%} baseline") + else: + gate(succ > 0, f"SR>0 [{suite}]", + f"{succ}/{eps} success ({sr:.1%}) over {len(seen)} tasks") # ---- server latency (gate: <= tol * baseline) ------------------------- # Per-suite models run several server processes; aggregate (sample-weighted) @@ -188,10 +198,11 @@ def gate(ok: bool, label: str, detail: str): return res -def render_md(platform: str, tol: float, results: list[dict]) -> str: +def render_md(platform: str, tol: float, sr_tol: float, results: list[dict]) -> str: overall = all(r["ok"] for r in results) out = [f"# CI gate - `{platform}` {'PASS' if overall else 'FAIL'}", - "", f"Tolerance: actual ≤ {tol:g}× reported baseline. SR gate: > 0.", ""] + "", f"Tolerance: actual ≤ {tol:g}× reported baseline. " + f"SR gate: ≥ {sr_tol:g}× sr_reported (> 0 without one).", ""] for r in results: out.append(f"## `{r['model']}` {'PASS' if r['ok'] else 'FAIL'}") for c in r["checks"]: @@ -215,6 +226,7 @@ def main() -> int: spec = json.loads(args.baseline.read_text()) tol = float(spec.get("tolerance", 1.10)) + sr_tol = float(spec.get("sr_tolerance", 0.5)) lat_metric = spec.get("latency_metric", "server_total_ms") mem_metric = spec.get("mem_metric") base_models = spec["models"] @@ -238,16 +250,16 @@ def main() -> int: "suites": {}}) continue results.append(check_model(name, model_dir, logs_dir, base_models[name], - lat_metric, mem_metric, tol)) + lat_metric, mem_metric, tol, sr_tol)) overall = bool(results) and all(r["ok"] for r in results) - verdict = {"platform": args.platform, "tolerance": tol, "ok": overall, - "results": results} + verdict = {"platform": args.platform, "tolerance": tol, "sr_tolerance": sr_tol, + "ok": overall, "results": results} out_dir = args.out or args.sweep out_dir.mkdir(parents=True, exist_ok=True) (out_dir / "verdict.json").write_text(json.dumps(verdict, indent=2)) - md = render_md(args.platform, tol, results) + md = render_md(args.platform, tol, sr_tol, results) (out_dir / "verdict.md").write_text(md) print(md) print(f"\n[gate] {'PASS' if overall else 'FAIL'} - wrote {out_dir / 'verdict.json'}") diff --git a/ci/orchestrate.sh b/ci/orchestrate.sh index 99f226c..f0dcc6a 100755 --- a/ci/orchestrate.sh +++ b/ci/orchestrate.sh @@ -34,6 +34,8 @@ sweep_and_gate() { # One LIBERO client per platform runs on the orchestrator at once. Gated metrics # are server-side, so orchestrator load cannot bias them. if [[ "${PLATFORM}" == "all" ]]; then + [[ -z "${VLA_CI_EXPECTED_COMMIT:-}" ]] || bash "${CI_DIR}/build_servers.sh" \ + || { echo "[all] server checkout/build failed - aborting before sim." >&2; exit 1; } # Commit consistency across the tested machines; stops here if they disagree. bash "${CI_DIR}/check_commits.sh" || { echo "[all] commit check failed - aborting before sim." >&2; exit 1; } @@ -59,6 +61,8 @@ fi # ── one platform ──────────────────────────────────────────────────────────── case "${PLATFORM}" in rtx3090|orin|m4) + [[ -z "${VLA_CI_EXPECTED_COMMIT:-}" ]] || bash "${CI_DIR}/build_servers.sh" "${PLATFORM}" \ + || { echo "[${PLATFORM}] server checkout/build failed - aborting before sim." >&2; exit 1; } bash "${CI_DIR}/check_commits.sh" "${PLATFORM}" \ || { echo "[${PLATFORM}] commit check failed - aborting before sim." >&2; exit 1; } sweep_and_gate "${PLATFORM}" ;; diff --git a/cmake/patch_sentencepiece_fpic.cmake b/cmake/patch_sentencepiece_fpic.cmake index 3e9eba6..3371619 100644 --- a/cmake/patch_sentencepiece_fpic.cmake +++ b/cmake/patch_sentencepiece_fpic.cmake @@ -1,4 +1,4 @@ -# sentencepiece v0.2.0 adds -fPIC for every compiler that is not MSVC, and +# sentencepiece v0.2.1 adds -fPIC for every compiler that is not MSVC, and # clang targeting arm64-pc-windows-msvc rejects the flag as a hard error. # Windows has no PIC to ask for, so drop it. Run as the FetchContent patch step # with the working directory at the sentencepiece source root. diff --git a/docs/ADOPTION.md b/docs/ADOPTION.md index 1fd0ba0..d1a10a4 100644 --- a/docs/ADOPTION.md +++ b/docs/ADOPTION.md @@ -7,25 +7,44 @@ Done: 1. **C ABI.** `include/vla.h` and `libvla`. `src/model.h` is C++ only, so without it nothing outside C++ can link the engine. 2. **Python bindings.** `bindings/python`, ctypes over the ABI. -3. **Prebuilt binaries.** `.github/workflows/release.yml` publishes - linux-x86_64 (CPU and CUDA), linux-aarch64 (CPU, for Jetson-class boards), - macos-arm64-metal and a Docker image on tag. -4. **One-command model fetch.** `-hf user/repo[:file.gguf]` on `vla-cli`, - `vla-server` and `vla-bench`, cached under `$VLA_CACHE`. -5. **Reproducible benchmarks.** `vla-bench` emits the README table rows. -6. **Contributor path.** `CONTRIBUTING.md` has the six-site walkthrough for + `pip install ./bindings/python` builds a self-contained wheel with + scikit-build-core. +3. **Prebuilt binaries.** `.github/workflows/release.yml` publishes on tag: + linux-x86_64 (CPU, CUDA 12.8, CUDA 13.4), linux-aarch64 (CPU, and CUDA 13.4 + for Orin, Thor and DGX Spark), macos-arm64-metal, and a Docker image. The + tarballs now run off the build machine: shared libraries ship next to the + binaries with an `$ORIGIN` rpath, builds use `GGML_NATIVE=OFF`, and CI runs + `vla-cli --help` with the build tree moved away. `vla-server` still needs + `libzmq5` from the system. +4. **Install rules.** `cmake --install` gives a relocatable tree with a + `$ORIGIN/../lib` rpath. +5. **One-command model fetch.** `-hf user/repo[:path/file.gguf|:tag]` on + `vla-cli`, `vla-server` and `vla-bench`, cached under `$VLA_CACHE`. A repo + with several GGUFs lists them instead of guessing. +6. **Reproducible benchmarks.** `vla-bench` emits the README table rows. +7. **Contributor path.** `CONTRIBUTING.md` has the six-site walkthrough for adding an architecture, plus issue and PR templates. -7. **Instruction in, action out.** `vla-cli --text` tokenizes with the - architecture's own tokenizer, so the quickstart no longer needs raw ids. +8. **Instruction in, action out.** `vla-cli --text` builds each arch's real + prompt. Octo carries its tokenizer in the GGUF, and + `scripts/add_tokenizer_to_gguf.py` adds one to pi0, pi0.5 and OpenVLA-OFT, so + those need no Python at run time. The other archs call + `scripts/tokenize_prompt.py`. Left: -- **Jetson CUDA binaries.** The aarch64 job is CPU only: the hosted arm64 image - carries no CUDA, so a Jetson GPU build still happens on the device. -- **PyPI.** The wheel is built from `bindings/python` but nothing publishes it. +- **Jetson on JetPack 6.** The tarballs are built on Ubuntu 24.04 and its glibc, + and the aarch64 CUDA one needs a CUDA 13 driver, so JetPack 6 Orins still + build on the device. +- **PyPI.** The wheel builds from `bindings/python`, but nothing publishes it. +- **Hugging Face library registration.** The `vrfai` model cards set + `library_name: vla.cpp`, but vla.cpp is not registered in huggingface.js, so + the Hub shows no "Use this model" snippet. +- **Windows zips and a Homebrew formula.** Windows builds from source only, and + there is no brew tap, though the install rules now make one possible. - **`ci/baselines/rtx3090.json`** still disagrees with the README table, which is now RTX 5090 numbers from `vla-bench`. Re-record the baselines on one machine. - **Success rates.** The README table comes from a May 2026 RTX 3060 sweep and - covers seven of the eleven archs. A fresh sweep would cover the rest. + covers seven of the thirteen archs, and π0, SmolVLA and GR00T N1.7 have moved + since. A fresh sweep would cover the rest. None of these change inference behaviour. diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 0ed9960..4ce8426 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -1,8 +1,8 @@ # Architecture vla.cpp runs Vision-Language-Action (VLA) policies on the ggml/llama.cpp runtime. -Every model is a self-contained GGUF (or a checkpoint plus a vision mmproj) that the -engine loads, detects, and drives on CPU, CUDA, or Metal. This page is the map; the +Every model is a self-contained GGUF that the engine loads, detects, and drives on +CPU, CUDA, Metal, SYCL, OpenVINO, OpenCL or Hexagon. This page is the map; the source is the detail. ## Layers @@ -14,11 +14,21 @@ source is the detail. (or safetensors namespace), and dispatches to the matching factory. - `src/models/*.cpp` - one translation unit per architecture. Each owns its ggml contexts, vision tower, weights, and compute graph. -- `src/models/gguf_reader.h` - the shared GGUF reader (metadata, tensor bytes, - on-demand embedding rows). -- `src/modules/preprocess.h` - small pure vision helpers (pixel-shuffle, view checks). +- `src/gguf_reader.h` - the shared GGUF reader (metadata, tensor bytes, + on-demand embedding rows). `src/loader.h` uploads weights at the resident dtype. +- `src/layers/` - ggml building blocks: linear, attention, FFN, norms, RoPE, + time embeddings. +- `src/modules/` - parts shared between archs: the SigLIP, Qwen3-VL and + DINOv2+SigLIP vision towers, the Gemma action expert, the Qwen3 LM, the DiT + head, and image preprocessing (`preprocess.h`: view checks, CHW normalization). +- `src/scratch_ctx.h` - compute contexts and `graph_cache`, reused across calls. +- `src/backend.h` - backend selection; `src/backend_fallback.cpp` runs the ops an + accelerator rejects on the CPU. +- `src/tokenizer.h` - SentencePiece tokenizers stored in the GGUF, for + `vla-cli --text`. +- `include/vla.h`, `src/vla_c_api.cpp` - the C ABI (`libvla`). - `src/serving/` - `vla-server` (ZeroMQ + protobuf, action prediction), `vlm-server` - (chat), and `vla-cli` (one-shot inference). + (chat), `vla-cli` (one-shot inference) and `vla-bench` (timing). - `src/kernels/bitvla/` - custom 1.58-bit ternary CUDA kernels for BitVLA. ## The prediction path @@ -32,36 +42,41 @@ A forward pass has two stages that most architectures share. 2. **Action head.** A smaller expert reads the prefix and produces an action chunk of shape `[num_steps, max_action_dim]`, where only the first `real_action_dim` columns - carry values and the rest are zero padding. The head comes in three flavours: + carry values and the rest are zero padding. The head comes in four flavours: - **Flow-matching expert** (SmolVLA, pi0, pi0.5): integrates a velocity field with Euler steps from noise at `t=1` to the action at `t=0`. SmolVLA alternates self-attention among action tokens with cross-attention to the prefix. - - **DiT** (GR00T N1.5/1.6/1.7, Evo-1, VLA-JEPA): a diffusion transformer head. + - **DiT** (GR00T N1.5/1.6/1.7, Evo-1, VLA-JEPA): a diffusion transformer head, + also integrated with flow-matching Euler steps. + - **Diffusion** (Octo): DDPM with a small MLP denoiser over the transformer readout. - **Parallel decode** (BitVLA, OpenVLA-OFT, VLA-Adapter): OpenVLA-OFT-style - bidirectional decode of the action tokens in a single pass. + bidirectional decode of the action tokens in a single pass. TurboVLA is also + one pass: an ACT decoder whose learned queries read the fused tokens. -Actions leave `predict` normalised to the training statistics; the caller -un-normalises into world units. +Actions leave `predict` in world units when `Config::denormalized` is true (the +default). GR00T N1.5/1.6/1.7 and VLA-JEPA return normalised actions and the caller +un-normalises. ## Vision deployment -Two patterns, chosen per architecture: - -- **Baked-in tower** (BitVLA, Evo-1, GR00T, OpenVLA-OFT, VLA-Adapter, VLA-JEPA): the - vision tower ships inside the combined GGUF, so a single file is enough. -- **Separate mmproj** (SmolVLA, pi0, pi0.5): the SigLIP tower ships as a second GGUF; - merge it into the checkpoint with `scripts/merge_*_mmproj_to_gguf.py` before loading. +Every architecture ships its vision tower inside the one GGUF. `model_load` and the +binaries still take an mmproj path so older command lines work, but every arch +ignores it. ## Backends and packaging -llama.cpp is fetched and pinned by CMake `FetchContent`; a bump is a one-line -`GIT_TAG` change. Weights are bf16 by default and can be repacked to Q8_0/Q4_0 with -`scripts/quantize_gguf.py`; the loader runs quantized GGUFs directly and lets -`ggml_mul_mat` dequantize at compute. CPU thread count scales to the machine core -count; CUDA and Metal run the towers and the transformer on the GPU. +llama.cpp is fetched by CMake `FetchContent` and pinned by `VLA_LLAMA_TAG` in +`CMakeLists.txt`. Each arch picks a resident dtype for its GEMM weights +(`--weight-dtype` overrides it), and a GGUF can be repacked to Q8_0/Q4_0 with +`scripts/quantize_gguf.py`. +The loader keeps packed weights packed. On CPU and CUDA, ggml quantizes the +activations to 8 bits and runs int8 dot products on the blocks. CPU thread count +scales to the machine core count; the GPU backends run the towers and the +transformer on the device. ## Adding an architecture Extend the `Arch` enum, declare a `*_create` factory in `arch.h`, implement it under `src/models/`, wire detection and dispatch in `src/model.cpp`, and add a converter in -`scripts/`. Reuse `gguf_reader.h` and `modules/preprocess.h` rather than copying them. +`scripts/`. Reuse `gguf_reader.h`, `loader.h`, `layers/` and `modules/` rather than +copying them. [CONTRIBUTING.md](../CONTRIBUTING.md) lists the exact sites. diff --git a/docs/DOCKER.md b/docs/DOCKER.md index 84af4eb..0ea0bf3 100644 --- a/docs/DOCKER.md +++ b/docs/DOCKER.md @@ -22,7 +22,9 @@ below. ### Prerequisites - [Docker Compose](https://docs.docker.com/compose/) v2.24+ -- NVIDIA GPU with proprietary driver ≥ 535. +- NVIDIA driver 525 or newer. The default image is CUDA 12.9, which runs on + older 12.x drivers under CDI. With `--gpus all` on a GeForce card the + driver has to be 575 or newer; see [Known issues](#known-issues). - CDI GPU access for Docker (`devices: - nvidia.com/gpu=all`). See [CUDA GPU access](#cuda-gpu-access) for runtime setup details. @@ -51,11 +53,17 @@ Build args accepted by the server `Dockerfile`: | Arg | Default | Notes | |-----|---------|-------| -| `BACKEND` | `cuda` | `cuda` or `cpu` | -| `CUDA_ARCH` | `120` in Compose, `89` in the `Dockerfile` | Blackwell; `89` for RTX40, `87` for Orin, `86` for RTX30 | -| `BASE_IMAGE` | `nvidia/cuda:12.9.1-devel-ubuntu24.04` | Set to `ubuntu:24.04` when building a CPU image | +| `BACKEND` | `cuda` | `cuda` or `cpu`. `cpu` builds and runs on `ubuntu:24.04` | +| `CUDA_VERSION` | `12.9.1` | Picks the `nvidia/cuda` `-devel` build and `-runtime` run images. `13.4.1` needs driver 580 or newer | +| `CUDA_ARCH` | `120` in Compose, `75-real;80-real;86-real;89-real;90;120-real;121-real` in the `Dockerfile` | One arch builds much faster: `86` RTX30, `89` RTX40, `90` H100, `87` Orin, `120` RTX50 | +| `GGML_NATIVE` | `ON` | Tunes the CPU code for the build machine. `OFF` gives a portable image | | `JOBS` | `nproc` | Lower if nvcc segfaults on flash-attn kernels | +The build runs in a `-devel` stage. The final image holds only `vla-server`, +`vla-cli` and their libraries in `/app`, on the matching `-runtime` base. +A local build is tuned for the CPU it was built on. The published image is +built with `GGML_NATIVE=OFF` and runs on any x86-64 CPU with AVX2. + Override the arch from the environment, `CUDA_ARCH=89 docker compose -f eval/docker-compose.yml build server`, or per build, `docker compose -f eval/docker-compose.yml build --build-arg CUDA_ARCH=89 server`. @@ -84,10 +92,6 @@ YAML docker compose -f eval/docker-compose.yml -f /tmp/vla-compose.override.yml up -d server ``` -> **π0 note**: π0 needs a separate `mmproj` vision GGUF. Pass both files: -> `--bind tcp://*:5555 /models/mmproj-....gguf /models/ckpt.gguf`. -> See the [README model table](../README.md#models) for details. - ### 4. Run a LIBERO evaluation episode ```bash @@ -132,9 +136,7 @@ server image with `BACKEND=cpu`: ### 1. Build the server image for CPU ```bash -docker build -t vla-cpp-cpu \ - --build-arg BACKEND=cpu \ - --build-arg BASE_IMAGE=ubuntu:24.04 . +docker build -t vla-cpp-cpu --build-arg BACKEND=cpu . ``` ### 2. Download the model @@ -153,7 +155,7 @@ docker run -d --name vla-cpp-server -p 5555:5555 \ vla-cpp-cpu --bind tcp://*:5555 /models/smolvla-libero.gguf ``` -Verify with `docker logs vla-cpp-server` — look for `vla-server: bound to tcp://*:5555. ready.` +Verify with `docker logs vla-cpp-server` and look for `vla-server: bound to tcp://*:5555. ready.` ### 4. Run a LIBERO evaluation episode @@ -220,7 +222,8 @@ via hostname `server`. The server service in `eval/docker-compose.yml` uses CDI (`devices: - nvidia.com/gpu=all`). This works when: -1. The NVIDIA proprietary driver is installed (≥ 535). +1. The NVIDIA proprietary driver is installed (525 or newer for the default + CUDA 12.9 image, 580 or newer for `CUDA_VERSION=13.4.1`). 2. A CDI-enabled container runtime is available (containerd ≥ 1.7, cri-o ≥ 1.29, or Docker with `nvidia-ctk` from `nvidia-container-toolkit` ≥ 1.15 to generate `/etc/cdi/nvidia.yaml`). @@ -246,12 +249,13 @@ docker run --rm --gpus all -p5555:5555 \ vla-cpp-server --bind tcp://*:5555 /models/model.gguf ``` +Each release also publishes this image, built for CUDA 12.9 and sm_75 to sm_120: +`ghcr.io/vinrobotics/vla.cpp:` or `:latest`. + ### Server only (CPU) ```bash -docker build -t vla-cpp-cpu \ - --build-arg BACKEND=cpu \ - --build-arg BASE_IMAGE=ubuntu:24.04 . +docker build -t vla-cpp-cpu --build-arg BACKEND=cpu . docker run --rm -p5555:5555 \ -v /tmp/smolvla-models:/models:ro \ @@ -275,7 +279,8 @@ docker run --rm -it --network host \ | Issue | Workaround | |-------|-----------| -| `Unsupported gpu architecture 'compute_120'` with CUDA < 12.8 | Use CUDA 12.8+ for `sm_120`, or set `CUDA_ARCH=89` for RTX40-series compatibility | +| `Unsupported gpu architecture 'compute_121'` (or `compute_120`) | `CUDA_VERSION` is too old for the arch list: 121 needs 12.9, 120 needs 12.8. Use the default `12.9.1`, or pass a `CUDA_ARCH` without them | +| `unsatisfied condition: cuda>=12.9` (or `cuda>=13.4`) with `--gpus all` | The container toolkit waives this check only for datacenter and workstation cards. On GeForce, update the driver, use CDI, or add `-e NVIDIA_DISABLE_REQUIRE=1` (driver 525+ for the default image, 580+ for `CUDA_VERSION=13.4.1`) | | NumPy 2.x: `module 'numpy' has no attribute 'core'` | `Dockerfile.client` pins `numpy==1.26.4` and patches accelerate | | `lerobot` pulls GPU torch | `Dockerfile.client` re-pins `torch==2.5.1` (CPU) after installing lerobot | | LIBERO data files not found | Editable install (`-e`) keeps `bddl_files/` / `init_files/` / `assets/` accessible at runtime | @@ -283,7 +288,7 @@ docker run --rm -it --network host \ | `pandas` segfaults on import | Pin `pandas==2.0.3` (last NumPy 1.x-compatible release) | | MuJoCo 3.x: robosuite init fails | Pin `mujoco<3.0` (2.3.7 known-good) | | `nvidia-container-toolkit` not installed | Use CDI (`devices: - nvidia.com/gpu=all`) instead of `runtime: nvidia` | -| CPU-only: no GPU available | Use the CPU-only `docker build` / `docker run` flow above, or maintain a Compose override that removes `devices: - nvidia.com/gpu=all` and builds with `BACKEND=cpu` plus `BASE_IMAGE=ubuntu:24.04` | +| CPU-only: no GPU available | Use the CPU-only `docker build` / `docker run` flow above, or maintain a Compose override that removes `devices: - nvidia.com/gpu=all` and builds with `BACKEND=cpu` | --- @@ -292,9 +297,9 @@ docker run --rm -it --network host \ The Docker evaluation stack provides a reproducible two-container workflow for vla.cpp: -1. **Server** — upstream `Dockerfile`, compiles `vla-server` with GPU by default - in Compose or with a CPU backend in the standalone CPU flow. -2. **Client** — `eval/Dockerfile.client`, Python simulation stack with pinned +1. **Server**: root `Dockerfile`, builds `vla-server` for CUDA by default or + for CPU with `BACKEND=cpu`, and ships it on a runtime-only base. +2. **Client**: `eval/Dockerfile.client`, Python simulation stack with pinned dependency versions (NumPy 1.x, MuJoCo 2.x, Pandas 2.0.x). 3. **CDI** is the GPU access path used by the checked-in Compose file. 4. **CPU-only** mode works without any GPU through the standalone Docker commands diff --git a/docs/EVAL.md b/docs/EVAL.md new file mode 100644 index 0000000..d6091bd --- /dev/null +++ b/docs/EVAL.md @@ -0,0 +1,86 @@ +# Simulator evaluation + +The eval scaffold under [`eval/`](../eval/) runs `vla-server` against two +simulators end-to-end: LIBERO and SimplerEnv. To run it in containers instead, +see [DOCKER.md](DOCKER.md). + +## Install simulators + +Each setup script bootstraps an isolated Python 3.10 `uv` venv next to itself and +clones the upstream sim repo. Both require [`uv`](https://github.com/astral-sh/uv) +on `PATH`. + +### LIBERO + +```bash +bash eval/sim/libero/setup_libero.sh +``` + +Clones LIBERO into [`eval/sim/libero/LIBERO/`](../eval/sim/libero/LIBERO), +creates `eval/sim/libero/libero_uv/.venv/`, and pins compatible versions of +torch, lerobot, transformers, and gymnasium. + +### SimplerEnv + +```bash +bash eval/sim/simpler/setup_SimplerEnv.sh +``` + +Clones SimplerEnv (and its nested `ManiSkill2_real2sim`) into +[`eval/sim/simpler/SimplerEnv/`](../eval/sim/simpler/SimplerEnv), creates +`eval/sim/simpler/simpler_uv/.venv/`. + +## Running the client + +[`eval/client/`](../eval/client/) drives `vla-server` directly over the protobuf +protocol. Start the server first, see [USAGE.md](USAGE.md#vla-server). + +```bash +export VLA_GGUF=models/smolvla/smolvla-libero.gguf # the checkpoint to serve +export VLA_ARCH=smolvla # client-side arch preset, see --help +``` + +### LIBERO + +With `vla-server` already running: + +```bash +source eval/sim/libero/libero_uv/.venv/bin/activate +python eval/client/run_sim_client_direct.py \ + --task libero_object --task-id 0 --n-episodes 1 \ + --output-dir /tmp/libero_outputs \ + --arch "$VLA_ARCH" +``` + +The GR00T models need two extras: + +- client side: `--stats-json /path/to/dataset_statistics.json` +- server side: `VLA_GR00T_EMBODIMENT` (`new_embodiment` for N1.5, `libero_panda` + for N1.6, `libero_sim` for N1.7). + +### SimplerEnv + +So far only **GR00T-N1.6** is wired (the `gr00t-n1d6-bridge` checkpoint with the +`oxe_widowx` embodiment). Start `vla-server` with the `oxe_widowx` embodiment: + +```bash +VLA_GR00T_EMBODIMENT=oxe_widowx \ + ./build/vla-server "$GR00T_N1D6_GGUF" +``` + +Then drive it from the SimplerEnv venv: + +```bash +source eval/sim/simpler/simpler_uv/.venv/bin/activate +python eval/client/run_simpler_client_direct.py \ + --arch gr00t_n1_6 \ + --task-id oxe_widowx/widowx_spoon_on_towel --n-episodes 1 \ + --embodiment oxe_widowx --image-size 252 \ + --stats-json "$VLA_STATS_JSON" +``` + +## Task success + +Success rate belongs to the checkpoint, not the engine; `vla_predict_check` in +[CONTRIBUTING.md](../CONTRIBUTING.md) is how a change is shown to leave it +alone. Measured LIBERO success rates are in [CHANGELOG.md](../CHANGELOG.md). diff --git a/docs/MODELS.md b/docs/MODELS.md new file mode 100644 index 0000000..f1b8787 --- /dev/null +++ b/docs/MODELS.md @@ -0,0 +1,45 @@ +# Converting and quantizing models + +Each model ships as a single self-contained GGUF; the published ones are listed in +the README's [support matrix](../README.md#support-matrix) and the +[vrfai collection](https://huggingface.co/collections/vrfai/vlacpp-model-bundles). + +## Conversion + +To convert a HuggingFace safetensors checkpoint yourself, [`scripts/`](../scripts/) +has a converter per arch. Set up its venv: + +```bash +python3 -m venv .venv-converter +source .venv-converter/bin/activate +pip install -e ".[convert]" +``` + +Then run any of the per-arch converters (`--help` for the full flag list): + +```bash +python scripts/convert_smolvla_to_gguf.py \ + --ckpt /path/to/smolvla-libero \ + --out /path/to/smolvla-libero-bf16.gguf +``` + +## Quantization + +Most shipped GGUFs are BF16. π0.5, Octo and TurboVLA ship F32, and GR00T N1.5 +and N1.6 are mostly F32. `scripts/quantize_gguf.py` repacks the LM-backbone weight +matrices to a smaller type and copies everything else unchanged. The loader keeps +the packed weights, so the file loads and runs like the original. + +```bash +python scripts/quantize_gguf.py --in model-bf16.gguf --out model-q8_0.gguf --type Q8_0 +``` + +On CPU and CUDA the packed matmuls do not dequantize to float first. ggml +quantizes the activations to 8 bits and runs integer dot products on the blocks +(`vec_dot_q8_0_q8_0` on CPU, the MMQ kernels on CUDA), so a Q8_0 LM is int8 +compute, not BF16. + +`Q8_0` is near-lossless and roughly halves the LM against BF16. `Q4_0` is 4-bit for +a bigger cut (`--type` also takes `Q4_1`, `Q5_0`, `Q5_1`). +Embeddings, the output head, norms and the action expert stay float; pass `--vision` to +pack the vision tower too (smaller, but more accuracy loss). diff --git a/docs/PREBUILT.md b/docs/PREBUILT.md new file mode 100644 index 0000000..e34a1b0 --- /dev/null +++ b/docs/PREBUILT.md @@ -0,0 +1,44 @@ +# Prebuilt binaries + +Each [release](https://github.com/VinRobotics/vla.cpp/releases) has a +`vla.cpp--.tar.gz` that extracts to a `vla.cpp--/` +directory holding `libvla`, `vla.h` and the binaries side by side. The Linux ones +ship `vla-cli`, `vla-bench`, `vla-server` and `vlm-server`; macOS ships `vla-cli` +and `vla-bench`. The platforms and what each needs are in the README's +[Install](../README.md#prebuilt-binaries) section. + +```bash +TAG= +PLATFORM=linux-x86_64-cuda-13.4 +curl -LO https://github.com/VinRobotics/vla.cpp/releases/download/$TAG/vla.cpp-$TAG-$PLATFORM.tar.gz +tar -xzf vla.cpp-$TAG-$PLATFORM.tar.gz +./vla.cpp-$TAG-$PLATFORM/vla-cli --help +``` + +## Linux + +The Linux tarballs are built on Ubuntu 24.04 and do not load on an older glibc +such as Ubuntu 22.04 or JetPack 6; build from source there. Apart from the CUDA +runtime and ZeroMQ, they carry every library they use. + +- `vla-server` and `vlm-server` need `sudo apt install libzmq5`. +- For a CUDA tarball, also extract the matching + `cudart-vla.cpp--.tar.gz` in the same place unless the CUDA + runtime is already installed; it drops `libcudart`, `libcublas` and + `libcublasLt` next to the binaries. + +```bash +curl -LO https://github.com/VinRobotics/vla.cpp/releases/download/$TAG/cudart-vla.cpp-$TAG-$PLATFORM.tar.gz +tar -xzf cudart-vla.cpp-$TAG-$PLATFORM.tar.gz +``` + +## macOS + +The macOS build has no SentencePiece, so `vla-cli --text` tokenizes through +`scripts/tokenize_prompt.py`, which ships in the tarball and needs `transformers` +in the active Python. Keep `default.metallib` next to the binaries. + +## Windows + +There are no Windows builds yet; see [backend/hexagon-windows.md](backend/hexagon-windows.md) +to build from source. The Docker image is covered in [DOCKER.md](DOCKER.md). diff --git a/docs/USAGE.md b/docs/USAGE.md new file mode 100644 index 0000000..52b5d44 --- /dev/null +++ b/docs/USAGE.md @@ -0,0 +1,105 @@ +# Using the binaries + +How to drive `vla-cli` and `vla-server`, and the flags and environment variables +they share. The README's [Quickstart](../README.md#quickstart) covers the +first run. + +## `vla-cli` + +`vla-cli` runs a single prediction without a server or simulator: give it a model, +an image, and an instruction, and it prints the action chunk. Handy for +smoke-testing a GGUF or scripting a quick inference. + +```bash +./build/vla-cli -hf vrfai/smolvla-libero-gguf \ + --image assets/front.jpg --text "pick up the black bowl" --pretty +``` + +`--text` builds the same prompt the eval client sends for that arch. +It is tokenized in-process when the GGUF carries a SentencePiece +tokenizer (Octo, or pi0, pi0.5 and OpenVLA-OFT after +`scripts/add_tokenizer_to_gguf.py --in model.gguf --out model-tok.gguf`). +Otherwise it calls `scripts/tokenize_prompt.py` with the tokenizer the +architecture was trained on, looking in `scripts/` next to `vla-cli`, then +`share/vla`, then the source tree (`VLA_PYTHON` picks the interpreter, +`VLA_TOKENIZE_SCRIPT` overrides the script). Pass `--tokens 1,100,200,2` instead +if you already have ids. +`--pretty` prints one action row per line; +`--state` sets proprioception (defaults to zeros). pi0.5 puts the state into its +prompt, so pi0.5 `--text` needs `--state`. + +## Fetching checkpoints with `-hf` + +`-hf` takes `user/repo`, `user/repo:path/in/repo.gguf`, or `user/repo:tag`, +where the tag is any part of the file path (case-insensitive, like `:Q8_0`). +The BitVLA and GR00T N1.7 repos hold one GGUF per LIBERO suite. When more than +one file matches, `-hf` lists them and stops, so pick one: + +```bash +./build/vla-cli -hf vrfai/gr00tn1d7-libero-gguf:libero_object/gr00tn1d7-libero-object.gguf ... +./build/vla-cli -hf vrfai/gr00tn1d7-libero-gguf:object ... +``` + +Checkpoints are cached under `$VLA_CACHE` (default `~/.cache/vla`). + +## `vla-server` + +`vla-server` loads the model once at startup and answers ZeroMQ REQ/REP requests +synchronously. + +```bash +./build/vla-server "$VLA_GGUF" +``` + +When ready, the server prints: + +``` +vla-server: bound to tcp://*:5555. ready. +``` + +Use `--bind` to change the address and port. Stop the server with `Ctrl-C`. +`vla-server` also takes `-hf user/repo[:file.gguf|:tag]` in place of a checkpoint path. + +Clients: the LIBERO and SimplerEnv runners in [EVAL.md](EVAL.md), and the +real-robot client in the README's +[Rollout on a real robot](../README.md#rollout-on-a-real-robot). + +## Runtime flags + +The same on `vla-server`, `vla-cli` and `vla-bench` (`--help` for the full +list). On `vla-server` and `vla-cli`, the `runtime` block of a `--config` JSON +sets them too, and the command line wins. The fastest configuration per model, +with measured latency and success rate, is in [CHANGELOG.md](../CHANGELOG.md); +per-device fastest flags are in [benchmark/](benchmark/). + +- `--weight-dtype f32|bf16|f16` - resident dtype for GEMM weights. +- `--act-dtype f32|bf16` - activation dtype; bf16 is π0 and Evo-1 only and + needs CUDA and bf16 weights. +- `--flash-attn` - faster on the larger towers, but changes numerics. +- `--mm-prec default|f32` - matmul accumulation precision. +- `--num-steps N` - flow-matching solver steps for π0, π0.5, SmolVLA, Evo-1, + GR00T and VLA-JEPA (default: the checkpoint's). The other archs refuse it. + +## Environment variables + +These apply to every arch: + +- `VLA_N_THREADS` - CPU backend thread count, default core count capped at 16. +- `VLA_DEVICE` - GPU ordinal for CUDA and SYCL builds, default 0. +- `VLA_CACHE` - where `-hf` stores checkpoints, default `~/.cache/vla`. + +Checkpoints that carry stats for several datasets need the one to un-normalize +with, for example `VLA_OCTO_UNNORM_DATASET=libero_object` for the Octo LIBERO +GGUF, which ships four. GR00T needs `VLA_GR00T_EMBODIMENT`, see +[EVAL.md](EVAL.md). + +## Benchmarking + +`vla-bench` times `predict()` in-process on synthetic inputs: engine only, no +transport, no simulator, no claim about task success. + +```bash +./build/vla-bench -hf vrfai/smolvla-libero-gguf --images 2 --size 512 --markdown +``` + +Results per device are in [benchmark/](benchmark/). diff --git a/docs/backend/hexagon-windows.md b/docs/backend/hexagon-windows.md index 686f2dc..0241be4 100644 --- a/docs/backend/hexagon-windows.md +++ b/docs/backend/hexagon-windows.md @@ -1,7 +1,7 @@ # vla.cpp on Snapdragon X (Windows on Arm): Hexagon NPU, Adreno GPU and CPU Measured 2026-09 against llama.cpp build 11201 (`2145525a4`), passed in with -`-LlamaDir`. The `b10729` tag that `CMakeLists.txt` pins was not tested. +`-LlamaDir`. The `b11223` tag that `CMakeLists.txt` pins was not tested. ## Summary @@ -13,7 +13,7 @@ vla.cpp now builds natively on a Snapdragon X laptop in three flavours: `vla-server`, `vla-cli`, `vla-bench` and the tests all build. -Eleven of the twelve published checkpoints run on all three. BitVLA does not: its only published GGUF is int2-packed, which only CUDA builds load. OpenVLA-OFT was not attempted, because at F16 it needs about 14 GB (more than the 8 GB budget). +Eleven of the twelve published checkpoints ran on all three. BitVLA does not: its only published GGUF is int2-packed, which only CUDA builds load. OpenVLA-OFT was not attempted, because at F16 it needs about 14 GB (more than the 8 GB budget). Every accelerator result below was checked against a CPU-backend reference on identical inputs before its latency was recorded. @@ -77,9 +77,9 @@ The build script sets up the Visual Studio shell, the compiler flags llama.cpp's .\scripts\build_windows_snapdragon.ps1 -Backend cpu -LlamaDir ``` -Each build goes into `build-wos-`, with every binary and DLL in `build-wos-\bin`. `-NoServer` skips `vla-server`, Octo and their protobuf and ZeroMQ dependencies. +Each build goes into `build-wos-`, with every binary and DLL in `build-wos-\bin`. `-NoServer` skips `vla-server`, SentencePiece and their protobuf and ZeroMQ dependencies, so Octo `--text` needs `--tokens` there. -`-LlamaDir` points the build at an existing llama.cpp checkout through `FETCHCONTENT_SOURCE_DIR_LLAMA`. Without it, the `b10729` pin in `CMakeLists.txt` applies, which was not tested here. +`-LlamaDir` points the build at an existing llama.cpp checkout through `FETCHCONTENT_SOURCE_DIR_LLAMA`. Without it, the `b11223` pin in `CMakeLists.txt` applies, which was not tested here. The HTP build also signs `libggml-htp-v*.so` with the certificate and copies the skels and their catalog next to the binaries. At startup the Hexagon backend points `ADSP_LIBRARY_PATH` at the executable's own folder, but only if the variable is unset. If it is already set, for example by a llama.cpp install, the skels it names must come from the same llama.cpp commit. @@ -249,6 +249,10 @@ TurboVLA and Octo ship F32 GGUFs, so their "CPU BF16" column is the F32 default. | TurboVLA | 7.1e-3 | 6.0e-4 | 1.1e-6 | | Octo-Small | 0 | 5.2e-4 | 8.2e-4 | +The GR00T N1.5 and N1.6 rows predate flash attention reaching their towers. +Hexagon turns it on by default, so both now run it there; they were not +re-measured. + † Against the CPU BF16 run. An F32 reference for π0.5 needs more memory than the 8 GB budget. The accelerators are often closer to F32 than the CPU running the same F16 weights. ggml's CPU matmul converts activations to the weight's type (F16 here), while HTP and Adreno keep them in F32. @@ -290,7 +294,7 @@ count on top of the host copy made during loading. | GR00T N1.6 | 9.16 → 7.99 GB | 4,930 ms, 7.9e-3 | 3,473 ms, 8.5e-3 | 6,509 ms, 6.6e-3 | | VLA-JEPA | 4.57 → 3.24 GB | 745 ms, 2.5e-2 | 1,477 ms, 4.9e-2 | 2,979 ms, 5.2e-2 | -SmolVLA's CPU Q8_0 run keeps its float weights at F16; the other rows keep the CPU default, BF16. The π0 Q8_0 result on the GPU (0.27) is far worse than the same file on the CPU, and was not investigated. Evo-1's Q8_0 file fails to load on every backend, including the CPU, with a `ggml_view` assertion; that is an arch bug, not a Snapdragon one. +SmolVLA's CPU Q8_0 run keeps its float weights at F16; the other rows keep the CPU default, BF16. The π0 Q8_0 result on the GPU (0.27) is far worse than the same file on the CPU, and was not investigated. Evo-1's Q8_0 file failed to load on every backend with a `ggml_view` assertion; it now loads on the CPU and the NPU. The Adreno GPU refuses one made by the old quantizer, because ggml-opencl ignores view offsets on quantized weights; requantize with the current `scripts/quantize_gguf.py`, which keeps the action expert float. ## Observations diff --git a/docs/backend/hexagon.md b/docs/backend/hexagon.md index ec1e123..d4767a6 100644 --- a/docs/backend/hexagon.md +++ b/docs/backend/hexagon.md @@ -19,8 +19,7 @@ to take once it ran. ## What "IQ9" and "IQ10" refer to -Two commits, both inside `b10729` (`458681e1`, 2026-09-01), and no Hexagon commit -lands after it as of this writing: +Two commits, both inside the pinned `b11223` (`4da63377`, 2026-09-27): | Commit | Date | What it actually did | |---|---|---| @@ -45,12 +44,12 @@ The generational split visible in the code, rather than in the marketing: supported without dynamic discovery"*. Two is what the fallback knows about; the discovery path is open-ended. -## What the backend gives you at `b10729` +## What the backend gives you at `b11223` - **Two libraries.** `libggml-hexagon.so` on the CPU side, `libggml-htp-vNN.so` - on the NPU side. Skels are built for v68, v69, v73, v75, v79 and v81 and the - right one is picked at runtime from `htpdrv_get_arch`; a failed query falls - back to v73. `GGML_HEXAGON_ARCH` overrides. + on the NPU side. Skels are built for v73, v75, v79 and v81 and the right one + is picked at runtime from `htpdrv_get_arch`; a failed query falls back to v73 + and older parts are capped to it. `GGML_HEXAGON_ARCH` overrides. - **Sessions are devices.** Each Hexagon process domain shows up to ggml as one device and behaves like a GPU for offload and model splitting. `GGML_HEXAGON_DEVICES` takes either a count or an explicit @@ -58,17 +57,18 @@ The generational split visible in the code, rather than in the marketing: - **~3.5 GB per session.** The backend now maps and unmaps execution buffers during graph execution to fit larger models into one session, and layer- or tensor-splitting across sessions is the alternative. -- **Repack buffers.** Q4_0, Q4_1, Q8_0, IQ4_NL and MXFP4 weights are repacked - into non-host buffers; since #26501 non-host is the default and - `GGML_HEXAGON_HOSTBUF=1` is the opt-out (needed to exercise `MUL_MAT` in - `test-backend-ops`). +- **Repack buffers.** Q4_0, Q4_1, Q8_0, IQ4_NL, MXFP4, Q4_K, Q5_K and Q6_K + weights are repacked into non-host buffers; since #26501 non-host is the + default and `GGML_HEXAGON_HOSTBUF=1` is the opt-out (needed to exercise + `MUL_MAT` in `test-backend-ops`). - **VTCM is the real budget.** `supports_op` precomputes kernel params for `MUL_MAT`, `FLASH_ATTN_EXT` and friends and returns false when the tile does not fit VTCM. An op is not rejected by shape rules so much as by whether it fits - which means coverage is a function of your tensor sizes, and has to be measured on the board, not predicted from a table. - **Fusion**, controlled by `GGML_HEXAGON_OPFUSION`: `RMS_NORM+MUL`, - `MUL_MAT+ADD`, N-way `MUL_MAT`, `ALLREDUCE+ADD`. + `MUL_MAT+ADD`, N-way `MUL_MAT` and `MUL_MAT_ID`, `ALLREDUCE+ADD`, + `GATED_DELTA_NET+CPY`. The knobs worth knowing on day one: diff --git a/docs/backend/metal.md b/docs/backend/metal.md index 34dbbd8..617cf1e 100644 --- a/docs/backend/metal.md +++ b/docs/backend/metal.md @@ -85,8 +85,8 @@ transport, no simulator, no claim about task success. Apple M5 Max (18-core CPU, 40-core GPU, 64 GB unified memory), macOS 26.6.1, High Power Mode, llama.cpp `b10331`, weights as shipped, 20 reps after 3 warmups, best of three sweeps (the sweep with the lowest p50), each model at its native input size and view count - -the same counts the RTX 5090 table in -[the README](../../README.md#benchmarks) uses, so the two compare cell for cell. +the same counts the +[RTX 5090 report](../benchmark/rtx-5090.md) uses, so the two compare cell for cell. | Model | Views | Input | min ms | p50 ms | p90 ms | vision ms | |---|--:|--:|--:|--:|--:|--:| diff --git a/docs/backend/ov-progress.md b/docs/backend/ov-progress.md new file mode 100644 index 0000000..817609b --- /dev/null +++ b/docs/backend/ov-progress.md @@ -0,0 +1,194 @@ +# OpenVINO backend - progress report + +Status of branch `backend/ov` at `7c2c89e` plus the working-tree changes below, +llama.cpp pinned at `b10729`. Reference doc: [ov.md](ov.md). + +Measured on an Intel Core Ultra X7 358H (Panther Lake), Arc B390 iGPU, AI Boost +NPU, Ubuntu 24.04, OpenVINO 2026.2.1. Every fidelity number here - CPU plugin, +iGPU and NPU - was taken on `b10729`. Only the latency table in [ov.md](ov.md) +still dates from `b10331`. + +## Where it stands + +All ten architectures that can reach this backend are inside the accuracy bar on +the OpenVINO CPU plugin, and nine of the ten on the iGPU as well. The three that used to be wrong +on the GPU - VLA-JEPA, GR00T N1.7 and π0 - are fixed. + +| Arch | CPU plugin | iGPU | NPU | +|---|---:|---:|---:| +| Evo-1 | 2.2e-6 | 6.0e-4 | compiler rejects | +| VLA-Adapter | 3.9e-6 | 5.3e-3 | compiler rejects | +| OpenVLA-OFT | 3.9e-6 | 2.2e-3 | compiler rejects | +| π0.5 | 6.1e-5 | 7.6e-4 | 1.6e-3 | +| VLA-JEPA | 7.7e-5 | 2.6e-3 | returns NaN | +| GR00T N1.7 | 3.9e-4 | 2.6e-3 | NPUW throws | +| π0 | 5.8e-4 | 6.6e-5 | **1.7e0 - wrong** | +| GR00T N1.5 | 6.0e-4 | 1.6e-3 | NPUW throws | +| GR00T N1.6 | 1.1e-3 | 1.4e-3 | NPUW throws | +| SmolVLA | 1.4e-3 | 2.6e-3 | 1.1e-2 | +| BitVLA | pins to CPU by design | - | - | + +The bar is 2.9e-3, the figure the SYCL backend is held to. Each number is +max\|delta\| against whichever CPU-backend reference is tighter for that arch (see +[Baselines](#baselines)); GR00T N1.5 and SmolVLA land closer to BF16, the rest to +F32. The iGPU computes in F16 and is legitimately looser: VLA-Adapter (5.3e-3) +is the one row outside the bar there, on a peak of 0.62, while being 3.9e-6 on the +CPU plugin. Judge translation fidelity on the CPU plugin and treat the GPU as a +separate precision target. π0 is tighter on the GPU than on the CPU plugin because +it is the one arch that runs the GPU at F32 - see below. + +## Fixed this round + +**The position-input fix had gone silently dead at `b10729`.** SmolVLA and π0.5 +returned `action_len=0` and a shape-inference failure: + +```text +Multiply (VariadicSplit[1]:f32[1,113,5,32], Multiply[0]:f32[1,50,1,32]) +Argument shapes are inconsistent. +``` + +113 is SmolVLA's prefix (64 image + 48 language + 1), 50 its suffix. One cos/sin +table, built from one position input, was being applied to both. + +The cause is worth recording because the failure mode is invisible. `b10729` +moved the graph-input naming out of `GgmlOvDecoder::get_graph_input_ov_name()` - +the member the patch guards - into a new free function +`get_tensor_graph_input_ov_name()` at `ggml-decoder.cpp:191`, which hardcodes +`return "inp_pos";`. `compute_model_inputs()` and `set_input_output()` call the +free one; the patched member has zero call sites. **The anchor still matched, the +hunk applied cleanly, and the fix stopped doing anything.** Both functions are +guarded now. + +A hunk that applies is not a hunk that runs. On every `VLA_LLAMA_TAG` bump, check +that each patched function still has a live caller. + +That prompted an audit of the other twelve hunks for the same failure mode: +eleven live, one dead, none refuted by a three-way adversarial check. The dead +one was the imrope shared sin/cos hunk - benign, because its only caller has been +commented out upstream (`// This optimization is error-prone`) since `b10331` and +the per-op path in `translate_rope()` already passes the flag. Retired; two new +ones landed later in the round, leaving thirteen. + +The audit also caught `scripts/upstream_split.py` addressing hunks by position in +the edit list. Adding a hunk to the front of a file's list silently handed every +later hunk to the wrong branch - two were swapped - while its own coverage count +still read 29/29, because each index was still used exactly once. Hunks are now +addressed by a unique substring of their anchor, which fails loudly instead. + +## What it took, in total + +Two changes in vla.cpp, both ordinary correctness fixes invisible on the other +backends: + +- **Weight buffers tagged** `..._WEIGHTS` instead of the default `ANY`, which + ggml-openvino reads as "KV cache". One call site in `src/loader.cpp`. +- **Graph tensors given unique names.** ggml derives a result's name from its + source, so unnamed intermediates collide; ggml-openvino keys its translation + map on those names and silently merges them. `vla::graph_unique_names` at each + `ggml_backend_graph_compute` site; compiles to nothing off OpenVINO. + +Plus one default set in `backend_init` (`GGML_OPENVINO_NAIVE_GRAPH_SIZE`, so +non-LLM graphs take the literal translation path), and thirteen fixes to the +fetched ggml OpenVINO backend applied by `scripts/patch_ggml_openvino.py` at +configure time. The load-bearing ones: + +| Fix | Assumption it breaks | +|---|---| +| Two elementwise adds never stacked on a GEMM | the plugin folds both into the GEMM and drops the second operand | +| PERMUTE op_case 2 requires a ROPE | any permute of a view is a rope'd query | +| GELU translated as tanh, not erf | ggml's `ggml_gelu` is the exact erf form | +| Position inputs keyed per tensor | a graph has exactly one position input | +| Folded weights padded to full rank | a 2-D weight is only ever a GEMM operand | +| Naive-path graph cache | (speed) that path recompiled on every graph_compute | +| GPU inference precision exposed | F16 is always the right trade on the GPU | + +## Fixed: the three archs that were wrong on the iGPU + +Two different bugs wearing the same symptom. + +**VLA-JEPA and GR00T N1.7: two elementwise adds stacked on a GEMM.** The GPU +plugin folds elementwise ops into the preceding GEMM as post-ops. Given +`ADD(ADD(residual, GEMM), graph_input)` it folds both and the second operand is +silently lost - the result equals the inner add, as though the outer one never +ran. A llama.cpp graph never builds that chain; a VLA does, wherever a tower's +features are added on top of an FFN residual. + +Bisected on VLA-JEPA with `GGML_OPENVINO_DEBUG_NODE`, which materialises an +intermediate as an extra `ov::Result` - the only way to observe an interior tensor +here. Its ViT and DiT graphs matched the CPU plugin to 0.2%; the VLM prefill was +already wrong at the end of layer 0; a binary search inside that layer landed on +`node_35`, the FFN residual add under the deepstack add. Only the first three +layers carry a deepstack add, which is why only three nodes mattered. + +Addition is associative, so re-hang the outer add on the inner one's non-GEMM +operand and the GEMM keeps a single post-op. VLA-JEPA 5.4e-1 -> 2.6e-3, GR00T N1.7 +1.9e0 -> 2.6e-3, CPU plugin unchanged. + +Two false starts worth recording. The first attempt tested `inner->src[0]` for the +MUL_MAT, but ggml puts it in `src[1]`, so the guard never fired and the experiment +read as "re-association does not help" when it had simply not run. And a probe +showing the deepstack input bound to different data on each device was an artifact +of the probe itself: a debug `Result` over a Parameter aliases the ggml buffer, +which gallocr had already reused. `ds_pad[0]` is identical on both devices. + +**π0: F16 compounding, not a translation bug.** Its error was almost entirely one +element - step 44, dim 6, off by 1.69 while every other value was within 0.013. +Dim 6 is the gripper, a saturating ±1 channel: the CPU flips it at step 44 and the +GPU at step 45. Split by channel, the continuous dims 0-5 are 4.0e-2 and the +gripper alone produces the 1.69. + +The cause is that π0 unrolls its whole 10-step denoise loop inside a single +6863-node graph, so the GPU plugin's F16 arithmetic compounds across every step +with nothing to reset it. `GGML_OPENVINO_GPU_PRECISION=f32` puts π0 at 6.6e-5. +It costs about 3x (383 ms -> 1,170 ms), so `backend_init` defaults it for π0 alone, +matched on the exact tag `vla(pi0)` - `vla(pi05)` contains that string and π0.5 +neither needs nor gets it, which was checked by measurement. + +## Also found, not fixed + +`r_ctx->device` is the default `"CPU"` on a GPU run. `ov_runtime_context` is +constructed with `device("CPU")` and `get_ov_runtime_context_ptr()` sets it from +`ggml_openvino_get_device_name()`, yet `naive_compute()` observes `"CPU"` while +`ggml_openvino_get_device_name()` returns `"GPU"` and a remote context exists. +Two decisions read that stale string: + +- the `ExecutionMode` hint is applied to the CPU plugin while the model compiles + for the GPU through the remote context, so upstream's ACCURACY/PERFORMANCE + choice never reaches the GPU at all; +- `manual_gqa_enabled` defaults to `device == "GPU"`, which is therefore always + false on a GPU run. + +Neither causes the failures above (both were tested directly), so this was left +alone rather than changed blind. It is a genuine upstream defect and worth a +separate report. + +## Baselines + +OpenVINO folds BF16 weights in as constants and executes them at F32, while +ggml's CPU backend keeps them BF16. Comparing against the default reference +charges the backend for a precision *upgrade* - it made Evo-1 look like 2.7e-3 +when it is 2.2e-6. The BF16 and F32 references bracket the answer and which is +tighter is arch-dependent, so report both. For scale, the CPU backend's own +output moves 2.0e-3 (SmolVLA) or 1.1e-2 (VLA-JEPA) from flipping that one flag. + +Compare with a guard. Four wrong conclusions in this project came from diffing +against a file that was missing or empty, whose signature is +`max|delta| ~= peak|reference|`. `cmpf.sh` refuses to compare unless both files +exist with equal, non-zero value counts. + +## Known issues + +- **Do not set `GGML_OPENVINO_CACHE_DIR`.** OpenVINO's on-disk blob cache returns + a graph that computes the wrong thing on reload - cold run correct, next run + wrong, nothing logged. `backend_init` clears it and says so. +- **NPU accepts three of ten archs and only two are correct.** Compiler + alignment rejections (Evo-1, VLA-Adapter), all-NaN output (VLA-JEPA), and one + NPUW partitioning assertion shared by all three GR00T models - the earlier note + that N1.5 and N1.6 failed *differently* was wrong; at `b10729` all three report + `NPUW: Assertion all_ok failed` at `partitioning.cpp:1350`. None are vla.cpp's + doing. π0 runs but is wrong there, and unlike on the GPU it cannot be fixed: + forcing F32 makes the NPU refuse to compile the model. +- **SmolVLA's `VLA_TIMING=phase` path is wrong under OpenVINO** on every device. + The default path that `vla-server` and `vla-cli` use is correct. +- **NPU needs two extra setup steps** beyond the driver: `libze1`, and + `ZE_ENABLE_ALT_DRIVERS` pointing at `libze_intel_npu.so.1`. diff --git a/docs/backend/ov.md b/docs/backend/ov.md index 103df65..2393349 100644 --- a/docs/backend/ov.md +++ b/docs/backend/ov.md @@ -26,6 +26,8 @@ Measured on an **Intel Core Ultra X7 358H** (Panther Lake) with the Arc B390 iGPU and the AI Boost NPU, Ubuntu 24.04, OpenVINO 2026.2.1, llama.cpp `b10729`, on the checkpoints under `vrfai/` on the Hub. Every **fidelity** number was re-measured on that pin; only the **latency** table still dates from `b10331`. +The build now pins llama.cpp `b11223` and `install_ov.sh` installs OpenVINO +2026.4; the numbers below have not been re-measured on that pin yet. ggml's backend translates a ggml compute graph into an OpenVINO model and hands it to the CPU, GPU or NPU plugin, which compiles and fuses it for the device. @@ -139,9 +141,9 @@ carry them across restarts; it produces silently wrong actions here - see one camera view, best of 4-6 iterations after 3 warmups. "CPU backend" is ggml's own CPU backend on the same 16-core host. No `GGML_OPENVINO_CACHE_DIR`. -Latencies were taken at `b10331` and have not been re-timed on `b10729`. Read the -GPU column for VLA-JEPA and GR00T N1.7 as the cost of a wrong answer at that pin; -both are correct now. +Latencies were taken at `b10331` and have not been re-timed on `b10729` or +`b11223`. Read the GPU column for VLA-JEPA and GR00T N1.7 as the cost of a wrong +answer at that pin; both are correct now. | Model | input | CPU backend | OpenVINO CPU | OpenVINO GPU | OpenVINO NPU | |---|---|---:|---:|---:|---:| @@ -268,7 +270,7 @@ the ggml contract, or fills a gap: | Folded weights padded to full rank | a 2-D weight becomes a rank-2 constant, but views index it at ggml rank | | CONCAT input ranks aligned | same rank-2 constants, and concat cannot broadcast rank | | Missing `GELU_ERF` translator | the exact-erf GELU op had no table entry at all, so a graph using it could not run | -| Naive-path graph cache | that path re-compiled the whole model on every graph_compute, and its `graph_key` is a node count plus two names, which two graphs can share | +| Naive-path graph cache | that path re-compiled the whole model on every graph_compute, and its `graph_key` is a node count plus tensor names, which two graphs of different shapes can share | | Interleaved-mrope sectors bounded | the sector cycle ignored `sections`, so the last few took the wrong stream | | Naive-path threshold settable | the 20-node constant is what picks the literal path | diff --git a/docs/backend/sycl.md b/docs/backend/sycl.md index 08dc291..01adcd5 100644 --- a/docs/backend/sycl.md +++ b/docs/backend/sycl.md @@ -151,24 +151,24 @@ where it pays off. ggml-sycl's copy table used to have `f16 -> f32` but no `bf16 -> f32`, so an arch whose graph contained that copy aborted at predict time. VLA-Adapter hit it with -its default BF16 weights, and the workaround was `VLA_ADAPTER_F32_WEIGHTS=1`. +its default BF16 weights, and the workaround was F32 weights (then +`VLA_ADAPTER_F32_WEIGHTS=1`, now `--weight-dtype f32`). llama.cpp b10326 adds the missing kernel (`cpy_1_bf16_f32` in `ggml/src/ggml-sycl/cpy.cpp`), so VLA-Adapter should run on stock BF16 weights now. Not yet re-tested on the A380 - if you hit the old abort, fall back to -`VLA_ADAPTER_F32_WEIGHTS=1` and file an issue. +`--weight-dtype f32` and file an issue. ## Performance note: F32 weights BF16 has no native DPAS path on Xe-HPG, so BF16 weights are slower there than -plain F32 despite the extra bandwidth. Each arch exposes a switch -(`VLA_WEIGHT_DTYPE=f32` for SmolVLA, `VLA_PI0_F32_WEIGHTS=1` for π0, and so on), -and on the A380 it is worth ~16%: +plain F32 despite the extra bandwidth. `--weight-dtype f32` switches any arch +to F32 weights, and on the A380 it is worth ~16%: | SmolVLA weights | vision | inference | total | |---|---:|---:|---:| | BF16 (default) | 158 ms | 474 ms | **630 ms** | -| F32 (`VLA_WEIGHT_DTYPE=f32`) | 173 ms | 355 ms | **528 ms** | +| F32 (`--weight-dtype f32`) | 173 ms | 355 ms | **528 ms** | The tradeoff is memory - F32 doubles the resident weights (1.07 GiB -> 2.09 GiB for SmolVLA), which matters on a 6 GB A380 for the larger checkpoints. The @@ -187,7 +187,7 @@ threads); GPU is the Arc A380. | Evo-1 | 448 | 7,695 ms | **1,176 ms** | 6.5x | | VLA-Adapter | 224 | 2,994 ms | **517 ms** | 5.8x | -VLA-Adapter is measured with `VLA_ADAPTER_F32_WEIGHTS=1` on both sides, which was +VLA-Adapter is measured with F32 weights on both sides, which was required at the time (see the `bf16 -> f32` section above); the others run their stock defaults. @@ -201,7 +201,7 @@ Per-stage for SmolVLA: SmolVLA gains least because its flow-matching denoise loop is a long chain of small GEMMs that cannot fill 128 EUs; its vision tower alone is 7.1x. With -`VLA_WEIGHT_DTYPE=f32` it reaches 528 ms (3.6x). +`--weight-dtype f32` it reaches 528 ms (3.6x). Outputs were checked against the CPU backend on every model above: max absolute deviation 2.9e-3 on actions peaking at 0.99 (2.9e-6 for the all-F32 diff --git a/docs/backend/wsl.md b/docs/backend/wsl.md index b70fa3e..806bec0 100644 --- a/docs/backend/wsl.md +++ b/docs/backend/wsl.md @@ -90,8 +90,8 @@ cache and that `nvcc` was on `PATH` at configure time. ## Run a model -SmolVLA ships a combined GGUF plus a separate `mmproj` vision tower (see -[Models](../../README.md#models); GGUF published +SmolVLA ships as one GGUF with its vision tower (see +[Models](../MODELS.md); GGUF published [here](https://huggingface.co/collections/vrfai/vlacpp-model-bundles)). ```bash diff --git a/docs/benchmark/README.md b/docs/benchmark/README.md new file mode 100644 index 0000000..805ac91 --- /dev/null +++ b/docs/benchmark/README.md @@ -0,0 +1,38 @@ +# Benchmarks + +Per-device latency and memory for every architecture, measured with `vla-bench` on +synthetic inputs at one commit: `c93ca0a` (PR #32), llama.cpp `b11223`. Each +report lists the device, the build, each model's configuration, latency and peak +memory at the defaults, and the fastest runtime flags for that device. + +The table below gives the minimum latency in ms at the defaults, the one number +every report has. `N/A` means the arch does not run on that backend, and `-` +means the model does not fit or fails there; each report says which and why. + +| Device | Octo-Small | TurboVLA | VLA-JEPA | VLA-Adapter | GR00T N1.5 | GR00T N1.6 | GR00T N1.7 | BitVLA | SmolVLA | π0 | π0.5 | OpenVLA-OFT | Evo-1 | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| [RTX 5090 (CUDA), author's numbers](rtx-5090.md) | 2.69 | 4.88 | 14.3 | 16.0 | 16.1 | 19.4 | 19.5 | 20.8 | 38.2 | 28.8 | 29.2 | 34.5 | 49.1 | +| [RTX 3090 (CUDA)](rtx-3090.md) | 4.3 | 9.0 | 27.3 | 34.2 | 36.6 | 36.3 | 35.3 | 26.9 | 61.0 | 97.5 | 97.8 | 134.2 | 108.6 | +| [RTX 3060 (CUDA)](rtx-3060.md) | 8.3 | 21.6 | 57.1 | 74.7 | 86.8 | 78.1 | 82.8 | 59.1 | 125.1 | 228.2 | 229.9 | - | 230.8 | +| [RTX 5070 Laptop (CUDA)](rtx-5070-laptop.md) | 4.7 | 14.1 | 38.0 | 56.3 | 73.0 | 65.5 | 63.3 | 54.4 | 89.3 | 177.1 | 180.5 | - | 184.6 | +| [Jetson AGX Orin (CUDA)](jetson-agx-orin.md) | 17.5 | 36.0 | 94.6 | 127.4 | 131.2 | 132.6 | 132.7 | 133.0 | 218.3 | 350.3 | 351.9 | 384.9 | 434.6 | +| [Jetson Orin Nano Super (CUDA)](jetson-orin-nano.md) | 31.9 | 89.3 | 193.9 | 297.5 | 293.4 | 272.3 | 271.7 | 335.0 | 462.8 | 845.2 | 849.7 | - | 1011.5 | +| [Apple M4 (Metal)](apple-m4.md) | 22.6 | 65.2 | 248.7 | 315.7 | 401.5 | 358.9 | 350.8 | N/A | 357.6 | 1141.1 | 1153.8 | 1650.7 | 829.3 | +| [Intel Arc A380 (SYCL)](arc-a380.md) | 60.5 | 103.5 | 355.9 | 408.9 | 472.0 | 646.0 | 615.3 | N/A | 734.5 | 921.9 | 945.6 | - | 1047.3 | +| [Snapdragon X Hexagon NPU](snapdragon-x-hexagon.md) | 239.7 | 2280.8 | 821.0 | 4072.1 | 2755.7 | 2940.0 | 1400.2 | N/A | 1239.3 | 5730.0 | 9321.7 | - | 7208.6 | +| [Intel Core i7-14700F (CPU)](core-i7-14700f.md) | 39.8 | 195.1 | 825.5 | 1140.6 | 1913.8 | 1627.9 | 1453.4 | N/A | 1689.2 | 6455.7 | 6068.1 | 9869.0 | 4159.6 | +| [Intel Core i9-14900HX (CPU)](core-i9-14900hx.md) | 67.0 | 314.7 | 1227.2 | 1659.3 | 1935.4 | 1688.9 | 1505.5 | N/A | 1822.6 | 6182.1 | 5735.0 | 9126.4 | 4118.2 | +| [Intel Core i5-12400F (CPU)](core-i5-12400f.md) | 78.1 | 392.7 | 1668.5 | 2165.7 | 2681.4 | 2216.8 | 1973.6 | N/A | 2288.4 | 8953.5 | 8637.9 | - | 5435.0 | +| [AMD Ryzen 5 5500 (CPU)](ryzen-5-5500.md) | 96.6 | 475.7 | 2133.0 | 2781.7 | 3447.0 | 2826.0 | 2518.8 | N/A | 3124.3 | 11321.3 | 11225.6 | - | 7347.2 | +| [Snapdragon X Oryon (CPU)](snapdragon-x-hexagon.md) | 109.5 | 548.9 | 1656.5 | 2013.1 | 2736.6 | 2226.0 | 1988.1 | N/A | 2080.9 | 9224.5 | 9140.6 | - | 5518.9 | + +Notes on the table: + +- The RTX 5090 row comes from the PR #32 author and was not re-measured. Its + numbers are a minimum over 5 alternating rounds, not the protocol the other + reports use (3 warmups + 20 timed reps, best of 3 processes). +- On the i7-14700F, `min` for TurboVLA, VLA-JEPA and VLA-Adapter catches boost + power at the start of a process and sits about 30% below `p50`. Its report + explains this. +- The Intel Core Ultra X7 358H (OpenVINO on the iGPU and NPU, and its CPU) is not + in this round, because the host was offline while it ran. diff --git a/docs/benchmark/TEMPLATE.md b/docs/benchmark/TEMPLATE.md new file mode 100644 index 0000000..3b1325c --- /dev/null +++ b/docs/benchmark/TEMPLATE.md @@ -0,0 +1,177 @@ + + +# () + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + + + +## Test setup + +| | | +|---|---| +| Device | | +| Host | . Drop this row for a CPU-only report | +| Power | | +| Threads | . CPU reports only | +| OS | | +| Driver / toolkit | | +| Commit | `` () | +| llama.cpp | `` | +| Date | to (UTC) | +| Build | `` (`GGML_NATIVE=`) | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + + +**** is . + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + + + +## Latency and memory + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + + + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | MiB | MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| | | | | | | | | | | + +### Fastest configuration + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`; +- `--act-dtype bf16 --flash-attn`, for π0 and Evo-1 on CUDA only. + + + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + + + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | MiB | MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:|--:| +| | `` or *(defaults)* | | | | | | | | <-N% or -> | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| | | | | | | | | | + +
+ + + +## Not run + + + +- ****: + + ```text + + ``` + +## Notes + + + +## Reproducing + +```bash +cmake -S . -B build -DCMAKE_BUILD_TYPE=Release +cmake --build build -j + +# one row, for example SmolVLA +./build/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/apple-m4.md b/docs/benchmark/apple-m4.md new file mode 100644 index 0000000..ace6382 --- /dev/null +++ b/docs/benchmark/apple-m4.md @@ -0,0 +1,141 @@ +# Apple M4 (Metal) + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +## Test setup + +| | | +|---|---| +| Device | Apple M4, 10-core GPU, 24 GB unified memory | +| Host CPU | Apple M4, 4 performance + 6 efficiency cores | +| OS | macOS 26.5.1 (25F80) | +| Toolchain | Apple clang 21.0.0, CMake 4.3.3 | +| Commit | `c93ca0a` (PR #32 head, branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 00:59 to 01:23 (UTC+07) | +| Build | `-DGGML_METAL=ON -DCMAKE_BUILD_TYPE=Release` | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + +Memory is unified, so there is no separate device pool. **Peak RSS** is the process's peak resident set (`getrusage`), which includes the Metal buffers. + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2`) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +## Latency and memory + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 22.6 | 22.9 | 22.9 | 23.0 | 5.7 | 838 | +| TurboVLA | 2 | 256 | 65.2 | 65.4 | 65.4 | 65.5 | - | 1127 | +| VLA-JEPA | 1 | 256 | 248.7 | 249.0 | 248.9 | 249.2 | 86.0 | 4043 | +| VLA-Adapter | 1 | 224 | 315.7 | 316.3 | 316.3 | 316.5 | 174.1 | 3712 | +| GR00T N1.7 | 1 | 256 | 350.8 | 355.0 | 355.2 | 355.6 | 85.9 | 5125 | +| SmolVLA | 2 | 512 | 357.6 | 358.1 | 358.2 | 358.4 | 215.8 | 1208 | +| GR00T N1.6 | 1 | 224 | 358.9 | 361.6 | 361.7 | 362.0 | 100.3 | 4762 | +| GR00T N1.5 | 1 | 224 | 401.5 | 401.9 | 401.9 | 402.2 | 99.4 | 3753 | +| Evo-1 | 1 | 448 | 829.3 | 830.1 | 830.1 | 830.5 | 382.1 | 1637 | +| π0 | 2 | 224 | 1141.1 | 1149.7 | 1146.3 | 1159.6 | 199.5 | 5549 | +| π0.5 | 2 | 224 | 1153.8 | 1165.1 | 1161.1 | 1180.4 | 199.3 | 5963 | +| OpenVLA-OFT | 1 | 224 | 1650.7 | 1716.1 | 1740.0 | 1751.1 | 184.2 | 15479 | + +### Fastest configuration + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | *(defaults)* | 22.6 | 22.9 | 22.9 | 23.0 | 5.7 | 838 | - | +| TurboVLA | `--weight-dtype f16 --flash-attn` | 54.8 | 55.0 | 55.0 | 55.2 | - | 747 | -16% | +| VLA-JEPA | *(defaults)* | 248.7 | 249.0 | 248.9 | 249.2 | 86.0 | 4043 | - | +| VLA-Adapter | *(defaults)* | 315.7 | 316.3 | 316.3 | 316.5 | 174.1 | 3712 | - | +| SmolVLA | `--weight-dtype f16 --flash-attn` | 325.6 | 326.0 | 325.9 | 326.2 | 185.9 | 1208 | -9% | +| GR00T N1.7 | *(defaults)* | 350.8 | 355.0 | 355.2 | 355.6 | 85.9 | 5125 | - | +| GR00T N1.6 | *(defaults)* | 358.9 | 361.6 | 361.7 | 362.0 | 100.3 | 4762 | - | +| GR00T N1.5 | *(defaults)* | 401.5 | 401.9 | 401.9 | 402.2 | 99.4 | 3753 | - | +| Evo-1 | `--weight-dtype f16 --flash-attn` | 771.4 | 772.1 | 772.2 | 772.4 | 327.1 | 1637 | -7% | +| π0 | *(defaults)* | 1141.1 | 1149.7 | 1146.3 | 1159.6 | 199.5 | 5549 | - | +| π0.5 | *(defaults)* | 1153.8 | 1165.1 | 1161.1 | 1180.4 | 199.3 | 5963 | - | +| OpenVLA-OFT | *(defaults)* | 1650.7 | 1716.1 | 1740.0 | 1751.1 | 184.2 | 15479 | - | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 22.9 | 23.1 | 23.2 | 22.9 | 23.1 | 23.1 | 23.1 | 23.0 | +| TurboVLA | 65.4 | 59.3 | 65.6 | 59.3 | 62.1 | 55.7 | 61.3 | 55.3 | +| VLA-JEPA | 249.2 | 251.3 | 249.0 | 250.7 | 249.2 | 250.4 | 247.0 | 248.6 | +| VLA-Adapter | 316.5 | 316.4 | 316.6 | 316.4 | 316.6 | 316.4 | 314.7 | 313.8 | +| GR00T N1.7 | 355.4 | 357.8 | 355.2 | 357.8 | 355.5 | 357.8 | 351.1 | 353.3 | +| SmolVLA | 358.8 | 329.8 | 358.6 | 330.6 | 358.4 | 329.2 | 354.2 | 326.4 | +| GR00T N1.6 | 361.9 | 363.3 | 361.6 | 363.2 | 361.5 | 363.4 | 357.6 | 357.9 | +| GR00T N1.5 | 402.0 | 401.9 | 402.2 | 402.5 | 401.9 | 402.5 | 398.3 | 399.0 | +| Evo-1 | 832.6 | 777.7 | 830.3 | 778.2 | 831.4 | 775.3 | 826.7 | 771.9 | +| π0 | 1179.1 | 1171.9 | 1178.9 | 1172.7 | 1182.7 | 1175.7 | 1172.0 | 1160.1 | +| π0.5 | 1160.7 | 1157.8 | 1156.8 | 1154.8 | 1156.8 | 1153.7 | 1132.6 | 1136.0 | +| OpenVLA-OFT | 1721.8 | 1699.7 | 1684.9 | 1681.7 | 1676.7 | 1672.2 | 1683.9 | 1684.9 | + +
+ +## Not run + +- **BitVLA**: N/A off CUDA. The only published GGUF is int2-packed, and only CUDA builds load it: + + ```text + vla(bitvla): int2-packed GGUF requires a CUDA build (VLA_BITVLA_CUDA_KERNELS); use the bf16 GGUF for CPU. + ``` + +## Reproducing + +```bash +cmake -S . -B build-metal -DGGML_METAL=ON -DCMAKE_BUILD_TYPE=Release +cmake --build build-metal -j + +# one row, for example SmolVLA +./build-metal/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/arc-a380.md b/docs/benchmark/arc-a380.md new file mode 100644 index 0000000..8a8352b --- /dev/null +++ b/docs/benchmark/arc-a380.md @@ -0,0 +1,145 @@ +# Intel Arc A380 (SYCL) + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +## Test setup + +| | | +|---|---| +| Device | Intel Arc A380, 6 GB GDDR6 (Alchemist, DG2) | +| Host | AMD Ryzen 5 5500 (6 cores, 12 threads), 15.5 GiB RAM | +| OS | Ubuntu 22.04.5 LTS, kernel 6.8 (i915) | +| Driver / toolkit | Intel compute runtime 24.39.31294.20, Level Zero 1.17.44 / oneAPI 2025.3 (icx/icpx) | +| Commit | `c93ca0a` (PR #32 head, branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 01:07 to 05:33 (UTC+07) | +| Build | `-DGGML_SYCL=ON -DCMAKE_C_COMPILER=icx -DCMAKE_CXX_COMPILER=icpx -DCMAKE_BUILD_TYPE=Release -G Ninja`, after `source /opt/intel/oneapi/setvars.sh` | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + +**Peak VRAM** is the process's device-local memory from the i915 DRM `fdinfo` counters (`drm-total-local0`), sampled every 0.5 s. **Peak host RSS** is the process's peak resident set (`getrusage`). + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2`) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +## Latency and memory + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak VRAM MiB | Peak host RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 60.5 | 61.0 | 61.0 | 61.3 | 4.8 | 585 | 497 | +| TurboVLA | 2 | 256 | 103.5 | 103.6 | 103.6 | 103.8 | - | 885 | 561 | +| VLA-JEPA | 1 | 256 | 355.9 | 356.8 | 356.8 | 357.3 | 109.3 | 3855 | 468 | +| VLA-Adapter | 1 | 224 | 408.9 | 410.0 | 410.2 | 410.4 | 193.7 | 2750 | 1102 | +| GR00T N1.5 | 1 | 224 | 472.0 | 472.7 | 472.7 | 473.3 | 114.8 | 3531 | 496 | +| GR00T N1.7 | 1 | 256 | 615.3 | 618.4 | 618.2 | 620.9 | 109.9 | 4913 | 517 | +| GR00T N1.6 | 1 | 224 | 646.0 | 647.6 | 647.9 | 648.4 | 114.8 | 4558 | 493 | +| SmolVLA | 2 | 512 | 734.5 | 736.5 | 736.5 | 738.0 | 231.6 | 1159 | 508 | +| π0 | 2 | 224 | 921.9 | 924.2 | 924.1 | 925.4 | 228.3 | 5366 | 513 | +| π0.5 | 2 | 224 | 945.6 | 948.1 | 948.2 | 950.0 | 228.3 | 5659 | 513 | +| Evo-1 | 1 | 448 | 1047.3 | 1048.8 | 1048.8 | 1049.8 | 354.6 | 1467 | 544 | + +### Fastest configuration + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak VRAM MiB | Peak host RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | *(defaults)* | 60.5 | 61.0 | 61.0 | 61.3 | 4.8 | 585 | 497 | - | +| TurboVLA | `--weight-dtype f16` | 89.7 | 89.9 | 89.9 | 90.1 | - | 541 | 561 | -13% | +| VLA-JEPA | *(defaults)* | 355.9 | 356.8 | 356.8 | 357.3 | 109.3 | 3855 | 468 | - | +| VLA-Adapter | *(defaults)* | 408.9 | 410.0 | 410.2 | 410.4 | 193.7 | 2750 | 1102 | - | +| GR00T N1.5 | *(defaults)* | 472.0 | 472.7 | 472.7 | 473.3 | 114.8 | 3531 | 496 | - | +| GR00T N1.7 | *(defaults)* | 615.3 | 618.4 | 618.2 | 620.9 | 109.9 | 4913 | 517 | - | +| GR00T N1.6 | *(defaults)* | 646.0 | 647.6 | 647.9 | 648.4 | 114.8 | 4558 | 493 | - | +| SmolVLA | *(defaults)* | 734.5 | 736.5 | 736.5 | 738.0 | 231.6 | 1159 | 508 | - | +| π0 | *(defaults)* | 921.9 | 924.2 | 924.1 | 925.4 | 228.3 | 5366 | 513 | - | +| π0.5 | *(defaults)* | 945.6 | 948.1 | 948.2 | 950.0 | 228.3 | 5659 | 513 | - | +| Evo-1 | *(defaults)* | 1047.3 | 1048.8 | 1048.8 | 1049.8 | 354.6 | 1467 | 544 | - | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 61.4 | 65.1 | 71.3 | 65.4 | 65.3 | 61.4 | 64.6 | 62.7 | +| TurboVLA | 103.5 | 110.7 | 104.0 | 110.7 | 100.3 | 103.9 | 90.0 | 94.1 | +| VLA-JEPA | 356.6 | 346.2 | 356.9 | 346.7 | 358.2 | 346.1 | 558.6 | 552.8 | +| VLA-Adapter | 410.2 | 410.0 | 410.2 | 410.2 | 410.2 | 410.0 | 573.5 | 573.4 | +| GR00T N1.5 | 471.3 | 485.3 | 470.3 | 485.3 | 472.6 | 484.3 | 816.2 | 829.6 | +| GR00T N1.7 | 619.4 | 608.7 | 618.7 | 610.8 | 619.4 | 610.5 | 1058.2 | 1053.0 | +| GR00T N1.6 | 646.8 | 658.3 | 646.5 | 661.8 | 646.4 | 660.8 | 1043.1 | 1061.3 | +| SmolVLA | 736.8 | 953.3 | 737.1 | 950.1 | 735.3 | 951.0 | 814.8 | 1035.3 | +| π0 | 923.4 | 1389.6 | 922.7 | 1391.1 | 923.3 | 1389.7 | 1885.7 | 2345.4 | +| π0.5 | 946.7 | 948.1 | 946.9 | 947.5 | 946.7 | 947.9 | 1910.3 | 1911.4 | +| Evo-1 | 1048.1 | 1317.8 | 1048.5 | 1318.8 | 1048.7 | 1318.5 | 1393.3 | 1666.7 | + +
+ +## Not run + +- **BitVLA**: N/A off CUDA. The only published GGUF is int2-packed, and only CUDA builds load it: + + ```text + vla(bitvla): int2-packed GGUF requires a CUDA build (VLA_BITVLA_CUDA_KERNELS); use the bf16 GGUF for CPU. + ``` +- **OpenVLA-OFT**: does not fit. The weights load, but the first predict runs out of memory: + + ```text + level_zero backend failed with error: 38 (UR_RESULT_ERROR_OUT_OF_HOST_MEMORY) exception caught at ggml/src/ggml-sycl/ggml-sycl.cpp, line:5835 + Error OP CONCAT + ``` + +## Reproducing + +```bash +source /opt/intel/oneapi/setvars.sh +cmake -S . -B build-sycl -G Ninja -DGGML_SYCL=ON -DCMAKE_C_COMPILER=icx -DCMAKE_CXX_COMPILER=icpx -DCMAKE_BUILD_TYPE=Release +cmake --build build-sycl -j + +# one row, for example SmolVLA +./build-sycl/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/core-i5-12400f.md b/docs/benchmark/core-i5-12400f.md new file mode 100644 index 0000000..c5bee69 --- /dev/null +++ b/docs/benchmark/core-i5-12400f.md @@ -0,0 +1,141 @@ +# Intel Core i5-12400F (CPU) + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +## Test setup + +| | | +|---|---| +| Device | Intel Core i5-12400F, 6 performance cores, 12 threads (Alder Lake) | +| Memory | 15.4 GiB | +| Power | PL1 = PL2 = 241 W (effectively unlimited); `intel_pstate` `powersave` governor | +| Threads | 12 (the default: all hardware threads, capped at 16) | +| OS | Ubuntu 22.04.5 LTS, kernel 6.8 | +| Toolchain | GCC 11.4, CMake 3.22 | +| Commit | `c93ca0a` (PR #32 head, branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 01:01 to 04:10 (UTC+07) | +| Build | `-DCMAKE_BUILD_TYPE=Release` (`GGML_NATIVE=ON`), CPU backend only | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + +**Peak RSS** is the process's peak resident set (`getrusage`). + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2`) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +## Latency and memory + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 78.1 | 78.9 | 78.5 | 80.4 | 9.5 | 623 | +| TurboVLA | 2 | 256 | 392.7 | 394.9 | 393.9 | 396.6 | - | 855 | +| VLA-JEPA | 1 | 256 | 1668.5 | 1674.6 | 1674.4 | 1680.9 | 568.6 | 3823 | +| GR00T N1.7 | 1 | 256 | 1973.6 | 1982.7 | 1981.6 | 1991.9 | 567.5 | 4884 | +| VLA-Adapter | 1 | 224 | 2165.7 | 2179.3 | 2177.6 | 2186.4 | 1285.5 | 2707 | +| GR00T N1.6 | 1 | 224 | 2216.8 | 2228.1 | 2227.0 | 2234.8 | 699.7 | 4524 | +| SmolVLA | 2 | 512 | 2288.4 | 2304.3 | 2296.6 | 2338.6 | 1520.8 | 1101 | +| GR00T N1.5 | 1 | 224 | 2681.4 | 2692.9 | 2691.6 | 2702.3 | 697.4 | 3500 | +| Evo-1 | 1 | 448 | 5435.0 | 5474.8 | 5456.3 | 5521.2 | 2682.3 | 1432 | +| π0.5 | 2 | 224 | 8637.9 | 8653.7 | 8651.9 | 8671.8 | 1388.6 | 5657 | +| π0 | 2 | 224 | 8953.5 | 8979.1 | 8975.9 | 8994.5 | 1413.9 | 5356 | + +### Fastest configuration + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | *(defaults)* | 78.1 | 78.9 | 78.5 | 80.4 | 9.5 | 623 | - | +| TurboVLA | `--weight-dtype f16 --flash-attn` | 357.7 | 361.0 | 359.6 | 364.8 | - | 490 | -9% | +| VLA-JEPA | *(defaults)* | 1668.5 | 1674.6 | 1674.4 | 1680.9 | 568.6 | 3823 | - | +| GR00T N1.7 | *(defaults)* | 1973.6 | 1982.7 | 1981.6 | 1991.9 | 567.5 | 4884 | - | +| SmolVLA | `--weight-dtype f16 --flash-attn` | 2044.6 | 2064.2 | 2057.7 | 2082.8 | 1274.9 | 1051 | -10% | +| VLA-Adapter | *(defaults)* | 2165.7 | 2179.3 | 2177.6 | 2186.4 | 1285.5 | 2707 | - | +| GR00T N1.6 | *(defaults)* | 2216.8 | 2228.1 | 2227.0 | 2234.8 | 699.7 | 4524 | - | +| GR00T N1.5 | *(defaults)* | 2681.4 | 2692.9 | 2691.6 | 2702.3 | 697.4 | 3500 | - | +| Evo-1 | `--flash-attn --mm-prec default` | 5061.5 | 5081.8 | 5076.1 | 5105.2 | 2299.2 | 1368 | -7% | +| π0.5 | *(defaults)* | 8637.9 | 8653.7 | 8651.9 | 8671.8 | 1388.6 | 5657 | - | +| π0 | *(defaults)* | 8953.5 | 8979.1 | 8975.9 | 8994.5 | 1413.9 | 5356 | - | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 78.4 | 82.5 | 78.6 | 79.2 | 79.3 | 79.4 | 79.5 | 80.3 | +| TurboVLA | 394.3 | 385.4 | 395.5 | 383.8 | 370.7 | 367.3 | 373.0 | 359.3 | +| VLA-JEPA | 1714.4 | 1666.4 | 1680.4 | 1668.4 | 1679.2 | 1663.7 | 1671.2 | 1671.1 | +| GR00T N1.7 | 1998.3 | 1977.2 | 1987.0 | 1977.7 | 1983.8 | 1975.4 | 1980.1 | 1978.2 | +| VLA-Adapter | 2197.7 | 2202.3 | 2182.5 | 2204.5 | 2180.8 | 2192.4 | 2191.2 | 2197.0 | +| GR00T N1.6 | 2226.2 | 2218.9 | 2233.0 | 2216.1 | 2225.4 | 2223.2 | 2226.1 | 2225.0 | +| SmolVLA | 2303.8 | 2063.9 | 2304.8 | 2082.2 | 2303.0 | 2078.4 | 2293.9 | 2053.6 | +| GR00T N1.5 | 2694.7 | 2671.9 | 2688.7 | 2667.7 | 2694.0 | 2665.6 | 2685.1 | 2666.2 | +| Evo-1 | 5474.7 | 5117.7 | 5490.6 | 5086.9 | 5469.6 | 5111.8 | 5470.4 | 5105.1 | +| π0.5 | 8663.5 | 8669.3 | 8665.7 | 8655.7 | 8666.1 | 8656.4 | 8648.9 | 8637.0 | +| π0 | 8974.5 | 8777.1 | 8973.5 | 8753.6 | 8966.5 | 8763.8 | 8980.0 | 8754.8 | + +
+ +## Not run + +- **BitVLA**: N/A off CUDA. The only published GGUF is int2-packed, and only CUDA builds load it: + + ```text + vla(bitvla): int2-packed GGUF requires a CUDA build (VLA_BITVLA_CUDA_KERNELS); use the bf16 GGUF for CPU. + ``` +- **OpenVLA-OFT**: not attempted. Its 14 GiB of weights do not fit in 15 GiB of RAM. + +## Reproducing + +```bash +cmake -S . -B build-cpu -DCMAKE_BUILD_TYPE=Release +cmake --build build-cpu -j + +# one row, for example SmolVLA +./build-cpu/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/core-i7-14700f.md b/docs/benchmark/core-i7-14700f.md new file mode 100644 index 0000000..0061deb --- /dev/null +++ b/docs/benchmark/core-i7-14700f.md @@ -0,0 +1,147 @@ +# Intel Core i7-14700F (CPU) + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +## Test setup + +| | | +|---|---| +| Device | Intel Core i7-14700F, 8 performance + 12 efficiency cores, 28 threads (Raptor Lake) | +| Memory | 62.6 GiB | +| Power | PL1 65 W (the rated base power) over a 27 s window, no short-term cap; `intel_pstate` `powersave` governor | +| Threads | 16 (the default: all hardware threads, capped at 16) | +| OS | Ubuntu 22.04.5 LTS, kernel 6.8 | +| Toolchain | GCC 11.4, CMake 3.22 | +| Commit | `c93ca0a` (PR #32 head, branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 00:59 to 03:34 (UTC+07) | +| Build | `-DCMAKE_BUILD_TYPE=Release` (`GGML_NATIVE=ON`), CPU backend only | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + +**Peak RSS** is the process's peak resident set (`getrusage`). + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2`) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +## Latency and memory + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 39.8 | 39.9 | 39.9 | 39.9 | 5.4 | 622 | +| TurboVLA | 2 | 256 | 195.1 | 257.5 | 285.7 | 294.8 | - | 854 | +| VLA-JEPA | 1 | 256 | 825.5 | 1192.8 | 1211.7 | 1213.2 | 394.6 | 3822 | +| VLA-Adapter | 1 | 224 | 1140.6 | 1585.4 | 1599.6 | 1602.2 | 918.7 | 2707 | +| GR00T N1.7 | 1 | 256 | 1453.4 | 1455.2 | 1455.5 | 1456.2 | 403.6 | 4882 | +| GR00T N1.6 | 1 | 224 | 1627.9 | 1631.5 | 1631.1 | 1633.8 | 500.6 | 4524 | +| SmolVLA | 2 | 512 | 1689.2 | 1691.4 | 1691.5 | 1692.7 | 1083.6 | 1100 | +| GR00T N1.5 | 1 | 224 | 1913.8 | 1915.1 | 1915.1 | 1916.2 | 500.4 | 3499 | +| Evo-1 | 1 | 448 | 4159.6 | 4164.7 | 4164.5 | 4168.4 | 1941.3 | 1431 | +| π0.5 | 2 | 224 | 6068.1 | 6078.0 | 6077.8 | 6083.2 | 993.9 | 5657 | +| π0 | 2 | 224 | 6455.7 | 6465.7 | 6467.3 | 6471.3 | 1022.3 | 5355 | +| OpenVLA-OFT | 1 | 224 | 9869.0 | 9881.1 | 9882.0 | 9892.2 | 963.3 | 14797 | + +### Fastest configuration + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | *(defaults)* | 39.8 | 39.9 | 39.9 | 39.9 | 5.4 | 622 | - | +| TurboVLA | *(defaults)* | 195.1 | 257.5 | 285.7 | 294.8 | - | 854 | - | +| VLA-JEPA | *(defaults)* | 825.5 | 1192.8 | 1211.7 | 1213.2 | 394.6 | 3822 | - | +| VLA-Adapter | *(defaults)* | 1140.6 | 1585.4 | 1599.6 | 1602.2 | 918.7 | 2707 | - | +| GR00T N1.7 | *(defaults)* | 1453.4 | 1455.2 | 1455.5 | 1456.2 | 403.6 | 4882 | - | +| SmolVLA | `--weight-dtype bf16 --flash-attn` | 1546.8 | 1548.7 | 1547.7 | 1549.2 | 938.6 | 1049 | -8% | +| GR00T N1.6 | *(defaults)* | 1627.9 | 1631.5 | 1631.1 | 1633.8 | 500.6 | 4524 | - | +| GR00T N1.5 | *(defaults)* | 1913.8 | 1915.1 | 1915.1 | 1916.2 | 500.4 | 3499 | - | +| Evo-1 | `--flash-attn --mm-prec default` | 3893.4 | 3897.0 | 3896.2 | 3900.1 | 1673.1 | 1367 | -6% | +| π0.5 | *(defaults)* | 6068.1 | 6078.0 | 6077.8 | 6083.2 | 993.9 | 5657 | - | +| π0 | *(defaults)* | 6455.7 | 6465.7 | 6467.3 | 6471.3 | 1022.3 | 5355 | - | +| OpenVLA-OFT | *(defaults)* | 9869.0 | 9881.1 | 9882.0 | 9892.2 | 963.3 | 14797 | - | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 39.8 | 40.4 | 40.4 | 40.2 | 39.9 | 40.3 | 39.8 | 39.9 | +| TurboVLA | 283.4 | 275.6 | 285.2 | 276.1 | 283.4 | 275.2 | 288.2 | 280.3 | +| VLA-JEPA | 1205.1 | 1188.5 | 1210.3 | 1181.9 | 1206.3 | 1180.9 | 1233.1 | 1212.2 | +| VLA-Adapter | 1600.7 | 1593.4 | 1603.6 | 1599.4 | 1594.6 | 1610.8 | 1630.2 | 1625.4 | +| GR00T N1.7 | 1455.0 | 1425.1 | 1453.0 | 1420.5 | 1460.8 | 1428.2 | 1449.1 | 1428.4 | +| GR00T N1.6 | 1635.1 | 1615.2 | 1638.0 | 1607.2 | 1638.1 | 1616.4 | 1621.8 | 1596.3 | +| SmolVLA | 1698.8 | 1548.5 | 1703.4 | 1549.6 | 1701.8 | 1546.0 | 1708.5 | 1553.6 | +| GR00T N1.5 | 1914.7 | 1904.8 | 1950.8 | 1984.3 | 2013.8 | 1983.6 | 2078.3 | 2050.8 | +| Evo-1 | 4171.1 | 3907.8 | 4163.4 | 3895.0 | 4161.3 | 3918.0 | 4231.3 | 3969.5 | +| π0.5 | 6048.1 | 6112.8 | 6111.3 | 6092.5 | 6132.1 | 6097.4 | 6142.4 | 6139.0 | +| π0 | 6475.1 | 6410.0 | 6488.2 | 6376.5 | 6519.9 | 6412.5 | 6527.6 | 6399.9 | +| OpenVLA-OFT | 9900.2 | 9946.7 | 9905.0 | 9888.6 | 9852.2 | 9933.6 | 10096.4 | 10101.2 | + +
+ +## Not run + +- **BitVLA**: N/A off CUDA. The only published GGUF is int2-packed, and only CUDA builds load it: + + ```text + vla(bitvla): int2-packed GGUF requires a CUDA build (VLA_BITVLA_CUDA_KERNELS); use the bf16 GGUF for CPU. + ``` + +## Notes + +The CPU runs at a long-term power limit of 65 W, its rated base power, averaged over 27 s, with no short-term cap. The first timed calls of a process can still run at boost power. For the models whose warmup ends inside that window (TurboVLA, VLA-JEPA and VLA-Adapter), `min` is therefore about 30% below `p50`. Read `mean` or `p50` for sustained latency. + +## Reproducing + +```bash +cmake -S . -B build-cpu -DCMAKE_BUILD_TYPE=Release +cmake --build build-cpu -j + +# one row, for example SmolVLA +./build-cpu/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/core-i9-14900hx.md b/docs/benchmark/core-i9-14900hx.md new file mode 100644 index 0000000..9740eee --- /dev/null +++ b/docs/benchmark/core-i9-14900hx.md @@ -0,0 +1,143 @@ +# Intel Core i9-14900HX (CPU) + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +## Test setup + +| | | +|---|---| +| Device | Intel Core i9-14900HX (laptop), 8 performance + 16 efficiency cores, 32 threads; on AC power, platform profile `performance` | +| Memory | 31 GiB | +| Power | PL1 = PL2 = 250 W; `intel_pstate` `powersave` governor | +| Threads | 16 (the default: all hardware threads, capped at 16) | +| OS | Ubuntu 22.04.5 LTS, kernel 6.8 | +| Toolchain | GCC 11.4, CMake 3.22 | +| Commit | `c93ca0a` (PR #32 head, branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 00:59 to 03:56 (UTC+07) | +| Build | `-DCMAKE_BUILD_TYPE=Release` (`GGML_NATIVE=ON`), CPU backend only | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + +**Peak RSS** is the process's peak resident set (`getrusage`). + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2`) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +## Latency and memory + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 67.0 | 71.8 | 72.2 | 75.0 | 8.4 | 622 | +| TurboVLA | 2 | 256 | 314.7 | 325.9 | 325.5 | 333.7 | - | 854 | +| VLA-JEPA | 1 | 256 | 1227.2 | 1273.3 | 1280.7 | 1294.7 | 400.9 | 3822 | +| GR00T N1.7 | 1 | 256 | 1505.5 | 1530.8 | 1533.0 | 1545.2 | 400.5 | 4883 | +| VLA-Adapter | 1 | 224 | 1659.3 | 1689.6 | 1684.4 | 1729.9 | 980.8 | 2706 | +| GR00T N1.6 | 1 | 224 | 1688.9 | 1721.9 | 1717.9 | 1745.8 | 496.1 | 4524 | +| SmolVLA | 2 | 512 | 1822.6 | 1855.6 | 1854.4 | 1874.7 | 1179.1 | 1099 | +| GR00T N1.5 | 1 | 224 | 1935.4 | 1966.1 | 1968.8 | 1983.0 | 494.0 | 3499 | +| Evo-1 | 1 | 448 | 4118.2 | 4155.7 | 4152.5 | 4184.5 | 1875.2 | 1431 | +| π0.5 | 2 | 224 | 5735.0 | 5913.9 | 5799.2 | 6092.9 | 1001.6 | 5657 | +| π0 | 2 | 224 | 6182.1 | 6342.8 | 6250.5 | 6507.3 | 1035.7 | 5353 | +| OpenVLA-OFT | 1 | 224 | 9126.4 | 9636.7 | 9836.8 | 9854.8 | 1008.6 | 14796 | + +### Fastest configuration + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | *(defaults)* | 67.0 | 71.8 | 72.2 | 75.0 | 8.4 | 622 | - | +| TurboVLA | *(defaults)* | 314.7 | 325.9 | 325.5 | 333.7 | - | 854 | - | +| VLA-JEPA | *(defaults)* | 1227.2 | 1273.3 | 1280.7 | 1294.7 | 400.9 | 3822 | - | +| GR00T N1.7 | *(defaults)* | 1505.5 | 1530.8 | 1533.0 | 1545.2 | 400.5 | 4883 | - | +| SmolVLA | `--flash-attn` | 1630.3 | 1661.5 | 1658.8 | 1681.5 | 982.4 | 1048 | -10% | +| VLA-Adapter | *(defaults)* | 1659.3 | 1689.6 | 1684.4 | 1729.9 | 980.8 | 2706 | - | +| GR00T N1.6 | *(defaults)* | 1688.9 | 1721.9 | 1717.9 | 1745.8 | 496.1 | 4524 | - | +| GR00T N1.5 | *(defaults)* | 1935.4 | 1966.1 | 1968.8 | 1983.0 | 494.0 | 3499 | - | +| Evo-1 | `--flash-attn --mm-prec default` | 3906.7 | 3952.3 | 3951.2 | 3981.4 | 1711.6 | 1366 | -5% | +| π0.5 | *(defaults)* | 5735.0 | 5913.9 | 5799.2 | 6092.9 | 1001.6 | 5657 | - | +| π0 | *(defaults)* | 6182.1 | 6342.8 | 6250.5 | 6507.3 | 1035.7 | 5353 | - | +| OpenVLA-OFT | *(defaults)* | 9126.4 | 9636.7 | 9836.8 | 9854.8 | 1008.6 | 14796 | - | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 72.5 | 74.2 | 72.3 | 73.4 | 75.7 | 72.3 | 73.5 | 72.0 | +| TurboVLA | 326.7 | 321.6 | 328.5 | 316.7 | 327.4 | 321.4 | 334.9 | 336.0 | +| VLA-JEPA | 1280.7 | 1289.1 | 1268.9 | 1284.5 | 1297.0 | 1275.7 | 1337.9 | 1349.0 | +| GR00T N1.7 | 1530.0 | 1523.1 | 1528.9 | 1522.2 | 1537.9 | 1530.8 | 1562.2 | 1575.5 | +| VLA-Adapter | 1710.7 | 1697.9 | 1714.1 | 1710.8 | 1705.3 | 1693.9 | 1794.8 | 1796.8 | +| GR00T N1.6 | 1736.3 | 1752.0 | 1726.9 | 1764.6 | 1723.6 | 1777.8 | 1725.8 | 1808.8 | +| SmolVLA | 1843.8 | 1696.3 | 1866.4 | 1700.3 | 1872.5 | 1699.9 | 1927.8 | 1742.6 | +| GR00T N1.5 | 1977.4 | 2009.7 | 1993.0 | 2003.7 | 1972.1 | 1983.2 | 2028.2 | 2083.7 | +| Evo-1 | 4195.4 | 3981.5 | 4182.1 | 3964.9 | 4183.0 | 3966.7 | 4254.8 | 4055.9 | +| π0.5 | 6015.6 | 6034.9 | 6020.5 | 6012.2 | 6022.9 | 6038.4 | 6055.9 | 6032.6 | +| π0 | 6393.7 | 6344.9 | 6397.6 | 6335.9 | 6407.8 | 6321.3 | 6460.6 | 6513.6 | +| OpenVLA-OFT | 9592.9 | 9575.3 | 9568.9 | 9583.8 | 9589.3 | 9713.6 | 10282.6 | 10295.8 | + +
+ +## Not run + +- **BitVLA**: N/A off CUDA. The only published GGUF is int2-packed, and only CUDA builds load it: + + ```text + vla(bitvla): int2-packed GGUF requires a CUDA build (VLA_BITVLA_CUDA_KERNELS); use the bf16 GGUF for CPU. + ``` + +## Reproducing + +```bash +cmake -S . -B build-cpu -DCMAKE_BUILD_TYPE=Release +cmake --build build-cpu -j + +# one row, for example SmolVLA +./build-cpu/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/core-ultra-x7-358h.md b/docs/benchmark/core-ultra-x7-358h.md new file mode 100644 index 0000000..1365a01 --- /dev/null +++ b/docs/benchmark/core-ultra-x7-358h.md @@ -0,0 +1,391 @@ +# Intel Core Ultra X7 358H: Arc B390 iGPU, AI Boost NPU and CPU + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +Four ways to run on this chip, from two builds. The OpenVINO build selects its device with +`GGML_OPENVINO_DEVICE`: `GPU` for the Arc B390 iGPU, `NPU` for the AI Boost NPU and `CPU` for +OpenVINO's CPU plugin. Every OpenVINO run was checked against ggml's `OpenVINO: using device` line, +so a silent fallback to another device would have been counted as a failure. A build without +OpenVINO gives ggml's own CPU backend. + +## Test setup + +| | | +|---|---| +| Device | Intel Core Ultra X7 358H (Panther Lake): 16 cores, 16 threads; Arc B390 iGPU; AI Boost NPU; 62 GiB of shared RAM | +| Threads | 16 (the default: all hardware threads, capped at 16), for both CPU sections | +| Power | platform profile `balanced`; PL1 25 W over a 27 s window, PL2 65 W; `intel_pstate` `powersave` governor | +| OS | Ubuntu 24.04.4 LTS, kernel 7.0 | +| Toolkit | OpenVINO 2026.4.0, the version `scripts/install_ov.sh` pins, unpacked into a private directory; the system's 2026.2.1 was not used. GPU compute runtime 26.22.38646.4, NPU driver 1.33.0, Level Zero loader 1.16.1, GCC 13.3 | +| Commit | `450992c` (PR #32 head, branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 09:49 to 16:20 (UTC+07) | +| Build | OpenVINO: `-DGGML_OPENVINO=ON -DCMAKE_BUILD_TYPE=Release` (`GGML_NATIVE=ON`), after sourcing OpenVINO's `setupvars.sh`. ggml CPU backend: `-DCMAKE_BUILD_TYPE=Release` (`GGML_NATIVE=ON`) | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + +Memory is shared with the CPU. **Peak device buffers** is the process's GPU or NPU buffer memory from the DRM `fdinfo` counters, sampled every 0.5 s: `drm-total-gtt`, `-system` and `-stolen` for the xe GPU driver, and `drm-total-memory` for the intel_vpu NPU driver. **Peak RSS** is the process's peak resident set (`getrusage`). The two are separate counters, so adding them can double-count a buffer the CPU also maps. The two CPU sections report peak RSS only. + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2`) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +### How the fastest configuration is picked + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +## Latency and memory: Arc B390 iGPU (OpenVINO GPU plugin) + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak device buffers MiB | Peak RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| TurboVLA | 2 | 256 | 35.6 | 36.3 | 36.3 | 36.9 | - | 1080 | 1537 | +| GR00T N1.5 | 1 | 224 | 145.5 | 146.6 | 146.6 | 147.5 | 43.0 | 10556 | 6726 | +| VLA-JEPA | 1 | 256 | 152.4 | 154.5 | 154.6 | 155.5 | 28.9 | 9387 | 5321 | +| GR00T N1.6 | 1 | 224 | 225.1 | 231.1 | 230.9 | 238.3 | 20.4 | 14630 | 10159 | +| GR00T N1.7 | 1 | 256 | 230.8 | 232.2 | 231.9 | 233.8 | 29.9 | 14931 | 11043 | +| VLA-Adapter | 1 | 224 | 232.6 | 258.5 | 237.3 | 303.2 | 122.3 | 7424 | 3747 | +| SmolVLA | 2 | 512 | 499.8 | 505.4 | 505.4 | 511.7 | 115.7 | 4769 | 3614 | +| π0.5 | 2 | 224 | 581.5 | 586.4 | 586.0 | 589.3 | 86.8 | 15754 | 11428 | +| π0 | 2 | 224 | 795.8 | 802.8 | 803.0 | 805.1 | 87.2 | 25041 | 16994 | +| Evo-1 | 1 | 448 | 841.5 | 849.9 | 850.4 | 853.6 | 232.6 | 4965 | 3075 | +| OpenVLA-OFT | 1 | 224 | 1494.4 | 1588.3 | 1554.1 | 1689.6 | 169.2 | 41486 | 15266 | + +### Fastest configuration: Arc B390 iGPU (OpenVINO GPU plugin) + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak device buffers MiB | Peak RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:|--:| +| TurboVLA | `--weight-dtype bf16 --flash-attn` | 31.9 | 33.0 | 32.6 | 33.4 | - | 1330 | 1216 | -9% | +| GR00T N1.5 | *(defaults)* | 145.5 | 146.6 | 146.6 | 147.5 | 43.0 | 10556 | 6726 | - | +| VLA-JEPA | *(defaults)* | 152.4 | 154.5 | 154.6 | 155.5 | 28.9 | 9387 | 5321 | - | +| GR00T N1.6 | *(defaults)* | 225.1 | 231.1 | 230.9 | 238.3 | 20.4 | 14630 | 10159 | - | +| GR00T N1.7 | *(defaults)* | 230.8 | 232.2 | 231.9 | 233.8 | 29.9 | 14931 | 11043 | - | +| VLA-Adapter | *(defaults)* | 232.6 | 258.5 | 237.3 | 303.2 | 122.3 | 7424 | 3747 | - | +| SmolVLA | `--weight-dtype f16 --flash-attn` | 455.2 | 459.9 | 460.4 | 461.7 | 68.8 | 2393 | 1851 | -9% | +| π0.5 | *(defaults)* | 581.5 | 586.4 | 586.0 | 589.3 | 86.8 | 15754 | 11428 | - | +| Evo-1 | `--flash-attn --mm-prec default` | 752.1 | 757.7 | 757.3 | 760.3 | 139.8 | 4854 | 3039 | -11% | +| π0 | `--flash-attn --mm-prec default` | 759.8 | 766.3 | 767.8 | 769.7 | 85.0 | 25043 | 16969 | -5% | +| OpenVLA-OFT | *(defaults)* | 1494.4 | 1588.3 | 1554.1 | 1689.6 | 169.2 | 41486 | 15266 | - | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| TurboVLA | 38.2 | 33.2 | 37.1 | 33.1 | 36.1 | 32.5 | 36.8 | 33.2 | +| GR00T N1.5 | 146.8 | 152.0 | 147.1 | 151.4 | 146.4 | 150.2 | 146.4 | 150.9 | +| VLA-JEPA | 155.3 | 152.1 | 154.5 | 152.0 | 154.7 | 151.2 | 154.4 | 151.1 | +| GR00T N1.6 | 228.6 | 225.0 | 230.4 | 224.6 | 237.6 | 224.9 | 230.1 | 224.8 | +| GR00T N1.7 | 232.5 | 230.1 | 232.2 | 229.8 | 232.5 | 228.9 | 232.8 | 229.4 | +| VLA-Adapter | 267.2 | 290.4 | 271.2 | 271.9 | 253.1 | 251.8 | 265.9 | 287.8 | +| SmolVLA | 506.3 | 462.0 | 502.3 | 459.1 | 507.7 | 457.1 | 504.9 | 456.9 | +| π0.5 | 586.2 | 584.8 | 587.7 | 586.1 | 589.0 | 586.0 | 585.4 | 585.3 | +| π0 | 804.2 | 769.5 | 801.1 | 765.6 | 803.0 | 768.6 | 800.1 | 769.6 | +| Evo-1 | 851.5 | 762.2 | 850.2 | 758.7 | 848.0 | 760.9 | 846.7 | 758.7 | +| OpenVLA-OFT | 1476.0 | 1480.6 | 1570.6 | 1471.4 | 1550.8 | 2222.0 | 1423.7 | 2479.9 | + +
+ +## Latency and memory: AI Boost NPU (OpenVINO NPU plugin) + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak device buffers MiB | Peak RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| TurboVLA | 2 | 256 | 63.8 | 67.3 | 65.9 | 72.7 | - | 895 | 1314 | +| VLA-Adapter | 1 | 224 | 316.0 | 426.4 | 433.3 | 437.1 | 151.8 | 2762 | 5038 | +| OpenVLA-OFT | 1 | 224 | 468.7 | 472.3 | 472.4 | 474.1 | 119.6 | 14521 | 29642 | +| π0 | 2 | 224 | 836.6 | 851.3 | 849.0 | 855.9 | 131.9 | 5756 | 11602 | +| SmolVLA | 2 | 512 | 1656.5 | 1678.9 | 1679.6 | 1702.1 | 1220.5 | 1486 | 2894 | + +### Fastest configuration: AI Boost NPU (OpenVINO NPU plugin) + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak device buffers MiB | Peak RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:|--:| +| TurboVLA | `--weight-dtype f16` | 59.9 | 63.6 | 62.4 | 69.0 | - | 523 | 877 | -5% | +| VLA-Adapter | *(defaults)* | 316.0 | 426.4 | 433.3 | 437.1 | 151.8 | 2762 | 5038 | - | +| OpenVLA-OFT | *(defaults)* | 468.7 | 472.3 | 472.4 | 474.1 | 119.6 | 14521 | 29642 | - | +| π0 | *(defaults)* | 836.6 | 851.3 | 849.0 | 855.9 | 131.9 | 5756 | 11602 | - | +| SmolVLA | *(defaults)* | 1656.5 | 1678.9 | 1679.6 | 1702.1 | 1220.5 | 1486 | 2894 | - | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| TurboVLA | 72.5 | 75.1 | 73.2 | 72.6 | 70.4 | 71.0 | 70.2 | 70.2 | +| VLA-Adapter | 425.2 | 424.1 | 432.3 | 430.7 | 431.8 | 429.1 | 436.7 | 421.5 | +| OpenVLA-OFT | 483.6 | 480.6 | 477.0 | 483.9 | 469.8 | 473.1 | 475.7 | 475.2 | +| π0 | 883.7 | 823.1 | 881.1 | 832.0 | 857.8 | 824.9 | 872.1 | 841.9 | +| SmolVLA | 1697.6 | 1703.9 | 1665.3 | 1659.4 | 1696.4 | 1679.3 | 1701.0 | 1688.8 | + +
+ +## Latency and memory: OpenVINO CPU plugin + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| TurboVLA | 2 | 256 | 240.7 | 242.3 | 241.9 | 243.6 | - | 1825 | +| VLA-JEPA | 1 | 256 | 1314.3 | 1420.7 | 1431.6 | 1445.5 | 475.5 | 11702 | +| VLA-Adapter | 1 | 224 | 1612.5 | 1623.4 | 1620.8 | 1636.0 | 905.3 | 7868 | +| SmolVLA | 2 | 512 | 1942.8 | 1950.2 | 1950.4 | 1955.3 | 1053.3 | 4545 | +| GR00T N1.7 | 1 | 256 | 2034.7 | 2048.9 | 2038.3 | 2053.4 | 460.9 | 14936 | +| GR00T N1.5 | 1 | 224 | 2221.0 | 2405.0 | 2406.8 | 2432.8 | 560.2 | 10627 | +| GR00T N1.6 | 1 | 224 | 2243.0 | 2269.8 | 2250.7 | 2342.5 | 523.1 | 13842 | +| Evo-1 | 1 | 448 | 4496.6 | 4681.7 | 4756.0 | 4776.2 | 1910.9 | 4748 | +| π0 | 2 | 224 | 6924.5 | 7436.4 | 7492.7 | 7516.8 | 1087.9 | 15819 | +| π0.5 | 2 | 224 | 7025.7 | 7528.9 | 7584.9 | 7616.4 | 1094.3 | 16522 | +| OpenVLA-OFT | 1 | 224 | 11604.9 | 11655.3 | 11647.8 | 11700.6 | 1040.6 | 43047 | + +### Fastest configuration: OpenVINO CPU plugin + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:| +| TurboVLA | *(defaults)* | 240.7 | 242.3 | 241.9 | 243.6 | - | 1825 | - | +| VLA-JEPA | *(defaults)* | 1314.3 | 1420.7 | 1431.6 | 1445.5 | 475.5 | 11702 | - | +| VLA-Adapter | *(defaults)* | 1612.5 | 1623.4 | 1620.8 | 1636.0 | 905.3 | 7868 | - | +| SmolVLA | *(defaults)* | 1942.8 | 1950.2 | 1950.4 | 1955.3 | 1053.3 | 4545 | - | +| GR00T N1.7 | *(defaults)* | 2034.7 | 2048.9 | 2038.3 | 2053.4 | 460.9 | 14936 | - | +| GR00T N1.5 | *(defaults)* | 2221.0 | 2405.0 | 2406.8 | 2432.8 | 560.2 | 10627 | - | +| GR00T N1.6 | *(defaults)* | 2243.0 | 2269.8 | 2250.7 | 2342.5 | 523.1 | 13842 | - | +| Evo-1 | *(defaults)* | 4496.6 | 4681.7 | 4756.0 | 4776.2 | 1910.9 | 4748 | - | +| π0 | *(defaults)* | 6924.5 | 7436.4 | 7492.7 | 7516.8 | 1087.9 | 15819 | - | +| π0.5 | *(defaults)* | 7025.7 | 7528.9 | 7584.9 | 7616.4 | 1094.3 | 16522 | - | +| OpenVLA-OFT | *(defaults)* | 11604.9 | 11655.3 | 11647.8 | 11700.6 | 1040.6 | 43047 | - | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| TurboVLA | 243.0 | 243.6 | 243.4 | 242.8 | 248.0 | 248.6 | 247.2 | 249.0 | +| VLA-JEPA | 1564.5 | 1515.8 | 1576.7 | 1565.6 | 1532.9 | 1547.8 | 1538.7 | 1556.6 | +| VLA-Adapter | 1851.5 | 1834.8 | 1825.1 | 1811.8 | 1852.4 | 1844.8 | 1793.9 | 1793.2 | +| SmolVLA | 1949.0 | 1996.8 | 1948.2 | 1990.5 | 1950.6 | 1994.9 | 1953.6 | 1997.5 | +| GR00T N1.7 | 2271.9 | 2040.0 | 2266.9 | 2261.1 | 2039.4 | 2222.5 | 2039.1 | 2034.5 | +| GR00T N1.5 | 2421.0 | 2425.6 | 2548.2 | 2521.8 | 2433.5 | 2531.9 | 2422.0 | 2427.9 | +| GR00T N1.6 | 2430.1 | 2250.6 | 2254.7 | 2398.4 | 2249.9 | 2461.8 | 2253.9 | 2253.4 | +| Evo-1 | 4906.3 | 4805.6 | 4889.3 | 4901.2 | 4842.8 | 4849.3 | 4766.9 | 4795.5 | +| π0 | 7567.3 | 7550.0 | 7528.8 | 7536.5 | 7541.7 | 7550.6 | 7569.9 | 7571.0 | +| π0.5 | 7666.3 | 7647.3 | 7613.1 | 7603.7 | 7620.1 | 7627.3 | 7599.9 | 7595.8 | +| OpenVLA-OFT | 11732.4 | 11706.9 | 11726.0 | 11705.5 | 11676.9 | 11744.3 | 11681.2 | 11690.1 | + +
+ +## Latency and memory: ggml CPU backend + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 79.9 | 130.0 | 138.4 | 142.2 | 16.6 | 623 | +| TurboVLA | 2 | 256 | 160.0 | 163.0 | 162.5 | 164.4 | - | 855 | +| VLA-JEPA | 1 | 256 | 1071.8 | 1306.5 | 1321.2 | 1332.1 | 399.2 | 3824 | +| GR00T N1.7 | 1 | 256 | 1267.9 | 1529.9 | 1540.3 | 1552.9 | 420.0 | 4884 | +| VLA-Adapter | 1 | 224 | 1666.3 | 1672.9 | 1672.2 | 1677.6 | 974.7 | 2708 | +| GR00T N1.6 | 1 | 224 | 1671.7 | 1713.2 | 1712.3 | 1727.0 | 519.7 | 4524 | +| SmolVLA | 2 | 512 | 1788.5 | 1920.7 | 1959.3 | 1975.0 | 901.7 | 1101 | +| GR00T N1.5 | 1 | 224 | 1981.2 | 1991.4 | 1990.3 | 1999.1 | 523.7 | 3500 | +| Evo-1 | 1 | 448 | 4104.1 | 4122.2 | 4120.4 | 4133.9 | 2053.6 | 1432 | +| π0.5 | 2 | 224 | 6110.4 | 6130.1 | 6130.5 | 6141.5 | 1048.2 | 5657 | +| π0 | 2 | 224 | 6153.5 | 6170.8 | 6169.4 | 6182.0 | 1053.9 | 5356 | +| OpenVLA-OFT | 1 | 224 | 9973.4 | 10011.6 | 10014.2 | 10044.8 | 1025.6 | 14797 | + +### Fastest configuration: ggml CPU backend + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | `--weight-dtype bf16 --flash-attn` | 108.0 | 111.3 | 111.5 | 114.3 | 16.1 | 623 | -14% | +| TurboVLA | *(defaults)* | 160.0 | 163.0 | 162.5 | 164.4 | - | 855 | - | +| VLA-JEPA | `--weight-dtype f16` | 881.3 | 1041.8 | 1074.4 | 1082.8 | 331.3 | 3824 | -20% | +| GR00T N1.7 | `--weight-dtype f16 --flash-attn` | 981.2 | 1238.8 | 1253.5 | 1304.5 | 273.3 | 4884 | -19% | +| GR00T N1.6 | `--weight-dtype f16 --flash-attn` | 1051.0 | 1384.6 | 1410.5 | 1438.1 | 357.2 | 4524 | -19% | +| VLA-Adapter | `--weight-dtype f16` | 1374.1 | 1380.0 | 1379.1 | 1387.2 | 808.9 | 2707 | -18% | +| SmolVLA | `--weight-dtype f16 --flash-attn` | 1501.1 | 1549.7 | 1550.8 | 1570.7 | 693.3 | 1051 | -19% | +| GR00T N1.5 | `--weight-dtype f16 --flash-attn` | 1555.9 | 1576.6 | 1578.0 | 1588.8 | 384.0 | 3498 | -21% | +| Evo-1 | `--weight-dtype f16 --flash-attn` | 3263.7 | 3277.0 | 3277.9 | 3285.5 | 1499.0 | 1368 | -21% | +| π0.5 | `--weight-dtype f16 --flash-attn` | 4903.4 | 4917.5 | 4917.2 | 4927.7 | 850.3 | 5657 | -20% | +| π0 | `--weight-dtype f16` | 4933.4 | 4951.2 | 4949.7 | 4966.4 | 860.6 | 5355 | -20% | +| OpenVLA-OFT | `--weight-dtype f16 --flash-attn` | 7925.4 | 7947.1 | 7947.8 | 7957.9 | 852.2 | 14797 | -21% | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 120.5 | 119.5 | 114.5 | 118.5 | 111.9 | 110.8 | 116.2 | 119.3 | +| TurboVLA | 201.4 | 190.7 | 201.9 | 189.3 | 300.4 | 287.6 | 245.8 | 231.4 | +| VLA-JEPA | 1206.8 | 1184.3 | 1574.5 | 1501.2 | 1494.5 | 1104.9 | 919.0 | 1139.7 | +| GR00T N1.7 | 1603.7 | 1641.2 | 1432.1 | 1568.9 | 1681.9 | 1389.4 | 1122.9 | 1089.8 | +| VLA-Adapter | 1720.5 | 1693.8 | 1695.9 | 1682.8 | 1697.5 | 1752.6 | 1533.7 | 1542.7 | +| GR00T N1.6 | 1652.7 | 1830.0 | 1736.6 | 1620.1 | 1871.4 | 1619.2 | 1287.4 | 1281.7 | +| SmolVLA | 1980.5 | 1861.1 | 1991.3 | 1873.8 | 1986.9 | 1872.2 | 1673.5 | 1566.2 | +| GR00T N1.5 | 1996.0 | 2135.1 | 1994.7 | 2136.8 | 1997.5 | 2137.7 | 1722.8 | 1702.3 | +| Evo-1 | 4136.1 | 3867.8 | 4140.3 | 3866.8 | 4139.2 | 3864.0 | 3569.0 | 3283.7 | +| π0.5 | 6164.1 | 6162.5 | 6164.4 | 6171.1 | 6160.7 | 6159.4 | 4920.4 | 4918.6 | +| π0 | 6193.3 | 6401.2 | 6179.6 | 6409.5 | 6174.2 | 6395.9 | 4954.6 | 5175.8 | +| OpenVLA-OFT | 10040.8 | 10041.5 | 10033.6 | 10025.2 | 10001.9 | 9994.9 | 7960.8 | 7958.7 | + +
+ +## Not run + +### Arc B390 iGPU (OpenVINO GPU plugin) + +- **BitVLA**: N/A off CUDA. The only published GGUF is int2-packed, and only CUDA builds load it: + + ```text + vla(bitvla): int2-packed GGUF requires a CUDA build (VLA_BITVLA_CUDA_KERNELS); use the bf16 GGUF for CPU. + ``` +- **Octo-Small**: not supported on OpenVINO (the README's support matrix lists it as planned). The translator has no conversion for one of its ops: + + ```text + GGML OpenVINO backend ov::Exception: Check 'it != m_translator_map.end()' failed at openvino/translate_session.cpp:304 + ``` + +### AI Boost NPU (OpenVINO NPU plugin) + +- **π0.5**: the NPU compiler rejects the graph's dynamic shape: + + ```text + Compilation failed. vclAllocatedExecutableCreate4 result: 0x78000004 - [NPU_VCL] Compiler returned msg: + IE.Reshape doesn't support dynamic shapes + vla(pi05): ggml_backend_graph_compute failed (-1) + ``` +- **GR00T N1.5**: the NPU compiler fails without giving a reason: + + ```text + Compilation failed. vclAllocatedExecutableCreate4 result: 0x78000004 - [NPU_VCL] Compiler returned msg: + Compilation failed + vla(gr00tn1d5): graph compute failed (-1) + ``` +- **GR00T N1.6**: the NPU compiler aborts on a dynamic dimension (exit code 250): + + ```text + LLVM ERROR: Failed to infer result type(s): + "IE.Add"(...) {} : (tensor<1x1x51x1536xf32>, tensor<1x1x?x1536xf32, {bounds = ... [1, 1, 132, 1536] ...}>) -> ( ??? ) + ``` +- **GR00T N1.7**: the NPU compiler fails without giving a reason: + + ```text + Compilation failed. vclAllocatedExecutableCreate4 result: 0x78000004 - [NPU_VCL] Compiler returned msg: + Compilation failed + vla(gr00tn1d7): graph compute failed (-1) + ``` +- **VLA-JEPA**: the NPU compiler aborts on a dynamic dimension (exit code 250): + + ```text + LLVM ERROR: Failed to infer result type(s): + "IE.Add"(...) {} : (tensor<1x1x40x768xf32>, tensor<1x1x?x768xf32, {bounds = ... [1, 1, 68, 768] ...}>) -> ( ??? ) + ``` +- **Evo-1**: the NPU compiler rejects the graph's dynamic shape: + + ```text + Compilation failed. vclAllocatedExecutableCreate4 result: 0x78000004 - [NPU_VCL] Compiler returned msg: + IE.Reshape doesn't support dynamic shapes + vla(evo1): ggml_backend_graph_compute failed (-1) + ``` +- **BitVLA**: N/A off CUDA. The only published GGUF is int2-packed, and only CUDA builds load it: + + ```text + vla(bitvla): int2-packed GGUF requires a CUDA build (VLA_BITVLA_CUDA_KERNELS); use the bf16 GGUF for CPU. + ``` +- **Octo-Small**: not supported on OpenVINO (the README's support matrix lists it as planned). The translator has no conversion for one of its ops: + + ```text + GGML OpenVINO backend ov::Exception: Check 'it != m_translator_map.end()' failed at openvino/translate_session.cpp:304 + ``` + +### OpenVINO CPU plugin + +- **BitVLA**: N/A off CUDA. The only published GGUF is int2-packed, and only CUDA builds load it: + + ```text + vla(bitvla): int2-packed GGUF requires a CUDA build (VLA_BITVLA_CUDA_KERNELS); use the bf16 GGUF for CPU. + ``` +- **Octo-Small**: not supported on OpenVINO (the README's support matrix lists it as planned). The translator has no conversion for one of its ops: + + ```text + GGML OpenVINO backend ov::Exception: Check 'it != m_translator_map.end()' failed at openvino/translate_session.cpp:304 + ``` + +### ggml CPU backend + +- **BitVLA**: N/A off CUDA. The only published GGUF is int2-packed, and only CUDA builds load it: + + ```text + vla(bitvla): int2-packed GGUF requires a CUDA build (VLA_BITVLA_CUDA_KERNELS); use the bf16 GGUF for CPU. + ``` + +## Notes + +OpenVINO compiles each graph on the first predict, which can take up to a minute on the GPU. The warmup calls absorb it, so none of the latencies above include it. Do not set `GGML_OPENVINO_CACHE_DIR`; `docs/backend/ov.md` explains why. + +π0 runs at F32 inference precision on the GPU by default (`GGML_OPENVINO_GPU_PRECISION=f32`, set by vla.cpp for π0 alone), which costs about 3x. + +A latency is not a correctness check. `docs/backend/ov.md` lists π0 as wrong on the NPU (max|Δ| 1.7), measured at an earlier llama.cpp pin. Its NPU output was not re-verified here. + +The chip runs under a 25 W long-term power limit. The first timed calls of a process can run at boost power, so on the CPU sections read `mean` or `p50` for sustained latency. + +## Reproducing + +```bash +# OpenVINO build: iGPU, NPU and OpenVINO CPU plugin +source /setupvars.sh +cmake -S . -B build-ov -G Ninja -DGGML_OPENVINO=ON -DCMAKE_BUILD_TYPE=Release +cmake --build build-ov -j + +# one row, for example SmolVLA +GGML_OPENVINO_DEVICE=GPU ./build-ov/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +ZE_ENABLE_ALT_DRIVERS=/lib/x86_64-linux-gnu/libze_intel_npu.so.1 GGML_OPENVINO_DEVICE=NPU \ + ./build-ov/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +GGML_OPENVINO_DEVICE=CPU ./build-ov/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 + +# ggml CPU backend +cmake -S . -B build-cpu -DCMAKE_BUILD_TYPE=Release && cmake --build build-cpu -j +./build-cpu/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/jetson-agx-orin.md b/docs/benchmark/jetson-agx-orin.md new file mode 100644 index 0000000..a0662ba --- /dev/null +++ b/docs/benchmark/jetson-agx-orin.md @@ -0,0 +1,138 @@ +# Jetson AGX Orin 64GB (CUDA) + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +## Test setup + +| | | +|---|---| +| Device | NVIDIA Jetson AGX Orin Developer Kit, 2048-core Ampere GPU (sm_87), 64 GB LPDDR5 unified | +| Host CPU | 12-core Arm Cortex-A78AE | +| Power mode | `MAXN` (nvpmodel 0); `jetson_clocks` state not changed | +| OS | JetPack 6.2 (L4T R36.4.3), Ubuntu 22.04.5 | +| Toolkit | CUDA 12.6, GCC 11.4 | +| Commit | `c93ca0a` (PR #32 head, branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 01:09 to 04:05 (UTC+07) | +| Build | `-DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=87 -DCMAKE_BUILD_TYPE=Release` (`GGML_NATIVE=ON`) | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + +Memory is unified, and `nvidia-smi` reports no per-process usage on Jetson. The CUDA buffers are mapped into the process, so **Peak RSS** (`getrusage`) counts them together with host memory, and it is the process's whole footprint. VLA-JEPA, for example, peaks at 4358 MiB here. On the RTX 3090 the same model takes 4138 MiB of VRAM plus 634 MiB of host RSS. + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2`) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +## Latency and memory + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 17.5 | 23.9 | 22.5 | 30.3 | 3.1 | 947 | +| TurboVLA | 2 | 256 | 36.0 | 36.8 | 36.8 | 36.9 | - | 1238 | +| VLA-JEPA | 1 | 256 | 94.6 | 96.3 | 95.1 | 98.7 | 38.2 | 4358 | +| VLA-Adapter | 1 | 224 | 127.4 | 128.2 | 128.2 | 128.5 | 75.3 | 3698 | +| GR00T N1.5 | 1 | 224 | 131.2 | 132.3 | 131.9 | 132.7 | 38.4 | 4079 | +| GR00T N1.6 | 1 | 224 | 132.6 | 133.1 | 133.0 | 133.4 | 38.0 | 5082 | +| GR00T N1.7 | 1 | 256 | 132.7 | 134.0 | 133.7 | 135.2 | 37.6 | 5442 | +| BitVLA | 1 | 224 | 133.0 | 134.7 | 133.4 | 134.9 | 27.4 | 2219 | +| SmolVLA | 2 | 512 | 218.3 | 223.4 | 223.4 | 225.9 | 130.6 | 2527 | +| π0 | 2 | 224 | 350.3 | 350.9 | 350.6 | 351.4 | 75.5 | 5982 | +| π0.5 | 2 | 224 | 351.9 | 353.1 | 353.1 | 353.7 | 75.0 | 6008 | +| OpenVLA-OFT | 1 | 224 | 384.9 | 385.3 | 385.3 | 385.4 | 75.7 | 15391 | +| Evo-1 | 1 | 448 | 434.6 | 435.2 | 435.3 | 435.6 | 208.8 | 2075 | + +### Fastest configuration + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`; +- `--act-dtype bf16 --flash-attn`, for π0 and Evo-1 on CUDA only. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | *(defaults)* | 17.5 | 23.9 | 22.5 | 30.3 | 3.1 | 947 | - | +| TurboVLA | `--weight-dtype bf16 --flash-attn` | 24.1 | 26.6 | 24.5 | 32.3 | - | 932 | -28% | +| VLA-JEPA | `--weight-dtype f16 --flash-attn` | 82.4 | 85.6 | 83.6 | 93.8 | 32.9 | 4346 | -11% | +| VLA-Adapter | `--weight-dtype f16 --flash-attn` | 116.5 | 117.8 | 117.7 | 118.7 | 64.0 | 3699 | -8% | +| GR00T N1.5 | `--weight-dtype f16 --flash-attn` | 123.0 | 124.0 | 123.3 | 125.7 | 35.9 | 4007 | -6% | +| GR00T N1.7 | `--weight-dtype f16 --flash-attn` | 125.1 | 125.9 | 125.4 | 126.8 | 31.5 | 5441 | -6% | +| GR00T N1.6 | `--weight-dtype f16` | 127.6 | 128.7 | 128.6 | 129.2 | 34.7 | 5077 | -3% | +| BitVLA | *(defaults)* | 133.0 | 134.7 | 133.4 | 134.9 | 27.4 | 2219 | - | +| SmolVLA | `--flash-attn --mm-prec default` | 146.4 | 147.0 | 146.9 | 147.3 | 78.1 | 1711 | -34% | +| π0 | `--act-dtype bf16 --flash-attn` | 300.4 | 301.2 | 301.2 | 301.5 | 67.7 | 6017 | -14% | +| Evo-1 | `--act-dtype bf16 --flash-attn` | 311.1 | 311.4 | 311.5 | 311.6 | 92.0 | 2154 | -28% | +| OpenVLA-OFT | `--weight-dtype f16` | 345.2 | 345.9 | 345.8 | 346.3 | 64.8 | 15392 | -10% | +| π0.5 | *(defaults)* | 351.9 | 353.1 | 353.1 | 353.7 | 75.0 | 6008 | - | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | `--act-dtype bf16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 33.4 | 32.4 | 31.9 | 33.5 | 33.4 | 31.3 | 32.0 | 30.9 | | +| TurboVLA | 40.1 | 38.3 | 39.5 | 36.7 | 35.1 | 34.8 | 51.2 | 35.4 | | +| VLA-JEPA | 117.6 | 113.2 | 115.6 | 115.3 | 117.2 | 115.7 | 108.8 | 102.6 | | +| VLA-Adapter | 130.7 | 133.4 | 128.7 | 128.8 | 128.6 | 129.3 | 119.5 | 118.0 | | +| GR00T N1.5 | 144.6 | 130.8 | 143.6 | 130.2 | 134.4 | 130.2 | 129.2 | 126.9 | | +| GR00T N1.6 | 134.8 | 136.4 | 135.9 | 135.2 | 133.4 | 135.4 | 128.6 | 130.9 | | +| GR00T N1.7 | 146.7 | 133.4 | 135.5 | 143.8 | 144.3 | 143.9 | 136.1 | 128.2 | | +| BitVLA | 144.4 | 144.5 | 144.6 | 144.9 | 145.5 | 144.5 | 145.4 | 144.8 | | +| SmolVLA | 222.9 | 168.9 | 203.8 | 146.2 | 249.9 | 196.7 | 250.4 | 200.2 | | +| π0 | 350.8 | 324.5 | 350.7 | 323.9 | 350.8 | 323.7 | 355.9 | 329.2 | 300.0 | +| π0.5 | 353.4 | 352.3 | 354.2 | 353.0 | 353.0 | 353.2 | 358.3 | 357.6 | | +| OpenVLA-OFT | 384.4 | 383.8 | 385.3 | 383.7 | 384.8 | 385.7 | 346.2 | 346.7 | | +| Evo-1 | 434.0 | 335.5 | 434.5 | 335.3 | 434.7 | 335.2 | 441.9 | 343.5 | 310.9 | + +
+ +## Reproducing + +```bash +cmake -S . -B build-cuda -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=87 -DCMAKE_BUILD_TYPE=Release +cmake --build build-cuda -j + +# one row, for example SmolVLA +./build-cuda/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/jetson-orin-nano.md b/docs/benchmark/jetson-orin-nano.md new file mode 100644 index 0000000..c1c975f --- /dev/null +++ b/docs/benchmark/jetson-orin-nano.md @@ -0,0 +1,139 @@ +# Jetson Orin Nano Super (CUDA) + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +## Test setup + +| | | +|---|---| +| Device | NVIDIA Jetson Orin Nano Developer Kit (Super), 1024-core Ampere GPU (sm_87), 8 GB LPDDR5 unified (7.4 GiB visible) | +| Host CPU | 6-core Arm Cortex-A78AE | +| Power mode | `MAXN_SUPER` (nvpmodel 2); `jetson_clocks` state not changed | +| OS | JetPack 6.2.2 (L4T R36.5), Ubuntu 22.04.5 | +| Toolkit | CUDA 12.6, GCC 11.4 | +| Commit | `c93ca0a` (PR #32 head, branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 01:09 to 02:45 (UTC+07) | +| Build | `-DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=87 -DCMAKE_BUILD_TYPE=Release` (`GGML_NATIVE=ON`) | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + +Memory is unified, and `nvidia-smi` reports no per-process usage on Jetson. The CUDA buffers are mapped into the process, so **Peak RSS** (`getrusage`) counts them together with host memory, and it is the process's whole footprint. VLA-JEPA, for example, peaks at 4332 MiB here. On the RTX 3090 the same model takes 4138 MiB of VRAM plus 634 MiB of host RSS. + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2`) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +## Latency and memory + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 31.9 | 34.4 | 33.1 | 36.3 | 4.1 | 923 | +| TurboVLA | 2 | 256 | 89.3 | 90.2 | 89.6 | 89.9 | - | 1215 | +| VLA-JEPA | 1 | 256 | 193.9 | 194.6 | 194.3 | 196.0 | 82.4 | 4332 | +| GR00T N1.7 | 1 | 256 | 271.7 | 272.1 | 272.0 | 272.2 | 82.3 | 5413 | +| GR00T N1.6 | 1 | 224 | 272.3 | 272.6 | 272.5 | 272.9 | 85.0 | 5051 | +| GR00T N1.5 | 1 | 224 | 293.4 | 295.0 | 295.2 | 296.0 | 84.8 | 4048 | +| VLA-Adapter | 1 | 224 | 297.5 | 297.9 | 297.9 | 298.1 | 167.4 | 3669 | +| BitVLA | 1 | 224 | 335.0 | 335.5 | 335.3 | 336.0 | 63.6 | 2193 | +| SmolVLA | 2 | 512 | 462.8 | 464.2 | 463.9 | 466.1 | 298.6 | 2488 | +| π0 | 2 | 224 | 845.2 | 846.1 | 846.1 | 846.7 | 169.2 | 5961 | +| π0.5 | 2 | 224 | 849.7 | 850.2 | 850.2 | 850.5 | 169.3 | 5982 | +| Evo-1 | 1 | 448 | 1011.5 | 1012.5 | 1012.6 | 1013.0 | 523.3 | 2048 | + +### Fastest configuration + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`; +- `--act-dtype bf16 --flash-attn`, for π0 and Evo-1 on CUDA only. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | *(defaults)* | 31.9 | 34.4 | 33.1 | 36.3 | 4.1 | 923 | - | +| TurboVLA | `--weight-dtype bf16 --flash-attn` | 52.4 | 53.1 | 53.1 | 53.2 | - | 895 | -41% | +| VLA-JEPA | `--weight-dtype f16 --flash-attn` | 173.7 | 174.8 | 174.6 | 176.6 | 70.6 | 4318 | -10% | +| GR00T N1.7 | `--weight-dtype f16 --flash-attn` | 254.5 | 255.0 | 254.9 | 255.6 | 70.4 | 5409 | -6% | +| VLA-Adapter | `--weight-dtype f16` | 266.2 | 266.6 | 266.5 | 266.8 | 148.1 | 3668 | -11% | +| GR00T N1.6 | *(defaults)* | 272.3 | 272.6 | 272.5 | 272.9 | 85.0 | 5051 | - | +| SmolVLA | `--flash-attn --mm-prec default` | 286.7 | 287.8 | 287.3 | 289.9 | 173.1 | 1676 | -38% | +| GR00T N1.5 | *(defaults)* | 293.4 | 295.0 | 295.2 | 296.0 | 84.8 | 4048 | - | +| BitVLA | *(defaults)* | 335.0 | 335.5 | 335.3 | 336.0 | 63.6 | 2193 | - | +| π0 | `--weight-dtype f16 --flash-attn` | 688.2 | 688.7 | 688.8 | 689.0 | 160.0 | 5901 | -19% | +| Evo-1 | `--act-dtype bf16 --flash-attn` | 720.8 | 722.7 | 722.7 | 723.6 | 233.4 | 2133 | -29% | +| π0.5 | `--weight-dtype f16 --flash-attn` | 764.8 | 765.8 | 765.7 | 766.4 | 153.2 | 5925 | -10% | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | `--act-dtype bf16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 43.1 | 45.1 | 41.8 | 41.8 | 41.5 | 43.0 | 43.5 | 48.7 | | +| TurboVLA | 91.2 | 76.3 | 97.3 | 72.8 | 73.5 | 60.3 | 81.0 | 65.9 | | +| VLA-JEPA | 205.7 | 193.1 | 201.0 | 199.0 | 205.6 | 193.4 | 194.8 | 186.1 | | +| GR00T N1.7 | 273.0 | 267.8 | 273.4 | 266.8 | 273.7 | 266.7 | 264.7 | 257.3 | | +| GR00T N1.6 | 275.0 | 279.0 | 274.3 | 279.6 | 273.0 | 278.9 | 271.5 | 270.8 | | +| GR00T N1.5 | 295.4 | 289.6 | 298.3 | 287.0 | 298.2 | 288.6 | 293.5 | 287.9 | | +| VLA-Adapter | 298.7 | 299.0 | 298.9 | 298.5 | 300.0 | 298.4 | 268.1 | 269.2 | | +| BitVLA | 336.1 | 336.3 | 335.9 | 335.9 | 335.7 | 337.0 | 336.0 | 335.6 | | +| SmolVLA | 465.3 | 336.9 | 414.5 | 287.3 | 533.9 | 406.7 | 535.0 | 407.8 | | +| π0 | 848.8 | 772.8 | 849.4 | 776.5 | 848.1 | 778.2 | 763.4 | 687.9 | 719.7 | +| π0.5 | 850.5 | 850.5 | 851.2 | 852.0 | 852.3 | 849.7 | 768.6 | 763.4 | | +| Evo-1 | 1014.7 | 763.0 | 1014.6 | 762.5 | 1014.8 | 762.3 | 1029.4 | 774.8 | 721.8 | + +
+ +## Not run + +- **OpenVLA-OFT**: not attempted. Its 15.1 GB of weights do not fit in 8 GB of memory. + +## Reproducing + +```bash +cmake -S . -B build-cuda -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=87 -DCMAKE_BUILD_TYPE=Release +cmake --build build-cuda -j + +# one row, for example SmolVLA +./build-cuda/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/libero.md b/docs/benchmark/libero.md new file mode 100644 index 0000000..fe2e1a5 --- /dev/null +++ b/docs/benchmark/libero.md @@ -0,0 +1,61 @@ +# LIBERO task success + +Each model below runs closed loop in LIBERO through `vla-server`, driven by +[`eval/run_libero.sh`](../../eval/run_libero.sh): 10 tasks and 20 episodes per +task, so 200 episodes per suite. An episode that hits the step limit counts as +a failure. + +## Test setup + +| | | +|---|---| +| Device | NVIDIA GeForce RTX 3090, 24 GB, CUDA 12.8 (the build in [rtx-3090.md](rtx-3090.md)) | +| Commit | `c93ca0a` (branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 11:28 to 16:53 (UTC+07) | +| Checkpoints | the `vrfai/*` GGUFs listed in [rtx-3090.md](rtx-3090.md#model-configuration); BitVLA and GR00T N1.7 use the per-suite checkpoint | +| Runtime flags | defaults (GR00T in BF16, as shipped); diffusion noise not seeded | +| Simulator | LIBERO with robosuite 1.4.0 and MuJoCo 3.9.0, 500-step limit in every suite | +| Replay | actions executed from each predicted chunk before the next query, the `run_libero.sh` default | + +## LIBERO-Object, all models + +| Model | Replay | Successes | Success rate | +|---|--:|--:|--:| +| BitVLA | 8 | 200/200 | 100.0% | +| TurboVLA | 12 | 200/200 | 100.0% | +| VLA-JEPA | 7 | 200/200 | 100.0% | +| Evo-1 | 14 | 197/200 | 98.5% | +| OpenVLA-OFT | 8 | 197/200 | 98.5% | +| GR00T N1.7 | 16 | 196/200 | 98.0% | +| π0.5 | 10 | 195/200 | 97.5% | +| GR00T N1.5 | 16 | 194/200 | 97.0% | +| VLA-Adapter | 8 | 191/200 | 95.5% | +| SmolVLA | 10 | 179/200 | 89.5% | +| GR00T N1.6 | 16 | 173/200 | 86.5% | +| π0 | 50 | 144/200 | 72.0% | +| Octo-Small | 4 | 16/200 | 8.0% | + +With 200 episodes, a rate near 95% carries a 95% confidence interval of about +±3 points, so gaps of a few points between models are not significant. + +Most replay counts match the model's upstream LIBERO eval. Three do not. +SmolVLA replays 10 of its 50-step chunk, the best setting in its paper's +ablation, where upstream replans every step. π0 replays the whole chunk, as +its LeRobot config does, where openpi's LIBERO eval replans every 5. GR00T +replays 16, where Isaac-GR00T uses 8 (1 for N1.5). π0 at 72% and Octo at 8% +are well below published results for these checkpoints and have not been +investigated yet. + +## Four suites + +| Model | Spatial | Object | Goal | Long (LIBERO-10) | Average | +|---|--:|--:|--:|--:|--:| +| BitVLA | 95.5% | 100.0% | 91.5% | 89.5% | 94.1% | +| GR00T N1.7 | 95.5% | 98.0% | 97.5% | 90.0% | 95.3% | + +Same replay as above: 8 for BitVLA, 16 for GR00T N1.7. + +Success rate belongs to the checkpoint, not the engine; +`vla_predict_check` in [CONTRIBUTING.md](../../CONTRIBUTING.md) is how a +change is shown to leave it alone. diff --git a/docs/benchmark/rtx-3060.md b/docs/benchmark/rtx-3060.md new file mode 100644 index 0000000..3a241b2 --- /dev/null +++ b/docs/benchmark/rtx-3060.md @@ -0,0 +1,145 @@ +# RTX 3060 (CUDA) + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +## Test setup + +| | | +|---|---| +| Device | NVIDIA GeForce RTX 3060, 12 GB GDDR6, Ampere (sm_86) | +| Host | Intel Core i5-12400F (6 cores, 12 threads), 15.4 GiB RAM | +| OS | Ubuntu 22.04.5 LTS, kernel 6.8 | +| Driver / toolkit | 580.178.04 / CUDA 12.8, GCC 11.4 | +| Commit | `c93ca0a` (PR #32 head, branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 01:01 to 04:49 (UTC+07) | +| Build | `-DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=86 -DCMAKE_BUILD_TYPE=Release` (`GGML_NATIVE=ON`) | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + +**Peak VRAM** is the process's device memory from `nvidia-smi --query-compute-apps`, sampled every 0.5 s. **Peak host RSS** is the process's peak resident set (`getrusage`). + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2`) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +## Latency and memory + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak VRAM MiB | Peak host RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 8.3 | 8.3 | 8.3 | 8.4 | 1.0 | 626 | 501 | +| TurboVLA | 2 | 256 | 21.6 | 22.2 | 22.1 | 22.7 | - | 982 | 496 | +| VLA-JEPA | 1 | 256 | 57.1 | 58.5 | 58.1 | 60.7 | 21.8 | 3968 | 631 | +| BitVLA | 1 | 224 | 59.1 | 59.8 | 59.4 | 61.1 | 11.9 | 1308 | 1180 | +| VLA-Adapter | 1 | 224 | 74.7 | 75.4 | 74.8 | 76.9 | 41.8 | 2862 | 1064 | +| GR00T N1.6 | 1 | 224 | 78.1 | 78.9 | 78.2 | 81.6 | 22.7 | 4682 | 636 | +| GR00T N1.7 | 1 | 256 | 82.8 | 83.7 | 83.0 | 85.8 | 21.7 | 5032 | 647 | +| GR00T N1.5 | 1 | 224 | 86.8 | 87.5 | 86.9 | 89.4 | 22.9 | 3632 | 688 | +| SmolVLA | 2 | 512 | 125.1 | 126.0 | 125.3 | 128.0 | 71.6 | 2038 | 721 | +| π0 | 2 | 224 | 228.2 | 228.8 | 228.6 | 229.2 | 44.9 | 5516 | 710 | +| π0.5 | 2 | 224 | 229.9 | 230.2 | 230.1 | 230.3 | 45.0 | 5728 | 733 | +| Evo-1 | 1 | 448 | 230.8 | 231.2 | 231.1 | 231.4 | 108.0 | 1594 | 705 | + +### Fastest configuration + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`; +- `--act-dtype bf16 --flash-attn`, for π0 and Evo-1 on CUDA only. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak VRAM MiB | Peak host RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | *(defaults)* | 8.3 | 8.3 | 8.3 | 8.4 | 1.0 | 626 | 501 | - | +| TurboVLA | `--weight-dtype f16 --flash-attn` | 13.7 | 13.7 | 13.7 | 13.8 | - | 616 | 578 | -38% | +| VLA-JEPA | `--weight-dtype f16 --flash-attn` | 46.9 | 47.6 | 47.6 | 48.1 | 17.5 | 3974 | 622 | -19% | +| BitVLA | *(defaults)* | 59.1 | 59.8 | 59.4 | 61.1 | 11.9 | 1308 | 1180 | - | +| VLA-Adapter | `--weight-dtype f16` | 65.6 | 66.3 | 65.9 | 68.2 | 35.9 | 2864 | 1064 | -12% | +| GR00T N1.7 | `--weight-dtype f16 --flash-attn` | 68.2 | 68.8 | 68.3 | 69.9 | 17.2 | 5036 | 637 | -18% | +| GR00T N1.5 | `--weight-dtype f16 --flash-attn` | 69.3 | 69.9 | 69.6 | 71.4 | 19.6 | 3630 | 612 | -20% | +| GR00T N1.6 | `--weight-dtype f16 --flash-attn` | 72.1 | 72.6 | 72.2 | 74.0 | 19.4 | 4684 | 633 | -8% | +| SmolVLA | `--flash-attn --mm-prec default` | 91.3 | 92.7 | 92.0 | 94.8 | 45.4 | 1228 | 727 | -26% | +| π0 | `--weight-dtype f16 --flash-attn` | 172.1 | 172.3 | 172.3 | 172.5 | 38.1 | 5534 | 641 | -25% | +| Evo-1 | `--act-dtype bf16 --flash-attn` | 176.2 | 176.8 | 176.4 | 176.9 | 56.2 | 1566 | 816 | -24% | +| π0.5 | `--weight-dtype f16` | 191.3 | 191.7 | 191.6 | 192.0 | 38.7 | 5728 | 657 | -17% | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | `--act-dtype bf16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 8.4 | 8.4 | 8.4 | 8.4 | 8.4 | 8.4 | 8.4 | 8.4 | | +| TurboVLA | 22.6 | 19.6 | 22.7 | 19.6 | 18.5 | 15.4 | 16.7 | 13.7 | | +| VLA-JEPA | 60.7 | 59.4 | 60.7 | 59.4 | 60.9 | 59.4 | 50.3 | 48.8 | | +| BitVLA | 61.5 | 61.5 | 61.4 | 61.3 | 61.6 | 61.6 | 61.7 | 61.7 | | +| VLA-Adapter | 79.0 | 78.6 | 78.4 | 78.3 | 78.4 | 78.2 | 67.9 | 68.0 | | +| GR00T N1.6 | 81.8 | 81.5 | 82.2 | 81.6 | 81.4 | 81.9 | 74.1 | 74.0 | | +| GR00T N1.7 | 86.2 | 85.3 | 86.1 | 85.0 | 86.0 | 85.3 | 71.8 | 70.7 | | +| GR00T N1.5 | 90.8 | 87.8 | 90.6 | 87.5 | 90.8 | 87.7 | 74.2 | 71.4 | | +| SmolVLA | 130.2 | 103.1 | 123.1 | 95.7 | 150.0 | 124.3 | 146.8 | 120.2 | | +| π0 | 231.0 | 213.7 | 230.7 | 214.3 | 231.2 | 213.8 | 190.1 | 173.4 | 202.8 | +| π0.5 | 232.4 | 233.7 | 233.3 | 232.7 | 233.1 | 232.8 | 192.5 | 192.9 | | +| Evo-1 | 233.5 | 193.4 | 233.7 | 193.8 | 234.1 | 192.2 | 222.3 | 180.5 | 179.5 | + +
+ +## Not run + +- **OpenVLA-OFT**: does not fit. Loading the weights runs out of memory: + + ```text + ggml_backend_cuda_buffer_type_alloc_buffer: allocating 14377.07 MiB on device 0: cudaMalloc failed: out of memory + alloc_tensor_range: failed to allocate CUDA0 buffer of size 15075447168 + vla(openvla_oft): alloc_weights failed (OOM?) + vla-bench: model_load failed + ``` + +## Reproducing + +```bash +cmake -S . -B build-cuda -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=86 -DCMAKE_BUILD_TYPE=Release +cmake --build build-cuda -j + +# one row, for example SmolVLA +./build-cuda/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/rtx-3090.md b/docs/benchmark/rtx-3090.md new file mode 100644 index 0000000..2cc5721 --- /dev/null +++ b/docs/benchmark/rtx-3090.md @@ -0,0 +1,137 @@ +# RTX 3090 (CUDA) + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +## Test setup + +| | | +|---|---| +| Device | NVIDIA GeForce RTX 3090, 24 GB GDDR6X, Ampere (sm_86); one of two cards in the host, the other idle | +| Host | Intel Core i7-14700F (20 cores, 28 threads), 62.6 GiB RAM | +| OS | Ubuntu 22.04.5 LTS, kernel 6.8 | +| Driver / toolkit | 595.84 / CUDA 12.8, GCC 11.4 | +| Commit | `c93ca0a` (PR #32 head, branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 00:59 to 03:28 (UTC+07) | +| Build | `-DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=86 -DCMAKE_BUILD_TYPE=Release` (`GGML_NATIVE=ON`) | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + +**Peak VRAM** is the process's device memory from `nvidia-smi --query-compute-apps`, sampled every 0.5 s. **Peak host RSS** is the process's peak resident set (`getrusage`). + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2`) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +## Latency and memory + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak VRAM MiB | Peak host RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 4.3 | 4.4 | 4.3 | 4.4 | 0.5 | 810 | 501 | +| TurboVLA | 2 | 256 | 9.0 | 9.3 | 9.4 | 9.4 | - | 1152 | 496 | +| BitVLA | 1 | 224 | 26.9 | 27.1 | 27.2 | 27.3 | 7.3 | 1460 | 1180 | +| VLA-JEPA | 1 | 256 | 27.3 | 27.8 | 27.6 | 28.1 | 8.9 | 4138 | 634 | +| VLA-Adapter | 1 | 224 | 34.2 | 34.5 | 34.5 | 34.7 | 17.0 | 3032 | 1063 | +| GR00T N1.7 | 1 | 256 | 35.3 | 35.7 | 35.4 | 36.1 | 9.0 | 5204 | 650 | +| GR00T N1.6 | 1 | 224 | 36.3 | 36.9 | 36.6 | 37.5 | 9.5 | 4854 | 643 | +| GR00T N1.5 | 1 | 224 | 36.6 | 36.9 | 36.7 | 37.2 | 9.5 | 3804 | 692 | +| SmolVLA | 2 | 512 | 61.0 | 61.3 | 61.3 | 61.6 | 28.9 | 2216 | 734 | +| π0 | 2 | 224 | 97.5 | 98.0 | 98.0 | 98.5 | 18.7 | 5670 | 712 | +| π0.5 | 2 | 224 | 97.8 | 99.6 | 99.7 | 99.9 | 18.5 | 5880 | 735 | +| Evo-1 | 1 | 448 | 108.6 | 109.0 | 108.9 | 109.5 | 39.8 | 1746 | 707 | +| OpenVLA-OFT | 1 | 224 | 134.2 | 135.2 | 135.2 | 135.8 | 17.6 | 14768 | 1036 | + +### Fastest configuration + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`; +- `--act-dtype bf16 --flash-attn`, for π0 and Evo-1 on CUDA only. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak VRAM MiB | Peak host RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | *(defaults)* | 4.3 | 4.4 | 4.3 | 4.4 | 0.5 | 810 | 501 | - | +| TurboVLA | `--weight-dtype f16 --flash-attn` | 5.7 | 5.8 | 5.8 | 5.8 | - | 714 | 577 | -38% | +| VLA-JEPA | `--weight-dtype f16` | 23.1 | 23.8 | 23.4 | 25.1 | 8.0 | 4148 | 618 | -14% | +| BitVLA | *(defaults)* | 26.9 | 27.1 | 27.2 | 27.3 | 7.3 | 1460 | 1180 | - | +| VLA-Adapter | `--weight-dtype f16 --flash-attn` | 29.3 | 29.5 | 29.4 | 29.6 | 14.0 | 3034 | 1064 | -14% | +| GR00T N1.5 | `--weight-dtype f16 --flash-attn` | 30.3 | 30.6 | 30.5 | 30.8 | 8.1 | 3802 | 612 | -17% | +| GR00T N1.7 | `--weight-dtype f16 --flash-attn` | 33.4 | 33.9 | 33.5 | 35.7 | 7.7 | 5210 | 638 | -5% | +| GR00T N1.6 | `--weight-dtype f16 --flash-attn` | 33.5 | 33.7 | 33.6 | 33.8 | 7.8 | 4852 | 633 | -9% | +| SmolVLA | `--flash-attn --mm-prec default` | 47.7 | 47.9 | 47.9 | 48.2 | 18.8 | 1402 | 737 | -22% | +| π0 | `--weight-dtype f16 --flash-attn` | 76.6 | 77.1 | 77.2 | 77.5 | 16.0 | 5706 | 642 | -21% | +| Evo-1 | `--weight-dtype f16 --flash-attn` | 82.2 | 82.8 | 82.8 | 83.2 | 21.7 | 1720 | 650 | -24% | +| π0.5 | `--weight-dtype f16` | 84.3 | 84.6 | 84.6 | 84.8 | 15.9 | 5880 | 658 | -15% | +| OpenVLA-OFT | `--weight-dtype f16` | 88.6 | 88.9 | 88.8 | 89.2 | 16.2 | 14770 | 1037 | -34% | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | `--act-dtype bf16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 4.4 | 4.4 | 5.0 | 5.0 | 5.0 | 5.0 | 5.0 | 5.0 | | +| TurboVLA | 9.1 | 8.0 | 9.1 | 8.0 | 7.6 | 6.6 | 7.5 | 6.4 | | +| BitVLA | 26.9 | 27.0 | 27.0 | 27.0 | 26.9 | 27.1 | 27.0 | 27.0 | | +| VLA-JEPA | 29.2 | 31.2 | 29.9 | 31.3 | 31.6 | 30.6 | 25.0 | 25.3 | | +| VLA-Adapter | 37.7 | 38.0 | 37.7 | 37.6 | 38.2 | 37.9 | 31.1 | 30.6 | | +| GR00T N1.7 | 37.8 | 38.8 | 39.0 | 37.2 | 37.0 | 38.2 | 34.8 | 34.0 | | +| GR00T N1.6 | 37.5 | 38.5 | 39.0 | 38.8 | 39.8 | 39.9 | 35.3 | 35.2 | | +| GR00T N1.5 | 40.5 | 39.2 | 38.9 | 37.3 | 39.9 | 39.8 | 33.1 | 32.5 | | +| SmolVLA | 61.5 | 51.1 | 58.5 | 48.1 | 67.9 | 57.4 | 66.0 | 55.7 | | +| π0 | 99.1 | 94.1 | 98.5 | 94.4 | 99.1 | 94.2 | 81.7 | 76.9 | 91.4 | +| π0.5 | 100.1 | 100.4 | 100.1 | 100.7 | 100.5 | 100.4 | 84.1 | 85.5 | | +| Evo-1 | 108.9 | 92.8 | 109.7 | 92.7 | 109.8 | 92.5 | 98.9 | 82.8 | 88.3 | +| OpenVLA-OFT | 135.4 | 135.6 | 136.0 | 135.9 | 135.9 | 135.9 | 88.9 | 88.9 | | + +
+ +## Reproducing + +```bash +cmake -S . -B build-cuda -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=86 -DCMAKE_BUILD_TYPE=Release +cmake --build build-cuda -j + +# one row, for example SmolVLA +CUDA_VISIBLE_DEVICES=0 ./build-cuda/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/rtx-5070-laptop.md b/docs/benchmark/rtx-5070-laptop.md new file mode 100644 index 0000000..8b9ba17 --- /dev/null +++ b/docs/benchmark/rtx-5070-laptop.md @@ -0,0 +1,145 @@ +# RTX 5070 Laptop GPU (CUDA) + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +## Test setup + +| | | +|---|---| +| Device | NVIDIA GeForce RTX 5070 Laptop GPU, 8 GB GDDR7, Blackwell (sm_120) | +| Host | Intel Core i9-14900HX (24 cores, 32 threads), 31 GiB RAM; laptop on AC power, platform profile `performance` | +| OS | Ubuntu 22.04.5 LTS, kernel 6.8 | +| Driver / toolkit | 590.48.01 / CUDA 12.8, GCC 11.4 | +| Commit | `c93ca0a` (PR #32 head, branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 00:59 to 03:56 (UTC+07) | +| Build | `-DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=120 -DCMAKE_BUILD_TYPE=Release` (`GGML_NATIVE=ON`) | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + +**Peak VRAM** is the process's device memory from `nvidia-smi --query-compute-apps`, sampled every 0.5 s. **Peak host RSS** is the process's peak resident set (`getrusage`). + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2`) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +## Latency and memory + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak VRAM MiB | Peak host RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 4.7 | 5.0 | 5.0 | 5.0 | 0.6 | 734 | 589 | +| TurboVLA | 2 | 256 | 14.1 | 14.5 | 14.5 | 14.8 | - | 1030 | 583 | +| VLA-JEPA | 1 | 256 | 38.0 | 40.8 | 38.6 | 50.5 | 14.2 | 4006 | 728 | +| BitVLA | 1 | 224 | 54.4 | 60.0 | 55.8 | 71.8 | 11.9 | 1328 | 1178 | +| VLA-Adapter | 1 | 224 | 56.3 | 63.0 | 59.4 | 75.9 | 34.1 | 2896 | 1071 | +| GR00T N1.7 | 1 | 256 | 63.3 | 68.0 | 64.3 | 81.6 | 16.1 | 5064 | 750 | +| GR00T N1.6 | 1 | 224 | 65.5 | 72.3 | 68.6 | 87.8 | 21.5 | 4708 | 740 | +| GR00T N1.5 | 1 | 224 | 73.0 | 79.2 | 74.4 | 95.0 | 21.2 | 3668 | 765 | +| SmolVLA | 2 | 512 | 89.3 | 98.7 | 96.5 | 111.3 | 54.2 | 2056 | 805 | +| π0 | 2 | 224 | 177.1 | 195.7 | 190.7 | 212.7 | 41.3 | 5534 | 788 | +| π0.5 | 2 | 224 | 180.5 | 196.7 | 192.1 | 209.8 | 40.9 | 5528 | 809 | +| Evo-1 | 1 | 448 | 184.6 | 201.6 | 201.7 | 217.7 | 92.4 | 1604 | 805 | + +### Fastest configuration + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`; +- `--act-dtype bf16 --flash-attn`, for π0 and Evo-1 on CUDA only. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak VRAM MiB | Peak host RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | `--flash-attn` | 4.4 | 4.7 | 4.8 | 4.8 | 0.6 | 734 | 590 | -6% | +| TurboVLA | `--weight-dtype f16 --flash-attn` | 8.6 | 9.1 | 9.1 | 9.3 | - | 586 | 642 | -37% | +| VLA-JEPA | *(defaults)* | 38.0 | 40.8 | 38.6 | 50.5 | 14.2 | 4006 | 728 | - | +| VLA-Adapter | `--weight-dtype f16 --flash-attn` | 48.1 | 53.6 | 49.6 | 63.7 | 27.4 | 2894 | 1071 | -15% | +| BitVLA | *(defaults)* | 54.4 | 60.0 | 55.8 | 71.8 | 11.9 | 1328 | 1178 | - | +| GR00T N1.5 | `--weight-dtype f16 --flash-attn` | 59.8 | 64.8 | 61.2 | 76.4 | 15.6 | 3670 | 688 | -18% | +| SmolVLA | `--flash-attn --mm-prec default` | 60.4 | 65.3 | 61.4 | 77.5 | 29.4 | 1238 | 810 | -34% | +| GR00T N1.7 | *(defaults)* | 63.3 | 68.0 | 64.3 | 81.6 | 16.1 | 5064 | 750 | - | +| GR00T N1.6 | *(defaults)* | 65.5 | 72.3 | 68.6 | 87.8 | 21.5 | 4708 | 740 | - | +| Evo-1 | `--weight-dtype f16 --flash-attn` | 129.0 | 141.8 | 138.4 | 153.9 | 36.0 | 1558 | 729 | -30% | +| π0 | `--weight-dtype f16 --flash-attn` | 135.4 | 149.0 | 145.3 | 169.3 | 31.5 | 5548 | 716 | -24% | +| π0.5 | `--weight-dtype f16 --flash-attn` | 147.1 | 161.0 | 156.6 | 174.9 | 29.0 | 5752 | 733 | -18% | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | `--act-dtype bf16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 4.8 | 4.4 | 4.4 | 4.5 | 4.5 | 4.5 | 4.4 | 4.5 | | +| TurboVLA | 14.7 | 12.1 | 14.3 | 12.0 | 12.4 | 10.0 | 11.2 | 9.1 | | +| VLA-JEPA | 38.2 | 37.3 | 39.2 | 37.5 | 38.7 | 37.5 | 44.4 | 43.6 | | +| BitVLA | 55.6 | 55.7 | 55.6 | 55.8 | 56.0 | 55.9 | 56.1 | 56.0 | | +| VLA-Adapter | 58.6 | 58.1 | 57.9 | 57.5 | 58.5 | 58.6 | 48.7 | 48.5 | | +| GR00T N1.7 | 64.0 | 62.9 | 63.8 | 63.3 | 64.5 | 62.9 | 73.2 | 72.2 | | +| GR00T N1.6 | 67.9 | 68.3 | 67.6 | 68.1 | 67.8 | 68.2 | 71.1 | 71.2 | | +| GR00T N1.5 | 73.7 | 71.3 | 73.0 | 71.2 | 73.8 | 71.9 | 62.1 | 60.5 | | +| SmolVLA | 90.2 | 67.0 | 85.6 | 61.0 | 95.1 | 71.3 | 88.9 | 65.5 | | +| π0 | 197.5 | 187.5 | 198.0 | 187.1 | 197.8 | 188.0 | 165.3 | 150.3 | 183.1 | +| π0.5 | 198.1 | 198.4 | 198.4 | 198.6 | 198.7 | 199.1 | 165.8 | 163.0 | | +| Evo-1 | 203.7 | 156.0 | 204.0 | 156.5 | 204.1 | 157.4 | 191.9 | 144.0 | 151.9 | + +
+ +## Not run + +- **OpenVLA-OFT**: does not fit. Loading the weights runs out of memory: + + ```text + ggml_backend_cuda_buffer_type_alloc_buffer: allocating 14377.07 MiB on device 0: cudaMalloc failed: out of memory + alloc_tensor_range: failed to allocate CUDA0 buffer of size 15075447168 + vla(openvla_oft): alloc_weights failed (OOM?) + vla-bench: model_load failed + ``` + +## Reproducing + +```bash +cmake -S . -B build-cuda -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=120 -DCMAKE_BUILD_TYPE=Release +cmake --build build-cuda -j + +# one row, for example SmolVLA +./build-cuda/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/rtx-5090.md b/docs/benchmark/rtx-5090.md new file mode 100644 index 0000000..ba53a72 --- /dev/null +++ b/docs/benchmark/rtx-5090.md @@ -0,0 +1,123 @@ +# RTX 5090 (CUDA) + +> These numbers were reported by the author of PR #32. We did not re-measure +> them on our hardware. Fields the author did not report are marked `-`. + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +## Test setup + +| | | +|---|---| +| Device | NVIDIA GeForce RTX 5090, 32 GB GDDR7, Blackwell (sm_120) | +| Host CPU | 24 cores (model not reported) | +| Driver / toolkit | 595.84 / CUDA 13.2 | +| Commit | PR #32 branch `b11223-numerics-perf` (head `c93ca0a`), baseline `7abe1b4` | +| llama.cpp | `b11223` (baseline: `b10729`) | +| Date | 2026-09-28 (PR opened; exact run time not reported) | +| Build | `-DGGML_CUDA=ON -DCMAKE_BUILD_TYPE=Release` | +| Runtime flags | defaults | +| Method | minimum over 5 rounds, alternating the baseline build and this branch | + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +Notes on the configuration: + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +## Latency and memory + +Rows are sorted by latency. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak VRAM MiB | Peak host RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 2.69 | - | - | - | - | - | - | +| TurboVLA | 2 | 256 | 4.88 | - | - | - | - | - | - | +| VLA-JEPA | 1 | 256 | 14.3 | - | - | - | - | - | - | +| VLA-Adapter | 1 | 224 | 16.0 | - | - | - | - | - | - | +| GR00T N1.5 | 1 | 224 | 16.1 | - | - | - | - | - | - | +| GR00T N1.6 | 1 | 224 | 19.4 | - | - | - | - | - | - | +| GR00T N1.7 | 1 | 256 | 19.5 | - | - | - | - | - | - | +| BitVLA | 1 | 224 | 20.8 | - | - | - | - | - | - | +| π0 | 2 | 224 | 28.8 | - | - | - | - | - | - | +| π0.5 | 2 | 224 | 29.2 | - | - | - | - | - | - | +| OpenVLA-OFT | 1 | 224 | 34.5 | - | - | - | - | - | - | +| SmolVLA | 2 | 512 | 38.2 | - | - | - | - | - | - | +| Evo-1 | 1 | 448 | 49.1 | - | - | - | - | - | - | + +The author did not report memory. The PR notes that keeping only the selected +GR00T embodiment resident saves 1.2 GiB of VRAM. + +### Fastest configuration + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`; +- `--act-dtype bf16 --flash-attn`, for π0 and Evo-1 on CUDA only. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +The author did not report a fastest-mode run, so this device has no numbers. + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak VRAM MiB | Peak host RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | - | - | - | - | - | - | - | - | - | +| TurboVLA | - | - | - | - | - | - | - | - | - | +| VLA-JEPA | - | - | - | - | - | - | - | - | - | +| VLA-Adapter | - | - | - | - | - | - | - | - | - | +| GR00T N1.5 | - | - | - | - | - | - | - | - | - | +| GR00T N1.6 | - | - | - | - | - | - | - | - | - | +| GR00T N1.7 | - | - | - | - | - | - | - | - | - | +| BitVLA | - | - | - | - | - | - | - | - | - | +| π0 | - | - | - | - | - | - | - | - | - | +| π0.5 | - | - | - | - | - | - | - | - | - | +| OpenVLA-OFT | - | - | - | - | - | - | - | - | - | +| SmolVLA | - | - | - | - | - | - | - | - | - | +| Evo-1 | - | - | - | - | - | - | - | - | - | + +## Reproducing + +```bash +cmake -B build-cuda -DGGML_CUDA=ON -DCMAKE_BUILD_TYPE=Release +cmake --build build-cuda -j + +# one row, for example SmolVLA +./build-cuda/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/ryzen-5-5500.md b/docs/benchmark/ryzen-5-5500.md new file mode 100644 index 0000000..4d5e3e1 --- /dev/null +++ b/docs/benchmark/ryzen-5-5500.md @@ -0,0 +1,141 @@ +# AMD Ryzen 5 5500 (CPU) + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +## Test setup + +| | | +|---|---| +| Device | AMD Ryzen 5 5500, 6 cores, 12 threads (Zen 3) | +| Memory | 15.5 GiB DDR4 | +| Power | stock; `amd-pstate-epp` `powersave` governor | +| Threads | 12 (the default: all hardware threads, capped at 16) | +| OS | Ubuntu 22.04.5 LTS, kernel 6.8 | +| Toolchain | GCC 11.4, CMake 3.22 | +| Commit | `c93ca0a` (PR #32 head, branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 01:08 to 04:53 (UTC+07) | +| Build | `-DCMAKE_BUILD_TYPE=Release` (`GGML_NATIVE=ON`), CPU backend only | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + +**Peak RSS** is the process's peak resident set (`getrusage`). + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2`) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +## Latency and memory + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 96.6 | 97.2 | 97.1 | 97.5 | 11.2 | 622 | +| TurboVLA | 2 | 256 | 475.7 | 476.1 | 476.1 | 476.4 | - | 855 | +| VLA-JEPA | 1 | 256 | 2133.0 | 2147.9 | 2136.5 | 2144.8 | 726.6 | 3823 | +| GR00T N1.7 | 1 | 256 | 2518.8 | 2525.3 | 2522.6 | 2534.5 | 728.0 | 4884 | +| VLA-Adapter | 1 | 224 | 2781.7 | 2804.8 | 2798.3 | 2803.6 | 1623.0 | 2707 | +| GR00T N1.6 | 1 | 224 | 2826.0 | 2833.3 | 2827.5 | 2842.3 | 892.4 | 4524 | +| SmolVLA | 2 | 512 | 3124.3 | 3131.8 | 3131.5 | 3138.8 | 2125.0 | 1101 | +| GR00T N1.5 | 1 | 224 | 3447.0 | 3461.6 | 3450.2 | 3471.3 | 887.2 | 3500 | +| Evo-1 | 1 | 448 | 7347.2 | 7385.4 | 7381.8 | 7400.9 | 3651.6 | 1432 | +| π0.5 | 2 | 224 | 11225.6 | 11333.6 | 11342.5 | 11402.1 | 1788.3 | 5656 | +| π0 | 2 | 224 | 11321.3 | 11439.9 | 11443.3 | 11505.5 | 1810.5 | 5355 | + +### Fastest configuration + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak RSS MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | *(defaults)* | 96.6 | 97.2 | 97.1 | 97.5 | 11.2 | 622 | - | +| TurboVLA | `--weight-dtype f16 --flash-attn` | 452.7 | 453.2 | 453.1 | 453.6 | - | 490 | -5% | +| VLA-JEPA | *(defaults)* | 2133.0 | 2147.9 | 2136.5 | 2144.8 | 726.6 | 3823 | - | +| GR00T N1.7 | *(defaults)* | 2518.8 | 2525.3 | 2522.6 | 2534.5 | 728.0 | 4884 | - | +| SmolVLA | `--weight-dtype f16 --flash-attn` | 2684.7 | 2687.5 | 2687.5 | 2688.7 | 1674.1 | 1050 | -14% | +| VLA-Adapter | *(defaults)* | 2781.7 | 2804.8 | 2798.3 | 2803.6 | 1623.0 | 2707 | - | +| GR00T N1.6 | *(defaults)* | 2826.0 | 2833.3 | 2827.5 | 2842.3 | 892.4 | 4524 | - | +| GR00T N1.5 | *(defaults)* | 3447.0 | 3461.6 | 3450.2 | 3471.3 | 887.2 | 3500 | - | +| Evo-1 | `--weight-dtype f16 --flash-attn` | 6798.7 | 6813.3 | 6802.8 | 6808.0 | 3057.5 | 1368 | -8% | +| π0.5 | *(defaults)* | 11225.6 | 11333.6 | 11342.5 | 11402.1 | 1788.3 | 5656 | - | +| π0 | *(defaults)* | 11321.3 | 11439.9 | 11443.3 | 11505.5 | 1810.5 | 5355 | - | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | +|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 97.0 | 97.2 | 97.2 | 97.0 | 97.2 | 96.9 | 97.0 | 97.2 | +| TurboVLA | 477.4 | 470.1 | 554.6 | 470.8 | 463.6 | 453.9 | 462.7 | 453.3 | +| VLA-JEPA | 2146.9 | 2151.8 | 2149.5 | 2149.1 | 2147.7 | 2150.4 | 2182.0 | 2144.0 | +| GR00T N1.7 | 2535.8 | 2584.2 | 2536.4 | 2536.3 | 2541.3 | 2533.0 | 2529.2 | 2530.2 | +| VLA-Adapter | 2830.7 | 2823.7 | 2851.0 | 2826.7 | 2823.4 | 2821.8 | 2810.3 | 2802.9 | +| GR00T N1.6 | 2890.1 | 2847.0 | 2884.9 | 2847.8 | 2887.4 | 2844.3 | 2838.1 | 2838.7 | +| SmolVLA | 3188.8 | 2691.2 | 3156.4 | 2709.7 | 3156.2 | 2696.1 | 3156.9 | 2680.8 | +| GR00T N1.5 | 3445.9 | 3469.2 | 3446.6 | 3436.9 | 3454.6 | 3455.1 | 3442.2 | 3425.2 | +| Evo-1 | 7431.9 | 6829.1 | 7427.4 | 6836.2 | 7465.8 | 6869.5 | 7445.8 | 6801.7 | +| π0.5 | 11338.0 | 11404.7 | 11399.2 | 11388.7 | 11454.6 | 11407.8 | 11320.1 | 11346.4 | +| π0 | 11552.5 | 11536.3 | 11520.4 | 11529.7 | 11538.6 | 11595.4 | 11408.4 | 11421.2 | + +
+ +## Not run + +- **BitVLA**: N/A off CUDA. The only published GGUF is int2-packed, and only CUDA builds load it: + + ```text + vla(bitvla): int2-packed GGUF requires a CUDA build (VLA_BITVLA_CUDA_KERNELS); use the bf16 GGUF for CPU. + ``` +- **OpenVLA-OFT**: not attempted. Its 14 GiB of weights do not fit in 15 GiB of RAM. + +## Reproducing + +```bash +cmake -S . -B build-cpu -DCMAKE_BUILD_TYPE=Release +cmake --build build-cpu -j + +# one row, for example SmolVLA +./build-cpu/vla-bench --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/docs/benchmark/snapdragon-x-hexagon.md b/docs/benchmark/snapdragon-x-hexagon.md new file mode 100644 index 0000000..6f3a629 --- /dev/null +++ b/docs/benchmark/snapdragon-x-hexagon.md @@ -0,0 +1,221 @@ +# Snapdragon X: Hexagon NPU and Oryon CPU + +`vla-bench` times `predict()` in-process on synthetic inputs. It measures the +engine only: no transport, no simulator, and no claim about task success. + +One `build-wos-htp` binary covers both sections. By default the graph runs on the Hexagon NPU, and +ops the NPU rejects run on the CPU through vla.cpp's fallback wrapper. `VLA_DEVICE=cpu` skips the +NPU, so the same binary gives the CPU numbers. + +## Test setup + +| | | +|---|---| +| Device | ASUS Vivobook 14 X1407QA laptop: Snapdragon X X1-26-100 (8 Oryon cores), Hexagon v73 NPU (driver 30.0.220.3000), 16 GB RAM (15.6 GiB visible) | +| Threads | 8 (the default: all hardware threads), for the CPU runs and the NPU's CPU fallback | +| Power | on AC power, Windows power plan `Balanced` | +| OS | Windows 11 Home, build 26200 | +| Toolchain | Visual Studio clang 22.1.3, Hexagon SDK 6.6.0.0 (tools 19.0.07) | +| Commit | `c93ca0a` (PR #32 head, branch `b11223-numerics-perf`) | +| llama.cpp | `b11223` | +| Date | 2026-09-30 01:40 to 05:58 (UTC+07) | +| Build | `scripts/build_windows_snapdragon.ps1 -Backend htp -HtpCert -LlamaDir ` (servers included) | +| Runtime flags | defaults; the fastest flags per model are in the second table | +| Method | 3 warmups + 20 timed reps per process, 3 processes. The process with the lowest mean is reported, and memory is the peak over all three. | + +**Peak working set** is the process's peak working set from `GetProcessMemoryInfo`. On the NPU the weights live in FastRPC shared memory, which counts toward the working set. **Peak private** is the process's peak private commit (`PeakPagefileUsage`). + +## Model configuration + +Each model runs at its native view count and input size, with weights as +shipped. Checkpoints are the `vrfai/*` GGUFs; where a repo ships several, we use +the `libero_object` variant. Settings not listed are the defaults. + +| Model | Checkpoint | Views | Input | Lang tokens | Action chunk | Action dim | Action head | Steps | +|---|---|--:|--:|--:|--:|--:|---|--:| +| SmolVLA | `smolvla-libero-gguf` | 2 | 512 | 16 | 50 | 7 | flow matching | 10 | +| π0 | `pi0-libero-finetuned-v044-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| π0.5 | `pi05-libero-gguf` | 2 | 224 | 16 | 50 | 7 | flow matching | 10 | +| GR00T N1.5 | `gr00tn1d5-libero-object-gguf` | 1 | 224 | 16 | 16 | 32 | flow matching (DiT) | 4 | +| GR00T N1.6 | `gr00tn1d6-libero-gguf` | 1 | 224 | 16 | 50 | 128 | flow matching (DiT) | 4 | +| GR00T N1.7 | `gr00tn1d7-libero-gguf` | 1 | 256 | 16 | 40 | 132 | flow matching (DiT) | 4 | +| VLA-JEPA | `vla-jepa-libero` | 1 | 256 | 16 + 32 | 7 | 7 | flow matching (DiT) | 4 | +| Evo-1 | `evo1-libero-gguf` | 1 | 448 | 16 | 50 | 7 | flow matching | 32 | +| BitVLA | `bitvla-libero-gguf` (int2`) | 1 | 224 | 16 | 8 | 7 | parallel decoding | 1 | +| VLA-Adapter | `vla-adapter-libero-gguf` | 1 | 224 | 16 | 8 | 7 | bridge-attention policy | 1 | +| OpenVLA-OFT | `openvla-oft-libero-gguf` | 1 | 224 | 16 | 8 | 7 | L1 regression head | 1 | +| Octo-Small | `octo-small-libero-gguf` | 2 | 256 + 128 | 16 | 4 | 7 | DDPM diffusion | 20 | +| TurboVLA | `turbovla-libero-gguf` | 2 | 256 | 16 | 12 | 7 | single-pass decoder | 1 | + +- **Lang tokens** is the synthetic prompt length. VLA-JEPA adds its 32 + `` tokens with `--extra-token 151697 --extra-count 32`. +- **Action dim** is the checkpoint's padded width. LIBERO uses the first 7 dims. +- **GR00T** needs its embodiment set in the environment: + `VLA_GR00T_EMBODIMENT=new_embodiment` for N1.5, `libero_panda` for N1.6 and + `libero_sim` for N1.7. **Octo** needs `VLA_OCTO_UNNORM_DATASET=libero_object`. +- **Octo** takes a 256 px primary view and a 128 px wrist view. + +### How the fastest configuration is picked + +We screen each model over these runtime flag sets, then re-run the fastest set +under the full protocol: + +- the defaults; +- `--flash-attn` and `--mm-prec default`, alone and together; +- `--weight-dtype bf16` and `--weight-dtype f16`, each alone and with + `--flash-attn`; +- `--flash-attn 0`, because flash attention is on by default in Hexagon builds. + +The screen runs 2 warmups and 5 reps in one process per flag set. A flag set +replaces the defaults only if its mean is more than 3% lower, both in the screen +and in the full re-run. + +Flash attention, bf16 activations and lower-precision weights can move the +action chunk. Success rates measured at the defaults therefore do not carry over +to these rows. + +## Latency and memory: Hexagon NPU (with CPU fallback) + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak working set MiB | Peak private MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 239.7 | 266.3 | 265.4 | 288.4 | 51.2 | 679 | 218 | +| VLA-JEPA | 1 | 256 | 821.0 | 846.2 | 846.0 | 859.0 | 369.2 | 3900 | 1105 | +| SmolVLA | 2 | 512 | 1239.3 | 1246.9 | 1247.7 | 1249.5 | 430.6 | 1129 | 334 | +| GR00T N1.7 | 1 | 256 | 1400.2 | 1421.3 | 1417.2 | 1438.2 | 367.9 | 4964 | 625 | +| TurboVLA | 2 | 256 | 2280.8 | 2292.7 | 2289.2 | 2305.5 | - | 948 | 208 | +| GR00T N1.5 | 1 | 224 | 2755.7 | 2773.9 | 2773.0 | 2784.5 | 1802.1 | 3560 | 151 | +| GR00T N1.6 | 1 | 224 | 2940.0 | 2957.1 | 2956.9 | 2970.7 | 1861.1 | 4601 | 164 | +| VLA-Adapter | 1 | 224 | 4072.1 | 4099.2 | 4101.9 | 4114.7 | 2888.8 | 2796 | 861 | +| π0 | 2 | 224 | 5730.0 | 5739.3 | 5740.4 | 5743.9 | 3614.4 | 5384 | 230 | +| Evo-1 | 1 | 448 | 7208.6 | 7292.2 | 7302.9 | 7344.8 | 1054.1 | 1499 | 276 | +| π0.5 | 2 | 224 | 9321.7 | 9363.2 | 9367.4 | 9395.1 | 4531.4 | 5701 | 230 | + +### Fastest configuration: Hexagon NPU (with CPU fallback) + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak working set MiB | Peak private MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | *(defaults)* | 239.7 | 266.3 | 265.4 | 288.4 | 51.2 | 679 | 218 | - | +| TurboVLA | `--weight-dtype f16 --flash-attn` | 490.1 | 554.7 | 541.9 | 600.9 | - | 586 | 208 | -76% | +| VLA-JEPA | *(defaults)* | 821.0 | 846.2 | 846.0 | 859.0 | 369.2 | 3900 | 1105 | - | +| SmolVLA | *(defaults)* | 1239.3 | 1246.9 | 1247.7 | 1249.5 | 430.6 | 1129 | 334 | - | +| GR00T N1.7 | *(defaults)* | 1400.2 | 1421.3 | 1417.2 | 1438.2 | 367.9 | 4964 | 625 | - | +| GR00T N1.5 | *(defaults)* | 2755.7 | 2773.9 | 2773.0 | 2784.5 | 1802.1 | 3560 | 151 | - | +| GR00T N1.6 | *(defaults)* | 2940.0 | 2957.1 | 2956.9 | 2970.7 | 1861.1 | 4601 | 164 | - | +| VLA-Adapter | *(defaults)* | 4072.1 | 4099.2 | 4101.9 | 4114.7 | 2888.8 | 2796 | 861 | - | +| π0 | *(defaults)* | 5730.0 | 5739.3 | 5740.4 | 5743.9 | 3614.4 | 5384 | 230 | - | +| Evo-1 | *(defaults)* | 7208.6 | 7292.2 | 7302.9 | 7344.8 | 1054.1 | 1499 | 276 | - | +| π0.5 | *(defaults)* | 9321.7 | 9363.2 | 9367.4 | 9395.1 | 4531.4 | 5701 | 230 | - | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | `--flash-attn 0` | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 275.6 | 270.9 | 262.0 | 272.9 | 278.9 | 270.7 | 262.8 | 280.2 | 271.8 | +| VLA-JEPA | 846.2 | 849.0 | 848.2 | 853.7 | 6512.3 | 6373.1 | 856.8 | 846.5 | 1424.1 | +| SmolVLA | 1251.1 | 1251.4 | 1245.3 | 1250.3 | 7625.0 | 7714.3 | 1248.8 | 1254.5 | 4753.7 | +| GR00T N1.7 | 1418.6 | 1415.1 | 1426.4 | 1415.0 | 7752.4 | 8351.3 | 1425.8 | 1416.1 | 1910.3 | +| TurboVLA | 2284.2 | 2285.7 | 2282.8 | 2294.4 | 1437.3 | 1426.4 | 552.1 | 516.9 | 2591.1 | +| GR00T N1.5 | 2786.6 | 2787.3 | 2775.8 | 2778.1 | 10053.6 | 10119.7 | 2777.8 | 2787.3 | 3772.6 | +| GR00T N1.6 | 2962.3 | 2960.8 | 2969.8 | 2962.5 | 8713.5 | 8558.1 | 2962.7 | 2959.9 | 3510.4 | +| VLA-Adapter | 4160.1 | 4123.3 | 4095.0 | 4095.5 | 9240.4 | 9441.1 | 4135.3 | 4121.3 | 4140.0 | +| π0 | 5736.0 | 5734.2 | 5731.2 | 5735.4 | 59145.8 | 57928.4 | 5724.3 | 5716.5 | 9426.5 | +| Evo-1 | 7331.0 | 7301.4 | 7314.1 | 7321.5 | 20437.4 | 20437.9 | 7325.5 | 7316.4 | 12709.7 | +| π0.5 | 9377.9 | 9397.3 | 9412.7 | 9348.2 | 58598.5 | 60551.1 | 9368.5 | 9394.8 | 9335.6 | + +
+ +## Latency and memory: Oryon CPU (`VLA_DEVICE=cpu`) + +Rows are sorted by latency. `vision ms` is `-` for archs that do not time their +vision stage separately. + +| Model | Views | Input | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak working set MiB | Peak private MiB | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 2 | 256 + 128 | 109.5 | 111.6 | 111.1 | 114.3 | 13.1 | 623 | 729 | +| TurboVLA | 2 | 256 | 548.9 | 591.2 | 617.6 | 625.4 | - | 847 | 1020 | +| VLA-JEPA | 1 | 256 | 1656.5 | 1668.1 | 1664.1 | 1674.2 | 575.5 | 3826 | 4886 | +| GR00T N1.7 | 1 | 256 | 1988.1 | 2018.0 | 2002.6 | 2043.9 | 570.8 | 4886 | 5460 | +| VLA-Adapter | 1 | 224 | 2013.1 | 2181.2 | 2180.4 | 2209.5 | 1279.4 | 2708 | 3508 | +| SmolVLA | 2 | 512 | 2080.9 | 2312.7 | 2321.2 | 2337.6 | 1540.8 | 1063 | 1352 | +| GR00T N1.6 | 1 | 224 | 2226.0 | 2352.5 | 2272.7 | 2524.1 | 717.9 | 4524 | 4648 | +| GR00T N1.5 | 1 | 224 | 2736.6 | 2863.4 | 2761.5 | 3080.6 | 734.4 | 3488 | 3599 | +| Evo-1 | 1 | 448 | 5518.9 | 5885.5 | 6116.9 | 6142.7 | 2785.1 | 1363 | 1493 | +| π0.5 | 2 | 224 | 9140.6 | 10325.4 | 10198.6 | 11022.6 | 1614.3 | 5643 | 5674 | +| π0 | 2 | 224 | 9224.5 | 10401.7 | 10298.1 | 11023.1 | 1556.6 | 5321 | 5472 | + +### Fastest configuration: Oryon CPU (`VLA_DEVICE=cpu`) + +| Model | Fastest flags | min ms | mean ms | p50 ms | p90 ms | vision ms | Peak working set MiB | Peak private MiB | vs defaults | +|---|---|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | *(defaults)* | 109.5 | 111.6 | 111.1 | 114.3 | 13.1 | 623 | 729 | - | +| TurboVLA | `--weight-dtype f16` | 358.8 | 363.3 | 363.1 | 367.3 | - | 485 | 655 | -39% | +| VLA-JEPA | *(defaults)* | 1656.5 | 1668.1 | 1664.1 | 1674.2 | 575.5 | 3826 | 4886 | - | +| GR00T N1.7 | *(defaults)* | 1988.1 | 2018.0 | 2002.6 | 2043.9 | 570.8 | 4886 | 5460 | - | +| VLA-Adapter | *(defaults)* | 2013.1 | 2181.2 | 2180.4 | 2209.5 | 1279.4 | 2708 | 3508 | - | +| SmolVLA | *(defaults)* | 2080.9 | 2312.7 | 2321.2 | 2337.6 | 1540.8 | 1063 | 1352 | - | +| GR00T N1.6 | *(defaults)* | 2226.0 | 2352.5 | 2272.7 | 2524.1 | 717.9 | 4524 | 4648 | - | +| GR00T N1.5 | *(defaults)* | 2736.6 | 2863.4 | 2761.5 | 3080.6 | 734.4 | 3488 | 3599 | - | +| Evo-1 | *(defaults)* | 5518.9 | 5885.5 | 6116.9 | 6142.7 | 2785.1 | 1363 | 1493 | - | +| π0.5 | *(defaults)* | 9140.6 | 10325.4 | 10198.6 | 11022.6 | 1614.3 | 5643 | 5674 | - | +| π0 | *(defaults)* | 9224.5 | 10401.7 | 10298.1 | 11023.1 | 1556.6 | 5321 | 5472 | - | + +
Screen means (ms) per flag set + +| Model | defaults | `--flash-attn` | `--mm-prec default` | `--flash-attn --mm-prec default` | `--weight-dtype bf16` | `--weight-dtype bf16 --flash-attn` | `--weight-dtype f16` | `--weight-dtype f16 --flash-attn` | `--flash-attn 0` | +|---|--:|--:|--:|--:|--:|--:|--:|--:|--:| +| Octo-Small | 125.1 | 133.6 | 126.2 | 126.0 | 124.5 | 125.9 | 128.9 | 126.1 | 126.2 | +| TurboVLA | 619.8 | 622.1 | 620.0 | 621.3 | 1138.8 | 1141.9 | 361.8 | 366.7 | 629.7 | +| VLA-JEPA | 1912.1 | 1763.2 | 1673.6 | 1753.8 | 6252.8 | 5802.0 | 2477.1 | 1910.1 | 1723.4 | +| GR00T N1.7 | 2244.6 | 2067.0 | 2253.5 | 2088.2 | 7437.8 | 7080.7 | 2077.1 | 2123.7 | 2272.3 | +| VLA-Adapter | 2465.6 | 2454.5 | 2684.9 | 2445.1 | 7794.7 | 7421.5 | 2381.1 | 2458.7 | 2463.8 | +| SmolVLA | 2883.6 | 2629.0 | 2642.4 | 2642.4 | 7254.5 | 7268.4 | 2698.5 | 2638.0 | 3001.2 | +| GR00T N1.6 | 2583.0 | 2395.4 | 2336.1 | 2386.6 | 8396.8 | 8137.0 | 2298.5 | 2452.0 | 2419.4 | +| GR00T N1.5 | 3192.7 | 3127.0 | 3110.0 | 3114.5 | 10416.3 | 10197.1 | 3089.9 | 3109.3 | 3353.4 | +| Evo-1 | 6223.3 | 6346.8 | 6301.9 | 6390.8 | 18169.2 | 17479.1 | 6138.2 | 6340.4 | 6515.3 | +| π0.5 | 11258.2 | 11366.8 | 11239.9 | 11168.1 | 33696.1 | 34020.8 | 11163.7 | 11248.8 | 11292.2 | +| π0 | 11321.0 | 11457.8 | 11441.3 | 11290.5 | 33961.6 | 34012.7 | 10983.4 | 11208.4 | 11245.1 | + +
+ +## Not run + +### Hexagon NPU (with CPU fallback) + +- **BitVLA**: N/A off CUDA. The only published GGUF is int2-packed, and only CUDA builds load it: + + ```text + vla(bitvla): int2-packed GGUF requires a CUDA build (VLA_BITVLA_CUDA_KERNELS); use the bf16 GGUF for CPU. + ``` +- **OpenVLA-OFT**: not attempted. Its 14 GiB of weights do not fit next to Windows in 16 GB of RAM. + +### Oryon CPU (`VLA_DEVICE=cpu`) + +- **BitVLA**: N/A off CUDA. The only published GGUF is int2-packed, and only CUDA builds load it: + + ```text + vla(bitvla): int2-packed GGUF requires a CUDA build (VLA_BITVLA_CUDA_KERNELS); use the bf16 GGUF for CPU. + ``` +- **OpenVLA-OFT**: not attempted. Its 14 GiB of weights do not fit next to Windows in 16 GB of RAM. + +## Notes + +`ADSP_LIBRARY_PATH` must be unset, or name the skels built with the binary. On this machine it pointed at the skels of an older llama.cpp build (b11201). With those loaded, every model aborted at its first graph with `ggml-hex: dspqueue_read failed: 0x00000072`. With it unset, the backend loads the skels next to the executable, and every model runs. + +In this build the default weight dtype is F16, for the CPU runs as well as the NPU. The screen shows it: `--weight-dtype f16` tracks the defaults, while `--weight-dtype bf16` is about 3x slower. TurboVLA's checkpoint is F32 and stays F32 at the defaults, which is why `--weight-dtype f16` helps it so much. + +The CPU rows vary more than on the desktop hosts. For example, π0.5's `min` is 9.1 s against a 10.3 s `mean`. The laptop ran on the `Balanced` power plan, and neither clocks nor temperature were controlled. + +## Reproducing + +```powershell +.\scripts\build_windows_snapdragon.ps1 -Backend htp -LlamaDir -HtpCert +Remove-Item Env:ADSP_LIBRARY_PATH -ErrorAction SilentlyContinue + +# one row on the NPU, for example SmolVLA +.\build-wos-htp\bin\vla-bench.exe --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +# the same row on the CPU +$env:VLA_DEVICE = "cpu"; .\build-wos-htp\bin\vla-bench.exe --ckpt smolvla-libero.gguf --images 2 --size 512 --warmup 3 --reps 20 +``` diff --git a/eval/Dockerfile.client b/eval/Dockerfile.client index fcba0b6..cae14e4 100644 --- a/eval/Dockerfile.client +++ b/eval/Dockerfile.client @@ -30,6 +30,7 @@ ENV DEBIAN_FRONTEND=noninteractive \ LANG=C.UTF-8 \ LC_ALL=C.UTF-8 \ PROTOCOL_BUFFERS_PYTHON_IMPLEMENTATION=python \ + PIP_NO_CACHE_DIR=1 \ VLA_CPP_PROTO=/workspace/vla.cpp/src/serving/vla.proto # System dependencies: build tools, protobuf, ZMQ, EGL (for MuJoCo offscreen) @@ -66,8 +67,7 @@ RUN pip3 install "mujoco<3.0" # PyTorch CPU-only (the server handles GPU inference) # install it now to prevent later deps from pulling in a GPU version -RUN pip3 install torch==2.5.1 --index-url https://download.pytorch.org/whl/cpu -RUN pip3 install torchvision==0.20.1 --index-url https://download.pytorch.org/whl/cpu +RUN pip3 install torch==2.7.1 torchvision==0.22.1 --index-url https://download.pytorch.org/whl/cpu # LIBERO — clone and install. # We use editable install (following eval/sim/libero/setup_libero.sh) so that @@ -78,18 +78,14 @@ RUN pip3 install torchvision==0.20.1 --index-url https://download.pytorch.org/wh # The CMAKE_POLICY_VERSION_MINIMUM env-var is needed by egl-probe (a transitive # dep of robomimic/robosuite) which requests cmake_minimum_required < 3.5. RUN CMAKE_POLICY_VERSION_MINIMUM=3.5 pip3 install --upgrade pip setuptools -RUN git clone --depth 1 https://github.com/Lifelong-Robot-Learning/LIBERO.git /tmp/LIBERO +RUN git clone https://github.com/Lifelong-Robot-Learning/LIBERO.git /tmp/LIBERO && git -C /tmp/LIBERO checkout 8f1084e3132a39270c3a13ebe37270a43ece2a01 RUN CMAKE_POLICY_VERSION_MINIMUM=3.5 pip3 install -r /tmp/LIBERO/requirements.txt RUN CMAKE_POLICY_VERSION_MINIMUM=3.5 pip3 install \ -e /tmp/LIBERO --config-settings editable_mode=compat && \ rm -rf /root/.libero && \ echo "N" | python3 -c "import libero.libero" 2>/dev/null || true -# lerobot — install with dependencies but immediately re-pin torch so it is -# not upgraded from our pinned CPU-only version. -RUN pip3 install lerobot==0.4.3 && \ - pip3 install torch==2.5.1 --index-url https://download.pytorch.org/whl/cpu && \ - pip3 install torchvision==0.20.1 --index-url https://download.pytorch.org/whl/cpu +RUN pip3 install lerobot==0.4.3 # Remaining client dependencies RUN pip3 install \ @@ -120,9 +116,6 @@ RUN pip3 install numpy==1.26.4 RUN sed -i 's/np_core = np._core if is_numpy_available("2.0.0") else np.core/np_core = np.core/' \ /usr/local/lib/python3.10/dist-packages/accelerate/utils/other.py 2>/dev/null || true -# Clean up pip cache to keep image size small -RUN rm -rf /root/.cache/pip - # The repo goes last: it changes every commit, and everything above it is a # half-hour of installs worth caching. COPY . . diff --git a/eval/client/benchmark.py b/eval/client/benchmark.py index 41873b5..37b8c95 100644 --- a/eval/client/benchmark.py +++ b/eval/client/benchmark.py @@ -244,6 +244,9 @@ def main() -> int: if done or trunc: obs, _info = env.reset() + if args.backend == "vla-cpp" and inner is not None: + inner._last_response = None + # The PyTorch server times select_action internally (see # pytorch_ref/utils/service.py). Drop the samples accumulated during warmup # so the compiled variants aren't charged for their first-call compilation. @@ -278,6 +281,7 @@ def main() -> int: if args.backend == "vla-cpp" and inner is not None: r = inner._last_response if r is not None: + inner._last_response = None server_latencies.append({ "total": r.latency_ms_total, "vision": r.latency_ms_vision, diff --git a/eval/client/run_ALOHA_client_direct.py b/eval/client/run_ALOHA_client_direct.py index cb416f7..fef1d03 100644 --- a/eval/client/run_ALOHA_client_direct.py +++ b/eval/client/run_ALOHA_client_direct.py @@ -419,6 +419,7 @@ def __init__(self, args: argparse.Namespace): self._async_lock = Lock() self._async_chunk = None # prefetched raw chunk (HORIZON × action_dim) self._async_obs = None # obs snapshot captured at trigger time + self._async_gen = 0 self._async_err: Exception | None = None self._async_cached = None # chunk ready for next _run_inference call self._async_worker = Thread(target=self._async_infer_worker, daemon=True) @@ -498,7 +499,7 @@ def _async_infer_worker(self): self._async_trigger.clear() with self._async_lock: - obs_snap = self._async_obs + gen, obs_snap = self._async_gen, self._async_obs if obs_snap is None: continue @@ -513,6 +514,8 @@ def _async_infer_worker(self): err = e with self._async_lock: + if gen != self._async_gen: + continue self._async_chunk = chunk self._async_err = err self._async_ready.set() @@ -540,6 +543,7 @@ def _trigger_next_infer(self, predicted_left=None, predicted_right=None): right_state, ) with self._async_lock: + self._async_gen += 1 self._async_obs = obs_snap self._async_chunk = None self._async_err = None @@ -548,17 +552,12 @@ def _trigger_next_infer(self, predicted_left=None, predicted_right=None): def _collect_prefetched(self, timeout: float = 2.0): """ - Wait for the prefetched chunk. On timeout, clear the cache so the - next _run_inference falls back to a fresh synchronous request. + Wait for the prefetched chunk. Returns the raw chunk or raises on error. """ if not self._async_ready.wait(timeout=timeout): - self.log.warning( - f"async prefetch timed out after {timeout:.1f}s - " - "next call will re-trigger synchronously" - ) - self._async_cached = None - raise TimeoutError("async inference timed out") + self.log.warning(f"async prefetch slower than {timeout:.1f}s, waiting") + self._async_ready.wait() with self._async_lock: err = self._async_err chunk = self._async_chunk @@ -634,7 +633,7 @@ def _run_inference_async(self): self._async_cached = None self.log.debug("async: using prefetched chunk") else: - # First call or after a timeout fallback: trigger and wait. + # First call or after a failed prefetch: trigger and wait. with self.lock: if self.front_rgb is None or self.wrist_left_rgb is None or self.left_state is None: self._log_waiting(self.front_rgb is not None, diff --git a/eval/client/run_sim_client_direct.py b/eval/client/run_sim_client_direct.py index 0bb62f5..7f3f878 100644 --- a/eval/client/run_sim_client_direct.py +++ b/eval/client/run_sim_client_direct.py @@ -144,6 +144,7 @@ success_count, inference_times = 0.0, [] skipped = 0 + episodes = [] for episode in range(args.n_episodes): print(f"*** Episode {episode + 1}/{args.n_episodes}") @@ -184,6 +185,8 @@ if episode_aborted: skipped += 1 + episodes.append((episode, 0 if episode_aborted else int(bool(info.get("is_success", 0))), + step_id, int(episode_aborted))) env.close() counted = max(1, args.n_episodes - skipped) @@ -196,6 +199,9 @@ f.write(f"Success rate: {success_count / counted:.2%} ({int(success_count)}/{counted})\n") f.write(f"Skipped (terminated mid-step): {skipped}/{args.n_episodes}\n") f.write(f"Average inference time per step: {avg_inf_ms} ms\n") + with open(output_dir / "episodes.csv", "w") as f: + f.write("episode,success,steps,aborted\n") + f.writelines(f"{e},{s},{n},{a}\n" for e, s, n, a in episodes) print("*** All episodes completed.") print(f"- Success rate: {success_count / counted:.2%} ({int(success_count)}/{counted})") diff --git a/eval/client/vla_cpp_client.py b/eval/client/vla_cpp_client.py index b295905..382cd32 100644 --- a/eval/client/vla_cpp_client.py +++ b/eval/client/vla_cpp_client.py @@ -129,6 +129,12 @@ def _resize_with_pad(img_chw: np.ndarray, target_h: int, target_w: int, t = F.pad(t, (pad_w, 0, pad_h, 0), value=pad_value) return t.squeeze(0).numpy() +def _minmax_norm(x: np.ndarray, lo: np.ndarray, hi: np.ndarray, mask: np.ndarray) -> np.ndarray: + + out = np.zeros_like(x, dtype=np.float32) + out[..., mask] = 2.0 * (x[..., mask] - lo[mask]) / (hi[mask] - lo[mask]) - 1.0 + return out + class VlaCppClient: DEFAULT_RECV_TIMEOUT_MS = 30_000 @@ -172,6 +178,8 @@ def __init__( self.sock = self.ctx.socket(zmq.REQ) self.sock.setsockopt(zmq.LINGER, 0) self.sock.setsockopt(zmq.RCVTIMEO, recv_timeout_ms) + self.sock.setsockopt(zmq.REQ_RELAXED, 1) + self.sock.setsockopt(zmq.REQ_CORRELATE, 1) self.sock.connect(vla_addr) print(f"vla-cpp-direct[arch={arch}]: connected to {vla_addr}", flush=True) @@ -193,7 +201,9 @@ def __init__( self.image_keys = list(image_keys) self.max_length = max_length self._step = 0 + self._episode = 0 self._last_response = None + self._noise_len = None if n_action_steps < 1: raise ValueError(f"n_action_steps must be >= 1, got {n_action_steps}") @@ -202,7 +212,7 @@ def __init__( self._bitvla_proprio_norm = None self._bitvla_unnorm_key = None - if arch == "bitvla": + if arch in ("bitvla", "vla_adapter"): if stats_json: stats_path = Path(stats_json) elif (Path(tokenizer_name) / "dataset_statistics.json").exists(): @@ -214,10 +224,12 @@ def __init__( stats_path = Path(hf_hub_download(tokenizer_name, "dataset_statistics.json")) if not stats_path.exists(): raise FileNotFoundError( - f"BitVLA dataset_statistics.json not found at {stats_path}. " + f"{arch} dataset_statistics.json not found at {stats_path}. " f"Pass --stats-json or point --tokenizer at a ckpt dir that has it.") blob = json.loads(stats_path.read_text()) key = bitvla_unnorm_key + if key is None and arch == "vla_adapter": + key = os.environ.get("VLA_ADAPTER_UNNORM_KEY") if key is None: if len(blob) != 1: raise ValueError( @@ -236,7 +248,7 @@ def _norm(x: np.ndarray, q01=q01, q99=q99, mask=mask) -> np.ndarray: out = np.where(mask, 2.0 * (y - q01) / (q99 - q01 + 1e-8) - 1.0, y) return np.clip(out, -1.0, 1.0).astype(np.float32) self._bitvla_proprio_norm = _norm - print(f"vla-cpp-direct[arch=bitvla]: proprio normalizer " + print(f"vla-cpp-direct[arch={arch}]: proprio normalizer " f"BOUNDS_Q99 via {stats_path}::{key}.proprio", flush=True) self._oft_proprio_norm = None @@ -381,6 +393,13 @@ def _action_unnorm(chunk, mn=action_min, mx=action_max): q01 = self._gr00t_quantile(action_stats, modalities, "q01") q99 = self._gr00t_quantile(action_stats, modalities, "q99") act_dim = int(q01.size) + state_stats = blob[key]["state"] + state_keys, state_dims = self._gr00t_modality_layout(state_stats) + state_cols = {} + s_off = 0 + for m, dim in zip(state_keys, state_dims): + state_cols[m] = slice(s_off, s_off + dim) + s_off += dim # Checkpoints trained with use_relative_action predict, for the # modalities listed in meta/relative_stats.json, the offset from the @@ -401,7 +420,7 @@ def _action_unnorm(chunk, mn=action_min, mx=action_max): horizon = min(len(rel_stats[m]["min"]) for m in rel_names) q01_t = np.tile(q01, (horizon, 1)).astype(np.float32) q99_t = np.tile(q99, (horizon, 1)).astype(np.float32) - is_rel = np.zeros(act_dim, dtype=bool) + rel_cols = [] off = 0 for m, dim in zip(modalities, mod_dims): if m in rel_stats: @@ -419,13 +438,18 @@ def _action_unnorm(chunk, mn=action_min, mx=action_max): raise ValueError( f"relative stats for {m!r} are {a.shape[1]}-wide, " f"statistics say {dim}") + sc = state_cols.get(m) + if sc is None or sc.stop - sc.start != dim: + raise ValueError( + f"relative modality {m!r} needs a {dim}-wide state.{m}, " + f"state statistics have {list(zip(state_keys, state_dims))}") q01_t[:, off:off + dim] = a q99_t[:, off:off + dim] = b - is_rel[off:off + dim] = True + rel_cols.append((slice(off, off + dim), sc)) off += dim rng_t = (q99_t - q01_t).astype(np.float32) - def _unnorm(chunk_132, q01_t=q01_t, rng_t=rng_t, is_rel=is_rel, + def _unnorm(chunk_132, q01_t=q01_t, rng_t=rng_t, rel_cols=tuple(rel_cols), act_dim=act_dim, horizon=horizon): n = min(len(chunk_132), horizon) norm = np.clip(chunk_132[:n, :act_dim].astype(np.float32), -1.0, 1.0) @@ -434,7 +458,9 @@ def _unnorm(chunk_132, q01_t=q01_t, rng_t=rng_t, is_rel=is_rel, if ref is None: raise RuntimeError("relative actions need the observation state; " "none was recorded for this request") - raw[:, is_rel] += np.asarray(ref, dtype=np.float32)[:act_dim][is_rel] + ref = np.asarray(ref, dtype=np.float32) + for a_cols, s_cols in rel_cols: + raw[:, a_cols] += ref[s_cols] return raw.astype(np.float32) else: rng = (q99 - q01).astype(np.float32) @@ -456,17 +482,13 @@ def _unnorm(chunk_132: np.ndarray, q01=q01, q99=q99, rng=rng, f"relative={rel_names or 'none'}]", flush=True) - state_stats = blob[key]["state"] - state_keys, state_dims = self._gr00t_modality_layout(state_stats) self._gr00t_state_keys = tuple(state_keys) self._gr00t_state_dims = tuple(state_dims) s_q01 = self._gr00t_quantile(state_stats, state_keys, "q01") s_q99 = self._gr00t_quantile(state_stats, state_keys, "q99") - s_rng = (s_q99 - s_q01).astype(np.float32) - def _state_norm(state_8d: np.ndarray, q01=s_q01, q99=s_q99, rng=s_rng) -> np.ndarray: - - norm = 2.0 * (state_8d - q01) / np.where(rng > 1e-8, rng, 1.0) - 1.0 - return np.clip(norm, -1.0, 1.0).astype(np.float32) + def _state_norm(state_8d: np.ndarray, q01=s_q01, q99=s_q99, + mask=~np.isclose(s_q99, s_q01)) -> np.ndarray: + return np.clip(_minmax_norm(state_8d, q01, q99, mask), -1.0, 1.0) self._gr00t_state_norm = _state_norm print(f"vla-cpp-direct[arch=gr00t_n1_7]: state normalizer " f"(q01/q99 + clip) via {stats_path}::{key}.state " @@ -522,10 +544,9 @@ def _unnorm_n16(chunk_full: np.ndarray, mn=a_min, mx=a_max, rng=a_rng) -> np.nda s_min_parts.append(mn); s_max_parts.append(mx) s_min = np.concatenate(s_min_parts) s_max = np.concatenate(s_max_parts) - s_rng = (s_max - s_min).astype(np.float32) - def _state_norm_n16(state_8d: np.ndarray, mn=s_min, mx=s_max, rng=s_rng) -> np.ndarray: - norm = 2.0 * (state_8d - mn) / np.where(rng > 1e-8, rng, 1.0) - 1.0 - return np.clip(norm, -1.0, 1.0).astype(np.float32) + def _state_norm_n16(state_8d: np.ndarray, mn=s_min, mx=s_max, + mask=~np.isclose(s_max, s_min)) -> np.ndarray: + return np.clip(_minmax_norm(state_8d, mn, mx, mask), -1.0, 1.0) self._gr00t_state_norm = _state_norm_n16 print(f"vla-cpp-direct[arch=gr00t_n1_6]: state normalizer " f"(min/max + clip) via {stats_path}::{key}.state " @@ -564,10 +585,8 @@ def _unnorm_n15(chunk_full, mn=a_min, mx=a_max, rng=a_rng): f"[min={a_min.tolist()}, max={a_max.tolist()}]", flush=True) s_min = np.asarray(blob[key]["state"]["min"], dtype=np.float32) s_max = np.asarray(blob[key]["state"]["max"], dtype=np.float32) - s_rng = (s_max - s_min).astype(np.float32) - def _state_norm_n15(state_vec, mn=s_min, mx=s_max, rng=s_rng): - - return (2.0 * (state_vec - mn) / np.where(rng > 1e-8, rng, 1.0) - 1.0).astype(np.float32) + def _state_norm_n15(state_vec, mn=s_min, mx=s_max, mask=s_min != s_max): + return _minmax_norm(state_vec, mn, mx, mask) self._gr00t_state_norm = _state_norm_n15 print(f"vla-cpp-direct[arch=gr00t_n1_5]: state normalizer " f"(flat min/max, no clip) via {stats_path}::{key}.state " @@ -627,6 +646,8 @@ def ping(self) -> bool: def reset(self) -> None: self._action_queue.clear() + self._episode += 1 + self._step = 0 def get_action(self, observations: dict[str, Any]) -> np.ndarray: @@ -706,6 +727,7 @@ def _predict_chunk(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(lang.tolist()) req.state.extend(state_padded.tolist()) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() resp = self.pb.PredictResponse() @@ -856,6 +878,7 @@ def _predict_chunk_vla_jepa(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(int(t) for t in lang) req.state.extend(float(x) for x in state_padded) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) resp = self.pb.PredictResponse() resp.ParseFromString(self.sock.recv()) @@ -916,6 +939,7 @@ def _predict_chunk_pi05(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(lang.tolist()) req.state.extend([0.0] * self.max_state_dim) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() resp = self.pb.PredictResponse() @@ -931,9 +955,12 @@ def _predict_chunk_pi05(self, observations: dict[str, Any]) -> np.ndarray: _EVO1_IMG_CTX = "" _EVO1_NUM_IMAGE_TOKEN = 256 _EVO1_MAX_TEXT_LENGTH = 1024 - _EVO1_NOISE_LEN = 50 * 24 # horizon * per_action_dim + _FIXED_NOISE_ARCHS = { + "smolvla", "pi0", "pi05", "evo1", "gr00t_n1_5", "gr00t_n1_6", "gr00t_n1_7", + "vla_jepa", "octo", + } - def _maybe_add_fixed_noise(self, req, n: int | None) -> None: + def _maybe_add_fixed_noise(self, req) -> None: """Attach a reproducible noise vector when VLA_FIXED_NOISE_SEED is set. Without it the server draws flow-matching noise from a clock-seeded RNG, @@ -942,14 +969,25 @@ def _maybe_add_fixed_noise(self, req, n: int | None) -> None: verifiable: same inputs plus same noise must give the same actions. """ seed = os.environ.get("VLA_FIXED_NOISE_SEED") - if seed is None or not n: + if seed is None or self.arch not in self._FIXED_NOISE_ARCHS: return + if self._noise_len is None: + self.sock.send(req.SerializeToString()) + r = self.pb.PredictResponse() + r.ParseFromString(self.sock.recv()) + if r.error: + raise RuntimeError(f"vla-server error: {r.error}") + self._noise_len = r.chunk_size * r.action_dim + n = self._noise_len # Vary per step but reproducibly, so a replay of the same episode sends # the same sequence of noise vectors. - rng = np.random.default_rng(int(seed) + self._step) + rng = np.random.default_rng([int(seed), self._episode, self._step]) # Evo-1 is trained on uniform[-1,1]; matching that keeps the check in # the distribution the model actually sees. - req.noise.extend(rng.uniform(-1.0, 1.0, size=n).astype(np.float32).tolist()) + if self.arch == "evo1": + req.noise.extend(rng.uniform(-1.0, 1.0, size=n).astype(np.float32).tolist()) + else: + req.noise.extend(rng.standard_normal(n, dtype=np.float32).tolist()) def _predict_chunk_evo1(self, observations: dict[str, Any]) -> np.ndarray: @@ -1028,7 +1066,7 @@ def _predict_chunk_evo1(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(input_ids_full[:n_real].tolist()) req.state.extend(state_padded.tolist()) req.attention_mask.extend(attn_mask.tolist()) - self._maybe_add_fixed_noise(req, self._EVO1_NOISE_LEN) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() @@ -1081,6 +1119,7 @@ def _predict_chunk_octo(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(input_ids.tolist()) req.attention_mask.extend(attn_mask.tolist()) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() resp = self.pb.PredictResponse() @@ -1094,33 +1133,31 @@ def _predict_chunk_octo(self, observations: dict[str, Any]) -> np.ndarray: return (np.array(resp.action_chunk, dtype=np.float32) .reshape(resp.chunk_size, resp.action_dim)) + def _oft_image(self, observations: dict[str, Any], key: str) -> np.ndarray: + if key not in observations: + raise KeyError(f"image key '{key}' missing; got {list(observations.keys())}") + img = observations[key] + if isinstance(img, torch.Tensor): + img = img.numpy() + img = np.asarray(img, dtype=np.float32) + if img.ndim != 3 or img.shape[0] != 3: + raise ValueError(f"{key}: expected CHW float [3, H, W], got {img.shape}") + img_u8 = np.clip(np.transpose(img, (1, 2, 0)) * 255.0 + 0.5, 0, 255).astype(np.uint8) + if img_u8.shape[0] != self.image_size or img_u8.shape[1] != self.image_size: + img_u8 = np.array(Image.fromarray(img_u8, mode="RGB").resize( + (self.image_size, self.image_size), resample=Image.LANCZOS), dtype=np.uint8) + h, w = img_u8.shape[:2] + s = 0.9 ** 0.5 + new_h, new_w = int(round(h * s)), int(round(w * s)) + off_h, off_w = (h - new_h) // 2, (w - new_w) // 2 + cropped = img_u8[off_h:off_h + new_h, off_w:off_w + new_w] + img_u8 = np.array(Image.fromarray(cropped, mode="RGB").resize( + (w, h), resample=Image.BILINEAR), dtype=np.uint8) + return np.ascontiguousarray(img_u8, dtype=np.uint8) + def _predict_chunk_bitvla(self, observations: dict[str, Any]) -> np.ndarray: - images_u8: list[np.ndarray] = [] - for key in self.image_keys[:BITVLA_N_VIEWS]: - if key not in observations: - raise KeyError(f"image key '{key}' missing; got {list(observations.keys())}") - img = observations[key] - if isinstance(img, torch.Tensor): - img = img.numpy() - img = np.asarray(img, dtype=np.float32) - if img.ndim != 3 or img.shape[0] != 3: - raise ValueError(f"{key}: expected CHW float [3, H, W], got {img.shape}") - img_u8 = np.clip(np.transpose(img, (1, 2, 0)) * 255.0 + 0.5, 0, 255).astype(np.uint8) - - if img_u8.shape[0] != self.image_size or img_u8.shape[1] != self.image_size: - img_u8 = np.array(Image.fromarray(img_u8, mode="RGB").resize( - (self.image_size, self.image_size), resample=Image.LANCZOS), - dtype=np.uint8) - - h, w = img_u8.shape[:2] - s = 0.9 ** 0.5 - new_h, new_w = int(round(h * s)), int(round(w * s)) - off_h, off_w = (h - new_h) // 2, (w - new_w) // 2 - cropped = img_u8[off_h:off_h + new_h, off_w:off_w + new_w] - img_u8 = np.array(Image.fromarray(cropped, mode="RGB").resize( - (w, h), resample=Image.BILINEAR), dtype=np.uint8) - images_u8.append(np.ascontiguousarray(img_u8, dtype=np.uint8)) + images_u8 = [self._oft_image(observations, k) for k in self.image_keys[:BITVLA_N_VIEWS]] s = observations["observation.state"] if isinstance(s, torch.Tensor): @@ -1167,33 +1204,13 @@ def _predict_chunk_bitvla(self, observations: dict[str, Any]) -> np.ndarray: return chunk def _predict_chunk_vla_adapter(self, observations: dict[str, Any]) -> np.ndarray: - images_u8: list[np.ndarray] = [] - for key in self.image_keys[:VLA_ADAPTER_N_VIEWS]: - if key not in observations: - raise KeyError(f"image key '{key}' missing; got {list(observations.keys())}") - img = observations[key] - if isinstance(img, torch.Tensor): - img = img.numpy() - img = np.asarray(img, dtype=np.float32) - if img.ndim != 3 or img.shape[0] != 3: - raise ValueError(f"{key}: expected CHW float [3, H, W], got {img.shape}") - img_u8 = np.clip(np.transpose(img, (1, 2, 0)) * 255.0 + 0.5, 0, 255).astype(np.uint8) - if img_u8.shape[0] != self.image_size or img_u8.shape[1] != self.image_size: - img_u8 = np.array(Image.fromarray(img_u8, mode="RGB").resize( - (self.image_size, self.image_size), resample=Image.LANCZOS), dtype=np.uint8) - h, w = img_u8.shape[:2] - s = 0.9 ** 0.5 - new_h, new_w = int(round(h * s)), int(round(w * s)) - off_h, off_w = (h - new_h) // 2, (w - new_w) // 2 - cropped = img_u8[off_h:off_h + new_h, off_w:off_w + new_w] - img_u8 = np.array(Image.fromarray(cropped, mode="RGB").resize( - (w, h), resample=Image.BILINEAR), dtype=np.uint8) - images_u8.append(np.ascontiguousarray(img_u8, dtype=np.uint8)) + images_u8 = [self._oft_image(observations, k) for k in self.image_keys[:VLA_ADAPTER_N_VIEWS]] st = observations["observation.state"] if isinstance(st, torch.Tensor): st = st.numpy() st = np.asarray(st, dtype=np.float32).reshape(-1)[:8] + st = self._bitvla_proprio_norm(st) task = observations.get("task", "") if isinstance(task, bytes): @@ -1227,28 +1244,7 @@ def _predict_chunk_vla_adapter(self, observations: dict[str, Any]) -> np.ndarray return chunk def _predict_chunk_openvla_oft(self, observations: dict[str, Any]) -> np.ndarray: - images_u8: list[np.ndarray] = [] - for key in self.image_keys[:OPENVLA_OFT_N_VIEWS]: - if key not in observations: - raise KeyError(f"image key '{key}' missing; got {list(observations.keys())}") - img = observations[key] - if isinstance(img, torch.Tensor): - img = img.numpy() - img = np.asarray(img, dtype=np.float32) - if img.ndim != 3 or img.shape[0] != 3: - raise ValueError(f"{key}: expected CHW float [3, H, W], got {img.shape}") - img_u8 = np.clip(np.transpose(img, (1, 2, 0)) * 255.0 + 0.5, 0, 255).astype(np.uint8) - if img_u8.shape[0] != self.image_size or img_u8.shape[1] != self.image_size: - img_u8 = np.array(Image.fromarray(img_u8, mode="RGB").resize( - (self.image_size, self.image_size), resample=Image.LANCZOS), dtype=np.uint8) - h, w = img_u8.shape[:2] - s = 0.9 ** 0.5 - new_h, new_w = int(round(h * s)), int(round(w * s)) - off_h, off_w = (h - new_h) // 2, (w - new_w) // 2 - cropped = img_u8[off_h:off_h + new_h, off_w:off_w + new_w] - img_u8 = np.array(Image.fromarray(cropped, mode="RGB").resize( - (w, h), resample=Image.BILINEAR), dtype=np.uint8) - images_u8.append(np.ascontiguousarray(img_u8, dtype=np.uint8)) + images_u8 = [self._oft_image(observations, k) for k in self.image_keys[:OPENVLA_OFT_N_VIEWS]] st = observations["observation.state"] if isinstance(st, torch.Tensor): @@ -1452,6 +1448,7 @@ def _predict_chunk_gr00t_n1_7(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(int(t) for t in lang) req.state.extend(float(x) for x in state_padded) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() resp = self.pb.PredictResponse() @@ -1572,6 +1569,7 @@ def _predict_chunk_gr00t_n1_6(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(int(t) for t in lang) req.state.extend(float(x) for x in state_padded) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() resp = self.pb.PredictResponse() @@ -1658,6 +1656,7 @@ def _predict_chunk_gr00t_n1_5(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(int(t) for t in lang) req.state.extend(float(x) for x in state_padded) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() resp = self.pb.PredictResponse() @@ -1759,11 +1758,7 @@ def get_action(self, observation: dict[str, Any]) -> dict[str, np.ndarray]: def _state_norm(self, sv: np.ndarray) -> np.ndarray: mn, mx = self._s_min, self._s_max - rng = mx - mn - mask = ~np.isclose(mx, mn) - out = np.zeros_like(sv) - out[mask] = 2.0 * (sv[mask] - mn[mask]) / rng[mask] - 1.0 - return np.clip(out, -1.0, 1.0).astype(np.float32) + return np.clip(_minmax_norm(sv, mn, mx, ~np.isclose(mx, mn)), -1.0, 1.0) def _decode_action(self, chunk: np.ndarray) -> np.ndarray: diff --git a/eval/docker-compose.yml b/eval/docker-compose.yml index 96a2a01..df0d1e2 100644 --- a/eval/docker-compose.yml +++ b/eval/docker-compose.yml @@ -43,9 +43,7 @@ services: - CUDA_VISIBLE_DEVICES=0 ports: - "5555:5555" - # For SmolVLA the single GGUF is enough. - # For π0 pass mmproj + model GGUF. - # For baked-in-vision models (BitVLA, GR00T, Evo-1) omit mmproj. + # Every arch ships one GGUF with its vision tower; no mmproj is needed. command: - --bind - tcp://*:5555 diff --git a/eval/reports/report-agx-orin.md b/eval/reports/report-agx-orin.md deleted file mode 100644 index 277fe35..0000000 --- a/eval/reports/report-agx-orin.md +++ /dev/null @@ -1,240 +0,0 @@ -# LIBERO sweep report - `aggregate_report` - -- Sweep root: `aggregate_report` -- Suite: `libero_object` (tasks 0..9) -- Generated: 2026-05-25T08:32:02 - -## Reproducibility - -Generated by `scripts/print_versions.sh` on `2026-05-25T01:32:02Z`. - -### vla.cpp - -- repo HEAD: `8659f464e63a9cc7f7d1a79e007894fd14ba13ec` (`8659f46-dirty`) -- patch `patches/llama.cpp-vla.patch` sha256: `6210a07bc756ed560e692d5c8c42c4136e715274359fa67826e4722655931758` - -### llama.cpp (third_party) - -- HEAD: `846262d7875dcabf502a150fa3d7b9c770dde7eb` (`b9016-dirty`) -- expected pinned tag (from `patches/patch.sh`): `b9016` - -### GGUF checksums - -_(pass GGUF paths as positional args to include their sha256s)_ - -### Host - -``` -uname: Linux 5.15.148-tegra #1 SMP PREEMPT Wed Apr 2 04:50:13 UTC 2025 aarch64 aarch64 aarch64 GNU/Linux -os: Ubuntu 22.04.5 LTS -nvidia: Orin (nvgpu), 540.4.0 - [N/A], 8.7 -cmake: cmake version 3.22.1 -g++: g++ (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0 -python3: Python 3.10.12 -uv: uv 0.11.14 (aarch64-unknown-linux-gnu) -``` - -### Python venv pins - -**LIBERO venv** (`/home/modelopt/khanh/vla.cpp/eval/sim/libero/libero_uv/.venv`): -``` -(venv not found at /home/modelopt/khanh/vla.cpp/eval/sim/libero/libero_uv/.venv) -``` - -### Determinism - -- LIBERO seeds: see `eval/client/run_libero_eval.py --seed` (default in that script). -- Server-side denoising noise: client-supplied if `PredictRequest.noise` is set; - otherwise sampled from the per-arch RNG seeded at model load (not currently - exposed on the CLI - set `PredictRequest.noise` from the client for bit-exact - reruns). - -## Success rate & client-side inference time - -- **SR** counts terminated episodes as failures: `successes / n_episodes`. -- **client/step** - wall-time per env step (amortized over chunk replay; matches `Average inference time per step` in each `summary.txt`). -- **client/call** = `client/step × n_action_steps` - wall-time per actual `vla-server` call. Includes client pre/post + ZMQ transport (TCP loopback) + server compute. - -| Model | n_act | Tasks | Successes | Terminated | SR | client/step (ms) | client/call (ms) | -|---|---:|---:|---:|---:|---:|---:|---:| -| `bitvla` | 8 | 10 | 200/200 | 0/200 | 100.00% | 101.11 | 808.88 | -| `evo1` | 8 | 10 | 191/200 | 0/200 | 95.50% | 131.01 | 1048.12 | -| `gr00t_n1_5` | 16 | 10 | 195/200 | 0/200 | 97.50% | 28.78 | 460.50 | -| `gr00t_n1_6` | 16 | 10 | 180/200 | 0/200 | 90.00% | 26.70 | 427.17 | -| `gr00t_n1_7` | 16 | 10 | 197/200 | 0/200 | 98.50% | 26.84 | 429.36 | -| `pi0` | 32 | 10 | 171/200 | 0/200 | 85.50% | 27.90 | 892.70 | -| `smolvla` | 4 | 10 | 181/200 | 0/200 | 90.50% | 65.41 | 261.64 | - -## Server-side inference breakdown - -Parsed from `_server_logs/.log` lines: - -``` -vla-server: rid=… served=… total=… ms vision=… inf=… other=… -``` - -These are server-side measurements only - they exclude ZMQ transport and client pre/post. `total = vision + inf + other`. - -| Model | Samples | total (ms) | vision | inf | other | -|---|---:|---:|---:|---:|---:| -| `bitvla` | 347 | 717.55 | 123.28 | 594.27 | 0.00 | -| `evo1` | 412 | 996.05 | 696.21 | 240.42 | 59.42 | -| `gr00t_n1_5` | 196 | 409.57 | 129.22 | 244.20 | 36.16 | -| `gr00t_n1_6` | 248 | 367.73 | 147.36 | 178.81 | 41.56 | -| `gr00t_n1_7` | 192 | 367.43 | 145.29 | 168.98 | 53.17 | -| `pi0` | 129 | 580.22 | 177.96 | 368.04 | 34.21 | -| `smolvla` | 880 | 205.95 | 109.44 | 92.85 | 3.66 | - -### Transport + client overhead - -`overhead = client/call − server total` - time spent outside vla-server (ZMQ over loopback + client preprocessing + protobuf round-trip). - -| Model | client/call (ms) | server total (ms) | overhead (ms) | -|---|---:|---:|---:| -| `bitvla` | 808.88 | 717.55 | 91.33 | -| `evo1` | 1048.12 | 996.05 | 52.07 | -| `gr00t_n1_5` | 460.50 | 409.57 | 50.92 | -| `gr00t_n1_6` | 427.17 | 367.73 | 59.44 | -| `gr00t_n1_7` | 429.36 | 367.43 | 61.93 | -| `pi0` | 892.70 | 580.22 | 312.48 | -| `smolvla` | 261.64 | 205.95 | 55.70 | - -## Peak memory - -Sampled by the inline `mem_sampler` function in [`eval/run_libero.sh`](../../eval/run_libero.sh) while `vla-server` was alive: - -- **Peak VRAM** - max of per-PID `used_memory` from `nvidia-smi --query-compute-apps`, polled every 1s. `(no GPU)` on Tegra/Jetson, which doesn't support that query. -- **Peak RAM** - `VmHWM` from `/proc//status` (kernel-tracked high-water mark of resident memory). Host only - does **not** include the iGPU's unified-memory allocations. -- **Peak sys RAM** / **sys Δ** - peak system-wide used RAM (`MemTotal - MemAvailable`) and its rise over the sampler-start baseline. On Tegra (unified memory) this is the only metric that captures the iGPU weights VRAM/VmHWM miss; the Δ is an upper bound on the server's footprint (a co-resident client/sim is included). - -| Model | Peak VRAM (MiB) | Peak RAM (MiB) | Peak sys RAM (MiB) | sys Δ (MiB) | Samples | -|---|---:|---:|---:|---:|---:| -| `bitvla` | (no GPU) | 1148.8 | 14699.1 | 9753.9 | 4410 | -| `evo1` | (no GPU) | 637.5 | n/a | n/a | 5939 | -| `gr00t_n1_5` | (no GPU) | 1331.3 | n/a | n/a | 2637 | -| `gr00t_n1_6` | (no GPU) | 1340.5 | n/a | n/a | 3002 | -| `gr00t_n1_7` | (no GPU) | 1316.5 | n/a | n/a | 2560 | -| `pi0` | (no GPU) | 640.4 | n/a | n/a | 3108 | -| `smolvla` | (no GPU) | 689.4 | n/a | n/a | 4247 | - -## Per-task breakdown - -
bitvla - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 20/20 | 0/20 | 100.00% | 102.06 | -| task_1 | 20/20 | 0/20 | 100.00% | 99.86 | -| task_2 | 20/20 | 0/20 | 100.00% | 100.38 | -| task_3 | 20/20 | 0/20 | 100.00% | 101.96 | -| task_4 | 20/20 | 0/20 | 100.00% | 100.31 | -| task_5 | 20/20 | 0/20 | 100.00% | 102.66 | -| task_6 | 20/20 | 0/20 | 100.00% | 101.11 | -| task_7 | 20/20 | 0/20 | 100.00% | 102.15 | -| task_8 | 20/20 | 0/20 | 100.00% | 99.68 | -| task_9 | 20/20 | 0/20 | 100.00% | 100.93 | - -
- -
evo1 - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 20/20 | 0/20 | 100.00% | 130.86 | -| task_1 | 20/20 | 0/20 | 100.00% | 129.26 | -| task_2 | 20/20 | 0/20 | 100.00% | 132.73 | -| task_3 | 20/20 | 0/20 | 100.00% | 130.62 | -| task_4 | 19/20 | 0/20 | 95.00% | 130.48 | -| task_5 | 18/20 | 0/20 | 90.00% | 130.99 | -| task_6 | 18/20 | 0/20 | 90.00% | 130.21 | -| task_7 | 20/20 | 0/20 | 100.00% | 131.55 | -| task_8 | 19/20 | 0/20 | 95.00% | 131.64 | -| task_9 | 17/20 | 0/20 | 85.00% | 131.81 | - -
- -
gr00t_n1_5 - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 20/20 | 0/20 | 100.00% | 28.90 | -| task_1 | 20/20 | 0/20 | 100.00% | 28.64 | -| task_2 | 20/20 | 0/20 | 100.00% | 29.32 | -| task_3 | 18/20 | 0/20 | 90.00% | 28.67 | -| task_4 | 19/20 | 0/20 | 95.00% | 28.43 | -| task_5 | 20/20 | 0/20 | 100.00% | 28.64 | -| task_6 | 20/20 | 0/20 | 100.00% | 28.54 | -| task_7 | 19/20 | 0/20 | 95.00% | 28.66 | -| task_8 | 19/20 | 0/20 | 95.00% | 28.86 | -| task_9 | 20/20 | 0/20 | 100.00% | 29.15 | - -
- -
gr00t_n1_6 - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 16/20 | 0/20 | 80.00% | 26.66 | -| task_1 | 20/20 | 0/20 | 100.00% | 27.16 | -| task_2 | 18/20 | 0/20 | 90.00% | 27.19 | -| task_3 | 20/20 | 0/20 | 100.00% | 27.14 | -| task_4 | 15/20 | 0/20 | 75.00% | 26.41 | -| task_5 | 17/20 | 0/20 | 85.00% | 26.67 | -| task_6 | 20/20 | 0/20 | 100.00% | 26.46 | -| task_7 | 14/20 | 0/20 | 70.00% | 25.67 | -| task_8 | 20/20 | 0/20 | 100.00% | 26.93 | -| task_9 | 20/20 | 0/20 | 100.00% | 26.69 | - -
- -
gr00t_n1_7 - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 19/20 | 0/20 | 95.00% | 27.24 | -| task_1 | 20/20 | 0/20 | 100.00% | 26.80 | -| task_2 | 20/20 | 0/20 | 100.00% | 27.94 | -| task_3 | 18/20 | 0/20 | 90.00% | 26.88 | -| task_4 | 20/20 | 0/20 | 100.00% | 26.84 | -| task_5 | 20/20 | 0/20 | 100.00% | 26.41 | -| task_6 | 20/20 | 0/20 | 100.00% | 26.08 | -| task_7 | 20/20 | 0/20 | 100.00% | 26.67 | -| task_8 | 20/20 | 0/20 | 100.00% | 26.31 | -| task_9 | 20/20 | 0/20 | 100.00% | 27.18 | - -
- -
pi0 - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 19/20 | 0/20 | 95.00% | 28.01 | -| task_1 | 19/20 | 0/20 | 95.00% | 27.77 | -| task_2 | 19/20 | 0/20 | 95.00% | 28.37 | -| task_3 | 12/20 | 0/20 | 60.00% | 27.35 | -| task_4 | 20/20 | 0/20 | 100.00% | 29.16 | -| task_5 | 16/20 | 0/20 | 80.00% | 27.55 | -| task_6 | 16/20 | 0/20 | 80.00% | 27.73 | -| task_7 | 13/20 | 0/20 | 65.00% | 28.14 | -| task_8 | 19/20 | 0/20 | 95.00% | 27.36 | -| task_9 | 18/20 | 0/20 | 90.00% | 27.53 | - -
- -
smolvla - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 14/20 | 0/20 | 70.00% | 65.55 | -| task_1 | 20/20 | 0/20 | 100.00% | 66.05 | -| task_2 | 20/20 | 0/20 | 100.00% | 65.97 | -| task_3 | 20/20 | 0/20 | 100.00% | 65.58 | -| task_4 | 17/20 | 0/20 | 85.00% | 65.44 | -| task_5 | 14/20 | 0/20 | 70.00% | 65.26 | -| task_6 | 18/20 | 0/20 | 90.00% | 65.12 | -| task_7 | 18/20 | 0/20 | 90.00% | 65.13 | -| task_8 | 20/20 | 0/20 | 100.00% | 65.22 | -| task_9 | 20/20 | 0/20 | 100.00% | 64.79 | - -
diff --git a/eval/reports/report-orin-nano.md b/eval/reports/report-orin-nano.md deleted file mode 100644 index 07cb49d..0000000 --- a/eval/reports/report-orin-nano.md +++ /dev/null @@ -1,191 +0,0 @@ -# LIBERO sweep report - `libero_object_sweep` - -- Sweep root: `/home/khanh/work/vla.cpp/outputs/libero_object_sweep` -- Suite: `libero_object` (tasks 0..9) -- Generated: 2026-05-24T19:32:17 - -## Reproducibility - -Generated by `scripts/print_versions.sh` on `2026-05-24T12:32:17Z`. - -### vla.cpp - -- repo HEAD: `26ba95495cac4eca20a53e2e09e84bca5b501d1a` (`26ba954`) -- patch `patches/llama.cpp-vla.patch` sha256: `6210a07bc756ed560e692d5c8c42c4136e715274359fa67826e4722655931758` - -### llama.cpp (third_party) - -- HEAD: `846262d7875dcabf502a150fa3d7b9c770dde7eb` (`b9016-dirty`) -- expected pinned tag (from `patches/patch.sh`): `b9016` - -### GGUF checksums - -_(pass GGUF paths as positional args to include their sha256s)_ - -### Host - -``` -uname: Linux 5.15.185-tegra #1 SMP PREEMPT Thu Jan 15 19:24:38 PST 2026 aarch64 aarch64 aarch64 GNU/Linux -os: Ubuntu 22.04.5 LTS -nvidia: Orin (nvgpu), 540.5.0 - [N/A], 8.7 -cmake: cmake version 3.22.1 -g++: g++ (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0 -python3: Python 3.10.12 -uv: uv 0.11.8 (aarch64-unknown-linux-gnu) -``` - -### Determinism - -- LIBERO seeds: see `eval/client/run_libero_eval.py --seed` (default in that script). -- Server-side denoising noise: client-supplied if `PredictRequest.noise` is set; - otherwise sampled from the per-arch RNG seeded at model load (not currently - exposed on the CLI - set `PredictRequest.noise` from the client for bit-exact - reruns). - -## Success rate & client-side inference time - -- **SR** counts terminated episodes as failures: `successes / n_episodes`. -- **client/step** - wall-time per env step (amortized over chunk replay; matches `Average inference time per step` in each `summary.txt`). -- **client/call** = `client/step × n_action_steps` - wall-time per actual `vla-server` call. Includes client pre/post + ZMQ transport (TCP loopback) + server compute. - -| Model | n_act | Tasks | Successes | Terminated | SR | client/step (ms) | client/call (ms) | -|---|---:|---:|---:|---:|---:|---:|---:| -| `bitvla` | 8 | 10 | 200/200 | 0/200 | 100.00% | 355.65 | 2845.24 | -| `evo1` | 8 | 10 | 195/200 | 0/200 | 97.50% | 458.84 | 3670.70 | -| `gr00t_n1_5` | 16 | 10 | 192/200 | 0/200 | 96.00% | 84.76 | 1356.21 | -| `pi0` | 50 | 10 | 161/200 | 0/200 | 80.50% | 39.10 | 1954.90 | -| `smolvla` | 4 | 10 | 176/200 | 0/200 | 88.00% | 141.81 | 567.25 | - -## Server-side inference breakdown - -Parsed from `_server_logs/.log` lines: - -``` -vla-server: rid=… served=… total=… ms vision=… inf=… other=… -``` - -These are server-side measurements only - they exclude ZMQ transport and client pre/post. `total = vision + inf + other`. - -| Model | Samples | total (ms) | vision | inf | other | -|---|---:|---:|---:|---:|---:| -| `bitvla` | 344 | 2710.95 | 416.81 | 2294.13 | 0.00 | -| `evo1` | 396 | 3551.56 | 2667.63 | 848.69 | 35.24 | -| `gr00t_n1_5` | 204 | 1182.67 | 299.38 | 860.90 | 22.39 | -| `pi0` | 89 | 1484.75 | 300.92 | 1164.73 | 19.10 | -| `smolvla` | 951 | 509.60 | 284.03 | 222.34 | 3.23 | - -### Transport + client overhead - -`overhead = client/call − server total` - time spent outside vla-server (ZMQ over loopback + client preprocessing + protobuf round-trip). - -| Model | client/call (ms) | server total (ms) | overhead (ms) | -|---|---:|---:|---:| -| `bitvla` | 2845.24 | 2710.95 | 134.29 | -| `evo1` | 3670.70 | 3551.56 | 119.14 | -| `gr00t_n1_5` | 1356.21 | 1182.67 | 173.54 | -| `pi0` | 1954.90 | 1484.75 | 470.15 | -| `smolvla` | 567.25 | 509.60 | 57.65 | - -## Peak memory - -Sampled by the inline `mem_sampler` function in [`eval/run_libero.sh`](../../eval/run_libero.sh) while `vla-server` was alive: - -- **Peak VRAM** - max of per-PID `used_memory` from `nvidia-smi --query-compute-apps`, polled every 1s. `(no GPU)` on Tegra/Jetson, which doesn't support that query. -- **Peak RAM** - `VmHWM` from `/proc//status` (kernel-tracked high-water mark of resident memory). Host only - does **not** include the iGPU's unified-memory allocations. -- **Peak sys RAM** / **sys Δ** - peak system-wide used RAM (`MemTotal - MemAvailable`) and its rise over the sampler-start baseline. On Tegra (unified memory) this is the only metric that captures the iGPU weights VRAM/VmHWM miss; the Δ is an upper bound on the server's footprint (a co-resident client/sim is included). - -| Model | Peak VRAM (MiB) | Peak RAM (MiB) | Peak sys RAM (MiB) | sys Δ (MiB) | Samples | -|---|---:|---:|---:|---:|---:| -| `bitvla` | (no GPU) | 2199.0 | 6503.6 | 4532.7 | 11550 | -| `evo1` | (no GPU) | 2135.0 | 6758.8 | 4873.9 | 16138 | -| `gr00t_n1_5` | (no GPU) | 5974.9 | 6338.6 | 4399.4 | 4531 | -| `pi0` | (no GPU) | 6067.7 | 7013.6 | 5067.8 | 2966 | -| `smolvla` | (no GPU) | 2031.2 | 6573.2 | 4656.6 | 7775 | - -## Per-task breakdown - -
bitvla - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 20/20 | 0/20 | 100.00% | 352.73 | -| task_1 | 20/20 | 0/20 | 100.00% | 349.59 | -| task_2 | 20/20 | 0/20 | 100.00% | 355.70 | -| task_3 | 20/20 | 0/20 | 100.00% | 360.28 | -| task_4 | 20/20 | 0/20 | 100.00% | 354.17 | -| task_5 | 20/20 | 0/20 | 100.00% | 360.81 | -| task_6 | 20/20 | 0/20 | 100.00% | 357.73 | -| task_7 | 20/20 | 0/20 | 100.00% | 360.99 | -| task_8 | 20/20 | 0/20 | 100.00% | 349.32 | -| task_9 | 20/20 | 0/20 | 100.00% | 355.23 | - -
- -
evo1 - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 20/20 | 0/20 | 100.00% | 459.33 | -| task_1 | 19/20 | 0/20 | 95.00% | 453.84 | -| task_2 | 20/20 | 0/20 | 100.00% | 464.00 | -| task_3 | 19/20 | 0/20 | 95.00% | 455.19 | -| task_4 | 20/20 | 0/20 | 100.00% | 458.94 | -| task_5 | 20/20 | 0/20 | 100.00% | 459.77 | -| task_6 | 20/20 | 0/20 | 100.00% | 458.26 | -| task_7 | 20/20 | 0/20 | 100.00% | 459.92 | -| task_8 | 19/20 | 0/20 | 95.00% | 458.44 | -| task_9 | 18/20 | 0/20 | 90.00% | 460.69 | - -
- -
gr00t_n1_5 - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 20/20 | 0/20 | 100.00% | 83.83 | -| task_1 | 18/20 | 0/20 | 90.00% | 84.94 | -| task_2 | 20/20 | 0/20 | 100.00% | 83.80 | -| task_3 | 17/20 | 0/20 | 85.00% | 84.99 | -| task_4 | 18/20 | 0/20 | 90.00% | 83.89 | -| task_5 | 20/20 | 0/20 | 100.00% | 84.67 | -| task_6 | 20/20 | 0/20 | 100.00% | 85.17 | -| task_7 | 19/20 | 0/20 | 95.00% | 85.45 | -| task_8 | 20/20 | 0/20 | 100.00% | 84.09 | -| task_9 | 20/20 | 0/20 | 100.00% | 86.80 | - -
- -
pi0 - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 17/20 | 0/20 | 85.00% | 39.03 | -| task_1 | 16/20 | 0/20 | 80.00% | 40.72 | -| task_2 | 16/20 | 0/20 | 80.00% | 41.81 | -| task_3 | 14/20 | 0/20 | 70.00% | 38.38 | -| task_4 | 18/20 | 0/20 | 90.00% | 37.24 | -| task_5 | 15/20 | 0/20 | 75.00% | 38.51 | -| task_6 | 16/20 | 0/20 | 80.00% | 39.35 | -| task_7 | 13/20 | 0/20 | 65.00% | 39.50 | -| task_8 | 17/20 | 0/20 | 85.00% | 36.50 | -| task_9 | 19/20 | 0/20 | 95.00% | 39.94 | - -
- -
smolvla - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 17/20 | 0/20 | 85.00% | 142.37 | -| task_1 | 19/20 | 0/20 | 95.00% | 141.66 | -| task_2 | 18/20 | 0/20 | 90.00% | 141.77 | -| task_3 | 17/20 | 0/20 | 85.00% | 141.89 | -| task_4 | 18/20 | 0/20 | 90.00% | 141.90 | -| task_5 | 16/20 | 0/20 | 80.00% | 142.02 | -| task_6 | 17/20 | 0/20 | 85.00% | 141.34 | -| task_7 | 15/20 | 0/20 | 75.00% | 141.66 | -| task_8 | 20/20 | 0/20 | 100.00% | 142.06 | -| task_9 | 19/20 | 0/20 | 95.00% | 141.45 | - -
diff --git a/eval/reports/report-rtx-3060.md b/eval/reports/report-rtx-3060.md deleted file mode 100644 index a8200e8..0000000 --- a/eval/reports/report-rtx-3060.md +++ /dev/null @@ -1,241 +0,0 @@ -# LIBERO sweep report - `aggregate_report` - -- Sweep root: `/home/khanh/work/vla.cpp/outputs/aggregate_report` -- Suite: `libero_object` (tasks 0..9) -- Generated: 2026-05-24T20:33:20 - -## Reproducibility - -Generated by `scripts/print_versions.sh` on `2026-05-24T13:33:20Z`. - -### vla.cpp - -- repo HEAD: `dcc29a37297e6d52f52a28eeb4e42bd4859f18c6` (`dcc29a3`) -- patch `patches/llama.cpp-vla.patch` sha256: `6210a07bc756ed560e692d5c8c42c4136e715274359fa67826e4722655931758` - -### llama.cpp (third_party) - -- HEAD: `846262d7875dcabf502a150fa3d7b9c770dde7eb` (`b9016-dirty`) -- expected pinned tag (from `patches/patch.sh`): `b9016` - -### GGUF checksums - -_(pass GGUF paths as positional args to include their sha256s)_ - -### Host - -``` -uname: Linux 6.8.0-111-generic #111~22.04.1-Ubuntu SMP PREEMPT_DYNAMIC Tue Apr 14 17:13:45 UTC x86_64 x86_64 x86_64 GNU/Linux -os: Ubuntu 22.04.5 LTS -nvidia: NVIDIA GeForce RTX 3060, 580.159.03 - 12288 MiB, 8.6 -nvcc: Cuda compilation tools, release 12.8, V12.8.93 -cmake: cmake version 3.22.1 -g++: g++ (Ubuntu 11.4.0-1ubuntu1~22.04.3) 11.4.0 -python3: Python 3.10.12 -uv: uv 0.11.7 (x86_64-unknown-linux-gnu) -``` - -### Python venv pins - -**LIBERO venv** (`/home/khanh/release/vla.cpp/eval/sim/libero/libero_uv/.venv`): -``` -(venv not found at /home/khanh/release/vla.cpp/eval/sim/libero/libero_uv/.venv) -``` - -### Determinism - -- LIBERO seeds: see `eval/client/run_libero_eval.py --seed` (default in that script). -- Server-side denoising noise: client-supplied if `PredictRequest.noise` is set; - otherwise sampled from the per-arch RNG seeded at model load (not currently - exposed on the CLI - set `PredictRequest.noise` from the client for bit-exact - reruns). - -## Success rate & client-side inference time - -- **SR** counts terminated episodes as failures: `successes / n_episodes`. -- **client/step** - wall-time per env step (amortized over chunk replay; matches `Average inference time per step` in each `summary.txt`). -- **client/call** = `client/step × n_action_steps` - wall-time per actual `vla-server` call. Includes client pre/post + ZMQ transport (TCP loopback) + server compute. - -| Model | n_act | Tasks | Successes | Terminated | SR | client/step (ms) | client/call (ms) | -|---|---:|---:|---:|---:|---:|---:|---:| -| `bitvla` | 8 | 10 | 200/200 | 0/200 | 100.00% | 37.85 | 302.78 | -| `evo1` | 8 | 10 | 189/200 | 0/200 | 94.50% | 63.60 | 508.84 | -| `gr00t_n1_5` | 16 | 10 | 192/200 | 0/200 | 96.00% | 14.17 | 226.69 | -| `gr00t_n1_6` | 16 | 10 | 173/200 | 0/200 | 86.50% | 10.29 | 164.64 | -| `gr00t_n1_7` | 16 | 10 | 196/200 | 0/200 | 98.00% | 10.26 | 164.16 | -| `pi0` | 32 | 10 | 175/200 | 0/200 | 87.50% | 9.74 | 311.68 | -| `smolvla` | 4 | 10 | 181/200 | 0/200 | 90.50% | 28.16 | 112.63 | - -## Server-side inference breakdown - -Parsed from `_server_logs/.log` lines: - -``` -vla-server: rid=… served=… total=… ms vision=… inf=… other=… -``` - -These are server-side measurements only - they exclude ZMQ transport and client pre/post. `total = vision + inf + other`. - -| Model | Samples | total (ms) | vision | inf | other | -|---|---:|---:|---:|---:|---:| -| `bitvla` | 345 | 283.41 | 47.53 | 235.89 | 0.00 | -| `evo1` | 437 | 486.74 | 348.75 | 130.97 | 7.02 | -| `gr00t_n1_5` | 202 | 203.73 | 51.09 | 147.01 | 5.64 | -| `gr00t_n1_6` | 271 | 140.33 | 49.62 | 83.62 | 7.09 | -| `gr00t_n1_7` | 195 | 139.89 | 44.63 | 84.07 | 11.19 | -| `pi0` | 126 | 261.58 | 49.94 | 207.18 | 4.45 | -| `smolvla` | 901 | 99.31 | 42.94 | 54.80 | 1.55 | - -### Transport + client overhead - -`overhead = client/call − server total` - time spent outside vla-server (ZMQ over loopback + client preprocessing + protobuf round-trip). - -| Model | client/call (ms) | server total (ms) | overhead (ms) | -|---|---:|---:|---:| -| `bitvla` | 302.78 | 283.41 | 19.36 | -| `evo1` | 508.84 | 486.74 | 22.10 | -| `gr00t_n1_5` | 226.69 | 203.73 | 22.96 | -| `gr00t_n1_6` | 164.64 | 140.33 | 24.31 | -| `gr00t_n1_7` | 164.16 | 139.89 | 24.27 | -| `pi0` | 311.68 | 261.58 | 50.10 | -| `smolvla` | 112.63 | 99.31 | 13.32 | - -## Peak memory - -Sampled by the inline `mem_sampler` function in [`eval/run_libero.sh`](../../eval/run_libero.sh) while `vla-server` was alive: - -- **Peak VRAM** - max of per-PID `used_memory` from `nvidia-smi --query-compute-apps`, polled every 1s. `(no GPU)` on Tegra/Jetson, which doesn't support that query. -- **Peak RAM** - `VmHWM` from `/proc//status` (kernel-tracked high-water mark of resident memory). Host only - does **not** include the iGPU's unified-memory allocations. -- **Peak sys RAM** / **sys Δ** - peak system-wide used RAM (`MemTotal - MemAvailable`) and its rise over the sampler-start baseline. On Tegra (unified memory) this is the only metric that captures the iGPU weights VRAM/VmHWM miss; the Δ is an upper bound on the server's footprint (a co-resident client/sim is included). - -| Model | Peak VRAM (MiB) | Peak RAM (MiB) | Peak sys RAM (MiB) | sys Δ (MiB) | Samples | -|---|---:|---:|---:|---:|---:| -| `bitvla` | 1312 | 1182.5 | 6446.2 | 4155.2 | 1663 | -| `evo1` | 1564 | 717.1 | n/a | n/a | 2837 | -| `gr00t_n1_5` | 4866 | 1440.6 | n/a | n/a | 1079 | -| `gr00t_n1_6` | 6048 | 1445.6 | n/a | n/a | 1166 | -| `gr00t_n1_7` | 6302 | 1422.2 | n/a | n/a | 934 | -| `pi0` | 5548 | 732.2 | n/a | n/a | 1067 | -| `smolvla` | 1410 | 767.7 | n/a | n/a | 1702 | - -## Per-task breakdown - -
bitvla - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 20/20 | 0/20 | 100.00% | 37.95 | -| task_1 | 20/20 | 0/20 | 100.00% | 37.26 | -| task_2 | 20/20 | 0/20 | 100.00% | 38.06 | -| task_3 | 20/20 | 0/20 | 100.00% | 37.97 | -| task_4 | 20/20 | 0/20 | 100.00% | 37.81 | -| task_5 | 20/20 | 0/20 | 100.00% | 38.81 | -| task_6 | 20/20 | 0/20 | 100.00% | 37.53 | -| task_7 | 20/20 | 0/20 | 100.00% | 37.87 | -| task_8 | 20/20 | 0/20 | 100.00% | 37.36 | -| task_9 | 20/20 | 0/20 | 100.00% | 37.85 | - -
- -
evo1 - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 20/20 | 0/20 | 100.00% | 63.63 | -| task_1 | 19/20 | 0/20 | 95.00% | 63.14 | -| task_2 | 20/20 | 0/20 | 100.00% | 63.73 | -| task_3 | 20/20 | 0/20 | 100.00% | 63.41 | -| task_4 | 19/20 | 0/20 | 95.00% | 63.75 | -| task_5 | 20/20 | 0/20 | 100.00% | 63.51 | -| task_6 | 19/20 | 0/20 | 95.00% | 63.85 | -| task_7 | 20/20 | 0/20 | 100.00% | 63.79 | -| task_8 | 18/20 | 0/20 | 90.00% | 63.82 | -| task_9 | 14/20 | 0/20 | 70.00% | 63.42 | - -
- -
gr00t_n1_5 - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 20/20 | 0/20 | 100.00% | 14.27 | -| task_1 | 20/20 | 0/20 | 100.00% | 14.12 | -| task_2 | 20/20 | 0/20 | 100.00% | 14.39 | -| task_3 | 17/20 | 0/20 | 85.00% | 14.13 | -| task_4 | 19/20 | 0/20 | 95.00% | 14.13 | -| task_5 | 19/20 | 0/20 | 95.00% | 14.04 | -| task_6 | 18/20 | 0/20 | 90.00% | 14.09 | -| task_7 | 19/20 | 0/20 | 95.00% | 14.32 | -| task_8 | 20/20 | 0/20 | 100.00% | 14.02 | -| task_9 | 20/20 | 0/20 | 100.00% | 14.17 | - -
- -
gr00t_n1_6 - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 19/20 | 0/20 | 95.00% | 10.29 | -| task_1 | 20/20 | 0/20 | 100.00% | 10.42 | -| task_2 | 17/20 | 0/20 | 85.00% | 10.24 | -| task_3 | 19/20 | 0/20 | 95.00% | 10.45 | -| task_4 | 14/20 | 0/20 | 70.00% | 10.26 | -| task_5 | 18/20 | 0/20 | 90.00% | 10.28 | -| task_6 | 19/20 | 0/20 | 95.00% | 10.39 | -| task_7 | 9/20 | 0/20 | 45.00% | 10.03 | -| task_8 | 19/20 | 0/20 | 95.00% | 10.25 | -| task_9 | 19/20 | 0/20 | 95.00% | 10.29 | - -
- -
gr00t_n1_7 - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 20/20 | 0/20 | 100.00% | 10.23 | -| task_1 | 20/20 | 0/20 | 100.00% | 10.29 | -| task_2 | 20/20 | 0/20 | 100.00% | 10.47 | -| task_3 | 18/20 | 0/20 | 90.00% | 10.24 | -| task_4 | 20/20 | 0/20 | 100.00% | 10.26 | -| task_5 | 19/20 | 0/20 | 95.00% | 10.30 | -| task_6 | 20/20 | 0/20 | 100.00% | 10.24 | -| task_7 | 20/20 | 0/20 | 100.00% | 10.15 | -| task_8 | 19/20 | 0/20 | 95.00% | 10.11 | -| task_9 | 20/20 | 0/20 | 100.00% | 10.31 | - -
- -
pi0 - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 18/20 | 0/20 | 90.00% | 10.01 | -| task_1 | 18/20 | 0/20 | 90.00% | 9.71 | -| task_2 | 20/20 | 0/20 | 100.00% | 9.81 | -| task_3 | 14/20 | 0/20 | 70.00% | 9.67 | -| task_4 | 19/20 | 0/20 | 95.00% | 9.94 | -| task_5 | 18/20 | 0/20 | 90.00% | 9.44 | -| task_6 | 19/20 | 0/20 | 95.00% | 9.83 | -| task_7 | 17/20 | 0/20 | 85.00% | 9.84 | -| task_8 | 15/20 | 0/20 | 75.00% | 9.42 | -| task_9 | 17/20 | 0/20 | 85.00% | 9.73 | - -
- -
smolvla - -| Task | Successes | Terminated | SR | client/step (ms) | -|---|---:|---:|---:|---:| -| task_0 | 16/20 | 0/20 | 80.00% | 28.21 | -| task_1 | 17/20 | 0/20 | 85.00% | 28.06 | -| task_2 | 19/20 | 0/20 | 95.00% | 28.14 | -| task_3 | 18/20 | 0/20 | 90.00% | 28.12 | -| task_4 | 18/20 | 0/20 | 90.00% | 28.12 | -| task_5 | 15/20 | 0/20 | 75.00% | 28.14 | -| task_6 | 20/20 | 0/20 | 100.00% | 28.18 | -| task_7 | 19/20 | 0/20 | 95.00% | 28.13 | -| task_8 | 20/20 | 0/20 | 100.00% | 28.24 | -| task_9 | 19/20 | 0/20 | 95.00% | 28.23 | - -
diff --git a/eval/run_libero.sh b/eval/run_libero.sh index 5c589ed..6b90d6c 100644 --- a/eval/run_libero.sh +++ b/eval/run_libero.sh @@ -13,7 +13,7 @@ # See the License for the specific language governing permissions and # limitations under the License. -# Run libero_object eval (task-id 0..9, N_EPISODES per task) against every VLA +# Run a LIBERO suite eval (task-id 0..9, N_EPISODES per task) against every VLA # model under MODELS_ROOT. For each model: build vla-server once, launch it, # wait for ready, drive run_sim_client_direct.py from the LIBERO venv, then # stop the server before moving on. @@ -25,21 +25,31 @@ REPO_ROOT="$(cd "${SCRIPT_DIR}/.." && pwd)" usage() { cat < [-o ] [-n ] [-m ] +Usage: $(basename "$0") -i [-o ] [-n ] [-m ] [-s ] -i MODELS_ROOT directory holding the per-model GGUF folders (e.g. $HOME/data/vrfai) [required] -o OUTPUT_ROOT destination for client outputs + server logs - (default: ${REPO_ROOT}/outputs/libero_object_sweep) + (default: ${REPO_ROOT}/outputs/_sweep) -n N_EPISODES episodes per task-id (default: 1) + -s SUITE libero_object | libero_spatial | libero_goal | libero_10 + (default: libero_object). BitVLA and GR00T-N1.7 load the + matching per-suite checkpoint; the others use their one GGUF. -m MODEL which model to run: smol | pi0 | pi05 | bit | evo1 | vla_adapter | openvla_oft | - gr00t_n1_5 | gr00t_n1_6 | gr00t_n1_7 | all + gr00t_n1_5 | gr00t_n1_6 | gr00t_n1_7 | + octo | turbovla | vla_jepa | all (default: all) -h show this help Env overrides: BIND_ADDR, CLIENT_ADDR, BITVLA_TOKENIZER, GR00T_N1_6_TOKENIZER, - GR00T_N1_5_STATS, GR00T_N1_6_STATS, GR00T_N1_7_STATS + GR00T_N1_5_STATS, GR00T_N1_6_STATS, GR00T_N1_7_STATS, + PALIGEMMA_TOKENIZER (pi0/pi05), TURBOVLA_STATS, VLA_JEPA_STATS, + TASK_IDS (default "0 1 2 3 4 5 6 7 8 9"), + SERVER_BIN (prebuilt vla-server; setting it skips the build), + VLA_JEPA_PYTHON (python for the vla_jepa client; it needs LIBERO and + transformers>=5.4, default is the LIBERO venv), + VLA_FIXED_NOISE_SEED (client-side noise, for paired A/B runs) EOF } @@ -47,13 +57,15 @@ MODELS_ROOT="" OUTPUT_ROOT="" N_EPISODES="1" MODEL="all" +TASK_SUITE="libero_object" -while getopts ":i:o:n:m:h" opt; do +while getopts ":i:o:n:m:s:h" opt; do case "${opt}" in i) MODELS_ROOT="${OPTARG}" ;; o) OUTPUT_ROOT="${OPTARG}" ;; n) N_EPISODES="${OPTARG}" ;; m) MODEL="${OPTARG}" ;; + s) TASK_SUITE="${OPTARG}" ;; h) usage; exit 0 ;; \?) echo "ERROR: unknown option -${OPTARG}" >&2; usage >&2; exit 1 ;; :) echo "ERROR: option -${OPTARG} requires an argument" >&2; usage >&2; exit 1 ;; @@ -62,13 +74,19 @@ done shift $((OPTIND - 1)) case "${MODEL}" in - smol|pi0|pi05|bit|evo1|vla_adapter|openvla_oft|gr00t_n1_5|gr00t_n1_6|gr00t_n1_7|all) ;; + smol|pi0|pi05|bit|evo1|vla_adapter|openvla_oft|gr00t_n1_5|gr00t_n1_6|gr00t_n1_7|octo|turbovla|vla_jepa|all) ;; *) - echo "ERROR: -m must be one of: smol | pi0 | pi05 | bit | evo1 | vla_adapter | openvla_oft | gr00t_n1_5 | gr00t_n1_6 | gr00t_n1_7 | all (got '${MODEL}')" >&2 + echo "ERROR: -m must be one of: smol | pi0 | pi05 | bit | evo1 | vla_adapter | openvla_oft | gr00t_n1_5 | gr00t_n1_6 | gr00t_n1_7 | octo | turbovla | vla_jepa | all (got '${MODEL}')" >&2 exit 1 ;; esac +case "${TASK_SUITE}" in + libero_object|libero_spatial|libero_goal|libero_10) ;; + *) echo "ERROR: -s must be one of: libero_object | libero_spatial | libero_goal | libero_10 (got '${TASK_SUITE}')" >&2; exit 1 ;; +esac +SUITE_NAME="${TASK_SUITE#libero_}" # object | spatial | goal | 10 + if [[ -z "${MODELS_ROOT}" ]]; then echo "ERROR: -i is required." >&2 usage >&2 @@ -80,19 +98,24 @@ if [[ ! -d "${MODELS_ROOT}" ]]; then fi MODELS_ROOT="$(cd "${MODELS_ROOT}" && pwd)" -OUTPUT_ROOT="${OUTPUT_ROOT:-${REPO_ROOT}/outputs/libero_object_sweep}" +OUTPUT_ROOT="${OUTPUT_ROOT:-${REPO_ROOT}/outputs/${TASK_SUITE}_sweep}" if ! [[ "${N_EPISODES}" =~ ^[1-9][0-9]*$ ]]; then echo "ERROR: N_EPISODES must be a positive integer (got '${N_EPISODES}')" >&2 exit 1 fi -SERVER_BIN="${REPO_ROOT}/build/vla-server" +if [[ -n "${SERVER_BIN:-}" ]]; then + SKIP_BUILD=1 +fi +SERVER_BIN="${SERVER_BIN:-${REPO_ROOT}/build/vla-server}" +TASK_IDS="${TASK_IDS:-0 1 2 3 4 5 6 7 8 9}" +PALIGEMMA_TOKENIZER="${PALIGEMMA_TOKENIZER:-}" VENV_PY="${REPO_ROOT}/eval/sim/libero/libero_uv/.venv/bin/python" +VLA_JEPA_PYTHON="${VLA_JEPA_PYTHON:-${VENV_PY}}" CLIENT="${REPO_ROOT}/eval/client/run_sim_client_direct.py" BIND_ADDR="${BIND_ADDR:-tcp://*:5555}" CLIENT_ADDR="${CLIENT_ADDR:-tcp://localhost:5555}" -TASK_SUITE="libero_object" # bitvla auto-loads its tokenizer + dataset_statistics.json from the GGUF repo on # the Hub. Optional override (offline): BITVLA_TOKENIZER=/path/to/bitvla-ckpt-dir @@ -115,12 +138,15 @@ N_ACTION_STEPS_PI0="${N_ACTION_STEPS_PI0:-50}" # pi0_libero_finetu N_ACTION_STEPS_PI05="${N_ACTION_STEPS_PI05:-10}" # pi0.5 n_action_steps (chunk 50, 10 denoise steps) N_ACTION_STEPS_VLA_ADAPTER="${N_ACTION_STEPS_VLA_ADAPTER:-8}" # VLA-Adapter action chunk N_ACTION_STEPS_OPENVLA_OFT="${N_ACTION_STEPS_OPENVLA_OFT:-8}" # OpenVLA-OFT parallel 8-step chunk -N_ACTION_STEPS_SMOL="${N_ACTION_STEPS_SMOL:-1}" # SmolVLA golden path (re-predict each step) -N_ACTION_STEPS_EVO1="${N_ACTION_STEPS_EVO1:-8}" # Evo-1 (chunk replay) +N_ACTION_STEPS_SMOL="${N_ACTION_STEPS_SMOL:-10}" # SmolVLA paper ablation: 10 beats 1 (82.8 vs 80.3%) +N_ACTION_STEPS_EVO1="${N_ACTION_STEPS_EVO1:-14}" # Evo-1 upstream LIBERO client (horizon = 14) N_ACTION_STEPS_BIT="${N_ACTION_STEPS_BIT:-8}" # BitVLA NUM_ACTIONS_CHUNK N_ACTION_STEPS_GR00T_N1_5="${N_ACTION_STEPS_GR00T_N1_5:-16}" # N1.5 lerobot closeout (10/10 on libero_object/task_0) N_ACTION_STEPS_GR00T_N1_6="${N_ACTION_STEPS_GR00T_N1_6:-16}" # N1.6 H4 closeout (10/10 on libero_object/task_0) N_ACTION_STEPS_GR00T_N1_7="${N_ACTION_STEPS_GR00T_N1_7:-16}" # N1.7 H4 closeout (10/10 on libero_object/task_0) +N_ACTION_STEPS_OCTO="${N_ACTION_STEPS_OCTO:-4}" +N_ACTION_STEPS_TURBOVLA="${N_ACTION_STEPS_TURBOVLA:-12}" +N_ACTION_STEPS_VLA_JEPA="${N_ACTION_STEPS_VLA_JEPA:-7}" mkdir -p "${OUTPUT_ROOT}" OUTPUT_ROOT="$(cd "${OUTPUT_ROOT}" && pwd)" @@ -132,11 +158,14 @@ echo "[config] MODELS_ROOT=${MODELS_ROOT}" echo "[config] OUTPUT_ROOT=${OUTPUT_ROOT}" echo "[config] N_EPISODES=${N_EPISODES}" echo "[config] MODEL=${MODEL}" +echo "[config] TASK_SUITE=${TASK_SUITE}" +echo "[config] TASK_IDS=${TASK_IDS}" +echo "[config] SERVER_BIN=${SERVER_BIN}" cd "${REPO_ROOT}" if [[ "${SKIP_BUILD:-0}" == "1" ]]; then - echo "[build] skipped (SKIP_BUILD=1)" + echo "[build] skipped (SKIP_BUILD=1 or SERVER_BIN set)" else echo "[build] cmake --build build" cmake --build build -j"$(nproc)" @@ -283,17 +312,27 @@ run_model() { shift 4 local server_args=("$@") local client_extra=() + local client_py="${VENV_PY}" + if [[ "${arch}" == vla_jepa ]]; then + client_py="${VLA_JEPA_PYTHON}" + fi # bitvla auto-loads tokenizer + dataset_statistics.json from the GGUF repo on # the Hub; only pass --tokenizer when BITVLA_TOKENIZER overrides with a local dir. if [[ "${arch}" == "bitvla" && -n "${BITVLA_TOKENIZER}" ]]; then client_extra+=(--tokenizer "${BITVLA_TOKENIZER}") + elif [[ "${arch}" == "bitvla" && "${TASK_SUITE}" != libero_object ]]; then + # the Hub default is the libero_object ckpt; other suites need their own stats + client_extra+=(--tokenizer "${model_dir}") fi # gr00t_n1_6 has no HF-default tokenizer; point the client at the vendored # Eagle tokenizer in the model dir (override via GR00T_N1_6_TOKENIZER). if [[ "${arch}" == "gr00t_n1_6" ]]; then client_extra+=(--tokenizer "${GR00T_N1_6_TOKENIZER:-${model_dir}}") fi + if [[ ( "${arch}" == pi0 || "${arch}" == pi05 ) && -n "${PALIGEMMA_TOKENIZER}" ]]; then + client_extra+=(--tokenizer "${PALIGEMMA_TOKENIZER}") + fi if [[ -n "${stats_json}" ]]; then client_extra+=(--stats-json "${stats_json}") fi @@ -326,8 +365,12 @@ run_model() { else unset VLA_OPENVLA_OFT_UNNORM_KEY fi + if [[ "${arch}" == octo ]]; then + export VLA_OCTO_UNNORM_DATASET="${VLA_OCTO_UNNORM_DATASET:-${TASK_SUITE}}" + echo "[${arch}] VLA_OCTO_UNNORM_DATASET=${VLA_OCTO_UNNORM_DATASET}" + fi - local log="${LOG_DIR}/${arch}.log" + local log="${LOG_DIR}/${arch}-${TASK_SUITE}.log" echo "====================" echo "[${arch}] model_dir=${model_dir}" echo "[${arch}] server args: ${server_args[*]}" @@ -336,9 +379,9 @@ run_model() { local out_dir="${OUTPUT_ROOT}/${arch}" mkdir -p "${out_dir}" - for task_id in $(seq 0 9); do + for task_id in ${TASK_IDS}; do echo "[${arch}] task_id=${task_id} episodes=${N_EPISODES}" - "${VENV_PY}" "${CLIENT}" \ + "${client_py}" "${CLIENT}" \ --arch "${arch}" \ --vla-addr "${CLIENT_ADDR}" \ --task "${TASK_SUITE}" \ @@ -394,21 +437,22 @@ fi # bitvla: vision baked in; tokenizer + dataset_statistics.json auto-load from the # GGUF repo on the Hub. Set BITVLA_TOKENIZER= to override (offline). if should_run bit; then + bit_suite="${SUITE_NAME/#10/long}" # the libero_10 ckpt is named libero_long run_model bitvla \ - "${MODELS_ROOT}/bitvla-libero-gguf/libero_object" \ + "${MODELS_ROOT}/bitvla-libero-gguf/libero_${bit_suite}" \ "${N_ACTION_STEPS_BIT}" \ "" \ - "${MODELS_ROOT}/bitvla-libero-gguf/libero_object/bitvla-libero-object.gguf" + "${MODELS_ROOT}/bitvla-libero-gguf/libero_${bit_suite}/bitvla-libero-${bit_suite}.gguf" fi # vla_adapter: Qwen2.5-0.5B + Bridge-Attention; vision baked in (no mmproj), # tokenizer auto-loads from the base ckpt on the Hub, stats baked into the GGUF. if should_run vla_adapter; then run_model vla_adapter \ - "${MODELS_ROOT}/vla-adapter-libero-object-gguf" \ + "${MODELS_ROOT}/vla-adapter-libero-gguf" \ "${N_ACTION_STEPS_VLA_ADAPTER}" \ "" \ - "${MODELS_ROOT}/vla-adapter-libero-object-gguf/libero_object/vla-adapter-libero-object.gguf" + "${MODELS_ROOT}/vla-adapter-libero-gguf/libero_object/vla-adapter-libero-object.gguf" fi # openvla_oft: Llama-2-7B + MLPResNet head; vision baked in (no mmproj). Needs the @@ -464,18 +508,49 @@ fi # gr00t_n1_7: needs dataset_statistics.json if should_run gr00t_n1_7; then - g7_stats_default="${MODELS_ROOT}/gr00tn1d7-libero-gguf/libero_object/dataset_statistics.json" + g7_stats_default="${MODELS_ROOT}/gr00tn1d7-libero-gguf/${TASK_SUITE}/dataset_statistics.json" g7_stats="${GR00T_N1_7_STATS:-${g7_stats_default}}" if [[ -f "${g7_stats}" ]]; then run_model gr00t_n1_7 \ - "${MODELS_ROOT}/gr00tn1d7-libero-gguf/libero_object" \ + "${MODELS_ROOT}/gr00tn1d7-libero-gguf/${TASK_SUITE}" \ "${N_ACTION_STEPS_GR00T_N1_7}" \ "${g7_stats}" \ - "${MODELS_ROOT}/gr00tn1d7-libero-gguf/libero_object/gr00tn1d7-libero-object.gguf" + "${MODELS_ROOT}/gr00tn1d7-libero-gguf/${TASK_SUITE}/gr00tn1d7-libero-${SUITE_NAME}.gguf" else echo "[skip] gr00t_n1_7: dataset_statistics.json not found at ${g7_stats}; set GR00T_N1_7_STATS to override" fi fi +if should_run octo; then + run_model octo \ + "${MODELS_ROOT}/octo-small-libero-gguf" \ + "${N_ACTION_STEPS_OCTO}" \ + "" \ + "${MODELS_ROOT}/octo-small-libero-gguf/octo-small-libero-f32.gguf" +fi + +if should_run turbovla; then + run_model turbovla \ + "${MODELS_ROOT}/turbovla-libero-gguf" \ + "${N_ACTION_STEPS_TURBOVLA}" \ + "${TURBOVLA_STATS:-}" \ + "${MODELS_ROOT}/turbovla-libero-gguf/turbovla-libero-f32.gguf" +fi + +if should_run vla_jepa; then + jepa_stats="${VLA_JEPA_STATS:-${MODELS_ROOT}/vla-jepa-libero}" + if [[ ! -f "${jepa_stats}/policy_preprocessor_step_3_normalizer_processor.safetensors" ]]; then + echo "[skip] vla_jepa: policy_{pre,post}processor safetensors not found in ${jepa_stats}; set VLA_JEPA_STATS to override" + elif ! "${VLA_JEPA_PYTHON}" -c 'import sys, transformers as t; sys.exit(tuple(int(x) for x in t.__version__.split(".")[:2]) < (5, 4))'; then + echo "[skip] vla_jepa: the client needs transformers>=5.4, which ${VLA_JEPA_PYTHON} lacks; set VLA_JEPA_PYTHON to a python with LIBERO and transformers>=5.4" + else + run_model vla_jepa \ + "${MODELS_ROOT}/vla-jepa-libero" \ + "${N_ACTION_STEPS_VLA_JEPA}" \ + "${jepa_stats}" \ + "${MODELS_ROOT}/vla-jepa-libero/vla-jepa.gguf" + fi +fi + echo "====================" echo "Done. Results under ${OUTPUT_ROOT}" diff --git a/examples/chat/README.md b/examples/chat/README.md index 3ee40dc..707614b 100644 --- a/examples/chat/README.md +++ b/examples/chat/README.md @@ -107,6 +107,10 @@ regenerate the binding next to the client: protoc --proto_path=src/serving --python_out=examples/chat src/serving/vlm.proto ``` +Use protoc 3.20 or 3.21 (Ubuntu 24.04's `protobuf-compiler` is 3.21.12). Its +output imports on every protobuf runtime from 3.20 to 7.x. protoc 5.26 and newer +adds a runtime version check, so its output fails on older runtimes. + **Interactive REPL** (streams tokens live; stateless multi-turn - the client resends the full history each turn): diff --git a/examples/chat/vlm_pb2.py b/examples/chat/vlm_pb2.py index c25c443..03a12ae 100644 --- a/examples/chat/vlm_pb2.py +++ b/examples/chat/vlm_pb2.py @@ -20,491 +20,42 @@ # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE # SOFTWARE. +# -*- coding: utf-8 -*- +# Generated by the protocol buffer compiler. DO NOT EDIT! +# source: vlm.proto +"""Generated protocol buffer code.""" +from google.protobuf.internal import builder as _builder from google.protobuf import descriptor as _descriptor -from google.protobuf import message as _message -from google.protobuf import reflection as _reflection +from google.protobuf import descriptor_pool as _descriptor_pool from google.protobuf import symbol_database as _symbol_database +# @@protoc_insertion_point(imports) _sym_db = _symbol_database.Default() -DESCRIPTOR = _descriptor.FileDescriptor( - name='vlm.proto', - package='vlm_chat', - syntax='proto3', - serialized_options=None, - create_key=_descriptor._internal_create_key, - serialized_pb=b'\n\tvlm.proto\x12\x08vlm_chat\"\x82\x01\n\x05Image\x12*\n\x08\x65ncoding\x18\x01 \x01(\x0e\x32\x18.vlm_chat.Image.Encoding\x12\r\n\x05width\x18\x02 \x01(\r\x12\x0e\n\x06height\x18\x03 \x01(\r\x12\x0c\n\x04\x64\x61ta\x18\x04 \x01(\x0c\" \n\x08\x45ncoding\x12\x08\n\x04JPEG\x10\x00\x12\n\n\x06RGB_U8\x10\x01\",\n\x0b\x43hatMessage\x12\x0c\n\x04role\x18\x01 \x01(\t\x12\x0f\n\x07\x63ontent\x18\x02 \x01(\t\"e\n\x0eSamplingParams\x12\x13\n\x0btemperature\x18\x01 \x01(\x02\x12\r\n\x05top_p\x18\x02 \x01(\x02\x12\r\n\x05top_k\x18\x03 \x01(\x05\x12\x12\n\nmax_tokens\x18\x04 \x01(\x05\x12\x0c\n\x04seed\x18\x05 \x01(\x04\"\xa7\x01\n\x0b\x43hatRequest\x12\'\n\x08messages\x18\x01 \x03(\x0b\x32\x15.vlm_chat.ChatMessage\x12\x1f\n\x06images\x18\x02 \x03(\x0b\x32\x0f.vlm_chat.Image\x12*\n\x08sampling\x18\x03 \x01(\x0b\x32\x18.vlm_chat.SamplingParams\x12\x0e\n\x06stream\x18\x04 \x01(\x08\x12\x12\n\nrequest_id\x18\x05 \x01(\x04\"\xd9\x01\n\x0c\x43hatResponse\x12\x12\n\nrequest_id\x18\x01 \x01(\x04\x12\x0c\n\x04text\x18\x02 \x01(\t\x12\x15\n\rfinish_reason\x18\x03 \x01(\t\x12\x15\n\rprompt_tokens\x18\x04 \x01(\r\x12\x19\n\x11\x63ompletion_tokens\x18\x05 \x01(\r\x12\x18\n\x10latency_ms_total\x18\x06 \x01(\x02\x12\x1a\n\x12latency_ms_prefill\x18\x07 \x01(\x02\x12\x19\n\x11latency_ms_decode\x18\x08 \x01(\x02\x12\r\n\x05\x65rror\x18\t \x01(\t\"4\n\x0f\x43hatStreamDelta\x12\x12\n\nrequest_id\x18\x01 \x01(\x04\x12\r\n\x05\x64\x65lta\x18\x02 \x01(\t\"l\n\rStreamMessage\x12*\n\x05\x64\x65lta\x18\x01 \x01(\x0b\x32\x19.vlm_chat.ChatStreamDeltaH\x00\x12\'\n\x05\x66inal\x18\x02 \x01(\x0b\x32\x16.vlm_chat.ChatResponseH\x00\x42\x06\n\x04kindb\x06proto3' -) -_IMAGE_ENCODING = _descriptor.EnumDescriptor( - name='Encoding', - full_name='vlm_chat.Image.Encoding', - filename=None, - file=DESCRIPTOR, - create_key=_descriptor._internal_create_key, - values=[ - _descriptor.EnumValueDescriptor( - name='JPEG', index=0, number=0, - serialized_options=None, - type=None, - create_key=_descriptor._internal_create_key), - _descriptor.EnumValueDescriptor( - name='RGB_U8', index=1, number=1, - serialized_options=None, - type=None, - create_key=_descriptor._internal_create_key), - ], - containing_type=None, - serialized_options=None, - serialized_start=122, - serialized_end=154, -) -_sym_db.RegisterEnumDescriptor(_IMAGE_ENCODING) -_IMAGE = _descriptor.Descriptor( - name='Image', - full_name='vlm_chat.Image', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='encoding', full_name='vlm_chat.Image.encoding', index=0, - number=1, type=14, cpp_type=8, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='width', full_name='vlm_chat.Image.width', index=1, - number=2, type=13, cpp_type=3, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='height', full_name='vlm_chat.Image.height', index=2, - number=3, type=13, cpp_type=3, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='data', full_name='vlm_chat.Image.data', index=3, - number=4, type=12, cpp_type=9, label=1, - has_default_value=False, default_value=b"", - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - _IMAGE_ENCODING, - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - ], - serialized_start=24, - serialized_end=154, -) -_CHATMESSAGE = _descriptor.Descriptor( - name='ChatMessage', - full_name='vlm_chat.ChatMessage', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='role', full_name='vlm_chat.ChatMessage.role', index=0, - number=1, type=9, cpp_type=9, label=1, - has_default_value=False, default_value=b"".decode('utf-8'), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='content', full_name='vlm_chat.ChatMessage.content', index=1, - number=2, type=9, cpp_type=9, label=1, - has_default_value=False, default_value=b"".decode('utf-8'), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - ], - serialized_start=156, - serialized_end=200, -) - -_SAMPLINGPARAMS = _descriptor.Descriptor( - name='SamplingParams', - full_name='vlm_chat.SamplingParams', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='temperature', full_name='vlm_chat.SamplingParams.temperature', index=0, - number=1, type=2, cpp_type=6, label=1, - has_default_value=False, default_value=float(0), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='top_p', full_name='vlm_chat.SamplingParams.top_p', index=1, - number=2, type=2, cpp_type=6, label=1, - has_default_value=False, default_value=float(0), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='top_k', full_name='vlm_chat.SamplingParams.top_k', index=2, - number=3, type=5, cpp_type=1, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='max_tokens', full_name='vlm_chat.SamplingParams.max_tokens', index=3, - number=4, type=5, cpp_type=1, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='seed', full_name='vlm_chat.SamplingParams.seed', index=4, - number=5, type=4, cpp_type=4, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - ], - serialized_start=202, - serialized_end=303, -) - -_CHATREQUEST = _descriptor.Descriptor( - name='ChatRequest', - full_name='vlm_chat.ChatRequest', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='messages', full_name='vlm_chat.ChatRequest.messages', index=0, - number=1, type=11, cpp_type=10, label=3, - has_default_value=False, default_value=[], - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='images', full_name='vlm_chat.ChatRequest.images', index=1, - number=2, type=11, cpp_type=10, label=3, - has_default_value=False, default_value=[], - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='sampling', full_name='vlm_chat.ChatRequest.sampling', index=2, - number=3, type=11, cpp_type=10, label=1, - has_default_value=False, default_value=None, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='stream', full_name='vlm_chat.ChatRequest.stream', index=3, - number=4, type=8, cpp_type=7, label=1, - has_default_value=False, default_value=False, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='request_id', full_name='vlm_chat.ChatRequest.request_id', index=4, - number=5, type=4, cpp_type=4, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - ], - serialized_start=306, - serialized_end=473, -) - -_CHATRESPONSE = _descriptor.Descriptor( - name='ChatResponse', - full_name='vlm_chat.ChatResponse', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='request_id', full_name='vlm_chat.ChatResponse.request_id', index=0, - number=1, type=4, cpp_type=4, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='text', full_name='vlm_chat.ChatResponse.text', index=1, - number=2, type=9, cpp_type=9, label=1, - has_default_value=False, default_value=b"".decode('utf-8'), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='finish_reason', full_name='vlm_chat.ChatResponse.finish_reason', index=2, - number=3, type=9, cpp_type=9, label=1, - has_default_value=False, default_value=b"".decode('utf-8'), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='prompt_tokens', full_name='vlm_chat.ChatResponse.prompt_tokens', index=3, - number=4, type=13, cpp_type=3, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='completion_tokens', full_name='vlm_chat.ChatResponse.completion_tokens', index=4, - number=5, type=13, cpp_type=3, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='latency_ms_total', full_name='vlm_chat.ChatResponse.latency_ms_total', index=5, - number=6, type=2, cpp_type=6, label=1, - has_default_value=False, default_value=float(0), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='latency_ms_prefill', full_name='vlm_chat.ChatResponse.latency_ms_prefill', index=6, - number=7, type=2, cpp_type=6, label=1, - has_default_value=False, default_value=float(0), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='latency_ms_decode', full_name='vlm_chat.ChatResponse.latency_ms_decode', index=7, - number=8, type=2, cpp_type=6, label=1, - has_default_value=False, default_value=float(0), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='error', full_name='vlm_chat.ChatResponse.error', index=8, - number=9, type=9, cpp_type=9, label=1, - has_default_value=False, default_value=b"".decode('utf-8'), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - ], - serialized_start=476, - serialized_end=693, -) - -_CHATSTREAMDELTA = _descriptor.Descriptor( - name='ChatStreamDelta', - full_name='vlm_chat.ChatStreamDelta', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='request_id', full_name='vlm_chat.ChatStreamDelta.request_id', index=0, - number=1, type=4, cpp_type=4, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='delta', full_name='vlm_chat.ChatStreamDelta.delta', index=1, - number=2, type=9, cpp_type=9, label=1, - has_default_value=False, default_value=b"".decode('utf-8'), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - ], - serialized_start=695, - serialized_end=747, -) - -_STREAMMESSAGE = _descriptor.Descriptor( - name='StreamMessage', - full_name='vlm_chat.StreamMessage', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='delta', full_name='vlm_chat.StreamMessage.delta', index=0, - number=1, type=11, cpp_type=10, label=1, - has_default_value=False, default_value=None, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='final', full_name='vlm_chat.StreamMessage.final', index=1, - number=2, type=11, cpp_type=10, label=1, - has_default_value=False, default_value=None, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - _descriptor.OneofDescriptor( - name='kind', full_name='vlm_chat.StreamMessage.kind', - index=0, containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[]), - ], - serialized_start=749, - serialized_end=857, -) - -_IMAGE.fields_by_name['encoding'].enum_type = _IMAGE_ENCODING -_IMAGE_ENCODING.containing_type = _IMAGE -_CHATREQUEST.fields_by_name['messages'].message_type = _CHATMESSAGE -_CHATREQUEST.fields_by_name['images'].message_type = _IMAGE -_CHATREQUEST.fields_by_name['sampling'].message_type = _SAMPLINGPARAMS -_STREAMMESSAGE.fields_by_name['delta'].message_type = _CHATSTREAMDELTA -_STREAMMESSAGE.fields_by_name['final'].message_type = _CHATRESPONSE -_STREAMMESSAGE.oneofs_by_name['kind'].fields.append( - _STREAMMESSAGE.fields_by_name['delta']) -_STREAMMESSAGE.fields_by_name['delta'].containing_oneof = _STREAMMESSAGE.oneofs_by_name['kind'] -_STREAMMESSAGE.oneofs_by_name['kind'].fields.append( - _STREAMMESSAGE.fields_by_name['final']) -_STREAMMESSAGE.fields_by_name['final'].containing_oneof = _STREAMMESSAGE.oneofs_by_name['kind'] -DESCRIPTOR.message_types_by_name['Image'] = _IMAGE -DESCRIPTOR.message_types_by_name['ChatMessage'] = _CHATMESSAGE -DESCRIPTOR.message_types_by_name['SamplingParams'] = _SAMPLINGPARAMS -DESCRIPTOR.message_types_by_name['ChatRequest'] = _CHATREQUEST -DESCRIPTOR.message_types_by_name['ChatResponse'] = _CHATRESPONSE -DESCRIPTOR.message_types_by_name['ChatStreamDelta'] = _CHATSTREAMDELTA -DESCRIPTOR.message_types_by_name['StreamMessage'] = _STREAMMESSAGE -_sym_db.RegisterFileDescriptor(DESCRIPTOR) - -Image = _reflection.GeneratedProtocolMessageType('Image', (_message.Message,), { - 'DESCRIPTOR' : _IMAGE, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(Image) - -ChatMessage = _reflection.GeneratedProtocolMessageType('ChatMessage', (_message.Message,), { - 'DESCRIPTOR' : _CHATMESSAGE, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(ChatMessage) - -SamplingParams = _reflection.GeneratedProtocolMessageType('SamplingParams', (_message.Message,), { - 'DESCRIPTOR' : _SAMPLINGPARAMS, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(SamplingParams) - -ChatRequest = _reflection.GeneratedProtocolMessageType('ChatRequest', (_message.Message,), { - 'DESCRIPTOR' : _CHATREQUEST, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(ChatRequest) - -ChatResponse = _reflection.GeneratedProtocolMessageType('ChatResponse', (_message.Message,), { - 'DESCRIPTOR' : _CHATRESPONSE, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(ChatResponse) - -ChatStreamDelta = _reflection.GeneratedProtocolMessageType('ChatStreamDelta', (_message.Message,), { - 'DESCRIPTOR' : _CHATSTREAMDELTA, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(ChatStreamDelta) - -StreamMessage = _reflection.GeneratedProtocolMessageType('StreamMessage', (_message.Message,), { - 'DESCRIPTOR' : _STREAMMESSAGE, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(StreamMessage) +DESCRIPTOR = _descriptor_pool.Default().AddSerializedFile(b'\n\tvlm.proto\x12\x08vlm_chat\"\x82\x01\n\x05Image\x12*\n\x08\x65ncoding\x18\x01 \x01(\x0e\x32\x18.vlm_chat.Image.Encoding\x12\r\n\x05width\x18\x02 \x01(\r\x12\x0e\n\x06height\x18\x03 \x01(\r\x12\x0c\n\x04\x64\x61ta\x18\x04 \x01(\x0c\" \n\x08\x45ncoding\x12\x08\n\x04JPEG\x10\x00\x12\n\n\x06RGB_U8\x10\x01\",\n\x0b\x43hatMessage\x12\x0c\n\x04role\x18\x01 \x01(\t\x12\x0f\n\x07\x63ontent\x18\x02 \x01(\t\"e\n\x0eSamplingParams\x12\x13\n\x0btemperature\x18\x01 \x01(\x02\x12\r\n\x05top_p\x18\x02 \x01(\x02\x12\r\n\x05top_k\x18\x03 \x01(\x05\x12\x12\n\nmax_tokens\x18\x04 \x01(\x05\x12\x0c\n\x04seed\x18\x05 \x01(\x04\"\xa7\x01\n\x0b\x43hatRequest\x12\'\n\x08messages\x18\x01 \x03(\x0b\x32\x15.vlm_chat.ChatMessage\x12\x1f\n\x06images\x18\x02 \x03(\x0b\x32\x0f.vlm_chat.Image\x12*\n\x08sampling\x18\x03 \x01(\x0b\x32\x18.vlm_chat.SamplingParams\x12\x0e\n\x06stream\x18\x04 \x01(\x08\x12\x12\n\nrequest_id\x18\x05 \x01(\x04\"\xd9\x01\n\x0c\x43hatResponse\x12\x12\n\nrequest_id\x18\x01 \x01(\x04\x12\x0c\n\x04text\x18\x02 \x01(\t\x12\x15\n\rfinish_reason\x18\x03 \x01(\t\x12\x15\n\rprompt_tokens\x18\x04 \x01(\r\x12\x19\n\x11\x63ompletion_tokens\x18\x05 \x01(\r\x12\x18\n\x10latency_ms_total\x18\x06 \x01(\x02\x12\x1a\n\x12latency_ms_prefill\x18\x07 \x01(\x02\x12\x19\n\x11latency_ms_decode\x18\x08 \x01(\x02\x12\r\n\x05\x65rror\x18\t \x01(\t\"4\n\x0f\x43hatStreamDelta\x12\x12\n\nrequest_id\x18\x01 \x01(\x04\x12\r\n\x05\x64\x65lta\x18\x02 \x01(\t\"l\n\rStreamMessage\x12*\n\x05\x64\x65lta\x18\x01 \x01(\x0b\x32\x19.vlm_chat.ChatStreamDeltaH\x00\x12\'\n\x05\x66inal\x18\x02 \x01(\x0b\x32\x16.vlm_chat.ChatResponseH\x00\x42\x06\n\x04kindb\x06proto3') + +_builder.BuildMessageAndEnumDescriptors(DESCRIPTOR, globals()) +_builder.BuildTopDescriptorsAndMessages(DESCRIPTOR, 'vlm_pb2', globals()) +if _descriptor._USE_C_DESCRIPTORS == False: + + DESCRIPTOR._options = None + _IMAGE._serialized_start=24 + _IMAGE._serialized_end=154 + _IMAGE_ENCODING._serialized_start=122 + _IMAGE_ENCODING._serialized_end=154 + _CHATMESSAGE._serialized_start=156 + _CHATMESSAGE._serialized_end=200 + _SAMPLINGPARAMS._serialized_start=202 + _SAMPLINGPARAMS._serialized_end=303 + _CHATREQUEST._serialized_start=306 + _CHATREQUEST._serialized_end=473 + _CHATRESPONSE._serialized_start=476 + _CHATRESPONSE._serialized_end=693 + _CHATSTREAMDELTA._serialized_start=695 + _CHATSTREAMDELTA._serialized_end=747 + _STREAMMESSAGE._serialized_start=749 + _STREAMMESSAGE._serialized_end=857 +# @@protoc_insertion_point(module_scope) diff --git a/pyproject.toml b/pyproject.toml index ec97106..e3d8938 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,18 +1,19 @@ [project] name = "vla-cpp-tooling" -version = "0.3.0" +version = "0.4.0" description = "Python tooling for vla.cpp: HuggingFace -> GGUF converters and the ZeroMQ eval client." readme = "README.md" requires-python = ">=3.10" -license = { text = "Apache-2.0" } +license = "Apache-2.0" dependencies = ["numpy>=1.24"] [project.optional-dependencies] # scripts/convert_*_to_gguf.py and the gguf_common / gguf_blocks modules they share convert = [ - "torch>=2.5", - "safetensors>=0.4", - "gguf>=0.10", + "torch>=2.6", + "safetensors>=0.4.3", + "gguf>=0.17", + "transformers>=4.57", "numpy>=1.24", ] # eval/client/* talks to vla-server over ZeroMQ. The simulators (LIBERO, SimplerEnv, @@ -20,13 +21,14 @@ convert = [ client = [ "pyzmq>=25", "msgpack>=1.1", - "msgpack-numpy>=0.4.8", "pillow>=10", - "torch>=2.5", - # Only used for AutoTokenizer/AutoProcessor.from_pretrained (the pi0 PaliGemma - # tokenizer). Capped below 5.0: the v5 line is a breaking rewrite we have not - # tested against. - "transformers>=4.51,<5", + "torch>=2.6", + "torchvision>=0.21", + # AutoTokenizer/AutoProcessor only. vla_jepa passes processor_kwargs to + # apply_chat_template, which 5.4 is the first release to accept. + "transformers>=5.4", + "protobuf>=4.21", + "opencv-python-headless>=4.8", "numpy>=1.24", ] diff --git a/scripts/add_tokenizer_to_gguf.py b/scripts/add_tokenizer_to_gguf.py new file mode 100644 index 0000000..0c67de7 --- /dev/null +++ b/scripts/add_tokenizer_to_gguf.py @@ -0,0 +1,89 @@ +#!/usr/bin/env python3 +# Copyright 2026 VinRobotics +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Copy a vla.cpp GGUF and embed the SentencePiece tokenizer its arch was trained +with as .tokenizer.spm_model, so vla-cli --text tokenizes in-process with +no Python and no HF login. + + python scripts/add_tokenizer_to_gguf.py --in pi0.gguf --out pi0-tok.gguf + +pi05 puts the robot state in its prompt, so its observation.state q01/q99 go in +too as pi05.state.q01/q99 (--stats, default: the LIBERO meta/stats.json the eval +client uses). +""" + +import argparse +import json +from pathlib import Path + +import numpy as np +import gguf + +from gguf_common import copy_kv, copy_tensor + +TOKENIZERS = { + "pi0": "google/paligemma-3b-pt-224", + "pi05": "google/paligemma-3b-pt-224", + "openvla_oft": "moojink/openvla-7b-oft-finetuned-libero-spatial-object-goal-10", +} + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--in", dest="src", required=True) + ap.add_argument("--out", dest="dst", required=True) + ap.add_argument("--tokenizer", help="HF repo holding tokenizer.model, or a local .model file " + "(default: the one the arch was trained with)") + ap.add_argument("--stats", help="pi05: stats.json with observation.state q01/q99") + args = ap.parse_args() + + r = gguf.GGUFReader(args.src) + arch = r.fields["general.architecture"].contents() + if arch not in TOKENIZERS: + raise SystemExit(f"vla-cli has no in-process --text prompt for arch {arch}; " + f"supported: {', '.join(TOKENIZERS)}") + tok = args.tokenizer or TOKENIZERS[arch] + if Path(tok).is_file(): + spm = Path(tok) + else: + from huggingface_hub import hf_hub_download + spm = Path(hf_hub_download(tok, "tokenizer.model")) + kv = {f"{arch}.tokenizer.spm_model": spm.read_bytes()} + + if arch == "pi05": + stats = args.stats + if not stats: + from huggingface_hub import hf_hub_download + stats = hf_hub_download("lerobot/libero", "meta/stats.json", repo_type="dataset") + st = json.loads(Path(stats).read_text())["observation.state"] + for q in ("q01", "q99"): + kv[f"pi05.state.{q}"] = np.asarray(st[q], dtype=np.float32).reshape(-1).tolist() + + w = gguf.GGUFWriter(args.dst, arch) + copy_kv(r, w, skip=kv) + for k, v in kv.items(): + w.add_array(k, v) + for t in r.tensors: + copy_tensor(w, t) + w.write_header_to_file() + w.write_kv_data_to_file() + w.write_tensors_to_file() + w.close() + print(f"{args.dst}: added {', '.join(kv)} from {spm}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/build_windows_snapdragon.ps1 b/scripts/build_windows_snapdragon.ps1 index ed2de01..3ee34fe 100644 --- a/scripts/build_windows_snapdragon.ps1 +++ b/scripts/build_windows_snapdragon.ps1 @@ -70,7 +70,7 @@ Set-Location $root $dir = "build-wos-$Backend" $jobs = $env:NUMBER_OF_PROCESSORS -$flags = "-march=armv8.7a+fp16+dotprod+i8mm -fvectorize -ffp-model=fast -D_GNU_SOURCE" +$flags = "-march=armv8.7a+fp16+dotprod+i8mm -fvectorize -ffp-model=fast -fno-finite-math-only -D_GNU_SOURCE" # protobuf and abseil must be built by clang too, with the triplet in # cmake/vcpkg-triplets: clang code does not link against an MSVC-built protobuf. # ZeroMQ is a C API and comes from the stock triplet, searched second. @@ -91,8 +91,8 @@ $args_ = @( ) if ($LlamaDir) { $args_ += "-DFETCHCONTENT_SOURCE_DIR_LLAMA=$($LlamaDir -replace '\\','/')" } if ($NoServer) { - # Octo tokenizes in-process through SentencePiece, which needs protobuf too. - $args_ += @("-DVLA_BUILD_SERVER=OFF", "-DVLA_OCTO=OFF") + # SentencePiece, for in-GGUF tokenizers, needs protobuf too. + $args_ += @("-DVLA_BUILD_SERVER=OFF", "-DVLA_SPM=OFF") } else { $args_ += @( "-DCMAKE_PREFIX_PATH=$($triplet -replace '\\','/');$($tripletC -replace '\\','/');$($env:OPENCL_SDK_ROOT -replace '\\','/')", diff --git a/scripts/convert_bitvla_to_gguf.py b/scripts/convert_bitvla_to_gguf.py index fe25dcf..3e29b74 100644 --- a/scripts/convert_bitvla_to_gguf.py +++ b/scripts/convert_bitvla_to_gguf.py @@ -15,9 +15,6 @@ from __future__ import annotations -import re -from pathlib import Path - import numpy as np import torch @@ -25,6 +22,7 @@ from gguf_common import ( add, arg_parser, + find_sidecar, finish, kv_f32, kv_prefix, @@ -196,14 +194,6 @@ def _add_bit_fused(writer, base: str, Ws: list[torch.Tensor]) -> None: packed, scales = pack_fused_projection(Ws) _add_packed(writer, base, packed, scales) -def _find_sidecar(ckpt: Path, stem: str) -> Path | None: - - cands = sorted( - ckpt.glob(f"{stem}--*_checkpoint.pt"), - key=lambda p: int(m.group(1)) if (m := re.search(r"--(\d+)_checkpoint\.pt$", p.name)) else -1, - ) - return cands[-1] if cands else None - def _add_kv(writer, statistics_json: str, processor_json: str, preproc_json: str) -> None: kv_u32( @@ -294,10 +284,8 @@ def main() -> int: W = load_safetensors(ckpt) print(f" {len(W)} main tensors") - ah_path = _find_sidecar(ckpt, "action_head") - pp_path = _find_sidecar(ckpt, "proprio_projector") - if ah_path is None or pp_path is None: - raise SystemExit(f"missing action_head/proprio_projector sidecars in {ckpt}") + ah_path = find_sidecar(ckpt, "action_head") + pp_path = find_sidecar(ckpt, "proprio_projector") print(f" sidecars: {ah_path.name}, {pp_path.name}") AH = load_pt_module(ah_path) PP = load_pt_module(pp_path) diff --git a/scripts/convert_evo1_to_gguf.py b/scripts/convert_evo1_to_gguf.py index 5ce88ee..7b0c1c4 100644 --- a/scripts/convert_evo1_to_gguf.py +++ b/scripts/convert_evo1_to_gguf.py @@ -137,13 +137,13 @@ def main() -> int: cfg["mlp_head_hidden"] = int(cfg_json.get("hidden_dim", 1024)) cfg["num_inference_timesteps"] = int(cfg_json.get("num_inference_timesteps", NUM_INFERENCE_TIMESTEPS)) cfg["image_size"] = int(cfg_json.get("image_size", VIT["image_size"])) - cfg["dit_heads"] = DIT_HEADS + cfg["dit_heads"] = int(cfg_json.get("num_heads", DIT_HEADS)) cfg["proj_ln_eps"] = PROJ_LN_EPS if cfg["action_dim"] != cfg["horizon"] * cfg["per_action_dim"]: raise SystemExit(f"action_dim {cfg['action_dim']} != horizon*per_action_dim {cfg['horizon']*cfg['per_action_dim']}") print(f"loading {pt_path} ...") - module = torch.load(pt_path, map_location="cpu", weights_only=False)["module"] + module = torch.load(pt_path, map_location="cpu", weights_only=True)["module"] keys = set(module.keys()) print(f" {len(module)} tensors") diff --git a/scripts/convert_octo_to_gguf.py b/scripts/convert_octo_to_gguf.py index cf4fb90..bd7aca9 100644 --- a/scripts/convert_octo_to_gguf.py +++ b/scripts/convert_octo_to_gguf.py @@ -16,6 +16,7 @@ import torch import gguf +from gguf_common import add_f32 ARCH = "octo" MODEL_ID = "hf://rail-berkeley/octo-small-1.5" @@ -90,10 +91,6 @@ def _add_meta(writer: gguf.GGUFWriter, key: str, value: Any) -> None: raise TypeError(f"unsupported metadata {full}={value!r}") -def _f32(t: torch.Tensor) -> np.ndarray: - return t.detach().to(dtype=torch.float32, device="cpu").contiguous().numpy() - - def _embed_tokenizer(writer: gguf.GGUFWriter, tokenizer_name: str = "t5-base") -> None: """Embed the raw T5 SentencePiece unigram model (spiece.model) as a UINT8 GGUF array (not a GGUF string: the serialized proto contains embedded NUL bytes, @@ -455,6 +452,7 @@ def main() -> int: loaded = OctoModelPt.load_pretrained_from_jax(model_id, step=args.step, skip_keys_regex=".*hf_model") m = loaded["octo_model"] sd = m.state_dict() + OCTO_META["ckpt_format"] = ckpt_format window_size = _resolve_window_size(m, args.ckpt, model_id, args.window_size) print(f"window_size = {window_size} (from " @@ -466,6 +464,11 @@ def main() -> int: OCTO_META["action.head_type"] = head_type OCTO_META["action.horizon"] = int(head_cfg["kwargs"]["action_horizon"]) OCTO_META["action.dim"] = int(head_cfg["kwargs"]["action_dim"]) + OCTO_META["diffusion.steps"] = int(head_cfg["kwargs"].get("diffusion_steps", 20)) + OCTO_META["diffusion.max_action"] = float(head_cfg["kwargs"].get("max_action", 5.0)) + mc = m.config["model"] + if not (mc.get("repeat_task_tokens") and mc.get("use_correct_attention")) or mc.get("readouts") != {"action": 1}: + raise SystemExit("vla.cpp Octo needs repeat_task_tokens, use_correct_attention and readouts={'action': 1}") print(f"action head: {head_cfg['name']} -> head_type={head_type} " f"horizon={OCTO_META['action.horizon']} dim={OCTO_META['action.dim']}") @@ -523,7 +526,7 @@ def main() -> int: rows = [] for src, dst in sorted(mapped.items(), key=lambda kv: kv[1]): tensor = sd[src] - writer.add_tensor(dst, _f32(tensor), raw_dtype=gguf.GGMLQuantizationType.F32) + add_f32(writer, dst, tensor) rows.append({"state_dict": src, "gguf": dst, "shape": list(tensor.shape)}) print(f"map {src} {tuple(tensor.shape)} -> {dst}") diff --git a/scripts/convert_openvla_oft_to_gguf.py b/scripts/convert_openvla_oft_to_gguf.py index 5bcf3a5..c6d882b 100644 --- a/scripts/convert_openvla_oft_to_gguf.py +++ b/scripts/convert_openvla_oft_to_gguf.py @@ -21,6 +21,7 @@ from gguf_common import ( add_bf16, arg_parser, + find_sidecar, finish, kv_prefix, kv_u32, @@ -101,8 +102,8 @@ def main() -> int: ckpt = args.ckpt.resolve() out = resolve_out(args, ckpt, ARCH) - ah_path = args.action_head or next(ckpt.glob("action_head--*checkpoint.pt")) - pp_path = args.proprio or next(ckpt.glob("proprio_projector--*checkpoint.pt")) + ah_path = args.action_head or find_sidecar(ckpt, "action_head") + pp_path = args.proprio or find_sidecar(ckpt, "proprio_projector") stats_path = ckpt / "dataset_statistics.json" require(stats_path) diff --git a/scripts/convert_pi05_to_gguf.py b/scripts/convert_pi05_to_gguf.py index e8e9280..565ae26 100644 --- a/scripts/convert_pi05_to_gguf.py +++ b/scripts/convert_pi05_to_gguf.py @@ -24,6 +24,7 @@ from gguf_blocks import ( norm_eps, + pi_root, probe_paligemma_vision, write_decoder_blocks, write_paligemma_vision, @@ -89,11 +90,11 @@ "action_out_proj.bias", ] -def _write_adarms_blocks(writer, sf, n_layers: int) -> None: +def _write_adarms_blocks(writer, sf, aex: str, n_layers: int) -> None: for i in range(n_layers): for src_suf, dst_suf in AEX_MAP: - add(writer, f"aex.blk.{i}.{dst_suf}", sf.get_tensor(f"{PFX_AEX}.layers.{i}.{src_suf}")) + add(writer, f"aex.blk.{i}.{dst_suf}", sf.get_tensor(f"{aex}.layers.{i}.{src_suf}")) def _load_dataset_stats( stats_json: Optional[Path], @@ -155,7 +156,7 @@ def main() -> int: "--dataset-stats", type=Path, default=None, - help="Path to a LIBERO dataset meta/stats.json for MEAN_STD norm stats" + help="Path to a LIBERO dataset meta/stats.json for QUANTILES norm stats" ) ap.add_argument( "--dataset-repo", @@ -173,6 +174,11 @@ def main() -> int: cfg_json = read_json(ckpt / "config.json") if cfg_json.get("type") != ARCH: raise SystemExit(f"config.json type is {cfg_json.get('type')!r}, expected 'pi05'") + norm_map = cfg_json.get("normalization_mapping") or {} + norm_mode = norm_map.get("ACTION", "QUANTILES") + if norm_mode != "QUANTILES" or norm_map.get("STATE", "QUANTILES") != norm_mode: + raise SystemExit(f"unsupported pi05 normalization_mapping {norm_map}: only QUANTILES is supported, " + f"the state prompt is always binned with q01/q99") cfg = dict(GEMMA_2B, **GEMMA_300M) cfg["paligemma_variant"] = str(cfg_json.get("paligemma_variant", "gemma_2b")) @@ -190,22 +196,25 @@ def main() -> int: cfg["rope_theta"] = ROPE_THETA cfg["rms_norm_eps"] = RMS_NORM_EPS cfg["norm_eps"] = norm_eps(ckpt) + cfg["norm_mode"] = norm_mode.lower() print(f"opening {sf_path}") sf = safe_open(sf_path, framework="pt") keys = set(sf.keys()) + root = pi_root(keys) + vlm, head, aex = root + PFX_VLM, root + PFX_VLM_HEAD, root + PFX_AEX - n_layers_vlm = max_layer(keys, f"{PFX_VLM}.layers.") - n_layers_aex = max_layer(keys, f"{PFX_AEX}.layers.") + n_layers_vlm = max_layer(keys, f"{vlm}.layers.") + n_layers_aex = max_layer(keys, f"{aex}.layers.") if n_layers_vlm <= 0: raise SystemExit("cannot find PaliGemma language-model layers in checkpoint") if n_layers_aex != n_layers_vlm: raise SystemExit(f"layer count mismatch: VLM={n_layers_vlm} expert={n_layers_aex}") cfg["n_layers"] = n_layers_vlm - q0 = sf.get_slice(f"{PFX_VLM}.layers.0.self_attn.q_proj.weight").get_shape() - kv0 = sf.get_slice(f"{PFX_VLM}.layers.0.self_attn.k_proj.weight").get_shape() - gate0 = sf.get_slice(f"{PFX_VLM}.layers.0.mlp.gate_proj.weight").get_shape() + q0 = sf.get_slice(f"{vlm}.layers.0.self_attn.q_proj.weight").get_shape() + kv0 = sf.get_slice(f"{vlm}.layers.0.self_attn.k_proj.weight").get_shape() + gate0 = sf.get_slice(f"{vlm}.layers.0.mlp.gate_proj.weight").get_shape() if q0[1] != cfg["hidden"]: raise SystemExit(f"hidden mismatch: cfg={cfg['hidden']} ckpt={q0[1]}") if q0[0] != cfg["n_q_heads"] * cfg["head_dim"]: @@ -215,15 +224,15 @@ def main() -> int: if gate0[0] != cfg["intermediate"]: raise SystemExit(f"intermediate mismatch: cfg={cfg['intermediate']} ckpt={gate0[0]}") - ada0 = sf.get_slice(f"{PFX_AEX}.layers.0.input_layernorm.dense.weight").get_shape() + ada0 = sf.get_slice(f"{aex}.layers.0.input_layernorm.dense.weight").get_shape() if ada0 != [3 * cfg["expert_h"], cfg["expert_h"]]: raise SystemExit(f"expert adaRMS dense shape {ada0} != [3*expert_h, expert_h] " f"{[3*cfg['expert_h'], cfg['expert_h']]}") - aex_o0 = sf.get_slice(f"{PFX_AEX}.layers.0.self_attn.o_proj.weight").get_shape() + aex_o0 = sf.get_slice(f"{aex}.layers.0.self_attn.o_proj.weight").get_shape() if aex_o0 != [cfg["expert_h"], cfg["n_q_heads"] * cfg["head_dim"]]: raise SystemExit(f"expert o_proj shape {aex_o0} unexpected") - cfg["vocab_size"] = int(sf.get_slice(PFX_VLM_HEAD).get_shape()[0]) + cfg["vocab_size"] = int(sf.get_slice(head).get_shape()[0]) print(f"resolved cfg: hidden={cfg['hidden']} n_layers={cfg['n_layers']} " f"expert_h={cfg['expert_h']} vocab={cfg['vocab_size']} chunk={cfg['chunk_size']} " @@ -231,7 +240,8 @@ def main() -> int: f"real_action={cfg['real_action_dim']} max_len={cfg['tokenizer_max_length']} " f"norm_eps={cfg['norm_eps']:g}") - cfg["vit"] = probe_paligemma_vision(sf, keys, cfg_json, PFX_VIS_CANDIDATES, PFX_MMP_CANDIDATES) + cfg["vit"] = probe_paligemma_vision(sf, keys, cfg_json, [root + p for p in PFX_VIS_CANDIDATES], + [root + p for p in PFX_MMP_CANDIDATES]) v = cfg["vit"] print(f"vision: SigLIP hidden={v['vit_hidden']} layers={v['vit_layers']} " f"heads={v['vit_heads']} image={v['image_size']} patch={v['patch_size']} " @@ -250,16 +260,16 @@ def main() -> int: writer = open_writer(out, ARCH) write_pi_kv(writer, KV, cfg, adarms=True) - add(writer, "token_embd.weight", sf.get_tensor(PFX_VLM_HEAD)) - add(writer, "vlm.output_norm.weight", sf.get_tensor(f"{PFX_VLM}.norm.weight")) - write_decoder_blocks(writer, sf.get_tensor, PFX_VLM, "vlm", cfg["n_layers"]) + add(writer, "token_embd.weight", sf.get_tensor(head)) + add(writer, "vlm.output_norm.weight", sf.get_tensor(f"{vlm}.norm.weight")) + write_decoder_blocks(writer, sf.get_tensor, vlm, "vlm", cfg["n_layers"]) - add(writer, "aex.output_norm.weight", sf.get_tensor(f"{PFX_AEX}.norm.dense.weight")) - add(writer, "aex.output_norm.bias", sf.get_tensor(f"{PFX_AEX}.norm.dense.bias")) - _write_adarms_blocks(writer, sf, cfg["n_layers"]) + add(writer, "aex.output_norm.weight", sf.get_tensor(f"{aex}.norm.dense.weight")) + add(writer, "aex.output_norm.bias", sf.get_tensor(f"{aex}.norm.dense.bias")) + _write_adarms_blocks(writer, sf, aex, cfg["n_layers"]) for suf in PROJ_SUFFIXES: - add(writer, suf, sf.get_tensor(suf)) + add(writer, suf, sf.get_tensor(root + suf)) write_paligemma_vision(writer, sf, cfg["vit"]) diff --git a/scripts/convert_pi0_to_gguf.py b/scripts/convert_pi0_to_gguf.py index 45658f1..8276562 100644 --- a/scripts/convert_pi0_to_gguf.py +++ b/scripts/convert_pi0_to_gguf.py @@ -15,15 +15,12 @@ from __future__ import annotations -from pathlib import Path - -import numpy as np from safetensors import safe_open from gguf_blocks import ( - identity_stats, - load_processor_stats, + lerobot_stats, norm_eps, + pi_root, probe_paligemma_vision, write_decoder_blocks, write_paligemma_vision, @@ -45,18 +42,17 @@ ARCH = "pi0" KV = kv_prefix(ARCH) -PFX_VLM = "model.paligemma_with_expert.paligemma.model.language_model" -PFX_VLM_HEAD = "model.paligemma_with_expert.paligemma.lm_head.weight" -PFX_AEX = "model.paligemma_with_expert.gemma_expert.model" -PFX_PROJ = "model" +PFX_VLM = "paligemma_with_expert.paligemma.model.language_model" +PFX_VLM_HEAD = "paligemma_with_expert.paligemma.lm_head.weight" +PFX_AEX = "paligemma_with_expert.gemma_expert.model" PFX_VIS_CANDIDATES = [ - "model.paligemma_with_expert.paligemma.model.vision_tower.vision_model", - "model.paligemma_with_expert.paligemma.vision_tower.vision_model", + "paligemma_with_expert.paligemma.model.vision_tower.vision_model", + "paligemma_with_expert.paligemma.vision_tower.vision_model", ] PFX_MMP_CANDIDATES = [ - "model.paligemma_with_expert.paligemma.model.multi_modal_projector", - "model.paligemma_with_expert.paligemma.multi_modal_projector", + "paligemma_with_expert.paligemma.model.multi_modal_projector", + "paligemma_with_expert.paligemma.multi_modal_projector", ] GEMMA_2B = dict(hidden=2048, n_q_heads=8, n_kv_heads=1, head_dim=256, intermediate=16384) @@ -78,71 +74,6 @@ "action_out_proj.bias", ] -def _load_stats(sf, ckpt: Path, state_dim: int, action_dim: int) -> dict[str, np.ndarray]: - - out = identity_stats(state_dim, action_dim) - - got_state = load_processor_stats( - ckpt, - "policy_preprocessor.json", - "normalizer_processor", - "observation.state", - state_dim - ) - got_action = load_processor_stats( - ckpt, - "policy_postprocessor.json", - "unnormalizer_processor", - "action", - action_dim - ) - - keys = set(sf.keys()) - - def _legacy(mk: str, sk: str, mean_dst: str, std_dst: str, dim: int) -> None: - if mk not in keys or sk not in keys: - print(f" stats: legacy {mk} / {sk} missing - using identity for {mean_dst[:-5]}") - return - mean = sf.get_tensor(mk).float().numpy().reshape(-1) - std = sf.get_tensor(sk).float().numpy().reshape(-1) - if mean.size != dim or std.size != dim: - print(f" stats: legacy {mk} dim mismatch ({mean.size} vs {dim}) - using identity") - return - out[mean_dst] = mean.astype(np.float32, copy=False) - out[std_dst] = std .astype(np.float32, copy=False) - print(f" stats: loaded {mean_dst[:-5]} from model.safetensors ({mk}/{sk}) [legacy]") - - if got_state is not None: - out["state_mean"], out["state_std"] = got_state - else: - _legacy( - "normalize_inputs.buffer_observation_state.mean", - "normalize_inputs.buffer_observation_state.std", - "state_mean", - "state_std", - state_dim - ) - - if got_action is not None: - out["action_mean"], out["action_std"] = got_action - elif "unnormalize_outputs.buffer_action.mean" in keys: - _legacy( - "unnormalize_outputs.buffer_action.mean", - "unnormalize_outputs.buffer_action.std", - "action_mean", - "action_std", - action_dim - ) - else: - _legacy( - "normalize_targets.buffer_action.mean", - "normalize_targets.buffer_action.std", - "action_mean", - "action_std", - action_dim - ) - return out - def main() -> int: ap = arg_parser(ARCH, "lerobot π₀ checkpoint dir (model.safetensors + config.json + policy_*processor.json)") args = ap.parse_args() @@ -156,6 +87,8 @@ def main() -> int: if cfg_json.get("type") != ARCH: raise SystemExit(f"config.json type is {cfg_json.get('type')!r}, expected 'pi0' " f"(π0.5 / other variants are not handled by this converter)") + if cfg_json.get("adapt_to_pi_aloha"): + raise SystemExit("adapt_to_pi_aloha=true is not supported") cfg = dict(GEMMA_2B, **GEMMA_300M) cfg["paligemma_variant"] = str(cfg_json.get("paligemma_variant", "gemma_2b")) @@ -177,9 +110,11 @@ def main() -> int: print(f"opening {sf_path}") sf = safe_open(sf_path, framework="pt") keys = set(sf.keys()) + root = pi_root(keys) + vlm, head, aex = root + PFX_VLM, root + PFX_VLM_HEAD, root + PFX_AEX - n_layers_vlm = max_layer(keys, f"{PFX_VLM}.layers.") - n_layers_aex = max_layer(keys, f"{PFX_AEX}.layers.") + n_layers_vlm = max_layer(keys, f"{vlm}.layers.") + n_layers_aex = max_layer(keys, f"{aex}.layers.") if n_layers_vlm <= 0: raise SystemExit("cannot find PaliGemma language-model layers in checkpoint") if n_layers_aex != n_layers_vlm: @@ -187,9 +122,9 @@ def main() -> int: f"(π0 expects them equal)") cfg["n_layers"] = n_layers_vlm - q0 = sf.get_slice(f"{PFX_VLM}.layers.0.self_attn.q_proj.weight").get_shape() - kv0 = sf.get_slice(f"{PFX_VLM}.layers.0.self_attn.k_proj.weight").get_shape() - gate0 = sf.get_slice(f"{PFX_VLM}.layers.0.mlp.gate_proj.weight").get_shape() + q0 = sf.get_slice(f"{vlm}.layers.0.self_attn.q_proj.weight").get_shape() + kv0 = sf.get_slice(f"{vlm}.layers.0.self_attn.k_proj.weight").get_shape() + gate0 = sf.get_slice(f"{vlm}.layers.0.mlp.gate_proj.weight").get_shape() if q0[1] != cfg["hidden"]: raise SystemExit(f"hidden mismatch: cfg={cfg['hidden']} ckpt={q0[1]}") if q0[0] != cfg["n_q_heads"] * cfg["head_dim"]: @@ -199,17 +134,17 @@ def main() -> int: if gate0[0] != cfg["intermediate"]: raise SystemExit(f"intermediate mismatch: cfg={cfg['intermediate']} ckpt={gate0[0]}") - aex_gate0 = sf.get_slice(f"{PFX_AEX}.layers.0.mlp.gate_proj.weight").get_shape() + aex_gate0 = sf.get_slice(f"{aex}.layers.0.mlp.gate_proj.weight").get_shape() if aex_gate0[1] != cfg["expert_h"]: raise SystemExit(f"expert_h mismatch: cfg={cfg['expert_h']} ckpt={aex_gate0[1]}") if aex_gate0[0] != cfg["expert_inter"]: raise SystemExit(f"expert_inter mismatch: cfg={cfg['expert_inter']} ckpt={aex_gate0[0]}") - aex_o0 = sf.get_slice(f"{PFX_AEX}.layers.0.self_attn.o_proj.weight").get_shape() + aex_o0 = sf.get_slice(f"{aex}.layers.0.self_attn.o_proj.weight").get_shape() if aex_o0 != [cfg["expert_h"], cfg["n_q_heads"] * cfg["head_dim"]]: raise SystemExit(f"expert o_proj shape {aex_o0} != [expert_h, n_q*head_dim] " f"{[cfg['expert_h'], cfg['n_q_heads']*cfg['head_dim']]}") - head_w = sf.get_slice(PFX_VLM_HEAD).get_shape() + head_w = sf.get_slice(head).get_shape() if head_w[1] != cfg["hidden"]: raise SystemExit(f"lm_head hidden mismatch: cfg={cfg['hidden']} ckpt={head_w[1]}") cfg["vocab_size"] = int(head_w[0]) @@ -222,9 +157,11 @@ def main() -> int: f"norm_eps={cfg['norm_eps']:g}") print("loading normalizer stats...") - stats = _load_stats(sf, ckpt, cfg["real_state_dim"], cfg["real_action_dim"]) + stats = lerobot_stats(sf, ckpt, cfg["real_state_dim"], cfg["real_action_dim"], + cfg_json.get("normalization_mapping") or {}) - cfg["vit"] = probe_paligemma_vision(sf, keys, cfg_json, PFX_VIS_CANDIDATES, PFX_MMP_CANDIDATES) + cfg["vit"] = probe_paligemma_vision(sf, keys, cfg_json, [root + p for p in PFX_VIS_CANDIDATES], + [root + p for p in PFX_MMP_CANDIDATES]) v = cfg["vit"] print(f"vision: SigLIP hidden={v['vit_hidden']} layers={v['vit_layers']} " f"heads={v['vit_heads']} image={v['image_size']} patch={v['patch_size']} " @@ -233,15 +170,15 @@ def main() -> int: writer = open_writer(out, ARCH) write_pi_kv(writer, KV, cfg) - add(writer, "token_embd.weight", sf.get_tensor(PFX_VLM_HEAD)) - add(writer, "vlm.output_norm.weight", sf.get_tensor(f"{PFX_VLM}.norm.weight")) - write_decoder_blocks(writer, sf.get_tensor, PFX_VLM, "vlm", cfg["n_layers"]) + add(writer, "token_embd.weight", sf.get_tensor(head)) + add(writer, "vlm.output_norm.weight", sf.get_tensor(f"{vlm}.norm.weight")) + write_decoder_blocks(writer, sf.get_tensor, vlm, "vlm", cfg["n_layers"]) - add(writer, "aex.output_norm.weight", sf.get_tensor(f"{PFX_AEX}.norm.weight")) - write_decoder_blocks(writer, sf.get_tensor, PFX_AEX, "aex", cfg["n_layers"]) + add(writer, "aex.output_norm.weight", sf.get_tensor(f"{aex}.norm.weight")) + write_decoder_blocks(writer, sf.get_tensor, aex, "aex", cfg["n_layers"]) for suf in PROJ_SUFFIXES: - add(writer, suf, sf.get_tensor(f"{PFX_PROJ}.{suf}")) + add(writer, suf, sf.get_tensor(root + suf)) write_paligemma_vision(writer, sf, cfg["vit"]) diff --git a/scripts/convert_smolvla_to_gguf.py b/scripts/convert_smolvla_to_gguf.py index f7c6463..5c741d1 100644 --- a/scripts/convert_smolvla_to_gguf.py +++ b/scripts/convert_smolvla_to_gguf.py @@ -15,14 +15,10 @@ from __future__ import annotations -from pathlib import Path - -import numpy as np from safetensors import safe_open from gguf_blocks import ( - identity_stats, - load_processor_stats, + lerobot_stats, norm_eps, probe_siglip, write_decoder_blocks, @@ -77,29 +73,6 @@ def _probe_vision(sf, keys, cfg_json: dict) -> dict: v["n_img_tokens"] = (grid // scale) ** 2 return v -def _load_stats(ckpt: Path, state_dim: int, action_dim: int) -> dict[str, np.ndarray]: - - out = identity_stats(state_dim, action_dim) - got_state = load_processor_stats( - ckpt, - "policy_preprocessor.json", - "normalizer_processor", - "observation.state", - state_dim - ) - got_action = load_processor_stats( - ckpt, - "policy_postprocessor.json", - "unnormalizer_processor", - "action", - action_dim - ) - if got_state is not None: - out["state_mean"], out["state_std"] = got_state - if got_action is not None: - out["action_mean"], out["action_std"] = got_action - return out - def _add_kv(writer, cfg: dict) -> None: writer.add_uint32 (KV("hidden"), cfg["hidden"]) @@ -145,6 +118,9 @@ def main() -> int: require(sf_path) cfg_json = read_json(ckpt / "config.json") + for k in ("adapt_to_pi_aloha", "add_image_special_tokens"): + if cfg_json.get(k): + raise SystemExit(f"{k}=true is not supported") cfg = dict(SMOLLM2_500M) cfg["chunk_size"] = int(cfg_json["chunk_size"]) @@ -189,7 +165,8 @@ def main() -> int: f"vocab={cfg['vocab_size']} chunk={cfg['chunk_size']}") print("loading normalizer stats...") - stats = _load_stats(ckpt, cfg["real_state_dim"], cfg["real_action_dim"]) + stats = lerobot_stats(sf, ckpt, cfg["real_state_dim"], cfg["real_action_dim"], + cfg_json.get("normalization_mapping") or {}) cfg["vit"] = _probe_vision(sf, keys, cfg_json) v = cfg["vit"] diff --git a/scripts/convert_turbovla_to_gguf.py b/scripts/convert_turbovla_to_gguf.py index e910827..5cbd74a 100644 --- a/scripts/convert_turbovla_to_gguf.py +++ b/scripts/convert_turbovla_to_gguf.py @@ -28,16 +28,14 @@ import json from pathlib import Path -import numpy as np import torch from safetensors import safe_open import gguf - -F32 = gguf.GGMLQuantizationType.F32 -BF16 = gguf.GGMLQuantizationType.BF16 +from gguf_common import add as add_tensor, finish, kv_prefix, max_layer, open_writer ARCH = "turbovla" +kv = kv_prefix(ARCH) # Prefixes used by official TurboVLA checkpoints (without a "model." prefix). @@ -52,45 +50,12 @@ KEY_TEXT_PROJ = "text_encoder.text_projection" -def kv_prefix(name: str) -> str: - return f"{ARCH}.{name}" - - -def bf16_u16(t: torch.Tensor) -> np.ndarray: - return t.contiguous().view(torch.uint16).cpu().numpy() - - -def add_tensor(writer: gguf.GGUFWriter, name: str, t: torch.Tensor) -> None: - """Add tensor with preserved dtype.""" - if t.dtype == torch.float32: - writer.add_tensor(name, t.contiguous().cpu().numpy(), raw_dtype=F32) - elif t.dtype == torch.bfloat16: - writer.add_tensor(name, bf16_u16(t), raw_shape=list(t.shape), raw_dtype=BF16) - elif t.dtype == torch.float16: - writer.add_tensor(name, t.contiguous().cpu().numpy().astype(np.float32), raw_dtype=F32) - else: - raise NotImplementedError(f"unsupported dtype {t.dtype} for {name}") - - -def max_layer(keys: set[str], pfx: str) -> int: - """Count number of layers with given prefix.""" - m = -1 - for k in keys: - if k.startswith(pfx): - try: - m = max(m, int(k[len(pfx):].split(".", 1)[0])) - except ValueError: - pass - return m + 1 - - # facebook/dinov3-vitb16-pretrain-lvd1689m is gated, but only these architecture # values are needed: the fine-tuned weights ship inside the TurboVLA checkpoint. -DINOV3_VITB16 = {"rope_theta": 100.0, "num_register_tokens": 4} +DINOV3_VITB16 = {"rope_theta": 100.0, "num_register_tokens": 4, "num_attention_heads": 12} -# Tensors the runtime never reads: DINOv3's final norm (TurboVLA taps -# hidden_states[-1], before it), BERT's pooler, and the MAE mask token. -UNUSED = (f"{PREFIX_VIT}.norm.", f"{PREFIX_TEXT}.pooler.", f"{PREFIX_VIT}.embeddings.mask_token") +# Tensors the runtime never reads: BERT's pooler and the MAE mask token. +UNUSED = (f"{PREFIX_TEXT}.pooler.", f"{PREFIX_VIT}.embeddings.mask_token") class TrackedTensors(dict): @@ -108,7 +73,7 @@ def __getitem__(self, key): def load_checkpoint(ckpt: Path) -> tuple[TrackedTensors, dict]: """Return (state dict, TurboVLA model_config) from a .pth or a directory.""" if ckpt.is_file(): - blob = torch.load(str(ckpt), map_location="cpu", weights_only=False) + blob = torch.load(str(ckpt), map_location="cpu", weights_only=True) if not isinstance(blob, dict) or "model_state_dict" not in blob: raise SystemExit(f"{ckpt} is not a TurboVLA checkpoint (no model_state_dict)") state = blob["model_state_dict"] @@ -232,6 +197,9 @@ def __init__(self, tensors: dict[str, torch.Tensor], keys: set[str], cfg_json: d ) if int(self.dinov3_cfg.get("num_register_tokens", self.num_register_tokens)) != self.num_register_tokens: raise SystemExit("DINOv3 config num_register_tokens disagrees with checkpoint weights") + self.vit_heads = int(self.dinov3_cfg.get("num_attention_heads", 0)) + if self.vit_heads <= 0 or self.vit_dim % self.vit_heads: + raise SystemExit("DINOv3 config num_attention_heads is missing or does not divide the ViT width") word_emb = self._get(f"{PREFIX_TEXT}.embeddings.word_embeddings.weight") self.vocab_size = int(word_emb.shape[0]) @@ -340,6 +308,9 @@ def write_vision_encoder(writer: gguf.GGUFWriter, tensors: dict, dims: TurboVLAD add_tensor(writer, f"vit.blk.{i}.fc2.weight", w_fc2) add_tensor(writer, f"vit.blk.{i}.fc2.bias", b_fc2) + add_tensor(writer, "vit.norm.weight", tensors[f"{root}.norm.weight"]) + add_tensor(writer, "vit.norm.bias", tensors[f"{root}.norm.bias"]) + def write_text_encoder(writer: gguf.GGUFWriter, tensors: dict, dims: TurboVLADims) -> None: """Write BERT text encoder.""" @@ -508,7 +479,7 @@ def write_text_groups(writer: gguf.GGUFWriter, text_cfg: dict) -> None: sees token ids, so the table is keyed by them. """ groups = text_cfg.get("padding_length_by_instruction") or {} - writer.add_uint32(kv_prefix("text_groups.count"), len(groups)) + writer.add_uint32(kv("text_groups.count"), len(groups)) if not groups: return from transformers import AutoTokenizer @@ -520,9 +491,9 @@ def write_text_groups(writer: gguf.GGUFWriter, text_cfg: dict) -> None: ids += [int(t) for t in seq] lengths.append(len(seq)) pad_to.append(int(length)) - writer.add_array(kv_prefix("text_groups.tokens"), ids) - writer.add_array(kv_prefix("text_groups.lengths"), lengths) - writer.add_array(kv_prefix("text_groups.pad_to"), pad_to) + writer.add_array(kv("text_groups.tokens"), ids) + writer.add_array(kv("text_groups.lengths"), lengths) + writer.add_array(kv("text_groups.pad_to"), pad_to) def verify_consumed_tensors(tensors: TrackedTensors) -> None: @@ -563,17 +534,21 @@ def main() -> int: dims = TurboVLADims(tensors, keys, cfg_json, dinov3_cfg) print(f" Detected: {dims}") - print(f"Writing GGUF to {out}...") - out.parent.mkdir(parents=True, exist_ok=True) - writer = gguf.GGUFWriter(str(out), ARCH) - - kv = kv_prefix - writer.add_string(kv("architecture"), ARCH) + text_cfg = cfg_json.get("text", {}) + inter_cfg = cfg_json.get("interaction", {}) + if (inter_cfg.get("residual_style", "normalized") != "normalized" + or inter_cfg.get("padding_strategy", "key_padding_mask") != "key_padding_mask" + or text_cfg.get("zero_padded_tokens", False) + or not text_cfg.get("sub_sentence_present", True)): + raise SystemExit("unsupported TurboVLA variant: the runtime implements residual_style=normalized, " + "padding_strategy=key_padding_mask, zero_padded_tokens=false, sub_sentence_present=true") + + writer = open_writer(out, ARCH) writer.add_uint32(kv("hidden"), dims.hidden_dim) writer.add_uint32(kv("vit_dim"), dims.vit_dim) writer.add_uint32(kv("vit_layers"), dims.vit_layers) - writer.add_uint32(kv("vit_head_dim"), 64) - writer.add_uint32(kv("vit_heads"), 12) + writer.add_uint32(kv("vit_head_dim"), dims.vit_dim // dims.vit_heads) + writer.add_uint32(kv("vit_heads"), dims.vit_heads) writer.add_uint32(kv("text_dim"), dims.text_dim) writer.add_uint32(kv("text_layers"), dims.text_layers) writer.add_uint32(kv("text_head_dim"), 64) @@ -608,7 +583,6 @@ def main() -> int: # TurboVLA's BERT wrapper uses these exact punctuation IDs when it creates # sub-sentence attention masks. Persist them so the GGUF runtime does not # silently depend on a tokenizer installation. - text_cfg = cfg_json.get("text", {}) model_name = text_cfg.get("model_name_or_path", "bert-base-uncased") if model_name not in ("bert-base-uncased", "google-bert/bert-base-uncased"): raise SystemExit( @@ -650,14 +624,7 @@ def main() -> int: print(" Verifying tensor consumption...") verify_consumed_tensors(tensors) - writer.write_header_to_file() - writer.write_kv_data_to_file() - writer.write_tensors_to_file() - writer.close() - - size_mb = out.stat().st_size / (1024 * 1024) - print(f" Done! Output: {out} ({size_mb:.1f} MiB)") - return 0 + return finish(writer, out) if __name__ == "__main__": diff --git a/scripts/gguf_blocks.py b/scripts/gguf_blocks.py index 1a3536c..917c0f6 100644 --- a/scripts/gguf_blocks.py +++ b/scripts/gguf_blocks.py @@ -85,6 +85,9 @@ def write_siglip_tower( add_n(writer, "vit.post_ln.weight", g(f"{root}.post_layernorm.weight")); add_n(writer, "vit.post_ln.bias", g(f"{root}.post_layernorm.bias")) +def pi_root(keys) -> str: + return "model." if "model.paligemma_with_expert.paligemma.lm_head.weight" in keys else "" + def probe_paligemma_vision(sf, keys, cfg_json: dict, vis_candidates, mmp_candidates) -> dict: vis = next((p for p in vis_candidates if f"{p}.embeddings.patch_embedding.weight" in keys), None) @@ -125,7 +128,7 @@ def write_pi_kv(writer, kv, cfg: dict, adarms: bool = False) -> None: if adarms: writer.add_bool (kv("use_adarms_expert"), True) writer.add_uint32(kv("adarms_cond_dim"), cfg["expert_h"]) - writer.add_string(kv("norm_mode"), "quantiles") + writer.add_string(kv("norm_mode"), cfg["norm_mode"]) writer.add_float64 (kv("min_period"), cfg["min_period"]) writer.add_float64 (kv("max_period"), cfg["max_period"]) writer.add_float64 (kv("rope_theta"), cfg["rope_theta"]) @@ -335,12 +338,12 @@ def load_processor_stats(ckpt: Path, meta_json: str, registry: str, key: str, di meta_path = ckpt / meta_json if not meta_path.exists(): - print(f" stats: {meta_json} missing - using identity for {key}") + print(f" stats: {meta_json} missing") return None try: meta = json.loads(meta_path.read_text()) except Exception as e: - print(f" stats: {meta_json} parse failed ({e}) - using identity for {key}") + print(f" stats: {meta_json} parse failed ({e})") return None state_file = None @@ -349,29 +352,68 @@ def load_processor_stats(ckpt: Path, meta_json: str, registry: str, key: str, di state_file = step.get("state_file") break if not state_file: - print(f" stats: no {registry} step in {meta_json} - using identity for {key}") + print(f" stats: no {registry} step in {meta_json}") return None sf_path = ckpt / state_file if not sf_path.is_file(): - print(f" stats: {sf_path.name} referenced by {meta_json} but missing - using identity for {key}") + print(f" stats: {sf_path.name} referenced by {meta_json} but missing") return None with safe_open(str(sf_path), framework="pt") as f: keys = set(f.keys()) mk, sk = f"{key}.mean", f"{key}.std" if mk not in keys or sk not in keys: - print(f" stats: {sf_path.name} lacks {mk}/{sk} - using identity for {key}") + print(f" stats: {sf_path.name} lacks {mk}/{sk}") return None mean = f.get_tensor(mk).float().numpy().reshape(-1) std = f.get_tensor(sk).float().numpy().reshape(-1) if mean.size != dim or std.size != dim: - print(f" stats: {mk} dim mismatch ({mean.size} vs {dim}) in {sf_path.name} - using identity") + print(f" stats: {mk} dim mismatch ({mean.size} vs {dim}) in {sf_path.name}") return None print(f" stats: loaded {key} from {sf_path.name} ({mk}/{sk})") return mean.astype(np.float32, copy=False), std.astype(np.float32, copy=False) +def lerobot_stats(sf, ckpt: Path, state_dim: int, action_dim: int, norm_map: dict) -> dict[str, np.ndarray]: + + out = identity_stats(state_dim, action_dim) + keys = set(sf.keys()) + + def _legacy(pfx: str, dim: int): + mk, sk = f"{pfx}.mean", f"{pfx}.std" + if mk not in keys or sk not in keys: + return None + mean = sf.get_tensor(mk).float().numpy().reshape(-1) + std = sf.get_tensor(sk).float().numpy().reshape(-1) + if mean.size != dim or std.size != dim: + print(f" stats: legacy {mk} dim mismatch ({mean.size} vs {dim})") + return None + print(f" stats: loaded {pfx} from model.safetensors [legacy]") + return mean.astype(np.float32, copy=False), std.astype(np.float32, copy=False) + + feats = ( + ("STATE", "state", "policy_preprocessor.json", "normalizer_processor", "observation.state", state_dim, + ("normalize_inputs.buffer_observation_state",)), + ("ACTION", "action", "policy_postprocessor.json", "unnormalizer_processor", "action", action_dim, + ("unnormalize_outputs.buffer_action", "normalize_targets.buffer_action")), + ) + for ftype, dst, meta_json, registry, key, dim, legacy in feats: + mode = norm_map.get(ftype, "MEAN_STD") + if mode == "IDENTITY": + continue + if mode != "MEAN_STD": + raise SystemExit(f"normalization_mapping {ftype}={mode} is not supported (MEAN_STD or IDENTITY only)") + got = load_processor_stats(ckpt, meta_json, registry, key, dim) + for pfx in legacy: + if got is None: + got = _legacy(pfx, dim) + if got is None: + raise SystemExit(f"no {key} mean/std in {meta_json} or model.safetensors ({', '.join(legacy)}); " + f"refusing to bake identity stats for MEAN_STD {ftype}") + out[f"{dst}_mean"], out[f"{dst}_std"] = got + return out + def identity_stats(state_dim: int, action_dim: int) -> dict[str, np.ndarray]: return { "state_mean": np.zeros(state_dim, dtype=np.float32), diff --git a/scripts/gguf_common.py b/scripts/gguf_common.py index 37541d2..b290015 100644 --- a/scripts/gguf_common.py +++ b/scripts/gguf_common.py @@ -20,14 +20,19 @@ import argparse import json +import re from pathlib import Path import numpy as np -import torch -from safetensors import safe_open import gguf +try: + import torch + from safetensors import safe_open +except ImportError: + pass + F32 = gguf.GGMLQuantizationType.F32 BF16 = gguf.GGMLQuantizationType.BF16 @@ -42,6 +47,8 @@ def add(writer: gguf.GGUFWriter, name: str, t: torch.Tensor) -> None: writer.add_tensor(name, t.contiguous().cpu().numpy(), raw_dtype=F32) elif t.dtype == torch.bfloat16: writer.add_tensor(name, bf16_u16(t), raw_shape=list(t.shape), raw_dtype=BF16) + elif t.dtype == torch.float16: + writer.add_tensor(name, t.contiguous().cpu().numpy().astype(np.float32), raw_dtype=F32) else: raise NotImplementedError(f"unsupported dtype {t.dtype} for {name}") @@ -54,6 +61,22 @@ def add_bf16(writer: gguf.GGUFWriter, name: str, t: torch.Tensor) -> None: def add_array(writer: gguf.GGUFWriter, name: str, a: np.ndarray) -> None: writer.add_tensor(name, np.ascontiguousarray(a, dtype=np.float32), raw_dtype=F32) +def copy_kv(reader: gguf.GGUFReader, writer: gguf.GGUFWriter, skip=()) -> None: + meta = {"GGUF.version", "GGUF.tensor_count", "GGUF.kv_count", "general.architecture", *skip} + for name, f in reader.fields.items(): + if name in meta: + continue + sub = f.types[-1] if f.types[0] == gguf.GGUFValueType.ARRAY else None + writer.add_key_value(name, f.contents(), f.types[0], sub_type=sub) + +def copy_tensor(writer: gguf.GGUFWriter, t) -> None: + data = np.ascontiguousarray(t.data) + if t.tensor_type == BF16: + data = data.view(np.uint16) + elif t.tensor_type == F32: + data = data.astype(np.float32, copy=False) + writer.add_tensor(t.name, data, raw_dtype=t.tensor_type) + def kv_u32(writer: gguf.GGUFWriter, kv, values: dict) -> None: for k, v in values.items(): writer.add_uint32(kv(k), int(v)) @@ -87,10 +110,20 @@ def load_safetensors(ckpt: Path, keep: tuple[str, ...] | None = None) -> dict[st def load_pt_module(path: Path) -> dict[str, torch.Tensor]: - sd = torch.load(str(path), map_location="cpu", weights_only=False) + sd = torch.load(str(path), map_location="cpu", weights_only=True) pfx = "module." return {(k[len(pfx):] if k.startswith(pfx) else k): v.contiguous() for k, v in sd.items()} +def find_sidecar(ckpt: Path, stem: str) -> Path: + + cands = sorted( + ckpt.glob(f"{stem}--*checkpoint.pt"), + key=lambda p: int(m.group(1)) if (m := re.search(r"--(\d+)_checkpoint\.pt$", p.name)) else -1, + ) + if not cands: + raise SystemExit(f"no {stem}--*checkpoint.pt in {ckpt}") + return cands[-1] + def read_json(path: Path) -> dict: if not path.exists(): raise SystemExit(f"missing {path}") diff --git a/scripts/install_ov.sh b/scripts/install_ov.sh index 28d450c..310df2d 100644 --- a/scripts/install_ov.sh +++ b/scripts/install_ov.sh @@ -21,8 +21,8 @@ need_cmd() { # Digests of the two archives the defaults below pin. These land in /opt under # sudo, so a bad download is a root-level problem. -OPENVINO_SHA256_2204="d701a115d3dc18088ff75b5b8e67a51fbf780022a3d40ee8ee7f2adfbd9915e6" -OPENVINO_SHA256_2404="6931e5a3c9b1fc9cb170137196df2c40489625703f2d184f511b7add2c110ef8" +OPENVINO_SHA256_2204="d327ede0a5dd29ad6e73d156aa7fe43b7d8fc0eac5e789f17bf9d381ab333e7d" +OPENVINO_SHA256_2404="0bd86d578beb1e8805655f593315c69bf909008e212f801ed76e3a350927736f" # verify_sha256 . An overridden version has no # digest here, so fall back to the one the mirror publishes: that catches a @@ -127,6 +127,10 @@ install_gpu_2204() { local igc_base_url="https://github.com/intel/intel-graphics-compiler/releases/download/v2.10.8" local crt_base_url="https://github.com/intel/compute-runtime/releases/download/25.13.33276.16" local checksum_file="ww13.sum" + local igc_sums=( + "85bb185f5c9f0700321c6ce9362a4aa85b6cdf02f74a7b8685424d0be024fb0e intel-igc-core-2_2.10.8+18926_amd64.deb" + "21e5ee9f0b5798335d82a88f98e80bccd9539380782e60893d6d050eb801d322 intel-igc-opencl-2_2.10.8+18926_amd64.deb" + ) local packages=( "${igc_base_url}/intel-igc-core-2_2.10.8+18926_amd64.deb" "${igc_base_url}/intel-igc-opencl-2_2.10.8+18926_amd64.deb" @@ -146,7 +150,8 @@ install_gpu_2204() { done wget --no-continue "${crt_base_url}/${checksum_file}" - sha256sum --ignore-missing -c "${checksum_file}" + sha256sum -c "${checksum_file}" + printf '%s\n' "${igc_sums[@]}" | sha256sum -c - shopt -s nullglob local artifacts=( *.deb *.ddeb ) @@ -167,6 +172,8 @@ install_npu_2204() { local npu_url="https://github.com/intel/linux-npu-driver/releases/download/v1.26.0/${npu_tarball}" local level_zero_deb="level-zero_1.24.2+u22.04_amd64.deb" local level_zero_url="https://github.com/oneapi-src/level-zero/releases/download/v1.24.2/${level_zero_deb}" + local npu_sha256="cfdbcc9adc1ea20d498ebd9cbdb5c212f6fc940e1034ef7a72e239a8636f653a" + local level_zero_sha256="7c304e93835d96025c90f6a3d8f2ce5edf142c24da9f4871113a4f0225fef22e" log "Installing Intel NPU drivers for Ubuntu 22.04..." mkdir -p "${download_dir}" @@ -176,6 +183,8 @@ install_npu_2204() { # at all if the fetch fails. wget --no-continue "${npu_url}" wget --no-continue "${level_zero_url}" + verify_sha256 "${npu_tarball}" "${npu_sha256}" "${npu_url}" + verify_sha256 "${level_zero_deb}" "${level_zero_sha256}" "${level_zero_url}" tar -xf "${npu_tarball}" mapfile -t npu_debs < <(find . -type f -name '*.deb' ! -name 'level-zero*.deb' | sort) @@ -201,8 +210,8 @@ install_npu_2204() { install_runtime_2204() { local download_dir="${WORK_DIR}/openvino_runtime_2204" - local openvino_version="${OPENVINO_VERSION:-2025.3}" - local openvino_build="${OPENVINO_BUILD:-19807.44526285f24}" + local openvino_version="${OPENVINO_VERSION:-2026.4}" + local openvino_build="${OPENVINO_BUILD:-22959.99c81491cc3}" local openvino_archive="openvino_toolkit_ubuntu22_${openvino_version}.0.${openvino_build}_x86_64.tgz" local openvino_dirname="openvino_toolkit_ubuntu22_${openvino_version}.0.${openvino_build}_x86_64" local openvino_url="https://storage.openvinotoolkit.org/repositories/openvino/packages/${openvino_version}/linux/${openvino_archive}" @@ -211,7 +220,7 @@ install_runtime_2204() { local symlink_path="${install_root}/openvino" local archive_path="${download_dir}/openvino_${openvino_version}.tgz" local expected_sha="" - if [[ "${openvino_version}" == "2025.3" && "${openvino_build}" == "19807.44526285f24" ]]; then + if [[ "${openvino_version}" == "2026.4" && "${openvino_build}" == "22959.99c81491cc3" ]]; then expected_sha="${OPENVINO_SHA256_2204}" fi @@ -238,19 +247,23 @@ install_runtime_2204() { install_gpu_2404() { local download_dir="${WORK_DIR}/intel_gpu_2404" - local igc_base_url="https://github.com/intel/intel-graphics-compiler/releases/download/v2.36.3" - local crt_base_url="https://github.com/intel/compute-runtime/releases/download/26.22.38646.4" - local checksum_file="ww22.sum" + local igc_base_url="https://github.com/intel/intel-graphics-compiler/releases/download/v2.41.5" + local crt_base_url="https://github.com/intel/compute-runtime/releases/download/26.35.39758.10" + local checksum_file="ww35.sum" + local igc_sums=( + "0a6e64a663ae65a0fa02d6912ae3b6b37cf85b90c21cc423fd9fef70aaf4f628 intel-igc-core-2_2.41.5+22716_amd64.deb" + "779e1b9e88098eb25711e9a8f67c2752665bad22f134aa40ed5649f6e1b87058 intel-igc-opencl-2_2.41.5+22716_amd64.deb" + ) local packages=( - "${igc_base_url}/intel-igc-core-2_2.36.3+21719_amd64.deb" - "${igc_base_url}/intel-igc-opencl-2_2.36.3+21719_amd64.deb" - "${crt_base_url}/intel-ocloc-dbgsym_26.22.38646.4-0_amd64.ddeb" - "${crt_base_url}/intel-ocloc_26.22.38646.4-0_amd64.deb" - "${crt_base_url}/intel-opencl-icd-dbgsym_26.22.38646.4-0_amd64.ddeb" - "${crt_base_url}/intel-opencl-icd_26.22.38646.4-0_amd64.deb" + "${igc_base_url}/intel-igc-core-2_2.41.5+22716_amd64.deb" + "${igc_base_url}/intel-igc-opencl-2_2.41.5+22716_amd64.deb" + "${crt_base_url}/intel-ocloc-dbgsym_26.35.39758.10-0_amd64.ddeb" + "${crt_base_url}/intel-ocloc_26.35.39758.10-0_amd64.deb" + "${crt_base_url}/intel-opencl-icd-dbgsym_26.35.39758.10-0_amd64.ddeb" + "${crt_base_url}/intel-opencl-icd_26.35.39758.10-0_amd64.deb" "${crt_base_url}/libigdgmm12_22.10.0_amd64.deb" - "${crt_base_url}/libze-intel-gpu1-dbgsym_26.22.38646.4-0_amd64.ddeb" - "${crt_base_url}/libze-intel-gpu1_26.22.38646.4-0_amd64.deb" + "${crt_base_url}/libze-intel-gpu1-dbgsym_26.35.39758.10-0_amd64.ddeb" + "${crt_base_url}/libze-intel-gpu1_26.35.39758.10-0_amd64.deb" ) log "Installing Intel GPU drivers for Ubuntu 24.04..." @@ -262,7 +275,8 @@ install_gpu_2404() { done wget --no-continue "${crt_base_url}/${checksum_file}" - sha256sum --ignore-missing -c "${checksum_file}" + sha256sum -c "${checksum_file}" + printf '%s\n' "${igc_sums[@]}" | sha256sum -c - shopt -s nullglob local artifacts=( *.deb *.ddeb ) @@ -279,9 +293,10 @@ install_gpu_2404() { install_npu_2404() { local download_dir="${WORK_DIR}/intel_npu_2404" - local npu_release="v1.33.0" - local npu_archive="linux-npu-driver-v1.33.0.20260529-26625960453-ubuntu2404.tar.gz" + local npu_release="v1.38.0" + local npu_archive="linux-npu-driver-v1.38.0.20260910-34487311128-ubuntu2404.tar.gz" local npu_url="https://github.com/intel/linux-npu-driver/releases/download/${npu_release}/${npu_archive}" + local npu_sha256="1efcd4b60c22abee751d8f2705962cbcc2a569de45c7e0e670cf08afbfcdc1d2" local npu_packages=( intel-driver-compiler-npu intel-fw-npu @@ -296,6 +311,7 @@ install_npu_2404() { # Download before purging, so a failed fetch does not leave the machine with # no NPU driver at all. wget --no-continue "${npu_url}" + verify_sha256 "${npu_archive}" "${npu_sha256}" "${npu_url}" tar -xf "${npu_archive}" shopt -s nullglob @@ -321,8 +337,8 @@ install_npu_2404() { install_runtime_2404() { local download_dir="${WORK_DIR}/openvino_runtime_2404" - local openvino_version="${OPENVINO_VERSION:-2026.2.1}" - local openvino_build="${OPENVINO_BUILD:-21919.ede283a88e3}" + local openvino_version="${OPENVINO_VERSION:-2026.4}" + local openvino_build="${OPENVINO_BUILD:-0.22959.99c81491cc3}" local openvino_archive="openvino_toolkit_ubuntu24_${openvino_version}.${openvino_build}_x86_64.tgz" local openvino_dirname="openvino_toolkit_ubuntu24_${openvino_version}.${openvino_build}_x86_64" local openvino_url="https://storage.openvinotoolkit.org/repositories/openvino/packages/${openvino_version}/linux/${openvino_archive}" @@ -331,7 +347,7 @@ install_runtime_2404() { local symlink_path="${install_root}/openvino" local archive_path="${download_dir}/openvino_${openvino_version}.tgz" local expected_sha="" - if [[ "${openvino_version}" == "2026.2.1" && "${openvino_build}" == "21919.ede283a88e3" ]]; then + if [[ "${openvino_version}" == "2026.4" && "${openvino_build}" == "0.22959.99c81491cc3" ]]; then expected_sha="${OPENVINO_SHA256_2404}" fi diff --git a/scripts/patch_ggml_cuda_ext_hook.py b/scripts/patch_ggml_cuda_ext_hook.py index d02bcb4..cf6aee4 100644 --- a/scripts/patch_ggml_cuda_ext_hook.py +++ b/scripts/patch_ggml_cuda_ext_hook.py @@ -33,10 +33,10 @@ 1. Two exported function pointers, null by default. 2. One call to the first at the top of ggml_cuda_compute_forward. Returning false means "not mine", and ggml runs the op exactly as before. - 3. The RMS_NORM+MUL fusion check GGML_ASSERTs F32 rather than declining, so a - BF16 rms_norm aborts the process before dispatch is ever reached. Those two - asserts become a return, which is what the surrounding checks already do - for every other unsupported type. + 3. The RMS_NORM+MUL and RMS_NORM+SCALE fusion checks GGML_ASSERT F32 rather + than declining, so a BF16 rms_norm aborts the process before dispatch is + ever reached. Each pair of asserts becomes a return, which is what the + surrounding checks already do for every other unsupported type. 4. One call to the second in the ADD/MUL fusion branch of ggml_cuda_try_fuse. Fusion happens in ggml_backend_cuda_graph_compute, upstream of ggml_cuda_compute_forward, so the hook in (2) never sees a fused node -- @@ -121,11 +121,11 @@ def main(): if MARKER in text: return # idempotent: re-configure over an already-patched tree - for old, new in (HOOK_DECL, FUSION_GUARD, FUSED_BINBCAST_GUARD): + for old, new, want in ((*HOOK_DECL, 1), (*FUSION_GUARD, 2), (*FUSED_BINBCAST_GUARD, 1)): n = text.count(old) - if n != 1: + if n != want: raise SystemExit( - f"{path}: anchor found {n} times, expected 1. The pinned llama.cpp " + f"{path}: anchor found {n} times, expected {want}. The pinned llama.cpp " f"probably moved; re-check this anchor against the new tag.\n" f"---\n{old[:400]}\n---" ) diff --git a/scripts/patch_ggml_openvino.py b/scripts/patch_ggml_openvino.py index 0bbffba..5fa4cc1 100755 --- a/scripts/patch_ggml_openvino.py +++ b/scripts/patch_ggml_openvino.py @@ -74,6 +74,15 @@ patching only the member applies cleanly and does nothing at all. Both are patched here so the fix survives whichever one upstream keeps. + openvino/op/rope.cpp - key the per-op sin/cos cache on the position input. + b11223's translate_rope() caches each sin/cos table under the ROPE's + op_params alone, so two ROPEs with the same parameters share one table + whatever their positions. pi0, pi0.5 and SmolVLA rotate the prefix and the + action suffix with the same parameters and different position tensors, and + the suffix then multiplies by the prefix's table: + "Multiply (Reshape[0]:f32[1,50,1,256], Concat[0]:f32[1,528,1,256]) + Argument shapes are inconsistent." on every OpenVINO device. + 5. utils.{h,cpp} - cache what the naive path compiles. The dynamic and static paths keep a `graph_key`-indexed cache of the decoder and the compiled infer request; the naive path has none, so it @@ -83,12 +92,12 @@ 1.4 s per prediction with the cache in place. A hit rebinds the cached decoder to the new graph through the existing update_io(), which is how the dynamic path already handles freshly built tensors. - The key is `naive_key`, not `graph_key`: the latter is n_nodes plus the - first and last node name, which two graphs of the same size can share, and - a compiled model is bound to the shapes it was built for. Reusing one - across a shape change returns another graph's answer with no error, so the - key mixes in every node's op and shape. The map is bounded; see the comment - on the flush. + The key is `naive_key`, not `graph_key`: the latter is n_nodes, the first + and last node name and the input names, which two graphs of the same size + can share, and a compiled model is bound to the shapes it was built for. + Reusing one across a shape change returns another graph's answer with no + error, so the key mixes in every node's op and shape. The map is bounded; + see the comment on the flush. 6. openvino/op_table.cpp - get both GELU flavours right. ggml has two: GGML_UNARY_OP_GELU is the tanh approximation, GGML_UNARY_OP_ @@ -241,11 +250,13 @@ NAIVE_COMPUTE_OLD = """enum ggml_status naive_compute(ggml_cgraph * cgraph, ov::Core & core, const std::string & device, - const ov::AnyMap & config) { + const ov::AnyMap & config, + ov_compiled_model_cache & cache) { if (cgraph->n_nodes == 1 && (cgraph->nodes[0]->op == GGML_OP_NONE || cgraph->nodes[0]->op == GGML_OP_VIEW)) { return GGML_STATUS_SUCCESS; } + std::unique_lock compile_lock(cache.mutex); bool naive = true; auto model_weights = GgmlOvDecoder::create_weight_nodes(cgraph, naive); auto decoder = std::make_shared(cgraph, model_weights); @@ -257,46 +268,59 @@ std::shared_ptr infer_request; auto remote_context = ggml_openvino_get_remote_context(); + ov::AnyMap compile_config = config; if (cgraph->nodes[0]->op == GGML_OP_MUL_MAT) { // TODO ACCURACY hint triggers a bug in GPU plugin/driver on Lunar Lake. Remove once CVS-182166 is resolved - core.set_property(device, ov::hint::execution_mode(ov::hint::ExecutionMode::PERFORMANCE)); + compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::PERFORMANCE; } else { - core.set_property(device, ov::hint::execution_mode(ov::hint::ExecutionMode::ACCURACY)); + compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::ACCURACY; } if (remote_context.has_value()) { infer_request = std::make_shared( - core.compile_model(model, remote_context.value(), config).create_infer_request()); + core.compile_model(model, remote_context.value(), compile_config).create_infer_request()); } else { - infer_request = - std::make_shared(core.compile_model(model, device, config).create_infer_request()); + infer_request = std::make_shared( + core.compile_model(model, device, compile_config).create_infer_request()); } - - auto ov_params = model->get_parameters();""" + std::vector input_names; + std::vector output_names; + for (const auto & param : model->get_parameters()) { + input_names.push_back(param->get_friendly_name()); + } + for (const auto & result : model->get_results()) { + output_names.push_back(result->get_friendly_name()); + } + // Destroy the frontend graph under the compilation lock as well: it can + // still own edges into the shared weight nodes. + model.reset(); + input_model.reset(); + decoder->clear_model_weights(); + model_weights.clear(); + compile_lock.unlock(); +""" NAIVE_COMPUTE_NEW = """enum ggml_status naive_compute(ggml_cgraph * cgraph, ov::Core & core, const std::string & device, const ov::AnyMap & config, - std::shared_ptr r_ctx) { + const std::shared_ptr & r_ctx) { if (cgraph->n_nodes == 1 && (cgraph->nodes[0]->op == GGML_OP_NONE || cgraph->nodes[0]->op == GGML_OP_VIEW)) { return GGML_STATUS_SUCCESS; } - // vla.cpp: reuse the decoder, the converted model and the compiled infer - // request across calls on the same graph, the way the dynamic and static - // paths already do. Conversion plus compile_model dominates a naive call, so - // without this every graph_compute pays it again. + // vla.cpp: reuse the decoder and the compiled infer request across calls on + // the same graph, the way the dynamic and static paths already do. + // Conversion plus compile_model dominates a naive call, so without this every + // graph_compute pays it again. static const bool cache_enabled = !ggml_openvino_getenv_int("GGML_OPENVINO_DISABLE_CACHE"); - const naive_key key(cgraph); std::shared_ptr entry; - bool cache_hit = false; - if (cache_enabled && r_ctx != nullptr) { + if (cache_enabled) { + const naive_key key(cgraph); std::lock_guard lock(r_ctx->ctx_mutex); auto it = r_ctx->naive_cache.find(key); if (it != r_ctx->naive_cache.end()) { entry = it->second; - cache_hit = true; } else { // Each entry holds a compiled model, so this cannot grow forever. // Flush rather than evict: a caller sees a handful of shapes, and an @@ -307,55 +331,68 @@ entry = std::make_shared(); r_ctx->naive_cache[key] = entry; } - } else { - entry = std::make_shared(); } - // One graph at a time: an ov::InferRequest is not re-entrant, and a hit - // rebinds the decoder to this cgraph. - std::lock_guard entry_lock(entry->mutex); - - bool naive = true; std::shared_ptr decoder; - std::shared_ptr model; std::shared_ptr infer_request; + std::vector input_names; + std::vector output_names; - if (cache_hit && entry->infer_request != nullptr) { + if (entry != nullptr && entry->infer_request != nullptr) { decoder = entry->decoder; - model = entry->model; infer_request = entry->infer_request; + input_names = entry->input_names; + output_names = entry->output_names; // Same shapes, new tensors: point the decoder at this call's graph. decoder->update_io(cgraph); } else { + std::unique_lock compile_lock(r_ctx->compiled_cache->mutex); + bool naive = true; auto model_weights = GgmlOvDecoder::create_weight_nodes(cgraph, naive); decoder = std::make_shared(cgraph, model_weights); auto input_model = std::make_shared(decoder); - model = ov::frontend::ggml::FrontEnd::convert(input_model, naive); + auto model = ov::frontend::ggml::FrontEnd::convert(input_model, naive); if (ggml_openvino_getenv_int("GGML_OPENVINO_DUMP_IR")) { ov::serialize(model, "IR_naive.xml"); } auto remote_context = ggml_openvino_get_remote_context(); + ov::AnyMap compile_config = config; if (cgraph->nodes[0]->op == GGML_OP_MUL_MAT) { // TODO ACCURACY hint triggers a bug in GPU plugin/driver on Lunar Lake. Remove once CVS-182166 is resolved - core.set_property(device, ov::hint::execution_mode(ov::hint::ExecutionMode::PERFORMANCE)); + compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::PERFORMANCE; } else { - core.set_property(device, ov::hint::execution_mode(ov::hint::ExecutionMode::ACCURACY)); + compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::ACCURACY; } if (remote_context.has_value()) { infer_request = std::make_shared( - core.compile_model(model, remote_context.value(), config).create_infer_request()); + core.compile_model(model, remote_context.value(), compile_config).create_infer_request()); } else { - infer_request = - std::make_shared(core.compile_model(model, device, config).create_infer_request()); + infer_request = std::make_shared( + core.compile_model(model, device, compile_config).create_infer_request()); + } + for (const auto & param : model->get_parameters()) { + input_names.push_back(param->get_friendly_name()); + } + for (const auto & result : model->get_results()) { + output_names.push_back(result->get_friendly_name()); + } + // Destroy the frontend graph under the compilation lock as well: it can + // still own edges into the shared weight nodes. + model.reset(); + input_model.reset(); + decoder->clear_model_weights(); + model_weights.clear(); + compile_lock.unlock(); + + if (entry != nullptr) { + entry->decoder = decoder; + entry->infer_request = infer_request; + entry->input_names = input_names; + entry->output_names = output_names; } - - entry->decoder = decoder; - entry->model = model; - entry->infer_request = infer_request; } - - auto ov_params = model->get_parameters();""" +""" # file -> [(anchor, replacement), ...]. Every anchor must match exactly once. EDITS = { @@ -383,11 +420,11 @@ (USM_LOOKUP % (("clEnqueueMemcpyINTEL",) * 2), USM_LOOKUP_NEW % (("clEnqueueMemcpyINTEL",) * 2)), ( """ "GGML_OPENVINO_LOG_UNSUPPORTED_OPS", - };""", +""", """ "GGML_OPENVINO_LOG_UNSUPPORTED_OPS", // vla.cpp: f16 (default) or f32 for the GPU plugin's inference precision. "GGML_OPENVINO_GPU_PRECISION", - };""", +""", ), ( """ } else if (cache_dir && strlen(cache_dir) > 0) { @@ -417,27 +454,35 @@ ], "ggml/src/ggml-openvino/openvino/op/add.cpp": [ ( - """ auto input_0 = process_view_input_new(context, 0); - auto input_1 = process_view_input_new(context, 1); - auto res = std::make_shared(input_0, input_1);""", - """ auto input_0 = process_view_input_new(context, 0); - auto input_1 = process_view_input_new(context, 1); - - // vla.cpp: re-hang the outer add on the inner one's non-GEMM operand so the + """ ov::Output res = std::make_shared(input_0, input_1);""", + """ // vla.cpp: re-hang the outer add on the inner one's non-GEMM operand so the // GEMM is left with a single post-op. Addition is associative. + ov::Output res; const int oc = context.get_op_case(); - if (oc == 2 || oc == 3) { - auto inner = input_0.get_node_shared_ptr(); - if (inner->get_input_size() == 2) { - const size_t keep = (oc == 2) ? 0 : 1; - const size_t fold = 1 - keep; - auto folded = std::make_shared(inner->input_value(fold), input_1); - auto res2 = std::make_shared(inner->input_value(keep), folded); - return rename_outputs_with_suffix({res2}, context.get_name()); + auto inner = input_0.get_node_shared_ptr(); + if ((oc == 2 || oc == 3) && inner->get_input_size() == 2) { + const size_t keep = (oc == 2) ? 0 : 1; + const size_t fold = 1 - keep; + auto folded = std::make_shared(inner->input_value(fold), input_1); + res = std::make_shared(inner->input_value(keep), folded); + } else { + res = std::make_shared(input_0, input_1); + }""", + ), + ], + "ggml/src/ggml-openvino/openvino/op/rope.cpp": [ + ( + """ if (context.get_input_size() == 3) { + cache_key += "_ff_" + context.get_input_names()[2]; } - } - - auto res = std::make_shared(input_0, input_1);""", +""", + """ if (context.get_input_size() == 3) { + cache_key += "_ff_" + context.get_input_names()[2]; + } + // vla.cpp: two ROPEs with the same op_params but different position + // inputs (a prefix and an action suffix) must not share one table. + cache_key += "_pos_" + context.get_input_names()[1]; +""", ), ], "ggml/src/ggml-openvino/openvino/op/concat.cpp": [ @@ -642,7 +687,7 @@ }""", """ if (GgmlOvDecoder::is_inp_pos(tensor, op)) { // vla.cpp: this free function is the live naming path -- the - // GgmlOvDecoder member of the same intent is unreferenced at b10729. + // GgmlOvDecoder member of the same intent is unreferenced at b11223. // Only collapse ROPE position inputs onto one "inp_pos" parameter when // the graph really has one. See scripts/patch_ggml_openvino.py. return decoder->has_multiple_inp_pos() ? get_tensor_ov_name(cgraph, tensor) : std::string("inp_pos"); @@ -707,16 +752,16 @@ // it that path rebuilt the decoder, re-converted the model and called // compile_model() on every ggml_backend_graph_compute, which dominated runtime. struct naive_runtime_ctx { - std::mutex mutex; std::shared_ptr decoder; - std::shared_ptr model; std::shared_ptr infer_request; + std::vector input_names; + std::vector output_names; }; -// vla.cpp: graph_key is {n_nodes, first name, last name}, which two graphs of the -// same size can share. A compiled model is bound to the shapes it was built for, -// so reusing one across a shape change returns another graph's answer with no -// error. Mix the ops and shapes in as well. +// vla.cpp: graph_key is {n_nodes, first name, last name, input names}, which two +// graphs of the same size can share. A compiled model is bound to the shapes it +// was built for, so reusing one across a shape change returns another graph's +// answer with no error. Mix the ops and shapes in as well. inline uint64_t naive_graph_sig(const ggml_cgraph * cgraph) { uint64_t h = 1469598103934665603ull; auto mix = [&h](uint64_t v) { h = (h ^ v) * 1099511628211ull; }; @@ -768,22 +813,11 @@ naive_cache.clear(); infer_request_cache.clear();""", ), - ( - """enum ggml_status naive_compute(struct ggml_cgraph * cgraph, - ov::Core & core, - const std::string & device, - const ov::AnyMap & config);""", - """enum ggml_status naive_compute(struct ggml_cgraph * cgraph, - ov::Core & core, - const std::string & device, - const ov::AnyMap & config, - std::shared_ptr r_ctx);""", - ), ], "ggml/src/ggml-openvino/utils.cpp": [ ( """ if (!model_is_splitted) { - return naive_compute(cgraph, core, device, config); + return naive_compute(cgraph, core, device, config, *r_ctx->compiled_cache); }""", """ if (!model_is_splitted) { return naive_compute(cgraph, core, device, config, r_ctx); @@ -791,7 +825,7 @@ ), ( """ if (is_naive(cgraph)) { - return naive_compute(cgraph, core, device, config); + return naive_compute(cgraph, core, device, config, *r_ctx->compiled_cache); }""", """ if (is_naive(cgraph)) { return naive_compute(cgraph, core, device, config, r_ctx); diff --git a/scripts/quantize_gguf.py b/scripts/quantize_gguf.py index 6e37647..130ea19 100644 --- a/scripts/quantize_gguf.py +++ b/scripts/quantize_gguf.py @@ -14,8 +14,8 @@ # limitations under the License. """Requantize a vla.cpp GGUF: pack large weight matrices to Q8_0/Q4_0 and copy -everything else unchanged. The loader keeps quantized weights packed and lets -ggml_mul_mat dequantize at compute, so a Q8_0 file is about half the size of the +everything else unchanged. The loader keeps quantized weights packed and ggml +runs them as int8 dot products, so a Q8_0 file is about half the size of the bf16 one with near-identical actions. Embeddings, the output head, norms, conv patch embeddings and position tables stay float (row-fetch and small tensors do not benefit and can lose accuracy). @@ -27,11 +27,14 @@ import numpy as np import gguf +from gguf_common import copy_kv, copy_tensor + # Substrings that keep a tensor at its source precision. Embeddings, the output # head, norms, conv, position tables and the action expert stay float. The vision # tower stays float too by default; add --vision to pack it as well. SKIP = ( "token_embd", + "tok_embd", "output.weight", "patch_embd", "norm", @@ -40,13 +43,16 @@ "cls", "action", "state", - "expert", + "aex.", + "ah.", + "act.", + "octo.head", "dit", "adaln", "ada_", "time" ) -SKIP_VISION = ("vit", "vision") +SKIP_VISION = ("vit", "vision", "vis.d.", "vis.s.", "octo.obs.") # Block size per row (ne0 must divide this). Only the types the gguf writer can # pack are offered; Q8_0 is near-lossless, Q4_0/Q4_1 are 4-bit. @@ -89,17 +95,9 @@ def main() -> None: arch = r.fields["general.architecture"].contents() w = gguf.GGUFWriter(args.dst, arch) - meta = {"GGUF.version", "GGUF.tensor_count", "GGUF.kv_count", "general.architecture"} - for name, f in r.fields.items(): - if name in meta: - continue - if f.types and f.types[0] == gguf.GGUFValueType.ARRAY: - w.add_array(name, f.contents()) - else: - w.add_key_value(name, f.contents(), f.types[0]) + copy_kv(r, w) qtype = getattr(gguf.GGMLQuantizationType, args.type) - F32, BF16 = gguf.GGMLQuantizationType.F32, gguf.GGMLQuantizationType.BF16 n_q = 0 bytes_in = bytes_out = 0 for t in r.tensors: @@ -112,13 +110,7 @@ def main() -> None: bytes_out += int(packed.nbytes) n_q += 1 else: - # Pass copies in their natural dtype so the writer keeps the size right. - data = np.ascontiguousarray(t.data) - if t.tensor_type == BF16: - data = data.view(np.uint16) - elif t.tensor_type == F32: - data = data.astype(np.float32, copy=False) - w.add_tensor(t.name, data, raw_dtype=t.tensor_type) + copy_tensor(w, t) bytes_out += src_bytes w.write_header_to_file() diff --git a/scripts/tokenize_prompt.py b/scripts/tokenize_prompt.py index aa604d6..7bd7d9c 100644 --- a/scripts/tokenize_prompt.py +++ b/scripts/tokenize_prompt.py @@ -13,19 +13,25 @@ # See the License for the specific language governing permissions and # limitations under the License. -"""Print the token ids for an instruction, using the tokenizer an arch was trained with. +"""Print the token ids for an instruction, using the tokenizer and prompt an arch was trained with. tokenize_prompt.py --arch smolvla --text "pick up the black bowl" - -> 1,4842,731,254,2482,7681,2 + -> 18188,614,260,2632,12505,198 -vla-cli --text calls this so the quickstart does not need raw ids. The eval -client keeps its own richer prompt handling; this only covers the plain case. +vla-cli --text calls this so the quickstart does not need raw ids. The prompt +templates are the ones eval/client/vla_cpp_client.py sends, so the ids match +the eval client token for token. pi05 puts the robot state in the prompt and +needs --state; --stats defaults to the LIBERO stats the eval client uses. +--views is the number of camera images for the archs whose prompt has image +slots. """ import argparse +import json +import re import sys +from pathlib import Path -# Same tokenizers the eval client uses (eval/client/vla_cpp_client.py). TOKENIZERS = { "smolvla": "HuggingFaceTB/SmolVLM2-500M-Instruct", "pi0": "google/paligemma-3b-pt-224", @@ -36,29 +42,150 @@ "openvla_oft": "moojink/openvla-7b-oft-finetuned-libero-spatial-object-goal-10", "vla_jepa": "Qwen/Qwen3-VL-2B-Instruct", "gr00t_n1_5": "lerobot/eagle2hg-processor-groot-n1p5", + "gr00t_n1_6": "vrfai/gr00tn1d6-libero-gguf", "gr00t_n1_7": "nvidia/Cosmos-Reason2-2B", "turbovla": "bert-base-uncased", } TRUST_REMOTE_CODE = {"evo1", "gr00t_n1_5"} +MAX_LENGTH = {"smolvla": 48, "pi0": 48, "pi05": 200, "turbovla": 21} +VIEWS = {"evo1": 3, "bitvla": 2, "vla_jepa": 2, "gr00t_n1_5": 2, "gr00t_n1_6": 2, "gr00t_n1_7": 2} + +BITVLA_PROMPT = "What action should the robot take to {}?" +VLA_ADAPTER_PROMPT = ( + "<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful " + "assistant.<|im_end|>\n<|im_start|>user\nWhat action should the robot take to " + "{}?<|im_end|>\n<|im_start|>assistant\n" +) +OPENVLA_OFT_PROMPT = "In: What action should the robot take to {}?\nOut:" +OPENVLA_OFT_EMPTY_TOKEN = 29871 +VLA_JEPA_PROMPT = ( + "Your task is {}. Infer the temporal dynamics from frames " + + "".join(f"<|action_{i}|>" * 8 for i in range(3)) + + " and produce the corresponding policy actions " + + "<|embodied_action|>" * 32 + + "." +) +GR00T_N1_7_IMAGE_PAD = 151655 + + +def fail(msg: str) -> int: + print(f"tokenize_prompt: {msg}", file=sys.stderr) + return 1 + + +def pi05_prompt(text: str, state: str, stats: str) -> str: + import numpy as np + if not stats: + from huggingface_hub import hf_hub_download + stats = hf_hub_download("lerobot/libero", "meta/stats.json", repo_type="dataset") + st = json.loads(Path(stats).read_text())["observation.state"] + q01 = np.asarray(st["q01"], dtype=np.float32).reshape(-1) + q99 = np.asarray(st["q99"], dtype=np.float32).reshape(-1) + s = np.asarray([float(x) for x in state.replace(",", " ").split()], dtype=np.float32) + if s.size < q01.size: + raise ValueError(f"--state has {s.size} values, the stats describe {q01.size}") + normed = 2.0 * (s[:q01.size] - q01) / (q99 - q01) - 1.0 + disc = np.digitize(normed, bins=np.linspace(-1.0, 1.0, 256 + 1)[:-1]) - 1 + cleaned = text.strip().replace("_", " ").replace("\n", " ") + return f"Task: {cleaned}, State: {' '.join(map(str, disc.tolist()))};\nAction: " + + +def token_ids(arch: str, text: str, tok, args) -> list: + views = args.views + if arch in ("smolvla", "pi0"): + text = text if text.endswith("\n") else text + "\n" + if arch == "pi05": + text = pi05_prompt(text, args.state, args.stats) + if arch in MAX_LENGTH: + return tok(text, truncation=True, max_length=MAX_LENGTH[arch])["input_ids"] + + if arch == "evo1": + block = "" + "" * 256 + "" + prompt = "".join(f"Image-{i+1}: {block}\n" for i in range(views)) + text.strip() + enc = tok(prompt, padding="max_length", truncation=True, max_length=1024) + return enc["input_ids"][:sum(enc["attention_mask"])] + if arch == "bitvla": + content = "<|image_pad|>" * (views * 256) + "" + BITVLA_PROMPT.format(text.lower()) + prompt = tok.apply_chat_template([{"role": "user", "content": content}], + tokenize=False, add_generation_prompt=True) + return tok(prompt, add_special_tokens=True)["input_ids"] + if arch == "vla_adapter": + return tok(VLA_ADAPTER_PROMPT.format(text.lower()), add_special_tokens=False)["input_ids"] + if arch == "openvla_oft": + ids = tok(OPENVLA_OFT_PROMPT.format(text.lower()), add_special_tokens=True)["input_ids"] + return ids if ids and ids[-1] == OPENVLA_OFT_EMPTY_TOKEN else ids + [OPENVLA_OFT_EMPTY_TOKEN] + if arch == "vla_jepa": + from PIL import Image + content = [{"type": "image", "image": Image.new("RGB", (224, 224))} for _ in range(views)] + content.append({"type": "text", "text": VLA_JEPA_PROMPT.format(text)}) + enc = tok.apply_chat_template([[{"role": "user", "content": content}]], tokenize=True, + add_generation_prompt=True, return_dict=True, + processor_kwargs={"padding": True, "return_tensors": "pt"}) + return enc["input_ids"][0].tolist() + if arch == "gr00t_n1_5": + images = "".join(f"" + "" * 256 + "" for i in range(views)) + prompt = ("<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n<|im_start|>user\n" + + images + str([text]) + "<|im_end|>\n<|im_start|>assistant\n") + return tok(prompt, add_special_tokens=False)["input_ids"] + + language = re.sub(r"[^\w\s]", "", text.lower()) + if arch == "gr00t_n1_6": + content = language + "".join(f"" + "" * 64 + "" + for i in range(views)) + else: + content = [{"type": "image"} for _ in range(views)] + [{"type": "text", "text": language}] + prompt = tok.apply_chat_template([{"role": "user", "content": content}], + tokenize=False, add_generation_prompt=False) + ids = tok(prompt, add_special_tokens=False)["input_ids"] + if arch == "gr00t_n1_6": + return ids + return [t for i in ids for t in ([i] * 64 if i == GR00T_N1_7_IMAGE_PAD else [i])] + + +def load_tokenizer(arch: str, name: str): + if arch == "vla_jepa": + from transformers import AutoProcessor + proc = AutoProcessor.from_pretrained(name) + proc.tokenizer.add_tokens([f"<|action_{i}|>" for i in range(28)], special_tokens=True) + proc.tokenizer.add_tokens(["<|embodied_action|>"], special_tokens=True) + return proc + from transformers import AutoTokenizer + tok = AutoTokenizer.from_pretrained(name, trust_remote_code=arch in TRUST_REMOTE_CODE, + use_fast=arch != "evo1") + if arch == "gr00t_n1_6": + path = Path(name) / "chat_template.json" + if not path.exists(): + from huggingface_hub import hf_hub_download + path = Path(hf_hub_download(name, "chat_template.json")) + tok.chat_template = json.loads(path.read_text())["chat_template"] + return tok def main() -> int: - ap = argparse.ArgumentParser(description=__doc__) + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) ap.add_argument("--arch", required=True, choices=sorted(TOKENIZERS)) ap.add_argument("--text", required=True) - ap.add_argument("--tokenizer", help="override the HuggingFace tokenizer id") + ap.add_argument("--tokenizer", help="override the HuggingFace tokenizer id or a local dir") + ap.add_argument("--views", type=int, help="camera images the prompt has slots for") + ap.add_argument("--state", help="pi05: robot state, comma-separated floats (--state=-0.1,... if it starts with -)") + ap.add_argument("--stats", help="pi05: stats.json with observation.state q01/q99 " + "(default: lerobot/libero meta/stats.json)") args = ap.parse_args() + if args.arch == "pi05" and not args.state: + return fail("pi05 puts the robot state in the prompt, so it needs --state f,f,...") + if args.views is None: + args.views = VIEWS.get(args.arch, 1) + if args.views < 1: + return fail("--views must be at least 1") + try: - from transformers import AutoTokenizer - except ImportError: - print("transformers is not installed: pip install -e \".[client]\"", file=sys.stderr) - return 1 - - name = args.tokenizer or TOKENIZERS[args.arch] - tok = AutoTokenizer.from_pretrained( - name, trust_remote_code=args.arch in TRUST_REMOTE_CODE) - ids = tok(args.text)["input_ids"] + tok = load_tokenizer(args.arch, args.tokenizer or TOKENIZERS[args.arch]) + ids = token_ids(args.arch, args.text, tok, args) + except ImportError as e: + return fail(f"{e}. Install the client extras: pip install -e \".[client]\"") + except (ValueError, KeyError, OSError) as e: + return fail(str(e)) print(",".join(str(int(i)) for i in ids)) return 0 diff --git a/scripts/upstream_split.py b/scripts/upstream_split.py index 2b328f7..aa89146 100755 --- a/scripts/upstream_split.py +++ b/scripts/upstream_split.py @@ -62,12 +62,12 @@ def H(rel, key): "A hit rebinds the cached decoder through the existing update_io(), the same way\n" "the dynamic path handles freshly built tensors.\n\n" "The cache is keyed on naive_key rather than graph_key. graph_key is n_nodes plus\n" - "the first and last node name, which two graphs of the same size can share, and a\n" + "tensor names, which two graphs of the same size can share, and a\n" "compiled model is bound to the shapes it was built for, so a collision returns\n" "another graph's answer with no error. naive_key mixes in every node's op, type\n" "and shape. The map is bounded and flushed when full.", [H(D+"utils.h","struct decoder_runtime_ctx"),H(D+"utils.h","graph_key_hash> decoder_cache"), - H(D+"utils.h","decoder_cache.clear()"),H(D+"utils.h","enum ggml_status naive_compute"), + H(D+"utils.h","decoder_cache.clear()"), H(D+"utils.cpp","if (!model_is_splitted)"),H(D+"utils.cpp","if (is_naive(cgraph))"), H(D+"utils.cpp","enum ggml_status naive_compute")]), @@ -108,9 +108,13 @@ def H(rel, key): "input. Single-position graphs are untouched.\n\n" "Guard the free get_tensor_graph_input_ov_name() as well as the GgmlOvDecoder\n" "member: the free function is the one compute_model_inputs() and\n" - "set_input_output() actually call, and the member currently has no callers.", + "set_input_output() actually call, and the member currently has no callers.\n\n" + "translate_rope() also caches each sin/cos table under the ROPE's op_params\n" + "alone, so two ROPEs with the same parameters and different position inputs\n" + "share one table and fail the same way. Key the cache on the position input too.", [H(D+"ggml-decoder.h","get_graph_input_ov_name"),H(D+"ggml-decoder.h","m_cgraph = nullptr"), - H(D+"ggml-decoder.cpp","is_inp_pos(tensor, op)"),H(D+"ggml-decoder.cpp","compute_op_case(const ggml_tensor")]), + H(D+"ggml-decoder.cpp","is_inp_pos(tensor, op)"),H(D+"ggml-decoder.cpp","compute_op_case(const ggml_tensor"), + H(D+"openvino/op/rope.cpp","_ff_")]), ("openvino-reshape-op-case", "openvino: narrow the RESHAPE op_case 3 guard", @@ -137,7 +141,7 @@ def H(rel, key): "of the inner add is the GEMM, since the order is not fixed.\n\n" "Same fusion path as the broadcast-DIV defect already handled in supports_op.", [H(D+"ggml-decoder.cpp","case GGML_OP_ADD: {"), - H(D+"openvino/op/add.cpp","auto input_0 = process_view_input_new(context, 0);")]), + H(D+"openvino/op/add.cpp","ov::Output res = std::make_shared")]), ("openvino-permute-op-case", "openvino: require a ROPE before taking PERMUTE op_case 2", diff --git a/src/arch.h b/src/arch.h index 47be355..146b6e4 100644 --- a/src/arch.h +++ b/src/arch.h @@ -59,7 +59,7 @@ inline int default_cpu_threads() { * (@ref detect_arch_from_ckpt) and routed to the corresponding factory. */ enum class Arch { - SMOLVLA, // Hugging Face SmolVLA (mmproj + LM + flow-matching head). + SMOLVLA, // Hugging Face SmolVLA (SigLIP + LM + flow-matching head). PI0, // Physical Intelligence pi0 (PaliGemma + flow-matching). PI05, // PaliGemma-3B + SigLIP-So400m + Gemma-300m + FM. EVO1, // MINT-SJTU Evo-1 (InternVL3 + cross-attention head). @@ -105,9 +105,9 @@ class ModelArchBase { }; /** - * @brief Build a SmolVLA model from its mmproj and checkpoint GGUFs. - * @param mmproj_path Path to the vision-tower GGUF. - * @param ckpt_path Path to the LM+action-expert GGUF. + * @brief Build a SmolVLA model from its checkpoint GGUF. + * @param mmproj_path Ignored; the vision tower is bundled in @p ckpt_path. + * @param ckpt_path Path to the vision+LM+action-expert GGUF. * @param config_path Optional JSON override; pass empty to use bundled config. * @return Owning pointer to the constructed model. */ @@ -117,7 +117,7 @@ std::unique_ptr smolvla_create(const std::string& mmproj_path, const Options& opts); /** - * @brief Build a pi0 model from its mmproj and checkpoint GGUFs. + * @brief Build a pi0 model from its checkpoint GGUF. * @copydetails smolvla_create */ std::unique_ptr pi0_create(const std::string& mmproj_path, @@ -126,7 +126,7 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, const Options& opts); /** - * @brief Build a pi0.5 model from its mmproj and checkpoint GGUFs. + * @brief Build a pi0.5 model from its checkpoint GGUF. * @copydetails smolvla_create */ std::unique_ptr pi05_create(const std::string& mmproj_path, @@ -173,14 +173,12 @@ std::unique_ptr gr00t_n1_7_create(const std::string& mmproj_path, /** * @brief Build an Octo model. Vision, T5 text encoder and tokenizer vocab are - * all baked into @p ckpt_path. Only compiled when VLA_OCTO is on. + * all baked into @p ckpt_path. * @copydetails smolvla_create */ -#ifdef VLA_USE_OCTO std::unique_ptr octo_create(const std::string& mmproj_path, const std::string& ckpt_path, const std::string& config_path); -#endif /** * @brief Build a BitVLA model. Vision is baked into @p ckpt_path. diff --git a/src/backend.h b/src/backend.h index 087638e..7f61a95 100644 --- a/src/backend.h +++ b/src/backend.h @@ -202,6 +202,15 @@ inline ggml_tensor * gelu(ggml_context * ctx, ggml_tensor * x) { #endif } +inline ggml_tensor * geglu(ggml_context * ctx, ggml_tensor * gate, ggml_tensor * up) { +#if !defined(GGML_USE_SYCL) && !defined(GGML_USE_METAL) && !defined(GGML_USE_OPENVINO) && \ + !defined(GGML_USE_HEXAGON) && !defined(GGML_USE_OPENCL) + if (gate->type == GGML_TYPE_F32) + return ggml_geglu_split(ctx, gate, up); +#endif + return ggml_mul(ctx, gelu(ctx, gate), up); +} + /** * @brief An arch's default resident type for GEMM weights, on this backend. * diff --git a/src/backend_fallback.cpp b/src/backend_fallback.cpp index 21d7b9a..7136bc7 100644 --- a/src/backend_fallback.cpp +++ b/src/backend_fallback.cpp @@ -45,9 +45,10 @@ namespace { enum class Where : uint8_t { Accel, Cpu, Either }; struct FallbackCtx { - ggml_backend_t accel = nullptr; - ggml_backend_t cpu = nullptr; - std::string name; + ggml_backend_t accel = nullptr; + ggml_backend_t cpu = nullptr; + ggml_threadpool_t tp = nullptr; + std::string name; // Host copies of weights some CPU-side op reads, made once. Keyed by tensor: // the wrapper lives exactly as long as the model that owns the weights. @@ -137,14 +138,15 @@ ggml_tensor * host_input(FallbackCtx * fc, ggml_context * meta, if (is_host(t)) { s->data = t->data; } else if (is_weight(t)) { - std::vector & w = fc->weights[t]; + const ggml_tensor * base = t->view_src ? t->view_src : t; + std::vector & w = fc->weights[base]; if (w.empty()) { // get_tensor undoes the accelerator's repacking, so this is plain ggml layout. - w.resize(n); - ggml_backend_tensor_get(t, w.data(), 0, n); - fc->bytes_in += n; + w.resize(ggml_nbytes(base)); + ggml_backend_tensor_get(base, w.data(), 0, w.size()); + fc->bytes_in += w.size(); } - s->data = w.data(); + s->data = w.data() + (t->view_src ? t->view_offs : 0); } else { s->data = take(fc, n); ggml_backend_tensor_get(t, s->data, 0, n); @@ -258,6 +260,7 @@ void fb_free(ggml_backend_t be) { } ggml_backend_free(fc->accel); ggml_backend_free(fc->cpu); + ggml_threadpool_free(fc->tp); delete fc; delete be; } @@ -418,6 +421,11 @@ ggml_backend_t fallback_backend_new(ggml_backend_t accel, int n_threads) { auto * fc = new FallbackCtx; fc->accel = accel; fc->cpu = cpu; + ggml_threadpool_params tpp = ggml_threadpool_params_default(n_threads); + tpp.poll = 0; + fc->tp = ggml_threadpool_new(&tpp); + if (fc->tp) + ggml_backend_cpu_set_threadpool(cpu, fc->tp); fc->name = std::string(ggml_backend_name(accel)) + "+CPU"; const char * st = std::getenv("VLA_FALLBACK_STATS"); fc->stats = st && st[0] == '1'; diff --git a/src/cuda/vla_cuda_bf16.cu b/src/cuda/vla_cuda_bf16.cu index 70a3d63..70b3766 100644 --- a/src/cuda/vla_cuda_bf16.cu +++ b/src/cuda/vla_cuda_bf16.cu @@ -30,7 +30,9 @@ // cannot break it the way an anchored source patch would. // // Every entry point returns false for anything it does not handle, and ggml -// then runs the op exactly as it would have. Nothing here changes the F32 path. +// then runs the op exactly as it would have. The exception is mul_mat with a +// BF16 result: ggml has no fallback for it, so an unsupported one aborts. +// Nothing here changes the F32 path. // // Accumulation is float throughout: only operand and result *storage* is BF16, // never a reduction. @@ -44,6 +46,8 @@ #include #include +#include +#include // Must match the typedef the hook patch inserts into ggml-cuda.cu. extern "C" { @@ -266,7 +270,8 @@ bool bin_bcast(ggml_tensor * dst, cudaStream_t stream) { g.ok && dst->ne[0]%8 == 0 && es(src0, 0) == 1 && es(dst, 0) == 1 && es(src1, 0) == 1 && src1->ne[0] == dst->ne[0] && - es(src0, 1)%8 == 0 && es(dst, 1)%8 == 0 && es(src1, 1)%8 == 0 && + es(src0, 1)%8 == 0 && es(src0, 2)%8 == 0 && es(src0, 3)%8 == 0 && + es(dst, 1)%8 == 0 && es(dst, 2)%8 == 0 && es(dst, 3)%8 == 0 && es(src1, 1)%8 == 0 && ((uintptr_t) src0->data%16) == 0 && ((uintptr_t) dst->data%16) == 0; if (vec8_shape) { const int64_t nvec = dst->ne[0]/8; @@ -619,29 +624,32 @@ bool norm(ggml_tensor * dst, cudaStream_t stream) { // dst->data unconditionally, which for a BF16 dst is both wrong and twice the // bytes the allocator reserved. -cublasHandle_t g_handle = nullptr; - bool mul_mat(ggml_tensor * dst, cudaStream_t stream) { + if (dst->type != GGML_TYPE_BF16) + return false; const ggml_tensor * src0 = dst->src[0]; const ggml_tensor * src1 = dst->src[1]; - if (!src0 || !src1) - return false; - if (dst->type != GGML_TYPE_BF16 || src0->type != GGML_TYPE_BF16 || src1->type != GGML_TYPE_BF16) { - return false; - } - if (!ggml_is_contiguous(src0) || !ggml_is_contiguous(src1) || !ggml_is_contiguous(dst)) { - return false; + if (src0->type != GGML_TYPE_BF16 || src1->type != GGML_TYPE_BF16 || + !ggml_is_contiguous(src0) || !ggml_is_contiguous(src1) || !ggml_is_contiguous(dst)) { + GGML_ABORT("vla: unsupported BF16 mul_mat %s", dst->name); } // src0 is either shared across the whole batch or batched 1:1 with src1 const bool batch_ok = (src0->ne[2] == 1 && src0->ne[3] == 1) || (src0->ne[2] == src1->ne[2] && src0->ne[3] == src1->ne[3]); if (!batch_ok) - return false; - - if (!g_handle && cublasCreate(&g_handle) != CUBLAS_STATUS_SUCCESS) - return false; - if (cublasSetStream(g_handle, stream) != CUBLAS_STATUS_SUCCESS) - return false; + GGML_ABORT("vla: unsupported BF16 mul_mat %s", dst->name); + + int dev = 0; + if (cudaGetDevice(&dev) != cudaSuccess) + GGML_ABORT("vla: cudaGetDevice failed for %s", dst->name); + static std::mutex mu; + static std::map handles; + std::lock_guard lock(mu); + cublasHandle_t & handle = handles[dev]; + if ((!handle && cublasCreate(&handle) != CUBLAS_STATUS_SUCCESS) || + cublasSetStream(handle, stream) != CUBLAS_STATUS_SUCCESS) { + GGML_ABORT("vla: cuBLAS setup failed for %s", dst->name); + } const int64_t ne00 = src0->ne[0], ne01 = src0->ne[1]; const int64_t ne10 = src1->ne[0], ne11 = src1->ne[1]; @@ -657,7 +665,7 @@ bool mul_mat(ggml_tensor * dst, cudaStream_t stream) { cublasStatus_t st; if (n_batch == 1) { - st = cublasGemmEx(g_handle, CUBLAS_OP_T, CUBLAS_OP_N, + st = cublasGemmEx(handle, CUBLAS_OP_T, CUBLAS_OP_N, (int) ne01, (int) ne11, (int) ne10, &alpha, a, CUDA_R_16BF, (int) ne00, b, CUDA_R_16BF, (int) ne10, @@ -666,7 +674,7 @@ bool mul_mat(ggml_tensor * dst, cudaStream_t stream) { } else { // stride_a == 0 broadcasts one weight matrix across the batch const long long stride_a = (src0->ne[2] == 1 && src0->ne[3] == 1) ? 0 : (long long) ne00*ne01; - st = cublasGemmStridedBatchedEx(g_handle, CUBLAS_OP_T, CUBLAS_OP_N, + st = cublasGemmStridedBatchedEx(handle, CUBLAS_OP_T, CUBLAS_OP_N, (int) ne01, (int) ne11, (int) ne10, &alpha, a, CUDA_R_16BF, (int) ne00, stride_a, b, CUDA_R_16BF, (int) ne10, (long long) ne10*ne11, @@ -674,32 +682,19 @@ bool mul_mat(ggml_tensor * dst, cudaStream_t stream) { (int) n_batch, CUBLAS_COMPUTE_32F, CUBLAS_GEMM_DEFAULT_TENSOR_OP); } - return st == CUBLAS_STATUS_SUCCESS; + if (st != CUBLAS_STATUS_SUCCESS) + GGML_ABORT("vla: BF16 mul_mat %s failed (%d)", dst->name, (int) st); + return true; } -} // namespace - -// --------------------------------------------------------------------------- -// hook entry point -// --------------------------------------------------------------------------- - -extern "C" bool vla_cuda_bf16_fused_binbcast(ggml_tensor * dst, int n_fuse, void * stream_v) { - if (!dst) - return false; - cudaStream_t stream = (cudaStream_t) stream_v; - - switch (dst->op) { - case GGML_OP_ADD: return fused_bin_bcast(dst, n_fuse, stream); - case GGML_OP_MUL: return fused_bin_bcast(dst, n_fuse, stream); - default: return false; - } +bool launched(const ggml_tensor * dst) { + const cudaError_t err = cudaGetLastError(); + if (err != cudaSuccess) + GGML_ABORT("vla: %s %s failed: %s", ggml_op_desc(dst), dst->name, cudaGetErrorString(err)); + return true; } -extern "C" bool vla_cuda_bf16_forward(ggml_tensor * dst, void * stream_v) { - if (!dst) - return false; - cudaStream_t stream = (cudaStream_t) stream_v; - +bool forward(ggml_tensor * dst, cudaStream_t stream) { switch (dst->op) { case GGML_OP_MUL_MAT: return mul_mat(dst, stream); case GGML_OP_ADD: return bin_bcast(dst, stream); @@ -719,6 +714,28 @@ extern "C" bool vla_cuda_bf16_forward(ggml_tensor * dst, void * stream_v) { } } +} // namespace + +// --------------------------------------------------------------------------- +// hook entry point +// --------------------------------------------------------------------------- + +extern "C" bool vla_cuda_bf16_fused_binbcast(ggml_tensor * dst, int n_fuse, void * stream_v) { + if (!dst) + return false; + cudaStream_t stream = (cudaStream_t) stream_v; + + switch (dst->op) { + case GGML_OP_ADD: return fused_bin_bcast(dst, n_fuse, stream) && launched(dst); + case GGML_OP_MUL: return fused_bin_bcast(dst, n_fuse, stream) && launched(dst); + default: return false; + } +} + +extern "C" bool vla_cuda_bf16_forward(ggml_tensor * dst, void * stream_v) { + return dst && forward(dst, (cudaStream_t) stream_v) && launched(dst); +} + namespace vla { // Called once, after the CUDA backend is up. Idempotent. diff --git a/src/kernels/bitvla/bitnet_kernels.cu b/src/kernels/bitvla/bitnet_kernels.cu index 2204474..91af3f0 100644 --- a/src/kernels/bitvla/bitnet_kernels.cu +++ b/src/kernels/bitvla/bitnet_kernels.cu @@ -28,42 +28,6 @@ std::abort(); } -extern "C" void bitlinear_int8xint2(int8_t* input0, int8_t* input1, __nv_bfloat16* output0, float* s, float* ws, int M, int N, int K, cudaStream_t stream){ - if (M == 1 && N == 3840 && K == 2560){ - ladder_int8xint2_kernel<1, 3840, 2560, 3, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if (M == 1 && N == 2560 && K == 2560){ - ladder_int8xint2_kernel<1, 2560, 2560, 1, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if (M == 1 && N == 13824 && K == 2560){ - ladder_int8xint2_kernel<1, 13824, 2560, 2, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if (M == 1 && N == 2560 && K == 6912){ - ladder_int8xint2_kernel<1, 2560, 6912, 1, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if(M == 1 && N == 4800 && K == 3200){ - ladder_int8xint2_kernel<1, 4800, 3200, 6, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if(M == 1 && N == 3200 && K == 3200){ - ladder_int8xint2_kernel<1, 3200, 3200, 1, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if(M == 1 && N == 20480 && K == 3200){ - ladder_int8xint2_kernel<1, 20480, 3200, 2, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if(M == 1 && N == 3200 && K == 10240){ - ladder_int8xint2_kernel<1, 3200, 10240, 1, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if(M == 1 && N == 5120 && K == 27648){ - ladder_int8xint2_kernel<1, 5120, 27648, 1, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if(M == 1 && N == 55296 && K == 5120){ - ladder_int8xint2_kernel<1, 55296, 5120, 1, 8, 16><<>>(input0, input1, output0, s, ws); - } - else{ - bitlinear_unsupported_shape("bitlinear_int8xint2", M, N, K); - } -} - // The wide kernel amortises the shared A block over 4 column tiles instead of // 1 (see ladder_int8xint2_kernel_m_wide). It is the default; set // VLA_BITVLA_NARROW_GEMM=1 to fall back to the one-tile-per-CTA kernel, which @@ -92,8 +56,6 @@ extern "C" void bitlinear_int8xint2_m( else if (N == 1152 && K == 1152) WIDE(1152, 1152, 1); else if (N == 4304 && K == 1152) WIDE(4304, 1152, 1); else if (N == 1152 && K == 4352) WIDE(1152, 4352, 1); - - else if (N == 3840 && K == 2560) WIDE(3840, 2560, 3); else bitlinear_unsupported_shape("bitlinear_int8xint2_m", M, N, K); #undef WIDE @@ -109,16 +71,14 @@ extern "C" void bitlinear_int8xint2_m( else if (N == 1152 && K == 1152) launch_ladder_int8xint2_m<1152, 1152, 1, 128>(input0, input1, output0, s, ws, M, stream); else if (N == 4304 && K == 1152) launch_ladder_int8xint2_m<4304, 1152, 1, 128>(input0, input1, output0, s, ws, M, stream); else if (N == 1152 && K == 4352) launch_ladder_int8xint2_m<1152, 4352, 1, 128>(input0, input1, output0, s, ws, M, stream); - - else if (N == 3840 && K == 2560) launch_ladder_int8xint2_m<3840, 2560, 3, 128>(input0, input1, output0, s, ws, M, stream); else bitlinear_unsupported_shape("bitlinear_int8xint2_m", M, N, K); } extern "C" void bitvla_act_quant_cuda( const __nv_bfloat16* in, int8_t* out, float* scales, - int M, int K, cudaStream_t stream) + int M, int K, int ld_out, cudaStream_t stream) { constexpr int BLOCK_THREADS = 256; - act_quant_kernel<<>>(in, out, scales, K); + act_quant_kernel<<>>(in, out, scales, K, ld_out); } diff --git a/src/kernels/bitvla/bitnet_kernels.h b/src/kernels/bitvla/bitnet_kernels.h index bae049d..833a106 100644 --- a/src/kernels/bitvla/bitnet_kernels.h +++ b/src/kernels/bitvla/bitnet_kernels.h @@ -23,11 +23,8 @@ * intrinsic to expand i2 to i8, then run a single warp-level @c wmma * fragment multiply per tile. * - * Two GEMM entry points are provided: - * * @ref ladder_int8xint2_kernel-single-row (M=1) decode kernel - * used for next-token/single-query inference. - * * @ref ladder_int8xint2_kernel_m + @ref launch_ladder_int8xint2_m - * - multi-row (M>1) variant for prefill and ViT batches. + * GEMM entry point: @ref ladder_int8xint2_kernel_m + @ref launch_ladder_int8xint2_m + * - multi-row (M>1) variant for prefill and ViT batches. * * This header is meant to be included by the per-tier CUDA `.cu` files * (@c bitvla_lm_cuda.cu, @c bitvla_vit_cuda.cu); it is not part of the @@ -38,23 +35,10 @@ #include #include #include -#include #include #include #include -#if (((__CUDACC_VER_MAJOR__ == 11) && (__CUDACC_VER_MINOR__ >= 4)) || (__CUDACC_VER_MAJOR__ > 11)) -#define TVM_ENABLE_L2_PREFETCH 1 -#else -#define TVM_ENABLE_L2_PREFETCH 0 -#endif - -#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ == 800 -#define TVM_ENBALE_EFFICIENT_SMEM_PTR_CAST 1 -#else -#define TVM_ENBALE_EFFICIENT_SMEM_PTR_CAST 0 -#endif - /** * @brief Decode a packed int2 word into N int8 values via @c lop3.b32. * @@ -89,61 +73,6 @@ __device__ void decode_i2s_to_i8s(T1 *_i2s, T2 *_i8s, const int N = 16) } } -/** - * @brief Single-row ternary GEMM kernel (M = 1). - * - * Computes one row of @c dtype_transform[0,:] = (A * B^T)/s[0]*ws, - * with @c A in int8, @c B packed as int2 (decoded on the fly), accumulated - * in int32 via @c __dp4a, then scaled back to bf16. The output bias - * @c ws is applied per @c ws_num column groups. - * - * @tparam M Always 1 in this overload; kept for symmetry with - * the multi-row kernel. - * @tparam N Output column count. - * @tparam K Reduction dimension. - * @tparam ws_num Number of column groups sharing one bias entry. - * @tparam K_block_size Threads collaborating along K (warp width). - * @tparam N_block_size Threads collaborating along N (warp height). - */ -template -__global__ void __launch_bounds__(128) ladder_int8xint2_kernel(int8_t* __restrict__ A, int8_t* __restrict__ B, __nv_bfloat16* __restrict__ dtype_transform, float* __restrict__ s, float* __restrict__ ws) { - constexpr int K_per_loop = 16; - constexpr int wmma_K = 32; - constexpr int wmma_N = 16; - int in_thread_C_local[1]; - signed char A_local[K_per_loop]; - int B_reshape_local[1]; - signed char B_decode_local[K_per_loop]; - int red_buf0[1]; - in_thread_C_local[0] = 0; - #pragma unroll - for (int k_0=0; k_0> 1)*wmma_K * wmma_N/4) + - ((((int)threadIdx.y) >> 3)*(wmma_K * wmma_N/2)/4) + - ((((int)threadIdx.x) & 1)*(wmma_K * wmma_N/4)/4) + - ((((int)threadIdx.y) & 7)*(wmma_K/2)/4) - ); - decode_i2s_to_i8s(B_reshape_local, B_decode_local, 16); - #pragma unroll - for (int k_2_0=0; k_2_0<4; ++k_2_0) { - in_thread_C_local[0] = __dp4a(*(int *)&A_local[((k_2_0*4))],*(int *)&B_decode_local[((k_2_0*4))], in_thread_C_local[0]); - } - } - red_buf0[0] = in_thread_C_local[0]; - #pragma unroll - for (int offset=K_block_size/2; offset>0; offset /= 2) { - red_buf0[0] += __shfl_down_sync(__activemask(), red_buf0[0], offset, K_block_size); - } - int out_idx = ((((int)blockIdx.x)*N_block_size)+((int)threadIdx.y)); - int ws_idx = out_idx/(N/ws_num); - if (threadIdx.x == 0) - dtype_transform[out_idx] = __float2bfloat16(((float)red_buf0[0])/s[0]*ws[ws_idx]); -} - /** * @brief Multi-row ternary GEMM kernel (M > 1) using @c wmma fragments. * @@ -180,9 +109,9 @@ __global__ void __launch_bounds__(128) ladder_int8xint2_kernel_m( const int warp = tid >> 5; const int m_base = (int)blockIdx.y*M_ROWS; - __shared__ signed char A_smem[M_ROWS][K_CHUNK]; - __shared__ signed char W_smem[16][K_CHUNK]; - __shared__ int C_smem[M_ROWS][16]; + __shared__ __align__(32) signed char A_smem[M_ROWS][K_CHUNK]; + __shared__ __align__(32) signed char W_smem[16][K_CHUNK]; + __shared__ __align__(32) int C_smem[M_ROWS][16]; int B_reshape_local[1]; signed char B_decode_local[K_per_loop]; @@ -316,9 +245,9 @@ __global__ void __launch_bounds__(128) ladder_int8xint2_kernel_m_wide( const int lane = tid & 31; const int m_base = (int)blockIdx.y*M_ROWS; - __shared__ signed char A_smem[M_ROWS][SM_STRIDE]; - __shared__ signed char W_smem[N_TILES][16][SM_STRIDE]; - __shared__ int C_smem[WARPS][16][16]; + __shared__ __align__(32) signed char A_smem[M_ROWS][SM_STRIDE]; + __shared__ __align__(32) signed char W_smem[N_TILES][16][SM_STRIDE]; + __shared__ __align__(32) int C_smem[WARPS][16][16]; // Column tile this warp owns. N is not always a multiple of 16*N_TILES // (the ViT's 4304 is 269 tiles), so tiles past the end are skipped rather @@ -430,7 +359,6 @@ static constexpr int bitvla_n_tiles_for(int N, int K) { : (N == 1152 && K == 1152) ? 2 // vit.q/k/v/o : (N == 4304 && K == 1152) ? 4 // vit.fc1 : (N == 1152 && K == 4352) ? 2 // vit.fc2 - : (N == 3840 && K == 2560) ? 4 // action head qkv : 2; } @@ -458,7 +386,7 @@ static inline void launch_ladder_int8xint2_m_wide( * @tparam BLOCK_THREADS Threads per CTA; must be a multiple of 32. * @param in bf16 activation matrix (M x K), device pointer. M is * passed implicitly via @c blockIdx.x. - * @param out int8 quantised matrix (M x K), device pointer. + * @param out int8 quantised matrix (M x ld_out), device pointer. * @param scales Per-row scales (length M), device pointer. * @param K Row length. */ @@ -467,12 +395,12 @@ __global__ void act_quant_kernel( const __nv_bfloat16* __restrict__ in, int8_t* __restrict__ out, float* __restrict__ scales, - int K) + int K, int ld_out) { const int m = (int)blockIdx.x; const int tid = (int)threadIdx.x; const __nv_bfloat16* row_in = in + m * K; - int8_t* row_out = out+m * K; + int8_t* row_out = out+m * ld_out; float local_max = 0.0f; for (int k=tid; kd_pp_out, (size_t)ctx->lm_hidden*sizeof(float), cudaMemcpyDeviceToHost, stream)); + CUDA_OK_RET(cudaGetLastError()); CUDA_OK_RET(cudaStreamSynchronize(stream)); return 0; } @@ -374,6 +376,7 @@ extern "C" int bitvla_fp32head_action_forward( CUDA_OK_RET(cudaMemcpyAsync(host_norm_actions, ctx->d_ah_out, (size_t)M * A * sizeof(float), cudaMemcpyDeviceToHost, stream)); + CUDA_OK_RET(cudaGetLastError()); CUDA_OK_RET(cudaStreamSynchronize(stream)); return 0; } diff --git a/src/kernels/bitvla/bitvla_lm_cuda.cu b/src/kernels/bitvla/bitvla_lm_cuda.cu index 7605a0e..1a50f9f 100644 --- a/src/kernels/bitvla/bitvla_lm_cuda.cu +++ b/src/kernels/bitvla/bitvla_lm_cuda.cu @@ -22,6 +22,7 @@ #include #include #include +#include #include extern "C" void bitlinear_int8xint2_m(int8_t* A, int8_t* B, __nv_bfloat16* out, @@ -29,7 +30,7 @@ extern "C" void bitlinear_int8xint2_m(int8_t* A, int8_t* B, __nv_bfloat16* out, int M, int N, int K, cudaStream_t stream); extern "C" void bitvla_act_quant_cuda(const __nv_bfloat16* in, int8_t* out, float* scales, - int M, int K, cudaStream_t stream); + int M, int K, int ld_out, cudaStream_t stream); extern "C" void gate_up_fused_sqrelu_mul_bf16(const __nv_bfloat16* gu, __nv_bfloat16* out, int seq, int ffn, cudaStream_t stream); @@ -132,6 +133,7 @@ __global__ void softmax_scaled_bf16_kernel(__nv_bfloat16* __restrict__ inout, } __syncthreads(); const float max_v = smem[0]; + __syncthreads(); float s_sum = 0.0f; for (int i=tid; i= N) - return; - float gv = __bfloat162float(g[i]); - if (gv < 0.0f) - gv = 0.0f; - out[i] = __float2bfloat16(gv * gv * __bfloat162float(u[i])); -} - __global__ void add_bf16_kernel(const __nv_bfloat16* a, const __nv_bfloat16* b, __nv_bfloat16* out, int N) { const int i = (int)(blockIdx.x*blockDim.x+threadIdx.x); @@ -240,11 +230,6 @@ extern "C" void bitvla_softmax_scaled_bf16(__nv_bfloat16* inout, float scale, constexpr int B = 256; softmax_scaled_bf16_kernel<<>>(inout, scale, S); } -extern "C" void bitvla_squared_relu_mul_bf16(const __nv_bfloat16* g, const __nv_bfloat16* u, - __nv_bfloat16* out, int N, cudaStream_t stream) { - constexpr int B = 256; - squared_relu_mul_bf16_kernel<<>>(g, u, out, N); -} extern "C" void bitvla_add_bf16(const __nv_bfloat16* a, const __nv_bfloat16* b, __nv_bfloat16* out, int N, cudaStream_t stream) { constexpr int B = 256; @@ -277,7 +262,7 @@ extern "C" void bitvla_gather_rows_bf16(const __nv_bfloat16* in, __nv_bfloat16* return -1; } } while (0) #define CUDA_OKV(call) do { cudaError_t e = (call); if (e != cudaSuccess) { \ std::fprintf(stderr, "vla(bitvla_lm_cuda): %s @ %s:%d\n", cudaGetErrorString(e), __FILE__, __LINE__); \ - return nullptr; } } while (0) + bitvla_lm_cuda_free(ctx); return nullptr; } } while (0) struct bitvla_lm_cuda_ctx { int hidden, n_q, n_kv, head_dim, ffn, n_layers, max_seq; @@ -422,7 +407,7 @@ static int run_layer(bitvla_lm_cuda_ctx* ctx, int L, int seq, cudaStream_t strea bitvla_rmsnorm_bf16(ctx->d_h, lr.attn_norm_w, ctx->d_h_norm, ctx->rms_eps, seq, hidden, stream); l0_dump("L0_01_attn_norm", ctx->d_h_norm, (size_t)seq * hidden); - bitvla_act_quant_cuda(ctx->d_h_norm, ctx->d_act_int8_h, ctx->d_act_s, seq, hidden, stream); + bitvla_act_quant_cuda(ctx->d_h_norm, ctx->d_act_int8_h, ctx->d_act_s, seq, hidden, hidden, stream); __nv_bfloat16* q_dense = ctx->d_qkv; __nv_bfloat16* k_dense = ctx->d_qkv+(size_t)seq * hq; @@ -490,7 +475,7 @@ static int run_layer(bitvla_lm_cuda_ctx* ctx, int L, int seq, cudaStream_t strea bitvla_rmsnorm_bf16(ctx->d_attn_merged, lr.attn_sub_norm_w, ctx->d_h_norm, ctx->rms_eps, seq, hq, stream); l0_dump("L0_04_attn_sub_norm", ctx->d_h_norm, (size_t)seq * hq); - bitvla_act_quant_cuda(ctx->d_h_norm, ctx->d_act_int8_h, ctx->d_act_s, seq, hq, stream); + bitvla_act_quant_cuda(ctx->d_h_norm, ctx->d_act_int8_h, ctx->d_act_s, seq, hq, hq, stream); bitlinear_int8xint2_m(ctx->d_act_int8_h, lr.o_packed, ctx->d_o_out, ctx->d_act_s, lr.o_ws, seq, hidden, hq, stream); @@ -501,7 +486,7 @@ static int run_layer(bitvla_lm_cuda_ctx* ctx, int L, int seq, cudaStream_t strea bitvla_rmsnorm_bf16(ctx->d_h, lr.ffn_norm_w, ctx->d_h_norm, ctx->rms_eps, seq, hidden, stream); l0_dump("L0_07_ffn_norm", ctx->d_h_norm, (size_t)seq * hidden); - bitvla_act_quant_cuda(ctx->d_h_norm, ctx->d_act_int8_h, ctx->d_act_s, seq, hidden, stream); + bitvla_act_quant_cuda(ctx->d_h_norm, ctx->d_act_int8_h, ctx->d_act_s, seq, hidden, hidden, stream); bitlinear_int8xint2_m(ctx->d_act_int8_h, lr.gate_up_packed, ctx->d_gate_up, ctx->d_act_s, lr.gate_up_ws, seq, 2*ffn, hidden, stream); @@ -513,7 +498,7 @@ static int run_layer(bitvla_lm_cuda_ctx* ctx, int L, int seq, cudaStream_t strea l0_dump("L0_08_ffn_sub_norm", ctx->d_gate_sq_up, (size_t)seq * ffn); - bitvla_act_quant_cuda(ctx->d_gate_sq_up, ctx->d_act_int8_ffn, ctx->d_act_s, seq, ffn, stream); + bitvla_act_quant_cuda(ctx->d_gate_sq_up, ctx->d_act_int8_ffn, ctx->d_act_s, seq, ffn, ffn, stream); bitlinear_int8xint2_m(ctx->d_act_int8_ffn, lr.down_packed, ctx->d_down_out, ctx->d_act_s, lr.down_ws, seq, hidden, ffn, stream); l0_dump("L0_09_down_out", ctx->d_down_out, (size_t)seq * hidden); @@ -578,6 +563,7 @@ __global__ void layernorm_bias_bf16_kernel(const __nv_bfloat16* __restrict__ x, } __syncthreads(); const float mean = smem[0]/(float)K; + __syncthreads(); float vsum = 0.0f; for (int k=tid; k= total_cols) - return; - x[(size_t)m * total_cols+k] = __float2bfloat16(0.0f); -} - extern "C" void bitvla_layernorm_bf16(const __nv_bfloat16* x, const __nv_bfloat16* w, const __nv_bfloat16* b, __nv_bfloat16* out, float eps, int M, int K, cudaStream_t stream) { @@ -655,15 +633,6 @@ extern "C" void bitvla_add_bias_bf16(const __nv_bfloat16* x, const __nv_bfloat16 const int n_kb = (K+B-1)/B; add_bias_bf16_kernel<<>>(x, bias, out, M, K); } -extern "C" void bitvla_zero_tail_bf16(__nv_bfloat16* x, int M, int total_cols, - int start_col, cudaStream_t stream) { - if (start_col >= total_cols) - return; - constexpr int B = 128; - const int len = total_cols-start_col; - const int n_kb = (len+B-1)/B; - zero_tail_bf16_kernel<<>>(x, total_cols, start_col); -} extern "C" int bitvla_lm_cuda_forward(bitvla_lm_cuda_ctx* ctx, const __nv_bfloat16* d_in, @@ -720,5 +689,6 @@ extern "C" int bitvla_lm_cuda_forward(bitvla_lm_cuda_ctx* ctx, cudaStreamSynchronize(stream); dump_to_file("lm_final", d_out); } + CUDA_OK(cudaGetLastError()); return 0; } diff --git a/src/kernels/bitvla/bitvla_lm_cuda.h b/src/kernels/bitvla/bitvla_lm_cuda.h index ee79799..004fbaf 100644 --- a/src/kernels/bitvla/bitvla_lm_cuda.h +++ b/src/kernels/bitvla/bitvla_lm_cuda.h @@ -80,17 +80,6 @@ void bitvla_rope_neox_bf16(__nv_bfloat16* inout, const float* cos_tab, void bitvla_softmax_scaled_bf16(__nv_bfloat16* inout, float scale, int n_rows, int S, cudaStream_t stream); -/** - * @brief Elementwise @c relu(g)^2*u (BitVLA squared-ReLU FFN gate). - * @param g Gate input (N), bf16 device pointer. - * @param u Up input (N), bf16 device pointer. - * @param out Output (N), bf16 device pointer. - * @param N Element count. - * @param stream CUDA stream. - */ -void bitvla_squared_relu_mul_bf16(const __nv_bfloat16* g, const __nv_bfloat16* u, - __nv_bfloat16* out, int N, cudaStream_t stream); - /** * @brief Elementwise bf16 add (@p out = @p a + @p b). * @param a Length-@p N input, device pointer. @@ -191,16 +180,6 @@ void bitvla_gelu_tanh_bf16(const __nv_bfloat16* x, __nv_bfloat16* out, void bitvla_add_bias_bf16(const __nv_bfloat16* x, const __nv_bfloat16* bias, __nv_bfloat16* out, int M, int K, cudaStream_t stream); -/** - * @brief Zero out columns [@p start_col, @p total_cols) of an - * (M x @p total_cols) bf16 matrix, leaving the first @p start_col - * columns untouched. - * - * Used to mask out padding lanes added by the ternary GEMM's column tiling. - */ -void bitvla_zero_tail_bf16(__nv_bfloat16* x, int M, int total_cols, - int start_col, cudaStream_t stream); - /** * @brief Per-layer weight pointers for one BitVLA LM transformer block. * diff --git a/src/kernels/bitvla/bitvla_vit_cuda.cu b/src/kernels/bitvla/bitvla_vit_cuda.cu index 2f48e3e..51cbbc0 100644 --- a/src/kernels/bitvla/bitvla_vit_cuda.cu +++ b/src/kernels/bitvla/bitvla_vit_cuda.cu @@ -28,7 +28,7 @@ extern "C" void bitlinear_int8xint2_m(int8_t* A, int8_t* B, __nv_bfloat16* out, int M, int N, int K, cudaStream_t stream); extern "C" void bitvla_act_quant_cuda(const __nv_bfloat16* in, int8_t* out, float* scales, - int M, int K, cudaStream_t stream); + int M, int K, int ld_out, cudaStream_t stream); __global__ void gelu_erf_bf16_kernel(const __nv_bfloat16* in, __nv_bfloat16* out, int N) { const int i = (int)(blockIdx.x*blockDim.x+threadIdx.x); @@ -48,7 +48,7 @@ static void gelu_erf_bf16(const __nv_bfloat16* in, __nv_bfloat16* out, int N, cu return -1; } } while (0) #define CUDA_OKV(call) do { cudaError_t e = (call); if (e != cudaSuccess) { \ std::fprintf(stderr, "vla(bitvla_vit_cuda): %s @ %s:%d\n", cudaGetErrorString(e), __FILE__, __LINE__); \ - return nullptr; } } while (0) + bitvla_vit_cuda_free(ctx); return nullptr; } } while (0) struct bitvla_vit_cuda_ctx { int n_layers, hidden, n_heads, head_dim, ffn, n_patches, patch_flat, mm_out; @@ -84,7 +84,6 @@ struct bitvla_vit_cuda_ctx { __nv_bfloat16* d_attn_merged = nullptr; __nv_bfloat16* d_o_out = nullptr; __nv_bfloat16* d_fc1_dense = nullptr; - __nv_bfloat16* d_fc1_padded = nullptr; __nv_bfloat16* d_fc2_out = nullptr; __nv_bfloat16* d_mm_h1 = nullptr; }; @@ -116,6 +115,7 @@ bitvla_vit_cuda_ctx* bitvla_vit_cuda_init(int n_layers, int hidden, int n_heads, CUDA_OKV(cudaMalloc(&ctx->d_h_norm, (size_t) n_patches * hidden * bf16)); CUDA_OKV(cudaMalloc(&ctx->d_act_int8_h, (size_t) n_patches * hidden)); CUDA_OKV(cudaMalloc(&ctx->d_act_int8_ffn, (size_t) n_patches * ctx->ffn_pad)); + CUDA_OKV(cudaMemset(ctx->d_act_int8_ffn, 0, (size_t) n_patches * ctx->ffn_pad)); CUDA_OKV(cudaMalloc(&ctx->d_act_s, (size_t) n_patches * sizeof(float))); CUDA_OKV(cudaMalloc(&ctx->d_q_proj, (size_t) n_patches * hidden * bf16)); CUDA_OKV(cudaMalloc(&ctx->d_k_proj, (size_t) n_patches * hidden * bf16)); @@ -128,9 +128,6 @@ bitvla_vit_cuda_ctx* bitvla_vit_cuda_init(int n_layers, int hidden, int n_heads, CUDA_OKV(cudaMalloc(&ctx->d_attn_merged, (size_t) n_patches * hidden * bf16)); CUDA_OKV(cudaMalloc(&ctx->d_o_out, (size_t) n_patches * hidden * bf16)); CUDA_OKV(cudaMalloc(&ctx->d_fc1_dense, (size_t) n_patches * ffn * bf16)); - CUDA_OKV(cudaMalloc(&ctx->d_fc1_padded, (size_t) n_patches * ctx->ffn_pad*bf16)); - - CUDA_OKV(cudaMemset(ctx->d_fc1_padded, 0, (size_t) n_patches * ctx->ffn_pad*bf16)); CUDA_OKV(cudaMalloc(&ctx->d_fc2_out, (size_t) n_patches * hidden * bf16)); CUDA_OKV(cudaMalloc(&ctx->d_mm_h1, (size_t) n_patches * mm_out * bf16)); return ctx; @@ -146,7 +143,7 @@ void bitvla_vit_cuda_free(bitvla_vit_cuda_ctx* ctx) { cudaFree(ctx->d_q_HShd); cudaFree(ctx->d_k_HShd); cudaFree(ctx->d_v_HShd); cudaFree(ctx->d_scores); cudaFree(ctx->d_attn_out); cudaFree(ctx->d_attn_merged); cudaFree(ctx->d_o_out); - cudaFree(ctx->d_fc1_dense); cudaFree(ctx->d_fc1_padded); cudaFree(ctx->d_fc2_out); + cudaFree(ctx->d_fc1_dense); cudaFree(ctx->d_fc2_out); cudaFree(ctx->d_mm_h1); delete ctx; } @@ -178,7 +175,7 @@ static int run_vit_layer(bitvla_vit_cuda_ctx* ctx, int L, cudaStream_t stream) { const int hd = ctx->head_dim, ffn = ctx->ffn, ffn_pad = ctx->ffn_pad; bitvla_layernorm_bf16(ctx->d_h, lr.ln1_w, lr.ln1_b, ctx->d_h_norm, ctx->ln_eps, seq, H, stream); - bitvla_act_quant_cuda(ctx->d_h_norm, ctx->d_act_int8_h, ctx->d_act_s, seq, H, stream); + bitvla_act_quant_cuda(ctx->d_h_norm, ctx->d_act_int8_h, ctx->d_act_s, seq, H, H, stream); bitlinear_int8xint2_m(ctx->d_act_int8_h, lr.q_packed, ctx->d_q_proj, ctx->d_act_s, lr.q_ws, seq, H, H, stream); bitvla_add_bias_bf16(ctx->d_q_proj, lr.q_b, ctx->d_q_proj, seq, H, stream); @@ -225,14 +222,14 @@ static int run_vit_layer(bitvla_vit_cuda_ctx* ctx, int L, cudaStream_t stream) { bitvla_transpose_NshHd_to_sNhd_bf16(ctx->d_attn_out, ctx->d_attn_merged, n_heads, seq, hd, stream); - bitvla_act_quant_cuda(ctx->d_attn_merged, ctx->d_act_int8_h, ctx->d_act_s, seq, H, stream); + bitvla_act_quant_cuda(ctx->d_attn_merged, ctx->d_act_int8_h, ctx->d_act_s, seq, H, H, stream); bitlinear_int8xint2_m(ctx->d_act_int8_h, lr.o_packed, ctx->d_o_out, ctx->d_act_s, lr.o_ws, seq, H, H, stream); bitvla_add_bias_bf16(ctx->d_o_out, lr.o_b, ctx->d_o_out, seq, H, stream); bitvla_add_bf16(ctx->d_h, ctx->d_o_out, ctx->d_h, seq * H, stream); bitvla_layernorm_bf16(ctx->d_h, lr.ln2_w, lr.ln2_b, ctx->d_h_norm, ctx->ln_eps, seq, H, stream); - bitvla_act_quant_cuda(ctx->d_h_norm, ctx->d_act_int8_h, ctx->d_act_s, seq, H, stream); + bitvla_act_quant_cuda(ctx->d_h_norm, ctx->d_act_int8_h, ctx->d_act_s, seq, H, H, stream); bitlinear_int8xint2_m(ctx->d_act_int8_h, lr.fc1_packed, ctx->d_fc1_dense, ctx->d_act_s, lr.fc1_ws, seq, ffn, H, stream); @@ -240,14 +237,7 @@ static int run_vit_layer(bitvla_vit_cuda_ctx* ctx, int L, cudaStream_t stream) { bitvla_gelu_tanh_bf16(ctx->d_fc1_dense, ctx->d_fc1_dense, seq * ffn, stream); - cudaMemcpy2DAsync( - ctx->d_fc1_padded, (size_t) ffn_pad * sizeof(__nv_bfloat16), - ctx->d_fc1_dense, (size_t) ffn * sizeof(__nv_bfloat16), - (size_t) ffn * sizeof(__nv_bfloat16), - seq, - cudaMemcpyDeviceToDevice, stream); - - bitvla_act_quant_cuda(ctx->d_fc1_padded, ctx->d_act_int8_ffn, ctx->d_act_s, seq, ffn_pad, stream); + bitvla_act_quant_cuda(ctx->d_fc1_dense, ctx->d_act_int8_ffn, ctx->d_act_s, seq, ffn, ffn_pad, stream); bitlinear_int8xint2_m(ctx->d_act_int8_ffn, lr.fc2_packed, ctx->d_fc2_out, ctx->d_act_s, lr.fc2_ws, seq, H, ffn_pad, stream); bitvla_add_bias_bf16(ctx->d_fc2_out, lr.fc2_b, ctx->d_fc2_out, seq, H, stream); @@ -316,5 +306,6 @@ int bitvla_vit_cuda_forward(bitvla_vit_cuda_ctx* ctx, return -1; } bitvla_add_bias_bf16(d_out, ctx->mm_b2, d_out, seq, M, stream); + CUDA_OK(cudaGetLastError()); return 0; } diff --git a/src/layers/attn.h b/src/layers/attn.h index bef6af2..25271e5 100644 --- a/src/layers/attn.h +++ b/src/layers/attn.h @@ -38,7 +38,7 @@ inline ggml_tensor * to_heads_v(ggml_context * C, ggml_tensor * p, int64_t hd, i inline ggml_tensor * attention(ggml_context * C, ggml_tensor * Q, ggml_tensor * K, ggml_tensor * V, ggml_tensor * mask, float scale, int64_t dim, int64_t T, int64_t nv = 1) { ggml_tensor * kq = ggml_mul_mat(C, K, Q); - ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_prec_set_acc(kq, GGML_PREC_F32); ggml_tensor * aw = ggml_soft_max_ext(C, kq, mask, scale, 0.0f); ggml_tensor * kqv = ggml_mul_mat(C, V, aw); @@ -52,8 +52,8 @@ inline ggml_tensor * flash_attention(ggml_context * C, ggml_tensor * Q, ggml_ten ggml_tensor * vf = V->type == GGML_TYPE_F16 ? V : ggml_cast(C, V, GGML_TYPE_F16); ggml_tensor * o = ggml_flash_attn_ext(C, Q, kf, vf, mask, scale, 0.0f, 0.0f); - ggml_flash_attn_ext_set_prec(o, GGML_PREC_F32); - return ggml_reshape_2d(C, o, o->ne[0]*o->ne[1], o->ne[2]*o->ne[3]); + ggml_prec_set_acc(o, GGML_PREC_F32); + return ggml_reshape_3d(C, o, o->ne[0]*o->ne[1], o->ne[2], o->ne[3]); } } diff --git a/src/layers/linear.h b/src/layers/linear.h index dd3a7a7..0de035a 100644 --- a/src/layers/linear.h +++ b/src/layers/linear.h @@ -26,16 +26,6 @@ inline ggml_tensor * linear(ggml_context * C, ggml_tensor * W, ggml_tensor * b, return b ? ggml_add(C, y, b) : y; } -// One row of a stacked [out, in, n_embodiment] weight. -inline ggml_tensor * cat_linear(ggml_context * C, ggml_tensor * W3d, ggml_tensor * b2d, int64_t id, ggml_tensor * x) { - const int64_t out = W3d->ne[0]; - const int64_t in = W3d->ne[1]; - - ggml_tensor * W_id = ggml_view_2d(C, W3d, out, in, W3d->nb[1], (size_t)id*W3d->nb[2]); - ggml_tensor * y = ggml_mul_mat(C, ggml_cont(C, ggml_transpose(C, W_id)), x); - return ggml_add(C, y, ggml_view_1d(C, b2d, out, (size_t)id*b2d->nb[1])); -} - // Block `blk` of a fused [nblk*E, T] projection, laid out as heads. inline ggml_tensor * head_view(ggml_context * C, ggml_tensor * proj, int64_t hd, int64_t heads, int64_t T, int64_t E, int nblk, int blk) { diff --git a/src/layers/norm.h b/src/layers/norm.h index a10acf7..62d77fe 100644 --- a/src/layers/norm.h +++ b/src/layers/norm.h @@ -19,9 +19,6 @@ #include "ggml.h" -#include -#include - namespace vla { inline ggml_tensor * layer_norm(ggml_context * C, ggml_tensor * x, ggml_tensor * w, ggml_tensor * b, float eps) { @@ -33,15 +30,4 @@ inline ggml_tensor * rms_norm(ggml_context * C, ggml_tensor * x, ggml_tensor * w return w ? ggml_mul(C, n, w) : n; } -// cond is (scale, shift) here; the DiT final projection uses (shift, scale). -inline ggml_tensor * adaln(ggml_context * C, ggml_tensor * x, ggml_tensor * temb, - ggml_tensor * lw, ggml_tensor * lb, int64_t dim, float eps) { - ggml_tensor * cond = linear(C, lw, lb, ggml_silu(C, temb)); - ggml_tensor * sc = ggml_view_1d(C, cond, dim, 0); - ggml_tensor * sh = ggml_view_1d(C, cond, dim, (size_t)dim*sizeof(float)); - - ggml_tensor * xn = ggml_norm(C, x, eps); - return ggml_add(C, ggml_add(C, xn, ggml_mul(C, xn, sc)), sh); -} - } diff --git a/src/layers/rope.h b/src/layers/rope.h index ca06156..ded9bde 100644 --- a/src/layers/rope.h +++ b/src/layers/rope.h @@ -21,7 +21,9 @@ #include "ggml.h" +#include #include +#include namespace vla { @@ -75,4 +77,19 @@ inline ggml_tensor * rope_pairwise(ggml_context * C, ggml_tensor * x, ggml_tenso return ggml_add(C, ggml_mul(C, x, c), ggml_mul(C, rope_pairwise_rot(C, x, HD), s)); } +inline void rope_pairwise_table(int64_t HD, int64_t T, float base, std::vector & cs, std::vector & sn) { + const int64_t half = HD/2; + cs.resize((size_t)(HD*T)); + sn.resize((size_t)(HD*T)); + for (int64_t t = 0; t < T; ++t) { + for (int64_t mi = 0; mi < HD; ++mi) { + const int64_t j = mi % half; + const double inv = 1.0/std::pow((double)base, (2.0*j)/(double)HD); + const double a = (double)t*inv; + cs[t*HD + mi] = (float)std::cos(a); + sn[t*HD + mi] = (float)std::sin(a); + } + } +} + } diff --git a/src/loader.cpp b/src/loader.cpp index 0137731..340fb2c 100644 --- a/src/loader.cpp +++ b/src/loader.cpp @@ -106,8 +106,9 @@ ggml_tensor * WeightLoader::fuse(ggml_type want, const char * out_name, const st return nullptr; } - const bool is1d = ggml_n_dims(first) == 1; - int64_t rows = 0; + const ggml_type rt = g_.resident_type(first, want); + const bool is1d = ggml_n_dims(first) == 1; + int64_t rows = 0; for (const std::string & s : srcs) { const ggml_tensor * gs = g_.meta(s.c_str()); if (!gs) { @@ -115,11 +116,16 @@ ggml_tensor * WeightLoader::fuse(ggml_type want, const char * out_name, const st ok_ = false; return nullptr; } + if (g_.resident_type(gs, want) != rt || (!is1d && gs->ne[0] != first->ne[0])) { + std::fprintf(stderr, "vla(%s): %s does not match %s for fusing\n", arch_, s.c_str(), srcs[0].c_str()); + ok_ = false; + return nullptr; + } rows += is1d ? gs->ne[0] : gs->ne[1]; } - ggml_tensor * t = is1d ? ggml_new_tensor_1d(ctx_, want, rows) - : ggml_new_tensor_2d(ctx_, want, first->ne[0], rows); + ggml_tensor * t = is1d ? ggml_new_tensor_1d(ctx_, rt, rows) + : ggml_new_tensor_2d(ctx_, rt, first->ne[0], rows); if (!t) { std::fprintf(stderr, "vla(%s): ggml_new_tensor failed for %s\n", arch_, out_name); ok_ = false; diff --git a/src/model.cpp b/src/model.cpp index 6bf4f55..ecbecdf 100644 --- a/src/model.cpp +++ b/src/model.cpp @@ -30,6 +30,8 @@ namespace vla { struct Model { std::unique_ptr impl; + bool fa = false; + bool mm = true; }; namespace { @@ -107,12 +109,10 @@ bool detect_arch_gguf(const std::string& path, Arch* out) { *out = Arch::GR00T_N1_7; ok = true; } -#ifdef VLA_USE_OCTO else if (arch_str == "octo" || arch_str == "octo-small-1.5") { *out = Arch::OCTO; ok = true; } -#endif else if (arch_str == "bitvla") { *out = Arch::BITVLA; ok = true; @@ -209,7 +209,13 @@ bool detect_arch_from_ckpt(const std::string& ckpt_path, Arch* out) { Model* model_load(const std::string& mmproj_path, const std::string& ckpt_path, const std::string& config_path) { - return model_load(mmproj_path, ckpt_path, config_path, Options{}); + Options o; + std::string err; + if (!o.load_json(config_path, err)) { + std::fprintf(stderr, "vla: %s\n", err.c_str()); + return nullptr; + } + return model_load(mmproj_path, ckpt_path, config_path, o); } Model* model_load(const std::string& mmproj_path, const std::string& ckpt_path, @@ -232,8 +238,26 @@ Model* model_load(const std::string& mmproj_path, const std::string& ckpt_path, } } - set_flash_attn(opts.flash_attn.value_or(default_flash_attn())); - set_mm_prec_f32(opts.mm_prec_f32.value_or(true)); + if (opts.num_steps) { + const char * name = arch == Arch::OCTO ? "octo" + : arch == Arch::BITVLA ? "bitvla" + : arch == Arch::VLA_ADAPTER ? "vla_adapter" + : arch == Arch::OPENVLA_OFT ? "openvla_oft" + : arch == Arch::TURBOVLA ? "turbovla" : nullptr; + if (name) { + std::fprintf(stderr, "vla(%s): num_steps is not supported\n", name); + return nullptr; + } + } + if (opts.act_dtype == GGML_TYPE_BF16 && arch != Arch::PI0 && arch != Arch::EVO1) { + std::fprintf(stderr, "vla: act_dtype bf16 is only supported by pi0 and evo1\n"); + return nullptr; + } + + const bool fa = opts.flash_attn.value_or(default_flash_attn()); + const bool mm = opts.mm_prec_f32.value_or(true); + set_flash_attn(fa); + set_mm_prec_f32(mm); switch (arch) { case Arch::SMOLVLA: @@ -264,12 +288,10 @@ Model* model_load(const std::string& mmproj_path, const std::string& ckpt_path, std::printf("vla: arch = gr00t_n1_7\n"); impl = gr00t_n1_7_create(mmproj_path, ckpt_path, config_path, opts); break; -#ifdef VLA_USE_OCTO case Arch::OCTO: std::printf("vla: arch = octo\n"); impl = octo_create(mmproj_path, ckpt_path, config_path); break; -#endif case Arch::BITVLA: std::printf("vla: arch = bitvla\n"); impl = bitvla_create(mmproj_path, ckpt_path, config_path, opts); @@ -300,6 +322,8 @@ Model* model_load(const std::string& mmproj_path, const std::string& ckpt_path, auto* m = new Model(); m->impl = std::move(impl); + m->fa = fa; + m->mm = mm; return m; } @@ -317,6 +341,8 @@ const Stats& last_stats(const Model* m) { std::vector predict(Model* m, const Inputs& in) { if (!m || !m->impl) return {}; + set_flash_attn(m->fa); + set_mm_prec_f32(m->mm); return m->impl->predict(in); } diff --git a/src/model.h b/src/model.h index a918831..e6b6622 100644 --- a/src/model.h +++ b/src/model.h @@ -95,9 +95,9 @@ struct Model; */ enum class TimingDetail { NONE, ///< Only @c ms_total is populated. - /// Per-phase timings (vision, prefill, denoise, ...). SmolVLA uses a second - /// builder here that does not pad the prefix to @c n_lang; same positions and - /// masking, so it differs from @c NONE only by float reduction order. + /// Per-phase timings (vision, prefill, denoise, ...). SmolVLA builds the graph + /// here without padding the prefix to @c n_lang; same positions and masking, + /// so it differs from @c NONE only by float reduction order. PHASE, }; @@ -132,8 +132,8 @@ struct Inputs { int n_images; ///< Number of @ref images. /// Pre-computed image embeddings, [n_img_views * n_img, hidden]; bypasses the - /// vision tower. Passed to the LM as-is, so the scale is arch-specific: pi0 - /// expects the projector output times 1/sqrt(hidden), pi0.5 expects it raw. + /// vision tower. Passed to the LM as-is; pi0 and pi0.5 expect the raw + /// projector output. const float* precomputed_img_emb = nullptr; int n_img_views = 0; ///< Number of views in /// @ref precomputed_img_emb. diff --git a/src/models/bitvla.cpp b/src/models/bitvla.cpp index 71df94f..a3509d2 100644 --- a/src/models/bitvla.cpp +++ b/src/models/bitvla.cpp @@ -27,6 +27,8 @@ #include "backend.h" #include "gguf_reader.h" #include "scratch_ctx.h" +#include "layers/norm.h" +#include "modules/preprocess.h" #ifdef VLA_BITVLA_CUDA_KERNELS #include "kernels/bitvla/bitvla_lm_cuda.h" @@ -102,7 +104,6 @@ struct BitvlaModelArch : public ModelArchBase { BitvlaModelArch() : ModelArchBase(Arch::BITVLA) {} ~BitvlaModelArch() override; - std::string gguf_path; gguf_reader emb_reader{"bitvla"}; // stays open for per-step token-embedding row fetches std::vector stop_embed; // cached constant stop-token embedding row ggml_backend_t backend = nullptr; @@ -133,7 +134,7 @@ struct BitvlaModelArch : public ModelArchBase { std::vector vit; ggml_tensor *mm_l1_w = nullptr, *mm_l1_b = nullptr, *mm_l2_w = nullptr, *mm_l2_b = nullptr; ggml_tensor *pp_fc1_w = nullptr, *pp_fc1_b = nullptr, *pp_fc2_w = nullptr, *pp_fc2_b = nullptr; - ggml_tensor *embed_tokens = nullptr, *lm_output_norm = nullptr; + ggml_tensor *lm_output_norm = nullptr; std::vector lm; ggml_tensor *ah_ln1_w = nullptr, *ah_ln1_b = nullptr, *ah_fc1_w = nullptr, *ah_fc1_b = nullptr; ggml_tensor *ah_b0_ln_w = nullptr, *ah_b0_ln_b = nullptr, *ah_b0_w = nullptr, *ah_b0_b = nullptr; @@ -151,6 +152,7 @@ struct BitvlaModelArch : public ModelArchBase { bool cuda_vit_ready = false; std::vector cpu_kept_ptrs; + int cuda_dev = 0; __nv_bfloat16* d_inputs_embeds = nullptr; __nv_bfloat16* d_last_hidden = nullptr; @@ -172,32 +174,25 @@ ggml_tensor * act_quant(ggml_context * C, ggml_tensor * x) { } ggml_tensor * bit_linear(ggml_context * C, ggml_tensor * W, ggml_tensor * b, ggml_tensor * x) { - ggml_tensor * y = ggml_mul_mat(C, W, act_quant(C, x)); - return b ? ggml_add(C, y, b) : y; -} -ggml_tensor * layernorm(ggml_context * C, ggml_tensor * x, ggml_tensor * w, ggml_tensor * b, float eps) { - return ggml_add(C, ggml_mul(C, ggml_norm(C, x, eps), w), b); -} -ggml_tensor * rmsnorm(ggml_context * C, ggml_tensor * x, ggml_tensor * w, float eps) { - return ggml_mul(C, ggml_rms_norm(C, x, eps), w); + return linear(C, W, b, act_quant(C, x)); } ggml_tensor * build_vit_layer(ggml_context * C, const VitLayerW & w, ggml_tensor * x, int64_t seq, int64_t heads, int64_t head_dim, int64_t hidden, float ln_eps) { const float scale = 1.0f/std::sqrt((float) head_dim); - ggml_tensor * x1 = layernorm(C, x, w.ln1w, w.ln1b, ln_eps); + ggml_tensor * x1 = layer_norm(C, x, w.ln1w, w.ln1b, ln_eps); ggml_tensor * q = bit_linear(C, w.Wq, w.bq, x1); ggml_tensor * k = bit_linear(C, w.Wk, w.bk, x1); ggml_tensor * v = bit_linear(C, w.Wv, w.bv, x1); ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, head_dim, heads, seq), 0, 2, 1, 3)); ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, head_dim, heads, seq), 0, 2, 1, 3)); ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_prec_set_acc(kq, GGML_PREC_F32); ggml_tensor * att= ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); ggml_tensor * y = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, att), 0, 2, 1, 3)), hidden, seq); ggml_tensor * o = bit_linear(C, w.Wo, w.bo, y); ggml_tensor * h1 = ggml_add(C, x, o); - ggml_tensor * x2 = layernorm(C, h1, w.ln2w, w.ln2b, ln_eps); + ggml_tensor * x2 = layer_norm(C, h1, w.ln2w, w.ln2b, ln_eps); ggml_tensor * f1 = ggml_gelu(C, bit_linear(C, w.Wfc1, w.bfc1, x2)); ggml_tensor * f2 = bit_linear(C, w.Wfc2, w.bfc2, f1); return ggml_add(C, h1, f2); @@ -208,7 +203,7 @@ ggml_tensor * build_lm_layer(ggml_context * C, const BitvlaModelArch & m, const const int64_t hd = m.lm_head_dim, n_q = m.lm_q, n_kv = m.lm_kv, hq = n_q * hd; const float scale = 1.0f/std::sqrt((float) hd); - ggml_tensor * hn = rmsnorm(C, h, w.attn_norm, m.lm_rms_eps); + ggml_tensor * hn = rms_norm(C, h, w.attn_norm, m.lm_rms_eps); ggml_tensor * qp = bit_linear(C, w.Wq, nullptr, hn); ggml_tensor * kp = bit_linear(C, w.Wk, nullptr, hn); ggml_tensor * vp = bit_linear(C, w.Wv, nullptr, hn); @@ -220,23 +215,23 @@ ggml_tensor * build_lm_layer(ggml_context * C, const BitvlaModelArch & m, const ggml_tensor * Q = ggml_cont(C, ggml_permute(C, qR, 0, 2, 1, 3)); ggml_tensor * K = ggml_cont(C, ggml_permute(C, kR, 0, 2, 1, 3)); ggml_tensor * V = ggml_cont(C, ggml_permute(C, v3, 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_prec_set_acc(kq, GGML_PREC_F32); // Unmasked on purpose, same as openvla_oft: BitVLA is fine-tuned with // OpenVLA-OFT's recipe, which swaps the causal mask for a bidirectional one // so the action chunk decodes in a single pass. ggml_tensor * att= ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); ggml_tensor * kqv= ggml_mul_mat(C, V, att); ggml_tensor * mer= ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, kqv, 0, 2, 1, 3)), hq, seq); - ggml_tensor * sub= rmsnorm(C, mer, w.attn_sub_norm, m.lm_rms_eps); + ggml_tensor * sub= rms_norm(C, mer, w.attn_sub_norm, m.lm_rms_eps); ggml_tensor * o = bit_linear(C, w.Wo, nullptr, sub); ggml_tensor * h1 = ggml_add(C, h, o); - ggml_tensor * h2 = rmsnorm(C, h1, w.ffn_norm, m.lm_rms_eps); + ggml_tensor * h2 = rms_norm(C, h1, w.ffn_norm, m.lm_rms_eps); ggml_tensor * g = bit_linear(C, w.Wgate, nullptr, h2); ggml_tensor * u = bit_linear(C, w.Wup, nullptr, h2); ggml_tensor * gsq= ggml_sqr(C, ggml_relu(C, g)); ggml_tensor * gu = ggml_mul(C, gsq, u); - ggml_tensor * fsub= rmsnorm(C, gu, w.ffn_sub_norm, m.lm_rms_eps); + ggml_tensor * fsub= rms_norm(C, gu, w.ffn_sub_norm, m.lm_rms_eps); ggml_tensor * dn = bit_linear(C, w.Wdown, nullptr, fsub); return ggml_add(C, h1, dn); } @@ -391,6 +386,11 @@ bool load_config(const gguf_reader & g, BitvlaModelArch & m, Config & cfg) { (long long) m.lm_q, (long long) m.lm_head_dim, (long long) m.lm_hidden); return false; } + if (m.vit_heads <= 0 || m.vit_head_dim <= 0 || m.vit_heads*m.vit_head_dim != m.vit_hidden) { + std::fprintf(stderr, "vla(bitvla): vit heads %lld x head_dim %lld does not match hidden %lld\n", + (long long) m.vit_heads, (long long) m.vit_head_dim, (long long) m.vit_hidden); + return false; + } const std::string js = g.str("bitvla.statistics_json"); if (js.empty()) { @@ -433,16 +433,13 @@ namespace { static void recover_ternary_and_scale(const float* W, int64_t n, std::vector& ternary, float& absmean) { - // Per-tensor absmean scale (1/mean|W|), matching scripts/convert_bitvla_to_gguf.py; - // the int2-packed path bakes the same scale. - double s = 0.0; + float amax = 0.0f; for (int64_t i=0; i 0 ? (float) (s/(double) n) : 0.0f; - if (mean < 1e-5f) - mean = 1e-5f; - absmean = mean; - const float inv = 1.0f/mean; + amax = std::max(amax, std::fabs(W[i])); + if (amax < 1e-5f) + amax = 1e-5f; + absmean = amax; + const float inv = 1.0f/amax; ternary.resize(n); for (int64_t i=0; i pack_ladder_int2(const int8_t* W, int64_t N, int64_t return out; } -static inline uint16_t f32_to_bf16_u16(float f) { - uint32_t u; std::memcpy(&u, &f, 4); - return (uint16_t)(u >> 16); -} - static __nv_bfloat16* upload_bf16_from_f32(const float* h, size_t n, std::vector& out_ptrs) { - std::vector tmp(n); - for (size_t i=0; i tmp(n); + ggml_fp32_to_bf16_row(h, tmp.data(), (int64_t) n); __nv_bfloat16* d = nullptr; cudaMalloc(&d, n * sizeof(__nv_bfloat16)); cudaMemcpy(d, tmp.data(), n * sizeof(__nv_bfloat16), cudaMemcpyHostToDevice); @@ -551,6 +542,8 @@ static int8_t* pack_and_upload_fused(const std::vector& wptrs, BitvlaModelArch::~BitvlaModelArch() { #ifdef VLA_BITVLA_CUDA_KERNELS + if (lm_cuda_ctx || !cuda_devptrs.empty()) + cudaSetDevice(cuda_dev); if (lm_cuda_ctx) bitvla_lm_cuda_free(lm_cuda_ctx); if (vit_cuda_ctx) @@ -592,7 +585,6 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, std::printf("vla(bitvla): note - mmproj '%s' is ignored (the BitSigLIP-L vision tower is bundled in the combined GGUF)\n", mmproj_path.c_str()); auto m = std::make_unique(); - m->gguf_path = ckpt_path; m->matmul_type = opts.weight_dtype.value_or(GGML_TYPE_F32); gguf_reader g("bitvla"); @@ -603,9 +595,11 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, } if (!load_config(g, *m, m->cfg)) return nullptr; + if (m->packed_int2) + m->matmul_type = GGML_TYPE_F32; // Keep one reader open for the per-step token-embedding fetches (token_embd - // stays on disk under int2 packing) and cache the constant stop-token row, + // stays on disk) and cache the constant stop-token row, // so predict() no longer re-opens and re-parses the GGUF twice per call. if (!m->emb_reader.open(ckpt_path)) return nullptr; @@ -644,13 +638,11 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, auto mk_f32 = [&](const char * name) { return L.f32 ("%s", name); }; auto mk_bit = [&](const char * name) { return L.typed(m->packed_int2 ? GGML_TYPE_I8 : m->matmul_type, "%s", name); }; - bool ok = true; - m->vit_patch_w = mk_mm("vit.patch_embd.weight"); m->vit_patch_b = mk_f32("vit.patch_embd.bias"); m->vit_pos = mk_f32("vit.pos_embd.weight"); m->vit.resize(m->vit_layers); - for (int64_t i=0; ivit_layers && ok; ++i) { + for (int64_t i=0; ivit_layers; ++i) { char p[64]; auto N = [&](const char * s) { std::snprintf(p, sizeof(p), "vit.blk.%lld.%s", (long long) i, s); return p; }; auto & w = m->vit[i]; w.ln1w=mk_f32(N("ln1.weight")); w.ln1b=mk_f32(N("ln1.bias")); @@ -661,7 +653,6 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, w.Wo=mk_bit(N("attn_o.weight")); w.bo=mk_f32(N("attn_o.bias")); w.Wfc1=mk_bit(N("fc1.weight")); w.bfc1=mk_f32(N("fc1.bias")); w.Wfc2=mk_bit(N("fc2.weight")); w.bfc2=mk_f32(N("fc2.bias")); - ok &= w.ln1w&&w.ln1b&&w.ln2w&&w.ln2b&&w.Wq&&w.bq&&w.Wk&&w.bk&&w.Wv&&w.bv&&w.Wo&&w.bo&&w.Wfc1&&w.bfc1&&w.Wfc2&&w.bfc2; } m->mm_l1_w=mk_mm("mm.linear_1.weight"); m->mm_l1_b=mk_f32("mm.linear_1.bias"); @@ -670,10 +661,9 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, m->pp_fc1_w=mk_f32("aex.proprio.fc1.weight"); m->pp_fc1_b=mk_f32("aex.proprio.fc1.bias"); m->pp_fc2_w=mk_f32("aex.proprio.fc2.weight"); m->pp_fc2_b=mk_f32("aex.proprio.fc2.bias"); - m->embed_tokens = m->packed_int2 ? nullptr : mk_mm("token_embd.weight"); m->lm_output_norm = mk_f32("lm.output_norm.weight"); m->lm.resize(m->lm_layers); - for (int64_t i=0; ilm_layers && ok; ++i) { + for (int64_t i=0; ilm_layers; ++i) { char p[64]; auto N = [&](const char * s) { std::snprintf(p, sizeof(p), "lm.blk.%lld.%s", (long long) i, s); return p; }; auto & w = m->lm[i]; w.attn_norm = mk_f32(N("attn_norm.weight")); @@ -688,8 +678,6 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, } else { w.Wgate=mk_mm(N("ffn_gate.weight")); w.Wup=mk_mm(N("ffn_up.weight")); } - ok &= w.attn_norm&&w.attn_sub_norm&&w.ffn_norm&&w.ffn_sub_norm&&w.Wq&&w.Wk&&w.Wv&&w.Wo&&w.Wdown&& - (m->packed_int2 ? (w.Wgate_up != nullptr) : (w.Wgate && w.Wup)); } m->ah_ln1_w =mk_f32("aex.head.ln1.weight"); m->ah_ln1_b =mk_f32("aex.head.ln1.bias"); @@ -701,17 +689,6 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, m->ah_ln2_w =mk_f32("aex.head.ln2.weight"); m->ah_ln2_b =mk_f32("aex.head.ln2.bias"); m->ah_fc2_w =mk_mm ("aex.head.fc2.weight"); m->ah_fc2_b =mk_f32("aex.head.fc2.bias"); - ok &= m->vit_patch_w&&m->vit_patch_b&&m->vit_pos&&m->mm_l1_w&&m->mm_l1_b&&m->mm_l2_w&&m->mm_l2_b&& - m->pp_fc1_w&&m->pp_fc1_b&&m->pp_fc2_w&&m->pp_fc2_b&&(m->embed_tokens||m->packed_int2)&&m->lm_output_norm&& - m->ah_ln1_w&&m->ah_ln1_b&&m->ah_fc1_w&&m->ah_fc1_b&& - m->ah_b0_ln_w&&m->ah_b0_ln_b&&m->ah_b0_w&&m->ah_b0_b&& - m->ah_b1_ln_w&&m->ah_b1_ln_b&&m->ah_b1_w&&m->ah_b1_b&& - m->ah_ln2_w&&m->ah_ln2_b&&m->ah_fc2_w&&m->ah_fc2_b; - if (!ok) { - std::fprintf(stderr, "vla(bitvla): weight tensor setup failed\n"); - return nullptr; - } - if (!L.upload(m->backend, &m->weight_buf)) return nullptr; @@ -731,25 +708,40 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, #ifdef VLA_BITVLA_CUDA_KERNELS - if ((m->packed_int2 || m->matmul_type == GGML_TYPE_F32) && !vla::env_flag("VLA_BITVLA_NO_CUDA_LM")) { + const ggml_tensor * quantized = nullptr; + for (ggml_tensor * t = ggml_get_first_tensor(m->ctx_weights); t && !quantized; t = ggml_get_next_tensor(m->ctx_weights, t)) + if (ggml_is_quantized(t->type)) + quantized = t; + if (quantized) + std::fprintf(stderr, "vla(bitvla): %s is quantized; the CUDA path needs F32 weights\n", ggml_get_name(quantized)); + + if (m->matmul_type == GGML_TYPE_F32 && !quantized && !vla::env_flag("VLA_BITVLA_NO_CUDA_LM")) { int dev_count = 0; - if (cudaGetDeviceCount(&dev_count) == cudaSuccess && dev_count > 0) { - cudaSetDevice(0); + m->cuda_dev = vla::backend_device_index(); + if (cudaGetDeviceCount(&dev_count) == cudaSuccess && m->cuda_dev < dev_count && + cudaSetDevice(m->cuda_dev) == cudaSuccess) { + cudaGetLastError(); // The ladder kernels dereference the scale pointer unconditionally, so // a missing sidecar is a device-side OOB read, not a soft failure. - bool scales_ok = true; + bool weights_ok = true; auto load_bit = [&](ggml_tensor * t, int64_t N, int64_t K) -> std::pair { + const size_t want = m->packed_int2 ? (size_t) (N*K/4) : (size_t) (N*K)*sizeof(float); + if (ggml_nbytes(t) != want) { + std::fprintf(stderr, "vla(bitvla): %s is %zu bytes, expected %zu\n", ggml_get_name(t), ggml_nbytes(t), want); + weights_ok = false; + return { nullptr, nullptr }; + } if (m->packed_int2) { int8_t * dp = upload_int8((const uint8_t*) t->data, ggml_nbytes(t), m->cuda_devptrs); std::string nm = ggml_get_name(t); std::string sn = nm.substr(0, nm.size()-7) + ".scale"; std::vector sc = g.read_f32(sn.c_str()); - if (sc.empty()) { - std::fprintf(stderr, "vla(bitvla): int2 tensor %s has no %s sidecar\n", + if (sc.size() != 1) { + std::fprintf(stderr, "vla(bitvla): int2 tensor %s needs a 1-element %s sidecar\n", nm.c_str(), sn.c_str()); - scales_ok = false; + weights_ok = false; return { dp, nullptr }; } float * dws = upload_f32_scales(sc.data(), (int) sc.size(), m->cuda_devptrs); @@ -766,7 +758,7 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, (int) m->lm_inter, (int) m->lm_layers, m->lm_rope_base, m->lm_rms_eps, max_seq); if (m->lm_cuda_ctx) { bool pack_ok = true; - for (int64_t L=0; Llm_layers && pack_ok && scales_ok; ++L) { + for (int64_t L=0; Llm_layers && pack_ok && weights_ok; ++L) { bitvla_lm_layer_cuda lyr{}; lyr.attn_norm_w = upload_bf16_from_f32((const float*) m->lm[L].attn_norm->data, m->lm_hidden, m->cuda_devptrs); @@ -799,15 +791,20 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, } if (m->packed_int2) { - lyr.gate_up_packed = upload_int8((const uint8_t*) m->lm[L].Wgate_up->data, ggml_nbytes(m->lm[L].Wgate_up), m->cuda_devptrs); const std::string sn = "lm.blk." + std::to_string(L) + ".ffn_gate_up.scale"; std::vector sc = g.read_f32(sn.c_str()); - if (sc.empty()) { - std::fprintf(stderr, "vla(bitvla): missing %s\n", sn.c_str()); - scales_ok = false; + if (ggml_nbytes(m->lm[L].Wgate_up) != (size_t) (2*m->lm_inter*m->lm_hidden/4) || sc.size() != 2) { + std::fprintf(stderr, "vla(bitvla): lm.blk.%lld.ffn_gate_up weight or its 2-element %s is malformed\n", + (long long) L, sn.c_str()); + weights_ok = false; } else { + lyr.gate_up_packed = upload_int8((const uint8_t*) m->lm[L].Wgate_up->data, ggml_nbytes(m->lm[L].Wgate_up), m->cuda_devptrs); lyr.gate_up_ws = upload_f32_scales(sc.data(), (int) sc.size(), m->cuda_devptrs); } + } else if (ggml_nbytes(m->lm[L].Wgate) != (size_t) (m->lm_inter*m->lm_hidden)*sizeof(float) || + ggml_nbytes(m->lm[L].Wup) != (size_t) (m->lm_inter*m->lm_hidden)*sizeof(float)) { + std::fprintf(stderr, "vla(bitvla): lm.blk.%lld ffn_gate/ffn_up size mismatch\n", (long long) L); + weights_ok = false; } else { std::vector ws2; lyr.gate_up_packed = pack_and_upload_fused( @@ -823,11 +820,12 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, lyr.down_ws = r.second; } bitvla_lm_cuda_set_layer(m->lm_cuda_ctx, (int) L, &lyr); + pack_ok = cudaGetLastError() == cudaSuccess; } - if (!scales_ok) { - std::fprintf(stderr, "vla(bitvla): int2 scale sidecars incomplete; refusing the CUDA LM\n"); + if (!weights_ok) { + std::fprintf(stderr, "vla(bitvla): LM weights or int2 scale sidecars invalid; refusing the CUDA LM\n"); } - if (pack_ok && scales_ok) { + if (pack_ok && weights_ok) { __nv_bfloat16* onorm = upload_bf16_from_f32((const float*) m->lm_output_norm->data, m->lm_hidden, m->cuda_devptrs); bitvla_lm_cuda_set_output_norm(m->lm_cuda_ctx, onorm); @@ -838,6 +836,8 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, lm_ce = cudaMalloc(&m->d_action_hidden, (size_t) (m->num_actions_chunk*m->action_dim)*m->lm_hidden*sizeof(__nv_bfloat16)); if (lm_ce == cudaSuccess) lm_ce = cudaMalloc(&m->d_action_ids, (size_t) (m->num_actions_chunk*m->action_dim)*sizeof(int32_t)); + if (lm_ce == cudaSuccess) + lm_ce = cudaGetLastError(); // only enable the CUDA LM once every work buffer is really allocated. if (lm_ce != cudaSuccess) { std::fprintf(stderr, "vla(bitvla): CUDA LM buffer alloc failed (%s); using CPU LM\n", @@ -865,6 +865,7 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, (int) m->vit_inter, (int) m->n_patches, patch_flat, m->vit_ln_eps, mm_out); if (m->vit_cuda_ctx) { + cudaGetLastError(); bool vit_ok = true; for (int64_t L=0; Lvit_layers && vit_ok; ++L) { bitvla_vit_layer_cuda vl{}; @@ -913,6 +914,9 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, vl.fc2_ws = r.second; } vl.fc2_b = upload_bf16_from_f32((const float*) m->vit[L].bfc2->data, m->vit_hidden, m->cuda_devptrs); + } else if (ggml_nbytes(m->vit[L].Wfc2) != (size_t) (m->vit_hidden*m->vit_inter)*sizeof(float)) { + std::fprintf(stderr, "vla(bitvla): vit.blk.%lld.fc2.weight size mismatch\n", (long long) L); + weights_ok = false; } else { const float* W = (const float*) m->vit[L].Wfc2->data; std::vector tern; @@ -932,9 +936,9 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, vl.fc2_b = upload_bf16_from_f32((const float*) m->vit[L].bfc2->data, m->vit_hidden, m->cuda_devptrs); } bitvla_vit_cuda_set_layer(m->vit_cuda_ctx, (int) L, &vl); + vit_ok = weights_ok && cudaGetLastError() == cudaSuccess; } if (vit_ok) { - __nv_bfloat16* pe_w = upload_bf16_from_f32((const float*) m->vit_patch_w->data, m->vit_hidden*patch_flat, m->cuda_devptrs); __nv_bfloat16* pe_b = upload_bf16_from_f32((const float*) m->vit_patch_b->data, m->vit_hidden, m->cuda_devptrs); __nv_bfloat16* pos_e = upload_bf16_from_f32((const float*) m->vit_pos->data, m->n_patches*m->vit_hidden, m->cuda_devptrs); @@ -946,8 +950,11 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, __nv_bfloat16* mm_b2 = upload_bf16_from_f32((const float*) m->mm_l2_b->data, mm_out, m->cuda_devptrs); bitvla_vit_cuda_set_mmproj(m->vit_cuda_ctx, mm_W1, mm_b1, mm_W2, mm_b2); - cudaMalloc(&m->d_vit_patches, (size_t) m->n_patches*patch_flat * sizeof(__nv_bfloat16)); - cudaMalloc(&m->d_vit_img_embeds, (size_t) m->n_patches*mm_out * sizeof(__nv_bfloat16)); + vit_ok = cudaMalloc(&m->d_vit_patches, (size_t) m->n_patches*patch_flat * sizeof(__nv_bfloat16)) == cudaSuccess && + cudaMalloc(&m->d_vit_img_embeds, (size_t) m->n_patches*mm_out * sizeof(__nv_bfloat16)) == cudaSuccess && + cudaGetLastError() == cudaSuccess; + } + if (vit_ok) { m->cuda_vit_ready = true; const size_t vit_packed_bytes = (size_t) m->vit_layers*( 4*(size_t) m->vit_hidden*m->vit_hidden/4 + @@ -958,6 +965,8 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, } else { bitvla_vit_cuda_free(m->vit_cuda_ctx); m->vit_cuda_ctx = nullptr; + cudaFree(m->d_vit_patches); cudaFree(m->d_vit_img_embeds); + m->d_vit_patches = nullptr; m->d_vit_img_embeds = nullptr; std::printf("vla(bitvla): CUDA ViT packing failed; falling back to CPU vision\n"); } } else { @@ -972,7 +981,7 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, std::fprintf(stderr, "vla(bitvla): bitvla_lm_cuda_init failed; falling back to CPU LM\n"); } } else { - std::printf("vla(bitvla): no CUDA device - using CPU LM forward\n"); + std::printf("vla(bitvla): CUDA device %d unavailable (%d visible) - using CPU LM forward\n", m->cuda_dev, dev_count); } } @@ -1038,6 +1047,7 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, } std::memcpy(copy, t->data, nb); t->data = copy; + t->buffer = nullptr; m->cpu_kept_ptrs.push_back(copy); bytes_kept += nb; } @@ -1068,6 +1078,15 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { const auto t_start = clk::now(); stats = Stats{}; const bool timing_phase = (in.timing_detail == TimingDetail::PHASE); +#ifdef VLA_BITVLA_CUDA_KERNELS + if (cuda_lm_ready || cuda_vit_ready) { + if (cudaSetDevice(cuda_dev) != cudaSuccess) { + std::fprintf(stderr, "vla(bitvla): cudaSetDevice(%d) failed\n", cuda_dev); + return {}; + } + cudaGetLastError(); + } +#endif const char* _dump_dir = std::getenv("VLA_BITVLA_DUMP_DIR"); auto _dump_bin = [&](const char* name, const float* data, size_t nelem) { @@ -1117,11 +1136,8 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { const auto t_v0 = clk::now(); for (int64_t v=0; v BitvlaModelArch::predict(const Inputs& in) { #ifdef VLA_BITVLA_CUDA_KERNELS if (cuda_vit_ready) { - std::vector patches_bf16((size_t) N * patch_flat); - for (size_t i=0; i img_bf16((size_t) N * hidden_l); - cudaMemcpy(img_bf16.data(), d_vit_img_embeds, img_bf16.size()*sizeof(uint16_t), cudaMemcpyDeviceToHost); - float* dst = img_embeds_host.data()+(size_t) v * N * hidden_l; - for (size_t i=0; i patches_bf16((size_t) N * patch_flat); + ggml_fp32_to_bf16_row(patches.data(), patches_bf16.data(), (int64_t) patches_bf16.size()); + std::vector img_bf16((size_t) N * hidden_l); + if (cudaMemcpy(d_vit_patches, patches_bf16.data(), patches_bf16.size()*sizeof(ggml_bf16_t), cudaMemcpyHostToDevice) != cudaSuccess || + bitvla_vit_cuda_forward(vit_cuda_ctx, d_vit_patches, d_vit_img_embeds, 0) != 0 || + cudaMemcpy(img_bf16.data(), d_vit_img_embeds, img_bf16.size()*sizeof(ggml_bf16_t), cudaMemcpyDeviceToHost) != cudaSuccess) { + std::fprintf(stderr, "vla(bitvla): CUDA ViT forward failed (view %lld)\n", (long long) v); + return {}; } + ggml_bf16_to_fp32_row(img_bf16.data(), img_embeds_host.data()+(size_t) v * N * hidden_l, (int64_t) img_bf16.size()); } else #endif { @@ -1271,6 +1283,13 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { std::fprintf(stderr, "vla(bitvla): seq=%lld > lm_max_pos=%lld\n", (long long) seq, (long long) lm_max_pos); return {}; } +#ifdef VLA_BITVLA_CUDA_KERNELS + if (cuda_lm_ready && seq > cuda_max_seq && (packed_int2 || !weight_buf)) { + std::fprintf(stderr, "vla(bitvla): seq=%lld > CUDA LM max_seq=%d and no CPU LM weights to fall back on\n", + (long long) seq, cuda_max_seq); + return {}; + } +#endif // Both LM paths index the action slots as seq-2-n_action+i and neither // ggml_get_rows nor the CUDA gather bound-checks, so a short sequence would // read out of bounds and come back as plausible hidden states. @@ -1284,13 +1303,22 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { if (full_prefix) { - std::vector ids(in.lang_tokens, in.lang_tokens+n_lang_in); - for (int32_t id : ids) { + std::vector ids; + std::vector pos; + for (int64_t i=0; i= vocab_size) { std::fprintf(stderr, "vla(bitvla): prompt token %d out of vocab\n", id); return {}; } + ids.push_back(id); + pos.push_back(i); } - if (!emb_reader.fetch_rows_f32("token_embd.weight", ids, inputs_embeds.data(), hidden_l)) return {}; + std::vector rows(ids.size()*(size_t) hidden_l); + if (!emb_reader.fetch_rows_f32("token_embd.weight", ids, rows.data(), hidden_l)) return {}; + for (size_t k=0; k BitvlaModelArch::predict(const Inputs& in) { #ifdef VLA_BITVLA_CUDA_KERNELS if (cuda_lm_ready && seq <= cuda_max_seq) { - std::vector in_bf16((size_t) seq * hidden_l); - for (size_t i=0; i in_bf16((size_t) seq * hidden_l); + ggml_fp32_to_bf16_row(inputs_embeds.data(), in_bf16.data(), (int64_t) in_bf16.size()); std::vector aids(n_action); for (int64_t i=0; i out_bf16((size_t) n_action * hidden_l); + if (cudaMemcpy(d_inputs_embeds, in_bf16.data(), in_bf16.size()*sizeof(ggml_bf16_t), cudaMemcpyHostToDevice) != cudaSuccess || + bitvla_lm_cuda_forward(lm_cuda_ctx, d_inputs_embeds, d_last_hidden, (int) seq, 0) != 0 || + cudaMemcpy(d_action_ids, aids.data(), n_action * sizeof(int32_t), cudaMemcpyHostToDevice) != cudaSuccess) { + std::fprintf(stderr, "vla(bitvla): CUDA LM forward failed\n"); + return {}; + } bitvla_gather_rows_bf16(d_last_hidden, d_action_hidden, d_action_ids, (int) n_action, (int) hidden_l, 0); - - std::vector out_bf16((size_t) n_action * hidden_l); - cudaMemcpy(out_bf16.data(), d_action_hidden, out_bf16.size()*sizeof(uint16_t), cudaMemcpyDeviceToHost); - for (size_t i=0; i BitvlaModelArch::predict(const Inputs& in) { for (int64_t L=0; L BitvlaModelArch::predict(const Inputs& in) { ggml_tensor * x = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, in_dim, chunk); ggml_set_name(x, "x"); - ggml_tensor * ln1 = layernorm(ctx, x, ah_ln1_w, ah_ln1_b, ah_ln_eps); + ggml_tensor * ln1 = layer_norm(ctx, x, ah_ln1_w, ah_ln1_b, ah_ln_eps); ggml_tensor * fc1 = ggml_add(ctx, ggml_mul_mat(ctx, ah_fc1_w, ln1), ah_fc1_b); ggml_tensor * h = ggml_relu(ctx, fc1); - ggml_tensor * b0_ln = layernorm(ctx, h, ah_b0_ln_w, ah_b0_ln_b, ah_ln_eps); + ggml_tensor * b0_ln = layer_norm(ctx, h, ah_b0_ln_w, ah_b0_ln_b, ah_ln_eps); ggml_tensor * b0 = ggml_add(ctx, ggml_mul_mat(ctx, ah_b0_w, b0_ln), ah_b0_b); h = ggml_add(ctx, h, ggml_relu(ctx, b0)); - ggml_tensor * b1_ln = layernorm(ctx, h, ah_b1_ln_w, ah_b1_ln_b, ah_ln_eps); + ggml_tensor * b1_ln = layer_norm(ctx, h, ah_b1_ln_w, ah_b1_ln_b, ah_ln_eps); ggml_tensor * b1 = ggml_add(ctx, ggml_mul_mat(ctx, ah_b1_w, b1_ln), ah_b1_b); h = ggml_add(ctx, h, ggml_relu(ctx, b1)); - ggml_tensor * ln2 = layernorm(ctx, h, ah_ln2_w, ah_ln2_b, ah_ln_eps); + ggml_tensor * ln2 = layer_norm(ctx, h, ah_ln2_w, ah_ln2_b, ah_ln_eps); ggml_tensor * y = ggml_add(ctx, ggml_mul_mat(ctx, ah_fc2_w, ln2), ah_fc2_b); ggml_set_name(y, "y"); diff --git a/src/models/evo1.cpp b/src/models/evo1.cpp index bdeb9a9..4276c3c 100644 --- a/src/models/evo1.cpp +++ b/src/models/evo1.cpp @@ -26,7 +26,7 @@ #include "scratch_ctx.h" #include "act_dtype.h" #include "cuda/vla_cuda_ops.h" -#include "env_flag.h" +#include "modules/preprocess.h" #include #include @@ -60,16 +60,12 @@ struct Evo1ModelArch : public ModelArchBase { Evo1ModelArch() : ModelArchBase(Arch::EVO1) {} ~Evo1ModelArch() override; - std::string gguf_path; // Opened once at load: reopening per predict re-parses the whole GGUF header. gguf_reader io{"evo1"}; ggml_backend_t backend = nullptr; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; - // scratch_ctx fixes its arena on first use, and the vision graph holds every - // view at once, so a later call with more views needs a bigger one. - size_t vision_arena = 0; struct MainKey { int64_t seq=-1, nsteps=-1; @@ -80,6 +76,7 @@ struct Evo1ModelArch : public ModelArchBase { struct MainIO { ggml_tensor *t_embeds=nullptr,*t_pos=nullptr,*t_lmmask=nullptr,*t_qmask=nullptr; ggml_tensor *t_state=nullptr,*t_x=nullptr,*t_amask=nullptr,*x_action=nullptr; + std::vector lm_ok; }; graph_cache main_graph; ggml_backend_buffer_t weight_buf = nullptr; @@ -90,12 +87,12 @@ struct Evo1ModelArch : public ModelArchBase { ggml_type act_type = GGML_TYPE_F32; int64_t lm_hidden=896, lm_layers=14, n_q=14, n_kv=2, lm_head_dim=64, lm_inter=4864; - int64_t embed_dim=896, dit_layers=8, dit_heads=8, mlp_head_hidden=1024; + int64_t embed_dim=896, dit_layers=8, dit_heads=8; int64_t horizon=50, per_a=24, action_dim=1200, num_steps=32; - int64_t num_image_token=256, n_images=3, vocab=151674, max_text_length=1024; + int64_t num_image_token=256, n_images=3, max_text_length=1024; int64_t img_ctx_id=151667, img_start_id=151665, img_end_id=151666, pad_token_id=151643; int64_t real_state_dim=8, real_action_dim=7; - int64_t vit_hidden=1024, vit_layers=24, vit_heads=16, vit_inter=4096, image_size=448, patch_size=14; + int64_t vit_hidden=1024, vit_layers=24, vit_heads=16, image_size=448, patch_size=14; float lm_rms_eps=1e-6f, proj_ln_eps=1e-5f, norm_eps_denom=1e-8f, vit_ln_eps=1e-6f; float lm_rope_base=1000000.0f; bool have_vision = false; @@ -120,7 +117,7 @@ struct Evo1ModelArch : public ModelArchBase { namespace { -// BF16 activation path (VLA_EVO1_BF16_ACT=1); see models/act_dtype.h for what +// BF16 activation path (--act-dtype bf16); see act_dtype.h for what // mm_act/as_type do and why. The split here: GEMMs, bias adds, residuals, // norms and activations in BF16; attention scores, softmax and RoPE in F32; // and the flow-matching Euler integrator in F32 so 32 steps of dt = 1/32 do @@ -146,7 +143,7 @@ ggml_tensor * build_qwen2_layer(ggml_context * C, const Evo1ModelArch & m, const ggml_tensor * Q = ggml_cont(C, ggml_permute(C, q_rope, 0, 2, 1, 3)); ggml_tensor * K = ggml_cont(C, ggml_permute(C, k_rope, 0, 2, 1, 3)); ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, vp, hd, n_kv, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_prec_set_acc(kq, GGML_PREC_F32); ggml_tensor * aw = ggml_soft_max_ext(C, kq, mask, scale, 0.0f); ggml_tensor * kqv = ggml_mul_mat(C, V, aw); ggml_tensor * att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, kqv, 0, 2, 1, 3)), hq, seq); @@ -160,28 +157,6 @@ ggml_tensor * build_qwen2_layer(ggml_context * C, const Evo1ModelArch & m, const return ggml_add(C, h_attn, mm_act(C, w.Wdown, ggml_mul(C, gate, up), at)); } -bool preprocess_image_chw(const ImageView & v, int64_t side, std::vector & out) { - static const float MEAN[3] = {0.485f, 0.456f, 0.406f}; - static const float STD [3] = {0.229f, 0.224f, 0.225f}; - if (v.w != (int) side || v.h != (int) side || !v.data) { - std::fprintf(stderr, "vla(evo1): image view is %dx%d, expected %lldx%lld\n", v.w, v.h, (long long) side, (long long) side); - return false; - } - out.assign((size_t) 3*side * side, 0.0f); - for (int64_t h=0; h // (see modeling_intern_vit.py), so this path is closer to the upstream model // than the explicit one, not a divergence from it. // -// OPT-IN (VLA_EVO1_FA=1), not default. It cuts the vision stage from ~132 ms to +// OPT-IN (--flash-attn), not default. It cuts the vision stage from ~132 ms to // ~82 ms, but ggml's CUDA flash attention computes K/V at F16 — fattn.cu accepts // an F32 K/V only by reinterpreting it as F16, so there is no full-precision FA // path on this backend. Over 24 ViT layers that moved actions by ~1e-2 and @@ -216,7 +191,7 @@ ggml_tensor * evo1_flash_attn(ggml_context * C, ggml_tensor * q, ggml_tensor * k ggml_tensor * o = ggml_flash_attn_ext(C, q, k, v, nullptr, scale, 0.0f, 0.0f); // F32 accumulation keeps the softmax/AV reduction at the precision the // explicit path used, so switching kernels does not move the actions. - ggml_flash_attn_ext_set_prec(o, GGML_PREC_F32); + ggml_prec_set_acc(o, GGML_PREC_F32); return ggml_reshape_2d(C, o, hidden, N); } @@ -240,7 +215,7 @@ ggml_tensor * build_internvit_layer(ggml_context * C, const Evo1ModelArch & m, c att = evo1_flash_attn(C, Q, K, V, scale, H, N); } else { ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, hd, n_heads, N), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_prec_set_acc(kq, GGML_PREC_F32); ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); ggml_tensor * kqv = ggml_mul_mat(C, V, aw); att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, kqv, 0, 2, 1, 3)), H, N); @@ -282,24 +257,24 @@ ggml_tensor * build_internvit_view(ggml_context * C, const Evo1ModelArch & m, gg } ggml_tensor * inproj_split_w(ggml_context * C, ggml_tensor * Win, int64_t E, int64_t k) { - return ggml_cont(C, ggml_view_2d(C, Win, E, E, Win->nb[1], (size_t) k * E * E * ggml_element_size(Win))); + return ggml_view_2d(C, Win, Win->ne[0], E, Win->nb[1], (size_t) k * E * Win->nb[1]); } ggml_tensor * inproj_split_b(ggml_context * C, ggml_tensor * bin, int64_t E, int64_t k) { - return ggml_cont(C, ggml_view_1d(C, bin, E, (size_t) k * E * ggml_element_size(bin))); + return ggml_view_1d(C, bin, E, (size_t) k * E * bin->nb[0]); } -bool load_config(const gguf_reader & g, Evo1ModelArch & m, Config & cfg) { +bool load_config(const gguf_reader & g, const Options & opts, Evo1ModelArch & m, Config & cfg) { auto u = [&](const char * k, int64_t & dst) { if (g.has((std::string("evo1.")+k).c_str())) dst = g.u32((std::string("evo1.")+k).c_str()); }; u("lm_hidden", m.lm_hidden); u("lm_layers_used", m.lm_layers); u("lm_q_heads", m.n_q); u("lm_kv_heads", m.n_kv); u("lm_head_dim", m.lm_head_dim); u("lm_inter", m.lm_inter); u("embed_dim", m.embed_dim); u("dit_layers", m.dit_layers); - u("dit_heads", m.dit_heads); u("mlp_head_hidden", m.mlp_head_hidden); u("horizon", m.horizon); u("per_action_dim", m.per_a); + u("dit_heads", m.dit_heads); u("horizon", m.horizon); u("per_action_dim", m.per_a); u("action_dim", m.action_dim); u("num_inference_timesteps", m.num_steps); u("num_image_token", m.num_image_token); - u("n_images", m.n_images); u("vocab_size", m.vocab); u("max_text_length", m.max_text_length); + u("n_images", m.n_images); u("max_text_length", m.max_text_length); u("img_context_token_id", m.img_ctx_id); u("img_start_token_id", m.img_start_id); u("img_end_token_id", m.img_end_id); u("pad_token_id", m.pad_token_id); u("real_state_dim", m.real_state_dim); u("real_action_dim", m.real_action_dim); u("vit_hidden", m.vit_hidden); u("vit_layers", m.vit_layers); u("vit_heads", m.vit_heads); - u("vit_inter", m.vit_inter); u("image_size", m.image_size); u("patch_size", m.patch_size); + u("image_size", m.image_size); u("patch_size", m.patch_size); if (g.has("evo1.lm_rms_eps")) m.lm_rms_eps = g.f32("evo1.lm_rms_eps"); if (g.has("evo1.proj_ln_eps")) @@ -321,6 +296,9 @@ bool load_config(const gguf_reader & g, Evo1ModelArch & m, Config & cfg) { (long long) m.action_dim, (long long) m.horizon, (long long) m.per_a); return false; } + if (!resolve_num_steps("evo1", opts, m.num_steps)) + return false; + cfg = Config{}; cfg.n_suffix = m.horizon; cfg.n_full = m.horizon; @@ -367,7 +345,6 @@ std::unique_ptr evo1_create(const std::string& mmproj_path, mmproj_path.c_str()); auto m = std::make_unique(); - m->gguf_path = ckpt_path; m->matmul_type = opts.weight_dtype.value_or(vla::default_weight_dtype(GGML_TYPE_BF16)); if (!m->io.open(ckpt_path)) @@ -376,7 +353,7 @@ std::unique_ptr evo1_create(const std::string& mmproj_path, if (!g.has("evo1.architecture")) { std::fprintf(stderr, "vla(evo1): %s is not an evo1 GGUF (no evo1.architecture KV)\n", ckpt_path.c_str()); return nullptr; } - if (!load_config(g, *m, m->cfg)) + if (!load_config(g, opts, *m, m->cfg)) return nullptr; std::printf("vla(evo1): lm=%lldd×%lldL (%lldq/%lldkv×%lld) inter=%lld embed=%lld dit=%lldL×%lldh " "horizon=%lld per_a=%lld N_steps=%lld resident matmul=%s\n", @@ -397,9 +374,9 @@ std::unique_ptr evo1_create(const std::string& mmproj_path, if (b.is_cuda && m->matmul_type == GGML_TYPE_BF16) { m->act_type = GGML_TYPE_BF16; cuda_register_bf16_ops(); // installs the in-tree BF16 CUDA kernels - std::printf("vla(evo1): activations = BF16 (VLA_EVO1_BF16_ACT)\n"); + std::printf("vla(evo1): activations = BF16\n"); } else { - std::fprintf(stderr, "vla(evo1): VLA_EVO1_BF16_ACT ignored - needs CUDA and BF16 weights\n"); + std::fprintf(stderr, "vla(evo1): --act-dtype bf16 ignored - needs CUDA and BF16 weights\n"); } } } @@ -448,6 +425,14 @@ std::unique_ptr evo1_create(const std::string& mmproj_path, w.f1w = mk_mm(N("ff1.weight")); w.f1b = mk_f32(N("ff1.bias")); w.f2w = mk_mm(N("ff2.weight")); w.f2b = mk_f32(N("ff2.bias")); ok &= w.n1w && w.n1b && w.n2w && w.n2b && w.Win && w.bin && w.Wo && w.bo && w.f1w && w.f1b && w.f2w && w.f2b; +#ifdef GGML_USE_OPENCL + if (ok && ggml_is_quantized(w.Win->type) && + std::strcmp(ggml_backend_reg_name(ggml_backend_dev_backend_reg(ggml_backend_get_device(m->backend))), "OpenCL") == 0) { + std::fprintf(stderr, "vla(evo1): %s is %s; OpenCL cannot split a quantized attn_in, requantize with attn_in kept float\n", + N("attn_in.weight"), ggml_type_name(w.Win->type)); + return nullptr; + } +#endif } m->norm_out_w = mk_f32("aex.norm_out.weight"); m->norm_out_b = mk_f32("aex.norm_out.bias"); m->seq_pool_w = mk_mm("aex.seq_pool.weight"); m->seq_pool_b = mk_f32("aex.seq_pool.bias"); @@ -540,12 +525,7 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { // // The branches are independent, so the arithmetic per view is unchanged // - only the submission pattern differs. - const size_t want_arena = (size_t) 32*1024*1024*(size_t) std::max(n_views, 1); - if (want_arena > vision_arena) { - vision_scratch.release(); - vision_arena = want_arena; - } - ggml_context * VC = vision_scratch.reset(vision_arena); + ggml_context * VC = vision_scratch.reset((size_t) 32*1024*1024*(size_t) std::max(n_views, 1)); if (!VC) { std::fprintf(stderr, "vla(evo1): ggml_init(vision ctx) failed\n"); return {}; } std::vector t_px((size_t) n_views), t_ie((size_t) n_views); for (int64_t v=0; v Evo1ModelArch::predict(const Inputs& in) { return {}; } img_emb_host.assign((size_t) n_views * num_image_token * lm_hidden, 0.0f); + static const float MEAN[3] = {0.485f, 0.456f, 0.406f}, STD[3] = {0.229f, 0.224f, 0.225f}; std::vector chw; const auto tv0 = std::chrono::steady_clock::now(); for (int64_t v=0; v Evo1ModelArch::predict(const Inputs& in) { for (int j=0; j Evo1ModelArch::predict(const Inputs& in) { const int64_t SEQ = max_text_length; std::vector inputs_embeds((size_t) SEQ * lm_hidden); - if (!io.fetch_rows_f32("token_embd.weight", input_ids, inputs_embeds.data(), lm_hidden)) return {}; + { + std::vector uniq(input_ids); + std::sort(uniq.begin(), uniq.end()); + uniq.erase(std::unique(uniq.begin(), uniq.end()), uniq.end()); + std::vector rows(uniq.size() * lm_hidden); + if (!io.fetch_rows_f32("token_embd.weight", uniq, rows.data(), lm_hidden)) return {}; + for (int64_t p=0; p Evo1ModelArch::predict(const Inputs& in) { } std::vector state_norm(per_a, 0.0f); - for (int64_t i=0; i Evo1ModelArch::predict(const Inputs& in) { // LM + DiT graph depends only on the padded length and step count. const MainKey mkey{ SEQ, num_steps }; - const bool built = main_graph.ensure(backend, mkey, (size_t) 96*1024*1024, + const size_t main_nodes = 32768 + (size_t) num_steps*64*(dit_layers+1); + const bool built = main_graph.ensure(backend, mkey, main_nodes*ggml_tensor_overhead() + ggml_graph_overhead_custom(main_nodes, false), [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { const int64_t E = embed_dim, hd_dit = E/dit_heads; const float scale_dit = 1.0f/std::sqrt((float) hd_dit); @@ -695,6 +684,7 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { ggml_tensor * t_state = ggml_new_tensor_1d(C, GGML_TYPE_F32, per_a); ggml_set_input(t_state); ggml_tensor * t_x = ggml_new_tensor_1d(C, GGML_TYPE_F32, action_dim); ggml_set_input(t_x); ggml_tensor * t_amask = ggml_new_tensor_1d(C, GGML_TYPE_F32, per_a); ggml_set_input(t_amask); + ggml_set_output(t_pos); ggml_set_output(t_lmmask); ggml_set_output(t_qmask); ggml_set_output(t_amask); const ggml_type at = act_type; @@ -735,7 +725,7 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { ggml_tensor * x_q = ggml_add(C, ggml_mul(C, ggml_norm(C, x, proj_ln_eps), w.n1w), w.n1b); ggml_tensor * qp = as_type(C, ggml_add(C, mm_act(C, c.Wq, x_q, at), c.bq), GGML_TYPE_F32); ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, qp, hd_dit, dit_heads, horizon), 0, 2, 1, 3)); - ggml_tensor * kq = ggml_mul_mat(C, c.K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * kq = ggml_mul_mat(C, c.K, Q); ggml_prec_set_acc(kq, GGML_PREC_F32); // The Evo-1 reference cross-attends over the full padded context // (no key mask), so the action queries see every LM position. ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale_dit, 0.0f); @@ -773,7 +763,7 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { gio.t_embeds=t_embeds; gio.t_pos=t_pos; gio.t_lmmask=t_lmmask; gio.t_qmask=t_qmask; gio.t_state=t_state; gio.t_x=t_x; gio.t_amask=t_amask; gio.x_action=x_action; - ggml_cgraph * gf = ggml_new_graph_custom(C, 32768, false); + ggml_cgraph * gf = ggml_new_graph_custom(C, main_nodes, false); ggml_build_forward_expand(gf, x_action); return gf; }); @@ -786,21 +776,22 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { ggml_tensor * t_amask = gio.t_amask, * x_action = gio.x_action; ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); - { + if (gio.lm_ok != attn_ok) { std::vector pp(SEQ); for (int64_t i=0; i mk((size_t) SEQ * SEQ); const float NEG = -std::numeric_limits::infinity(); + for (int64_t q=0; q am(per_a, 0.0f); for (int64_t i=0; i qm(SEQ, 0.0f); for (int64_t p=0; p mk((size_t) SEQ * SEQ); const float NEG = -std::numeric_limits::infinity(); - for (int64_t q=0; q am(per_a, 0.0f); for (int64_t i=0; i qm(SEQ, 0.0f); for (int64_t p=0; p #include #include -#include -#include #include -#include #include #include @@ -56,7 +52,9 @@ struct Gr00tN1d5ModelArch : public ModelArchBase { ggml_context * ctx_weights = nullptr; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; - scratch_ctx vision_scratch; + FlowTimes times; + std::vector c_mask; + int64_t c_mask_seq = -1; struct MainKey { int64_t seq=-1, nsteps=-1; @@ -66,9 +64,12 @@ struct Gr00tN1d5ModelArch : public ModelArchBase { }; struct MainIO { ggml_tensor *t_embeds=nullptr,*t_pos=nullptr,*t_lmmask=nullptr,*t_state=nullptr,*t_x0=nullptr,*actions=nullptr; - std::vector t_tau, t_tproj; }; graph_cache main_graph; + struct VisIO { + ggml_tensor *t_px=nullptr,*vit_emb=nullptr; + }; + graph_cache vis_graph; SigLipTower vit; Qwen3LM lm; @@ -82,7 +83,7 @@ struct Gr00tN1d5ModelArch : public ModelArchBase { int64_t vit_layers=27, vit_inter=4304, image_size=224, patch_size=14, n_img_tokens=256; int64_t lm_inter=6144, vocab=151680, image_token_index=151669; int64_t bb_embed_dim=2048, in_embed_dim=1536, dit_interleave=1, vlsa_layers=4; - int64_t num_future=32, action_horizon=16, action_dim=32, max_state_dim=64; + int64_t action_horizon=16, action_dim=32, max_state_dim=64; int64_t num_steps=4, num_buckets=1000, max_embodiments=32, max_seq_len=1024; float vlln_eps=1e-5f; @@ -91,10 +92,10 @@ struct Gr00tN1d5ModelArch : public ModelArchBase { namespace { -bool load_config(const gguf_reader & g, Gr00tN1d5ModelArch & m, Config & cfg) { +bool load_config(const gguf_reader & g, const Options & opts, Gr00tN1d5ModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; auto F = [&](const char * k, float & dst) { if (g.has(k)) dst = g.f32(k); }; - auto fk = [&](const char * s) { static char b[64]; std::snprintf(b, sizeof(b), "gr00t_n1_5.%s", s); return b; }; + auto fk = [&](const char * s) { thread_local char b[64]; std::snprintf(b, sizeof(b), "gr00t_n1_5.%s", s); return b; }; U(fk("vit_hidden" ), m.vit.enc.cfg.hidden); U(fk("vit_layers" ), m.vit_layers); @@ -121,7 +122,6 @@ bool load_config(const gguf_reader & g, Gr00tN1d5ModelArch & m, Config & cfg) { U(fk("vlsa_layers" ), m.vlsa_layers); U(fk("vlsa_heads" ), m.vlsa.cfg.heads); U(fk("vlsa_head_dim" ), m.vlsa.cfg.head_dim); - U(fk("num_target_vision_tokens"), m.num_future); U(fk("action_horizon" ), m.action_horizon); U(fk("action_dim" ), m.action_dim); U(fk("max_state_dim" ), m.max_state_dim); @@ -139,31 +139,25 @@ bool load_config(const gguf_reader & g, Gr00tN1d5ModelArch & m, Config & cfg) { if (g.has(fk("lm_rope_theta"))) m.lm.cfg.rope.freq_base = (float) g.f64(fk("lm_rope_theta")); + if (m.vit.enc.cfg.heads <= 0 || m.patch_size <= 0 || + (m.image_size/m.patch_size)*(m.image_size/m.patch_size) != m.n_img_tokens) { + std::fprintf(stderr, "vla(gr00tn1d5): vit_heads %lld, image %lld / patch %lld do not give n_img_tokens %lld\n", + (long long) m.vit.enc.cfg.heads, (long long) m.image_size, (long long) m.patch_size, + (long long) m.n_img_tokens); + return false; + } m.vit.enc.cfg.head_dim = m.vit.enc.cfg.hidden/m.vit.enc.cfg.heads; m.vlsa.cfg.hidden = m.bb_embed_dim; m.vlsa.cfg.ln_eps = m.dit.cfg.ln_eps; m.lm.cfg.rope.n_dims = (int) m.lm.cfg.head_dim; + m.vit.enc.cfg.flash_attn = m.vlsa.cfg.flash_attn = m.lm.cfg.flash_attn = flash_attn_enabled(); m.aex.embodiment_id = 24; - if (const char * e = std::getenv("VLA_GR00T_EMBODIMENT")) { - char * end = nullptr; - const long v = std::strtol(e, &end, 10); - if (end && *end == '\0') { - m.aex.embodiment_id = (int64_t) v; - } else { - const std::string js = g.str(fk("embodiment_tag_mapping")); - const std::string key = std::string("\"")+e+"\":"; - const size_t p = js.find(key); - if (p != std::string::npos) - m.aex.embodiment_id = std::strtol(js.c_str()+p+key.size(), nullptr, 10); - else std::fprintf(stderr, "vla(gr00tn1d5): embodiment tag '%s' not in embodiment_tag_mapping; using id %lld\n", e, (long long) m.aex.embodiment_id); - } - } - if (m.aex.embodiment_id < 0 || m.aex.embodiment_id >= m.max_embodiments) { - std::fprintf(stderr, "vla(gr00tn1d5): embodiment id %lld out of range [0,%lld)\n", - (long long) m.aex.embodiment_id, (long long) m.max_embodiments); + if (!resolve_embodiment("gr00tn1d5", g.str(fk("embodiment_tag_mapping")), nullptr, m.max_embodiments, m.aex.embodiment_id)) + return false; + + if (!resolve_num_steps("gr00tn1d5", opts, m.num_steps)) return false; - } cfg = Config{}; cfg.n_img = m.n_img_tokens; @@ -220,7 +214,7 @@ std::unique_ptr gr00t_n1_5_create(const std::string& mmproj_path, std::fprintf(stderr, "vla(gr00tn1d5): %s is not a gr00t_n1_5 GGUF\n", ckpt_path.c_str()); return nullptr; } - if (!load_config(g, *m, m->cfg)) + if (!load_config(g, opts, *m, m->cfg)) return nullptr; std::printf("vla(gr00tn1d5): vit=%lldd×%lldL×%lldh n_img_tok=%lld lm=Qwen3 %lldd×%lldL (%lldq/%lldkv×%lld) " @@ -256,12 +250,15 @@ std::unique_ptr gr00t_n1_5_create(const std::string& mmproj_path, m->vlln_b = L.f32("aex.vlln.bias"); m->vlsa.declare(L, "aex.vlsa", m->vlsa_layers, EncNames{"norm1", "norm3", "ff0", "ff2"}); - m->aex.declare(L, "aex"); + if (!m->aex.declare(L, g, m->backend, "aex")) + return nullptr; m->future_tokens = L.f32("aex.future_tokens"); m->dit.declare(L, "aex.dit"); if (!L.upload(m->backend, &m->weight_buf)) return nullptr; + if (!m->times.build("gr00tn1d5", m->backend, m->dit, m->num_steps, m->num_buckets, m->in_embed_dim, m->action_horizon)) + return nullptr; std::printf("vla(gr00tn1d5): weights resident in %.2f GiB (%s) - incl. SigLIP vision tower; embodiment id %lld\n", ggml_backend_buffer_get_size(m->weight_buf)/(1024.0*1024.0*1024.0), @@ -275,10 +272,8 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { const int64_t H = lm.cfg.hidden; const int64_t K = n_img_tokens; - const int64_t E = in_embed_dim; const int64_t AD = action_dim; const int64_t AH = action_horizon; - const int64_t Nsa = 1+num_future+AH; int64_t n_views = 0; std::vector img_emb_host; @@ -291,20 +286,23 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { n_views = in.n_images; img_emb_host.assign((size_t) n_views*K*H, 0.0f); - ggml_context * VC = vision_scratch.reset((size_t) 64*1024*1024); - if (!VC) { std::fprintf(stderr, "vla(gr00tn1d5): ggml_init(vision ctx) failed\n"); return {}; } - - const int64_t grid = image_size/patch_size; - ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, image_size, image_size, 3); - ggml_set_input(t_px); + const bool vbuilt = vis_graph.ensure(backend, 0, (size_t) 64*1024*1024, [&](ggml_context * VC, VisIO & vio) -> ggml_cgraph * { + const int64_t grid = image_size/patch_size; + ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, image_size, image_size, 3); + ggml_set_input(t_px); - ggml_tensor * h = vit.build(VC, vit.embed_conv(VC, t_px, patch_size, grid), K); - ggml_tensor * vit_emb = linear(VC, mm_W, mm_b, h); - ggml_set_output(vit_emb); + ggml_tensor * h = vit.build(VC, vit.embed_conv(VC, t_px, patch_size, grid), K); + ggml_tensor * vit_emb = linear(VC, mm_W, mm_b, h); + ggml_set_output(vit_emb); + vio.t_px = t_px; vio.vit_emb = vit_emb; - ggml_cgraph * vg = ggml_new_graph_custom(VC, 8192, false); - ggml_build_forward_expand(vg, vit_emb); - if (!vision_scratch.alloc(backend, vg)) { std::fprintf(stderr, "vla(gr00tn1d5): vision gallocr alloc failed\n"); return {}; } + ggml_cgraph * vg = ggml_new_graph_custom(VC, 8192, false); + ggml_build_forward_expand(vg, vit_emb); + return vg; + }); + if (!vbuilt) { std::fprintf(stderr, "vla(gr00tn1d5): vision graph build failed\n"); return {}; } + ggml_tensor * t_px = vis_graph.io().t_px, * vit_emb = vis_graph.io().vit_emb; + ggml_cgraph * vg = vis_graph.graph(); const auto tv0 = std::chrono::steady_clock::now(); std::vector chw; @@ -331,13 +329,14 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { const int64_t SEQ = prompt.len(); std::vector inputs_embeds; - if (!fetch_embeds("gr00tn1d5", io, prompt, img_emb_ptr, H, inputs_embeds)) return {}; + if (!fetch_embeds(io, prompt, img_emb_ptr, H, inputs_embeds)) return {}; std::vector x_init; init_noise(in, (size_t) AH*AD, x_init); const MainKey mkey{ SEQ, num_steps }; - const bool built = main_graph.ensure(backend, mkey, (size_t) 128*1024*1024, + const size_t main_nodes = 32768 + (size_t) num_steps*64*(dit.cfg.layers+1); + const bool built = main_graph.ensure(backend, mkey, main_nodes*ggml_tensor_overhead() + ggml_graph_overhead_custom(main_nodes, false), [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { ggml_tensor * t_embeds = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, SEQ); ggml_set_input(t_embeds); ggml_tensor * t_pos = ggml_new_tensor_1d(C, GGML_TYPE_I32, SEQ); ggml_set_input(t_pos); @@ -345,48 +344,19 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { ggml_tensor * t_state = ggml_new_tensor_2d(C, GGML_TYPE_F32, max_state_dim, 1); ggml_set_input(t_state); ggml_tensor * t_x0 = ggml_new_tensor_2d(C, GGML_TYPE_F32, AD, AH); ggml_set_input(t_x0); - std::vector t_tau(num_steps), t_tproj(num_steps); - for (int64_t s=0; s Kc(dit.cfg.layers, nullptr), Vc(dit.cfg.layers, nullptr); - for (int64_t i=0; inb[1], (size_t)(Nsa-AH)*pred->nb[1])); - actions = ggml_add(C, actions, ggml_scale(C, vel, dt)); - } + ggml_tensor * actions = aex.denoise(C, dit, times, dit_interleave != 0, 1, t_state, future_tokens, vl_embs, vl_embs, + t_x0); ggml_set_name(actions, "action_pred"); ggml_set_output(actions); gio.t_embeds=t_embeds; gio.t_pos=t_pos; gio.t_lmmask=t_lmmask; gio.t_state=t_state; - gio.t_x0=t_x0; gio.t_tau=t_tau; gio.t_tproj=t_tproj; gio.actions=actions; + gio.t_x0=t_x0; gio.actions=actions; - ggml_cgraph * gf = ggml_new_graph_custom(C, 32768, false); + ggml_cgraph * gf = ggml_new_graph_custom(C, main_nodes, false); ggml_build_forward_expand(gf, actions); return gf; }); @@ -401,9 +371,11 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { pp[i] = (int32_t) i; ggml_backend_tensor_set(gio.t_pos, pp.data(), 0, ggml_nbytes(gio.t_pos)); - std::vector mask; - build_causal_mask(SEQ, mask); - ggml_backend_tensor_set(gio.t_lmmask, mask.data(), 0, ggml_nbytes(gio.t_lmmask)); + if (c_mask_seq != SEQ) { + build_causal_mask(SEQ, c_mask); + c_mask_seq = SEQ; + } + ggml_backend_tensor_set(gio.t_lmmask, c_mask.data(), 0, ggml_nbytes(gio.t_lmmask)); std::vector st(max_state_dim, 0.0f); for (int64_t i=0; i Gr00tN1d5ModelArch::predict(const Inputs& in) { ggml_backend_tensor_set(gio.t_x0, x_init.data(), 0, ggml_nbytes(gio.t_x0)); - for (int64_t s=0; s tau, tpr; - action_sinusoid(bucket, E, AH, tau); - timesteps_proj(bucket, tpr); - ggml_backend_tensor_set(gio.t_tau[s], tau.data(), 0, ggml_nbytes(gio.t_tau[s])); - ggml_backend_tensor_set(gio.t_tproj[s], tpr.data(), 0, ggml_nbytes(gio.t_tproj[s])); - } - graph_unique_names(main_graph.graph()); const auto tc0 = std::chrono::steady_clock::now(); const ggml_status status = ggml_backend_graph_compute(backend, main_graph.graph()); diff --git a/src/models/gr00tn1d6.cpp b/src/models/gr00tn1d6.cpp index c017f36..0824a39 100644 --- a/src/models/gr00tn1d6.cpp +++ b/src/models/gr00tn1d6.cpp @@ -15,7 +15,6 @@ #include "arch.h" #include "options.h" #include "backend.h" -#include "env_flag.h" #include "gguf_reader.h" #include "layers/embed.h" #include "layers/ffn.h" @@ -36,10 +35,8 @@ #include #include #include -#include #include #include -#include #include #include @@ -56,8 +53,9 @@ struct Gr00tN1d6ModelArch : public ModelArchBase { ggml_context * ctx_weights = nullptr; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; - scratch_ctx vision_scratch; - scratch_ctx merge_scratch; + FlowTimes times; + std::vector c_mask; + int64_t c_mask_seq = -1; struct MainKey { int64_t seq=-1, n_img=-1, seq_txt=-1, nsteps=-1; @@ -68,9 +66,12 @@ struct Gr00tN1d6ModelArch : public ModelArchBase { struct MainIO { ggml_tensor *t_embeds=nullptr,*t_pos=nullptr,*t_lmmask=nullptr,*t_state=nullptr,*t_x0=nullptr; ggml_tensor *t_img_idx=nullptr,*t_txt_idx=nullptr,*actions=nullptr; - std::vector t_tau, t_tproj; }; graph_cache main_graph; + struct VisIO { + ggml_tensor *t_patches=nullptr,*vit_embeds=nullptr; + }; + graph_cache vis_graph; SigLipTower vit; Qwen3LM lm; @@ -93,10 +94,10 @@ struct Gr00tN1d6ModelArch : public ModelArchBase { namespace { -bool load_config(const gguf_reader & g, Gr00tN1d6ModelArch & m, Config & cfg) { +bool load_config(const gguf_reader & g, const Options & opts, Gr00tN1d6ModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; auto F = [&](const char * k, float & dst) { if (g.has(k)) dst = g.f32(k); }; - auto fk = [&](const char * s) { static char b[64]; std::snprintf(b, sizeof(b), "gr00t_n1_6.%s", s); return b; }; + auto fk = [&](const char * s) { thread_local char b[64]; std::snprintf(b, sizeof(b), "gr00t_n1_6.%s", s); return b; }; U(fk("vit_hidden" ), m.vit.enc.cfg.hidden); U(fk("vit_layers" ), m.vit_layers); @@ -142,47 +143,20 @@ bool load_config(const gguf_reader & g, Gr00tN1d6ModelArch & m, Config & cfg) { if (g.has(fk("lm_rope_theta"))) m.lm.cfg.rope.freq_base = (float) g.f64(fk("lm_rope_theta")); + if (m.vit.enc.cfg.heads <= 0 || m.attend_text_every_n <= 0) { + std::fprintf(stderr, "vla(gr00tn1d6): vit_heads %lld and attend_text_every_n_blocks %lld must be positive\n", + (long long) m.vit.enc.cfg.heads, (long long) m.attend_text_every_n); + return false; + } m.vit.enc.cfg.head_dim = m.vit.enc.cfg.hidden/m.vit.enc.cfg.heads; m.lm.cfg.rope.n_dims = (int) m.lm.cfg.head_dim; + m.vit.enc.cfg.flash_attn = m.lm.cfg.flash_attn = flash_attn_enabled(); m.aex.embodiment_id = 20; - { - const std::string js = g.str(fk("embodiment_id_mapping")); - auto lookup = [&](const char * key) -> long { - const std::string k = std::string("\"")+key+"\""; - size_t p = js.find(k); - if (p == std::string::npos) - return -1; - p = js.find(':', p+k.size()); - if (p == std::string::npos) - return -1; - return std::strtol(js.c_str()+p+1, nullptr, 10); - }; - - const long gr1 = lookup("gr1"); - if (gr1 >= 0) - m.aex.embodiment_id = gr1; - - if (const char * e = std::getenv("VLA_GR00T_EMBODIMENT")) { - char * end = nullptr; - const long v = std::strtol(e, &end, 10); - if (end && *end == '\0') { - m.aex.embodiment_id = v; - } else { - const long id = lookup(e); - if (id >= 0) - m.aex.embodiment_id = id; - else std::fprintf(stderr, "vla(gr00tn1d6): embodiment tag '%s' not in embodiment_id_mapping; using id %lld\n", e, (long long) m.aex.embodiment_id); - } - } - } - if (m.aex.embodiment_id < 0 || m.aex.embodiment_id >= m.max_embodiments) { - std::fprintf(stderr, "vla(gr00tn1d6): embodiment id %lld out of range [0,%lld)\n", - (long long) m.aex.embodiment_id, (long long) m.max_embodiments); + if (!resolve_embodiment("gr00tn1d6", g.str(fk("embodiment_id_mapping")), "gr1", m.max_embodiments, m.aex.embodiment_id)) return false; - } - // pixel_shuffle_back writes (grid/shuffle)^2 tokens into a buffer sized from + // The pixel unshuffle yields (grid/shuffle)^2 tokens and is reshaped to // n_img_tokens, so the KV has to agree with the grid it is derived from. if (m.patch_size <= 0 || m.vit_pixel_shuffle <= 0 || m.image_size%m.patch_size != 0 || (m.image_size/m.patch_size)%m.vit_pixel_shuffle != 0) { @@ -199,6 +173,9 @@ bool load_config(const gguf_reader & g, Gr00tN1d6ModelArch & m, Config & cfg) { } } + if (!resolve_num_steps("gr00tn1d6", opts, m.num_steps)) + return false; + cfg = Config{}; cfg.n_img = m.n_img_tokens; cfg.n_lang = m.max_seq_len; @@ -254,7 +231,7 @@ std::unique_ptr gr00t_n1_6_create(const std::string& mmproj_path, std::fprintf(stderr, "vla(gr00tn1d6): %s is not a gr00t_n1_6 GGUF\n", ckpt_path.c_str()); return nullptr; } - if (!load_config(g, *m, m->cfg)) + if (!load_config(g, opts, *m, m->cfg)) return nullptr; std::printf("vla(gr00tn1d6): vit=%lldd×%lldL×%lldh (Linear patch embed) pixel_shuffle÷%lld ⇒ n_img_tok=%lld mlp1=LN(%lld)→Linear→GELU→Linear " @@ -294,11 +271,14 @@ std::unique_ptr gr00t_n1_6_create(const std::string& mmproj_path, m->vlln_w = L.f32("aex.vlln.weight"); m->vlln_b = L.f32("aex.vlln.bias"); - m->aex.declare(L, "aex"); + if (!m->aex.declare(L, g, m->backend, "aex")) + return nullptr; m->dit.declare(L, "aex.dit"); if (!L.upload(m->backend, &m->weight_buf)) return nullptr; + if (!m->times.build("gr00tn1d6", m->backend, m->dit, m->num_steps, m->num_buckets, m->in_embed_dim, m->action_horizon)) + return nullptr; std::printf("vla(gr00tn1d6): weights resident in %.2f GiB (%s) - incl. SigLIP2 vision tower; embodiment id %lld\n", ggml_backend_buffer_get_size(m->weight_buf)/(1024.0*1024.0*1024.0), @@ -312,7 +292,6 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { const int64_t H = lm.cfg.hidden; const int64_t K = n_img_tokens; - const int64_t E = in_embed_dim; const int64_t grid = image_size/patch_size; const int64_t r = vit_pixel_shuffle; const int64_t n_patches = grid*grid; @@ -320,7 +299,6 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { const int64_t c4 = vit.enc.cfg.hidden*r*r; const int64_t AD = action_dim; const int64_t AH = action_horizon; - const int64_t Nsa = 1+AH; int64_t n_views = 0; std::vector img_emb_host; @@ -333,36 +311,33 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { n_views = in.n_images; img_emb_host.assign((size_t) n_views*K*H, 0.0f); - ggml_context * VC = vision_scratch.reset((size_t) 64*1024*1024); - if (!VC) { std::fprintf(stderr, "vla(gr00tn1d6): ggml_init(vision ctx A) failed\n"); return {}; } - - ggml_tensor * t_patches = ggml_new_tensor_3d(VC, GGML_TYPE_F32, patch_dim, n_patches, n_views); - ggml_set_input(t_patches); - ggml_tensor * post_ln = vit.build(VC, vit.embed_patches(VC, t_patches), n_patches, n_views); - ggml_set_output(post_ln); - - ggml_cgraph * vgA = ggml_new_graph_custom(VC, 8192, false); - ggml_build_forward_expand(vgA, post_ln); - if (!vision_scratch.alloc(backend, vgA)) { std::fprintf(stderr, "vla(gr00tn1d6): vision gallocr A alloc failed\n"); return {}; } - - ggml_context * MC = merge_scratch.reset((size_t) 16*1024*1024); - if (!MC) { std::fprintf(stderr, "vla(gr00tn1d6): ggml_init(vision ctx B) failed\n"); return {}; } - - ggml_tensor * t_shuf = ggml_new_tensor_3d(MC, GGML_TYPE_F32, c4, K, n_views); - ggml_set_input(t_shuf); - ggml_tensor * mln = layer_norm(MC, t_shuf, mm_ln_w, mm_ln_b, connector_ln_eps); - ggml_tensor * vit_embeds = ffn_gelu_erf(MC, mm_fc1_w, mm_fc1_b, mm_fc2_w, mm_fc2_b, mln); - ggml_set_output(vit_embeds); - - ggml_cgraph * vgB = ggml_new_graph(MC); - ggml_build_forward_expand(vgB, vit_embeds); - if (!merge_scratch.alloc(backend, vgB)) { std::fprintf(stderr, "vla(gr00tn1d6): vision gallocr B alloc failed\n"); return {}; } + const bool vbuilt = vis_graph.ensure(backend, n_views, (size_t) 64*1024*1024, [&](ggml_context * VC, VisIO & vio) -> ggml_cgraph * { + ggml_tensor * t_patches = ggml_new_tensor_3d(VC, GGML_TYPE_F32, patch_dim, n_patches, n_views); + ggml_set_input(t_patches); + ggml_tensor * post_ln = vit.build(VC, vit.embed_patches(VC, t_patches), n_patches, n_views); + + const int64_t hid = vit.enc.cfg.hidden, g2 = grid/r; + ggml_tensor * shuf = ggml_reshape_3d(VC, post_ln, hid, r, g2*r*g2*n_views); + shuf = ggml_cont(VC, ggml_permute(VC, shuf, 1, 0, 2, 3)); + shuf = ggml_reshape_4d(VC, shuf, r, hid*g2, r, g2*n_views); + shuf = ggml_cont(VC, ggml_permute(VC, shuf, 0, 2, 1, 3)); + shuf = ggml_reshape_3d(VC, shuf, c4, K, n_views); + ggml_tensor * mln = layer_norm(VC, shuf, mm_ln_w, mm_ln_b, connector_ln_eps); + ggml_tensor * vit_embeds = ffn_gelu_erf(VC, mm_fc1_w, mm_fc1_b, mm_fc2_w, mm_fc2_b, mln); + ggml_set_output(vit_embeds); + vio.t_patches = t_patches; vio.vit_embeds = vit_embeds; + + ggml_cgraph * vg = ggml_new_graph_custom(VC, 8192, false); + ggml_build_forward_expand(vg, vit_embeds); + return vg; + }); + if (!vbuilt) { std::fprintf(stderr, "vla(gr00tn1d6): vision graph build failed\n"); return {}; } + ggml_tensor * t_patches = vis_graph.io().t_patches, * vit_embeds = vis_graph.io().vit_embeds; + ggml_cgraph * vg = vis_graph.graph(); const auto tv0 = std::chrono::steady_clock::now(); std::vector patches; std::vector patches_all((size_t) patch_dim*n_patches*n_views); - std::vector post_ln_host((size_t) vit.enc.cfg.hidden*n_patches*n_views); - std::vector shuf_host((size_t) c4*K*n_views); bool vok = true; for (int64_t v=0; v Gr00tN1d6ModelArch::predict(const Inputs& in) { } if (vok) { ggml_backend_tensor_set(t_patches, patches_all.data(), 0, ggml_nbytes(t_patches)); - graph_unique_names(vgA); - if (ggml_backend_graph_compute(backend, vgA) != GGML_STATUS_SUCCESS) { - std::fprintf(stderr, "vla(gr00tn1d6): vision compute A failed\n"); - vok = false; - } - } - if (vok) { - ggml_backend_tensor_get(post_ln, post_ln_host.data(), 0, ggml_nbytes(post_ln)); - for (int64_t v=0; v Gr00tN1d6ModelArch::predict(const Inputs& in) { const int64_t SEQ_TXT = prompt.n_text(); std::vector inputs_embeds; - if (!fetch_embeds("gr00tn1d6", io, prompt, img_emb_ptr, H, inputs_embeds)) return {}; + if (!fetch_embeds(io, prompt, img_emb_ptr, H, inputs_embeds)) return {}; std::vector x_init; init_noise(in, (size_t) AH*AD, x_init); const MainKey mkey{ SEQ, n_img, SEQ_TXT, num_steps }; - const bool built = main_graph.ensure(backend, mkey, (size_t) 256*1024*1024, + const size_t main_nodes = 65536 + (size_t) num_steps*64*(dit.cfg.layers+1); + const bool built = main_graph.ensure(backend, mkey, main_nodes*ggml_tensor_overhead() + ggml_graph_overhead_custom(main_nodes, false), [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { ggml_tensor * t_embeds = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, SEQ); ggml_set_input(t_embeds); ggml_tensor * t_pos = ggml_new_tensor_1d(C, GGML_TYPE_I32, SEQ); ggml_set_input(t_pos); @@ -430,56 +393,20 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { if (t_txt_idx) ggml_set_input(t_txt_idx); - std::vector t_tau(num_steps), t_tproj(num_steps); - for (int64_t s=0; s Kc(dit.cfg.layers, nullptr), Vc(dit.cfg.layers, nullptr); - for (int64_t i=0; inb[1], (size_t)(Nsa-AH)*pred->nb[1])); - actions = ggml_add(C, actions, ggml_scale(C, vel, dt)); - } + ggml_tensor * actions = aex.denoise(C, dit, times, dit_interleave != 0, 2*attend_text_every_n, t_state, nullptr, + vl_txt, vl_img, t_x0); ggml_set_name(actions, "action_pred"); ggml_set_output(actions); gio.t_embeds=t_embeds; gio.t_pos=t_pos; gio.t_lmmask=t_lmmask; gio.t_state=t_state; gio.t_x0=t_x0; - gio.t_img_idx=t_img_idx; gio.t_txt_idx=t_txt_idx; gio.t_tau=t_tau; gio.t_tproj=t_tproj; gio.actions=actions; + gio.t_img_idx=t_img_idx; gio.t_txt_idx=t_txt_idx; gio.actions=actions; - ggml_cgraph * gf = ggml_new_graph_custom(C, 65536, false); + ggml_cgraph * gf = ggml_new_graph_custom(C, main_nodes, false); ggml_build_forward_expand(gf, actions); return gf; }); @@ -494,9 +421,11 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { pp[i] = (int32_t) i; ggml_backend_tensor_set(gio.t_pos, pp.data(), 0, ggml_nbytes(gio.t_pos)); - std::vector mask; - build_causal_mask(SEQ, mask); - ggml_backend_tensor_set(gio.t_lmmask, mask.data(), 0, ggml_nbytes(gio.t_lmmask)); + if (c_mask_seq != SEQ) { + build_causal_mask(SEQ, c_mask); + c_mask_seq = SEQ; + } + ggml_backend_tensor_set(gio.t_lmmask, c_mask.data(), 0, ggml_nbytes(gio.t_lmmask)); std::vector st(max_state_dim, 0.0f); for (int64_t i=0; i Gr00tN1d6ModelArch::predict(const Inputs& in) { if (gio.t_txt_idx) ggml_backend_tensor_set(gio.t_txt_idx, prompt.text_pos.data(), 0, ggml_nbytes(gio.t_txt_idx)); - for (int64_t s=0; s tau, tpr; - action_sinusoid(bucket, E, AH, tau); - timesteps_proj(bucket, tpr); - ggml_backend_tensor_set(gio.t_tau[s], tau.data(), 0, ggml_nbytes(gio.t_tau[s])); - ggml_backend_tensor_set(gio.t_tproj[s], tpr.data(), 0, ggml_nbytes(gio.t_tproj[s])); - } - graph_unique_names(main_graph.graph()); const auto tc0 = std::chrono::steady_clock::now(); const ggml_status status = ggml_backend_graph_compute(backend, main_graph.graph()); diff --git a/src/models/gr00tn1d7.cpp b/src/models/gr00tn1d7.cpp index 0e959b3..433f184 100644 --- a/src/models/gr00tn1d7.cpp +++ b/src/models/gr00tn1d7.cpp @@ -13,72 +13,52 @@ // limitations under the License. #include "arch.h" -#include "layers/attn.h" -#include "layers/linear.h" #include "layers/norm.h" #include "modules/action_expert.h" #include "modules/dit_head.h" #include "modules/encoder.h" +#include "modules/prompt.h" #include "modules/qwen3_lm.h" #include "options.h" #include "model.h" #include "ggml.h" -#include "ggml-cpu.h" #include "ggml-backend.h" #include "backend.h" -#include "gguf.h" #include "gguf_reader.h" #include "scratch_ctx.h" #include "layers/embed.h" #include "modules/qwen3vl_vit.h" -#include "env_flag.h" #include -#include #include #include #include #include -#include #include -#include -#include #include #include namespace vla { -namespace { - - -struct VlsaLayerW { ggml_tensor *n1w,*n1b,*n3w,*n3b,*Wq,*bq,*Wk,*bk,*Wv,*bv,*Wo,*bo,*Wff0,*bff0,*Wff2,*bff2; }; - -} struct Gr00tN1d7ModelArch : public ModelArchBase { Gr00tN1d7ModelArch() : ModelArchBase(Arch::GR00T_N1_7) {} ~Gr00tN1d7ModelArch() override; - std::string gguf_path; ggml_backend_t backend = nullptr; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; - scratch_ctx vision_scratch; + graph_cache vision_graph; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; - int64_t vit_hidden=1024, vit_layers=24, vit_heads=16, vit_inter=4096; - int64_t patch_size=16, temporal_patch=2, spatial_merge=2, vit_num_pos=2304, vit_patch_flat=1536, vit_merged_dim=4096; - int64_t deepstack_idx[3] = {5, 11, 17}; int64_t lm_hidden=2048, lm_layers=16, n_q=16, n_kv=8, lm_head_dim=128, lm_inter=6144, vocab=151936, image_token_index=151655; int64_t vlsa_layers=4, vlsa_heads=32, vlsa_head_dim=64, vlsa_ff_inner=8192; int64_t bb_embed_dim=2048, in_embed_dim=1536, dit_hidden=1536, dit_heads=32, dit_head_dim=48, dit_layers=32, dit_interleave=1, attend_text_every_n=2; int64_t action_horizon=40, action_dim=132, max_state_dim=132; int64_t num_steps=4, num_buckets=1000, max_embodiments=32, max_seq_len=1024; - int64_t image_target_size=256; - float vit_ln_eps=1e-6f, vit_rope_base=10000.0f, lm_rms_eps=1e-6f, lm_rope_base=5000000.0f; - float vlln_eps=1e-5f, vlsa_ln_eps=1e-5f, ln_eps=1e-5f, norm_out_eps=1e-6f, connector_ln_eps=1e-6f; - int64_t embodiment_id = 2; + float lm_rms_eps=1e-6f, lm_rope_base=5000000.0f; + float vlln_eps=1e-5f, vlsa_ln_eps=1e-5f, ln_eps=1e-5f, norm_out_eps=1e-6f; Qwen3VLTower vit; Qwen3LM lm; @@ -87,14 +67,9 @@ struct Gr00tN1d7ModelArch : public ModelArchBase { DitHead dit; ggml_tensor *vlln_w=nullptr,*vlln_b=nullptr; - bool caches_ready = false; - std::vector c_grow, c_gcol; - std::vector c_rope_cos, c_rope_sin; - std::vector c_pos_interp; - std::vector> c_tau, c_tproj; - std::vector c_mask; int64_t c_mask_seq = -1; - gguf_reader io; - bool build_caches(); + FlowTimes times; + std::vector c_mask; int64_t c_mask_seq = -1; + gguf_reader io{"gr00tn1d7"}; struct MainKey { int64_t seq=-1, n_img=-1, seq_txt=-1, nsteps=-1; bool deepstack=false; @@ -107,7 +82,6 @@ struct Gr00tN1d7ModelArch : public ModelArchBase { ggml_tensor *t_embeds=nullptr,*t_pos=nullptr,*t_lmmask=nullptr,*t_state=nullptr,*t_x0=nullptr; ggml_tensor *t_ds[3]={nullptr,nullptr,nullptr}; ggml_tensor *t_img_idx=nullptr,*t_txt_idx=nullptr,*actions=nullptr; - std::vector t_tau, t_tproj; }; graph_cache mg; @@ -116,18 +90,12 @@ struct Gr00tN1d7ModelArch : public ModelArchBase { namespace { - - - - -bool load_config(const gguf_reader & g, Gr00tN1d7ModelArch & m, Config & cfg) { +bool load_config(const gguf_reader & g, const Options & opts, Gr00tN1d7ModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; auto F = [&](const char * k, float & dst) { if (g.has(k)) dst = g.f32(k); }; - auto fk = [&](const char * s) { static char b[64]; std::snprintf(b, sizeof(b), "gr00t_n1_7.%s", s); return b; }; - U(fk("vit_hidden"), m.vit_hidden); U(fk("vit_layers"), m.vit_layers); U(fk("vit_heads"), m.vit_heads); U(fk("vit_inter"), m.vit_inter); - U(fk("patch_size"), m.patch_size); U(fk("temporal_patch_size"), m.temporal_patch); U(fk("spatial_merge_size"), m.spatial_merge); - U(fk("vit_num_position_embeddings"), m.vit_num_pos); U(fk("vit_patch_flat"), m.vit_patch_flat); U(fk("vit_merged_dim"), m.vit_merged_dim); - U(fk("deepstack_idx_0"), m.deepstack_idx[0]); U(fk("deepstack_idx_1"), m.deepstack_idx[1]); U(fk("deepstack_idx_2"), m.deepstack_idx[2]); + auto fk = [&](const char * s) { thread_local char b[64]; std::snprintf(b, sizeof(b), "gr00t_n1_7.%s", s); return b; }; + if (!m.vit.load_config("gr00tn1d7", g, "gr00t_n1_7")) + return false; U(fk("lm_hidden"), m.lm_hidden); U(fk("lm_layers_used"), m.lm_layers); U(fk("lm_q_heads"), m.n_q); U(fk("lm_kv_heads"), m.n_kv); U(fk("lm_head_dim"), m.lm_head_dim); U(fk("lm_inter"), m.lm_inter); U(fk("vocab_size"), m.vocab); U(fk("image_token_index"), m.image_token_index); U(fk("vlsa_layers"), m.vlsa_layers); U(fk("vlsa_heads"), m.vlsa_heads); U(fk("vlsa_head_dim"), m.vlsa_head_dim); U(fk("vlsa_ff_inner"), m.vlsa_ff_inner); @@ -136,26 +104,16 @@ bool load_config(const gguf_reader & g, Gr00tN1d7ModelArch & m, Config & cfg) { U(fk("attend_text_every_n_blocks"), m.attend_text_every_n); U(fk("action_horizon"), m.action_horizon); U(fk("action_dim"), m.action_dim); U(fk("max_state_dim"), m.max_state_dim); U(fk("num_inference_timesteps"), m.num_steps); U(fk("num_timestep_buckets"), m.num_buckets); U(fk("max_num_embodiments"), m.max_embodiments); U(fk("max_seq_len"), m.max_seq_len); - U(fk("image_target_size"), m.image_target_size); - - // merge_block_coords only enumerates the patch grid exactly when the spatial - // merge divides it; otherwise it emits rows past the position table. - if (m.patch_size <= 0 || m.spatial_merge <= 0 || m.image_target_size%m.patch_size != 0 || - (m.image_target_size/m.patch_size)%m.spatial_merge != 0) { - std::fprintf(stderr, "vla(gr00tn1d7): image %lld / patch %lld / merge %lld do not divide evenly\n", - (long long) m.image_target_size, (long long) m.patch_size, (long long) m.spatial_merge); + + if (m.attend_text_every_n <= 0) { + std::fprintf(stderr, "vla(gr00tn1d7): attend_text_every_n_blocks %lld must be positive\n", (long long) m.attend_text_every_n); return false; } - if (const char * ns = std::getenv("VLA_NUM_STEPS")) { - char * end = nullptr; long v = std::strtol(ns, &end, 10); - if (end && *end == '\0' && v >= 1) { - m.num_steps = (int64_t) v; - std::fprintf(stderr, "vla(gr00tn1d7): VLA_NUM_STEPS override → num_steps=%lld\n", (long long) v); - } - } - F(fk("vit_ln_eps"), m.vit_ln_eps); F(fk("lm_rms_eps"), m.lm_rms_eps); F(fk("ln_eps"), m.ln_eps); F(fk("norm_out_eps"), m.norm_out_eps); - F(fk("vlln_eps"), m.vlln_eps); F(fk("vlsa_ln_eps"), m.vlsa_ln_eps); F(fk("connector_ln_eps"), m.connector_ln_eps); F(fk("vit_rope_theta"), m.vit_rope_base); + if (!resolve_num_steps("gr00tn1d7", opts, m.num_steps)) + return false; + F(fk("lm_rms_eps"), m.lm_rms_eps); F(fk("ln_eps"), m.ln_eps); F(fk("norm_out_eps"), m.norm_out_eps); + F(fk("vlln_eps"), m.vlln_eps); F(fk("vlsa_ln_eps"), m.vlsa_ln_eps); if (g.has(fk("lm_rope_theta"))) m.lm_rope_base = (float) g.f64(fk("lm_rope_theta")); @@ -189,29 +147,12 @@ bool load_config(const gguf_reader & g, Gr00tN1d7ModelArch & m, Config & cfg) { m.dit.cfg.norm_out_eps = m.norm_out_eps; m.aex.embodiment_id = 2; - { - const std::string js = g.str(fk("embodiment_id_mapping")); - auto lookup = [&](const char * key) -> long { - const std::string k = std::string("\"")+key + "\""; - size_t p = js.find(k); if (p == std::string::npos) return -1; - p = js.find(':', p+k.size()); if (p == std::string::npos) return -1; - return std::strtol(js.c_str()+p+1, nullptr, 10); - }; - long ls = lookup("libero_sim"); if (ls >= 0) m.aex.embodiment_id = ls; - if (const char * e = std::getenv("VLA_GR00T_EMBODIMENT")) { - char * end = nullptr; long v = std::strtol(e, &end, 10); - if (end && *end == '\0') - m.aex.embodiment_id = v; - else { long id = lookup(e); if (id >= 0) m.aex.embodiment_id = id; else std::fprintf(stderr, "vla(gr00tn1d7): embodiment tag '%s' not in embodiment_id_mapping; using id %lld\n", e, (long long) m.aex.embodiment_id); } - } - } - if (m.aex.embodiment_id < 0 || m.aex.embodiment_id >= m.max_embodiments) { - std::fprintf(stderr, "vla(gr00tn1d7): embodiment id %lld out of range [0,%lld)\n", (long long) m.aex.embodiment_id, (long long) m.max_embodiments); + if (!resolve_embodiment("gr00tn1d7", g.str(fk("embodiment_id_mapping")), "libero_sim", m.max_embodiments, m.aex.embodiment_id)) return false; - } cfg = Config{}; - cfg.n_img = 64; cfg.n_lang = m.max_seq_len; cfg.n_state = 1; + cfg.n_img = m.vit.n_tokens(); + cfg.n_lang = m.max_seq_len; cfg.n_state = 1; cfg.n_suffix = m.action_horizon; cfg.max_state_dim = m.max_state_dim; cfg.max_action_dim = m.action_dim; cfg.real_state_dim = m.max_state_dim; cfg.real_action_dim = m.action_dim; cfg.hidden = m.lm_hidden; cfg.n_q_heads = m.n_q; cfg.n_kv_heads = m.n_kv; cfg.head_dim = m.lm_head_dim; cfg.n_layers = m.lm_layers; @@ -244,23 +185,22 @@ std::unique_ptr gr00t_n1_7_create(const std::string& mmproj_path, std::printf("vla(gr00tn1d7): note - mmproj '%s' is ignored (the vision tower is bundled in the combined GGUF)\n", mmproj_path.c_str()); auto m = std::make_unique(); - m->gguf_path = ckpt_path; m->matmul_type = opts.weight_dtype.value_or(vla::default_weight_dtype(GGML_TYPE_BF16)); - gguf_reader g("gr00tn1d7"); - if (!g.open(ckpt_path)) + if (!m->io.open(ckpt_path)) return nullptr; + gguf_reader & g = m->io; if (!g.has("gr00t_n1_7.architecture")) { std::fprintf(stderr, "vla(gr00tn1d7): %s is not a gr00t_n1_7 GGUF\n", ckpt_path.c_str()); return nullptr; } - if (!load_config(g, *m, m->cfg)) + if (!load_config(g, opts, *m, m->cfg)) return nullptr; std::printf("vla(gr00tn1d7): vit=Qwen3-VL %lldd×%lldL×%lldh (Conv3d patch %lld², temporal %lld; learned pos %lld + 2D rope; deepstack@{%lld,%lld,%lld}; merge÷%lld) " "lm=Qwen3-VL %lldd×%lldL (%lldq/%lldkv×%lld, θ=%g) vlsa=%lldL×%lldh×%lld dit=AlternateVLDiT %lldL×%lldh×%lld(inner %lld) attend_text_every_n=%lld " "in_emb=%lld horizon=%lld action_dim=%lld max_state=%lld N_steps=%lld embodiment=%lld resident=%s\n", - (long long) m->vit_hidden, (long long) m->vit_layers, (long long) m->vit_heads, (long long) m->patch_size, (long long) m->temporal_patch, - (long long) m->vit_num_pos, (long long) m->deepstack_idx[0], (long long) m->deepstack_idx[1], (long long) m->deepstack_idx[2], (long long) m->spatial_merge, + (long long) m->vit.hidden, (long long) m->vit.layers, (long long) m->vit.heads, (long long) m->vit.patch, (long long) m->vit.temporal, + (long long) m->vit.num_pos, (long long) m->vit.deepstack_idx[0], (long long) m->vit.deepstack_idx[1], (long long) m->vit.deepstack_idx[2], (long long) m->vit.merge, (long long) m->lm_hidden, (long long) m->lm_layers, (long long) m->n_q, (long long) m->n_kv, (long long) m->lm_head_dim, (double) m->lm_rope_base, (long long) m->vlsa_layers, (long long) m->vlsa_heads, (long long) m->vlsa_head_dim, (long long) m->dit_layers, (long long) m->dit_heads, (long long) m->dit_head_dim, (long long) m->dit_hidden, (long long) m->attend_text_every_n, (long long) m->in_embed_dim, @@ -284,146 +224,49 @@ std::unique_ptr gr00t_n1_7_create(const std::string& mmproj_path, WeightLoader L("gr00tn1d7", g, m->ctx_weights, m->matmul_type); - m->vit.declare(L, "vit", m->vit_layers); + m->vit.declare(L, "vit"); m->lm.declare(L, "vlm"); m->vlln_w = L.f32("aex.vlln.weight"); m->vlln_b = L.f32("aex.vlln.bias"); m->vlsa.declare(L, "aex.vlsa", m->vlsa_layers, EncNames{"norm1", "norm3", "ff0", "ff2"}); - m->aex.declare(L, "aex"); + if (!m->aex.declare(L, g, m->backend, "aex")) + return nullptr; m->dit.declare(L, "aex.dit", true, m->dit_interleave != 0); if (!L.upload(m->backend, &m->weight_buf)) return nullptr; + if (!m->times.build("gr00tn1d7", m->backend, m->dit, m->num_steps, m->num_buckets, m->in_embed_dim, m->action_horizon)) + return nullptr; std::printf("vla(gr00tn1d7): QKV-fused DiT (self Wqkv / cross Wkv)\n"); std::printf("vla(gr00tn1d7): weights resident in %.2f GiB (%s) - incl. Qwen3-VL vision tower + deepstack + vl_self_attention; embodiment id %lld\n", ggml_backend_buffer_get_size(m->weight_buf)/(1024.0*1024.0*1024.0), dtype_name(m->matmul_type), (long long) m->aex.embodiment_id); - if (!m->build_caches()) { - std::fprintf(stderr, "vla(gr00tn1d7): build_caches failed\n"); + if (!m->vit.build_caches("gr00tn1d7", m->io)) return nullptr; - } return m; } -bool Gr00tN1d7ModelArch::build_caches() { - if (caches_ready) - return true; - const int64_t side = image_target_size, ps = patch_size, m2 = spatial_merge; - const int64_t grid = side/ps; - const int64_t hd_vit = vit_hidden/vit_heads; - const int64_t num_side = (int64_t) std::lround(std::sqrt((double) vit_num_pos)); - const int64_t E = in_embed_dim, AH = action_horizon; - - merge_block_coords(grid, grid, m2, c_grow, c_gcol); - vit_rope_tables(c_grow, c_gcol, hd_vit, (double) vit_rope_base, c_rope_cos, c_rope_sin); - - if (!io.open(gguf_path)) { - std::fprintf(stderr, "vla(gr00tn1d7): build_caches: io.open(%s) failed\n", gguf_path.c_str()); - return false; - } - std::vector pos_table = io.read_f32("vit.pos_embd"); - if (pos_table.empty() || (int64_t) pos_table.size() != vit_num_pos * vit_hidden) { - std::fprintf(stderr, "vla(gr00tn1d7): build_caches: vit.pos_embd unreadable\n"); return false; - } - interp_pos_embed(pos_table, num_side, vit_hidden, c_grow, c_gcol, grid, grid, c_pos_interp); - - c_tau.assign((size_t) num_steps, {}); c_tproj.assign((size_t) num_steps, {}); - for (int64_t s=0; s Gr00tN1d7ModelArch::predict(const Inputs& in) { const auto t0 = std::chrono::steady_clock::now(); stats = Stats{}; - const int64_t H = lm_hidden, E = in_embed_dim; - const int64_t side = image_target_size; - const int64_t ps = patch_size, m2 = spatial_merge; - const int64_t grid = side/ps; - const int64_t n_patches = grid * grid; - const int64_t K = (grid/m2)*(grid/m2); - const int64_t hd_vit = vit_hidden/vit_heads; - const int64_t AD = action_dim, AH = action_horizon, Nsa = 1+AH; + const int64_t H = lm_hidden; + const int64_t K = vit.n_tokens(); + const int64_t AD = action_dim, AH = action_horizon; const bool do_dump = (std::getenv("VLA_GR00T_N17_DUMP") != nullptr); - if (!caches_ready) { std::fprintf(stderr, "vla(gr00tn1d7): caches not ready\n"); return {}; } - const std::vector & grow = c_grow, & gcol = c_gcol; - const std::vector & rope_cos = c_rope_cos, & rope_sin = c_rope_sin, & pos_interp = c_pos_interp; - int64_t n_views = 0; std::vector img_emb_host, ds_host[3]; const float * img_emb_ptr = nullptr; if (in.precomputed_img_emb && in.n_img_views > 0) { n_views = in.n_img_views; img_emb_ptr = in.precomputed_img_emb; - - for (int j=0; j<3; ++j) - ds_host[j].assign((size_t) n_views * K * H, 0.0f); } else if (in.images && in.n_images > 0) { n_views = in.n_images; - img_emb_host.assign((size_t) n_views * K * H, 0.0f); - for (int j=0; j<3; ++j) - ds_host[j].assign((size_t) n_views * K * H, 0.0f); - - ggml_context * VC = vision_scratch.reset((size_t) 512*1024*1024); - if (!VC) { std::fprintf(stderr, "vla(gr00tn1d7): ggml_init(vision ctx) failed\n"); return {}; } - ggml_tensor * t_patches = ggml_new_tensor_2d(VC, GGML_TYPE_F32, vit_patch_flat, n_patches); ggml_set_input(t_patches); - ggml_tensor * t_pos = ggml_new_tensor_2d(VC, GGML_TYPE_F32, vit_hidden, n_patches); ggml_set_input(t_pos); - ggml_tensor * t_cos = ggml_new_tensor_2d(VC, GGML_TYPE_F32, hd_vit, n_patches); ggml_set_input(t_cos); - ggml_tensor * t_sin = ggml_new_tensor_2d(VC, GGML_TYPE_F32, hd_vit, n_patches); ggml_set_input(t_sin); - ggml_tensor * h = ggml_add(VC, ggml_add(VC, ggml_mul_mat(VC, vit.patch_w, t_patches), vit.patch_b), t_pos); - - ggml_set_output(h); - ggml_tensor * stash[3] = {nullptr, nullptr, nullptr}; - for (int64_t i=0; i patches; - bool vok = true; - for (int64_t v=0; v(std::chrono::steady_clock::now()-tv0).count(); if (!vok) return {}; img_emb_ptr = img_emb_host.data(); @@ -432,68 +275,28 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { } const int64_t n_img = n_views * K; - std::vector input_ids; - int64_t n_img_slots = 0; - for (int j=0; j max_seq_len) { std::fprintf(stderr, "vla(gr00tn1d7): prompt too long (%lld > %lld)\n", (long long) SEQ, (long long) max_seq_len); return {}; } - - std::vector inputs_embeds((size_t) SEQ * H); - if (!io.fetch_rows_f32("token_embd.weight", input_ids, inputs_embeds.data(), H)) return {}; - { int64_t k = 0; - for (int64_t p=0; p= n_img) { std::fprintf(stderr, "vla(gr00tn1d7): more tokens than ViT embeds\n"); return {}; } - std::memcpy(inputs_embeds.data()+p * H, img_emb_ptr+k * H, H * sizeof(float)); ++k; - } - } + Prompt prompt; + if (!build_prompt("gr00tn1d7", in, n_img, (int32_t) image_token_index, max_seq_len, prompt)) return {}; + const int64_t SEQ = prompt.len(), SEQ_TXT = prompt.n_text(); - std::vector image_pos_idx, text_pos_idx; - image_pos_idx.reserve((size_t) n_img); text_pos_idx.reserve((size_t) (SEQ-n_img)); - for (int64_t p=0; p inputs_embeds; + if (!fetch_embeds(io, prompt, img_emb_ptr, H, inputs_embeds)) return {}; + + std::vector pp; + if (!mrope_positions("gr00tn1d7", prompt.ids, (int32_t) image_token_index, vit.grid()/vit.merge, pp)) return {}; std::vector> ds_pad(3); - const bool inject_deepstack = (in.images && in.n_images > 0); + const bool inject_deepstack = !img_emb_host.empty(); if (inject_deepstack) for (int j=0; j<3; ++j) { ds_pad[j].assign((size_t) SEQ * H, 0.0f); for (int64_t k=0; k x_init((size_t) AH * AD); - if (in.noise) - std::memcpy(x_init.data(), in.noise, x_init.size()*sizeof(float)); - else { - std::mt19937 rng((uint32_t) std::chrono::steady_clock::now().time_since_epoch().count()); - std::normal_distribution nd(0.f, 1.f); - for (auto & v : x_init) - v = nd(rng); - } + std::vector x_init; + init_noise(in, (size_t) AH*AD, x_init); // On by default: 16% faster, bit-identical. Set VLA_GR00T_GRAPH_CACHE=0 to opt out. // Dumping adds graph outputs, so it always rebuilds. @@ -505,7 +308,8 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { ggml_tensor * eagle = nullptr, * vl_embs = nullptr; std::vector lm_h_dump, vlsa_dump; const MainKey mkey{ SEQ, n_img, SEQ_TXT, num_steps, inject_deepstack }; - const bool built = mg.ensure(backend, mkey, (size_t) 256*1024*1024, + const size_t main_nodes = 65536 + (size_t) num_steps*64*(dit_layers+1); + const bool built = mg.ensure(backend, mkey, main_nodes*ggml_tensor_overhead() + ggml_graph_overhead_custom(main_nodes, false), [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { ggml_tensor * t_embeds = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, SEQ); ggml_set_input(t_embeds); ggml_tensor * t_pos = ggml_new_tensor_1d(C, GGML_TYPE_I32, 4*SEQ); ggml_set_input(t_pos); @@ -523,11 +327,6 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { ? ggml_new_tensor_1d(C, GGML_TYPE_I32, SEQ_TXT) : nullptr; if (t_txt_idx) ggml_set_input(t_txt_idx); - std::vector t_tau(num_steps), t_tproj(num_steps); - for (int64_t s=0; s Gr00tN1d7ModelArch::predict(const Inputs& in) { } eagle = h; - ggml_set_name(eagle, "eagle"); ggml_set_output(eagle); + ggml_set_name(eagle, "eagle"); - vl_embs = ggml_add(C, ggml_mul(C, ggml_norm(C, eagle, vlln_eps), vlln_w), vlln_b); + vl_embs = layer_norm(C, eagle, vlln_w, vlln_b, vlln_eps); if (do_dump) { ggml_set_output(vl_embs); vlsa_dump.push_back(vl_embs); @@ -555,57 +354,20 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { vlsa_dump.push_back(vl_embs); } } - ggml_set_name(vl_embs, "vl_embs"); ggml_set_output(vl_embs); + ggml_set_name(vl_embs, "vl_embs"); ggml_tensor * vl_img = ggml_get_rows(C, vl_embs, t_img_idx); ggml_tensor * vl_txt = (t_txt_idx ? ggml_get_rows(C, vl_embs, t_txt_idx) : vl_img); - ggml_tensor * state_features = cat_linear(C, aex.se_l2W, aex.se_l2b, aex.embodiment_id, ggml_relu(C, cat_linear(C, aex.se_l1W, aex.se_l1b, aex.embodiment_id, t_state))); - - const float dt = 1.0f/(float) num_steps; - const int64_t every2 = 2*attend_text_every_n; - - std::vector Kc(dit_layers, nullptr), Vc(dit_layers, nullptr); - for (int64_t i=0; inb[1], 0)); - ggml_tensor * sa = ggml_concat(C, state_features, af, 1); - ggml_tensor * hh = sa; - for (int64_t i=0; inb[1], (size_t) (Nsa-AH)*pred->nb[1])); - actions = ggml_add(C, actions, ggml_scale(C, vel, dt)); - } + ggml_tensor * actions = aex.denoise(C, dit, times, dit_interleave != 0, 2*attend_text_every_n, t_state, nullptr, + vl_txt, vl_img, t_x0); ggml_set_name(actions, "action_pred"); ggml_set_output(actions); gio.t_embeds=t_embeds; gio.t_pos=t_pos; gio.t_lmmask=t_lmmask; gio.t_state=t_state; gio.t_x0=t_x0; gio.t_ds[0]=t_ds[0]; gio.t_ds[1]=t_ds[1]; gio.t_ds[2]=t_ds[2]; - gio.t_img_idx=t_img_idx; gio.t_txt_idx=t_txt_idx; gio.t_tau=t_tau; gio.t_tproj=t_tproj; gio.actions=actions; + gio.t_img_idx=t_img_idx; gio.t_txt_idx=t_txt_idx; gio.actions=actions; - ggml_cgraph * gf = ggml_new_graph_custom(C, 65536, false); + ggml_cgraph * gf = ggml_new_graph_custom(C, main_nodes, false); ggml_build_forward_expand(gf, actions); return gf; }); @@ -613,92 +375,25 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { MainIO & gio = mg.io(); ggml_cgraph * gf = mg.graph(); - ggml_tensor * t_embeds = gio.t_embeds, * t_pos = gio.t_pos, * t_lmmask = gio.t_lmmask, * t_state = gio.t_state, * t_x0 = gio.t_x0; - ggml_tensor * t_ds[3] = { gio.t_ds[0], gio.t_ds[1], gio.t_ds[2] }; - ggml_tensor * t_img_idx = gio.t_img_idx, * t_txt_idx = gio.t_txt_idx, * actions = gio.actions; - std::vector & t_tau = gio.t_tau; std::vector & t_tproj = gio.t_tproj; - - ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); - { - const int64_t llm_grid_h = image_target_size/patch_size/spatial_merge; - const int64_t llm_grid_w = llm_grid_h; - - std::vector pp((size_t) 4*SEQ, 0); - int64_t st = 0, st_idx = 0; - while (st < SEQ) { - int64_t img_start = -1; - for (int64_t i=st; i max_image_pos) - max_image_pos = llm_grid_h-1; - if (llm_grid_w-1 > max_image_pos) - max_image_pos = llm_grid_w-1; - st_idx = image_offset+max_image_pos+1; - st = img_end; - } - - std::memcpy(pp.data()+(size_t) 3*SEQ, pp.data()+(size_t) 0*SEQ, (size_t) SEQ * sizeof(int32_t)); - ggml_backend_tensor_set(t_pos, pp.data(), 0, ggml_nbytes(t_pos)); - } + ggml_backend_tensor_set(gio.t_embeds, inputs_embeds.data(), 0, ggml_nbytes(gio.t_embeds)); + ggml_backend_tensor_set(gio.t_pos, pp.data(), 0, ggml_nbytes(gio.t_pos)); if (c_mask_seq != SEQ) { build_causal_mask(SEQ, c_mask); c_mask_seq = SEQ; } - ggml_backend_tensor_set(t_lmmask, c_mask.data(), 0, ggml_nbytes(t_lmmask)); + ggml_backend_tensor_set(gio.t_lmmask, c_mask.data(), 0, ggml_nbytes(gio.t_lmmask)); { std::vector st(max_state_dim, 0.0f); for (int64_t i=0; i Gr00tN1d7ModelArch::predict(const Inputs& in) { stats.ms_inference = std::chrono::duration(tc1-tc0).count(); std::vector out((size_t) AH * AD); - ggml_backend_tensor_get(actions, out.data(), 0, out.size()*sizeof(float)); + ggml_backend_tensor_get(gio.actions, out.data(), 0, out.size()*sizeof(float)); if (const char * dump = std::getenv("VLA_GR00T_N17_DUMP")) { auto dump_t = [&](const char * name, ggml_tensor * t) { diff --git a/src/models/octo.cpp b/src/models/octo.cpp index 0bd5bc3..fca0b31 100644 --- a/src/models/octo.cpp +++ b/src/models/octo.cpp @@ -15,9 +15,12 @@ #include "arch.h" #include "backend.h" #include "gguf_reader.h" +#include "layers/attn.h" +#include "layers/ffn.h" +#include "layers/linear.h" +#include "layers/norm.h" #include "loader.h" #include "model.h" -#include "models/octo.h" #include "modules/preprocess.h" #include "scratch_ctx.h" @@ -26,7 +29,6 @@ #include "gguf.h" #include "nlohmann/json.hpp" -#include "sentencepiece_processor.h" #include #include @@ -63,8 +65,8 @@ constexpr uint32_t kReplaySeed = 20260921u; struct OctoRuntime { ggml_backend_t backend = nullptr; - ggml_context * ctx_w = nullptr; std::unordered_map by_name; + bool pytorch_ckpt = false; // One cache per camera: the views differ in side and token count, so they // cannot share a graph. @@ -125,10 +127,9 @@ struct OctoRuntime { graph_cache language; struct BtKey { - int seq = -1; - int n_readout = -1; + std::vector runs; bool operator==(const BtKey& o) const { - return seq == o.seq && n_readout == o.n_readout; + return runs == o.runs; } }; struct BtIO { @@ -178,6 +179,7 @@ struct OctoRuntime { std::vector lang_pos; std::vector lang_repeated; int lang_steps = -1; + std::vector readout_pos; struct ActionStats { std::vector mean, stdv; @@ -194,7 +196,6 @@ struct OctoRuntime { void init(ggml_backend_t b, ggml_context * w) { backend = b; - ctx_w = w; // ggml_get_tensor is a linear strcmp scan, and building the stage graphs // looks up a few hundred weights by name. by_name.clear(); @@ -248,7 +249,6 @@ struct OctoModelArch : public ModelArchBase { if (backend) ggml_backend_free(backend); } - std::string gguf_path; ggml_backend_t backend = nullptr; ggml_context * ctx_weights = nullptr; ggml_backend_buffer_t weight_buf = nullptr; @@ -372,6 +372,7 @@ bool load_config(const gguf_reader& g, OctoModelArch& m) { m.max_action = scalar_key(g, "octo.diffusion.max_action"); m.pad_id = g.has("octo.tokenizer.pad_id") ? (int32_t) g.u32("octo.tokenizer.pad_id") : 0; m.head_type = g.has("octo.action.head_type") ? g.str("octo.action.head_type") : "diffusion"; + m.rt.pytorch_ckpt = g.str("octo.ckpt_format") == "pytorch"; detect_proprio(g, m.has_proprio, m.proprio_in_dim); if (m.has_proprio && m.proprio_in_dim != 1 && m.proprio_in_dim != 256) { @@ -443,7 +444,7 @@ bool is_stem_conv_weight(const char * name) { std::strstr(name, ".conv.weight") != nullptr; } -void standardize_conv_weight(float * w, int64_t oc, int64_t n) { +void standardize_conv_weight(float * w, int64_t oc, int64_t n, bool pytorch) { for (int64_t o=0; one[3], t->ne[0]*t->ne[1]*t->ne[2]); + standardize_conv_weight(row.data(), t->ne[3], t->ne[0]*t->ne[1]*t->ne[2], m.rt.pytorch_ckpt); ggml_backend_tensor_set(t, row.data(), 0, ggml_nbytes(t)); } return true; @@ -503,37 +505,46 @@ std::vector tensor_to_vec(const ggml_tensor * t) { return out; } +ggml_tensor * conv_2d_f32(ggml_context * C, ggml_tensor * w, ggml_tensor * x, int stride, int pad) { + ggml_tensor * col = ggml_im2col(C, w, x, stride, stride, pad, pad, 1, 1, true, GGML_TYPE_F32); + ggml_tensor * y = ggml_mul_mat(C, + ggml_reshape_2d(C, col, col->ne[0], col->ne[3]*col->ne[2]*col->ne[1]), + ggml_reshape_2d(C, w, w->ne[0]*w->ne[1]*w->ne[2], w->ne[3])); + y = ggml_reshape_4d(C, y, col->ne[1], col->ne[2], col->ne[3], w->ne[3]); + return ggml_cont(C, ggml_permute(C, y, 0, 1, 3, 2)); +} + // SmallStem16 for one camera view: four standardized-conv + GroupNorm + ReLU // stages at stride 2, a 1x1 patch embedding, a projection to the model width, // and the per-timestep position embedding. // -// `obs` holds the normalized CHW frames for `steps`, `task` the single goal -// frame, which is concatenated onto every one of them as channels 3..5. +// `frame` is the normalized CHW frame at every one of the timesteps in +// `pos_rows`, and the goal frame concatenated onto each as channels 3..5 is the +// constant -1. bool run_obs_tokenizer_graph(OctoRuntime& rt, graph_cache& cache, const char * view, - const std::vector& obs, - const std::vector& task, + const std::vector& frame, int side, int n_tok, - int steps, const std::vector& pos_rows, std::vector& pos) { - const size_t frame = (size_t) 3*side*side; - if (obs.size() != frame*(size_t) steps || task.size() != frame) { + if (frame.size() != (size_t) 3*side*side) { std::fprintf(stderr, "vla(octo): unexpected input image shape for side=%d\n", side); return false; } - if ((int) pos_rows.size() != steps) - return false; + const int steps = (int) pos_rows.size(); pos.resize((size_t) steps*n_tok*kHidden); const OctoRuntime::ObsKey key{side, n_tok, steps}; + bool fresh = false; const bool built = cache.ensure(rt.backend, key, (size_t) 32*1024*1024, [&](ggml_context * C, OctoRuntime::ObsIO& io) -> ggml_cgraph * { + fresh = true; ggml_tensor * x = ggml_new_tensor_4d(C, GGML_TYPE_F32, side, side, 6, steps); ggml_set_name(x, "octo.obs.input_norm"); ggml_set_input(x); + ggml_set_output(x); io.input = x; char rname[160]; @@ -556,9 +567,9 @@ bool run_obs_tokenizer_graph(OctoRuntime& rt, if (!cw || !cb || !gw || !gb) return nullptr; - x = ggml_conv_2d(C, cw, x, 2, 2, 1, 1, 1, 1); + x = conv_2d_f32(C, cw, x, 2, 1); x = ggml_add(C, x, cb); - x = ggml_group_norm(C, x, 32, 1e-5f); + x = ggml_group_norm(C, x, 32, rt.pytorch_ckpt ? 1e-5f : 1e-6f); x = ggml_add(C, ggml_mul(C, x, gw), gb); x = ggml_relu(C, x); } @@ -571,7 +582,7 @@ bool run_obs_tokenizer_graph(OctoRuntime& rt, if (!pw || !pb || !jw || !jb || !pos_r) return nullptr; - ggml_tensor * patch = ggml_add(C, ggml_conv_2d(C, pw, x, 1, 1, 0, 0, 1, 1), pb); + ggml_tensor * patch = ggml_add(C, conv_2d_f32(C, pw, x, 1, 0), pb); ggml_tensor * tok = ggml_cont(C, ggml_reshape_3d(C, ggml_cont(C, ggml_permute(C, patch, 1, 2, 0, 3)), kPatchEmbed, n_tok, steps)); @@ -585,7 +596,7 @@ bool run_obs_tokenizer_graph(OctoRuntime& rt, ggml_get_rows(C, ggml_reshape_2d(C, pos_r, kHidden*n_tok, pos_r->ne[2]), rows), kHidden, n_tok, steps); - ggml_tensor * out = ggml_add(C, ggml_add(C, ggml_mul_mat(C, jw, tok), jb), pe); + ggml_tensor * out = ggml_add(C, linear(C, jw, jb, tok), pe); ggml_set_name(out, "obs.tokenizer.pos"); ggml_set_output(out); io.pos = out; @@ -600,13 +611,13 @@ bool run_obs_tokenizer_graph(OctoRuntime& rt, } OctoRuntime::ObsIO& io = cache.io(); - std::vector input((size_t) steps*2*frame); - for (int t=0; t task((size_t) ggml_nelements(io.input), -1.0f); + ggml_backend_tensor_set(io.input, task.data(), 0, ggml_nbytes(io.input)); } - ggml_backend_tensor_set(io.input, input.data(), 0, ggml_nbytes(io.input)); + const size_t n = frame.size()*sizeof(float); + for (int t=0; t thresholds = tensor_to_vec(thresholds_r); - const int n_thresh = (int) thresholds.size(); + if (thresholds.size() != (size_t) in_dim+1) { + std::fprintf(stderr, "vla(octo): octo.obs.proprio.bin_thresholds has %zu entries, expected %d\n", + thresholds.size(), in_dim+1); + return false; + } + const float lo = thresholds.front(); + const float hi = thresholds.back(); + const bool uniform = std::fabs((thresholds[1]-lo)-(hi-lo)/in_dim) < 1e-3f*(hi-lo)/in_dim; for (int t=0; t= in_dim) + bucket = uniform && v >= hi ? in_dim-1 : 0; tokens_in[((size_t) t*n_dims+d)*in_dim+bucket] = 1.0f; } } @@ -683,7 +699,7 @@ bool run_proprio_tokenizer_graph(OctoRuntime& rt, ggml_tensor * pe = ggml_reshape_3d(C, ggml_get_rows(C, ggml_reshape_2d(C, pos_r, kHidden*n_dims, pos_r->ne[2]), rows), kHidden, n_dims, steps); - ggml_tensor * out = ggml_add(C, ggml_add(C, ggml_mul_mat(C, proj_w, x), proj_b), pe); + ggml_tensor * out = ggml_add(C, linear(C, proj_w, proj_b, x), pe); ggml_set_name(out, "obs.proprio.pos"); ggml_set_output(out); io.pos = out; @@ -733,7 +749,7 @@ bool run_language_graph(OctoRuntime& rt, ggml_set_input(in); io.in = in; - ggml_tensor * pos_t = ggml_add(C, ggml_add(C, ggml_mul_mat(C, jw, in), jb), pe); + ggml_tensor * pos_t = ggml_add(C, linear(C, jw, jb, in), pe); ggml_set_name(pos_t, "task_language.pos"); ggml_set_output(pos_t); io.pos = pos_t; @@ -798,6 +814,16 @@ bool run_t5_encoder_graph(OctoRuntime& rt, std::fprintf(stderr, "vla(octo): T5 encoder expected %d input_ids/attention_mask\n", seq); return false; } + const ggml_tensor * te = rt.weight("octo.t5.tok_embd.weight"); + if (!te) + return false; + for (int i=0; i= te->ne[1]) { + std::fprintf(stderr, "vla(octo): lang_tokens[%d]=%d out of vocab range [0, %lld)\n", + i, input_ids[(size_t) i], (long long) te->ne[1]); + return false; + } + } std::vector bucket_idx((size_t) seq*seq); std::vector padmask((size_t) seq*seq); @@ -853,28 +879,19 @@ bool run_t5_encoder_graph(OctoRuntime& rt, ggml_tensor * mask = ggml_add(C, pos_bias, padmask_t); for (int i=0; i<12; ++i) { - ggml_tensor * n1 = ggml_mul(C, ggml_rms_norm(C, x, ln_eps), blk_w[i][0]); - ggml_tensor * Q = ggml_mul_mat(C, blk_w[i][1], n1); - ggml_tensor * K = ggml_mul_mat(C, blk_w[i][2], n1); - ggml_tensor * V = ggml_mul_mat(C, blk_w[i][3], n1); - ggml_tensor * Qh = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, Q, head_dim, heads, seq), 0, 2, 1, 3)); - ggml_tensor * Kh = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, K, head_dim, heads, seq), 0, 2, 1, 3)); - ggml_tensor * Vh = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, V, head_dim, heads, seq), 1, 2, 0, 3)); - - ggml_tensor * scores = ggml_mul_mat(C, Kh, Qh); - ggml_mul_mat_set_prec(scores, GGML_PREC_F32); + ggml_tensor * n1 = rms_norm(C, x, blk_w[i][0], ln_eps); + ggml_tensor * Qh = to_heads(C, ggml_mul_mat(C, blk_w[i][1], n1), head_dim, heads, seq); + ggml_tensor * Kh = to_heads(C, ggml_mul_mat(C, blk_w[i][2], n1), head_dim, heads, seq); + ggml_tensor * Vh = to_heads_v(C, ggml_mul_mat(C, blk_w[i][3], n1), head_dim, heads, seq); // T5 folds 1/sqrt(d_k) into the weights, so the scale here is 1. - ggml_tensor * probs = ggml_soft_max_ext(C, scores, mask, 1.0f, 0.0f); - ggml_tensor * attended = ggml_mul_mat(C, Vh, probs); - ggml_tensor * merged = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, attended, 0, 2, 1, 3)), hidden, seq); + ggml_tensor * merged = attention(C, Qh, Kh, Vh, mask, 1.0f, hidden, seq); x = ggml_add(C, x, ggml_mul_mat(C, blk_w[i][4], merged)); - ggml_tensor * n2 = ggml_mul(C, ggml_rms_norm(C, x, ln_eps), blk_w[i][5]); - ggml_tensor * h = ggml_relu(C, ggml_mul_mat(C, blk_w[i][6], n2)); - x = ggml_add(C, x, ggml_mul_mat(C, blk_w[i][7], h)); + ggml_tensor * n2 = rms_norm(C, x, blk_w[i][5], ln_eps); + x = ggml_add(C, x, ffn_relu(C, blk_w[i][6], nullptr, blk_w[i][7], nullptr, n2)); } - ggml_tensor * out = ggml_mul(C, ggml_rms_norm(C, x, ln_eps), outw); + ggml_tensor * out = rms_norm(C, x, outw, ln_eps); ggml_set_name(out, "t5.out"); ggml_set_output(out); io.out = out; @@ -1050,12 +1067,18 @@ void build_transformer_mask(const OctoSeqLayout& layout, std::vector& mas } } +ggml_tensor * octo_gelu(ggml_context * C, ggml_tensor * x, bool erf) { + if (erf) + return ggml_gelu_erf(C, x); + ggml_tensor * z = ggml_add(C, x, ggml_scale(C, ggml_mul(C, ggml_mul(C, x, x), x), 0.044715f)); + return ggml_mul(C, x, ggml_sigmoid(C, ggml_scale(C, z, 2.0f*0.7978845608028654f))); +} + // 12 pre-norm encoder blocks over the assembled sequence. Only the readout rows // are gathered back out; everything else the blocks compute is intermediate. bool run_transformer_graph(OctoRuntime& rt, const OctoSeqLayout& layout, const std::vector& input, - const std::vector& blocked_mask, std::vector& readout_action) { constexpr int heads = 6; constexpr int head_dim = 64; @@ -1063,12 +1086,16 @@ bool run_transformer_graph(OctoRuntime& rt, constexpr float attn_scale = 0.125f; const int seq = layout.seq; const int n_readout = (int) layout.readout_seq_idx.size(); - if (input.size() != (size_t) kHidden*seq || blocked_mask.size() != (size_t) seq*seq) + if (input.size() != (size_t) kHidden*seq) return false; - const OctoRuntime::BtKey key{seq, n_readout}; + OctoRuntime::BtKey key; + for (const OctoSeqRun& r : layout.runs) + key.runs.insert(key.runs.end(), {(int32_t) r.group, r.timestep, r.n_tokens, r.key_valid}); + bool fresh = false; const bool built = rt.transformer.ensure(rt.backend, key, (size_t) 32*1024*1024, [&](ggml_context * C, OctoRuntime::BtIO& io) -> ggml_cgraph * { + fresh = true; char rname[160]; ggml_tensor * blk_w[12][12]; const char * leaves[12] = {"attn_norm.weight", "attn_norm.bias", "attn_qkv.weight", "attn_qkv.bias", @@ -1095,39 +1122,31 @@ bool run_transformer_graph(OctoRuntime& rt, ggml_tensor * mask = ggml_new_tensor_2d(C, GGML_TYPE_F32, seq, seq); ggml_set_name(mask, "octo.block_transformer.additive_mask"); ggml_set_input(mask); + ggml_set_output(mask); io.mask = mask; ggml_tensor * readout_idx = ggml_new_tensor_1d(C, GGML_TYPE_I32, n_readout); ggml_set_name(readout_idx, "octo.block_transformer.readout_idx"); ggml_set_input(readout_idx); + ggml_set_output(readout_idx); io.readout_idx = readout_idx; for (int i=0; i<12; ++i) { - ggml_tensor * n1 = ggml_add(C, ggml_mul(C, ggml_norm(C, x, ln_eps), blk_w[i][0]), blk_w[i][1]); - ggml_tensor * qkv = ggml_add(C, ggml_mul_mat(C, blk_w[i][2], n1), blk_w[i][3]); - ggml_tensor * q = ggml_cont(C, ggml_view_2d(C, qkv, kHidden, seq, qkv->nb[1], 0)); - ggml_tensor * k = ggml_cont(C, ggml_view_2d(C, qkv, kHidden, seq, qkv->nb[1], (size_t) kHidden*qkv->nb[0])); - ggml_tensor * v = ggml_cont(C, ggml_view_2d(C, qkv, kHidden, seq, qkv->nb[1], (size_t) 2*kHidden*qkv->nb[0])); - ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, head_dim, heads, seq), 0, 2, 1, 3)); - ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, head_dim, heads, seq), 0, 2, 1, 3)); - ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); - - ggml_tensor * scores = ggml_mul_mat(C, K, Q); - ggml_mul_mat_set_prec(scores, GGML_PREC_F32); - ggml_tensor * probs = ggml_soft_max_ext(C, scores, mask, attn_scale, 0.0f); - ggml_tensor * attended = ggml_mul_mat(C, V, probs); - ggml_tensor * merged = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, attended, 0, 2, 1, 3)), kHidden, seq); - ggml_tensor * attn_out = ggml_add(C, ggml_mul_mat(C, blk_w[i][4], merged), blk_w[i][5]); - ggml_tensor * residual = ggml_add(C, x, attn_out); - - ggml_tensor * n2 = ggml_add(C, ggml_mul(C, ggml_norm(C, residual, ln_eps), blk_w[i][6]), blk_w[i][7]); - ggml_tensor * mlp = ggml_add(C, ggml_mul_mat(C, blk_w[i][8], n2), blk_w[i][9]); - mlp = ggml_gelu_erf(C, mlp); - mlp = ggml_add(C, ggml_mul_mat(C, blk_w[i][10], mlp), blk_w[i][11]); - x = ggml_add(C, residual, mlp); + ggml_tensor * n1 = layer_norm(C, x, blk_w[i][0], blk_w[i][1], ln_eps); + ggml_tensor * qkv = linear(C, blk_w[i][2], blk_w[i][3], n1); + ggml_tensor * Q = ggml_cont(C, ggml_permute(C, head_view(C, qkv, head_dim, heads, seq, kHidden, 3, 0), 0, 2, 1, 3)); + ggml_tensor * K = ggml_cont(C, ggml_permute(C, head_view(C, qkv, head_dim, heads, seq, kHidden, 3, 1), 0, 2, 1, 3)); + ggml_tensor * V = ggml_cont(C, ggml_permute(C, head_view(C, qkv, head_dim, heads, seq, kHidden, 3, 2), 1, 2, 0, 3)); + + ggml_tensor * merged = attention(C, Q, K, V, mask, attn_scale, kHidden, seq); + ggml_tensor * residual = ggml_add(C, x, linear(C, blk_w[i][4], blk_w[i][5], merged)); + + ggml_tensor * n2 = layer_norm(C, residual, blk_w[i][6], blk_w[i][7], ln_eps); + ggml_tensor * mlp = octo_gelu(C, linear(C, blk_w[i][8], blk_w[i][9], n2), rt.pytorch_ckpt); + x = ggml_add(C, residual, linear(C, blk_w[i][10], blk_w[i][11], mlp)); } - ggml_tensor * output = ggml_add(C, ggml_mul(C, ggml_norm(C, x, ln_eps), out_w), out_b); + ggml_tensor * output = layer_norm(C, x, out_w, out_b, ln_eps); // The readouts are not evenly spaced once padded timesteps drop their // observation groups, so they are gathered rather than strided. ggml_tensor * readout = ggml_get_rows(C, output, readout_idx); @@ -1145,9 +1164,13 @@ bool run_transformer_graph(OctoRuntime& rt, } OctoRuntime::BtIO& io = rt.transformer.io(); - ggml_backend_tensor_set(io.input, input.data(), 0, ggml_nbytes(io.input)); - ggml_backend_tensor_set(io.mask, blocked_mask.data(), 0, ggml_nbytes(io.mask)); - ggml_backend_tensor_set(io.readout_idx, layout.readout_seq_idx.data(), 0, ggml_nbytes(io.readout_idx)); + if (fresh) { + std::vector mask; + build_transformer_mask(layout, mask); + ggml_backend_tensor_set(io.mask, mask.data(), 0, ggml_nbytes(io.mask)); + ggml_backend_tensor_set(io.readout_idx, layout.readout_seq_idx.data(), 0, ggml_nbytes(io.readout_idx)); + } + ggml_backend_tensor_set(io.input, input.data(), 0, ggml_nbytes(io.input)); if (!octo_compute(rt, rt.transformer.graph(), "block transformer")) return false; readout_action.resize((size_t) kHidden*n_readout); @@ -1245,19 +1268,16 @@ ggml_tensor * build_score_actor(ggml_context * ctx, ggml_tensor * f = ggml_scale(ctx, ggml_mul_mat(ctx, w.time_w, time), two_pi); ggml_tensor * time_ff = ggml_concat(ctx, ggml_cos(ctx, f), ggml_sin(ctx, f), 0); - ggml_tensor * cond = ggml_silu(ctx, ggml_add(ctx, ggml_mul_mat(ctx, w.c0w, time_ff), w.c0b)); - cond = ggml_add(ctx, ggml_mul_mat(ctx, w.c1w, cond), w.c1b); + ggml_tensor * cond = linear(ctx, w.c1w, w.c1b, ggml_silu(ctx, linear(ctx, w.c0w, w.c0b, time_ff))); ggml_tensor * reverse_input = ggml_concat(ctx, ggml_concat(ctx, cond, obs, 0), actions, 0); - ggml_tensor * x = ggml_add(ctx, ggml_mul_mat(ctx, w.rinw, reverse_input), w.rinb); + ggml_tensor * x = linear(ctx, w.rinw, w.rinb, reverse_input); for (int i=0; i<3; ++i) { - ggml_tensor * residual = x; - ggml_tensor * h = ggml_add(ctx, ggml_mul(ctx, ggml_norm(ctx, x, ln_eps), w.blk[i][0]), w.blk[i][1]); - h = ggml_silu(ctx, ggml_add(ctx, ggml_mul_mat(ctx, w.blk[i][2], h), w.blk[i][3])); - h = ggml_add(ctx, ggml_mul_mat(ctx, w.blk[i][4], h), w.blk[i][5]); - x = ggml_add(ctx, residual, h); + ggml_tensor * h = layer_norm(ctx, x, w.blk[i][0], w.blk[i][1], ln_eps); + h = ggml_silu(ctx, linear(ctx, w.blk[i][2], w.blk[i][3], h)); + x = ggml_add(ctx, x, linear(ctx, w.blk[i][4], w.blk[i][5], h)); } - return ggml_add(ctx, ggml_mul_mat(ctx, w.routw, ggml_silu(ctx, x)), w.routb); + return linear(ctx, w.routw, w.routb, ggml_silu(ctx, x)); } // The DDPM reverse process as ONE graph. The steps are sequentially dependent so @@ -1432,34 +1452,33 @@ bool run_l1_action_head_graph(OctoRuntime& rt, // in_proj_weight, so each needs its own mul_mat and drops the slices it // does not use -- nn.MultiheadAttention keeps the [Wq;Wk;Wv] row blocks // whatever is fed through it. - ggml_tensor * qkv_probe = ggml_add(C, ggml_mul_mat(C, qkv_w, probe), qkv_b); + ggml_tensor * qkv_probe = linear(C, qkv_w, qkv_b, probe); ggml_tensor * q = ggml_cont(C, ggml_view_2d(C, qkv_probe, kHidden, 1, qkv_probe->nb[1], 0)); - ggml_tensor * qkv_x = ggml_add(C, ggml_mul_mat(C, qkv_w, x), qkv_b); + ggml_tensor * qkv_x = linear(C, qkv_w, qkv_b, x); ggml_tensor * k = ggml_cont(C, ggml_view_2d(C, qkv_x, kHidden, width, qkv_x->nb[1], (size_t) kHidden*qkv_x->nb[0])); ggml_tensor * v = ggml_cont(C, ggml_view_2d(C, qkv_x, kHidden, width, qkv_x->nb[1], (size_t) 2*kHidden*qkv_x->nb[0])); // Heads on ne2 and window on ne3 are both batch axes mul_mat loops over, // never cross-multiplied, which keeps each timestep independent. - ggml_tensor * Qh = ggml_cont(C, ggml_permute(C, ggml_reshape_4d(C, q, map_head_dim, map_heads, 1, 1), 0, 2, 1, 3)); - ggml_tensor * Kh = ggml_cont(C, ggml_permute(C, ggml_reshape_4d(C, k, map_head_dim, map_heads, 1, width), 0, 2, 1, 3)); - ggml_tensor * Vh = ggml_cont(C, ggml_permute(C, ggml_reshape_4d(C, v, map_head_dim, map_heads, 1, width), 1, 2, 0, 3)); + ggml_tensor * Qh = to_heads(C, q, map_head_dim, map_heads, 1); + ggml_tensor * Kh = to_heads(C, k, map_head_dim, map_heads, 1, width); + ggml_tensor * Vh = to_heads_v(C, v, map_head_dim, map_heads, 1, width); ggml_tensor * scores = ggml_mul_mat(C, Qh, Kh); - ggml_mul_mat_set_prec(scores, GGML_PREC_F32); + ggml_prec_set_acc(scores, GGML_PREC_F32); // One readout token per timestep, so this softmax is over a single logit // and always yields 1.0. Kept as the real op in case that changes. ggml_tensor * probs = ggml_soft_max_ext(C, scores, nullptr, 1.0f/std::sqrt((float) map_head_dim), 0.0f); ggml_tensor * attended = ggml_mul_mat(C, Vh, probs); ggml_tensor * merged = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, attended, 0, 2, 1, 3)), kHidden, width); - ggml_tensor * attn_out = ggml_add(C, ggml_mul_mat(C, o_w, merged), o_b); - ggml_tensor * y = ggml_add(C, ggml_mul(C, ggml_norm(C, attn_out, ln_eps), norm_w), norm_b); - ggml_tensor * h = ggml_gelu_erf(C, ggml_add(C, ggml_mul_mat(C, ffn_up_w, y), ffn_up_b)); - h = ggml_add(C, ggml_mul_mat(C, ffn_down_w, h), ffn_down_b); + ggml_tensor * attn_out = linear(C, o_w, o_b, merged); + ggml_tensor * y = layer_norm(C, attn_out, norm_w, norm_b, ln_eps); // The residual is onto attn_out, before the norm, as in MAPHead. - ggml_tensor * emb = ggml_add(C, attn_out, h); + ggml_tensor * h = octo_gelu(C, linear(C, ffn_up_w, ffn_up_b, y), rt.pytorch_ckpt); + ggml_tensor * emb = ggml_add(C, attn_out, linear(C, ffn_down_w, ffn_down_b, h)); - ggml_tensor * mean_raw = ggml_add(C, ggml_mul_mat(C, mean_w, emb), mean_b); + ggml_tensor * mean_raw = linear(C, mean_w, mean_b, emb); ggml_tensor * out = ggml_scale(C, ggml_tanh(C, ggml_scale(C, mean_raw, 1.0f/max_action)), max_action); ggml_set_name(out, "l1_head.mean_normalized"); ggml_set_output(out); @@ -1493,22 +1512,6 @@ bool run_l1_action_head_graph(OctoRuntime& rt, return true; } -bool read_kv_u8_array(const gguf_reader& g, const char * key, std::vector& out) { - const int64_t id = gguf_find_key(g.gctx, key); - if (id < 0) { - std::fprintf(stderr, "vla(octo): missing metadata %s\n", key); - return false; - } - if (gguf_get_kv_type(g.gctx, id) != GGUF_TYPE_ARRAY || gguf_get_arr_type(g.gctx, id) != GGUF_TYPE_UINT8) { - std::fprintf(stderr, "vla(octo): %s is not a UINT8 array\n", key); - return false; - } - const size_t n = gguf_get_arr_n(g.gctx, id); - const uint8_t * data = (const uint8_t *) gguf_get_arr_data(g.gctx, id); - out.assign(data, data+n); - return true; -} - // Which top-level key of octo.dataset_statistics to un-normalize against, when // the caller did not pin one down. In order: VLA_OCTO_UNNORM_DATASET, the sole // key if there is only one, then bridge_dataset, which is what the pretrain @@ -1557,7 +1560,8 @@ bool resolve_stats_block(const nlohmann::json& j, const std::string& dataset_key // Parses octo.dataset_statistics once and caches the blocks on `rt`: it is a // JSON blob in the metadata, ~25 datasets wide for the pretrain checkpoint, and // re-reading 21 floats out of it per request is pure overhead. -bool ensure_stats(OctoRuntime& rt, gguf_reader& g, const std::string& dataset_key_in, int64_t action_dim) { +bool ensure_stats(OctoRuntime& rt, gguf_reader& g, const std::string& dataset_key_in, int64_t action_dim, + bool need_proprio) { if (rt.stats_loaded && rt.stats_key == dataset_key_in) return true; @@ -1581,27 +1585,35 @@ bool ensure_stats(OctoRuntime& rt, gguf_reader& g, const std::string& dataset_ke const auto& act = (*block)["action"]; OctoRuntime::ActionStats a; - a.mean = act.at("mean").get>(); - a.stdv = act.at("std").get>(); - for (bool b : act.at("mask").get>()) - a.mask.push_back(b ? 1 : 0); + OctoRuntime::ProprioStats pr; + bool has_pr = false; + try { + a.mean = act.at("mean").get>(); + a.stdv = act.at("std").get>(); + if (act.contains("mask")) { + for (bool b : act["mask"].get>()) + a.mask.push_back(b ? 1 : 0); + } else { + a.mask.assign(a.mean.size(), 1); + } + if (need_proprio && block->contains("proprio")) { + const auto& p = (*block)["proprio"]; + pr.mean = p.at("mean").get>(); + pr.stdv = p.at("std").get>(); + has_pr = true; + } + } catch (const nlohmann::json::exception& e) { + std::fprintf(stderr, "vla(octo): bad dataset_statistics: %s\n", e.what()); + return false; + } if (a.mask.size() != (size_t) action_dim || a.mean.size() != (size_t) action_dim || a.stdv.size() != (size_t) action_dim) { std::fprintf(stderr, "vla(octo): dataset_statistics/action is not %lld-dim\n", (long long) action_dim); return false; } - - OctoRuntime::ProprioStats pr; - bool has_pr = false; - if (block->contains("proprio")) { - const auto& p = (*block)["proprio"]; - pr.mean = p.at("mean").get>(); - pr.stdv = p.at("std").get>(); - if (pr.mean.size() != (size_t) kProprioTokens || pr.stdv.size() != (size_t) kProprioTokens) { - std::fprintf(stderr, "vla(octo): dataset_statistics/proprio is not %d-dim\n", kProprioTokens); - return false; - } - has_pr = true; + if (has_pr && (pr.mean.size() != (size_t) kProprioTokens || pr.stdv.size() != (size_t) kProprioTokens)) { + std::fprintf(stderr, "vla(octo): dataset_statistics/proprio is not %d-dim\n", kProprioTokens); + return false; } rt.action_stats = std::move(a); @@ -1690,7 +1702,7 @@ bool run_pipeline(OctoModelArch& m, const OctoFrame& f, const int window_size = (int) m.window_size; const int action_total = (int) (m.action_horizon*m.action_dim); const int n_proprio = m.has_proprio ? kProprioTokens : 0; - if (!ensure_stats(rt, m.io, "", m.action_dim)) + if (!ensure_stats(rt, m.io, "", m.action_dim, m.has_proprio)) return false; // Cold start: history is filled with copies of the one live frame and every @@ -1709,29 +1721,13 @@ bool run_pipeline(OctoModelArch& m, const OctoFrame& f, // Every live window slot holds the same frame, so it is replicated rather // than re-decoded. Language-only conditioning means no goal image: the task // frame is a zero uint8 image, which normalizes to -1. - auto tokenize_view = [&](graph_cache& cache, - const char * view, const std::vector& frame, int side, int n_tok, - const std::vector& steps, std::vector& out) { - const size_t n = (size_t) 3*side*side; - if (frame.size() != n) { - std::fprintf(stderr, "vla(octo): %s frame is %zu floats, expected %zu\n", view, frame.size(), n); - return false; - } - std::vector obs(n*steps.size()); - for (size_t i=0; i task(n, -1.0f); - return run_obs_tokenizer_graph(rt, cache, view, obs, task, side, n_tok, (int) steps.size(), steps, out); - }; - if (!layout.primary_steps.empty() && - !tokenize_view(rt.obs_primary, "primary", f.primary, (int) m.primary_size, - (int) m.primary_tokens, layout.primary_steps, primary_pos)) + !run_obs_tokenizer_graph(rt, rt.obs_primary, "primary", f.primary, (int) m.primary_size, + (int) m.primary_tokens, layout.primary_steps, primary_pos)) return false; if (!layout.wrist_steps.empty() && - !tokenize_view(rt.obs_wrist, "wrist", f.wrist, (int) m.wrist_size, - (int) m.wrist_tokens, layout.wrist_steps, wrist_pos)) + !run_obs_tokenizer_graph(rt, rt.obs_wrist, "wrist", f.wrist, (int) m.wrist_size, + (int) m.wrist_tokens, layout.wrist_steps, wrist_pos)) return false; if (!layout.proprio_steps.empty()) { @@ -1747,7 +1743,7 @@ bool run_pipeline(OctoModelArch& m, const OctoFrame& f, for (int t=0; t readout_pos = tensor_to_vec(readout_pos_r); + if (rt.readout_pos.empty()) { + ggml_tensor * readout_pos_r = rt.weight("octo.readout.action.pos_embd"); + if (!readout_pos_r) + return false; + rt.readout_pos = tensor_to_vec(readout_pos_r); + } - std::vector input, mask; + std::vector input; if (!assemble_transformer_input(layout, rt.lang_pos, primary_pos, wrist_pos, proprio_pos, - rt.lang_repeated, readout_pos, input)) + rt.lang_repeated, rt.readout_pos, input)) return false; - build_transformer_mask(layout, mask); std::vector readout_action; - if (!run_transformer_graph(rt, layout, input, mask, readout_action)) + if (!run_transformer_graph(rt, layout, input, readout_action)) return false; std::vector normalized; @@ -1808,7 +1805,6 @@ std::unique_ptr octo_create(const std::string& mmproj_path, std::printf("vla(octo): note - mmproj '%s' is ignored (Octo ships one GGUF)\n", mmproj_path.c_str()); auto m = std::make_unique(); - m->gguf_path = ckpt_path; if (!m->io.open(ckpt_path)) return nullptr; @@ -1836,41 +1832,6 @@ std::unique_ptr octo_create(const std::string& mmproj_path, return m; } -bool octo_tokenize_text(const std::string& ckpt_path, - const std::string& text, - std::vector& input_ids, - std::vector& attention_mask) { - gguf_reader g{"octo"}; - if (!g.open(ckpt_path)) - return false; - - std::vector spm_bytes; - if (!read_kv_u8_array(g, "octo.tokenizer.spm_model", spm_bytes)) - return false; - const uint32_t eos_id = g.has("octo.tokenizer.eos_id") ? g.u32("octo.tokenizer.eos_id") : 1; - const uint32_t pad_id = g.has("octo.tokenizer.pad_id") ? g.u32("octo.tokenizer.pad_id") : 0; - const int64_t max_length = g.has("octo.tokens.language") ? g.u32("octo.tokens.language") : kTaskTokens; - - sentencepiece::SentencePieceProcessor sp; - const auto status = sp.LoadFromSerializedProto( - absl::string_view(reinterpret_cast(spm_bytes.data()), spm_bytes.size())); - if (!status.ok()) { - std::fprintf(stderr, "vla(octo): sentencepiece LoadFromSerializedProto failed: %s\n", - status.ToString().c_str()); - return false; - } - - std::vector ids = sp.EncodeAsIds(text); - if ((int64_t) ids.size() > max_length-1) - ids.resize((size_t) (max_length-1)); - input_ids.assign(ids.begin(), ids.end()); - input_ids.push_back((int32_t) eos_id); - attention_mask.assign(input_ids.size(), 1); - input_ids.resize((size_t) max_length, (int32_t) pad_id); - attention_mask.resize((size_t) max_length, 0); - return true; -} - // Unlike the other archs, this returns the action in world units rather than the // normalized one: Octo's dataset_statistics lives inside the multi-hundred-MB // checkpoint, not in a small sibling stats.json a client could hold, so diff --git a/src/models/openvla_oft.cpp b/src/models/openvla_oft.cpp index 735b1fe..b8c7a0f 100644 --- a/src/models/openvla_oft.cpp +++ b/src/models/openvla_oft.cpp @@ -19,21 +19,17 @@ #include "modules/dual_tower.h" #include "ggml.h" -#include "ggml-cpu.h" #include "ggml-backend.h" #include "backend.h" -#include "gguf.h" #include "gguf_reader.h" #include "scratch_ctx.h" -#include "env_flag.h" +#include "layers/linear.h" +#include "layers/norm.h" #include #include #include #include -#include -#include -#include #include #include #include @@ -41,63 +37,6 @@ namespace vla { namespace { -bool parse_stats(const std::string & js, int64_t want, std::vector & q01, - std::vector & q99, std::vector & mask, std::string & suite) { - auto find_key = [&](size_t from, const std::string & key) -> size_t { - const std::string pat = "\"" + key + "\""; - return js.find(pat, from); - }; - const char * env = std::getenv("VLA_OPENVLA_OFT_UNNORM_KEY"); - size_t suite_pos; - if (env) { - suite = env; - suite_pos = find_key(0, suite); - } - else { - size_t b = js.find('{'); size_t q = js.find('"', b); - size_t qe = js.find('"', q+1); - suite = js.substr(q+1, qe-q-1); suite_pos = q; - } - if (suite_pos == std::string::npos) { - std::fprintf(stderr, "vla(openvla_oft): suite '%s' not in stats\n", suite.c_str()); - return false; - } - size_t act = find_key(suite_pos, "action"); - if (act == std::string::npos) - return false; - auto read_arr = [&](const std::string & key, std::vector & out) -> bool { - size_t k = find_key(act, key); if (k == std::string::npos) return false; - size_t lb = js.find('[', k); size_t rb = js.find(']', lb); - if (lb == std::string::npos || rb == std::string::npos) - return false; - out.clear(); size_t p = lb+1; - while (p < rb) { - while (p < rb && (js[p] == ',' || js[p] == ' ' || js[p] == '\n' || js[p] == '\t' || js[p] == '\r')) - ++p; - if (p >= rb) - break; - bool t = (js.compare(p, 4, "true") == 0), f = (js.compare(p, 5, "false") == 0); - if (t || f) { - out.push_back(t ? 1.0f : 0.0f); - p += t ? 4 : 5; - } - else { - out.push_back(std::strtof(js.c_str()+p, nullptr)); - while (p < rb && js[p] != ',') - ++p; - } - } - return true; - }; - std::vector mk; - if (!read_arr("q01", q01) || !read_arr("q99", q99)) - return false; - if (!read_arr("mask", mk)) - mk.assign(want, 1.0f); - mask.assign(mk.size(), 1); for (size_t i=0; i vision_graph; struct MainKey { int64_t seq=-1, n_views=-1, n_lang=-1; @@ -127,18 +66,15 @@ struct OpenVlaOftModelArch : public ModelArchBase { }; struct MainIO { ggml_tensor *t_ids=nullptr,*t_state=nullptr,*t_proj=nullptr,*act0=nullptr,*t_pos=nullptr,*norm_actions=nullptr; + bool consts=false; }; graph_cache main_graph; ggml_backend_buffer_t weight_buf = nullptr; ggml_type mt = GGML_TYPE_BF16; - int64_t d_hidden=1024,d_layers=23,d_heads=16,d_head_dim=64,d_inter=4096; - int64_t s_hidden=1152,s_layers=26,s_heads=16,s_head_dim=72,s_inter=4304; - int64_t image_size=224,patch_size=14,n_patches=256,proj_mid=8704,vdim=2176,num_images=2; - float vit_ln_eps=1e-6f; - int64_t lm_hidden=4096,lm_layers=32,n_q=32,n_kv=32,lm_head_dim=128,lm_inter=11008,vocab=32064; + int64_t lm_hidden=4096,lm_layers=32,n_q=32,n_kv=32,lm_head_dim=128,vocab=32064; float lm_rope_base=1e4f, lm_rms_eps=1e-6f; - int64_t chunk=8,action_dim=7,proprio_dim=8,head_hidden=4096,head_blocks=2; + int64_t chunk=8,action_dim=7,proprio_dim=8,head_blocks=2; float head_ln_eps=1e-5f; int64_t stop_id=2; @@ -147,7 +83,7 @@ struct OpenVlaOftModelArch : public ModelArchBase { ggml_tensor *pp_fc1w,*pp_fc1b,*pp_fc2w,*pp_fc2b; ggml_tensor *h_ln1w,*h_ln1b,*h_fc1w,*h_fc1b,*h_ln2w,*h_ln2b,*h_fc2w,*h_fc2b; std::vector hblk; - std::vector q01,q99; std::vector unnorm_mask; std::string suite; + Q99Stats act_stats; std::vector predict(const Inputs& in) override; }; @@ -171,33 +107,32 @@ std::unique_ptr openvla_oft_create(const std::string& mmproj_path auto U=[&](const char*k,int64_t&d){ if(g.has(k)) d=(int64_t)g.u32(k); }; auto F=[&](const char*k,float&d){ if(g.has(k)) d=g.f32(k); }; - U("openvla_oft.vit.dino.hidden",m->d_hidden); U("openvla_oft.vit.dino.layers",m->d_layers); - U("openvla_oft.vit.dino.heads",m->d_heads); U("openvla_oft.vit.dino.head_dim",m->d_head_dim); U("openvla_oft.vit.dino.inter",m->d_inter); - U("openvla_oft.vit.sig.hidden",m->s_hidden); U("openvla_oft.vit.sig.layers",m->s_layers); - U("openvla_oft.vit.sig.heads",m->s_heads); U("openvla_oft.vit.sig.head_dim",m->s_head_dim); U("openvla_oft.vit.sig.inter",m->s_inter); - U("openvla_oft.vit.image_size",m->image_size); U("openvla_oft.vit.patch_size",m->patch_size); - U("openvla_oft.vit.n_patches",m->n_patches); F("openvla_oft.vit.ln_eps",m->vit_ln_eps); - U("openvla_oft.vit.proj_mid",m->proj_mid); U("openvla_oft.vit.vdim",m->vdim); U("openvla_oft.vit.num_images",m->num_images); + m->vis.read_config(g, "openvla_oft"); U("openvla_oft.lm.hidden",m->lm_hidden); U("openvla_oft.lm.layers",m->lm_layers); U("openvla_oft.lm.q_heads",m->n_q); U("openvla_oft.lm.kv_heads",m->n_kv); U("openvla_oft.lm.head_dim",m->lm_head_dim); - U("openvla_oft.lm.inter",m->lm_inter); U("openvla_oft.lm.vocab",m->vocab); + U("openvla_oft.lm.vocab",m->vocab); F("openvla_oft.lm.rope_theta",m->lm_rope_base); F("openvla_oft.lm.rms_eps",m->lm_rms_eps); U("openvla_oft.action.chunk",m->chunk); U("openvla_oft.action.action_dim",m->action_dim); - U("openvla_oft.action.proprio_dim",m->proprio_dim); U("openvla_oft.action.head_hidden",m->head_hidden); + U("openvla_oft.action.proprio_dim",m->proprio_dim); U("openvla_oft.action.head_blocks",m->head_blocks); F("openvla_oft.action.head_ln_eps",m->head_ln_eps); // No empty_id: the reference zeroes the action-slot embeddings instead // (modeling_prismatic.py:891), which is what act0 below does. U("openvla_oft.tokens.stop_id",m->stop_id); + if (m->vis.patch_size <= 0 || m->n_q <= 0) { + std::fprintf(stderr, "vla(openvla_oft): bad geometry (patch_size %lld q_heads %lld)\n", + (long long) m->vis.patch_size, (long long) m->n_q); + return nullptr; + } if (m->lm_head_dim==0) m->lm_head_dim = m->lm_hidden/m->n_q; if (g.has("openvla_oft.statistics_json")) { - if (!parse_stats(g.str("openvla_oft.statistics_json"), m->action_dim, m->q01, m->q99, m->unnorm_mask, m->suite)) + if (!m->act_stats.parse(g.str("openvla_oft.statistics_json"), "VLA_OPENVLA_OFT_UNNORM_KEY", "openvla_oft", m->action_dim)) { std::fprintf(stderr, "vla(openvla_oft): failed to parse statistics_json\n"); return nullptr; } - std::printf("vla(openvla_oft): unnorm suite = %s (q99 dim %zu)\n", m->suite.c_str(), m->q99.size()); + std::printf("vla(openvla_oft): unnorm suite = %s (q99 dim %zu)\n", m->act_stats.suite.c_str(), m->act_stats.q99.size()); } { @@ -210,12 +145,11 @@ std::unique_ptr openvla_oft_create(const std::string& mmproj_path ggml_init_params wp = { (size_t)64*1024*1024, nullptr, true }; m->ctx_weights = ggml_init(wp); - bool ok = true; WeightLoader L("openvla_oft", g, m->ctx_weights, m->mt); auto mm = [&](const char * n) { return L.gemm("%s", n); }; auto f32 = [&](const char * n) { return L.f32("%s", n); }; - m->vis.declare(L, m->d_layers, m->s_layers); + m->vis.declare(L); m->token_embd=mm("token_embd.weight"); m->lm_out_norm=f32("lm.output_norm.weight"); m->lm.resize(m->lm_layers); @@ -236,10 +170,6 @@ std::unique_ptr openvla_oft_create(const std::string& mmproj_path for(int i=0;ihead_blocks;++i){ auto&w=m->hblk[i]; char b[64]; auto N=[&](const char*s){ std::snprintf(b,sizeof(b),"aex.head.blk.%d.%s",i,s); return (const char*)b; }; w.lnw=f32(N("ln.weight")); w.lnb=f32(N("ln.bias")); w.linw=mm(N("lin.weight")); w.linb=f32(N("lin.bias")); } - if(!ok){ - std::fprintf(stderr,"vla(openvla_oft): weight setup failed\n"); - return nullptr; - } if (!L.upload(m->backend, &m->weight_buf)) return nullptr; @@ -249,7 +179,7 @@ std::unique_ptr openvla_oft_create(const std::string& mmproj_path m->cfg.n_suffix = m->chunk; m->cfg.max_action_dim = m->action_dim; m->cfg.real_action_dim = m->action_dim; m->cfg.real_state_dim = m->proprio_dim; - m->cfg.max_state_dim = m->proprio_dim; m->cfg.n_img = m->n_patches; m->cfg.hidden = m->lm_hidden; + m->cfg.max_state_dim = m->proprio_dim; m->cfg.n_img = m->vis.n_patches; m->cfg.hidden = m->lm_hidden; m->cfg.n_lang = 512; return m; } @@ -258,52 +188,20 @@ std::vector OpenVlaOftModelArch::predict(const Inputs& in) { using clock = std::chrono::steady_clock; const auto t0 = clock::now(); stats = Stats{}; - const int64_t S=image_size, NP=n_patches, HC=lm_hidden; + const int64_t NP=vis.n_patches, HC=lm_hidden; const int64_t n_views = in.n_images; if (in.precomputed_img_emb) { std::fprintf(stderr, "vla(openvla_oft): precomputed_img_emb is not supported; the DINOv2+SigLIP tower is baked into the GGUF, pass raw images\n"); return {}; } + if (in.n_lang < 1 || !in.lang_tokens) { std::fprintf(stderr, "vla(openvla_oft): need >=1 lang token\n"); return {}; } if (n_views < 1) { std::fprintf(stderr, "vla(openvla_oft): need >=1 image view\n"); return {}; } if (!in.images) { std::fprintf(stderr, "vla(openvla_oft): n_images=%d but the images pointer is null\n", in.n_images); return {}; } - // towers read S*S*3 per view; reject any view that is not exactly SxS. - for (int64_t v=0; v 0.484375). The reference - // preprocesses in bf16, so these are the values it actually sees. - static const float DMEAN[3]={0.484375f,0.455078125f,0.40625f}, DSTD[3]={0.228515625f,0.2236328125f,0.224609375f}; - static const float SMEAN[3]={0.5f,0.5f,0.5f}, SSTD[3]={0.5f,0.5f,0.5f}; const int64_t NPATCH = NP * n_views; - std::vector proj_host((size_t)HC*NPATCH); + ggml_tensor*proj=nullptr; { const auto tv=clock::now(); - ggml_context*C=vision_scratch.reset((size_t)64*1024*1024); - std::vector px_d(n_views), px_s(n_views), cmb(n_views); - for(int v=0; v dbuf, sbuf; - for(int v=0;v(clock::now()-tv).count(); } @@ -343,14 +241,14 @@ std::vector OpenVlaOftModelArch::predict(const Inputs& in) { ggml_tensor*t_proj=ggml_new_tensor_2d(C,GGML_TYPE_F32,HC,NPATCH); ggml_set_input(t_proj); ggml_tensor*patches=ggml_concat(C,t_proj,pvec,1); - ggml_tensor*act0=ggml_new_tensor_2d(C,GGML_TYPE_F32,HC,n_act); ggml_set_input(act0); + ggml_tensor*act0=ggml_new_tensor_2d(C,GGML_TYPE_F32,HC,n_act); ggml_set_input(act0); ggml_set_output(act0); ggml_tensor*seq=ggml_concat(C,bos,patches,1); seq=ggml_concat(C,seq,rest,1); seq=ggml_concat(C,seq,act0,1); seq=ggml_concat(C,seq,stop,1); - ggml_tensor*t_pos=ggml_new_tensor_1d(C,GGML_TYPE_I32,SEQ); ggml_set_input(t_pos); + ggml_tensor*t_pos=ggml_new_tensor_1d(C,GGML_TYPE_I32,SEQ); ggml_set_input(t_pos); ggml_set_output(t_pos); const float lsc=1.0f/std::sqrt((float)lm_head_dim); ggml_tensor*x=seq; for(int i=0;i OpenVlaOftModelArch::predict(const Inputs& in) { ggml_tensor*qr=ggml_rope_ext(C,qh,t_pos,nullptr,(int)lm_head_dim,GGML_ROPE_TYPE_NEOX,0,lm_rope_base,1.0f,0.0f,1.0f,32.0f,1.0f); ggml_tensor*kr=ggml_rope_ext(C,kh,t_pos,nullptr,(int)lm_head_dim,GGML_ROPE_TYPE_NEOX,0,lm_rope_base,1.0f,0.0f,1.0f,32.0f,1.0f); ggml_tensor*Q=ggml_cont(C,ggml_permute(C,qr,0,2,1,3)),*K=ggml_cont(C,ggml_permute(C,kr,0,2,1,3)),*V=ggml_cont(C,ggml_permute(C,vh,1,2,0,3)); - ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_mul_mat_set_prec(kq,GGML_PREC_F32); + ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_prec_set_acc(kq,GGML_PREC_F32); // Unmasked on purpose: OpenVLA-OFT patches transformers to replace the // causal mask across the whole sequence (modeling_llama.py:719-723). ggml_tensor*aw=ggml_soft_max_ext(C,kq,nullptr,lsc,0.0f); @@ -377,14 +275,14 @@ std::vector OpenVlaOftModelArch::predict(const Inputs& in) { ggml_tensor*hin=ggml_reshape_2d(C,ah,action_dim*HC,chunk); - ggml_tensor*hh=LN(C,hin,h_ln1w,h_ln1b,head_ln_eps); - hh=ggml_relu(C,ggml_add(C,ggml_mul_mat(C,h_fc1w,hh),h_fc1b)); + ggml_tensor*hh=layer_norm(C,hin,h_ln1w,h_ln1b,head_ln_eps); + hh=ggml_relu(C,linear(C,h_fc1w,h_fc1b,hh)); for(int i=0;i OpenVlaOftModelArch::predict(const Inputs& in) { ids[i]=in.lang_tokens[i]; ids[L]=(int32_t)stop_id; ggml_backend_tensor_set(t_ids,ids.data(),0,ggml_nbytes(t_ids)); } - ggml_backend_tensor_set(t_proj,proj_host.data(),0,ggml_nbytes(t_proj)); - { + if(!ggml_are_same_shape(proj,t_proj)){ std::fprintf(stderr,"vla(openvla_oft): projector output does not match the LM width\n"); return {}; } + ggml_backend_tensor_copy(proj,t_proj); + if(!gio.consts){ std::vector pp(SEQ); for(int64_t i=0;i sv(proprio_dim,0.0f); for(int64_t i=0;i z((size_t)HC*n_act,0.0f); ggml_backend_tensor_set(act0,z.data(),0,ggml_nbytes(act0)); + gio.consts=true; } + { std::vector sv(proprio_dim,0.0f); for(int64_t i=0;i OpenVlaOftModelArch::predict(const Inputs& in) { ggml_backend_tensor_get(norm_actions,na.data(),0,na.size()*sizeof(float)); stats.ms_inference = std::chrono::duration(clock::now()-ti).count(); - const int64_t Wd = cfg.max_action_dim>0 ? cfg.max_action_dim : action_dim; - std::vector out((size_t)chunk*Wd,0.0f); - for(int64_t c=0;c out=act_stats.unnorm(na,chunk,action_dim,cfg.max_action_dim>0 ? cfg.max_action_dim : action_dim); stats.ms_total = std::chrono::duration(clock::now()-t0).count(); return out; } diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index 27f25ff..f91da15 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -14,49 +14,33 @@ #include "arch.h" #include "modules/gemma_expert.h" -#include "modules/siglip_vit.h" #include "options.h" #include "model.h" #include "ggml.h" -#include "ggml-cpu.h" #include "ggml-backend.h" -#include "ggml-alloc.h" #include "backend.h" -#include "gguf.h" #include "gguf_reader.h" #include "scratch_ctx.h" +#include "layers/attn.h" #include "layers/embed.h" +#include "layers/norm.h" #include "modules/preprocess.h" +#include "modules/prompt.h" #include "act_dtype.h" #include "cuda/vla_cuda_ops.h" -#include "env_flag.h" -#include #include #include #include #include -#include #include #include -#include #include -#include #include namespace vla { -namespace { - - -bool ends_with(const std::string & s, const char * sfx) { - const size_t n = std::strlen(sfx); - return s.size() >= n && s.compare(s.size()-n, n, sfx) == 0; -} - -} - struct Pi0ModelArch : public ModelArchBase { Pi0ModelArch() : ModelArchBase(Arch::PI0) {} ~Pi0ModelArch() override; @@ -66,6 +50,8 @@ struct Pi0ModelArch : public ModelArchBase { ggml_backend_t backend = nullptr; ggml_backend_buffer_t weight_buf = nullptr; ggml_context * ctx_weights = nullptr; + ggml_context * ctx_const = nullptr; + ggml_backend_buffer_t const_buf = nullptr; scratch_ctx vision_scratch; struct MainKey { @@ -77,25 +63,17 @@ struct Pi0ModelArch : public ModelArchBase { struct MainIO { ggml_tensor *t_image_emb=nullptr,*t_lang_emb=nullptr,*t_prefix_pos=nullptr,*t_state=nullptr; ggml_tensor *t_x0=nullptr,*t_suffix_pos=nullptr,*t_full_mask=nullptr,*x_final=nullptr; - std::vector t_time; }; graph_cache main_graph; - std::string ckpt_path_; // Opened once at load: reopening per predict re-parses the whole GGUF header. gguf_reader io{"pi0"}; ggml_type matmul_type = GGML_TYPE_BF16; // Activation dtype carried between ops. F32 by default; BF16 under - // VLA_PI0_BF16_ACT, which removes the per-GEMM F32<->BF16 round trip ggml + // --act-dtype bf16, which removes the per-GEMM F32<->BF16 round trip ggml // pays when BF16 weights meet F32 activations. See mm_act/as_type below. ggml_type act_type = GGML_TYPE_F32; - // In-tree SigLIP-So400m/14 vision tower (was llama.cpp clip.cpp mmproj). - int64_t vit_hidden = 1152, vit_layers = 27, vit_heads = 16; - int64_t vit_image_size = 224, vit_patch_size = 14, vit_n_tokens = 256; - float vit_ln_eps = 1e-6f; - SigLipTower vit; - ggml_tensor * mm_proj_w = nullptr, * mm_proj_b = nullptr; - + PaliVision vis; GemmaStack pl; GemmaStack ex; @@ -105,9 +83,10 @@ struct Pi0ModelArch : public ModelArchBase { ggml_tensor * W_at2 = nullptr, * b_at2 = nullptr; ggml_tensor * W_aout = nullptr,* b_aout = nullptr; + std::vector t_time; + std::vector state_mean, state_std, action_mean, action_std; - std::mt19937 rng{std::random_device{}()}; int n_threads = default_cpu_threads(); }; @@ -117,127 +96,45 @@ namespace { // (the PaliGemma vision tower is the same SigLIP-So400m/14). Bidirectional // attention (nullptr mask), F32 score accumulation, tanh GELU FFN. // Fused attention for the SigLIP tower and the PaliGemma/expert stack. -// OPT-IN (VLA_PI0_FA=1): pi0's score matrices are small (~560 keys, 8 heads), so +// OPT-IN (--flash-attn): pi0's score matrices are small (~560 keys, 8 heads), so // fusing them only moved 111.4 ms -> 107.5 ms (3.5%), and ggml's FA computes K/V // at F16 regardless of the input type. pi0's flash-attention SR was never // measured, so it stays opt-in on an unquantified risk rather than a measured // cost. (The evo1 SR drop this used to cite did not reproduce.) -// VLA_PI0_BF16_ACT is the better lever here: 9.1%, and its SR was measured. +// --act-dtype bf16 is the better lever here: 9.1%, and its SR was measured. ggml_tensor * build_siglip_layer(ggml_context * C, const EncBlockW & w, ggml_tensor * x, - int64_t seq, int64_t heads, int64_t head_dim, int64_t hidden, float ln_eps, - ggml_type at) { - const float scale = 1.0f/std::sqrt((float) head_dim); - ggml_tensor * n1 = ggml_add(C, ggml_mul(C, ggml_norm(C, x, ln_eps), w.ln1w), w.ln1b); + int64_t seq, const EncCfg & c, ggml_type at) { + const float scale = 1.0f/std::sqrt((float) c.head_dim); + ggml_tensor * n1 = layer_norm(C, x, w.ln1w, w.ln1b, c.ln_eps); ggml_tensor * q = as_type(C, ggml_add(C, mm_act(C, w.Wq, n1, at), w.bq), GGML_TYPE_F32); ggml_tensor * k = as_type(C, ggml_add(C, mm_act(C, w.Wk, n1, at), w.bk), GGML_TYPE_F32); ggml_tensor * v = as_type(C, ggml_add(C, mm_act(C, w.Wv, n1, at), w.bv), GGML_TYPE_F32); - ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, head_dim, heads, seq), 0, 2, 1, 3)); - ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, head_dim, heads, seq), 0, 2, 1, 3)); + ggml_tensor * Q = to_heads(C, q, c.head_dim, c.heads, seq); + ggml_tensor * K = to_heads(C, k, c.head_dim, c.heads, seq); ggml_tensor * att; if (vla::flash_attn_enabled()) { // Avoids materialising the per-head score matrix; K/V stay F32 so the // numerics track the explicit path below (except on Hexagon, whose // kernel takes F16 K/V only; see fa_kv). - ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 0, 2, 1, 3)); + ggml_tensor * V = to_heads(C, v, c.head_dim, c.heads, seq); ggml_tensor * fa = ggml_flash_attn_ext(C, Q, vla::fa_kv(C, K), vla::fa_kv(C, V), nullptr, scale, 0.0f, 0.0f); - ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); - att = ggml_reshape_2d(C, fa, hidden, seq); + ggml_prec_set_acc(fa, GGML_PREC_F32); + att = ggml_reshape_2d(C, fa, c.hidden, seq); } else { - ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); - ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); - att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); + att = attention(C, Q, K, to_heads_v(C, v, c.head_dim, c.heads, seq), nullptr, scale, c.hidden, seq); } ggml_tensor * h1 = ggml_add(C, x, ggml_add(C, mm_act(C, w.Wo, as_type(C, att, at), at), w.bo)); - ggml_tensor * n2 = ggml_add(C, ggml_mul(C, ggml_norm(C, h1, ln_eps), w.ln2w), w.ln2b); + ggml_tensor * n2 = layer_norm(C, h1, w.ln2w, w.ln2b, c.ln_eps); ggml_tensor * ff = ggml_add(C, mm_act(C, w.Wfc2, vla::gelu(C, ggml_add(C, mm_act(C, w.Wfc1, n2, at), w.bfc1)), at), w.bfc2); return ggml_add(C, h1, ff); } -// CHW-planar float image in [-1,1] for ggml_conv_2d (SigLIP mean/std 0.5). - -ggml_tensor * build_gemma_layer( - ggml_context * ctx, const GemmaLayerW & w, - ggml_tensor * x_in, ggml_tensor * positions, - const Config & cfg, int64_t seq, float rope_base, - ggml_tensor * cached_K, ggml_tensor * cached_V, ggml_tensor * mask, - ggml_tensor ** k_out, ggml_tensor ** v_out, ggml_type at) { - const int64_t hd = cfg.head_dim; - const int64_t nq = cfg.n_q_heads; - const int64_t nkv = cfg.n_kv_heads; - const int64_t qf = nq * hd; - - ggml_tensor * x_norm = ggml_mul(ctx, ggml_rms_norm(ctx, x_in, cfg.rms_eps), w.ln_in); - - // Q/K/V land in F32: RoPE, the KV cache the suffix passes re-read, and the - // score/softmax core all stay full precision. - ggml_tensor * q = as_type(ctx, mm_act(ctx, w.Wq, x_norm, at), GGML_TYPE_F32); - ggml_tensor * k = as_type(ctx, mm_act(ctx, w.Wk, x_norm, at), GGML_TYPE_F32); - ggml_tensor * v = as_type(ctx, mm_act(ctx, w.Wv, x_norm, at), GGML_TYPE_F32); - - ggml_tensor * q_h = ggml_reshape_3d(ctx, q, hd, nq, seq); - ggml_tensor * k_h = ggml_reshape_3d(ctx, k, hd, nkv, seq); - ggml_tensor * v_h = ggml_reshape_3d(ctx, v, hd, nkv, seq); - - auto rope_call = [&](ggml_tensor * t) { - return ggml_rope_ext(ctx, t, positions, nullptr, - (int) hd, GGML_ROPE_TYPE_NEOX, 0, - rope_base, 1.f, 0.f, 1.f, - 32.f, 1.f); - }; - ggml_tensor * q_rope = rope_call(q_h); - ggml_tensor * k_rope = rope_call(k_h); - - if (k_out) - *k_out = k_rope; - if (v_out) - *v_out = v_h; - - ggml_tensor * K_full = k_rope; - ggml_tensor * V_full = v_h; - if (cached_K && cached_V) { - K_full = ggml_concat(ctx, cached_K, k_rope, 2); - V_full = ggml_concat(ctx, cached_V, v_h, 2); - } - - const float scale = 1.f/std::sqrt((float) hd); - ggml_tensor * Q = ggml_cont(ctx, ggml_permute(ctx, q_rope, 0, 2, 1, 3)); - ggml_tensor * K = ggml_cont(ctx, ggml_permute(ctx, K_full, 0, 2, 1, 3)); - ggml_tensor * att_pre; - if (vla::flash_attn_enabled()) { - ggml_tensor * V = ggml_cont(ctx, ggml_permute(ctx, V_full, 0, 2, 1, 3)); - // ggml_flash_attn_ext asserts an F16 mask. The mask holds only 0 and - // -inf, both exactly representable in F16, so the cast is lossless. - ggml_tensor * mask_f16 = mask ? ggml_cast(ctx, mask, GGML_TYPE_F16) : nullptr; - ggml_tensor * fa = ggml_flash_attn_ext(ctx, Q, vla::fa_kv(ctx, K), vla::fa_kv(ctx, V), mask_f16, scale, 0.0f, 0.0f); - ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); - att_pre = ggml_reshape_2d(ctx, fa, qf, seq); - } else { - ggml_tensor * V = ggml_cont(ctx, ggml_permute(ctx, V_full, 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(ctx, K, Q); - ggml_mul_mat_set_prec(kq, GGML_PREC_F32); - ggml_tensor * attn = ggml_soft_max_ext(ctx, kq, mask, scale, 0.f); - ggml_tensor * kqv = ggml_mul_mat(ctx, V, attn); - att_pre = ggml_reshape_2d(ctx, - ggml_cont(ctx, ggml_permute(ctx, kqv, 0, 2, 1, 3)), qf, seq); - } - ggml_tensor * o_out = mm_act(ctx, w.Wo, as_type(ctx, att_pre, at), at); - ggml_tensor * h1 = ggml_add(ctx, x_in, o_out); - - ggml_tensor * x_norm_mlp = ggml_mul(ctx, ggml_rms_norm(ctx, h1, cfg.rms_eps), w.ln_post); - ggml_tensor * gate = mm_act(ctx, w.Wgate, x_norm_mlp, at); - ggml_tensor * up = mm_act(ctx, w.Wup, x_norm_mlp, at); - ggml_tensor * inter_t = ggml_mul(ctx, vla::gelu(ctx, gate), up); - ggml_tensor * mlp_out = mm_act(ctx, w.Wdown, inter_t, at); - return ggml_add(ctx, h1, mlp_out); -} - ggml_tensor * build_embed_suffix(ggml_context * ctx, const Pi0ModelArch & m, ggml_tensor * state, ggml_tensor * x, ggml_tensor * time_bcast) { const ggml_type at = m.act_type; ggml_tensor * state_emb = ggml_add(ctx, mm_act(ctx, m.W_sp, as_type(ctx, state, at), at), m.b_sp); - // x is the F32 flow-matching state; time_bcast is an F32 input tensor. Both + // x is the F32 flow-matching state; time_bcast is an F32 tensor. Both // enter the expert in the activation dtype, and ggml_concat needs them to agree. ggml_tensor * action_emb = ggml_add(ctx, mm_act(ctx, m.W_ain, as_type(ctx, x, at), at), m.b_ain); ggml_tensor * action_time_in = ggml_concat(ctx, action_emb, as_type(ctx, time_bcast, at), 0); @@ -248,88 +145,17 @@ ggml_tensor * build_embed_suffix(ggml_context * ctx, const Pi0ModelArch & m, return ggml_concat(ctx, state_emb_2d, action_time_emb, 1); } -bool load_config(const gguf_reader & g, Config & cfg) { - auto need = [&](const char * k) { - if (!g.has(k)) { - std::fprintf(stderr, "vla(pi0): gguf missing key %s\n", k); - return false; - } - return true; - }; - for (const char * k : {"pi0.hidden", "pi0.intermediate", "pi0.n_q_heads", "pi0.n_kv_heads", - "pi0.head_dim", "pi0.n_layers", "pi0.expert_h", "pi0.expert_inter", - "pi0.chunk_size", "pi0.num_steps", "pi0.max_state_dim", "pi0.max_action_dim", - "pi0.real_state_dim", "pi0.real_action_dim", "pi0.tokenizer_max_length", - "pi0.min_period", "pi0.max_period"}) { - if (!need(k)) - return false; - } - cfg = Config{}; - cfg.hidden = g.u32("pi0.hidden"); - cfg.intermediate = g.u32("pi0.intermediate"); - cfg.n_q_heads = g.u32("pi0.n_q_heads"); - cfg.n_kv_heads = g.u32("pi0.n_kv_heads"); - cfg.head_dim = g.u32("pi0.head_dim"); - cfg.n_layers = g.u32("pi0.n_layers"); - cfg.expert_h = g.u32("pi0.expert_h"); - cfg.expert_inter = g.u32("pi0.expert_inter"); - cfg.n_suffix = g.u32("pi0.chunk_size"); - cfg.num_steps = g.u32("pi0.num_steps"); - cfg.max_state_dim = g.u32("pi0.max_state_dim"); - cfg.max_action_dim = g.u32("pi0.max_action_dim"); - cfg.real_state_dim = g.u32("pi0.real_state_dim"); - cfg.real_action_dim = g.u32("pi0.real_action_dim"); - cfg.n_lang = g.u32("pi0.tokenizer_max_length"); - cfg.min_period = g.f64("pi0.min_period"); - cfg.max_period = g.f64("pi0.max_period"); - - cfg.n_state = 1; - cfg.n_img = 256; - cfg.q_full_dim = cfg.n_q_heads * cfg.head_dim; - cfg.kv_full_dim = cfg.n_kv_heads*cfg.head_dim; - cfg.self_attn_every_n = 0; - cfg.rms_eps = g.has("pi0.rms_norm_eps") ? g.f32("pi0.rms_norm_eps") : 1e-6f; - cfg.norm_eps = g.has("pi0.norm_eps") ? g.f32("pi0.norm_eps") : 1e-8f; - cfg.rope_mode = GGML_ROPE_TYPE_NEOX; - cfg.rope_n_dims = (int) cfg.head_dim; - cfg.rope_freq_base = g.has("pi0.rope_theta") ? (float) g.f64("pi0.rope_theta") : 10000.f; - cfg.n_prefix = 0; - cfg.n_full = 0; - return true; -} - bool load_stats(gguf_reader & g, Pi0ModelArch & m) { const auto & cfg = m.cfg; m.state_mean .assign(cfg.real_state_dim, 0.f); m.state_std .assign(cfg.real_state_dim, 1.f); m.action_mean.assign(cfg.real_action_dim, 0.f); m.action_std .assign(cfg.real_action_dim, 1.f); - // Absent stats are a valid checkpoint: identity, carry on. Stats that are - // present but unreadable are not - falling back to identity there hands back - // un-denormalised actions with nothing in the log. Note stderr, not stdout: - // stdout is the action stream tests/predict_check.cpp diffs. - auto read1d = [&](const char * name, std::vector & dst) { - const ggml_tensor * t = g.meta(name); - if (!t) { - std::fprintf(stderr, "vla(pi0): %s missing - identity\n", name); - return true; - } - if (t->ne[0] != (int64_t) dst.size()) { - std::fprintf(stderr, "vla(pi0): %s is %lld wide, expected %zu\n", - name, (long long) t->ne[0], dst.size()); - return false; - } - if (!g.read_raw(name, dst.data(), dst.size()*sizeof(float))) { - std::fprintf(stderr, "vla(pi0): %s read failed\n", name); - return false; - } - return true; - }; bool ok = true; - ok &= read1d("state_mean", m.state_mean); - ok &= read1d("state_std", m.state_std); - ok &= read1d("action_mean", m.action_mean); - ok &= read1d("action_std", m.action_std); + ok &= read_pi_stat(g, "state_mean", m.state_mean); + ok &= read_pi_stat(g, "state_std", m.state_std); + ok &= read_pi_stat(g, "action_mean", m.action_mean); + ok &= read_pi_stat(g, "action_std", m.action_std); return ok; } @@ -340,6 +166,10 @@ Pi0ModelArch::~Pi0ModelArch() { ggml_backend_buffer_free(weight_buf); if (ctx_weights) ggml_free(ctx_weights); + if (const_buf) + ggml_backend_buffer_free(const_buf); + if (ctx_const) + ggml_free(ctx_const); if (backend) ggml_backend_free(backend); } @@ -350,27 +180,11 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, const Options& opts) { (void) config_path; - if (!ends_with(ckpt_path, ".gguf")) { - std::fprintf(stderr, - "vla(pi0): ckpt must be a GGUF produced by scripts/convert_pi0_to_gguf.py " - "(got '%s'); direct .safetensors loading for π₀ is not yet supported\n", - ckpt_path.c_str()); - return nullptr; - } - auto m = std::make_unique(); - m->ckpt_path_ = ckpt_path; m->matmul_type = opts.weight_dtype.value_or(vla::default_weight_dtype(GGML_TYPE_BF16)); - if (!m->io.open(ckpt_path)) - return nullptr; gguf_reader & g = m->io; - if (!g.has("pi0.architecture") || g.str("pi0.architecture") != "pi0") { - std::fprintf(stderr, "vla(pi0): '%s' is not a π₀ GGUF (pi0.architecture missing/wrong)\n", - ckpt_path.c_str()); - return nullptr; - } - if (!load_config(g, m->cfg)) + if (!load_pi_config(g, ckpt_path, 1, m->cfg) || !resolve_num_steps("pi0", opts, m->cfg.num_steps)) return nullptr; const Config & cfg = m->cfg; std::printf("vla(pi0): hidden=%lld inter=%lld heads=%lldq/%lldkv x%lld n_layers=%lld " @@ -382,7 +196,6 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, cfg.num_steps, (long long) cfg.real_state_dim, (long long) cfg.real_action_dim, m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); - m->n_threads = default_cpu_threads(); { const Backend b = backend_init("vla(pi0)", m->n_threads); if (!b.handle) { @@ -395,29 +208,17 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, if (b.is_cuda && m->matmul_type == GGML_TYPE_BF16) { m->act_type = GGML_TYPE_BF16; cuda_register_bf16_ops(); // installs the in-tree BF16 CUDA kernels - std::printf("vla(pi0): activations = BF16 (VLA_PI0_BF16_ACT)\n"); + std::printf("vla(pi0): activations = BF16\n"); } else { - std::fprintf(stderr, "vla(pi0): VLA_PI0_BF16_ACT ignored - needs CUDA and BF16 weights\n"); + std::fprintf(stderr, "vla(pi0): --act-dtype bf16 ignored - needs CUDA and BF16 weights\n"); } } } // The SigLIP tower is now bundled in the ckpt GGUF; mmproj_path is ignored. (void) mmproj_path; - { - auto vu = [&](const char * k, int64_t & d) { if (g.has(k)) d = (int64_t) g.u32(k); }; - vu("pi0.vit_hidden", m->vit_hidden); vu("pi0.vit_layers", m->vit_layers); - vu("pi0.vit_heads", m->vit_heads); vu("pi0.image_size", m->vit_image_size); - vu("pi0.patch_size", m->vit_patch_size); vu("pi0.n_img_tokens", m->vit_n_tokens); - if (g.has("pi0.vit_ln_eps")) - m->vit_ln_eps = g.f32("pi0.vit_ln_eps"); - const int64_t grid = m->vit_image_size/m->vit_patch_size; - if (grid * grid != m->vit_n_tokens || m->vit_n_tokens != cfg.n_img) { - std::fprintf(stderr, "vla(pi0): vit geometry mismatch (grid^2=%lld n_img_tokens=%lld cfg.n_img=%lld)\n", - (long long) (grid * grid), (long long) m->vit_n_tokens, (long long) cfg.n_img); - return nullptr; - } - } + if (!m->vis.load(g, cfg.n_img)) + return nullptr; { ggml_init_params wp = { (size_t) 16*1024*1024, nullptr, true }; @@ -429,9 +230,7 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, } WeightLoader L("pi0", g, m->ctx_weights, m->matmul_type); - m->vit.declare(L, "vit", m->vit_layers); - m->mm_proj_w = L.gemm ("mm.proj.weight"); - m->mm_proj_b = L.opt_f32("mm.proj.bias"); + m->vis.declare(L); m->pl.declare(L, "vlm", cfg.n_layers, false); m->ex.declare(L, "aex", cfg.n_layers, true); @@ -450,6 +249,31 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, if (!load_stats(g, *m)) return nullptr; + + { + ggml_init_params p = { (size_t) cfg.num_steps*ggml_tensor_overhead(), nullptr, true }; + m->ctx_const = ggml_init(p); + if (!m->ctx_const) { + std::fprintf(stderr, "vla(pi0): ggml_init(ctx_const) failed\n"); + return nullptr; + } + m->t_time.resize(cfg.num_steps); + for (ggml_tensor * & t : m->t_time) + t = ggml_new_tensor_2d(m->ctx_const, GGML_TYPE_F32, cfg.expert_h, cfg.n_suffix); + m->const_buf = alloc_weights(m->ctx_const, m->backend); + if (!m->const_buf) { + std::fprintf(stderr, "vla(pi0): alloc_weights(time embeddings) failed\n"); + return nullptr; + } + const float dt = -1.0f/(float) cfg.num_steps; + std::vector tile((size_t) cfg.expert_h * cfg.n_suffix); + for (int s=0; s tv = sinusoidal_time_emb(1.0f+(float) s * dt, cfg.expert_h, cfg.min_period, cfg.max_period); + for (int64_t c=0; ct_time[s], tile.data(), 0, ggml_nbytes(m->t_time[s])); + } + } std::printf("vla(pi0): model loaded (n_threads=%d)\n", m->n_threads); return m; } @@ -469,61 +293,21 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { const int64_t max_ad = cfg.max_action_dim; const int num_steps = cfg.num_steps; const float dt = -1.0f/(float) num_steps; - const float rope_base = cfg.rope_freq_base; + const bool fa = flash_attn_enabled(); - std::vector img_emb_host; int64_t n_img_tokens = 0; if (in.precomputed_img_emb) { + if (in.n_img_views < 1) { + std::fprintf(stderr, "vla(pi0): precomputed_img_emb set but n_img_views=%d\n", in.n_img_views); + return {}; + } n_img_tokens = (int64_t) in.n_img_views*cfg.n_img; - img_emb_host.assign(in.precomputed_img_emb, - in.precomputed_img_emb+(size_t) n_img_tokens * hidden_pl); } else { if (in.n_images < 1 || !in.images) { std::fprintf(stderr, "vla(pi0): predict: no images and no precomputed_img_emb\n"); return {}; } - const int64_t K = vit_n_tokens, H = hidden_pl, grid = vit_image_size/vit_patch_size; - n_img_tokens = (int64_t) in.n_images*K; - img_emb_host.assign((size_t) in.n_images*K * H, 0.0f); - - ggml_context * VC = vision_scratch.reset((size_t) 128*1024*1024); - if (!VC) { std::fprintf(stderr, "vla(pi0): ggml_init(vision ctx) failed\n"); return {}; } - ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, vit_image_size, vit_image_size, 3); ggml_set_input(t_px); - ggml_tensor * conv = ggml_conv_2d(VC, vit.patch_w, t_px, (int) vit_patch_size, (int) vit_patch_size, 0, 0, 1, 1); - ggml_tensor * patches = ggml_cont(VC, ggml_transpose(VC, ggml_reshape_2d(VC, conv, grid * grid, vit_hidden))); - // patch embed (conv_2d) stays F32; the tower runs in the activation dtype - ggml_tensor * h = as_type(VC, ggml_add(VC, ggml_add(VC, patches, vit.patch_b), vit.pos), act_type); - for (int64_t i=0; ine[0])), GGML_TYPE_F32); - ggml_set_output(vit_emb); - - ggml_cgraph * vg = ggml_new_graph_custom(VC, 8192, false); - ggml_build_forward_expand(vg, vit_emb); - - if (!vision_scratch.alloc(backend, vg)) { - std::fprintf(stderr, "vla(pi0): vision gallocr alloc failed\n"); - return {}; - } - const auto tv0 = clk::now(); - std::vector chw; - for (int v=0; v(clk::now()-tv0).count(); + n_img_tokens = (int64_t) in.n_images*vis.n_tokens; } if (in.n_lang < 1 || !in.lang_tokens) { @@ -542,7 +326,9 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { // Prefix + expert graph depends only on the token counts and step count. const MainKey mkey{ n_img_tokens, n_lang, num_steps }; - const bool built = main_graph.ensure(backend, mkey, (size_t) 64*1024*1024, + const size_t max_nodes = (size_t) 64*n_layers*(num_steps+1) + 1024; + const bool built = main_graph.ensure(backend, mkey, + ggml_tensor_overhead()*max_nodes + ggml_graph_overhead_custom(max_nodes, false), [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { ggml_tensor * t_image_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_img_tokens); ggml_set_input(t_image_emb); ggml_tensor * t_lang_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_lang); ggml_set_input(t_lang_emb); @@ -551,11 +337,6 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { ggml_tensor * t_x0 = ggml_new_tensor_2d(C, GGML_TYPE_F32, max_ad, chunk); ggml_set_input(t_x0); ggml_tensor * t_suffix_pos= ggml_new_tensor_1d(C, GGML_TYPE_I32, n_suf); ggml_set_input(t_suffix_pos); ggml_tensor * t_full_mask = ggml_new_tensor_2d(C, GGML_TYPE_F32, n_total, n_suf); ggml_set_input(t_full_mask); - std::vector t_time(num_steps); - for (int s=0; s Pi0ModelArch::predict(const Inputs& in) { { ggml_tensor * h = prefix_embs; for (int64_t i=0; i Pi0ModelArch::predict(const Inputs& in) { for (int step=0; step Pi0ModelArch::predict(const Inputs& in) { gio.t_image_emb=t_image_emb; gio.t_lang_emb=t_lang_emb; gio.t_prefix_pos=t_prefix_pos; gio.t_state=t_state; gio.t_x0=t_x0; gio.t_suffix_pos=t_suffix_pos; - gio.t_full_mask=t_full_mask; gio.t_time=t_time; gio.x_final=x_final; + gio.t_full_mask=t_full_mask; gio.x_final=x_final; - ggml_cgraph * gf = ggml_new_graph_custom(C, 16384, false); + ggml_cgraph * gf = ggml_new_graph_custom(C, max_nodes, false); ggml_build_forward_expand(gf, x_final); return gf; }); @@ -612,9 +391,52 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { ggml_tensor * t_prefix_pos = gio.t_prefix_pos, * t_state = gio.t_state, * t_x0 = gio.t_x0; ggml_tensor * t_suffix_pos = gio.t_suffix_pos, * t_full_mask = gio.t_full_mask; ggml_tensor * x_final = gio.x_final; - std::vector & t_time = gio.t_time; - ggml_backend_tensor_set(t_image_emb, img_emb_host.data(), 0, ggml_nbytes(t_image_emb)); + if (in.precomputed_img_emb) { + ggml_backend_tensor_set(t_image_emb, in.precomputed_img_emb, 0, ggml_nbytes(t_image_emb)); + } else { + const int64_t K = vis.n_tokens, H = hidden_pl, grid = vis.image_size/vis.patch_size; + ggml_context * VC = vision_scratch.reset((size_t) 128*1024*1024); + if (!VC) { std::fprintf(stderr, "vla(pi0): ggml_init(vision ctx) failed\n"); return {}; } + ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, vis.image_size, vis.image_size, 3); ggml_set_input(t_px); + // patch embed (conv_2d) stays F32; the tower runs in the activation dtype + ggml_tensor * h = as_type(VC, vis.vit.embed_conv(VC, t_px, vis.patch_size, grid), act_type); + for (const EncBlockW & w : vis.vit.enc.blk) + h = build_siglip_layer(VC, w, h, K, vis.vit.enc.cfg, act_type); + h = layer_norm(VC, h, vis.vit.post_ln_w, vis.vit.post_ln_b, vis.vit.enc.cfg.ln_eps); + // PaliGemma projector: linear (+ optional bias). + ggml_tensor * proj = mm_act(VC, vis.proj_w, h, act_type); + if (vis.proj_b) + proj = ggml_add(VC, proj, vis.proj_b); + ggml_tensor * vit_emb = as_type(VC, proj, GGML_TYPE_F32); + ggml_set_output(vit_emb); + + ggml_cgraph * vg = ggml_new_graph_custom(VC, 8192, false); + ggml_build_forward_expand(vg, vit_emb); + + if (!vision_scratch.alloc(backend, vg)) { + std::fprintf(stderr, "vla(pi0): vision gallocr alloc failed\n"); + return {}; + } + const auto tv0 = clk::now(); + std::vector chw; + for (int v=0; vnb[1], (size_t) v * K * t_image_emb->nb[1]); + if (ggml_backend_view_init(dst) != GGML_STATUS_SUCCESS) { + std::fprintf(stderr, "vla(pi0): image embedding view failed (view %d)\n", v); + return {}; + } + ggml_backend_tensor_copy(vit_emb, dst); + } + stats.ms_vision = std::chrono::duration(clk::now()-tv0).count(); + } ggml_backend_tensor_set(t_lang_emb, lang_rows.data(), 0, ggml_nbytes(t_lang_emb)); { std::vector pp(n_prefix); for (int64_t i=0; i Pi0ModelArch::predict(const Inputs& in) { ggml_backend_tensor_set(t_state, sh.data(), 0, ggml_nbytes(t_state)); } { - std::vector x0h((size_t) max_ad * chunk); - if (in.noise) - std::memcpy(x0h.data(), in.noise, x0h.size()*sizeof(float)); - else { - std::normal_distribution nd(0.f, 1.f); - for (auto & v : x0h) - v = nd(rng); - } + std::vector x0h; + init_noise(in, (size_t) max_ad * chunk, x0h); ggml_backend_tensor_set(t_x0, x0h.data(), 0, ggml_nbytes(t_x0)); } { @@ -658,15 +474,6 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { } ggml_backend_tensor_set(t_full_mask, mk.data(), 0, ggml_nbytes(t_full_mask)); } - for (int s=0; s tv = sinusoidal_time_emb(timestep, hidden_ex, cfg.min_period, cfg.max_period); - std::vector tile((size_t) hidden_ex * chunk); - for (int64_t c=0; c #include #include #include #include -#include -#include #include -#include #include -#include #include namespace vla { -namespace { - - -struct ExpertLayerW { - ggml_tensor * ada_in_w = nullptr; - ggml_tensor * ada_in_b = nullptr; - ggml_tensor * Wq = nullptr; - ggml_tensor * Wk = nullptr; - ggml_tensor * Wv = nullptr; - ggml_tensor * Wo = nullptr; - ggml_tensor * ada_post_w = nullptr; - ggml_tensor * ada_post_b = nullptr; - ggml_tensor * Wgate = nullptr; - ggml_tensor * Wup = nullptr; - ggml_tensor * Wdown = nullptr; -}; - -bool ends_with(const std::string & s, const char * sfx) { - const size_t n = std::strlen(sfx); - return s.size() >= n && s.compare(s.size()-n, n, sfx) == 0; -} -} - struct Pi05ModelArch : public ModelArchBase { Pi05ModelArch() : ModelArchBase(Arch::PI05) {} ~Pi05ModelArch() override; @@ -77,6 +46,8 @@ struct Pi05ModelArch : public ModelArchBase { ggml_backend_t backend = nullptr; ggml_backend_buffer_t weight_buf = nullptr; ggml_context * ctx_weights = nullptr; + ggml_context * ctx_const = nullptr; + ggml_backend_buffer_t const_buf = nullptr; scratch_ctx vision_scratch; struct MainKey { @@ -88,128 +59,33 @@ struct Pi05ModelArch : public ModelArchBase { struct MainIO { ggml_tensor *t_image_emb=nullptr,*t_lang_emb=nullptr,*t_prefix_pos=nullptr; ggml_tensor *t_x0=nullptr,*t_suffix_pos=nullptr,*x_final=nullptr; - std::vector t_time; }; graph_cache main_graph; - std::string ckpt_path_; // Opened once at load: reopening per predict re-parses the whole GGUF header. gguf_reader io{"pi05"}; ggml_type matmul_type = GGML_TYPE_BF16; - int64_t adarms_cond_dim = 0; - - // In-tree SigLIP-So400m/14 vision tower (was llama.cpp clip.cpp mmproj). - int64_t vit_hidden = 1152, vit_layers = 27, vit_heads = 16; - int64_t vit_image_size = 224, vit_patch_size = 14, vit_n_tokens = 256; - float vit_ln_eps = 1e-6f; - SigLipTower vit; - ggml_tensor * mm_proj_w = nullptr, * mm_proj_b = nullptr; + PaliVision vis; GemmaStack pl; - std::vector ex_layers; - - ggml_tensor * ex_final_w = nullptr; - ggml_tensor * ex_final_b = nullptr; + std::vector ex_layers; ggml_tensor * W_ain = nullptr, * b_ain = nullptr; - ggml_tensor * W_tin = nullptr, * b_tin = nullptr; - ggml_tensor * W_tout = nullptr, * b_tout = nullptr; ggml_tensor * W_aout = nullptr, * b_aout = nullptr; - std::vector state_mean, state_std, action_mean, action_std; + ggml_tensor * ada_mod = nullptr; + + std::vector action_mean, action_std; std::vector action_q01, action_q99; bool quantile_norm = false; - std::mt19937 rng{std::random_device{}()}; int n_threads = default_cpu_threads(); }; namespace { -// One pre-norm SigLIP encoder block, identical to gr00tn1d5's in-tree tower -// (the PaliGemma vision tower is the same SigLIP-So400m/14). Bidirectional -// attention (nullptr mask), F32 score accumulation, tanh GELU FFN. -ggml_tensor * build_siglip_layer(ggml_context * C, const EncBlockW & w, ggml_tensor * x, - int64_t seq, int64_t heads, int64_t head_dim, int64_t hidden, float ln_eps) { - const float scale = 1.0f/std::sqrt((float) head_dim); - ggml_tensor * n1 = ggml_add(C, ggml_mul(C, ggml_norm(C, x, ln_eps), w.ln1w), w.ln1b); - ggml_tensor * q = ggml_add(C, ggml_mul_mat(C, w.Wq, n1), w.bq); - ggml_tensor * k = ggml_add(C, ggml_mul_mat(C, w.Wk, n1), w.bk); - ggml_tensor * v = ggml_add(C, ggml_mul_mat(C, w.Wv, n1), w.bv); - ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, head_dim, heads, seq), 0, 2, 1, 3)); - ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, head_dim, heads, seq), 0, 2, 1, 3)); - ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); - ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); - ggml_tensor * att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); - ggml_tensor * h1 = ggml_add(C, x, ggml_add(C, ggml_mul_mat(C, w.Wo, att), w.bo)); - ggml_tensor * n2 = ggml_add(C, ggml_mul(C, ggml_norm(C, h1, ln_eps), w.ln2w), w.ln2b); - ggml_tensor * ff = ggml_add(C, ggml_mul_mat(C, w.Wfc2, vla::gelu(C, ggml_add(C, ggml_mul_mat(C, w.Wfc1, n2), w.bfc1))), w.bfc2); - return ggml_add(C, h1, ff); -} - -// CHW-planar float image in [-1,1] for ggml_conv_2d (SigLIP mean/std 0.5). - -ggml_tensor * build_vlm_layer( - ggml_context * ctx, const GemmaLayerW & w, - ggml_tensor * x_in, ggml_tensor * positions, - const Config & cfg, int64_t seq, float rope_base, - ggml_tensor ** k_out, ggml_tensor ** v_out) { - const int64_t hd = cfg.head_dim; - const int64_t nq = cfg.n_q_heads; - const int64_t nkv = cfg.n_kv_heads; - const int64_t qf = nq * hd; - - ggml_tensor * x_norm = ggml_mul(ctx, ggml_rms_norm(ctx, x_in, cfg.rms_eps), w.ln_in); - - ggml_tensor * q = ggml_mul_mat(ctx, w.Wq, x_norm); - ggml_tensor * k = ggml_mul_mat(ctx, w.Wk, x_norm); - ggml_tensor * v = ggml_mul_mat(ctx, w.Wv, x_norm); - - ggml_tensor * q_h = ggml_reshape_3d(ctx, q, hd, nq, seq); - ggml_tensor * k_h = ggml_reshape_3d(ctx, k, hd, nkv, seq); - ggml_tensor * v_h = ggml_reshape_3d(ctx, v, hd, nkv, seq); - - auto rope_call = [&](ggml_tensor * t) { - return ggml_rope_ext(ctx, t, positions, nullptr, - (int) hd, GGML_ROPE_TYPE_NEOX, 0, - rope_base, 1.f, 0.f, 1.f, 32.f, 1.f); - }; - ggml_tensor * q_rope = rope_call(q_h); - ggml_tensor * k_rope = rope_call(k_h); - - if (k_out) - *k_out = k_rope; - if (v_out) - *v_out = v_h; - - ggml_tensor * Q = ggml_cont(ctx, ggml_permute(ctx, q_rope, 0, 2, 1, 3)); - ggml_tensor * K = ggml_cont(ctx, ggml_permute(ctx, k_rope, 0, 2, 1, 3)); - ggml_tensor * V = ggml_cont(ctx, ggml_permute(ctx, v_h, 1, 2, 0, 3)); - - ggml_tensor * kq = ggml_mul_mat(ctx, K, Q); - ggml_mul_mat_set_prec(kq, GGML_PREC_F32); - const float scale = 1.f/std::sqrt((float) hd); - ggml_tensor * attn = ggml_soft_max_ext(ctx, kq, nullptr, scale, 0.f); - ggml_tensor * kqv = ggml_mul_mat(ctx, V, attn); - - ggml_tensor * att_pre = ggml_reshape_2d(ctx, - ggml_cont(ctx, ggml_permute(ctx, kqv, 0, 2, 1, 3)), qf, seq); - ggml_tensor * o_out = ggml_mul_mat(ctx, w.Wo, att_pre); - ggml_tensor * h1 = ggml_add(ctx, x_in, o_out); - - ggml_tensor * x_norm_mlp = ggml_mul(ctx, ggml_rms_norm(ctx, h1, cfg.rms_eps), w.ln_post); - ggml_tensor * gate = ggml_mul_mat(ctx, w.Wgate, x_norm_mlp); - ggml_tensor * up = ggml_mul_mat(ctx, w.Wup, x_norm_mlp); - ggml_tensor * inter_t = ggml_mul(ctx, vla::gelu(ctx, gate), up); - ggml_tensor * mlp_out = ggml_mul_mat(ctx, w.Wdown, inter_t); - return ggml_add(ctx, h1, mlp_out); -} - ggml_tensor * build_adarms( - ggml_context * ctx, ggml_tensor * x, ggml_tensor * dense_w, - ggml_tensor * dense_b, ggml_tensor * cond, int64_t h, float eps, + ggml_context * ctx, ggml_tensor * x, ggml_tensor * mod, int64_t h, float eps, ggml_tensor ** gate_out) { - ggml_tensor * mod = ggml_add(ctx, ggml_mul_mat(ctx, dense_w, cond), dense_b); ggml_tensor * scale = ggml_view_1d(ctx, mod, h, 0); ggml_tensor * shift = ggml_view_1d(ctx, mod, h, (size_t) h * sizeof(float)); ggml_tensor * gate = ggml_view_1d(ctx, mod, h, (size_t) 2*h * sizeof(float)); @@ -223,152 +99,122 @@ ggml_tensor * build_adarms( } ggml_tensor * build_expert_layer( - ggml_context * ctx, const ExpertLayerW & w, - ggml_tensor * x_in, ggml_tensor * positions, ggml_tensor * cond, - const Config & cfg, int64_t seq, float rope_base, + ggml_context * ctx, const GemmaLayerW & w, + ggml_tensor * x_in, ggml_tensor * positions, ggml_tensor * mod_attn, ggml_tensor * mod_ffn, + const Config & cfg, int64_t seq, ggml_tensor * cached_K, ggml_tensor * cached_V) { - const int64_t hd = cfg.head_dim; - const int64_t nq = cfg.n_q_heads; - const int64_t nkv = cfg.n_kv_heads; - const int64_t qf = nq * hd; - const int64_t h = cfg.expert_h; + const int64_t h = cfg.expert_h; ggml_tensor * gate_attn = nullptr; - ggml_tensor * x_norm = build_adarms(ctx, x_in, w.ada_in_w, w.ada_in_b, cond, h, cfg.rms_eps, &gate_attn); - - ggml_tensor * q = ggml_mul_mat(ctx, w.Wq, x_norm); - ggml_tensor * k = ggml_mul_mat(ctx, w.Wk, x_norm); - ggml_tensor * v = ggml_mul_mat(ctx, w.Wv, x_norm); - - ggml_tensor * q_h = ggml_reshape_3d(ctx, q, hd, nq, seq); - ggml_tensor * k_h = ggml_reshape_3d(ctx, k, hd, nkv, seq); - ggml_tensor * v_h = ggml_reshape_3d(ctx, v, hd, nkv, seq); - - auto rope_call = [&](ggml_tensor * t) { - return ggml_rope_ext(ctx, t, positions, nullptr, - (int) hd, GGML_ROPE_TYPE_NEOX, 0, - rope_base, 1.f, 0.f, 1.f, 32.f, 1.f); - }; - ggml_tensor * q_rope = rope_call(q_h); - ggml_tensor * k_rope = rope_call(k_h); - - ggml_tensor * K_full = ggml_concat(ctx, cached_K, k_rope, 2); - ggml_tensor * V_full = ggml_concat(ctx, cached_V, v_h, 2); - - ggml_tensor * Q = ggml_cont(ctx, ggml_permute(ctx, q_rope, 0, 2, 1, 3)); - ggml_tensor * K = ggml_cont(ctx, ggml_permute(ctx, K_full, 0, 2, 1, 3)); - ggml_tensor * V = ggml_cont(ctx, ggml_permute(ctx, V_full, 1, 2, 0, 3)); - - ggml_tensor * kq = ggml_mul_mat(ctx, K, Q); - ggml_mul_mat_set_prec(kq, GGML_PREC_F32); - const float scale = 1.f/std::sqrt((float) hd); - ggml_tensor * attn = ggml_soft_max_ext(ctx, kq, nullptr, scale, 0.f); - ggml_tensor * kqv = ggml_mul_mat(ctx, V, attn); - - ggml_tensor * att_pre = ggml_reshape_2d(ctx, - ggml_cont(ctx, ggml_permute(ctx, kqv, 0, 2, 1, 3)), qf, seq); - ggml_tensor * o_out = ggml_mul_mat(ctx, w.Wo, att_pre); + ggml_tensor * x_norm = build_adarms(ctx, x_in, mod_attn, h, cfg.rms_eps, &gate_attn); + ggml_tensor * o_out = gemma_attn(ctx, w, x_norm, positions, cfg, seq, cached_K, cached_V, nullptr, + nullptr, nullptr, GGML_TYPE_F32, false); ggml_tensor * h1 = ggml_add(ctx, x_in, ggml_mul(ctx, o_out, gate_attn)); ggml_tensor * gate_ffn = nullptr; - ggml_tensor * x_norm_mlp = build_adarms(ctx, h1, w.ada_post_w, w.ada_post_b, cond, h, cfg.rms_eps, &gate_ffn); - ggml_tensor * gate = ggml_mul_mat(ctx, w.Wgate, x_norm_mlp); - ggml_tensor * up = ggml_mul_mat(ctx, w.Wup, x_norm_mlp); - ggml_tensor * inter_t = ggml_mul(ctx, vla::gelu(ctx, gate), up); - ggml_tensor * mlp_out = ggml_mul_mat(ctx, w.Wdown, inter_t); + ggml_tensor * x_norm_mlp = build_adarms(ctx, h1, mod_ffn, h, cfg.rms_eps, &gate_ffn); + ggml_tensor * mlp_out = gemma_mlp(ctx, w, x_norm_mlp, GGML_TYPE_F32); return ggml_add(ctx, h1, ggml_mul(ctx, mlp_out, gate_ffn)); } -bool load_config(const gguf_reader & g, Config & cfg) { - auto need = [&](const char * k) { - if (!g.has(k)) { - std::fprintf(stderr, "vla(pi05): gguf missing key %s\n", k); - return false; +bool bake_adarms(gguf_reader & g, Pi05ModelArch & m) { + const Config & cfg = m.cfg; + const int64_t h = cfg.expert_h, nj = 2*cfg.n_layers+1, ns = cfg.num_steps; + const float dt = -1.0f/(float) ns; + + ggml_init_params wp = { (size_t) (2*nj+4)*ggml_tensor_overhead(), nullptr, true }; + std::unique_ptr wctx(ggml_init(wp), ggml_free); + if (!wctx) { + std::fprintf(stderr, "vla(pi05): ggml_init(adaRMS weights) failed\n"); + return false; + } + WeightLoader L("pi05", g, wctx.get(), m.matmul_type); + std::vector W(nj), B(nj); + for (int64_t i=0; i wbuf(wbuf_raw, ggml_backend_buffer_free); + if (!uploaded) + return false; + + const size_t max_nodes = (size_t) ns*(2*nj+8) + 64; + scratch_ctx sc; + ggml_context * C = sc.reset(ggml_tensor_overhead()*max_nodes + ggml_graph_overhead_custom(max_nodes, false)); + if (!C) { + std::fprintf(stderr, "vla(pi05): ggml_init(adaRMS graph) failed\n"); + return false; + } + ggml_cgraph * gf = ggml_new_graph_custom(C, max_nodes, false); + std::vector t_time(ns), mods((size_t) ns*nj); + for (int64_t s=0; s tv = sinusoidal_time_emb(1.0f+(float) s * dt, h, cfg.min_period, cfg.max_period); + ggml_backend_tensor_set(t_time[s], tv.data(), 0, ggml_nbytes(t_time[s])); + } + graph_unique_names(gf); + if (ggml_backend_graph_compute(m.backend, gf) != GGML_STATUS_SUCCESS) { + std::fprintf(stderr, "vla(pi05): adaRMS compute failed\n"); + return false; + } + std::vector table((size_t) ns*nj*3*h); + for (size_t k=0; k & dst) { - const ggml_tensor * t = g.meta(name); - if (!t) { - std::fprintf(stderr, "vla(pi05): %s missing - identity\n", name); - return true; - } - if (t->ne[0] != (int64_t) dst.size()) { - std::fprintf(stderr, "vla(pi05): %s is %lld wide, expected %zu\n", - name, (long long) t->ne[0], dst.size()); - return false; - } - if (!g.read_raw(name, dst.data(), dst.size()*sizeof(float))) { - std::fprintf(stderr, "vla(pi05): %s read failed\n", name); - return false; - } - return true; - }; bool ok = true; - ok &= read1d("state_mean", m.state_mean); - ok &= read1d("state_std", m.state_std); - ok &= read1d("action_mean", m.action_mean); - ok &= read1d("action_std", m.action_std); + ok &= read_pi_stat(g, "action_mean", m.action_mean); + ok &= read_pi_stat(g, "action_std", m.action_std); if (m.quantile_norm) { m.action_q01.assign(cfg.real_action_dim, -1.f); m.action_q99.assign(cfg.real_action_dim, 1.f); - ok &= read1d("action_q01", m.action_q01); - ok &= read1d("action_q99", m.action_q99); + ok &= read_pi_stat(g, "action_q01", m.action_q01); + ok &= read_pi_stat(g, "action_q99", m.action_q99); } return ok; } @@ -380,6 +226,10 @@ Pi05ModelArch::~Pi05ModelArch() { ggml_backend_buffer_free(weight_buf); if (ctx_weights) ggml_free(ctx_weights); + if (const_buf) + ggml_backend_buffer_free(const_buf); + if (ctx_const) + ggml_free(ctx_const); if (backend) ggml_backend_free(backend); } @@ -390,41 +240,23 @@ std::unique_ptr pi05_create(const std::string& mmproj_path, const Options& opts) { (void) config_path; - if (!ends_with(ckpt_path, ".gguf")) { - std::fprintf(stderr, - "vla(pi05): ckpt must be a GGUF produced by scripts/convert_pi05_to_gguf.py " - "(got '%s')\n", ckpt_path.c_str()); - return nullptr; - } - auto m = std::make_unique(); - m->ckpt_path_ = ckpt_path; m->matmul_type = opts.weight_dtype.value_or(vla::default_weight_dtype(GGML_TYPE_BF16)); - if (!m->io.open(ckpt_path)) - return nullptr; gguf_reader & g = m->io; - if (!g.has("pi05.architecture") || g.str("pi05.architecture") != "pi05") { - std::fprintf(stderr, "vla(pi05): '%s' is not a π0.5 GGUF (pi05.architecture missing/wrong)\n", - ckpt_path.c_str()); - return nullptr; - } - if (!load_config(g, m->cfg)) + if (!load_pi_config(g, ckpt_path, 0, m->cfg) || !resolve_num_steps("pi05", opts, m->cfg.num_steps)) return nullptr; const Config & cfg = m->cfg; - m->adarms_cond_dim = g.has("pi05.adarms_cond_dim") ? g.u32("pi05.adarms_cond_dim") : cfg.expert_h; - m->quantile_norm = g.has("pi05.norm_mode") && g.str("pi05.norm_mode") == "quantiles"; + m->quantile_norm = g.has("pi05.norm_mode") && g.str("pi05.norm_mode") == "quantiles"; std::printf("vla(pi05): hidden=%lld inter=%lld heads=%lldq/%lldkv x%lld n_layers=%lld " "expert_h=%lld expert_inter=%lld chunk=%lld steps=%d real_state=%lld real_action=%lld " - "max_len=%lld adarms_cond=%lld matmul_weights=%s\n", + "max_len=%lld matmul_weights=%s\n", (long long) cfg.hidden, (long long) cfg.intermediate, (long long) cfg.n_q_heads, (long long) cfg.n_kv_heads, (long long) cfg.head_dim, (long long) cfg.n_layers, (long long) cfg.expert_h, (long long) cfg.expert_inter, (long long) cfg.n_suffix, cfg.num_steps, (long long) cfg.real_state_dim, (long long) cfg.real_action_dim, - (long long) cfg.n_lang, (long long) m->adarms_cond_dim, - m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); + (long long) cfg.n_lang, m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); - m->n_threads = default_cpu_threads(); { const Backend b = backend_init("vla(pi05)", m->n_threads); if (!b.handle) { @@ -435,20 +267,8 @@ std::unique_ptr pi05_create(const std::string& mmproj_path, // The SigLIP tower is now bundled in the ckpt GGUF; mmproj_path is ignored. (void) mmproj_path; - { - auto vu = [&](const char * k, int64_t & d) { if (g.has(k)) d = (int64_t) g.u32(k); }; - vu("pi05.vit_hidden", m->vit_hidden); vu("pi05.vit_layers", m->vit_layers); - vu("pi05.vit_heads", m->vit_heads); vu("pi05.image_size", m->vit_image_size); - vu("pi05.patch_size", m->vit_patch_size); vu("pi05.n_img_tokens", m->vit_n_tokens); - if (g.has("pi05.vit_ln_eps")) - m->vit_ln_eps = g.f32("pi05.vit_ln_eps"); - const int64_t grid = m->vit_image_size/m->vit_patch_size; - if (grid * grid != m->vit_n_tokens || m->vit_n_tokens != cfg.n_img) { - std::fprintf(stderr, "vla(pi05): vit geometry mismatch (grid^2=%lld n_img_tokens=%lld cfg.n_img=%lld)\n", - (long long) (grid * grid), (long long) m->vit_n_tokens, (long long) cfg.n_img); - return nullptr; - } - } + if (!m->vis.load(g, cfg.n_img)) + return nullptr; { ggml_init_params wp = { (size_t) 16*1024*1024, nullptr, true }; @@ -460,33 +280,23 @@ std::unique_ptr pi05_create(const std::string& mmproj_path, } WeightLoader L("pi05", g, m->ctx_weights, m->matmul_type); - m->vit.declare(L, "vit", m->vit_layers); - m->mm_proj_w = L.gemm ("mm.proj.weight"); - m->mm_proj_b = L.opt_f32("mm.proj.bias"); + m->vis.declare(L); m->pl.declare(L, "vlm", cfg.n_layers, false); m->ex_layers.resize(cfg.n_layers); for (int64_t i=0; iex_layers[i]; - w.ada_in_w = L.f32 ("aex.blk.%lld.attn_norm.weight", (long long)i); - w.ada_in_b = L.f32 ("aex.blk.%lld.attn_norm.bias", (long long)i); - w.Wq = L.gemm("aex.blk.%lld.attn_q.weight", (long long)i); - w.Wk = L.gemm("aex.blk.%lld.attn_k.weight", (long long)i); - w.Wv = L.gemm("aex.blk.%lld.attn_v.weight", (long long)i); - w.Wo = L.gemm("aex.blk.%lld.attn_o.weight", (long long)i); - w.ada_post_w = L.f32 ("aex.blk.%lld.ffn_norm.weight", (long long)i); - w.ada_post_b = L.f32 ("aex.blk.%lld.ffn_norm.bias", (long long)i); - w.Wgate = L.gemm("aex.blk.%lld.ffn_gate.weight", (long long)i); - w.Wup = L.gemm("aex.blk.%lld.ffn_up.weight", (long long)i); - w.Wdown = L.gemm("aex.blk.%lld.ffn_down.weight", (long long)i); + GemmaLayerW & w = m->ex_layers[i]; + w.Wq = L.gemm("aex.blk.%lld.attn_q.weight", (long long)i); + w.Wk = L.gemm("aex.blk.%lld.attn_k.weight", (long long)i); + w.Wv = L.gemm("aex.blk.%lld.attn_v.weight", (long long)i); + w.Wo = L.gemm("aex.blk.%lld.attn_o.weight", (long long)i); + w.Wgate = L.gemm("aex.blk.%lld.ffn_gate.weight", (long long)i); + w.Wup = L.gemm("aex.blk.%lld.ffn_up.weight", (long long)i); + w.Wdown = L.gemm("aex.blk.%lld.ffn_down.weight", (long long)i); } - m->ex_final_w = L.f32("aex.output_norm.weight"); - m->ex_final_b = L.f32("aex.output_norm.bias"); m->W_ain = L.f32("action_in_proj.weight"); m->b_ain = L.f32("action_in_proj.bias"); - m->W_tin = L.f32("time_mlp_in.weight"); m->b_tin = L.f32("time_mlp_in.bias"); - m->W_tout = L.f32("time_mlp_out.weight"); m->b_tout = L.f32("time_mlp_out.bias"); m->W_aout = L.f32("action_out_proj.weight"); m->b_aout = L.f32("action_out_proj.bias"); if (!L.upload(m->backend, &m->weight_buf)) @@ -495,7 +305,7 @@ std::unique_ptr pi05_create(const std::string& mmproj_path, std::printf("vla(pi05): resident weights = %.2f GiB\n", ggml_backend_buffer_get_size(m->weight_buf)/(1024.0*1024.0*1024.0)); - if (!load_stats(g, *m)) + if (!bake_adarms(g, *m) || !load_stats(g, *m)) return nullptr; std::printf("vla(pi05): model loaded (n_threads=%d)\n", m->n_threads); return m; @@ -515,66 +325,20 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { const int64_t max_ad = cfg.max_action_dim; const int num_steps = cfg.num_steps; const float dt = -1.0f/(float) num_steps; - const float rope_base = cfg.rope_freq_base; - std::vector img_emb_host; int64_t n_img_tokens = 0; if (in.precomputed_img_emb) { + if (in.n_img_views < 1) { + std::fprintf(stderr, "vla(pi05): precomputed_img_emb set but n_img_views=%d\n", in.n_img_views); + return {}; + } n_img_tokens = (int64_t) in.n_img_views*cfg.n_img; - img_emb_host.assign(in.precomputed_img_emb, - in.precomputed_img_emb+(size_t) n_img_tokens * hidden_pl); } else { if (in.n_images < 1 || !in.images) { std::fprintf(stderr, "vla(pi05): predict: no images and no precomputed_img_emb\n"); return {}; } - const int64_t K = vit_n_tokens, H = hidden_pl, grid = vit_image_size/vit_patch_size; - n_img_tokens = (int64_t) in.n_images*K; - img_emb_host.assign((size_t) in.n_images*K * H, 0.0f); - - ggml_context * VC = vision_scratch.reset((size_t) 128*1024*1024); - if (!VC) { std::fprintf(stderr, "vla(pi05): ggml_init(vision ctx) failed\n"); return {}; } - ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, vit_image_size, vit_image_size, 3); ggml_set_input(t_px); - ggml_tensor * conv = ggml_conv_2d(VC, vit.patch_w, t_px, (int) vit_patch_size, (int) vit_patch_size, 0, 0, 1, 1); - ggml_tensor * patches = ggml_cont(VC, ggml_transpose(VC, ggml_reshape_2d(VC, conv, grid * grid, vit_hidden))); - ggml_tensor * h = ggml_add(VC, ggml_add(VC, patches, vit.patch_b), vit.pos); - for (int64_t i=0; ine[0])); - ggml_set_output(vit_emb); - - ggml_cgraph * vg = ggml_new_graph_custom(VC, 8192, false); - ggml_build_forward_expand(vg, vit_emb); - - if (!vision_scratch.alloc(backend, vg)) { - std::fprintf(stderr, "vla(pi05): vision gallocr alloc failed\n"); - return {}; - } - const auto tv0 = clk::now(); - std::vector chw; - for (int v=0; v(clk::now()-tv0).count(); - - // Undo the 1/sqrt(hidden) the shared vision graph applies; pi05 wants raw - // projector features. Inside this branch on purpose: precomputed_img_emb - // replaces the tower and is already LM-ready. - const float img_scale = (float) std::sqrt((double) hidden_pl); - for (float & x : img_emb_host) - x *= img_scale; + n_img_tokens = (int64_t) in.n_images*vis.n_tokens; } if (in.n_lang < 1 || !in.lang_tokens) { @@ -592,7 +356,9 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { // Prefix + expert graph depends only on the token counts and step count. const MainKey mkey{ n_img_tokens, n_lang, num_steps }; - const bool built = main_graph.ensure(backend, mkey, (size_t) 64*1024*1024, + const size_t max_nodes = (size_t) 64*n_layers*(num_steps+1) + 1024; + const bool built = main_graph.ensure(backend, mkey, + ggml_tensor_overhead()*max_nodes + ggml_graph_overhead_custom(max_nodes, false), [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { ggml_tensor * t_image_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_img_tokens); ggml_set_input(t_image_emb); ggml_tensor * t_lang_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_lang); ggml_set_input(t_lang_emb); @@ -600,12 +366,6 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { ggml_tensor * t_x0 = ggml_new_tensor_2d(C, GGML_TYPE_F32, max_ad, chunk); ggml_set_input(t_x0); ggml_tensor * t_suffix_pos= ggml_new_tensor_1d(C, GGML_TYPE_I32, n_suf); ggml_set_input(t_suffix_pos); - std::vector t_time(num_steps); - for (int s=0; s Pi05ModelArch::predict(const Inputs& in) { { ggml_tensor * h = prefix_embs; for (int64_t i=0; inb[2]+j*ada_mod->nb[1]); + }; ggml_tensor * h = ggml_add(C, ggml_mul_mat(C, W_ain, x_t), b_ain); for (int64_t i=0; i Pi05ModelArch::predict(const Inputs& in) { ggml_set_output(x_final); gio.t_image_emb=t_image_emb; gio.t_lang_emb=t_lang_emb; gio.t_prefix_pos=t_prefix_pos; - gio.t_x0=t_x0; gio.t_suffix_pos=t_suffix_pos; gio.t_time=t_time; gio.x_final=x_final; + gio.t_x0=t_x0; gio.t_suffix_pos=t_suffix_pos; gio.x_final=x_final; - ggml_cgraph * gf = ggml_new_graph_custom(C, 16384, false); + ggml_cgraph * gf = ggml_new_graph_custom(C, max_nodes, false); ggml_build_forward_expand(gf, x_final); return gf; }); @@ -652,9 +412,45 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { ggml_tensor * t_image_emb = gio.t_image_emb, * t_lang_emb = gio.t_lang_emb; ggml_tensor * t_prefix_pos = gio.t_prefix_pos, * t_x0 = gio.t_x0; ggml_tensor * t_suffix_pos = gio.t_suffix_pos, * x_final = gio.x_final; - std::vector & t_time = gio.t_time; - ggml_backend_tensor_set(t_image_emb, img_emb_host.data(), 0, ggml_nbytes(t_image_emb)); + if (in.precomputed_img_emb) { + ggml_backend_tensor_set(t_image_emb, in.precomputed_img_emb, 0, ggml_nbytes(t_image_emb)); + } else { + const int64_t K = vis.n_tokens, H = hidden_pl, grid = vis.image_size/vis.patch_size; + ggml_context * VC = vision_scratch.reset((size_t) 128*1024*1024); + if (!VC) { std::fprintf(stderr, "vla(pi05): ggml_init(vision ctx) failed\n"); return {}; } + ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, vis.image_size, vis.image_size, 3); ggml_set_input(t_px); + ggml_tensor * h = vis.vit.build(VC, vis.vit.embed_conv(VC, t_px, vis.patch_size, grid), K); + // PaliGemma projector: linear (+ optional bias). + ggml_tensor * vit_emb = linear(VC, vis.proj_w, vis.proj_b, h); + ggml_set_output(vit_emb); + + ggml_cgraph * vg = ggml_new_graph_custom(VC, 8192, false); + ggml_build_forward_expand(vg, vit_emb); + + if (!vision_scratch.alloc(backend, vg)) { + std::fprintf(stderr, "vla(pi05): vision gallocr alloc failed\n"); + return {}; + } + const auto tv0 = clk::now(); + std::vector chw; + for (int v=0; vnb[1], (size_t) v * K * t_image_emb->nb[1]); + if (ggml_backend_view_init(dst) != GGML_STATUS_SUCCESS) { + std::fprintf(stderr, "vla(pi05): image embedding view failed (view %d)\n", v); + return {}; + } + ggml_backend_tensor_copy(vit_emb, dst); + } + stats.ms_vision = std::chrono::duration(clk::now()-tv0).count(); + } ggml_backend_tensor_set(t_lang_emb, lang_rows.data(), 0, ggml_nbytes(t_lang_emb)); { std::vector pp(n_prefix); for (int64_t i=0; i Pi05ModelArch::predict(const Inputs& in) { ggml_backend_tensor_set(t_suffix_pos, sp.data(), 0, ggml_nbytes(t_suffix_pos)); } { - std::vector x0h((size_t) max_ad * chunk); - if (in.noise) - std::memcpy(x0h.data(), in.noise, x0h.size()*sizeof(float)); - else { - std::normal_distribution nd(0.f, 1.f); - for (auto & v : x0h) - v = nd(rng); - } + std::vector x0h; + init_noise(in, (size_t) max_ad * chunk, x0h); ggml_backend_tensor_set(t_x0, x0h.data(), 0, ggml_nbytes(t_x0)); } - for (int s=0; s tv = sinusoidal_time_emb(timestep, hidden_ex, cfg.min_period, cfg.max_period); - ggml_backend_tensor_set(t_time[s], tv.data(), 0, ggml_nbytes(t_time[s])); - } - graph_unique_names(gf); const auto ti0 = clk::now(); const ggml_status st = ggml_backend_graph_compute(backend, gf); @@ -691,17 +475,15 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { std::vector out((size_t) chunk * max_ad); ggml_backend_tensor_get(x_final, out.data(), 0, out.size()*sizeof(float)); - if (!vla::env_flag("VLA_PI05_SKIP_UNNORM")) { - for (int64_t t=0; t= cfg.real_action_dim) - row[j] = 0.0f; - else if (quantile_norm) - row[j] = (row[j]+1.0f)*(action_q99[j]-action_q01[j])*0.5f+action_q01[j]; - else - row[j] = row[j]*(action_std[j]+cfg.norm_eps)+action_mean[j]; - } + for (int64_t t=0; t= cfg.real_action_dim) + row[j] = 0.0f; + else if (quantile_norm) + row[j] = (row[j]+1.0f)*(action_q99[j]-action_q01[j])*0.5f+action_q01[j]; + else + row[j] = row[j]*(action_std[j]+cfg.norm_eps)+action_mean[j]; } } diff --git a/src/models/smolvla.cpp b/src/models/smolvla.cpp index d9fd0fc..41f991c 100644 --- a/src/models/smolvla.cpp +++ b/src/models/smolvla.cpp @@ -16,21 +16,24 @@ // distilled action-expert weights and force num_steps = 1 at the denoise loops. #include "arch.h" -#include "modules/encoder.h" +#include "gguf_reader.h" +#include "layers/attn.h" +#include "layers/embed.h" +#include "layers/ffn.h" +#include "layers/norm.h" +#include "layers/rope.h" +#include "modules/siglip_vit.h" #include "options.h" #include "model.h" #include "modules/preprocess.h" #include "scratch_ctx.h" -#include "layers/embed.h" #include "ggml.h" #include "ggml-backend.h" -#include "ggml-cpu.h" #include "gguf.h" #include "backend.h" #include "nlohmann/json.hpp" -#include "env_flag.h" #include #include @@ -39,8 +42,10 @@ #include #include #include +#include #include #include +#include #include namespace vla { @@ -148,40 +153,8 @@ struct safetensors { } }; -struct gguf_source { - struct gguf_context * gctx = nullptr; - struct ggml_context * meta_ctx = nullptr; - FILE * fp = nullptr; - size_t data_off = 0; - - bool open(const std::string & path) { - gguf_init_params p{}; - p.no_alloc = true; - p.ctx = &meta_ctx; - gctx = gguf_init_from_file(path.c_str(), p); - if (!gctx) { - std::fprintf(stderr, "vla: gguf_init_from_file failed for %s\n", path.c_str()); - return false; - } - fp = std::fopen(path.c_str(), "rb"); - if (!fp) { - std::fprintf(stderr, "vla: fopen failed for %s\n", path.c_str()); - gguf_free(gctx); gctx = nullptr; - ggml_free(meta_ctx); meta_ctx = nullptr; - return false; - } - data_off = gguf_get_data_offset(gctx); - return true; - } - - ~gguf_source() { - if (fp) - std::fclose(fp); - if (gctx) - gguf_free(gctx); - if (meta_ctx) - ggml_free(meta_ctx); - } +struct gguf_source : gguf_reader { + gguf_source() : gguf_reader("smolvla") {} static bool shape_matches(const ggml_tensor * t, const std::vector & pt_shape) { const int nd_used = std::max(1, (int) pt_shape.size()); @@ -201,131 +174,50 @@ struct gguf_source { bool read_to_f32(const std::string & name, float * dst, const std::vector & expected_shape) { - ggml_tensor * t = ggml_get_tensor(meta_ctx, name.c_str()); + const ggml_tensor * t = meta(name.c_str()); if (!t) { - std::fprintf(stderr, "vla: gguf tensor not found: %s\n", name.c_str()); + std::fprintf(stderr, "vla(smolvla): gguf tensor not found: %s\n", name.c_str()); return false; } if (!shape_matches(t, expected_shape)) { - std::fprintf(stderr, "vla: gguf shape mismatch for %s\n", name.c_str()); + std::fprintf(stderr, "vla(smolvla): gguf shape mismatch for %s\n", name.c_str()); return false; } - const int64_t id = gguf_find_tensor(gctx, name.c_str()); - const size_t offset = data_off+gguf_get_tensor_offset(gctx, id); - const size_t bytes = gguf_get_tensor_size(gctx, id); - if (vla_fseek64(fp, offset) != 0) { - std::fprintf(stderr, "vla: fseek failed for %s\n", name.c_str()); + const size_t bytes = ggml_nbytes(t); + if (t->type == GGML_TYPE_F32) + return read_raw(name.c_str(), dst, bytes); + // A requantized file (scripts/quantize_gguf.py) may pack a tensor + // this model keeps float; unpack it rather than refuse the file. + const ggml_type_traits * tt = ggml_get_type_traits(t->type); + if (!tt->to_float) { + std::fprintf(stderr, "vla(smolvla): gguf cannot convert %s for %s\n", + ggml_type_name(t->type), name.c_str()); return false; } - if (t->type == GGML_TYPE_F32) { - if (std::fread(dst, 1, bytes, fp) != bytes) - return false; - } else if (t->type == GGML_TYPE_BF16) { - std::vector tmp(bytes/sizeof(ggml_bf16_t)); - if (std::fread(tmp.data(), 1, bytes, fp) != bytes) - return false; - ggml_bf16_to_fp32_row(tmp.data(), dst, tmp.size()); - } else if (t->type == GGML_TYPE_F16 || ggml_is_quantized(t->type)) { - // A requantized file (scripts/quantize_gguf.py) may pack a tensor - // this model keeps float; unpack it rather than refuse the file. - const ggml_type_traits * tt = ggml_get_type_traits(t->type); - if (!tt->to_float) { - std::fprintf(stderr, "vla: gguf cannot dequantize %s for %s\n", - ggml_type_name(t->type), name.c_str()); - return false; - } - std::vector tmp(bytes); - if (std::fread(tmp.data(), 1, bytes, fp) != bytes) - return false; - tt->to_float(tmp.data(), dst, ggml_nelements(t)); - } else { - std::fprintf(stderr, "vla: gguf unsupported dtype %d for %s\n", - (int) t->type, name.c_str()); + std::vector tmp(bytes); + if (!read_raw(name.c_str(), tmp.data(), bytes)) return false; - } + tt->to_float(tmp.data(), dst, ggml_nelements(t)); return true; } /// Type of a tensor in the file, or GGML_TYPE_COUNT if it is absent. ggml_type file_type(const std::string & name) const { - const ggml_tensor * t = ggml_get_tensor(meta_ctx, name.c_str()); + const ggml_tensor * t = meta(name.c_str()); return t ? t->type : GGML_TYPE_COUNT; } - /// Packed bytes, stored as they are: the caller made the tensor that type. + /// Bytes stored as they are: the caller made the tensor that type. bool read_packed(const std::string & name, void * dst, ggml_type want, size_t expected_bytes) { - const int64_t id = gguf_find_tensor(gctx, name.c_str()); - if (id < 0 || file_type(name) != want || gguf_get_tensor_size(gctx, id) != expected_bytes) { - std::fprintf(stderr, "vla: gguf bad packed read for %s\n", name.c_str()); - return false; - } - if (vla_fseek64(fp, data_off+gguf_get_tensor_offset(gctx, id)) != 0) - return false; - return std::fread(dst, 1, expected_bytes, fp) == expected_bytes; - } - - bool read_raw(const std::string & name, void * dst, size_t expected_bytes, - const char * expected_dtype) { - ggml_tensor * t = ggml_get_tensor(meta_ctx, name.c_str()); - if (!t) { - std::fprintf(stderr, "vla: gguf tensor not found: %s\n", name.c_str()); + if (file_type(name) != want) { + std::fprintf(stderr, "vla(smolvla): gguf bad packed read for %s\n", name.c_str()); return false; } - const int64_t id = gguf_find_tensor(gctx, name.c_str()); - const size_t bytes = gguf_get_tensor_size(gctx, id); - const bool ok_dtype = (std::strcmp(expected_dtype, "BF16") == 0 && t->type == GGML_TYPE_BF16) - || (std::strcmp(expected_dtype, "F32") == 0 && t->type == GGML_TYPE_F32); - if (bytes != expected_bytes || !ok_dtype) { - std::fprintf(stderr, "vla: gguf bad raw read for %s\n", name.c_str()); - return false; - } - const size_t offset = data_off+gguf_get_tensor_offset(gctx, id); - if (vla_fseek64(fp, offset) != 0) - return false; - return std::fread(dst, 1, bytes, fp) == bytes; - } - - int64_t find_key(const char * key) const { - return gguf_find_key(gctx, key); - } - bool has_key(const char * key) const { - return find_key(key) >= 0; - } - uint32_t get_u32(const char * key) const { - return gguf_get_val_u32(gctx, find_key(key)); - } - int32_t get_i32(const char * key) const { - return gguf_get_val_i32(gctx, find_key(key)); - } - float get_f32(const char * key) const { - return gguf_get_val_f32(gctx, find_key(key)); - } - double get_f64(const char * key) const { - return gguf_get_val_f64(gctx, find_key(key)); + return read_raw(name.c_str(), dst, expected_bytes); } - std::string get_str(const char * key) const { - return gguf_get_val_str(gctx, find_key(key)); - } - - bool has_tensor(const char * name) const { - return ggml_get_tensor(meta_ctx, name) != nullptr; - } -}; - -struct VlmLayerW { - ggml_tensor * Wln_in; - ggml_tensor * Wq; - ggml_tensor * Wk; - ggml_tensor * Wv; - ggml_tensor * Wo; - ggml_tensor * Wln_post; - ggml_tensor * Wgate; - ggml_tensor * Wup; - ggml_tensor * Wdown; }; -struct ExpertLayerW { - bool is_self_attn; +struct LayerW { ggml_tensor * Wln_in; ggml_tensor * Wq; ggml_tensor * Wk; @@ -337,7 +229,9 @@ struct ExpertLayerW { ggml_tensor * Wdown; }; -// SigLIP-B/16 vision block weights (SmolVLM2 tower, built in-tree). +bool expert_self_attn(const Config & cfg, int64_t i) { + return cfg.self_attn_every_n > 0 && i%cfg.self_attn_every_n == 0; +} } @@ -351,27 +245,27 @@ struct SmolVLAModelArch : public ModelArchBase { int64_t vit_hidden = 768, vit_layers = 12, vit_heads = 12, vit_inter = 3072; int64_t vit_patch = 16, vit_image = 512, vit_scale = 4, vit_n_tokens = 64; float vit_ln_eps = 1e-6f; - ggml_tensor * vit_patch_w = nullptr, * vit_patch_b = nullptr, * vit_pos = nullptr; - ggml_tensor * vit_post_ln_w = nullptr, * vit_post_ln_b = nullptr, * mm_fc = nullptr; - std::vector vit; + SigLipTower vit; + ggml_tensor * mm_fc = nullptr; ggml_backend_t backend = nullptr; ggml_backend_buffer_t weight_buf = nullptr; + bool is_cuda = false; ggml_type weight_dtype = GGML_TYPE_BF16; ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; - scratch_ctx connector_scratch; - ggml_tensor * E_lang = nullptr; + gguf_source gst; + std::vector E_lang; + int64_t n_vocab = 0; ggml_tensor * Wstate = nullptr; ggml_tensor * bstate = nullptr; - std::vector vlm_layers; - ggml_tensor * Wnorm_vlm = nullptr; + std::vector vlm_layers; - std::vector expert_layers; + std::vector expert_layers; ggml_tensor * Wnorm_expert = nullptr; ggml_tensor * W_ain = nullptr; @@ -389,30 +283,20 @@ struct SmolVLAModelArch : public ModelArchBase { std::mt19937 rng{std::random_device{}()}; - ggml_context * ctx_compute = nullptr; - ggml_gallocr_t galloc = nullptr; - ggml_cgraph * gf_cached = nullptr; - int cached_n_views = 0; - - ggml_tensor * in_img_emb = nullptr; - ggml_tensor * in_lang_ids = nullptr; - ggml_tensor * in_state = nullptr; - ggml_tensor * in_x0 = nullptr; - ggml_tensor * in_mask_prefill = nullptr; - ggml_tensor * in_pos_prefill = nullptr; - ggml_tensor * in_mask_full = nullptr; - ggml_tensor * in_mask_pfx_only = nullptr; - ggml_tensor * in_pos_full = nullptr; - ggml_tensor * in_pos_rebased = nullptr; - - ggml_tensor * out_x_t = nullptr; + struct MainIO { + ggml_tensor *img_emb = nullptr, *lang_emb = nullptr, *state = nullptr, *x0 = nullptr; + ggml_tensor *mask_prefill = nullptr, *pos_prefill = nullptr, *mask_full = nullptr; + ggml_tensor *mask_pfx_only = nullptr, *pos_full = nullptr, *pos_rebased = nullptr, *x_t = nullptr; + std::vector k_cache, v_cache, k_leaf, v_leaf; + }; + graph_cache main_graph; std::vector time_bcasts; }; namespace { -// Fused attention in the SigLIP tower. OPT-IN (VLA_SMOLVLA_FA=1), not default. +// Fused attention in the SigLIP tower. OPT-IN (--flash-attn), not default. // It cuts the vision stage from 33.2 ms to 22.1 ms (total 68.5 -> 55.8 ms), which // is enough to beat compiled PyTorch — but ggml's CUDA flash attention computes // K/V at F16 regardless of input type (fattn.cu accepts F32 K/V only by @@ -422,15 +306,13 @@ namespace { // One pre-norm SigLIP encoder block (SmolVLM2 tower), same graph as the other // in-tree models. Bidirectional attention, F32 score accumulation, tanh GELU. -ggml_tensor * build_siglip_layer(ggml_context * C, const EncBlockW & w, ggml_tensor * x, - int64_t seq, int64_t heads, int64_t head_dim, int64_t hidden, float ln_eps) { - const float scale = 1.0f/std::sqrt((float) head_dim); - ggml_tensor * n1 = ggml_add(C, ggml_mul(C, ggml_norm(C, x, ln_eps), w.ln1w), w.ln1b); - ggml_tensor * q = ggml_add(C, ggml_mul_mat(C, w.Wq, n1), w.bq); - ggml_tensor * k = ggml_add(C, ggml_mul_mat(C, w.Wk, n1), w.bk); - ggml_tensor * v = ggml_add(C, ggml_mul_mat(C, w.Wv, n1), w.bv); - ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, head_dim, heads, seq), 0, 2, 1, 3)); - ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, head_dim, heads, seq), 0, 2, 1, 3)); +ggml_tensor * build_siglip_layer(ggml_context * C, const EncCfg & c, const EncBlockW & w, + ggml_tensor * x, int64_t seq) { + const float scale = 1.0f/std::sqrt((float) c.head_dim); + ggml_tensor * n1 = layer_norm(C, x, w.ln1w, w.ln1b, c.ln_eps); + ggml_tensor * Q = to_heads(C, linear(C, w.Wq, w.bq, n1), c.head_dim, c.heads, seq); + ggml_tensor * K = to_heads(C, linear(C, w.Wk, w.bk, n1), c.head_dim, c.heads, seq); + ggml_tensor * v = linear(C, w.Wv, w.bv, n1); ggml_tensor * att; if (vla::flash_attn_enabled()) { // The tower runs 1024 tokens (512/16 grid) over 12 layers, so the @@ -440,23 +322,27 @@ ggml_tensor * build_siglip_layer(ggml_context * C, const EncBlockW & w, ggml_ten // stage is roughly half of smolvla's latency. K/V stay F32 so the // numerics match the explicit path; the expert layers below already call // this op the same way. - ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 0, 2, 1, 3)); + ggml_tensor * V = to_heads(C, v, c.head_dim, c.heads, seq); ggml_tensor * fa = ggml_flash_attn_ext(C, Q, vla::fa_kv(C, K), vla::fa_kv(C, V), nullptr, scale, 0.0f, 0.0f); - ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); - att = ggml_reshape_2d(C, fa, hidden, seq); + ggml_prec_set_acc(fa, GGML_PREC_F32); + att = ggml_reshape_2d(C, fa, c.hidden, seq); } else { - ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); - ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); - att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); + att = attention(C, Q, K, to_heads_v(C, v, c.head_dim, c.heads, seq), nullptr, scale, c.hidden, seq); } - ggml_tensor * h1 = ggml_add(C, x, ggml_add(C, ggml_mul_mat(C, w.Wo, att), w.bo)); - ggml_tensor * n2 = ggml_add(C, ggml_mul(C, ggml_norm(C, h1, ln_eps), w.ln2w), w.ln2b); - ggml_tensor * ff = ggml_add(C, ggml_mul_mat(C, w.Wfc2, vla::gelu(C, ggml_add(C, ggml_mul_mat(C, w.Wfc1, n2), w.bfc1))), w.bfc2); - return ggml_add(C, h1, ff); + ggml_tensor * h1 = ggml_add(C, x, linear(C, w.Wo, w.bo, att)); + return ggml_add(C, h1, ffn_gelu(C, w.Wfc1, w.bfc1, w.Wfc2, w.bfc2, layer_norm(C, h1, w.ln2w, w.ln2b, c.ln_eps))); } -// CHW-planar float image in [-1,1] for ggml_conv_2d (SigLIP mean/std 0.5). +void set_derived_config(Config & cfg) { + cfg.rms_eps = 1e-5f; + cfg.rope_mode = GGML_ROPE_TYPE_NEOX; + cfg.rope_freq_base = 10000.f; + + cfg.n_state = 1; + cfg.q_full_dim = cfg.n_q_heads * cfg.head_dim; + cfg.kv_full_dim = cfg.n_kv_heads*cfg.head_dim; + cfg.rope_n_dims = static_cast(cfg.head_dim); +} bool load_config_from_json(const std::string & path, Config & cfg) { std::ifstream f(path); @@ -471,6 +357,12 @@ bool load_config_from_json(const std::string & path, Config & cfg) { std::fprintf(stderr, "vla: failed to parse %s: %s\n", path.c_str(), e.what()); return false; } + for (const char * k : {"adapt_to_pi_aloha", "add_image_special_tokens"}) { + if (j.contains(k) && j[k].is_boolean() && j[k].get()) { + std::fprintf(stderr, "vla(smolvla): %s=true in %s is not supported\n", k, path.c_str()); + return false; + } + } cfg.hidden = 960; cfg.n_q_heads = 15; @@ -478,10 +370,6 @@ bool load_config_from_json(const std::string & path, Config & cfg) { cfg.head_dim = 64; cfg.intermediate = 2560; - cfg.rms_eps = 1e-5f; - cfg.rope_mode = GGML_ROPE_TYPE_NEOX; - cfg.rope_freq_base = 10000.f; - try { cfg.n_suffix = j.at("chunk_size").get(); cfg.num_steps = j.at("num_steps").get(); @@ -504,16 +392,12 @@ bool load_config_from_json(const std::string & path, Config & cfg) { return false; } - cfg.n_state = 1; - cfg.q_full_dim = cfg.n_q_heads * cfg.head_dim; - cfg.kv_full_dim = cfg.n_kv_heads*cfg.head_dim; - cfg.rope_n_dims = static_cast(cfg.head_dim); - + set_derived_config(cfg); cfg.norm_eps = 1e-8f; return true; } -bool load_normalizer_stats(const std::string & model_dir, SmolVLAModelArch & m) { +void load_normalizer_stats(const std::string & model_dir, SmolVLAModelArch & m) { const auto & cfg = m.cfg; m.state_mean .assign(cfg.real_state_dim, 0.f); @@ -582,14 +466,11 @@ bool load_normalizer_stats(const std::string & model_dir, SmolVLAModelArch & m) load_one(model_dir + "/policy_postprocessor.json", "unnormalizer_processor", "action.mean", "action.std", m.action_mean, m.action_std, cfg.real_action_dim, "action"); - return true; } -std::string default_config_path(const std::string & ckpt_path) { - const auto pos = ckpt_path.find_last_of("/\\"); - const std::string dir = (pos == std::string::npos) ? std::string(".") - : ckpt_path.substr(0, pos); - return dir + "/config.json"; +std::string dir_of(const std::string & path) { + const auto pos = path.find_last_of("/\\"); + return (pos == std::string::npos) ? std::string(".") : path.substr(0, pos); } bool ends_with_gguf(const std::string & path) { @@ -599,20 +480,17 @@ bool ends_with_gguf(const std::string & path) { } bool load_config_from_gguf(const gguf_source & st, Config & cfg) { - if (!st.has_key("smolvla.architecture")) { - std::fprintf(stderr, "vla: gguf missing key 'smolvla.architecture'\n"); - return false; - } - const std::string arch = st.get_str("smolvla.architecture"); + const std::string arch = st.str("smolvla.architecture"); if (arch != "smolvla") { - std::fprintf(stderr, "vla: gguf architecture = '%s' (expected 'smolvla')\n", + std::fprintf(stderr, "vla(smolvla): gguf architecture = '%s' (expected 'smolvla')\n", arch.c_str()); return false; } - auto need = [&](const char * key) -> bool { - if (!st.has_key(key)) { - std::fprintf(stderr, "vla: gguf missing key '%s'\n", key); + auto need = [&](const char * key, gguf_type type) -> bool { + int64_t id; + if (!st.typed_key(key, type, &id)) { + std::fprintf(stderr, "vla(smolvla): gguf missing key '%s'\n", key); return false; } return true; @@ -626,41 +504,34 @@ bool load_config_from_gguf(const gguf_source & st, Config & cfg) { "smolvla.chunk_size", "smolvla.num_steps", "smolvla.max_state_dim", "smolvla.max_action_dim", "smolvla.real_state_dim", "smolvla.real_action_dim", - "smolvla.self_attn_every_n_layers", "smolvla.tokenizer_max_length", - "smolvla.min_period", "smolvla.max_period"}) { - if (!need(k)) + "smolvla.self_attn_every_n_layers", "smolvla.tokenizer_max_length"}) { + if (!need(k, GGUF_TYPE_UINT32)) return false; } + if (!need("smolvla.min_period", GGUF_TYPE_FLOAT64) || !need("smolvla.max_period", GGUF_TYPE_FLOAT64)) + return false; - cfg.hidden = st.get_u32("smolvla.hidden"); - cfg.intermediate = st.get_u32("smolvla.intermediate"); - cfg.n_q_heads = st.get_u32("smolvla.n_q_heads"); - cfg.n_kv_heads = st.get_u32("smolvla.n_kv_heads"); - cfg.head_dim = st.get_u32("smolvla.head_dim"); - cfg.n_layers = st.get_u32("smolvla.n_layers"); - cfg.expert_h = st.get_u32("smolvla.expert_h"); - cfg.expert_inter = st.get_u32("smolvla.expert_inter"); - cfg.n_suffix = st.get_u32("smolvla.chunk_size"); - cfg.num_steps = st.get_u32("smolvla.num_steps"); - cfg.max_state_dim = st.get_u32("smolvla.max_state_dim"); - cfg.max_action_dim= st.get_u32("smolvla.max_action_dim"); - cfg.real_state_dim = st.get_u32("smolvla.real_state_dim"); - cfg.real_action_dim = st.get_u32("smolvla.real_action_dim"); - cfg.self_attn_every_n = st.get_u32("smolvla.self_attn_every_n_layers"); - cfg.n_lang = st.get_u32("smolvla.tokenizer_max_length"); - cfg.min_period = st.get_f64("smolvla.min_period"); - cfg.max_period = st.get_f64("smolvla.max_period"); - - cfg.rms_eps = 1e-5f; - cfg.rope_mode = GGML_ROPE_TYPE_NEOX; - cfg.rope_freq_base = 10000.f; - - cfg.n_state = 1; - cfg.q_full_dim = cfg.n_q_heads * cfg.head_dim; - cfg.kv_full_dim = cfg.n_kv_heads*cfg.head_dim; - cfg.rope_n_dims = static_cast(cfg.head_dim); - - cfg.norm_eps = st.has_key("smolvla.norm_eps") ? st.get_f32("smolvla.norm_eps") : 1e-8f; + cfg.hidden = st.u32("smolvla.hidden"); + cfg.intermediate = st.u32("smolvla.intermediate"); + cfg.n_q_heads = st.u32("smolvla.n_q_heads"); + cfg.n_kv_heads = st.u32("smolvla.n_kv_heads"); + cfg.head_dim = st.u32("smolvla.head_dim"); + cfg.n_layers = st.u32("smolvla.n_layers"); + cfg.expert_h = st.u32("smolvla.expert_h"); + cfg.expert_inter = st.u32("smolvla.expert_inter"); + cfg.n_suffix = st.u32("smolvla.chunk_size"); + cfg.num_steps = st.u32("smolvla.num_steps"); + cfg.max_state_dim = st.u32("smolvla.max_state_dim"); + cfg.max_action_dim= st.u32("smolvla.max_action_dim"); + cfg.real_state_dim = st.u32("smolvla.real_state_dim"); + cfg.real_action_dim = st.u32("smolvla.real_action_dim"); + cfg.self_attn_every_n = st.u32("smolvla.self_attn_every_n_layers"); + cfg.n_lang = st.u32("smolvla.tokenizer_max_length"); + cfg.min_period = st.f64("smolvla.min_period"); + cfg.max_period = st.f64("smolvla.max_period"); + + set_derived_config(cfg); + cfg.norm_eps = st.has("smolvla.norm_eps") ? st.f32("smolvla.norm_eps") : 1e-8f; return true; } @@ -709,8 +580,6 @@ std::string hf_to_gguf(const std::string & n) { if (n == "model.vlm_with_expert.vlm.model.text_model.embed_tokens.weight") return "token_embd.weight"; - if (n == "model.vlm_with_expert.vlm.model.text_model.norm.weight") - return "vlm.output_norm.weight"; if (n == "model.vlm_with_expert.lm_expert.norm.weight") return "aex.output_norm.weight"; @@ -787,101 +656,64 @@ std::string hf_to_gguf(const std::string & n) { } -ggml_tensor * rope_q_or_k(ggml_context * ctx, ggml_tensor * x, - ggml_tensor * positions, const Config & cfg) { - return ggml_rope_ext(ctx, x, positions, nullptr, - cfg.rope_n_dims, cfg.rope_mode, 0, - cfg.rope_freq_base, 1.f, - 0.f, 1.f, - 32.f, 1.f); +RopeSpec rope_spec(const Config & cfg) { + return RopeSpec{cfg.rope_mode, cfg.rope_n_dims, {}, cfg.rope_freq_base}; } -static inline bool tower_mm_f32_prec() { - return vla::mm_prec_f32_enabled(); -} -static inline ggml_tensor * mm_w(ggml_context * ctx, ggml_tensor * w, ggml_tensor * x) { +ggml_tensor * mm_w(ggml_context * ctx, ggml_tensor * w, ggml_tensor * x) { ggml_tensor * r = ggml_mul_mat(ctx, w, x); - if (tower_mm_f32_prec()) - ggml_mul_mat_set_prec(r, GGML_PREC_F32); + if (vla::mm_prec_f32_enabled()) + ggml_prec_set_acc(r, GGML_PREC_F32); return r; } -ggml_tensor * build_vlm_layer(ggml_context * ctx, const VlmLayerW & w, +ggml_tensor * attn_mlp(ggml_context * ctx, const LayerW & w, ggml_tensor * x_in, + ggml_tensor * q, ggml_tensor * k, ggml_tensor * v, ggml_tensor * mask, + const Config & cfg, int64_t seq) { + ggml_tensor * Q = ggml_permute(ctx, q, 0, 2, 1, 3); + ggml_tensor * K = ggml_permute(ctx, vla::fa_kv(ctx, k), 0, 2, 1, 3); + ggml_tensor * V = ggml_permute(ctx, vla::fa_kv(ctx, v), 0, 2, 1, 3); + const float scale = 1.f/std::sqrt(static_cast(cfg.head_dim)); + ggml_tensor * fa = ggml_flash_attn_ext(ctx, Q, K, V, mask, scale, 0.f, 0.f); + ggml_prec_set_acc(fa, GGML_PREC_F32); + ggml_tensor * h1 = ggml_add(ctx, x_in, mm_w(ctx, w.Wo, ggml_reshape_2d(ctx, fa, cfg.q_full_dim, seq))); + + ggml_tensor * x_norm_mlp = rms_norm(ctx, h1, w.Wln_post, cfg.rms_eps); + ggml_tensor * inter = ggml_mul(ctx, ggml_silu(ctx, mm_w(ctx, w.Wgate, x_norm_mlp)), + mm_w(ctx, w.Wup, x_norm_mlp)); + return ggml_add(ctx, h1, mm_w(ctx, w.Wdown, inter)); +} + +ggml_tensor * build_vlm_layer(ggml_context * ctx, const LayerW & w, ggml_tensor * x_in, ggml_tensor * mask, ggml_tensor * positions, const Config & cfg, ggml_tensor ** k_out, ggml_tensor ** v_out) { - ggml_tensor * x_norm = ggml_mul(ctx, ggml_rms_norm(ctx, x_in, cfg.rms_eps), w.Wln_in); - ggml_tensor * q_proj = mm_w(ctx, w.Wq, x_norm); - ggml_tensor * k_proj = mm_w(ctx, w.Wk, x_norm); - ggml_tensor * v_proj = mm_w(ctx, w.Wv, x_norm); - - ggml_tensor * q_h = ggml_reshape_3d(ctx, q_proj, cfg.head_dim, cfg.n_q_heads, cfg.n_prefix); - ggml_tensor * k_h = ggml_reshape_3d(ctx, k_proj, cfg.head_dim, cfg.n_kv_heads, cfg.n_prefix); - ggml_tensor * v_h = ggml_reshape_3d(ctx, v_proj, cfg.head_dim, cfg.n_kv_heads, cfg.n_prefix); - - ggml_tensor * q_rope = rope_q_or_k(ctx, q_h, positions, cfg); - ggml_tensor * k_rope = rope_q_or_k(ctx, k_h, positions, cfg); - *k_out = k_rope; - *v_out = v_h; - - ggml_tensor * Q = ggml_permute(ctx, q_rope, 0, 2, 1, 3); - ggml_tensor * K = ggml_permute(ctx, vla::fa_kv(ctx, k_rope), 0, 2, 1, 3); - ggml_tensor * V = ggml_permute(ctx, vla::fa_kv(ctx, v_h), 0, 2, 1, 3); - const float scale = 1.f/std::sqrt(static_cast(cfg.head_dim)); - ggml_tensor * fa = ggml_flash_attn_ext(ctx, Q, K, V, mask, scale, - 0.f, 0.f); - ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); - ggml_tensor * att_pre_o = ggml_reshape_2d(ctx, fa, cfg.q_full_dim, cfg.n_prefix); - ggml_tensor * o_out = mm_w(ctx, w.Wo, att_pre_o); - ggml_tensor * h1 = ggml_add(ctx, x_in, o_out); - - ggml_tensor * x_norm_mlp = ggml_mul(ctx, ggml_rms_norm(ctx, h1, cfg.rms_eps), w.Wln_post); - ggml_tensor * gate = mm_w(ctx, w.Wgate, x_norm_mlp); - ggml_tensor * up = mm_w(ctx, w.Wup, x_norm_mlp); - ggml_tensor * inter = ggml_mul(ctx, ggml_silu(ctx, gate), up); - ggml_tensor * mlp_out = mm_w(ctx, w.Wdown, inter); - return ggml_add(ctx, h1, mlp_out); + ggml_tensor * x_norm = rms_norm(ctx, x_in, w.Wln_in, cfg.rms_eps); + ggml_tensor * q_h = ggml_reshape_3d(ctx, mm_w(ctx, w.Wq, x_norm), cfg.head_dim, cfg.n_q_heads, cfg.n_prefix); + ggml_tensor * k_h = ggml_reshape_3d(ctx, mm_w(ctx, w.Wk, x_norm), cfg.head_dim, cfg.n_kv_heads, cfg.n_prefix); + *v_out = ggml_reshape_3d(ctx, mm_w(ctx, w.Wv, x_norm), cfg.head_dim, cfg.n_kv_heads, cfg.n_prefix); + *k_out = rope(ctx, rope_spec(cfg), k_h, positions); + return attn_mlp(ctx, w, x_in, rope(ctx, rope_spec(cfg), q_h, positions), *k_out, *v_out, mask, cfg, cfg.n_prefix); } ggml_tensor * build_expert_self_attn_layer( - ggml_context * ctx, const ExpertLayerW & w, + ggml_context * ctx, const LayerW & w, ggml_tensor * x_in, ggml_tensor * cached_K, ggml_tensor * cached_V, ggml_tensor * positions_full, ggml_tensor * mask_full, const Config & cfg) { - ggml_tensor * x_norm = ggml_mul(ctx, ggml_rms_norm(ctx, x_in, cfg.rms_eps), w.Wln_in); - ggml_tensor * q_proj = mm_w(ctx, w.Wq, x_norm); - ggml_tensor * k_proj = mm_w(ctx, w.Wk, x_norm); - ggml_tensor * v_proj = mm_w(ctx, w.Wv, x_norm); - - ggml_tensor * q_h = ggml_reshape_3d(ctx, q_proj, cfg.head_dim, cfg.n_q_heads, cfg.n_suffix); - ggml_tensor * k_h = ggml_reshape_3d(ctx, k_proj, cfg.head_dim, cfg.n_kv_heads, cfg.n_suffix); - ggml_tensor * v_h = ggml_reshape_3d(ctx, v_proj, cfg.head_dim, cfg.n_kv_heads, cfg.n_suffix); - - ggml_tensor * q_rope = rope_q_or_k(ctx, q_h, positions_full, cfg); - ggml_tensor * k_rope = rope_q_or_k(ctx, k_h, positions_full, cfg); - - ggml_tensor * K_full = ggml_concat(ctx, cached_K, k_rope, 2); - ggml_tensor * V_full = ggml_concat(ctx, cached_V, v_h, 2); - - ggml_tensor * Q = ggml_permute(ctx, q_rope, 0, 2, 1, 3); - ggml_tensor * Kp = ggml_permute(ctx, vla::fa_kv(ctx, K_full), 0, 2, 1, 3); - ggml_tensor * Vp = ggml_permute(ctx, vla::fa_kv(ctx, V_full), 0, 2, 1, 3); - const float scale = 1.f/std::sqrt(static_cast(cfg.head_dim)); - ggml_tensor * fa = ggml_flash_attn_ext(ctx, Q, Kp, Vp, mask_full, scale, - 0.f, 0.f); - ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); - ggml_tensor * att_pre_o = ggml_reshape_2d(ctx, fa, cfg.q_full_dim, cfg.n_suffix); - ggml_tensor * h1 = ggml_add(ctx, x_in, mm_w(ctx, w.Wo, att_pre_o)); - - ggml_tensor * x_norm_mlp = ggml_mul(ctx, ggml_rms_norm(ctx, h1, cfg.rms_eps), w.Wln_post); - ggml_tensor * inter = ggml_mul(ctx, ggml_silu(ctx, mm_w(ctx, w.Wgate, x_norm_mlp)), - mm_w(ctx, w.Wup, x_norm_mlp)); - return ggml_add(ctx, h1, mm_w(ctx, w.Wdown, inter)); + ggml_tensor * x_norm = rms_norm(ctx, x_in, w.Wln_in, cfg.rms_eps); + ggml_tensor * q_h = ggml_reshape_3d(ctx, mm_w(ctx, w.Wq, x_norm), cfg.head_dim, cfg.n_q_heads, cfg.n_suffix); + ggml_tensor * k_h = ggml_reshape_3d(ctx, mm_w(ctx, w.Wk, x_norm), cfg.head_dim, cfg.n_kv_heads, cfg.n_suffix); + ggml_tensor * v_h = ggml_reshape_3d(ctx, mm_w(ctx, w.Wv, x_norm), cfg.head_dim, cfg.n_kv_heads, cfg.n_suffix); + + ggml_tensor * K_full = ggml_concat(ctx, cached_K, rope(ctx, rope_spec(cfg), k_h, positions_full), 2); + ggml_tensor * V_full = ggml_concat(ctx, cached_V, v_h, 2); + return attn_mlp(ctx, w, x_in, rope(ctx, rope_spec(cfg), q_h, positions_full), K_full, V_full, mask_full, cfg, cfg.n_suffix); } // Reproject the constant prefix cache into a cross-attn layer's K/V. Depends only // on the prefix, so it is built once and shared across all denoise steps. -void expert_cross_kv(ggml_context * ctx, const ExpertLayerW & w, +void expert_cross_kv(ggml_context * ctx, const LayerW & w, ggml_tensor * cached_K, ggml_tensor * cached_V, const Config & cfg, ggml_tensor ** K_repro, ggml_tensor ** V_repro) { ggml_tensor * cK_flat = ggml_reshape_2d(ctx, cached_K, cfg.kv_full_dim, cfg.n_prefix); @@ -891,31 +723,15 @@ void expert_cross_kv(ggml_context * ctx, const ExpertLayerW & w, } ggml_tensor * build_expert_cross_attn_layer( - ggml_context * ctx, const ExpertLayerW & w, + ggml_context * ctx, const LayerW & w, ggml_tensor * x_in, ggml_tensor * K_repro, ggml_tensor * V_repro, ggml_tensor * positions_rebased, ggml_tensor * mask_prefix_only, const Config & cfg) { - ggml_tensor * x_norm = ggml_mul(ctx, ggml_rms_norm(ctx, x_in, cfg.rms_eps), w.Wln_in); - - ggml_tensor * q_proj = mm_w(ctx, w.Wq, x_norm); - ggml_tensor * q_h = ggml_reshape_3d(ctx, q_proj, cfg.head_dim, cfg.n_q_heads, cfg.n_suffix); - ggml_tensor * q_rope = rope_q_or_k(ctx, q_h, positions_rebased, cfg); - - ggml_tensor * Q = ggml_permute(ctx, q_rope, 0, 2, 1, 3); - ggml_tensor * Kp = ggml_permute(ctx, vla::fa_kv(ctx, K_repro), 0, 2, 1, 3); - ggml_tensor * Vp = ggml_permute(ctx, vla::fa_kv(ctx, V_repro), 0, 2, 1, 3); - const float scale = 1.f/std::sqrt(static_cast(cfg.head_dim)); - ggml_tensor * fa = ggml_flash_attn_ext(ctx, Q, Kp, Vp, mask_prefix_only, scale, - 0.f, 0.f); - ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); - ggml_tensor * att_pre_o = ggml_reshape_2d(ctx, fa, cfg.q_full_dim, cfg.n_suffix); - ggml_tensor * h1 = ggml_add(ctx, x_in, mm_w(ctx, w.Wo, att_pre_o)); - - ggml_tensor * x_norm_mlp = ggml_mul(ctx, ggml_rms_norm(ctx, h1, cfg.rms_eps), w.Wln_post); - ggml_tensor * inter = ggml_mul(ctx, ggml_silu(ctx, mm_w(ctx, w.Wgate, x_norm_mlp)), - mm_w(ctx, w.Wup, x_norm_mlp)); - return ggml_add(ctx, h1, mm_w(ctx, w.Wdown, inter)); + ggml_tensor * x_norm = rms_norm(ctx, x_in, w.Wln_in, cfg.rms_eps); + ggml_tensor * q_h = ggml_reshape_3d(ctx, mm_w(ctx, w.Wq, x_norm), cfg.head_dim, cfg.n_q_heads, cfg.n_suffix); + return attn_mlp(ctx, w, x_in, rope(ctx, rope_spec(cfg), q_h, positions_rebased), K_repro, V_repro, + mask_prefix_only, cfg, cfg.n_suffix); } } @@ -969,43 +785,35 @@ static void backend_set_from_f32(ggml_tensor * t, const float * src, int64_t n) } } -SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, - const std::string& mmproj_path, - const std::string& ckpt_path, - const std::string& config_path) { - auto* m = new SmolVLAModelArch(); +std::unique_ptr smolvla_load_impl(ggml_type weight_dtype, + const std::string& ckpt_path, + const std::string& config_path, + const Options& opts) { + auto m = std::make_unique(); const bool use_gguf = ends_with_gguf(ckpt_path); - safetensors st; - gguf_source gst; + const std::string cfg_path = config_path.empty() ? dir_of(ckpt_path) + "/config.json" : config_path; + safetensors st; + gguf_source & gst = m->gst; if (use_gguf) { - if (!gst.open(ckpt_path)) { - delete m; + if (!gst.open(ckpt_path) || !load_config_from_gguf(gst, m->cfg)) return nullptr; - } - if (!load_config_from_gguf(gst, m->cfg)) { - delete m; - return nullptr; - } std::printf("vla: config = %s (gguf KV)\n", ckpt_path.c_str()); } else { - const std::string cfg_path = config_path.empty() ? default_config_path(ckpt_path) - : config_path; - if (!load_config_from_json(cfg_path, m->cfg)) { - delete m; + if (!load_config_from_json(cfg_path, m->cfg)) return nullptr; - } std::printf("vla: config = %s\n", cfg_path.c_str()); } + if (!resolve_num_steps("smolvla", opts, m->cfg.num_steps)) + return nullptr; { const Backend b = backend_init("vla", default_cpu_threads()); - if (!b.handle) { - delete m; + if (!b.handle) return nullptr; - } m->backend = b.handle; + m->is_cuda = b.is_cuda; } vram_probe(m->backend, "after backend init"); @@ -1013,26 +821,33 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, std::printf("vla: tower weights resident as %s\n", ggml_type_name(m->weight_dtype)); // Vision tower geometry: from gguf KV (self-contained ckpt), else SmolVLM2-500M defaults. - (void) mmproj_path; if (use_gguf) { - auto vu = [&](const char * k, int64_t & d) { if (gst.has_key(k)) d = (int64_t) gst.get_u32(k); }; + auto vu = [&](const char * k, int64_t & d) { if (gst.has(k)) d = (int64_t) gst.u32(k); }; vu("smolvla.vit_hidden", m->vit_hidden); vu("smolvla.vit_layers", m->vit_layers); vu("smolvla.vit_heads", m->vit_heads); vu("smolvla.patch_size", m->vit_patch); vu("smolvla.image_size", m->vit_image); vu("smolvla.vit_pixel_shuffle", m->vit_scale); vu("smolvla.n_img_tokens", m->vit_n_tokens); vu("smolvla.vit_inter", m->vit_inter); - if (gst.has_key("smolvla.vit_ln_eps")) - m->vit_ln_eps = gst.get_f32("smolvla.vit_ln_eps"); + if (gst.has("smolvla.vit_ln_eps")) + m->vit_ln_eps = gst.f32("smolvla.vit_ln_eps"); } { + if (m->vit_patch <= 0 || m->vit_scale <= 0 || m->vit_heads <= 0 || + m->vit_image % m->vit_patch || (m->vit_image/m->vit_patch) % m->vit_scale || + m->vit_hidden % m->vit_heads) { + std::fprintf(stderr, "vla(smolvla): bad vit geometry (image %lld patch %lld shuffle %lld hidden %lld heads %lld)\n", + (long long) m->vit_image, (long long) m->vit_patch, (long long) m->vit_scale, + (long long) m->vit_hidden, (long long) m->vit_heads); + return nullptr; + } const int64_t grid = m->vit_image/m->vit_patch; const int64_t k = grid/m->vit_scale; if (k * k != m->vit_n_tokens) { std::fprintf(stderr, "vla: smolvla vit geometry mismatch (grid=%lld scale=%lld -> %lld tokens, KV says %lld)\n", (long long) grid, (long long) m->vit_scale, (long long) (k * k), (long long) m->vit_n_tokens); - delete m; return nullptr; } m->cfg.n_img = m->vit_n_tokens; + m->vit.enc.cfg = {m->vit_hidden, m->vit_heads, m->vit_hidden/m->vit_heads, m->vit_ln_eps, false}; } m->cfg.n_prefix = m->cfg.n_img+m->cfg.n_lang+m->cfg.n_state; m->cfg.n_full = m->cfg.n_prefix+m->cfg.n_suffix; @@ -1040,7 +855,6 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, if (!use_gguf) { if (!st.open(ckpt_path)) { std::fprintf(stderr, "vla: failed to open %s\n", ckpt_path.c_str()); - delete m; return nullptr; } if (m->cfg.n_layers <= 0) { @@ -1055,7 +869,6 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, } if (max_layer < 0) { std::fprintf(stderr, "vla: cannot infer n_layers from %s\n", ckpt_path.c_str()); - delete m; return nullptr; } m->cfg.n_layers = max_layer+1; @@ -1064,7 +877,6 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, const auto it = st.tensors.find("model.vlm_with_expert.lm_expert.layers.0.mlp.gate_proj.weight"); if (it == st.tensors.end() || it->second.shape.size() != 2) { std::fprintf(stderr, "vla: missing/malformed expert gate_proj for shape derivation\n"); - delete m; return nullptr; } @@ -1073,7 +885,6 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, std::fprintf(stderr, "vla: expert_h mismatch - config implies %lld, " "checkpoint gate_proj has %lld\n", (long long) m->cfg.expert_h, (long long) it->second.shape[1]); - delete m; return nullptr; } } @@ -1092,16 +903,10 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, if (use_gguf) { if (!load_normalizer_stats_from_gguf(gst, *m)) { std::fprintf(stderr, "vla: failed to load normalizer stats from gguf\n"); - delete m; return nullptr; } } else { - const std::string cfg_path_local = config_path.empty() - ? default_config_path(ckpt_path) : config_path; - const auto pos = cfg_path_local.find_last_of("/\\"); - const std::string model_dir = (pos == std::string::npos) ? std::string(".") - : cfg_path_local.substr(0, pos); - load_normalizer_stats(model_dir, *m); + load_normalizer_stats(dir_of(cfg_path), *m); } ggml_init_params gparams = { @@ -1112,7 +917,6 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, m->ctx_weights = ggml_init(gparams); if (!m->ctx_weights) { std::fprintf(stderr, "vla: ggml_init (weights) failed\n"); - delete m; return nullptr; } @@ -1121,13 +925,7 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, const ggml_type wdt = m->weight_dtype; struct PendingF32 { std::string name; ggml_tensor * t; std::vector shape; }; - struct PendingBF16 { std::string name; ggml_tensor * t; }; std::vector pending_f32; - std::vector pending_bf16; - - m->E_lang = ggml_new_tensor_2d(ctx, GGML_TYPE_BF16, cfg.hidden, 49280); - pending_bf16.push_back({ - "model.vlm_with_expert.vlm.model.text_model.embed_tokens.weight", m->E_lang}); m->Wstate = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.max_state_dim, cfg.hidden); m->bstate = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, cfg.hidden); @@ -1140,19 +938,20 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, const int64_t grid = m->vit_image/P, n_patches = grid * grid; const int64_t c4 = H * m->vit_scale*m->vit_scale; const char * VP = "model.vlm_with_expert.vlm.model.vision_model."; - m->vit_patch_w = ggml_new_tensor_4d(ctx, GGML_TYPE_F32, P, P, 3, H); - m->vit_patch_b = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, H); - m->vit_pos = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, H, n_patches); - m->vit_post_ln_w = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, H); - m->vit_post_ln_b = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, H); - pending_f32.push_back({std::string(VP) + "embeddings.patch_embedding.weight", m->vit_patch_w, {H, 3, P, P}}); - pending_f32.push_back({std::string(VP) + "embeddings.patch_embedding.bias", m->vit_patch_b, {H}}); - pending_f32.push_back({std::string(VP) + "embeddings.position_embedding.weight", m->vit_pos, {n_patches, H}}); - pending_f32.push_back({std::string(VP) + "post_layernorm.weight", m->vit_post_ln_w, {H}}); - pending_f32.push_back({std::string(VP) + "post_layernorm.bias", m->vit_post_ln_b, {H}}); - m->vit.resize(m->vit_layers); + SigLipTower & vt = m->vit; + vt.patch_w = ggml_new_tensor_4d(ctx, GGML_TYPE_F32, P, P, 3, H); + vt.patch_b = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, H); + vt.pos = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, H, n_patches); + vt.post_ln_w = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, H); + vt.post_ln_b = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, H); + pending_f32.push_back({std::string(VP) + "embeddings.patch_embedding.weight", vt.patch_w, {H, 3, P, P}}); + pending_f32.push_back({std::string(VP) + "embeddings.patch_embedding.bias", vt.patch_b, {H}}); + pending_f32.push_back({std::string(VP) + "embeddings.position_embedding.weight", vt.pos, {n_patches, H}}); + pending_f32.push_back({std::string(VP) + "post_layernorm.weight", vt.post_ln_w, {H}}); + pending_f32.push_back({std::string(VP) + "post_layernorm.bias", vt.post_ln_b, {H}}); + vt.enc.blk.resize(m->vit_layers); for (int64_t i=0; ivit_layers; ++i) { - EncBlockW & w = m->vit[i]; + EncBlockW & w = vt.enc.blk[i]; char pb[256]; std::snprintf(pb, sizeof(pb), "%sencoder.layers.%lld.", VP, (long long) i); const std::string pf = pb; w.ln1w = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, H); w.ln1b = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, H); @@ -1179,7 +978,7 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, m->vlm_layers.resize(cfg.n_layers); for (int i=0; ivlm_layers[i]; + LayerW & w = m->vlm_layers[i]; w.Wln_in = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, cfg.hidden); w.Wq = ggml_new_tensor_2d(ctx, wdt, cfg.hidden, cfg.q_full_dim); w.Wk = ggml_new_tensor_2d(ctx, wdt, cfg.hidden, cfg.kv_full_dim); @@ -1203,22 +1002,19 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, pending_f32.push_back({pf + "mlp.up_proj.weight", w.Wup, {cfg.intermediate, cfg.hidden}}); pending_f32.push_back({pf + "mlp.down_proj.weight", w.Wdown, {cfg.hidden, cfg.intermediate}}); } - m->Wnorm_vlm = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, cfg.hidden); - pending_f32.push_back({"model.vlm_with_expert.vlm.model.text_model.norm.weight", - m->Wnorm_vlm, {cfg.hidden}}); m->expert_layers.resize(cfg.n_layers); for (int i=0; iexpert_layers[i]; - w.is_self_attn = (i%cfg.self_attn_every_n == 0); + LayerW & w = m->expert_layers[i]; + const bool self_attn = expert_self_attn(cfg, i); w.Wln_in = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, cfg.expert_h); w.Wq = ggml_new_tensor_2d(ctx, wdt, cfg.expert_h, cfg.q_full_dim); - if (w.is_self_attn) { + if (self_attn) { w.Wk = ggml_new_tensor_2d(ctx, wdt, cfg.expert_h, cfg.kv_full_dim); w.Wv = ggml_new_tensor_2d(ctx, wdt, cfg.expert_h, cfg.kv_full_dim); } else { - w.Wk = ggml_new_tensor_2d(ctx, wdt, cfg.kv_full_dim, cfg.kv_full_dim); - w.Wv = ggml_new_tensor_2d(ctx, wdt, cfg.kv_full_dim, cfg.kv_full_dim); + w.Wk = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.kv_full_dim, cfg.kv_full_dim); + w.Wv = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.kv_full_dim, cfg.kv_full_dim); } w.Wo = ggml_new_tensor_2d(ctx, wdt, cfg.q_full_dim, cfg.expert_h); w.Wln_post = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, cfg.expert_h); @@ -1231,7 +1027,7 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, const std::string pf = p; pending_f32.push_back({pf + "input_layernorm.weight", w.Wln_in, {cfg.expert_h}}); pending_f32.push_back({pf + "self_attn.q_proj.weight", w.Wq, {cfg.q_full_dim, cfg.expert_h}}); - if (w.is_self_attn) { + if (self_attn) { pending_f32.push_back({pf + "self_attn.k_proj.weight", w.Wk, {cfg.kv_full_dim, cfg.expert_h}}); pending_f32.push_back({pf + "self_attn.v_proj.weight", w.Wv, {cfg.kv_full_dim, cfg.expert_h}}); } else { @@ -1276,6 +1072,13 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, // ggml_mul_mat dequantizes at compute, as in the shared WeightLoader. Every // tensor above that uses wdt is a mul_mat operand, so those are the ones // eligible. Retyping is safe here: nothing is allocated yet. + auto retype = [](ggml_tensor * t, ggml_type type) { + t->type = type; + t->nb[0] = ggml_type_size(type); + t->nb[1] = t->nb[0] * (t->ne[0] / ggml_blck_size(type)); + for (int d = 2; d < GGML_MAX_DIMS; ++d) + t->nb[d] = t->nb[d-1] * t->ne[d-1]; + }; struct PendingPacked { std::string name; ggml_tensor * t; }; std::vector pending_packed; if (use_gguf) { @@ -1285,11 +1088,7 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, const ggml_type ft = gst.file_type(hf_to_gguf(p.name)); if (p.t->type == wdt && ft != GGML_TYPE_COUNT && ggml_is_quantized(ft) && p.t->ne[0] % ggml_blck_size(ft) == 0) { - p.t->type = ft; - p.t->nb[0] = ggml_type_size(ft); - p.t->nb[1] = p.t->nb[0] * (p.t->ne[0] / ggml_blck_size(ft)); - for (int d = 2; d < GGML_MAX_DIMS; ++d) - p.t->nb[d] = p.t->nb[d-1] * p.t->ne[d-1]; + retype(p.t, ft); pending_packed.push_back({p.name, p.t}); } else { keep.push_back(std::move(p)); @@ -1301,10 +1100,41 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, pending_packed.size(), ggml_type_name(pending_packed[0].t->type)); } + std::unordered_set widened; + if (m->is_cuda && vla::mm_prec_f32_enabled() && !opts.weight_dtype) { + for (const auto * layers : {&m->vlm_layers, &m->expert_layers}) + for (const LayerW & w : *layers) + for (ggml_tensor * t : {w.Wq, w.Wk, w.Wv, w.Wo, w.Wgate, w.Wup, w.Wdown}) + if (t->type == GGML_TYPE_BF16) { + retype(t, GGML_TYPE_F32); + widened.insert(t); + } + if (!widened.empty()) + std::printf("vla: %zu LM GEMM weights widened to f32 for --mm-prec f32\n", widened.size()); + } + + const std::string emb = "model.vlm_with_expert.vlm.model.text_model.embed_tokens.weight"; + if (use_gguf) { + const ggml_tensor * te = gst.meta(hf_to_gguf(emb).c_str()); + if (!te || te->ne[0] != cfg.hidden || te->ne[2] != 1 || te->ne[3] != 1 || + (te->type != GGML_TYPE_BF16 && te->type != GGML_TYPE_F32)) { + std::fprintf(stderr, "vla(smolvla): gguf %s missing or not a [%lld, vocab] f32/bf16 table\n", + hf_to_gguf(emb).c_str(), (long long) cfg.hidden); + return nullptr; + } + m->n_vocab = te->ne[1]; + } else { + m->n_vocab = 49280; + m->E_lang.resize(size_t(cfg.hidden)*m->n_vocab); + if (!st.read_raw(emb, m->E_lang.data(), m->E_lang.size()*sizeof(ggml_bf16_t), "BF16")) { + std::fprintf(stderr, "vla: read_raw failed for %s\n", emb.c_str()); + return nullptr; + } + } + m->weight_buf = alloc_weights(m->ctx_weights, m->backend); if (!m->weight_buf) { std::fprintf(stderr, "vla: alloc_weights failed\n"); - delete m; return nullptr; } std::printf("vla: [vram] weight_buf = %.1f MiB\n", @@ -1321,40 +1151,23 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, std::fprintf(stderr, "vla: read_to_f32 failed for %s\n", hf_name.c_str()); return false; } - backend_set_from_f32(t, hbuf.data(), ggml_nelements(t)); - return true; - }; - auto stream_bf16 = [&](const std::string & hf_name, ggml_tensor * t) -> bool { - std::vector hbuf(ggml_nelements(t)); - const bool ok = use_gguf - ? gst.read_raw(hf_to_gguf(hf_name), hbuf.data(), ggml_nbytes(t), "BF16") - : st .read_raw(hf_name, hbuf.data(), ggml_nbytes(t), "BF16"); - if (!ok) { - std::fprintf(stderr, "vla: read_raw failed for %s\n", hf_name.c_str()); - return false; + if (widened.count(t)) { + std::vector tmp(hbuf.size()); + ggml_fp32_to_bf16_row(hbuf.data(), tmp.data(), (int64_t) hbuf.size()); + ggml_bf16_to_fp32_row(tmp.data(), hbuf.data(), (int64_t) hbuf.size()); } - ggml_backend_tensor_set(t, hbuf.data(), 0, ggml_nbytes(t)); + backend_set_from_f32(t, hbuf.data(), ggml_nelements(t)); return true; }; for (auto & p : pending_f32) { - if (!stream_f32(p.name, p.t, p.shape)) { - delete m; + if (!stream_f32(p.name, p.t, p.shape)) return nullptr; - } - } - for (auto & p : pending_bf16) { - if (!stream_bf16(p.name, p.t)) { - delete m; - return nullptr; - } } for (auto & p : pending_packed) { std::vector hbuf(ggml_nbytes(p.t)); - if (!gst.read_packed(hf_to_gguf(p.name), hbuf.data(), p.t->type, hbuf.size())) { - delete m; + if (!gst.read_packed(hf_to_gguf(p.name), hbuf.data(), p.t->type, hbuf.size())) return nullptr; - } ggml_backend_tensor_set(p.t, hbuf.data(), 0, hbuf.size()); } @@ -1379,161 +1192,98 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, } namespace { -bool build_compute_graph(SmolVLAModelArch* m, int n_views) { - if (n_views < 1) { - std::fprintf(stderr, "vla: build_compute_graph: n_views=%d invalid\n", n_views); - return false; - } - const Config & cfg_model = m->cfg; - Config cfg = cfg_model; - - cfg.n_img = cfg_model.n_img*int64_t(n_views); - - ggml_init_params gparams = { - size_t(64)*1024*1024, - nullptr, - true, - }; - ggml_context * ctx = ggml_init(gparams); - if (!ctx) { - std::fprintf(stderr, "vla: build_compute_graph: ggml_init failed\n"); - return false; - } +void build_graph(SmolVLAModelArch * m, ggml_context * ctx, SmolVLAModelArch::MainIO & io, + int n_views, int64_t n_lang, bool kv_leaves) { + Config cfg = m->cfg; + cfg.n_img = m->cfg.n_img*int64_t(n_views); + cfg.n_lang = n_lang; + cfg.n_prefix = cfg.n_img+cfg.n_lang+cfg.n_state; + cfg.n_full = cfg.n_prefix+cfg.n_suffix; - const int64_t n_lang_max = cfg.n_lang; - const int64_t n_prefix_max = cfg.n_img+n_lang_max+cfg.n_state; - const int64_t n_full_max = n_prefix_max+cfg.n_suffix; - - ggml_tensor * img_emb_in = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.hidden, cfg.n_img); - ggml_tensor * lang_ids = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, n_lang_max); - ggml_tensor * state_t = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, cfg.max_state_dim); - ggml_tensor * x0 = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.max_action_dim, cfg.n_suffix); - - ggml_tensor * mask_prefill = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, n_prefix_max, n_prefix_max); - ggml_tensor * pos_prefill = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, n_prefix_max); - ggml_tensor * mask_full = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, n_full_max, cfg.n_suffix); - ggml_tensor * mask_pfx_only = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, n_prefix_max, cfg.n_suffix); - ggml_tensor * pos_full = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, cfg.n_suffix); - ggml_tensor * pos_rebased = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, cfg.n_suffix); - - for (ggml_tensor * t : {img_emb_in, lang_ids, state_t, x0, - mask_prefill, pos_prefill, mask_full, - mask_pfx_only, pos_full, pos_rebased}) { + io.img_emb = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.hidden, cfg.n_img); + io.lang_emb = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.hidden, cfg.n_lang); + io.state = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, cfg.max_state_dim); + io.x0 = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.max_action_dim, cfg.n_suffix); + io.mask_prefill = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.n_prefix, cfg.n_prefix); + io.pos_prefill = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, cfg.n_prefix); + io.mask_full = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.n_full, cfg.n_suffix); + io.mask_pfx_only = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.n_prefix, cfg.n_suffix); + io.pos_full = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, cfg.n_suffix); + io.pos_rebased = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, cfg.n_suffix); + for (ggml_tensor * t : {io.img_emb, io.lang_emb, io.state, io.x0, + io.mask_prefill, io.pos_prefill, io.mask_full, + io.mask_pfx_only, io.pos_full, io.pos_rebased}) { ggml_set_input(t); } - Config cfg_built = cfg; - cfg_built.n_lang = n_lang_max; - cfg_built.n_prefix = n_prefix_max; - cfg_built.n_full = n_full_max; - const float lang_scale = std::sqrt(static_cast(cfg.hidden)); - ggml_tensor * img_emb_scaled = ggml_scale(ctx, img_emb_in, lang_scale); - ggml_tensor * lang_raw = ggml_get_rows(ctx, m->E_lang, lang_ids); - ggml_tensor * lang_emb_scaled = ggml_scale(ctx, lang_raw, lang_scale); - ggml_tensor * state_pre = ggml_add(ctx, ggml_mul_mat(ctx, m->Wstate, state_t), m->bstate); - ggml_tensor * state_emb = ggml_reshape_2d(ctx, state_pre, cfg.hidden, 1); - ggml_tensor * embs_il = ggml_concat(ctx, img_emb_scaled, lang_emb_scaled, 1); - ggml_tensor * prefix_embs = ggml_concat(ctx, embs_il, state_emb, 1); - - ggml_tensor * mask_prefill_f16 = ggml_cast(ctx, mask_prefill, GGML_TYPE_F16); - ggml_tensor * mask_full_f16 = ggml_cast(ctx, mask_full, GGML_TYPE_F16); - ggml_tensor * mask_pfx_only_f16= ggml_cast(ctx, mask_pfx_only, GGML_TYPE_F16); - std::vector k_cache(cfg.n_layers); - std::vector v_cache(cfg.n_layers); + ggml_tensor * img_emb_scaled = ggml_scale(ctx, io.img_emb, lang_scale); + ggml_tensor * lang_emb_scaled = ggml_scale(ctx, io.lang_emb, lang_scale); + ggml_tensor * state_emb = ggml_reshape_2d(ctx, linear(ctx, m->Wstate, m->bstate, io.state), cfg.hidden, 1); + ggml_tensor * prefix_embs = ggml_concat(ctx, ggml_concat(ctx, img_emb_scaled, lang_emb_scaled, 1), state_emb, 1); + + ggml_tensor * mask_prefill_f16 = ggml_cast(ctx, io.mask_prefill, GGML_TYPE_F16); + ggml_tensor * mask_full_f16 = ggml_cast(ctx, io.mask_full, GGML_TYPE_F16); + ggml_tensor * mask_pfx_only_f16= ggml_cast(ctx, io.mask_pfx_only, GGML_TYPE_F16); + io.k_cache.resize(cfg.n_layers); + io.v_cache.resize(cfg.n_layers); { ggml_tensor * h = prefix_embs; for (int i=0; ivlm_layers[i], h, mask_prefill_f16, pos_prefill, - cfg_built, &k_cache[i], &v_cache[i]); + h = build_vlm_layer(ctx, m->vlm_layers[i], h, mask_prefill_f16, io.pos_prefill, + cfg, &io.k_cache[i], &io.v_cache[i]); + } + } + + if (kv_leaves) { + for (int i=0; i & K = kv_leaves ? io.k_leaf : io.k_cache; + const std::vector & V = kv_leaves ? io.v_leaf : io.v_cache; // Reproject each cross-attn layer's prefix K/V once; reused every denoise step. std::vector xk_cache(cfg.n_layers, nullptr); std::vector xv_cache(cfg.n_layers, nullptr); for (int li=0; liexpert_layers[li].is_self_attn) - expert_cross_kv(ctx, m->expert_layers[li], k_cache[li], v_cache[li], - cfg_built, &xk_cache[li], &xv_cache[li]); + if (expert_self_attn(cfg, li)) + continue; + expert_cross_kv(ctx, m->expert_layers[li], K[li], V[li], cfg, &xk_cache[li], &xv_cache[li]); + if (m->is_cuda) { + xk_cache[li] = ggml_cast(ctx, xk_cache[li], GGML_TYPE_F16); + xv_cache[li] = ggml_cast(ctx, xv_cache[li], GGML_TYPE_F16); + } } const float dt = -1.f/static_cast(cfg.num_steps); - ggml_tensor * x_t = x0; - + ggml_tensor * x_t = io.x0; for (int step=0; steptime_bcasts[step]; - ggml_tensor * action_emb = ggml_add(ctx, ggml_mul_mat(ctx, m->W_ain, x_t), m->b_ain); - ggml_tensor * action_time_in = ggml_concat(ctx, action_emb, time_bcast, 0); - ggml_tensor * mlp1 = ggml_add(ctx, ggml_mul_mat(ctx, m->W_at1, action_time_in), m->b_at1); - ggml_tensor * suffix_embs = ggml_add(ctx, - ggml_mul_mat(ctx, m->W_at2, ggml_silu(ctx, mlp1)), - m->b_at2); - ggml_tensor * h = suffix_embs; + ggml_tensor * action_emb = linear(ctx, m->W_ain, m->b_ain, x_t); + ggml_tensor * action_time_in = ggml_concat(ctx, action_emb, m->time_bcasts[step], 0); + ggml_tensor * mlp1 = linear(ctx, m->W_at1, m->b_at1, action_time_in); + ggml_tensor * h = linear(ctx, m->W_at2, m->b_at2, ggml_silu(ctx, mlp1)); for (int li=0; liexpert_layers[li].is_self_attn) { - h = build_expert_self_attn_layer(ctx, m->expert_layers[li], h, - k_cache[li], v_cache[li], - pos_full, mask_full_f16, cfg_built); + if (expert_self_attn(cfg, li)) { + h = build_expert_self_attn_layer(ctx, m->expert_layers[li], h, K[li], V[li], + io.pos_full, mask_full_f16, cfg); } else { h = build_expert_cross_attn_layer(ctx, m->expert_layers[li], h, xk_cache[li], xv_cache[li], - pos_rebased, mask_pfx_only_f16, cfg_built); + io.pos_rebased, mask_pfx_only_f16, cfg); } } - ggml_tensor * h_final = ggml_mul(ctx, ggml_rms_norm(ctx, h, cfg.rms_eps), m->Wnorm_expert); - ggml_tensor * v_t = ggml_add(ctx, ggml_mul_mat(ctx, m->W_aout, h_final), m->b_aout); + ggml_tensor * v_t = linear(ctx, m->W_aout, m->b_aout, rms_norm(ctx, h, m->Wnorm_expert, cfg.rms_eps)); x_t = ggml_add(ctx, x_t, ggml_scale(ctx, v_t, dt)); } ggml_set_output(x_t); - - ggml_cgraph * gf = ggml_new_graph_custom(ctx, 16384, false); - ggml_build_forward_expand(gf, x_t); - - ggml_backend_buffer_type_t buft = ggml_backend_get_default_buffer_type(m->backend); - ggml_gallocr_t galloc = ggml_gallocr_new(buft); - if (!galloc) { - std::fprintf(stderr, "vla: build_compute_graph: ggml_gallocr_new failed\n"); - ggml_free(ctx); - return false; - } - if (!ggml_gallocr_reserve(galloc, gf)) { - std::fprintf(stderr, "vla: build_compute_graph: ggml_gallocr_reserve failed\n"); - ggml_gallocr_free(galloc); - ggml_free(ctx); - return false; - } - - std::printf("vla: [vram] gallocr compute buf = %.1f MiB\n", - ggml_gallocr_get_buffer_size(galloc, 0)/(1024.0*1024.0)); - vram_probe(m->backend, "after gallocr reserve"); - - m->ctx_compute = ctx; - m->galloc = galloc; - m->gf_cached = gf; - m->in_img_emb = img_emb_in; - m->in_lang_ids = lang_ids; - m->in_state = state_t; - m->in_x0 = x0; - m->in_mask_prefill = mask_prefill; - m->in_pos_prefill = pos_prefill; - m->in_mask_full = mask_full; - m->in_mask_pfx_only = mask_pfx_only; - m->in_pos_full = pos_full; - m->in_pos_rebased = pos_rebased; - m->out_x_t = x_t; - - return true; + io.x_t = x_t; } } SmolVLAModelArch::~SmolVLAModelArch() { - - if (galloc) - ggml_gallocr_free(galloc); - if (ctx_compute) - ggml_free(ctx_compute); if (weight_buf) ggml_backend_buffer_free(weight_buf); if (ctx_weights) @@ -1549,9 +1299,25 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { m->stats = Stats{}; - Config cfg = m->cfg; + const Config & cfg = m->cfg; - const size_t per_view_n = size_t(m->cfg.n_img*cfg.hidden); + if (in.n_lang < 1 || in.n_lang > int(cfg.n_lang)) { + std::fprintf(stderr, "vla: lang_tokens length %d out of range [1, %lld]\n", + in.n_lang, (long long) cfg.n_lang); + return {}; + } + + // Reject any token id outside the embedding table before the gather so an + // out-of-range id cannot read past the host table. + for (int i=0; i= m->n_vocab) { + std::fprintf(stderr, "vla: lang_tokens[%d]=%d out of vocab range [0, %lld)\n", + i, in.lang_tokens[i], (long long) m->n_vocab); + return {}; + } + } + + const size_t per_view_n = size_t(cfg.n_img*cfg.hidden); int n_views = 0; size_t img_emb_n = 0; @@ -1580,38 +1346,26 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { const int64_t s = m->vit_scale, c4 = H * s * s, K = m->vit_n_tokens; const auto t_vision_begin = clock::now(); - // Graph A: SigLIP ViT (conv patch-embed -> +pos -> layers -> post_ln), plain sequential positions. + // SigLIP ViT (conv patch-embed -> +pos -> layers -> post_ln -> pixel shuffle -> connector), plain sequential positions. ggml_context * VC = m->vision_scratch.reset(size_t(256)*1024*1024); if (!VC) { std::fprintf(stderr, "vla(smolvla): ggml_init(vision ctx) failed\n"); return {}; } ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, m->vit_image, m->vit_image, 3); ggml_set_input(t_px); - ggml_tensor * conv = ggml_conv_2d(VC, m->vit_patch_w, t_px, (int) m->vit_patch, (int) m->vit_patch, 0, 0, 1, 1); - ggml_tensor * patches = ggml_cont(VC, ggml_transpose(VC, ggml_reshape_2d(VC, conv, n_patches, H))); - ggml_tensor * hv = ggml_add(VC, ggml_add(VC, patches, m->vit_patch_b), m->vit_pos); - for (int64_t i=0; ivit_layers; ++i) - hv = build_siglip_layer(VC, m->vit[i], hv, n_patches, m->vit_heads, H/m->vit_heads, H, m->vit_ln_eps); - ggml_tensor * post_ln = ggml_add(VC, ggml_mul(VC, ggml_norm(VC, hv, m->vit_ln_eps), m->vit_post_ln_w), m->vit_post_ln_b); - ggml_set_output(post_ln); + ggml_tensor * hv = m->vit.embed_conv(VC, t_px, m->vit_patch, grid); + for (const EncBlockW & w : m->vit.enc.blk) + hv = build_siglip_layer(VC, m->vit.enc.cfg, w, hv, n_patches); + ggml_tensor * shuf = layer_norm(VC, hv, m->vit.post_ln_w, m->vit.post_ln_b, m->vit.enc.cfg.ln_eps); + shuf = ggml_cont(VC, ggml_permute(VC, ggml_reshape_3d(VC, shuf, H*s, grid/s, grid), 0, 2, 1, 3)); + shuf = ggml_cont(VC, ggml_permute(VC, ggml_reshape_3d(VC, shuf, c4, grid/s, grid/s), 0, 2, 1, 3)); + ggml_tensor * img_embeds = ggml_mul_mat(VC, m->mm_fc, ggml_reshape_2d(VC, shuf, c4, K)); + ggml_set_output(img_embeds); ggml_cgraph * gA = ggml_new_graph_custom(VC, 8192, false); - ggml_build_forward_expand(gA, post_ln); + ggml_build_forward_expand(gA, img_embeds); if (!m->vision_scratch.alloc(m->backend, gA)) { - std::fprintf(stderr, "vla(smolvla): vision gallocr A alloc failed\n"); - return {}; - } - - // Graph B: pixel-shuffle connector, a single bias-free matmul (c4 -> hidden). - ggml_context * MC = m->connector_scratch.reset(size_t(64)*1024*1024); - if (!MC) { std::fprintf(stderr, "vla(smolvla): ggml_init(connector ctx) failed\n"); return {}; } - ggml_tensor * t_shuf = ggml_new_tensor_2d(MC, GGML_TYPE_F32, c4, K); ggml_set_input(t_shuf); - ggml_tensor * img_embeds = ggml_mul_mat(MC, m->mm_fc, t_shuf); - ggml_set_output(img_embeds); - ggml_cgraph * gB = ggml_new_graph(MC); - ggml_build_forward_expand(gB, img_embeds); - if (!m->connector_scratch.alloc(m->backend, gB)) { - std::fprintf(stderr, "vla(smolvla): vision gallocr B alloc failed\n"); + std::fprintf(stderr, "vla(smolvla): vision gallocr alloc failed\n"); return {}; } - std::vector chw, post_host((size_t) H * n_patches), shuf_host((size_t) c4*K); + std::vector chw; bool vok = true; for (int v=0; vvit_image, chw)) { @@ -1621,14 +1375,7 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { ggml_backend_tensor_set(t_px, chw.data(), 0, ggml_nbytes(t_px)); graph_unique_names(gA); if (ggml_backend_graph_compute(m->backend, gA) != GGML_STATUS_SUCCESS) { - std::fprintf(stderr, "vla(smolvla): vision compute A failed (view %d)\n", v); vok = false; break; - } - ggml_backend_tensor_get(post_ln, post_host.data(), 0, ggml_nbytes(post_ln)); - pixel_shuffle_hf(post_host.data(), shuf_host.data(), H, grid, s); - ggml_backend_tensor_set(t_shuf, shuf_host.data(), 0, ggml_nbytes(t_shuf)); - graph_unique_names(gB); - if (ggml_backend_graph_compute(m->backend, gB) != GGML_STATUS_SUCCESS) { - std::fprintf(stderr, "vla(smolvla): connector compute failed (view %d)\n", v); vok = false; break; + std::fprintf(stderr, "vla(smolvla): vision compute failed (view %d)\n", v); vok = false; break; } ggml_backend_tensor_get(img_embeds, img_emb_pre.data()+size_t(v)*per_view_n, 0, ggml_nbytes(img_embeds)); } @@ -1636,183 +1383,56 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { m->stats.ms_vision = std::chrono::duration(clock::now()-t_vision_begin).count(); } - cfg.n_img = m->cfg.n_img*int64_t(n_views); - cfg.n_prefix = cfg.n_img+cfg.n_lang+cfg.n_state; - cfg.n_full = cfg.n_prefix+cfg.n_suffix; - - if (in.n_lang < 1 || in.n_lang > int(cfg.n_lang)) { - std::fprintf(stderr, "vla: lang_tokens length %d out of range [1, %lld]\n", - in.n_lang, (long long) cfg.n_lang); - return {}; - } - - // The language tokens index E_lang via ggml_get_rows, which does not bound - // its indices. Reject any token id outside the embedding table before the - // gather so an out-of-range id cannot read past the weights. - const int64_t vocab_rows = m->E_lang ? m->E_lang->ne[1] : 0; - for (int i=0; i= vocab_rows) { - std::fprintf(stderr, "vla: lang_tokens[%d]=%d out of vocab range [0, %lld)\n", - i, in.lang_tokens[i], (long long) vocab_rows); + const bool phase = in.timing_detail == TimingDetail::PHASE; + const int64_t n_img = cfg.n_img*int64_t(n_views); + const int64_t n_lang = phase ? in.n_lang : cfg.n_lang; + const int64_t n_prefix = n_img+n_lang+cfg.n_state; + const int64_t n_full = n_prefix+cfg.n_suffix; + const int64_t pad_start = n_img+in.n_lang; + const int64_t pad_end = n_img+n_lang; + + const size_t max_nodes = size_t(64)*cfg.n_layers*(cfg.num_steps+1) + 1024; + const size_t arena = ggml_tensor_overhead()*max_nodes + ggml_graph_overhead_custom(max_nodes, false); + + SmolVLAModelArch::MainIO * io = nullptr; + ggml_cgraph * gf = nullptr; + SmolVLAModelArch::MainIO phase_io; + std::unique_ptr phase_ctx(nullptr, ggml_free); + std::unique_ptr phase_buf(nullptr, ggml_backend_buffer_free); + if (!phase) { + const bool built = m->main_graph.ensure(m->backend, n_views, arena, + [&](ggml_context * C, SmolVLAModelArch::MainIO & gio) -> ggml_cgraph * { + build_graph(m, C, gio, n_views, n_lang, false); + ggml_cgraph * g = ggml_new_graph_custom(C, max_nodes, false); + ggml_build_forward_expand(g, gio.x_t); + return g; + }); + if (!built) { + std::fprintf(stderr, "vla: cached graph build failed\n"); return {}; } - } - - if (in.timing_detail == TimingDetail::NONE) { - - if (m->gf_cached == nullptr || m->cached_n_views != n_views) { - if (m->galloc) - ggml_gallocr_free(m->galloc); - if (m->ctx_compute) - ggml_free(m->ctx_compute); - m->galloc = nullptr; - m->ctx_compute = nullptr; - m->gf_cached = nullptr; - if (!build_compute_graph(m, n_views)) { - std::fprintf(stderr, "vla: cached graph build failed\n"); - return {}; - } - m->cached_n_views = n_views; - } - - const int64_t n_lang_max = m->cfg.n_lang; - const int64_t n_prefix_max = cfg.n_img+n_lang_max+cfg.n_state; - const int64_t n_full_max = n_prefix_max+cfg.n_suffix; - const int64_t pad_start = cfg.n_img+in.n_lang; - const int64_t pad_end = cfg.n_img+n_lang_max; - - std::vector state_host(cfg.max_state_dim, 0.0f); - if (in.state) - std::memcpy(state_host.data(), in.state, cfg.max_state_dim*sizeof(float)); - for (int64_t i=0; istate_mean[i])/(m->state_std[i]+cfg.norm_eps); - } - - std::vector noise_host(cfg.n_suffix*cfg.max_action_dim); - if (in.noise) { - std::memcpy(noise_host.data(), in.noise, noise_host.size()*sizeof(float)); - } else { - std::normal_distribution dist(0.f, 1.f); - for (auto & v : noise_host) - v = dist(m->rng); - } - - std::vector lang_host(n_lang_max, 0); - std::memcpy(lang_host.data(), in.lang_tokens, in.n_lang*sizeof(int32_t)); - - const int64_t state_pos = cfg.n_img+in.n_lang; - const int64_t suffix_pos_base = state_pos+1; - std::vector mask_prefill_host(n_prefix_max * n_prefix_max); - std::vector pos_prefill_host (n_prefix_max); - for (int64_t i=0; i= pad_start && j < pad_end) - blocked = true; - mask_prefill_host[i * n_prefix_max+j] = blocked ? -INFINITY : 0.f; - } - pos_prefill_host[i] = (i == n_prefix_max-1) - ? static_cast(state_pos) - : static_cast(i); - } - - std::vector mask_full_host (n_full_max * cfg.n_suffix); - std::vector mask_prefix_only_host(n_prefix_max * cfg.n_suffix, 0.f); - std::vector pos_full_host (cfg.n_suffix); - std::vector pos_rebased_host (cfg.n_suffix); - for (int64_t i=0; i= pad_start && j < pad_end); - } else { - blocked = ((j-n_prefix_max) > i); - } - mask_full_host[i * n_full_max+j] = blocked ? -INFINITY : 0.f; - } - for (int64_t j=0; j= pad_start && j < pad_end) { - mask_prefix_only_host[i * n_prefix_max+j] = -INFINITY; - } - } - pos_full_host [i] = static_cast(suffix_pos_base+i); - pos_rebased_host[i] = static_cast(i); - } - - if (!ggml_gallocr_alloc_graph(m->galloc, m->gf_cached)) { - std::fprintf(stderr, "vla: ggml_gallocr_alloc_graph failed\n"); + io = &m->main_graph.io(); + gf = m->main_graph.graph(); + } else { + ggml_init_params gparams = { arena + ggml_graph_overhead_custom(4096, false), nullptr, true }; + phase_ctx.reset(ggml_init(gparams)); + if (!phase_ctx) { + std::fprintf(stderr, "vla: ggml_init (compute) failed\n"); return {}; } - - ggml_backend_tensor_set(m->in_img_emb, img_emb_pre.data(), 0, img_emb_n * sizeof(float)); - ggml_backend_tensor_set(m->in_lang_ids, lang_host.data(), 0, n_lang_max * sizeof(int32_t)); - ggml_backend_tensor_set(m->in_state, state_host.data(), 0, cfg.max_state_dim*sizeof(float)); - ggml_backend_tensor_set(m->in_x0, noise_host.data(), 0, noise_host.size()*sizeof(float)); - ggml_backend_tensor_set(m->in_mask_prefill, mask_prefill_host.data(), 0, mask_prefill_host.size() * sizeof(float)); - ggml_backend_tensor_set(m->in_pos_prefill, pos_prefill_host.data(), 0, pos_prefill_host.size() * sizeof(int32_t)); - ggml_backend_tensor_set(m->in_mask_full, mask_full_host.data(), 0, mask_full_host.size() * sizeof(float)); - ggml_backend_tensor_set(m->in_mask_pfx_only, mask_prefix_only_host.data(), 0, mask_prefix_only_host.size()*sizeof(float)); - ggml_backend_tensor_set(m->in_pos_full, pos_full_host.data(), 0, pos_full_host.size() * sizeof(int32_t)); - ggml_backend_tensor_set(m->in_pos_rebased, pos_rebased_host.data(), 0, pos_rebased_host.size() * sizeof(int32_t)); - - graph_unique_names(m->gf_cached); - const auto t0 = clock::now(); - if (ggml_backend_graph_compute(m->backend, m->gf_cached) != GGML_STATUS_SUCCESS) { - std::fprintf(stderr, "vla: ggml compute (cached) failed\n"); + build_graph(m, phase_ctx.get(), phase_io, n_views, n_lang, true); + // Graph inputs, not weights: tag them so a backend that reads buffer usage + // (ggml-openvino) does not mistake the default ANY for a KV cache. gallocr + // tags its own arena the same way. + phase_buf.reset(ggml_backend_alloc_ctx_tensors(phase_ctx.get(), m->backend)); + if (!phase_buf) { + std::fprintf(stderr, "vla: ggml_backend_alloc_ctx_tensors (compute) failed\n"); return {}; } - m->stats.ms_inference = std::chrono::duration( - clock::now()-t0).count(); - - std::vector out(cfg.n_suffix*cfg.max_action_dim); - ggml_backend_tensor_get(m->out_x_t, out.data(), 0, out.size()*sizeof(float)); - for (int64_t r=0; raction_std[j]+cfg.norm_eps)+m->action_mean[j] : 0.0f; - } - } - - m->stats.ms_total = std::chrono::duration( - clock::now()-t_total_begin).count(); - return out; - } - - ggml_init_params gparams = { - size_t(64)*1024*1024, - nullptr, - true, - }; - ggml_context * ctx = ggml_init(gparams); - if (!ctx) { - std::fprintf(stderr, "vla: ggml_init (compute) failed\n"); - return {}; - } - - if (in.n_lang < 1 || in.n_lang > int(cfg.n_lang)) { - std::fprintf(stderr, "vla: lang_tokens length %d out of range [1, %lld]\n", - in.n_lang, (long long) cfg.n_lang); - ggml_free(ctx); - return {}; + ggml_backend_buffer_set_usage(phase_buf.get(), GGML_BACKEND_BUFFER_USAGE_COMPUTE); + io = &phase_io; } - cfg.n_lang = in.n_lang; - cfg.n_prefix = cfg.n_img+cfg.n_lang+cfg.n_state; - cfg.n_full = cfg.n_prefix+cfg.n_suffix; - ggml_tensor * img_emb_in = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.hidden, cfg.n_img); - ggml_tensor * lang_ids = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, cfg.n_lang); - ggml_tensor * state_t = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, cfg.max_state_dim); - ggml_tensor * x0 = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.max_action_dim, cfg.n_suffix); - - ggml_tensor * mask_prefill = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.n_prefix, cfg.n_prefix); - ggml_tensor * pos_prefill = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, cfg.n_prefix); - ggml_tensor * mask_full = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.n_full, cfg.n_suffix); - ggml_tensor * mask_prefix_only = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.n_prefix, cfg.n_suffix); - ggml_tensor * pos_full = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, cfg.n_suffix); - ggml_tensor * pos_rebased = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, cfg.n_suffix); - std::vector state_host(cfg.max_state_dim, 0.0f); if (in.state) std::memcpy(state_host.data(), in.state, cfg.max_state_dim*sizeof(float)); @@ -1829,193 +1449,105 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { v = dist(m->rng); } - std::vector mask_prefill_host(cfg.n_prefix*cfg.n_prefix); - std::vector pos_prefill_host (cfg.n_prefix); - for (int64_t i=0; i(i); - } - - std::vector mask_full_host (cfg.n_full * cfg.n_suffix); - std::vector mask_prefix_only_host(cfg.n_prefix*cfg.n_suffix, 0.f); + std::vector lang_host(n_lang, 0); + std::memcpy(lang_host.data(), in.lang_tokens, in.n_lang*sizeof(int32_t)); + std::vector lang_emb_host(size_t(n_lang)*cfg.hidden); + if (m->E_lang.empty()) { + if (!m->gst.fetch_rows_f32("token_embd.weight", lang_host, lang_emb_host.data(), cfg.hidden)) + return {}; + } else { + for (int64_t i=0; iE_lang.data()+size_t(lang_host[i])*cfg.hidden, + lang_emb_host.data()+size_t(i)*cfg.hidden, cfg.hidden); + } + + const int64_t state_pos = n_img+in.n_lang; + const int64_t suffix_pos_base = state_pos+1; + std::vector mask_prefill_host(n_prefix * n_prefix); + std::vector pos_prefill_host (n_prefix); + for (int64_t i=0; i= pad_start && j < pad_end) + blocked = true; + mask_prefill_host[i * n_prefix+j] = blocked ? -INFINITY : 0.f; + } + pos_prefill_host[i] = (i == n_prefix-1) + ? static_cast(state_pos) + : static_cast(i); + } + + std::vector mask_full_host (n_full * cfg.n_suffix); + std::vector mask_prefix_only_host(n_prefix * cfg.n_suffix, 0.f); std::vector pos_full_host (cfg.n_suffix); std::vector pos_rebased_host (cfg.n_suffix); for (int64_t i=0; i i); - mask_full_host[i * cfg.n_full+j] = blocked ? -INFINITY : 0.f; - } - pos_full_host [i] = static_cast(cfg.n_prefix+i); - pos_rebased_host[i] = static_cast(i); - } - - const float lang_scale = std::sqrt(static_cast(cfg.hidden)); - ggml_tensor * img_emb_scaled = ggml_scale(ctx, img_emb_in, lang_scale); - ggml_tensor * lang_raw = ggml_get_rows(ctx, m->E_lang, lang_ids); - ggml_tensor * lang_emb_scaled = ggml_scale(ctx, lang_raw, lang_scale); - ggml_tensor * state_pre = ggml_add(ctx, ggml_mul_mat(ctx, m->Wstate, state_t), m->bstate); - ggml_tensor * state_emb = ggml_reshape_2d(ctx, state_pre, cfg.hidden, 1); - ggml_tensor * embs_il = ggml_concat(ctx, img_emb_scaled, lang_emb_scaled, 1); - ggml_tensor * prefix_embs = ggml_concat(ctx, embs_il, state_emb, 1); - - ggml_tensor * mask_prefill_f16 = ggml_cast(ctx, mask_prefill, GGML_TYPE_F16); - ggml_tensor * mask_full_f16 = ggml_cast(ctx, mask_full, GGML_TYPE_F16); - ggml_tensor * mask_prefix_only_f16 = ggml_cast(ctx, mask_prefix_only, GGML_TYPE_F16); - std::vector k_cache(cfg.n_layers); - std::vector v_cache(cfg.n_layers); - { - ggml_tensor * h = prefix_embs; - for (int i=0; ivlm_layers[i], h, mask_prefill_f16, pos_prefill, - cfg, &k_cache[i], &v_cache[i]); - } - } - - std::vector K_storage(cfg.n_layers); - std::vector V_storage(cfg.n_layers); - if (in.timing_detail == TimingDetail::PHASE) { - for (int i=0; i(clock::now()-t0).count(); - }; - - auto & K_ref = (in.timing_detail == TimingDetail::PHASE) ? K_storage : k_cache; - auto & V_ref = (in.timing_detail == TimingDetail::PHASE) ? V_storage : v_cache; - - const float dt = -1.f/static_cast(cfg.num_steps); - ggml_tensor * x_t = x0; - std::vector time_bcasts(cfg.num_steps, nullptr); - std::vector> time_host (cfg.num_steps); - - std::vector xk_cache(cfg.n_layers, nullptr); - std::vector xv_cache(cfg.n_layers, nullptr); - for (int li=0; liexpert_layers[li].is_self_attn) - expert_cross_kv(ctx, m->expert_layers[li], K_ref[li], V_ref[li], cfg, &xk_cache[li], &xv_cache[li]); - } - - for (int step=0; step(step)*static_cast(dt); - time_host[step] = sinusoidal_time_emb(time, cfg.expert_h, cfg.min_period, cfg.max_period); - - ggml_tensor * time_bcast = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, cfg.expert_h, cfg.n_suffix); - time_bcasts[step] = time_bcast; - - ggml_tensor * action_emb = ggml_add(ctx, ggml_mul_mat(ctx, m->W_ain, x_t), m->b_ain); - ggml_tensor * action_time_in = ggml_concat(ctx, action_emb, time_bcast, 0); - ggml_tensor * mlp1 = ggml_add(ctx, ggml_mul_mat(ctx, m->W_at1, action_time_in), m->b_at1); - ggml_tensor * suffix_embs = ggml_add(ctx, - ggml_mul_mat(ctx, m->W_at2, ggml_silu(ctx, mlp1)), - m->b_at2); - - ggml_tensor * h = suffix_embs; - for (int li=0; liexpert_layers[li].is_self_attn) { - h = build_expert_self_attn_layer(ctx, m->expert_layers[li], h, - K_ref[li], V_ref[li], - pos_full, mask_full_f16, cfg); + if (j < n_prefix) { + blocked = (j >= pad_start && j < pad_end); } else { - h = build_expert_cross_attn_layer(ctx, m->expert_layers[li], h, - xk_cache[li], xv_cache[li], - pos_rebased, mask_prefix_only_f16, cfg); + blocked = ((j-n_prefix) > i); } + mask_full_host[i * n_full+j] = blocked ? -INFINITY : 0.f; } - ggml_tensor * h_final = ggml_mul(ctx, ggml_rms_norm(ctx, h, cfg.rms_eps), m->Wnorm_expert); - ggml_tensor * v_t = ggml_add(ctx, ggml_mul_mat(ctx, m->W_aout, h_final), m->b_aout); - x_t = ggml_add(ctx, x_t, ggml_scale(ctx, v_t, dt)); - } - - // Graph inputs, not weights: tag them so a backend that reads buffer usage - // (ggml-openvino) does not mistake the default ANY for a KV cache. gallocr - // tags its own arena the same way. - ggml_backend_buffer_t compute_buf = ggml_backend_alloc_ctx_tensors(ctx, m->backend); - if (compute_buf) { - ggml_backend_buffer_set_usage(compute_buf, GGML_BACKEND_BUFFER_USAGE_COMPUTE); - } else { - std::fprintf(stderr, "vla: ggml_backend_alloc_ctx_tensors (compute) failed\n"); - ggml_free(ctx); - return {}; - } - - ggml_backend_tensor_set(img_emb_in, img_emb_pre.data(), 0, img_emb_n * sizeof(float)); - ggml_backend_tensor_set(lang_ids, in.lang_tokens, 0, cfg.n_lang * sizeof(int32_t)); - ggml_backend_tensor_set(state_t, state_host.data(), 0, cfg.max_state_dim*sizeof(float)); - ggml_backend_tensor_set(x0, noise_host.data(), 0, noise_host.size() * sizeof(float)); - ggml_backend_tensor_set(mask_prefill, mask_prefill_host.data(), 0, mask_prefill_host.size() * sizeof(float)); - ggml_backend_tensor_set(pos_prefill, pos_prefill_host.data(), 0, pos_prefill_host.size() * sizeof(int32_t)); - ggml_backend_tensor_set(mask_full, mask_full_host.data(), 0, mask_full_host.size() * sizeof(float)); - ggml_backend_tensor_set(mask_prefix_only, mask_prefix_only_host.data(), 0, mask_prefix_only_host.size()*sizeof(float)); - ggml_backend_tensor_set(pos_full, pos_full_host.data(), 0, pos_full_host.size() * sizeof(int32_t)); - ggml_backend_tensor_set(pos_rebased, pos_rebased_host.data(), 0, pos_rebased_host.size() * sizeof(int32_t)); - for (int step=0; step tile(cfg.expert_h*cfg.n_suffix); - for (int64_t t=0; t= pad_start && j < pad_end) { + mask_prefix_only_host[i * n_prefix+j] = -INFINITY; + } } - ggml_backend_tensor_set(time_bcasts[step], tile.data(), 0, tile.size()*sizeof(float)); + pos_full_host [i] = static_cast(suffix_pos_base+i); + pos_rebased_host[i] = static_cast(i); } - if (in.timing_detail == TimingDetail::PHASE) { + ggml_backend_tensor_set(io->img_emb, img_emb_pre.data(), 0, img_emb_n * sizeof(float)); + ggml_backend_tensor_set(io->lang_emb, lang_emb_host.data(), 0, lang_emb_host.size()*sizeof(float)); + ggml_backend_tensor_set(io->state, state_host.data(), 0, cfg.max_state_dim*sizeof(float)); + ggml_backend_tensor_set(io->x0, noise_host.data(), 0, noise_host.size()*sizeof(float)); + ggml_backend_tensor_set(io->mask_prefill, mask_prefill_host.data(), 0, mask_prefill_host.size() * sizeof(float)); + ggml_backend_tensor_set(io->pos_prefill, pos_prefill_host.data(), 0, pos_prefill_host.size() * sizeof(int32_t)); + ggml_backend_tensor_set(io->mask_full, mask_full_host.data(), 0, mask_full_host.size() * sizeof(float)); + ggml_backend_tensor_set(io->mask_pfx_only, mask_prefix_only_host.data(), 0, mask_prefix_only_host.size()*sizeof(float)); + ggml_backend_tensor_set(io->pos_full, pos_full_host.data(), 0, pos_full_host.size() * sizeof(int32_t)); + ggml_backend_tensor_set(io->pos_rebased, pos_rebased_host.data(), 0, pos_rebased_host.size() * sizeof(int32_t)); - ggml_cgraph * gf_pre = ggml_new_graph_custom(ctx, 4096, false); + if (phase) { + ggml_cgraph * gf_pre = ggml_new_graph_custom(phase_ctx.get(), 4096, false); for (int i=0; ik_cache[i]); + ggml_build_forward_expand(gf_pre, io->v_cache[i]); } graph_unique_names(gf_pre); const auto t0 = clock::now(); if (ggml_backend_graph_compute(m->backend, gf_pre) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla: ggml prefill compute failed\n"); - ggml_backend_buffer_free(compute_buf); - ggml_free(ctx); return {}; } - m->stats.ms_prefill = ms_since(t0); + m->stats.ms_prefill = std::chrono::duration(clock::now()-t0).count(); for (int i=0; ik_cache[i], io->k_leaf[i]); + ggml_backend_tensor_copy(io->v_cache[i], io->v_leaf[i]); } + gf = ggml_new_graph_custom(phase_ctx.get(), max_nodes, false); + ggml_build_forward_expand(gf, io->x_t); } - { - ggml_cgraph * gf = ggml_new_graph_custom(ctx, 16384, false); - ggml_build_forward_expand(gf, x_t); - graph_unique_names(gf); - const auto t0 = clock::now(); - if (ggml_backend_graph_compute(m->backend, gf) != GGML_STATUS_SUCCESS) { - std::fprintf(stderr, "vla: ggml compute failed\n"); - ggml_backend_buffer_free(compute_buf); - ggml_free(ctx); - return {}; - } - const float ms = ms_since(t0); - if (in.timing_detail == TimingDetail::PHASE) { - m->stats.ms_denoise = ms; - } - - m->stats.ms_inference = m->stats.ms_prefill+ms; + graph_unique_names(gf); + const auto t0 = clock::now(); + if (ggml_backend_graph_compute(m->backend, gf) != GGML_STATUS_SUCCESS) { + std::fprintf(stderr, "vla: ggml compute failed\n"); + return {}; } + const float ms = std::chrono::duration(clock::now()-t0).count(); + if (phase) + m->stats.ms_denoise = ms; + m->stats.ms_inference = m->stats.ms_prefill+ms; std::vector out(cfg.n_suffix*cfg.max_action_dim); - ggml_backend_tensor_get(x_t, out.data(), 0, out.size()*sizeof(float)); - + ggml_backend_tensor_get(io->x_t, out.data(), 0, out.size()*sizeof(float)); for (int64_t r=0; r predict_impl(SmolVLAModelArch* m, const Inputs& in) { } } - ggml_backend_buffer_free(compute_buf); - ggml_free(ctx); - m->stats.ms_total = std::chrono::duration( clock::now()-t_total_begin).count(); return out; @@ -2036,15 +1565,12 @@ std::vector SmolVLAModelArch::predict(const Inputs& in) { return predict_impl(this, in); } -std::unique_ptr smolvla_create(const std::string& mmproj_path, +std::unique_ptr smolvla_create(const std::string&, const std::string& ckpt_path, const std::string& config_path, const Options& opts) { - SmolVLAModelArch* raw = smolvla_load_impl(opts.weight_dtype.value_or(vla::default_weight_dtype(GGML_TYPE_BF16)), - mmproj_path, ckpt_path, config_path); - if (!raw) - return nullptr; - return std::unique_ptr(raw); + return smolvla_load_impl(opts.weight_dtype.value_or(vla::default_weight_dtype(GGML_TYPE_BF16)), + ckpt_path, config_path, opts); } } diff --git a/src/models/turbovla.cpp b/src/models/turbovla.cpp index 236aa75..9909cea 100644 --- a/src/models/turbovla.cpp +++ b/src/models/turbovla.cpp @@ -16,7 +16,8 @@ // BERT-base over the instruction, six Grounding-DINO bi-attention fusion layers // each followed by a text enhancer, and a three-layer ACT decoder whose twelve // learned queries read the fused tokens plus two state tokens. One forward pass, -// no denoising loop, so the whole model is one cached graph. +// no denoising loop, so the model is one cached graph plus a BERT graph that +// reruns only when the instruction changes. #include "arch.h" #include "backend.h" @@ -100,6 +101,7 @@ struct TurboVlaModelArch : public ModelArchBase { TurboVlaModelArch() : ModelArchBase(Arch::TURBOVLA) {} ~TurboVlaModelArch() override { graph.release(); + text.release(); if (const_buf) ggml_backend_buffer_free(const_buf); if (ctx_const) ggml_free(ctx_const); if (weight_buf) ggml_backend_buffer_free(weight_buf); @@ -127,6 +129,7 @@ struct TurboVlaModelArch : public ModelArchBase { ggml_tensor *patch_w = nullptr, *patch_b = nullptr, *cls_tok = nullptr, *reg_tok = nullptr; std::vector vit; + ggml_tensor *vit_norm_w = nullptr, *vit_norm_b = nullptr; ggml_tensor *vp_in_w = nullptr, *vp_in_b = nullptr, *vp_fc1_w = nullptr, *vp_fc1_b = nullptr; ggml_tensor *vp_fc2_w = nullptr, *vp_fc2_b = nullptr, *vp_skip_w = nullptr; ggml_tensor *vp_out_w = nullptr, *vp_out_b = nullptr, *view_emb = nullptr; @@ -156,17 +159,22 @@ struct TurboVlaModelArch : public ModelArchBase { bool operator==(const Key & o) const { return bert_len == o.bert_len; } }; struct IO { - ggml_tensor *patches = nullptr, *ids = nullptr, *pos = nullptr, *bert_mask = nullptr; - ggml_tensor *enh_mask = nullptr, *fus_mask = nullptr, *state = nullptr, *actions = nullptr; + ggml_tensor *patches = nullptr, *enh_mask = nullptr, *fus_mask = nullptr, *state = nullptr, *actions = nullptr; }; - graph_cache graph; + struct TextIO { + ggml_tensor *ids = nullptr, *pos = nullptr, *bert_mask = nullptr, *lang = nullptr; + }; + graph_cache graph; + graph_cache text; + std::vector text_ids; int64_t grid() const { return image_size / patch; } int64_t n_patches() const { return grid() * grid(); } int64_t vit_seq() const { return 1 + n_reg + n_patches(); } int64_t text_len(const int32_t * ids, int64_t n) const; - ggml_cgraph * build(ggml_context * C, IO & io, int64_t bert_len) const; + ggml_cgraph * build_text(ggml_context * C, TextIO & io, int64_t bert_len) const; + ggml_cgraph * build(ggml_context * C, IO & io, ggml_tensor * text_out) const; bool upload_rope_tables(); std::vector predict(const Inputs& in) override; @@ -210,9 +218,8 @@ ggml_tensor * vit_attention(ggml_context * C, ggml_tensor * qkv, ggml_tensor * c ggml_tensor * k = rope_heads(C, split_heads(C, qkv, hd, heads, T, nv, 1), cos_t, sin_signed); ggml_tensor * v = split_heads(C, qkv, hd, heads, T, nv, 2); if (flash) { - ggml_tensor * o = flash_attention(C, ggml_permute(C, q, 0, 2, 1, 3), ggml_permute(C, k, 0, 2, 1, 3), - ggml_permute(C, v, 0, 2, 1, 3), nullptr, scale); - return ggml_reshape_3d(C, o, hd*heads, T, nv); + return flash_attention(C, ggml_permute(C, q, 0, 2, 1, 3), ggml_permute(C, k, 0, 2, 1, 3), + ggml_permute(C, v, 0, 2, 1, 3), nullptr, scale); } ggml_tensor * Q = ggml_cont(C, ggml_permute(C, q, 0, 2, 1, 3)); ggml_tensor * K = ggml_cont(C, ggml_permute(C, k, 0, 2, 1, 3)); @@ -305,7 +312,35 @@ bool TurboVlaModelArch::upload_rope_tables() { return true; } -ggml_cgraph * TurboVlaModelArch::build(ggml_context * C, IO & io, int64_t bert_len) const { +// BERT over the checkpoint's padded length, token type 0 throughout. +ggml_cgraph * TurboVlaModelArch::build_text(ggml_context * C, TextIO & io, int64_t bert_len) const { + io.ids = ggml_new_tensor_1d(C, GGML_TYPE_I32, bert_len); + io.pos = ggml_new_tensor_1d(C, GGML_TYPE_I32, bert_len); + io.bert_mask = ggml_new_tensor_2d(C, GGML_TYPE_F32, bert_len, bert_len); + ggml_set_input(io.ids); + ggml_set_input(io.pos); + ggml_set_input(io.bert_mask); + ggml_tensor * t = ggml_add(C, ggml_get_rows(C, word_emb, io.ids), ggml_get_rows(C, pos_emb, io.pos)); + t = ggml_add(C, t, ggml_view_1d(C, type_emb, bert_dim, 0)); + t = layer_norm(C, t, emb_ln_w, emb_ln_b, kBertEps); + for (const BertLayerW & l : bert) { + ggml_tensor * att = self_attention(C, linear(C, l.qkv_w, l.qkv_b, t), io.bert_mask, + bert_dim / bert_heads, bert_heads, bert_len, 1); + t = layer_norm(C, ggml_add(C, t, linear(C, l.o_w, l.o_b, att)), l.ln1_w, l.ln1_b, kBertEps); + t = layer_norm(C, ggml_add(C, t, ffn_gelu_erf(C, l.fc1_w, l.fc1_b, l.fc2_w, l.fc2_b, t)), + l.ln2_w, l.ln2_b, kBertEps); + } + io.lang = linear(C, text_proj_w, text_proj_b, t); + if (bert_len < text_len_max) + io.lang = ggml_concat(C, io.lang, ggml_repeat_4d(C, text_proj_b, hidden, text_len_max - bert_len, 1, 1), 1); + ggml_set_output(io.lang); + + ggml_cgraph * gf = ggml_new_graph_custom(C, 8192, false); + ggml_build_forward_expand(gf, io.lang); + return gf; +} + +ggml_cgraph * TurboVlaModelArch::build(ggml_context * C, IO & io, ggml_tensor * text_out) const { const int64_t NP = n_patches(), T = vit_seq(), nv = n_views, VS = nv*NP, LT = text_len_max; const int64_t pdim = 3*patch*patch; @@ -325,10 +360,12 @@ ggml_cgraph * TurboVlaModelArch::build(ggml_context * C, IO & io, int64_t bert_l x = ggml_add(C, x, ffn_gelu_erf(C, l.fc1_w, l.fc1_b, l.fc2_w, l.fc2_b, layer_norm(C, x, l.ln2_w, l.ln2_b, kLnEps))); } - // TurboVLA reads hidden_states[-1], which is before DINOv3's final norm - // (models/vision_encoder.py:109), then drops the CLS and register tokens. + // TurboVLA reads hidden_states[-1] (models/vision_encoder.py:109), which is + // after DINOv3's final norm in the transformers it was trained with, then + // drops the CLS and register tokens. The norm is per token. x = ggml_view_3d(C, x, vit_dim, NP, nv, x->nb[1], x->nb[2], (size_t) (1 + n_reg)*x->nb[1]); x = ggml_reshape_2d(C, ggml_cont(C, x), vit_dim, VS); + x = layer_norm(C, x, vit_norm_w, vit_norm_b, kLnEps); // VisionProjection, then one learned embedding per camera. ggml_tensor * mlp = linear(C, vp_fc1_w, vp_fc1_b, layer_norm(C, x, vp_in_w, vp_in_b, kLnEps)); @@ -337,26 +374,7 @@ ggml_cgraph * TurboVlaModelArch::build(ggml_context * C, IO & io, int64_t bert_l v = ggml_reshape_2d(C, ggml_add(C, ggml_reshape_3d(C, v, hidden, NP, nv), ggml_reshape_3d(C, view_emb, hidden, 1, nv)), hidden, VS); - // BERT over the checkpoint's padded length, token type 0 throughout. - io.ids = ggml_new_tensor_1d(C, GGML_TYPE_I32, bert_len); - io.pos = ggml_new_tensor_1d(C, GGML_TYPE_I32, bert_len); - io.bert_mask = ggml_new_tensor_2d(C, GGML_TYPE_F32, bert_len, bert_len); - ggml_set_input(io.ids); - ggml_set_input(io.pos); - ggml_set_input(io.bert_mask); - ggml_tensor * t = ggml_add(C, ggml_get_rows(C, word_emb, io.ids), ggml_get_rows(C, pos_emb, io.pos)); - t = ggml_add(C, t, ggml_view_1d(C, type_emb, bert_dim, 0)); - t = layer_norm(C, t, emb_ln_w, emb_ln_b, kBertEps); - for (const BertLayerW & l : bert) { - ggml_tensor * att = self_attention(C, linear(C, l.qkv_w, l.qkv_b, t), io.bert_mask, - bert_dim / bert_heads, bert_heads, bert_len, 1); - t = layer_norm(C, ggml_add(C, t, linear(C, l.o_w, l.o_b, att)), l.ln1_w, l.ln1_b, kBertEps); - t = layer_norm(C, ggml_add(C, t, ffn_gelu_erf(C, l.fc1_w, l.fc1_b, l.fc2_w, l.fc2_b, t)), - l.ln2_w, l.ln2_b, kBertEps); - } - ggml_tensor * lang = linear(C, text_proj_w, text_proj_b, t); - if (bert_len < LT) - lang = ggml_concat(C, lang, ggml_repeat_4d(C, text_proj_b, hidden, LT - bert_len, 1, 1), 1); + ggml_tensor * lang = ggml_view_tensor(C, text_out); // Fusion. One score matrix serves both directions upstream (the language // side softmaxes its transpose); two small products are cheaper than a @@ -365,6 +383,8 @@ ggml_cgraph * TurboVlaModelArch::build(ggml_context * C, IO & io, int64_t bert_l io.fus_mask = ggml_new_tensor_2d(C, GGML_TYPE_F32, LT, VS); ggml_set_input(io.enh_mask); ggml_set_input(io.fus_mask); + ggml_set_output(io.enh_mask); + ggml_set_output(io.fus_mask); const int64_t fhd = fusion_dim / fusion_heads, ehd = hidden / enh_heads; const float fscale = 1.0f / std::sqrt((float) fhd); for (int64_t i = 0; i < fusion_layers; ++i) { @@ -470,11 +490,21 @@ bool load_config(const gguf_reader & g, TurboVlaModelArch & m) { I("period_token_id", m.period_id); I("question_token_id", m.question_id); if (g.has("turbovla.rope_theta")) m.rope_theta = g.f32("turbovla.rope_theta"); + bool ok = std::isfinite(m.rope_theta) && m.rope_theta > 0.f && m.image_size <= 4096; + for (int64_t v : { m.hidden, m.n_views, m.image_size, m.patch, m.vit_dim, m.vit_layers, m.vit_heads, + m.bert_dim, m.bert_layers, m.bert_heads, m.vocab, m.fusion_layers, m.fusion_heads, + m.enh_heads, m.dec_layers, m.dec_heads, m.horizon, m.action_dim, m.state_dim, + m.n_state_tok, m.text_len_max }) + ok = ok && v >= 1; + if (!ok) { + std::fprintf(stderr, "vla(turbovla): inconsistent dimensions in GGUF metadata\n"); + return false; + } int64_t fhd = m.fusion_dim / m.fusion_heads; U("fusion_head_dim", fhd); m.fusion_dim = fhd * m.fusion_heads; - if (m.image_size % m.patch || m.vit_dim % m.vit_heads || (m.vit_dim / m.vit_heads) % 4 || + if (m.fusion_dim < 1 || m.image_size % m.patch || m.vit_dim % m.vit_heads || (m.vit_dim / m.vit_heads) % 4 || m.bert_dim % m.bert_heads || m.hidden % m.enh_heads || m.hidden % m.dec_heads) { std::fprintf(stderr, "vla(turbovla): inconsistent dimensions in GGUF metadata\n"); return false; @@ -556,6 +586,13 @@ bool load_weights(TurboVlaModelArch & m, gguf_reader & g) { w.fc2_w = L.gemm("%s", N(f, i, "fc2.weight").c_str()); w.fc2_b = L.f32("%s", N(f, i, "fc2.bias").c_str()); } + if (gguf_find_tensor(g.gctx, "vit.norm.weight") < 0) { + std::fprintf(stderr, "vla(turbovla): GGUF has no DINOv3 final norm (vit.norm); re-convert it with " + "scripts/convert_turbovla_to_gguf.py\n"); + return false; + } + m.vit_norm_w = L.f32("vit.norm.weight"); + m.vit_norm_b = L.f32("vit.norm.bias"); m.vp_in_w = L.f32("vit_proj.input_norm.weight"); m.vp_in_b = L.f32("vit_proj.input_norm.bias"); @@ -681,7 +718,7 @@ bool load_weights(TurboVlaModelArch & m, gguf_reader & g) { if (m.vocab != m.word_emb->ne[1] || m.text_len_max > m.bert_max_pos || m.patch_w->ne[0] != 3*m.patch*m.patch || (m.reg_tok && m.reg_tok->ne[1] != m.n_reg) || ggml_nelements(m.view_emb) != m.hidden*m.n_views || m.act_q->ne[1] != m.horizon || - m.dec[0].cross_qkv_w->ne[1] != 3*m.hidden) { + m.act_q->type != GGML_TYPE_F32 || m.dec[0].cross_qkv_w->ne[1] != 3*m.hidden) { std::fprintf(stderr, "vla(turbovla): tensor shapes disagree with GGUF metadata\n"); return false; } @@ -794,23 +831,21 @@ std::vector TurboVlaModelArch::predict(const Inputs& in) { std::vector ids((size_t) bert_len, pad_id); std::copy(in.lang_tokens, in.lang_tokens + in.n_lang, ids.begin()); - std::vector allowed; - std::vector pos; - special_token_blocks(ids, *this, allowed, pos); - const float NEG = -INFINITY; - std::vector bert_mask((size_t) bert_len*bert_len), enh_mask((size_t) LT*LT, NEG), fus_mask((size_t) LT*VS); - for (int64_t q = 0; q < bert_len; ++q) - for (int64_t k = 0; k < bert_len; ++k) { - bert_mask[(size_t) q*bert_len + k] = allowed[(size_t) q*bert_len + k] ? 0.0f : NEG; - enh_mask[(size_t) q*LT + k] = bert_mask[(size_t) q*bert_len + k]; - } - for (int64_t q = bert_len; q < LT; ++q) - enh_mask[(size_t) q*LT + q] = 0.0f; - for (int64_t k = 0; k < LT; ++k) { - const float val = (k < bert_len && ids[(size_t) k] != pad_id) ? 0.0f : NEG; - for (int64_t q = 0; q < VS; ++q) - fus_mask[(size_t) q*LT + k] = val; + const size_t arena = ggml_tensor_overhead()*8192 + ggml_graph_overhead_custom(8192, false); + bool fresh = false; + if (!text.ensure(backend, Key{bert_len}, arena, + [&](ggml_context * C, TextIO & t) { fresh = true; return build_text(C, t, bert_len); })) { + std::fprintf(stderr, "vla(turbovla): graph build/alloc failed\n"); + return {}; } + if (fresh) + graph.release(); + if (!graph.ensure(backend, Key{bert_len}, arena, + [&](ggml_context * C, IO & io) { fresh = true; return build(C, io, text.io().lang); })) { + std::fprintf(stderr, "vla(turbovla): graph build/alloc failed\n"); + return {}; + } + IO & io = graph.io(); std::vector patches((size_t) 3*patch*patch*VS); for (int64_t i = 0; i < n_views; ++i) @@ -819,22 +854,44 @@ std::vector TurboVlaModelArch::predict(const Inputs& in) { if (in.state) std::copy(in.state, in.state + state_dim, state.begin()); - const size_t arena = ggml_tensor_overhead()*8192 + ggml_graph_overhead_custom(8192, false); - if (!graph.ensure(backend, Key{bert_len}, arena, - [&](ggml_context * C, IO & io) { return build(C, io, bert_len); })) { - std::fprintf(stderr, "vla(turbovla): graph build/alloc failed\n"); - return {}; - } - IO & io = graph.io(); - ggml_backend_tensor_set(io.patches, patches.data(), 0, ggml_nbytes(io.patches)); - ggml_backend_tensor_set(io.ids, ids.data(), 0, ggml_nbytes(io.ids)); - ggml_backend_tensor_set(io.pos, pos.data(), 0, ggml_nbytes(io.pos)); - ggml_backend_tensor_set(io.bert_mask, bert_mask.data(), 0, ggml_nbytes(io.bert_mask)); - ggml_backend_tensor_set(io.enh_mask, enh_mask.data(), 0, ggml_nbytes(io.enh_mask)); - ggml_backend_tensor_set(io.fus_mask, fus_mask.data(), 0, ggml_nbytes(io.fus_mask)); - ggml_backend_tensor_set(io.state, state.data(), 0, ggml_nbytes(io.state)); + ggml_backend_tensor_set(io.patches, patches.data(), 0, ggml_nbytes(io.patches)); + ggml_backend_tensor_set(io.state, state.data(), 0, ggml_nbytes(io.state)); const auto tc = clock::now(); + if (fresh || ids != text_ids) { + text_ids.clear(); + std::vector allowed; + std::vector pos; + special_token_blocks(ids, *this, allowed, pos); + const float NEG = -INFINITY; + std::vector bert_mask((size_t) bert_len*bert_len), enh_mask((size_t) LT*LT, NEG), fus_mask((size_t) LT*VS); + for (int64_t q = 0; q < bert_len; ++q) + for (int64_t k = 0; k < bert_len; ++k) { + bert_mask[(size_t) q*bert_len + k] = allowed[(size_t) q*bert_len + k] ? 0.0f : NEG; + enh_mask[(size_t) q*LT + k] = bert_mask[(size_t) q*bert_len + k]; + } + for (int64_t q = bert_len; q < LT; ++q) + enh_mask[(size_t) q*LT + q] = 0.0f; + for (int64_t k = 0; k < LT; ++k) { + const float val = (k < bert_len && ids[(size_t) k] != pad_id) ? 0.0f : NEG; + for (int64_t q = 0; q < VS; ++q) + fus_mask[(size_t) q*LT + k] = val; + } + + TextIO & t = text.io(); + ggml_backend_tensor_set(t.ids, ids.data(), 0, ggml_nbytes(t.ids)); + ggml_backend_tensor_set(t.pos, pos.data(), 0, ggml_nbytes(t.pos)); + ggml_backend_tensor_set(t.bert_mask, bert_mask.data(), 0, ggml_nbytes(t.bert_mask)); + ggml_backend_tensor_set(io.enh_mask, enh_mask.data(), 0, ggml_nbytes(io.enh_mask)); + ggml_backend_tensor_set(io.fus_mask, fus_mask.data(), 0, ggml_nbytes(io.fus_mask)); + graph_unique_names(text.graph()); + if (ggml_backend_graph_compute(backend, text.graph()) != GGML_STATUS_SUCCESS) { + std::fprintf(stderr, "vla(turbovla): compute failed\n"); + return {}; + } + text_ids = ids; + } + graph_unique_names(graph.graph()); if (ggml_backend_graph_compute(backend, graph.graph()) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(turbovla): compute failed\n"); diff --git a/src/models/vla_adapter.cpp b/src/models/vla_adapter.cpp index 3a308da..e32cbbb 100644 --- a/src/models/vla_adapter.cpp +++ b/src/models/vla_adapter.cpp @@ -19,22 +19,20 @@ #include "modules/dual_tower.h" #include "ggml.h" -#include "ggml-cpu.h" #include "ggml-backend.h" #include "backend.h" -#include "gguf.h" #include "gguf_reader.h" #include "scratch_ctx.h" +#include "layers/attn.h" #include "layers/embed.h" -#include "env_flag.h" +#include "layers/linear.h" +#include "layers/norm.h" +#include "layers/rope.h" #include #include #include #include -#include -#include -#include #include #include #include @@ -42,64 +40,6 @@ namespace vla { namespace { -bool parse_stats(const std::string & js, int64_t want, std::vector & q01, - std::vector & q99, std::vector & mask, std::string & suite) { - auto find_key = [&](size_t from, const std::string & key) -> size_t { - const std::string pat = "\"" + key + "\""; - return js.find(pat, from); - }; - - const char * env = std::getenv("VLA_ADAPTER_UNNORM_KEY"); - size_t suite_pos; - if (env) { - suite = env; - suite_pos = find_key(0, suite); - } - else { - size_t b = js.find('{'); size_t q = js.find('"', b); - size_t qe = js.find('"', q+1); - suite = js.substr(q+1, qe-q-1); suite_pos = q; - } - if (suite_pos == std::string::npos) { - std::fprintf(stderr, "vla(vla_adapter): suite '%s' not in stats\n", suite.c_str()); - return false; - } - size_t act = find_key(suite_pos, "action"); - if (act == std::string::npos) - return false; - auto read_arr = [&](const std::string & key, std::vector & out) -> bool { - size_t k = find_key(act, key); if (k == std::string::npos) return false; - size_t lb = js.find('[', k); size_t rb = js.find(']', lb); - if (lb == std::string::npos || rb == std::string::npos) - return false; - out.clear(); size_t p = lb+1; - while (p < rb) { - while (p < rb && (js[p] == ',' || js[p] == ' ' || js[p] == '\n' || js[p] == '\t' || js[p] == '\r')) - ++p; - if (p >= rb) - break; - bool t = (js.compare(p, 4, "true") == 0), f = (js.compare(p, 5, "false") == 0); - if (t || f) { - out.push_back(t ? 1.0f : 0.0f); - p += t ? 4 : 5; - } - else { - out.push_back(std::strtof(js.c_str()+p, nullptr)); - while (p < rb && js[p] != ',') - ++p; - } - } - return true; - }; - std::vector mk; - if (!read_arr("q01", q01) || !read_arr("q99", q99)) - return false; - if (!read_arr("mask", mk)) - mk.assign(want, 1.0f); - mask.assign(mk.size(), 1); for (size_t i=0; i vision_graph; struct MainKey { int64_t seq=-1, n_views=-1, nprompt=-1; @@ -131,16 +71,13 @@ struct VlaAdapterModelArch : public ModelArchBase { ggml_tensor *t_ids=nullptr,*t_proj=nullptr,*t_pos=nullptr,*t_mask=nullptr; ggml_tensor *t_state=nullptr,*t_x0=nullptr,*norm_actions=nullptr; ggml_tensor *cT=nullptr,*sT=nullptr,*cA=nullptr,*sA=nullptr,*cK=nullptr,*sK=nullptr; + bool consts=false; }; graph_cache main_graph; ggml_backend_buffer_t weight_buf = nullptr; ggml_type mt = GGML_TYPE_BF16; - int64_t d_hidden=1024,d_layers=23,d_heads=16,d_head_dim=64,d_inter=4096; - int64_t s_hidden=1152,s_layers=26,s_heads=16,s_head_dim=72,s_inter=4304; - int64_t image_size=224,patch_size=14,n_patches=256,proj_mid=8704,vdim=2176; - float vit_ln_eps=1e-6f; - int64_t lm_hidden=896,lm_layers=24,n_q=14,n_kv=2,lm_head_dim=64,lm_inter=4864,vocab=151936; + int64_t lm_hidden=896,lm_layers=24,n_q=14,n_kv=2,lm_head_dim=64,vocab=151936; float lm_rope_base=1e6f, lm_rms_eps=1e-6f; int64_t chunk=8,action_dim=7,proprio_dim=8,num_tokens=64,head_blocks=24,head_heads=8,head_dim=112; float head_rope_base=1e4f, head_ln_eps=1e-5f; @@ -151,33 +88,11 @@ struct VlaAdapterModelArch : public ModelArchBase { ggml_tensor *h_ln1w,*h_ln1b,*h_fc1w,*h_fc1b,*h_ln2w,*h_ln2b,*h_fc2w,*h_fc2b; ggml_tensor *pp_fc1w,*pp_fc1b,*pp_fc2w,*pp_fc2b; std::vector hblk; - std::vector q01,q99; std::vector unnorm_mask; std::string suite; + Q99Stats act_stats; std::vector predict(const Inputs& in) override; }; -namespace { - -// Interleaved rotation, paired with the half-split frequency table in fill_cs, so -// a rotation pair gets two different angles. The reference does the same -// (action_heads.py:163 vs :137-140) and the weights were trained on it. Leave it. -static ggml_tensor* hrot(ggml_context*C, ggml_tensor*x, int64_t HD){ - int64_t L=x->ne[1],H=x->ne[2]; - ggml_tensor*xp=ggml_reshape_4d(C,x,2,HD/2,L,H); - ggml_tensor*ev=ggml_cont(C,ggml_view_4d(C,xp,1,HD/2,L,H,xp->nb[1],xp->nb[2],xp->nb[3],0)); - ggml_tensor*od=ggml_cont(C,ggml_view_4d(C,xp,1,HD/2,L,H,xp->nb[1],xp->nb[2],xp->nb[3],xp->nb[0])); - return ggml_reshape_3d(C,ggml_concat(C,ggml_scale(C,od,-1.0f),ev,0),HD,L,H); -} -static ggml_tensor* hheads(ggml_context*C, ggml_tensor*p, int64_t HD, int64_t NH){ - return ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,p,HD,NH,p->ne[1]),0,2,1,3)); -} -static ggml_tensor* hrope(ggml_context*C, ggml_tensor*x, ggml_tensor*cs, ggml_tensor*sn, int64_t HD){ - ggml_tensor*c=ggml_reshape_3d(C,cs,HD,x->ne[1],1),*s=ggml_reshape_3d(C,sn,HD,x->ne[1],1); - return ggml_add(C,ggml_mul(C,x,c),ggml_mul(C,hrot(C,x,HD),s)); -} - -} - std::unique_ptr vla_adapter_create(const std::string& mmproj_path, const std::string& ckpt_path, const std::string&, @@ -197,16 +112,10 @@ std::unique_ptr vla_adapter_create(const std::string& mmproj_path auto U=[&](const char*k,int64_t&d){ if(g.has(k)) d=(int64_t)g.u32(k); }; auto F=[&](const char*k,float&d){ if(g.has(k)) d=g.f32(k); }; - U("vla_adapter.vit.dino.hidden",m->d_hidden); U("vla_adapter.vit.dino.layers",m->d_layers); - U("vla_adapter.vit.dino.heads",m->d_heads); U("vla_adapter.vit.dino.head_dim",m->d_head_dim); U("vla_adapter.vit.dino.inter",m->d_inter); - U("vla_adapter.vit.sig.hidden",m->s_hidden); U("vla_adapter.vit.sig.layers",m->s_layers); - U("vla_adapter.vit.sig.heads",m->s_heads); U("vla_adapter.vit.sig.head_dim",m->s_head_dim); U("vla_adapter.vit.sig.inter",m->s_inter); - U("vla_adapter.vit.image_size",m->image_size); U("vla_adapter.vit.patch_size",m->patch_size); - U("vla_adapter.vit.n_patches",m->n_patches); F("vla_adapter.vit.ln_eps",m->vit_ln_eps); - U("vla_adapter.vit.proj_mid",m->proj_mid); U("vla_adapter.vit.vdim",m->vdim); + m->vis.read_config(g, "vla_adapter"); U("vla_adapter.lm.hidden",m->lm_hidden); U("vla_adapter.lm.layers",m->lm_layers); U("vla_adapter.lm.q_heads",m->n_q); U("vla_adapter.lm.kv_heads",m->n_kv); U("vla_adapter.lm.head_dim",m->lm_head_dim); - U("vla_adapter.lm.inter",m->lm_inter); U("vla_adapter.lm.vocab",m->vocab); + U("vla_adapter.lm.vocab",m->vocab); F("vla_adapter.lm.rope_theta",m->lm_rope_base); F("vla_adapter.lm.rms_eps",m->lm_rms_eps); U("vla_adapter.action.chunk",m->chunk); U("vla_adapter.action.action_dim",m->action_dim); U("vla_adapter.action.proprio_dim",m->proprio_dim); U("vla_adapter.action.num_tokens",m->num_tokens); @@ -223,12 +132,12 @@ std::unique_ptr vla_adapter_create(const std::string& mmproj_path } if (g.has("vla_adapter.statistics_json")) { - if (!parse_stats(g.str("vla_adapter.statistics_json"), m->action_dim, m->q01, m->q99, m->unnorm_mask, m->suite)) + if (!m->act_stats.parse(g.str("vla_adapter.statistics_json"), "VLA_ADAPTER_UNNORM_KEY", "vla_adapter", m->action_dim)) { std::fprintf(stderr, "vla(vla_adapter): failed to parse statistics_json\n"); return nullptr; } - std::printf("vla(vla_adapter): unnorm suite = %s (q99 dim %zu)\n", m->suite.c_str(), m->q99.size()); + std::printf("vla(vla_adapter): unnorm suite = %s (q99 dim %zu)\n", m->act_stats.suite.c_str(), m->act_stats.q99.size()); } { @@ -241,14 +150,11 @@ std::unique_ptr vla_adapter_create(const std::string& mmproj_path ggml_init_params wp = { (size_t)64*1024*1024, nullptr, true }; m->ctx_weights = ggml_init(wp); - bool ok = true; WeightLoader L("vla_adapter", g, m->ctx_weights, m->mt); auto mm = [&](const char * n) { return L.gemm("%s", n); }; auto f32 = [&](const char * n) { return L.f32("%s", n); }; - char nm[96]; - auto P=[&](const char*fmt,int i){ std::snprintf(nm,sizeof(nm),fmt,i); return (const char*)nm; }; - m->vis.declare(L, m->d_layers, m->s_layers); + m->vis.declare(L); m->token_embd=mm("token_embd.weight"); m->action_queries=mm("action_queries.weight"); m->lm_out_norm=f32("lm.output_norm.weight"); m->lm.resize(m->lm_layers); @@ -273,12 +179,10 @@ std::unique_ptr vla_adapter_create(const std::string& mmproj_path w.Wva=mm(N("v_adapter.weight")); w.bva=f32(N("v_adapter.bias")); w.Wkt=mm(N("k_task.weight")); w.bkt=f32(N("k_task.bias")); w.Wvt=mm(N("v_task.weight")); w.bvt=f32(N("v_task.bias")); w.Wo=mm(N("o_proj.weight")); w.bo=f32(N("o_proj.bias")); w.flnw=f32(N("ffn_ln.weight")); w.flnb=f32(N("ffn_ln.bias")); w.flw=mm(N("ffn_lin.weight")); w.flb=f32(N("ffn_lin.bias")); - std::vector gv=g.read_f32(N("gating")); w.rg = gv.empty()?0.0f:std::tanh(gv[0]); } - (void)P; - if(!ok){ - std::fprintf(stderr,"vla(vla_adapter): weight setup failed\n"); - return nullptr; - } + std::vector gv=g.read_f32(N("gating")); + if(gv.empty()) + return nullptr; + w.rg=std::tanh(gv[0]); } if (!L.upload(m->backend, &m->weight_buf)) return nullptr; @@ -288,65 +192,28 @@ std::unique_ptr vla_adapter_create(const std::string& mmproj_path m->cfg.n_suffix = m->chunk; m->cfg.max_action_dim = m->action_dim; m->cfg.real_action_dim = m->action_dim; m->cfg.real_state_dim = m->proprio_dim; - m->cfg.max_state_dim = m->proprio_dim; m->cfg.n_img = m->n_patches; m->cfg.hidden = m->lm_hidden; + m->cfg.max_state_dim = m->proprio_dim; m->cfg.n_img = m->vis.n_patches; m->cfg.hidden = m->lm_hidden; m->cfg.n_lang = 512; return m; } -namespace { - -} - std::vector VlaAdapterModelArch::predict(const Inputs& in) { using clock = std::chrono::steady_clock; const auto t0 = clock::now(); stats = Stats{}; - const int64_t S=image_size, NP=n_patches, HC=lm_hidden, HD=head_dim, NH=head_heads; + const int64_t NP=vis.n_patches, HC=lm_hidden, HD=head_dim, NH=head_heads; const int64_t n_views = in.n_images; if (in.precomputed_img_emb) { std::fprintf(stderr, "vla(vla_adapter): precomputed_img_emb is not supported; the DINOv2+SigLIP tower is baked into the GGUF, pass raw images\n"); return {}; } + if (in.n_lang < 1 || !in.lang_tokens) { std::fprintf(stderr, "vla(vla_adapter): need >=1 lang token\n"); return {}; } if (n_views < 1) { std::fprintf(stderr, "vla(vla_adapter): need >=1 image view\n"); return {}; } if (!in.images) { std::fprintf(stderr, "vla(vla_adapter): n_images=%d but the images pointer is null\n", in.n_images); return {}; } - // towers read S*S*3 per view; reject any view that is not exactly SxS. - for (int64_t v=0; v 0.484375). The reference - // preprocesses in bf16, so these are the values it actually sees. - static const float DMEAN[3]={0.484375f,0.455078125f,0.40625f}, DSTD[3]={0.228515625f,0.2236328125f,0.224609375f}; - static const float SMEAN[3]={0.5f,0.5f,0.5f}, SSTD[3]={0.5f,0.5f,0.5f}; - std::vector proj_host((size_t)HC*NP*n_views); + ggml_tensor*proj=nullptr; { const auto tv=clock::now(); - ggml_context*C=vision_scratch.reset((size_t)64*1024*1024); - std::vector px_d(n_views), px_s(n_views); - std::vector cmb(n_views); - for(int v=0; v dbuf, sbuf; - for(int v=0;v(clock::now()-tv).count(); } @@ -382,8 +249,8 @@ std::vector VlaAdapterModelArch::predict(const Inputs& in) { ggml_tensor*t_proj=ggml_new_tensor_2d(C,GGML_TYPE_F32,HC,NPATCH); ggml_set_input(t_proj); ggml_tensor*mm_seq=ggml_concat(C,ggml_concat(C,e0,t_proj,1),erest,1); - ggml_tensor*t_pos=ggml_new_tensor_1d(C,GGML_TYPE_I32,SEQ); ggml_set_input(t_pos); - ggml_tensor*t_mask=ggml_new_tensor_2d(C,GGML_TYPE_F32,SEQ,SEQ); ggml_set_input(t_mask); + ggml_tensor*t_pos=ggml_new_tensor_1d(C,GGML_TYPE_I32,SEQ); ggml_set_input(t_pos); ggml_set_output(t_pos); + ggml_tensor*t_mask=ggml_new_tensor_2d(C,GGML_TYPE_F32,SEQ,SEQ); ggml_set_input(t_mask); ggml_set_output(t_mask); const float lsc=1.0f/std::sqrt((float)lm_head_dim); std::vector lout(lm_layers); ggml_tensor*x=mm_seq; for(int i=0;i VlaAdapterModelArch::predict(const Inputs& in) { ggml_tensor*qr=ggml_rope_ext(C,qh,t_pos,nullptr,(int)lm_head_dim,GGML_ROPE_TYPE_NEOX,0,lm_rope_base,1.0f,0.0f,1.0f,32.0f,1.0f); ggml_tensor*kr=ggml_rope_ext(C,kh,t_pos,nullptr,(int)lm_head_dim,GGML_ROPE_TYPE_NEOX,0,lm_rope_base,1.0f,0.0f,1.0f,32.0f,1.0f); ggml_tensor*Q=ggml_cont(C,ggml_permute(C,qr,0,2,1,3)),*K=ggml_cont(C,ggml_permute(C,kr,0,2,1,3)),*V=ggml_cont(C,ggml_permute(C,vh,1,2,0,3)); - ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_mul_mat_set_prec(kq,GGML_PREC_F32); + ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_prec_set_acc(kq,GGML_PREC_F32); ggml_tensor*aw=ggml_soft_max_ext(C,kq,t_mask,lsc,0.0f); ggml_tensor*kqv=ggml_mul_mat(C,V,aw); ggml_tensor*att=ggml_reshape_2d(C,ggml_cont(C,ggml_permute(C,kqv,0,2,1,3)),HC,SEQ); @@ -405,54 +272,50 @@ std::vector VlaAdapterModelArch::predict(const Inputs& in) { ggml_tensor*final_norm=ggml_mul(C,ggml_rms_norm(C,lout[lm_layers-1],lm_rms_eps),lm_out_norm); std::vector cond(head_blocks); - for(int i=0;istd::pair{ - ggml_tensor*cc=ggml_new_tensor_2d(C,GGML_TYPE_F32,HD,Lh); ggml_set_input(cc); - ggml_tensor*ss=ggml_new_tensor_2d(C,GGML_TYPE_F32,HD,Lh); ggml_set_input(ss); + ggml_tensor*cc=ggml_new_tensor_2d(C,GGML_TYPE_F32,HD,Lh); ggml_set_input(cc); ggml_set_output(cc); + ggml_tensor*ss=ggml_new_tensor_2d(C,GGML_TYPE_F32,HD,Lh); ggml_set_input(ss); ggml_set_output(ss); return {cc,ss}; }; auto [cT,sT]=cs_tensor(chunk); auto [cA,sA]=cs_tensor(num_tokens+1); auto [cK,sK]=cs_tensor(NPATCH); - ggml_tensor*t_x0=ggml_new_tensor_2d(C,GGML_TYPE_F32,action_dim*HC,chunk); ggml_set_input(t_x0); - ggml_tensor*hx=ggml_relu(C,ggml_add(C,ggml_mul_mat(C,h_fc1w,LN(C,t_x0,h_ln1w,h_ln1b,head_ln_eps)),h_fc1b)); + ggml_tensor*t_x0=ggml_new_tensor_2d(C,GGML_TYPE_F32,action_dim*HC,chunk); ggml_set_input(t_x0); ggml_set_output(t_x0); + ggml_tensor*hx=ggml_relu(C,linear(C,h_fc1w,h_fc1b,layer_norm(C,t_x0,h_ln1w,h_ln1b,head_ln_eps))); const float hsc=1.0f/std::sqrt((float)HD); for(int i=0;inb[1],0)); ggml_tensor*ha=ggml_cont(C,ggml_view_2d(C,cond[i],HC,num_tokens,cond[i]->nb[1],(NPATCH+NUM_PROMPT_TOKENS)*cond[i]->nb[1])); ggml_tensor*had=ggml_concat(C,ha,pvec,1); - auto lin=[&](ggml_tensor*Wt,ggml_tensor*bt,ggml_tensor*inp){ return ggml_add(C,ggml_mul_mat(C,Wt,inp),bt); }; - ggml_tensor*q=hrope(C,hheads(C,lin(w.Wq,w.bq,hx),HD,NH),cT,sT,HD); - ggml_tensor*kse=hrope(C,hheads(C,lin(w.Wks,w.bks,hx),HD,NH),cT,sT,HD); - ggml_tensor*vse=lin(w.Wvs,w.bvs,hx); - ggml_tensor*kad=hrope(C,hheads(C,lin(w.Wka,w.bka,had),HD,NH),cA,sA,HD); ggml_tensor*vad=lin(w.Wva,w.bva,had); - ggml_tensor*kta=hrope(C,hheads(C,lin(w.Wkt,w.bkt,ht),HD,NH),cK,sK,HD); ggml_tensor*vta=lin(w.Wvt,w.bvt,ht); - auto tov=[&](ggml_tensor*pp){ int64_t L=pp->ne[1]; return ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,pp,HD,NH,L),1,2,0,3)); }; + auto qk=[&](ggml_tensor*p,ggml_tensor*cs,ggml_tensor*sn){ return rope_pairwise(C,to_heads(C,p,HD,NH,p->ne[1]),cs,sn,HD); }; + auto tov=[&](ggml_tensor*p){ return to_heads_v(C,p,HD,NH,p->ne[1]); }; + ggml_tensor*q=qk(linear(C,w.Wq,w.bq,hx),cT,sT); + ggml_tensor*kse=qk(linear(C,w.Wks,w.bks,hx),cT,sT); + ggml_tensor*vse=linear(C,w.Wvs,w.bvs,hx); + ggml_tensor*kad=qk(linear(C,w.Wka,w.bka,had),cA,sA); ggml_tensor*vad=linear(C,w.Wva,w.bva,had); + ggml_tensor*kta=qk(linear(C,w.Wkt,w.bkt,ht),cK,sK); ggml_tensor*vta=linear(C,w.Wvt,w.bvt,ht); ggml_tensor*Vs=tov(vse),*Va=tov(vad),*VT=tov(vta); ggml_tensor*ss2=ggml_mul_mat(C,kse,q),*sa=ggml_mul_mat(C,kad,q),*sr=ggml_mul_mat(C,kta,q); - ggml_mul_mat_set_prec(ss2,GGML_PREC_F32); ggml_mul_mat_set_prec(sa,GGML_PREC_F32); ggml_mul_mat_set_prec(sr,GGML_PREC_F32); + ggml_prec_set_acc(ss2,GGML_PREC_F32); ggml_prec_set_acc(sa,GGML_PREC_F32); ggml_prec_set_acc(sr,GGML_PREC_F32); ggml_tensor*st2=ggml_scale(C,sr,w.rg); ggml_tensor*scr=ggml_concat(C,ggml_concat(C,ss2,sa,0),st2,0); ggml_tensor*attn=ggml_soft_max_ext(C,scr,nullptr,hsc,0.0f); ggml_tensor*Vc=ggml_concat(C,ggml_concat(C,Vs,Va,0),VT,0); ggml_tensor*kqv=ggml_mul_mat(C,Vc,attn); ggml_tensor*mg=ggml_reshape_2d(C,ggml_cont(C,ggml_permute(C,kqv,0,2,1,3)),HC,chunk); - ggml_tensor*out=lin(w.Wo,w.bo,mg); - ggml_tensor*res=ggml_add(C,out,hx); - ggml_tensor*ln=LN(C,res,w.flnw,w.flnb,head_ln_eps); - hx=ggml_relu(C,ggml_add(C,ggml_mul_mat(C,w.flw,ln),w.flb)); + ggml_tensor*res=ggml_add(C,linear(C,w.Wo,w.bo,mg),hx); + hx=ggml_relu(C,linear(C,w.flw,w.flb,layer_norm(C,res,w.flnw,w.flnb,head_ln_eps))); } - ggml_tensor*xn=LN(C,hx,h_ln2w,h_ln2b,head_ln_eps); - ggml_tensor*norm_actions=ggml_add(C,ggml_mul_mat(C,h_fc2w,xn),h_fc2b); ggml_set_output(norm_actions); + ggml_tensor*norm_actions=linear(C,h_fc2w,h_fc2b,layer_norm(C,hx,h_ln2w,h_ln2b,head_ln_eps)); ggml_set_output(norm_actions); gio.t_ids=t_ids; gio.t_proj=t_proj; gio.t_pos=t_pos; gio.t_mask=t_mask; gio.t_state=t_state; gio.t_x0=t_x0; gio.norm_actions=norm_actions; @@ -476,32 +339,24 @@ std::vector VlaAdapterModelArch::predict(const Inputs& in) { ids[NPROMPT+i]=1; ids[NPROMPT+num_tokens]=(int32_t)stop_id; ggml_backend_tensor_set(t_ids,ids.data(),0,ggml_nbytes(t_ids)); } - ggml_backend_tensor_set(t_proj,proj_host.data(),0,ggml_nbytes(t_proj)); - { + if(!ggml_are_same_shape(proj,t_proj)){ std::fprintf(stderr,"vla(vla_adapter): projector output does not match the LM width\n"); return {}; } + ggml_backend_tensor_copy(proj,t_proj); + if(!gio.consts){ std::vector pp(SEQ); for(int64_t i=0;i mk; build_causal_mask(SEQ, mk); - ggml_backend_tensor_set(t_mask,mk.data(),0,ggml_nbytes(t_mask)); } - { std::vector sv(proprio_dim,0.0f); for(int64_t i=0;i mk; build_causal_mask(SEQ, mk); + ggml_backend_tensor_set(t_mask,mk.data(),0,ggml_nbytes(t_mask)); std::vector zx((size_t)action_dim*HC*chunk,0.0f); ggml_backend_tensor_set(t_x0,zx.data(),0,ggml_nbytes(t_x0)); + auto fill_cs=[&](ggml_tensor*cc,ggml_tensor*ss,int64_t Lh){ std::vector cb,sb; rope_pairwise_table(HD,Lh,head_rope_base,cb,sb); + ggml_backend_tensor_set(cc,cb.data(),0,ggml_nbytes(cc)); ggml_backend_tensor_set(ss,sb.data(),0,ggml_nbytes(ss)); }; + fill_cs(cT,sT,chunk); fill_cs(cA,sA,num_tokens+1); fill_cs(cK,sK,NPATCH); + gio.consts=true; } - - auto fill_cs=[&](ggml_tensor*cc,ggml_tensor*ss,int64_t Lh){ std::vector cb(HD*Lh),sb(HD*Lh); const int64_t half=HD/2; - for(int64_t t=0;t sv(proprio_dim,0.0f); for(int64_t i=0;i VlaAdapterModelArch::predict(const Inputs& in) { ggml_backend_tensor_get(norm_actions,na.data(),0,na.size()*sizeof(float)); stats.ms_inference = std::chrono::duration(clock::now()-ti).count(); - const int64_t W = cfg.max_action_dim>0 ? cfg.max_action_dim : action_dim; - std::vector out((size_t)chunk*W,0.0f); - for(int64_t c=0;c out=act_stats.unnorm(na,chunk,action_dim,cfg.max_action_dim>0 ? cfg.max_action_dim : action_dim); stats.ms_total = std::chrono::duration(clock::now()-t0).count(); return out; } diff --git a/src/models/vla_jepa.cpp b/src/models/vla_jepa.cpp index 00667ec..6e55b11 100644 --- a/src/models/vla_jepa.cpp +++ b/src/models/vla_jepa.cpp @@ -17,48 +17,35 @@ #include "model.h" #include "ggml.h" -#include "ggml-cpu.h" #include "ggml-backend.h" #include "backend.h" -#include "gguf.h" #include "gguf_reader.h" #include "scratch_ctx.h" #include "layers/embed.h" -#include "layers/linear.h" -#include "layers/norm.h" +#include "layers/ffn.h" #include "modules/dit_head.h" +#include "modules/prompt.h" #include "modules/qwen3_lm.h" #include "modules/qwen3vl_vit.h" -#include "env_flag.h" #include -#include #include #include -#include #include -#include #include -#include -#include #include #include namespace vla { -namespace { - - -} struct VlaJepaModelArch : public ModelArchBase { VlaJepaModelArch() : ModelArchBase(Arch::VLA_JEPA) {} ~VlaJepaModelArch() override; - std::string gguf_path; ggml_backend_t backend = nullptr; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; - scratch_ctx vision_scratch; + graph_cache vision_graph; struct LmKey { int64_t seq=-1, nfuture=-1; bool operator==(const LmKey & o) const { @@ -68,7 +55,7 @@ struct VlaJepaModelArch : public ModelArchBase { struct LmIO { ggml_tensor *t_embeds=nullptr,*t_pos2=nullptr,*t_lmmask=nullptr,*t_emb_idx=nullptr; ggml_tensor *t_ds[3]={nullptr,nullptr,nullptr}; - ggml_tensor *eagle=nullptr,*conditioning=nullptr; + ggml_tensor *conditioning=nullptr; }; struct HeadKey { int64_t nsteps=-1; @@ -78,23 +65,18 @@ struct VlaJepaModelArch : public ModelArchBase { }; struct HeadIO { ggml_tensor *t_cond=nullptr,*t_state=nullptr,*t_x0=nullptr,*actions=nullptr; - std::vector t_tau, t_tproj; }; graph_cache lm_graph; graph_cache head_graph; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; - int64_t vit_hidden=1024, vit_layers=24, vit_heads=16, vit_inter=4096; - int64_t patch_size=16, temporal_patch=2, spatial_merge=2, vit_num_pos=2304, vit_patch_flat=1536, vit_merged_dim=4096; - int64_t deepstack_idx[3] = {5, 11, 17}; int64_t lm_hidden=2048, lm_layers=28, n_q=16, n_kv=8, lm_head_dim=128, lm_inter=6144, vocab=151936; int64_t image_token_index=151655, embodied_token_id=151697; - int64_t image_target_size=256; int64_t dit_hidden=768, dit_heads=12, dit_head_dim=64, dit_layers=16, cross_dim=2048, output_dim=1024, time_proj_dim=256; int64_t action_dim=7, state_dim=8, action_horizon=7, num_future=32, num_steps=4, num_buckets=1000; - float vit_ln_eps=1e-6f, vit_rope_base=10000.0f, lm_rms_eps=1e-6f, lm_rope_base=5000000.0f, connector_ln_eps=1e-6f; + float lm_rms_eps=1e-6f, lm_rope_base=5000000.0f; float dit_ln_eps=1e-5f, dit_norm_out_eps=1e-6f; Qwen3VLTower vit; @@ -106,57 +88,34 @@ struct VlaJepaModelArch : public ModelArchBase { ggml_tensor *ad_l1W=nullptr,*ad_l1b=nullptr,*ad_l2W=nullptr,*ad_l2b=nullptr; ggml_tensor *future_tokens=nullptr,*pos_embd=nullptr; - bool caches_ready = false; - std::vector c_grow, c_gcol; - std::vector c_rope_cos, c_rope_sin, c_pos_interp; - std::vector> c_tau, c_tproj; - std::vector c_mask; int64_t c_mask_seq = -1; - gguf_reader io; - bool build_caches(); + FlowTimes times; + std::vector c_mask; int64_t c_mask_seq = -1; + gguf_reader io{"vla_jepa"}; std::vector predict(const Inputs& in) override; }; namespace { - - -bool load_config(const gguf_reader & g, VlaJepaModelArch & m, Config & cfg) { +bool load_config(const gguf_reader & g, const Options & opts, VlaJepaModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; auto F = [&](const char * k, float & dst) { if (g.has(k)) dst = g.f32(k); }; - auto fk = [&](const char * s) { static char b[64]; std::snprintf(b, sizeof(b), "vla_jepa.%s", s); return b; }; - U(fk("vit_hidden"), m.vit_hidden); U(fk("vit_layers"), m.vit_layers); U(fk("vit_heads"), m.vit_heads); U(fk("vit_inter"), m.vit_inter); - U(fk("patch_size"), m.patch_size); U(fk("temporal_patch_size"), m.temporal_patch); U(fk("spatial_merge_size"), m.spatial_merge); - U(fk("vit_num_position_embeddings"), m.vit_num_pos); U(fk("vit_patch_flat"), m.vit_patch_flat); U(fk("vit_merged_dim"), m.vit_merged_dim); - U(fk("deepstack_idx_0"), m.deepstack_idx[0]); U(fk("deepstack_idx_1"), m.deepstack_idx[1]); U(fk("deepstack_idx_2"), m.deepstack_idx[2]); + auto fk = [&](const char * s) { thread_local char b[64]; std::snprintf(b, sizeof(b), "vla_jepa.%s", s); return b; }; + if (!m.vit.load_config("vla_jepa", g, "vla_jepa")) + return false; U(fk("lm_hidden"), m.lm_hidden); U(fk("lm_layers"), m.lm_layers); U(fk("lm_q_heads"), m.n_q); U(fk("lm_kv_heads"), m.n_kv); U(fk("lm_head_dim"), m.lm_head_dim); U(fk("lm_inter"), m.lm_inter); U(fk("vocab_size"), m.vocab); U(fk("image_token_index"), m.image_token_index); U(fk("embodied_action_token_id"), m.embodied_token_id); - U(fk("image_target_size"), m.image_target_size); U(fk("dit_hidden"), m.dit_hidden); U(fk("dit_heads"), m.dit_heads); U(fk("dit_head_dim"), m.dit_head_dim); U(fk("dit_layers"), m.dit_layers); U(fk("cross_dim"), m.cross_dim); U(fk("output_dim"), m.output_dim); U(fk("time_proj_dim"), m.time_proj_dim); U(fk("action_dim"), m.action_dim); U(fk("state_dim"), m.state_dim); U(fk("action_horizon"), m.action_horizon); U(fk("num_future_tokens"), m.num_future); U(fk("num_inference_timesteps"), m.num_steps); U(fk("num_timestep_buckets"), m.num_buckets); - if (const char * ns = std::getenv("VLA_NUM_STEPS")) { - char * end = nullptr; long v = std::strtol(ns, &end, 10); - if (end && *end == '\0' && v >= 1) { - m.num_steps = (int64_t) v; - std::fprintf(stderr, "vla(vla_jepa): VLA_NUM_STEPS override → num_steps=%lld\n", (long long) v); - } - } - F(fk("vit_ln_eps"), m.vit_ln_eps); F(fk("lm_rms_eps"), m.lm_rms_eps); F(fk("connector_ln_eps"), m.connector_ln_eps); - F(fk("vit_rope_theta"), m.vit_rope_base); F(fk("dit_ln_eps"), m.dit_ln_eps); F(fk("dit_norm_out_eps"), m.dit_norm_out_eps); + if (!resolve_num_steps("vla_jepa", opts, m.num_steps)) + return false; + F(fk("lm_rms_eps"), m.lm_rms_eps); F(fk("dit_ln_eps"), m.dit_ln_eps); F(fk("dit_norm_out_eps"), m.dit_norm_out_eps); if (g.has(fk("lm_rope_theta"))) m.lm_rope_base = (float) g.f64(fk("lm_rope_theta")); - // merge_block_coords only enumerates the patch grid exactly when the spatial - // merge divides it; otherwise it emits rows past the position table. - if (m.patch_size <= 0 || m.spatial_merge <= 0 || m.image_target_size%m.patch_size != 0 || - (m.image_target_size/m.patch_size)%m.spatial_merge != 0) { - std::fprintf(stderr, "vla(vla_jepa): image %lld / patch %lld / merge %lld do not divide evenly\n", - (long long) m.image_target_size, (long long) m.patch_size, (long long) m.spatial_merge); - return false; - } // timesteps_proj always emits 256 floats into the time-projection input. if (m.time_proj_dim != 256) { std::fprintf(stderr, "vla(vla_jepa): time_proj_dim %lld, expected 256\n", (long long) m.time_proj_dim); @@ -187,7 +146,7 @@ bool load_config(const gguf_reader & g, VlaJepaModelArch & m, Config & cfg) { m.dit.cfg.norm_out_eps = m.dit_norm_out_eps; cfg = Config{}; - cfg.n_img = (m.image_target_size/m.patch_size/m.spatial_merge)*(m.image_target_size/m.patch_size/m.spatial_merge); + cfg.n_img = m.vit.n_tokens(); cfg.n_lang = 1024; cfg.n_state = 1; cfg.n_suffix = m.action_horizon; cfg.max_state_dim = m.state_dim; cfg.max_action_dim = m.action_dim; cfg.real_state_dim = m.state_dim; cfg.real_action_dim = m.action_dim; @@ -220,21 +179,20 @@ std::unique_ptr vla_jepa_create(const std::string& mmproj_path, std::printf("vla(vla_jepa): note - mmproj '%s' is ignored (the vision tower is bundled in the combined GGUF)\n", mmproj_path.c_str()); auto m = std::make_unique(); - m->gguf_path = ckpt_path; m->matmul_type = opts.weight_dtype.value_or(vla::default_weight_dtype(GGML_TYPE_BF16)); - gguf_reader g("vla_jepa"); - if (!g.open(ckpt_path)) + if (!m->io.open(ckpt_path)) return nullptr; + gguf_reader & g = m->io; if (!g.has("vla_jepa.architecture")) { std::fprintf(stderr, "vla(vla_jepa): %s is not a vla_jepa GGUF\n", ckpt_path.c_str()); return nullptr; } - if (!load_config(g, *m, m->cfg)) + if (!load_config(g, opts, *m, m->cfg)) return nullptr; std::printf("vla(vla_jepa): vit=Qwen3-VL %lldd×%lldL (deepstack@{%lld,%lld,%lld}, merge÷%lld) lm=Qwen3-VL %lldd×%lldL (%lldq/%lldkv×%lld, θ=%g) " "dit-B %lldL×%lldh×%lld(inner %lld, cross %lld, out %lld) horizon=%lld action_dim=%lld state_dim=%lld future=%lld N_steps=%lld resident=%s\n", - (long long) m->vit_hidden, (long long) m->vit_layers, (long long) m->deepstack_idx[0], (long long) m->deepstack_idx[1], (long long) m->deepstack_idx[2], (long long) m->spatial_merge, + (long long) m->vit.hidden, (long long) m->vit.layers, (long long) m->vit.deepstack_idx[0], (long long) m->vit.deepstack_idx[1], (long long) m->vit.deepstack_idx[2], (long long) m->vit.merge, (long long) m->lm_hidden, (long long) m->lm_layers, (long long) m->n_q, (long long) m->n_kv, (long long) m->lm_head_dim, (double) m->lm_rope_base, (long long) m->dit_layers, (long long) m->dit_heads, (long long) m->dit_head_dim, (long long) m->dit_hidden, (long long) m->cross_dim, (long long) m->output_dim, (long long) m->action_horizon, (long long) m->action_dim, (long long) m->state_dim, (long long) m->num_future, (long long) m->num_steps, @@ -257,7 +215,7 @@ std::unique_ptr vla_jepa_create(const std::string& mmproj_path, WeightLoader L("vla_jepa", g, m->ctx_weights, m->matmul_type); - m->vit.declare(L, "vit", m->vit_layers); + m->vit.declare(L, "vit"); m->lm.declare(L, "vlm"); m->ae_l1W = L.f32("ah.act_enc.l1.weight"); m->ae_l1b = L.f32("ah.act_enc.l1.bias"); @@ -274,92 +232,29 @@ std::unique_ptr vla_jepa_create(const std::string& mmproj_path, if (!L.upload(m->backend, &m->weight_buf)) return nullptr; + if (!m->times.build("vla_jepa", m->backend, m->dit, m->num_steps, m->num_buckets, m->dit_hidden, m->action_horizon)) + return nullptr; std::printf("vla(vla_jepa): weights resident in %.2f GiB (%s)\n", ggml_backend_buffer_get_size(m->weight_buf)/(1024.0*1024.0*1024.0), dtype_name(m->matmul_type)); - if (!m->build_caches()) { - std::fprintf(stderr, "vla(vla_jepa): build_caches failed\n"); + if (!m->vit.build_caches("vla_jepa", m->io)) return nullptr; - } return m; } -bool VlaJepaModelArch::build_caches() { - if (caches_ready) - return true; - const int64_t side = image_target_size, ps = patch_size, m2 = spatial_merge; - const int64_t grid = side/ps; - const int64_t hd_vit = vit_hidden/vit_heads; - const int64_t num_side = (int64_t) std::lround(std::sqrt((double) vit_num_pos)); - - merge_block_coords(grid, grid, m2, c_grow, c_gcol); - vit_rope_tables(c_grow, c_gcol, hd_vit, (double) vit_rope_base, c_rope_cos, c_rope_sin); - - if (!io.open(gguf_path)) { - std::fprintf(stderr, "vla(vla_jepa): build_caches: io.open(%s) failed\n", gguf_path.c_str()); - return false; - } - std::vector pos_table = io.read_f32("vit.pos_embd"); - if (pos_table.empty() || (int64_t) pos_table.size() != vit_num_pos * vit_hidden) { - std::fprintf(stderr, "vla(vla_jepa): build_caches: vit.pos_embd unreadable\n"); return false; - } - interp_pos_embed(pos_table, num_side, vit_hidden, c_grow, c_gcol, grid, grid, c_pos_interp); - - c_tau.assign((size_t) num_steps, {}); c_tproj.assign((size_t) num_steps, {}); - for (int64_t s=0; s VlaJepaModelArch::predict(const Inputs& in) { const auto t0 = std::chrono::steady_clock::now(); stats = Stats{}; const int64_t H = lm_hidden, E = dit_hidden, AD = action_dim, AH = action_horizon, OUTD = output_dim; - const int64_t side = image_target_size, ps = patch_size, m2 = spatial_merge; - const int64_t grid = side/ps, n_patches = grid * grid, K = (grid/m2)*(grid/m2); - const int64_t hd_vit = vit_hidden/vit_heads; + const int64_t K = vit.n_tokens(); const int64_t Nseq = 1+num_future+AH; - const char * dump_prefix = std::getenv("VLA_JEPA_DUMP"); - if (!caches_ready) { std::fprintf(stderr, "vla(vla_jepa): caches not ready\n"); return {}; } - - auto dump_t = [&](const char * name, ggml_tensor * t) { - if (!dump_prefix) - return; - const int64_t n0 = t->ne[0], n1 = t->ne[1]; - std::vector buf((size_t) n0*std::max(1, n1)); - ggml_backend_tensor_get(t, buf.data(), 0, buf.size()*sizeof(float)); - char path[1024]; std::snprintf(path, sizeof(path), "%s_%s_%lldx%lld.f32", dump_prefix, name, (long long) n0, (long long) n1); - FILE * fp = std::fopen(path, "wb"); if (fp) { - std::fwrite(buf.data(), sizeof(float), buf.size(), fp); - std::fclose(fp); - } - }; - std::vector x_init((size_t) AH * AD); - if (in.noise) - std::memcpy(x_init.data(), in.noise, x_init.size()*sizeof(float)); - else { - std::mt19937 rng((uint32_t) std::chrono::steady_clock::now().time_since_epoch().count()); - std::normal_distribution nd(0.f, 1.f); - for (auto & v : x_init) - v = nd(rng); - } + std::vector x_init; + init_noise(in, (size_t) AH*AD, x_init); std::vector cond_host((size_t) H * num_future, 0.0f); - const char * cond_file = std::getenv("VLA_JEPA_COND"); - if (cond_file) { - FILE * fp = std::fopen(cond_file, "rb"); - if (!fp) { std::fprintf(stderr, "vla(vla_jepa): VLA_JEPA_COND open failed: %s\n", cond_file); return {}; } - const size_t want = cond_host.size(); - if (std::fread(cond_host.data(), sizeof(float), want, fp) != want) { std::fprintf(stderr, "vla(vla_jepa): VLA_JEPA_COND short read\n"); std::fclose(fp); return {}; } - std::fclose(fp); - std::printf("vla(vla_jepa): conditioning injected from %s (action-head isolation)\n", cond_file); - } else { + { if (in.precomputed_img_emb) { std::fprintf(stderr, "vla(vla_jepa): precomputed_img_emb is not supported. The V-JEPA tower also " "emits deepstack features that a single embedding buffer cannot carry; pass raw images.\n"); @@ -367,116 +262,36 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { } int64_t n_views = in.n_images; if (n_views <= 0) { std::fprintf(stderr, "vla(vla_jepa): no images in the request\n"); return {}; } - std::vector img_emb_host((size_t) n_views * K * H), ds_host[3]; - for (int j=0; j<3; ++j) - ds_host[j].assign((size_t) n_views * K * H, 0.0f); - - std::vector inj_patches; const char * patches_file = std::getenv("VLA_JEPA_PATCHES"); - if (patches_file) { - FILE * fp = std::fopen(patches_file, "rb"); - if (!fp) { std::fprintf(stderr, "vla(vla_jepa): VLA_JEPA_PATCHES open failed\n"); return {}; } - inj_patches.resize((size_t) n_views * n_patches * vit_patch_flat); - if (std::fread(inj_patches.data(), sizeof(float), inj_patches.size(), fp) != inj_patches.size()) { std::fprintf(stderr, "vla(vla_jepa): VLA_JEPA_PATCHES short read\n"); std::fclose(fp); return {}; } - std::fclose(fp); - std::printf("vla(vla_jepa): pixel_values injected from %s\n", patches_file); - } - if (inj_patches.empty() && !in.images) { std::fprintf(stderr, "vla(vla_jepa): n_images=%d but the images pointer is null\n", in.n_images); return {}; } - - ggml_context * VC = vision_scratch.reset((size_t) 512*1024*1024); - if (!VC) { std::fprintf(stderr, "vla(vla_jepa): ggml_init(vision ctx) failed\n"); return {}; } - ggml_tensor * t_patches = ggml_new_tensor_2d(VC, GGML_TYPE_F32, vit_patch_flat, n_patches); ggml_set_input(t_patches); - ggml_tensor * t_pos = ggml_new_tensor_2d(VC, GGML_TYPE_F32, vit_hidden, n_patches); ggml_set_input(t_pos); - ggml_tensor * t_cos = ggml_new_tensor_2d(VC, GGML_TYPE_F32, hd_vit, n_patches); ggml_set_input(t_cos); - ggml_tensor * t_sin = ggml_new_tensor_2d(VC, GGML_TYPE_F32, hd_vit, n_patches); ggml_set_input(t_sin); - ggml_tensor * h = ggml_add(VC, ggml_add(VC, ggml_mul_mat(VC, vit.patch_w, t_patches), vit.patch_b), t_pos); - ggml_set_output(h); - ggml_tensor * stash[3] = {nullptr, nullptr, nullptr}; - for (int64_t i=0; i img_emb_host, ds_host[3]; + if (!in.images) { std::fprintf(stderr, "vla(vla_jepa): n_images=%d but the images pointer is null\n", in.n_images); return {}; } + const auto tv0 = std::chrono::steady_clock::now(); - std::vector patches; - bool vok = true; - for (int64_t v=0; v(std::chrono::steady_clock::now()-tv0).count(); if (!vok) return {}; const int64_t n_img = n_views * K; - std::vector input_ids; - int64_t n_img_slots = 0; - for (int j=0; j inputs_embeds((size_t) SEQ * H); - if (!io.fetch_rows_f32("token_embd.weight", input_ids, inputs_embeds.data(), H)) return {}; - { int64_t k = 0; for (int64_t p=0; p inputs_embeds; + if (!fetch_embeds(io, prompt, img_emb_host.data(), H, inputs_embeds)) return {}; - std::vector image_pos_idx, emb_pos_idx; - for (int64_t p=0; p emb_pos_idx; + for (int64_t p=0; p pp; + if (!mrope_positions("vla_jepa", prompt.ids, (int32_t) image_token_index, vit.grid()/vit.merge, pp)) return {}; + std::vector> ds_pad(3); for (int j=0; j<3; ++j) { ds_pad[j].assign((size_t) SEQ * H, 0.0f); for (int64_t k=0; k VlaJepaModelArch::predict(const Inputs& in) { if (i < 3) hh = ggml_add(C, hh, t_ds[i]); } - ggml_tensor * eagle = hh; - ggml_set_output(eagle); - ggml_tensor * conditioning = ggml_get_rows(C, eagle, t_emb_idx); + ggml_tensor * conditioning = ggml_get_rows(C, hh, t_emb_idx); ggml_set_output(conditioning); gio.t_embeds=t_embeds; gio.t_pos2=t_pos2; gio.t_lmmask=t_lmmask; gio.t_emb_idx=t_emb_idx; gio.t_ds[0]=t_ds[0]; gio.t_ds[1]=t_ds[1]; gio.t_ds[2]=t_ds[2]; - gio.eagle=eagle; gio.conditioning=conditioning; + gio.conditioning=conditioning; ggml_cgraph * lg = ggml_new_graph_custom(C, 32768, false); ggml_build_forward_expand(lg, conditioning); @@ -516,46 +329,10 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { ggml_tensor * t_embeds = gio.t_embeds, * t_pos2 = gio.t_pos2, * t_lmmask = gio.t_lmmask; ggml_tensor * t_emb_idx = gio.t_emb_idx; ggml_tensor * t_ds[3] = { gio.t_ds[0], gio.t_ds[1], gio.t_ds[2] }; - ggml_tensor * eagle = gio.eagle, * conditioning = gio.conditioning; + ggml_tensor * conditioning = gio.conditioning; ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); - - { - const int64_t llm_grid = side/ps/m2; - std::vector pp((size_t) 4*SEQ, 0); - int64_t st = 0, st_idx = 0; - while (st < SEQ) { - int64_t img_start = -1; - for (int64_t i=st; i max_image_pos) max_image_pos = llm_grid-1; - st_idx = image_offset+max_image_pos+1; st = img_end; - } - std::memcpy(pp.data()+(size_t) 3*SEQ, pp.data(), (size_t) SEQ * sizeof(int32_t)); - ggml_backend_tensor_set(t_pos2, pp.data(), 0, ggml_nbytes(t_pos2)); - } + ggml_backend_tensor_set(t_pos2, pp.data(), 0, ggml_nbytes(t_pos2)); if (c_mask_seq != SEQ) { build_causal_mask(SEQ, c_mask); c_mask_seq = SEQ; @@ -569,86 +346,48 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { graph_unique_names(lg); if (ggml_backend_graph_compute(backend, lg) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(vla_jepa): LM compute failed\n"); return {}; } stats.ms_prefill = std::chrono::duration(std::chrono::steady_clock::now()-tp0).count(); - if (dump_prefix) { - dump_t("eagle", eagle); - dump_t("conditioning", conditioning); - } ggml_backend_tensor_get(conditioning, cond_host.data(), 0, cond_host.size()*sizeof(float)); } - // Dumping adds graph outputs, so it always rebuilds. - std::vector step_seq, step_pred, step_vel, step_act; - if (dump_prefix) - head_graph.release(); const HeadKey hkey{ num_steps }; - const bool head_built = head_graph.ensure(backend, hkey, (size_t) 256*1024*1024, + const size_t head_nodes = 8192 + (size_t) num_steps*64*(dit_layers+1); + const bool head_built = head_graph.ensure(backend, hkey, head_nodes*ggml_tensor_overhead() + ggml_graph_overhead_custom(head_nodes, false), [&](ggml_context * C, HeadIO & gio) -> ggml_cgraph * { ggml_tensor * t_cond = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, num_future); ggml_set_input(t_cond); ggml_tensor * t_state = ggml_new_tensor_2d(C, GGML_TYPE_F32, state_dim, 1); ggml_set_input(t_state); ggml_tensor * t_x0 = ggml_new_tensor_2d(C, GGML_TYPE_F32, AD, AH); ggml_set_input(t_x0); - std::vector t_tau(num_steps), t_tproj(num_steps); - for (int64_t s=0; s Kc(dit_layers, nullptr), Vc(dit_layers, nullptr); + for (int64_t i=0; inb[1], 0)); - ggml_tensor * seq = ggml_concat(C, ggml_concat(C, state_features, future, 1), af, 1); - step_seq[s] = seq; - ggml_tensor * x = seq; + ggml_tensor * x = ggml_concat(C, ggml_concat(C, state_features, future, 1), af, 1); for (int64_t i=0; inb[1], (size_t) (Nseq-AH)*model_output->nb[1])); - ggml_tensor * vel = ggml_add(C, ggml_mul_mat(C, ad_l2W, ggml_relu(C, ggml_add(C, ggml_mul_mat(C, ad_l1W, last), ad_l1b))), ad_l2b); - step_vel[s] = vel; + ggml_tensor * vel = ffn_relu(C, ad_l1W, ad_l1b, ad_l2W, ad_l2b, last); actions = ggml_add(C, actions, ggml_scale(C, vel, dt)); - step_act[s] = actions; - if (dump_prefix) { - ggml_set_output(step_seq[s]); - ggml_set_output(step_pred[s]); - ggml_set_output(step_vel[s]); - ggml_set_output(step_act[s]); - } } ggml_set_output(actions); gio.t_cond=t_cond; gio.t_state=t_state; gio.t_x0=t_x0; gio.actions=actions; - gio.t_tau=t_tau; gio.t_tproj=t_tproj; - ggml_cgraph * hg = ggml_new_graph_custom(C, 65536, false); + ggml_cgraph * hg = ggml_new_graph_custom(C, head_nodes, false); ggml_build_forward_expand(hg, actions); - if (dump_prefix) for (int64_t s=0; s VlaJepaModelArch::predict(const Inputs& in) { HeadIO & hio = head_graph.io(); ggml_cgraph * hg = head_graph.graph(); ggml_tensor * t_cond = hio.t_cond, * t_state = hio.t_state, * t_x0 = hio.t_x0, * actions = hio.actions; - std::vector & t_tau = hio.t_tau; std::vector & t_tproj = hio.t_tproj; ggml_backend_tensor_set(t_cond, cond_host.data(), 0, ggml_nbytes(t_cond)); { @@ -666,10 +404,6 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { ggml_backend_tensor_set(t_state, st.data(), 0, ggml_nbytes(t_state)); } ggml_backend_tensor_set(t_x0, x_init.data(), 0, ggml_nbytes(t_x0)); - for (int64_t s=0; s VlaJepaModelArch::predict(const Inputs& in) { stats.ms_denoise = std::chrono::duration(std::chrono::steady_clock::now()-td0).count(); stats.ms_inference = stats.ms_prefill+stats.ms_denoise; - if (dump_prefix) for (int64_t s=0; s out((size_t) AH * AD); ggml_backend_tensor_get(actions, out.data(), 0, out.size()*sizeof(float)); stats.ms_total = std::chrono::duration(std::chrono::steady_clock::now()-t0).count(); diff --git a/src/modules/action_expert.cpp b/src/modules/action_expert.cpp index 0f21fca..4849ef1 100644 --- a/src/modules/action_expert.cpp +++ b/src/modules/action_expert.cpp @@ -14,47 +14,202 @@ #include "modules/action_expert.h" +#include "backend.h" #include "layers/linear.h" +#include +#include +#include +#include + namespace vla { -void ActionExpert::declare(WeightLoader & L, const char * prefix) { - se_l1W = L.f32("%s.state_enc.l1.W", prefix); - se_l1b = L.f32("%s.state_enc.l1.b", prefix); - se_l2W = L.f32("%s.state_enc.l2.W", prefix); - se_l2b = L.f32("%s.state_enc.l2.b", prefix); +namespace { - ae_W1W = L.f32("%s.act_enc.W1.W", prefix); - ae_W1b = L.f32("%s.act_enc.W1.b", prefix); - ae_W2W = L.f32("%s.act_enc.W2.W", prefix); - ae_W2b = L.f32("%s.act_enc.W2.b", prefix); - ae_W3W = L.f32("%s.act_enc.W3.W", prefix); - ae_W3b = L.f32("%s.act_enc.W3.b", prefix); +bool read_slab(gguf_reader & g, const char * name, int64_t id, int64_t n, std::vector & out) { + const ggml_tensor * t = g.meta(name); + const int64_t tid = gguf_find_tensor(g.gctx, name); + if (!t || tid < 0) { + std::fprintf(stderr, "vla(%s): missing tensor %s\n", g.arch, name); + return false; + } + if (id < 0 || id >= ggml_nelements(t)/n) { + std::fprintf(stderr, "vla(%s): embodiment id %lld out of range for %s\n", g.arch, (long long) id, name); + return false; + } + if (t->type != GGML_TYPE_F32 && t->type != GGML_TYPE_BF16 && t->type != GGML_TYPE_F16) { + std::fprintf(stderr, "vla(%s): tensor %s unsupported type %d\n", g.arch, name, (int) t->type); + return false; + } + const size_t es = ggml_type_size(t->type); + std::vector raw((size_t) n*es); + if (vla_fseek64(g.fp, g.data_off+gguf_get_tensor_offset(g.gctx, tid)+(uint64_t) id*raw.size()) != 0 || + std::fread(raw.data(), 1, raw.size(), g.fp) != raw.size()) { + std::fprintf(stderr, "vla(%s): read %s failed\n", g.arch, name); + return false; + } + out.resize((size_t) n); + if (t->type == GGML_TYPE_F32) + std::memcpy(out.data(), raw.data(), raw.size()); + else if (t->type == GGML_TYPE_BF16) + ggml_bf16_to_fp32_row(reinterpret_cast(raw.data()), out.data(), n); + else + ggml_fp16_to_fp32_row(reinterpret_cast(raw.data()), out.data(), n); + return true; +} - ad_l1W = L.f32("%s.act_dec.l1.W", prefix); - ad_l1b = L.f32("%s.act_dec.l1.b", prefix); - ad_l2W = L.f32("%s.act_dec.l2.W", prefix); - ad_l2b = L.f32("%s.act_dec.l2.b", prefix); +} + +ActionExpert::~ActionExpert() { + if (buf) + ggml_backend_buffer_free(buf); + if (ctx) + ggml_free(ctx); +} +bool ActionExpert::declare(WeightLoader & L, gguf_reader & g, ggml_backend_t backend, const char * prefix) { pos_embd = L.f32("%s.pos_embd", prefix); + + struct Proj { const char * name; ggml_tensor ** W; ggml_tensor ** b; }; + const Proj projs[] = { + {"state_enc.l1", &se_l1W, &se_l1b}, {"state_enc.l2", &se_l2W, &se_l2b}, + {"act_enc.W1", &ae_W1W, &ae_W1b}, {"act_enc.W2", &ae_W2W, &ae_W2b}, {"act_enc.W3", &ae_W3W, &ae_W3b}, + {"act_dec.l1", &ad_l1W, &ad_l1b}, {"act_dec.l2", &ad_l2W, &ad_l2b}, + }; + + ggml_init_params p = { 2*std::size(projs)*ggml_tensor_overhead(), nullptr, true }; + ctx = ggml_init(p); + if (!ctx) + return false; + char wn[192], bn[192]; + for (const Proj & pr : projs) { + std::snprintf(wn, sizeof(wn), "%s.%s.W", prefix, pr.name); + std::snprintf(bn, sizeof(bn), "%s.%s.b", prefix, pr.name); + const ggml_tensor * W = g.meta(wn); + const ggml_tensor * b = g.meta(bn); + if (!W || !b || ggml_n_dims(W) > 3 || ggml_n_dims(b) > 2 || b->ne[0] != W->ne[0] || b->ne[1] != W->ne[2]) { + std::fprintf(stderr, "vla(%s): %s/%s missing or not a stacked [out,in,n] weight and [out,n] bias\n", + g.arch, wn, bn); + return false; + } + *pr.W = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, W->ne[1], W->ne[0]); + *pr.b = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, W->ne[0]); + ggml_set_name(*pr.W, wn); + ggml_set_name(*pr.b, bn); + } + buf = alloc_weights(ctx, backend); + if (!buf) { + std::fprintf(stderr, "vla(%s): action expert buffer alloc failed\n", g.arch); + return false; + } + + std::vector slab, t; + for (const Proj & pr : projs) { + const int64_t out = (*pr.W)->ne[1], in = (*pr.W)->ne[0]; + std::snprintf(wn, sizeof(wn), "%s.%s.W", prefix, pr.name); + std::snprintf(bn, sizeof(bn), "%s.%s.b", prefix, pr.name); + if (!read_slab(g, wn, embodiment_id, in*out, slab)) + return false; + t.resize(slab.size()); + for (int64_t o=0; onb[1], 0); - return ggml_add(C, cat_linear(C, ae_W3W, ae_W3b, embodiment_id, x_w2), pos); + return ggml_add(C, linear(C, ae_W3W, ae_W3b, x_w2), pos); } ggml_tensor * ActionExpert::decode(ggml_context * C, ggml_tensor * model_out) const { - ggml_tensor * h = ggml_relu(C, cat_linear(C, ad_l1W, ad_l1b, embodiment_id, model_out)); - return cat_linear(C, ad_l2W, ad_l2b, embodiment_id, h); + ggml_tensor * h = ggml_relu(C, linear(C, ad_l1W, ad_l1b, model_out)); + return linear(C, ad_l2W, ad_l2b, h); +} + +ggml_tensor * ActionExpert::denoise(ggml_context * C, const DitHead & dit, const FlowTimes & times, bool interleave, int64_t every2, + ggml_tensor * state, ggml_tensor * future, ggml_tensor * txt, ggml_tensor * img, + ggml_tensor * x0) const { + const int64_t n_layers = dit.cfg.layers, AD = x0->ne[0], AH = x0->ne[1]; + ggml_tensor * state_features = encode_state(C, state); + + std::vector enc(n_layers, nullptr), Kc(n_layers, nullptr), Vc(n_layers, nullptr); + for (int64_t i=0; ine[0], AH); + ggml_tensor * sa = future ? ggml_concat(C, state_features, future, 1) : state_features; + ggml_tensor * hh = ggml_concat(C, sa, af, 1); + + for (int64_t i=0; inb[1], (size_t)(pred->ne[1]-AH)*pred->nb[1])); + actions = ggml_add(C, actions, ggml_scale(C, vel, dt)); + } + return actions; +} + +bool resolve_embodiment(const char * arch, const std::string & mapping, const char * default_tag, + int64_t max_id, int64_t & id) { + auto lookup = [&](const char * key) -> long { + const std::string k = std::string("\"")+key+"\""; + size_t p = mapping.find(k); + if (p == std::string::npos) + return -1; + p = mapping.find(':', p+k.size()); + if (p == std::string::npos) + return -1; + return std::strtol(mapping.c_str()+p+1, nullptr, 10); + }; + + if (default_tag) { + const long d = lookup(default_tag); + if (d >= 0) + id = d; + } + if (const char * e = std::getenv("VLA_GR00T_EMBODIMENT")) { + char * end = nullptr; + const long v = std::strtol(e, &end, 10); + if (end && *end == '\0') { + id = v; + } else { + const long t = lookup(e); + if (t >= 0) + id = t; + else + std::fprintf(stderr, "vla(%s): embodiment tag '%s' not in the GGUF embodiment mapping; using id %lld\n", + arch, e, (long long) id); + } + } + if (id < 0 || id >= max_id) { + std::fprintf(stderr, "vla(%s): embodiment id %lld out of range [0,%lld)\n", arch, (long long) id, (long long) max_id); + return false; + } + return true; } } diff --git a/src/modules/action_expert.h b/src/modules/action_expert.h index 0804a74..86e3088 100644 --- a/src/modules/action_expert.h +++ b/src/modules/action_expert.h @@ -12,15 +12,20 @@ // See the License for the specific language governing permissions and // limitations under the License. -// Every projection is a cat_linear row selected by embodiment_id. +// Every projection is the embodiment_id row of a stacked weight, sliced at load. #pragma once +#include "gguf_reader.h" #include "loader.h" +#include "modules/dit_head.h" #include "ggml.h" +#include "ggml-backend.h" #include +#include +#include namespace vla { @@ -33,7 +38,12 @@ struct ActionExpert { int64_t embodiment_id = 0; - void declare(WeightLoader & L, const char * prefix); + ActionExpert() = default; + ActionExpert(const ActionExpert &) = delete; + ActionExpert & operator=(const ActionExpert &) = delete; + ~ActionExpert(); + + bool declare(WeightLoader & L, gguf_reader & g, ggml_backend_t backend, const char * prefix); ggml_tensor * encode_state(ggml_context * C, ggml_tensor * state) const; @@ -41,6 +51,17 @@ struct ActionExpert { int64_t embed_dim, int64_t horizon) const; ggml_tensor * decode(ggml_context * C, ggml_tensor * model_out) const; + + ggml_tensor * denoise(ggml_context * C, const DitHead & dit, const FlowTimes & times, bool interleave, int64_t every2, + ggml_tensor * state, ggml_tensor * future, ggml_tensor * txt, ggml_tensor * img, + ggml_tensor * x0) const; + +private: + ggml_context * ctx = nullptr; + ggml_backend_buffer_t buf = nullptr; }; +bool resolve_embodiment(const char * arch, const std::string & mapping, const char * default_tag, + int64_t max_id, int64_t & id); + } diff --git a/src/modules/dit_head.cpp b/src/modules/dit_head.cpp index f081288..addee52 100644 --- a/src/modules/dit_head.cpp +++ b/src/modules/dit_head.cpp @@ -14,10 +14,14 @@ #include "modules/dit_head.h" +#include "backend.h" #include "layers/attn.h" +#include "layers/embed.h" #include "layers/ffn.h" #include "layers/linear.h" -#include "layers/norm.h" + +#include "ggml-alloc.h" +#include "ggml-backend.h" #include #include @@ -100,7 +104,7 @@ void DitHead::kv(ggml_context * C, const DitLayerW & w, ggml_tensor * src, *V_out = to_heads_v(C, linear(C, w.Wv, w.bv, src), hd, heads, Tkv); } -ggml_tensor * DitHead::block(ggml_context * C, const DitLayerW & w, ggml_tensor * h, ggml_tensor * temb, +ggml_tensor * DitHead::block(ggml_context * C, const DitLayerW & w, ggml_tensor * h, ggml_tensor * mod, ggml_tensor * enc, ggml_tensor * K_pre, ggml_tensor * V_pre) const { const int64_t hd = cfg.head_dim; const int64_t heads = cfg.heads; @@ -108,7 +112,10 @@ ggml_tensor * DitHead::block(ggml_context * C, const DitLayerW & w, ggml_tensor const int64_t Tk = h->ne[1]; const float scale = 1.0f/std::sqrt((float)hd); - ggml_tensor * n = adaln(C, h, temb, w.adaln_w, w.adaln_b, dim, cfg.ln_eps); + ggml_tensor * sc = ggml_view_1d(C, mod, dim, 0); + ggml_tensor * sh = ggml_view_1d(C, mod, dim, (size_t)dim*sizeof(float)); + ggml_tensor * xn = ggml_norm(C, h, cfg.ln_eps); + ggml_tensor * n = ggml_add(C, ggml_add(C, xn, ggml_mul(C, xn, sc)), sh); ggml_tensor *Q, *K, *V; if (!enc && w.Wqkv) { ggml_tensor * qkv = linear(C, w.Wqkv, w.bqkv, n); @@ -136,14 +143,106 @@ ggml_tensor * DitHead::time_emb(ggml_context * C, ggml_tensor * tproj) const { return linear(C, te_l2W, te_l2b, ggml_silu(C, linear(C, te_l1W, te_l1b, tproj))); } -ggml_tensor * DitHead::proj_out(ggml_context * C, ggml_tensor * h, ggml_tensor * temb) const { - ggml_tensor * po = linear(C, po1W, po1b, ggml_silu(C, temb)); - ggml_tensor * sh = ggml_view_1d(C, po, cfg.hidden, 0); - ggml_tensor * sc = ggml_view_1d(C, po, cfg.hidden, (size_t)cfg.hidden*sizeof(float)); +ggml_tensor * DitHead::proj_out(ggml_context * C, ggml_tensor * h, ggml_tensor * mod) const { + ggml_tensor * sh = ggml_view_1d(C, mod, cfg.hidden, 0); + ggml_tensor * sc = ggml_view_1d(C, mod, cfg.hidden, (size_t)cfg.hidden*sizeof(float)); ggml_tensor * hn = ggml_norm(C, h, cfg.norm_out_eps); ggml_tensor * h_mod = ggml_add(C, ggml_add(C, hn, ggml_mul(C, hn, sc)), sh); return linear(C, po2W, po2b, h_mod); } +FlowTimes::~FlowTimes() { + if (buf) + ggml_backend_buffer_free(buf); + if (ctx) + ggml_free(ctx); +} + +bool FlowTimes::build(const char * arch, ggml_backend_t backend, const DitHead & dit, + int64_t steps, int64_t buckets, int64_t embed_dim, int64_t horizon) { + const int64_t dim = dit.cfg.hidden, layers = dit.cfg.layers; + for (const DitLayerW & w : dit.blk) + if (w.adaln_w->ne[1] != 2*dim) { + std::fprintf(stderr, "vla(%s): adaln weight has %lld rows, expected %lld\n", + arch, (long long) w.adaln_w->ne[1], (long long) (2*dim)); + return false; + } + if (dit.po1W->ne[1] != 2*dim) { + std::fprintf(stderr, "vla(%s): proj_out1 weight has %lld rows, expected %lld\n", + arch, (long long) dit.po1W->ne[1], (long long) (2*dim)); + return false; + } + + per_step = layers+1; + ggml_init_params rp = { (size_t) (steps+1)*ggml_tensor_overhead(), nullptr, true }; + ctx = ggml_init(rp); + if (!ctx) + return false; + tau.assign((size_t) steps, nullptr); + for (int64_t s=0; s t_tproj((size_t) steps), outs; + for (int64_t s=0; s tau_h, tproj_h; + for (int64_t s=0; s host((size_t) ggml_nelements(mods)); + for (size_t k=0; kne[0], (size_t) (s*per_step+i)*mods->nb[1]); +} + } diff --git a/src/modules/dit_head.h b/src/modules/dit_head.h index 6c00b2c..9a003e2 100644 --- a/src/modules/dit_head.h +++ b/src/modules/dit_head.h @@ -21,6 +21,7 @@ #include "loader.h" #include "ggml.h" +#include "ggml-backend.h" #include #include @@ -58,13 +59,33 @@ struct DitHead { void kv(ggml_context * C, const DitLayerW & w, ggml_tensor * src, ggml_tensor ** K_out, ggml_tensor ** V_out) const; - ggml_tensor * block(ggml_context * C, const DitLayerW & w, ggml_tensor * h, ggml_tensor * temb, + ggml_tensor * block(ggml_context * C, const DitLayerW & w, ggml_tensor * h, ggml_tensor * mod, ggml_tensor * enc, ggml_tensor * K_pre = nullptr, ggml_tensor * V_pre = nullptr) const; ggml_tensor * time_emb(ggml_context * C, ggml_tensor * tproj) const; - // (shift, scale) adaLN, opposite to layers/norm.h adaln. - ggml_tensor * proj_out(ggml_context * C, ggml_tensor * h, ggml_tensor * temb) const; + // (shift, scale), opposite to the blocks. + ggml_tensor * proj_out(ggml_context * C, ggml_tensor * h, ggml_tensor * mod) const; +}; + +struct FlowTimes { + std::vector tau; + + FlowTimes() = default; + FlowTimes(const FlowTimes &) = delete; + FlowTimes & operator=(const FlowTimes &) = delete; + ~FlowTimes(); + + bool build(const char * arch, ggml_backend_t backend, const DitHead & dit, + int64_t steps, int64_t buckets, int64_t embed_dim, int64_t horizon); + + ggml_tensor * mod(ggml_context * C, int64_t s, int64_t i) const; + +private: + ggml_tensor * mods = nullptr; + int64_t per_step = 0; + ggml_context * ctx = nullptr; + ggml_backend_buffer_t buf = nullptr; }; } diff --git a/src/modules/dual_tower.h b/src/modules/dual_tower.h index ffee524..a21f77c 100644 --- a/src/modules/dual_tower.h +++ b/src/modules/dual_tower.h @@ -12,18 +12,32 @@ // See the License for the specific language governing permissions and // limitations under the License. -// DINOv2 + SigLIP dual vision tower, shared by OpenVLA-OFT and VLA-Adapter. +// DINOv2 + SigLIP dual vision tower and q01/q99 action stats, shared by +// OpenVLA-OFT and VLA-Adapter. // DINOv2 passes prefix=true (CLS + 4 register tokens, dropped after the blocks) // and uses LayerScale; SigLIP passes prefix=false. #pragma once -#include "ggml.h" +#include "backend.h" +#include "gguf_reader.h" +#include "layers/attn.h" +#include "layers/ffn.h" +#include "layers/linear.h" +#include "layers/norm.h" #include "loader.h" #include "model.h" +#include "modules/preprocess.h" +#include "scratch_ctx.h" + +#include "ggml.h" +#include "ggml-backend.h" #include #include +#include +#include +#include #include namespace vla { @@ -33,12 +47,26 @@ struct ViTLayerW { ggml_tensor *n1w,*n1b,*n2w,*n2b,*ls1,*ls2,*Wqkv,*bqkv,*Wproj, // DINOv2 carries CLS + 4 register tokens and LayerScale; SigLIP carries // neither, so its ls1/ls2 stay null and vit_block is told to skip them. struct DualTower { + int64_t d_hidden=1024,d_layers=23,d_heads=16,d_head_dim=64; + int64_t s_hidden=1152,s_layers=26,s_heads=16,s_head_dim=72; + int64_t image_size=224,patch_size=14,n_patches=256; + float ln_eps=1e-6f; + ggml_tensor *d_patch_w=nullptr,*d_patch_b=nullptr,*d_cls=nullptr,*d_reg=nullptr,*d_pos=nullptr; ggml_tensor *s_patch_w=nullptr,*s_patch_b=nullptr,*s_pos=nullptr; std::vector dvit, svit; ggml_tensor *pj_fc1w=nullptr,*pj_fc1b=nullptr,*pj_fc2w=nullptr,*pj_fc2b=nullptr,*pj_fc3w=nullptr,*pj_fc3b=nullptr; - void declare(WeightLoader & L, int64_t d_layers, int64_t s_layers) { + void read_config(const gguf_reader & g, const std::string & arch) { + auto U=[&](const char*k,int64_t&d){ const std::string key=arch+".vit."+k; if(g.has(key.c_str())) d=(int64_t)g.u32(key.c_str()); }; + U("dino.hidden",d_hidden); U("dino.layers",d_layers); U("dino.heads",d_heads); U("dino.head_dim",d_head_dim); + U("sig.hidden",s_hidden); U("sig.layers",s_layers); U("sig.heads",s_heads); U("sig.head_dim",s_head_dim); + U("image_size",image_size); U("patch_size",patch_size); U("n_patches",n_patches); + const std::string eps=arch+".vit.ln_eps"; + if(g.has(eps.c_str())) ln_eps=g.f32(eps.c_str()); + } + + void declare(WeightLoader & L) { auto blocks = [&](std::vector & v, const char * pre, int64_t n, bool layer_scale) { v.resize(n); for (int64_t i=0; i px_d, px_s; ggml_tensor*proj=nullptr; }; + ggml_tensor* encode(ggml_backend_t backend, graph_cache & cache, const Inputs & in, const char * tag) const; +}; inline ggml_tensor* vit_block(ggml_context*C, const ViTLayerW&w, ggml_tensor*x, int64_t N, int64_t hidden, int64_t heads, int64_t hd, float eps, bool ls){ - const float sc=1.0f/std::sqrt((float)hd); - ggml_tensor*xn=LN(C,x,w.n1w,w.n1b,eps); - ggml_tensor*qkv=ggml_add(C,ggml_mul_mat(C,w.Wqkv,xn),w.bqkv); - ggml_tensor*q=ggml_cont(C,ggml_view_2d(C,qkv,hidden,N,qkv->nb[1],0*hidden*sizeof(float))); - ggml_tensor*k=ggml_cont(C,ggml_view_2d(C,qkv,hidden,N,qkv->nb[1],1*hidden*sizeof(float))); - ggml_tensor*v=ggml_cont(C,ggml_view_2d(C,qkv,hidden,N,qkv->nb[1],2*hidden*sizeof(float))); - ggml_tensor*Q=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,q,hd,heads,N),0,2,1,3)); - ggml_tensor*K=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,k,hd,heads,N),0,2,1,3)); - ggml_tensor*V=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,v,hd,heads,N),1,2,0,3)); - ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_mul_mat_set_prec(kq,GGML_PREC_F32); - ggml_tensor*aw=ggml_soft_max_ext(C,kq,nullptr,sc,0.0f); - ggml_tensor*kqv=ggml_mul_mat(C,V,aw); - ggml_tensor*att=ggml_reshape_2d(C,ggml_cont(C,ggml_permute(C,kqv,0,2,1,3)),hidden,N); - ggml_tensor*ao=ggml_add(C,ggml_mul_mat(C,w.Wproj,att),w.bproj); + ggml_tensor*qkv=linear(C,w.Wqkv,w.bqkv,layer_norm(C,x,w.n1w,w.n1b,eps)); + ggml_tensor*Q=ggml_cont(C,ggml_permute(C,head_view(C,qkv,hd,heads,N,hidden,3,0),0,2,1,3)); + ggml_tensor*K=ggml_cont(C,ggml_permute(C,head_view(C,qkv,hd,heads,N,hidden,3,1),0,2,1,3)); + ggml_tensor*V=ggml_cont(C,ggml_permute(C,head_view(C,qkv,hd,heads,N,hidden,3,2),1,2,0,3)); + ggml_tensor*ao=linear(C,w.Wproj,w.bproj,attention(C,Q,K,V,nullptr,1.0f/std::sqrt((float)hd),hidden,N)); x=ggml_add(C,x,ls?ggml_mul(C,ao,w.ls1):ao); - ggml_tensor*xn2=LN(C,x,w.n2w,w.n2b,eps); - ggml_tensor*h=ggml_add(C,ggml_mul_mat(C,w.Wfc1,xn2),w.bfc1); h=ggml_gelu_erf(C,h); - h=ggml_add(C,ggml_mul_mat(C,w.Wfc2,h),w.bfc2); + ggml_tensor*h=ffn_gelu_erf(C,w.Wfc1,w.bfc1,w.Wfc2,w.bfc2,layer_norm(C,x,w.n2w,w.n2b,eps)); return ggml_add(C,x,ls?ggml_mul(C,h,w.ls2):h); } inline ggml_tensor* tower(ggml_context*C, ggml_tensor*pix, ggml_tensor*pw, ggml_tensor*pb, ggml_tensor*pos, ggml_tensor*cls, ggml_tensor*reg, const std::vector&blk, - int64_t hidden, int64_t heads, int64_t hd, int64_t inter, int64_t patch, float eps, bool prefix){ - (void)inter; + int64_t hidden, int64_t heads, int64_t hd, int64_t patch, float eps, bool prefix){ ggml_tensor*conv=ggml_conv_2d(C,pw,pix,patch,patch,0,0,1,1); // Patch count from the conv, not a constant: both callers run 224/14 today, // and a different input size would otherwise reshape into the wrong grid. @@ -126,14 +141,120 @@ inline ggml_tensor* tower(ggml_context*C, ggml_tensor*pix, ggml_tensor*pw, ggml_ return x; } -// HWC to CHW planar with per-channel mean/std: ImageNet for DINOv2, 0.5 for SigLIP. -inline void normalize_tower(const ImageView& v, int64_t S, const float mean[3], const float std_[3], std::vector& out){ - out.assign((size_t)3*S*S,0.0f); - for(int64_t h=0;h & cache, const Inputs & in, const char * tag) const { + const int64_t S=image_size, n_views=in.n_images; + // towers read S*S*3 per view; reject any view that is not exactly SxS. + for (int64_t v=0; v 0.484375). The reference + // preprocesses in bf16, so these are the values it actually sees. + static const float DMEAN[3]={0.484375f,0.455078125f,0.40625f}, DSTD[3]={0.228515625f,0.2236328125f,0.224609375f}; + + const size_t max_nodes=(size_t)64*(d_layers+s_layers+1)*n_views+1024; + const bool built=cache.ensure(backend,n_views,ggml_tensor_overhead()*max_nodes+ggml_graph_overhead_custom(max_nodes,false), + [&](ggml_context*C, VisIO&io)->ggml_cgraph*{ + io.px_d.resize(n_views); io.px_s.resize(n_views); + std::vector cmb(n_views); + for(int v=0; v dbuf, sbuf; + for(int v=0;v q01, q99; + std::vector mask; + std::string suite; + + bool parse(const std::string & js, const char * env_key, const char * tag, int64_t want) { + auto find_key = [&](size_t from, const std::string & key) -> size_t { + const std::string pat = "\"" + key + "\""; + return js.find(pat, from); + }; + const char * env = std::getenv(env_key); + size_t suite_pos; + if (env) { + suite = env; + suite_pos = find_key(0, suite); + } + else { + size_t b = js.find('{'); size_t q = js.find('"', b); + size_t qe = js.find('"', q+1); + suite = js.substr(q+1, qe-q-1); suite_pos = q; + } + if (suite_pos == std::string::npos) { + std::fprintf(stderr, "vla(%s): suite '%s' not in stats\n", tag, suite.c_str()); + return false; + } + size_t act = find_key(suite_pos, "action"); + if (act == std::string::npos) + return false; + auto read_arr = [&](const std::string & key, std::vector & out) -> bool { + size_t k = find_key(act, key); if (k == std::string::npos) return false; + size_t lb = js.find('[', k); size_t rb = js.find(']', lb); + if (lb == std::string::npos || rb == std::string::npos) + return false; + out.clear(); size_t p = lb+1; + while (p < rb) { + while (p < rb && (js[p] == ',' || js[p] == ' ' || js[p] == '\n' || js[p] == '\t' || js[p] == '\r')) + ++p; + if (p >= rb) + break; + bool t = (js.compare(p, 4, "true") == 0), f = (js.compare(p, 5, "false") == 0); + if (t || f) { + out.push_back(t ? 1.0f : 0.0f); + p += t ? 4 : 5; + } + else { + out.push_back(std::strtof(js.c_str()+p, nullptr)); + while (p < rb && js[p] != ',') + ++p; + } + } + return true; + }; + std::vector mk; + if (!read_arr("q01", q01) || !read_arr("q99", q99)) + return false; + if (!read_arr("mask", mk)) + mk.assign(want, 1.0f); + mask.assign(mk.size(), 1); for (size_t i=0; i unnorm(const std::vector & na, int64_t chunk, int64_t dim, int64_t W) const { + std::vector out((size_t)chunk*W,0.0f); + for(int64_t c=0;c #include +#include +#include #include namespace vla { @@ -62,4 +71,215 @@ struct GemmaStack { } }; +inline ggml_tensor * gemma_attn( + ggml_context * ctx, const GemmaLayerW & w, + ggml_tensor * x_norm, ggml_tensor * positions, + const Config & cfg, int64_t seq, + ggml_tensor * cached_K, ggml_tensor * cached_V, ggml_tensor * mask, + ggml_tensor ** k_out, ggml_tensor ** v_out, ggml_type at, bool flash) { + const int64_t hd = cfg.head_dim; + const int64_t nq = cfg.n_q_heads; + const int64_t nkv = cfg.n_kv_heads; + const int64_t qf = nq * hd; + + // Q/K/V land in F32: RoPE, the KV cache the suffix passes re-read, and the + // score/softmax core all stay full precision. + ggml_tensor * q = as_type(ctx, mm_act(ctx, w.Wq, x_norm, at), GGML_TYPE_F32); + ggml_tensor * k = as_type(ctx, mm_act(ctx, w.Wk, x_norm, at), GGML_TYPE_F32); + ggml_tensor * v = as_type(ctx, mm_act(ctx, w.Wv, x_norm, at), GGML_TYPE_F32); + + ggml_tensor * q_h = ggml_reshape_3d(ctx, q, hd, nq, seq); + ggml_tensor * k_h = ggml_reshape_3d(ctx, k, hd, nkv, seq); + ggml_tensor * v_h = ggml_reshape_3d(ctx, v, hd, nkv, seq); + + auto rope_call = [&](ggml_tensor * t) { + return ggml_rope_ext(ctx, t, positions, nullptr, + (int) hd, GGML_ROPE_TYPE_NEOX, 0, + cfg.rope_freq_base, 1.f, 0.f, 1.f, 32.f, 1.f); + }; + ggml_tensor * q_rope = rope_call(q_h); + ggml_tensor * k_rope = rope_call(k_h); + + if (k_out) + *k_out = k_rope; + if (v_out) + *v_out = v_h; + + ggml_tensor * K_full = k_rope; + ggml_tensor * V_full = v_h; + if (cached_K && cached_V) { + K_full = ggml_concat(ctx, cached_K, k_rope, 2); + V_full = ggml_concat(ctx, cached_V, v_h, 2); + } + + const float scale = 1.f/std::sqrt((float) hd); + ggml_tensor * Q = ggml_cont(ctx, ggml_permute(ctx, q_rope, 0, 2, 1, 3)); + ggml_tensor * K = ggml_cont(ctx, ggml_permute(ctx, K_full, 0, 2, 1, 3)); + ggml_tensor * att_pre; + if (flash) { + ggml_tensor * V = ggml_cont(ctx, ggml_permute(ctx, V_full, 0, 2, 1, 3)); + // ggml_flash_attn_ext asserts an F16 mask. The mask holds only 0 and + // -inf, both exactly representable in F16, so the cast is lossless. + ggml_tensor * mask_f16 = mask ? ggml_cast(ctx, mask, GGML_TYPE_F16) : nullptr; + ggml_tensor * fa = ggml_flash_attn_ext(ctx, Q, fa_kv(ctx, K), fa_kv(ctx, V), mask_f16, scale, 0.0f, 0.0f); + ggml_prec_set_acc(fa, GGML_PREC_F32); + att_pre = ggml_reshape_2d(ctx, fa, qf, seq); + } else { + ggml_tensor * V = ggml_cont(ctx, ggml_permute(ctx, V_full, 1, 2, 0, 3)); + ggml_tensor * kq = ggml_mul_mat(ctx, K, Q); + ggml_prec_set_acc(kq, GGML_PREC_F32); + ggml_tensor * attn = ggml_soft_max_ext(ctx, kq, mask, scale, 0.f); + ggml_tensor * kqv = ggml_mul_mat(ctx, V, attn); + att_pre = ggml_reshape_2d(ctx, + ggml_cont(ctx, ggml_permute(ctx, kqv, 0, 2, 1, 3)), qf, seq); + } + return mm_act(ctx, w.Wo, as_type(ctx, att_pre, at), at); +} + +inline ggml_tensor * gemma_mlp(ggml_context * ctx, const GemmaLayerW & w, ggml_tensor * x_norm, ggml_type at) { + ggml_tensor * gate = mm_act(ctx, w.Wgate, x_norm, at); + ggml_tensor * up = mm_act(ctx, w.Wup, x_norm, at); + return mm_act(ctx, w.Wdown, geglu(ctx, gate, up), at); +} + +inline ggml_tensor * gemma_layer( + ggml_context * ctx, const GemmaLayerW & w, + ggml_tensor * x_in, ggml_tensor * positions, + const Config & cfg, int64_t seq, + ggml_tensor * cached_K, ggml_tensor * cached_V, ggml_tensor * mask, + ggml_tensor ** k_out, ggml_tensor ** v_out, + ggml_type at = GGML_TYPE_F32, bool flash = false) { + ggml_tensor * h1 = ggml_add(ctx, x_in, + gemma_attn(ctx, w, rms_norm(ctx, x_in, w.ln_in, cfg.rms_eps), positions, cfg, seq, + cached_K, cached_V, mask, k_out, v_out, at, flash)); + return ggml_add(ctx, h1, gemma_mlp(ctx, w, rms_norm(ctx, h1, w.ln_post, cfg.rms_eps), at)); +} + +inline std::string pi_key(const gguf_reader & g, const char * s) { + return std::string(g.arch) + "." + s; +} + +// PaliGemma's SigLIP-So400m/14 tower and projector, bundled in the ckpt GGUF. +struct PaliVision { + int64_t layers = 27, image_size = 224, patch_size = 14, n_tokens = 256; + SigLipTower vit; + ggml_tensor * proj_w = nullptr, * proj_b = nullptr; + + bool load(const gguf_reader & g, int64_t n_img) { + EncCfg & c = vit.enc.cfg; + c.hidden = 1152; + c.heads = 16; + auto u = [&](const char * s, int64_t & d) { if (g.has(pi_key(g, s).c_str())) d = (int64_t) g.u32(pi_key(g, s).c_str()); }; + u("vit_hidden", c.hidden); u("vit_layers", layers); + u("vit_heads", c.heads); u("image_size", image_size); + u("patch_size", patch_size); u("n_img_tokens", n_tokens); + if (g.has(pi_key(g, "vit_ln_eps").c_str())) + c.ln_eps = g.f32(pi_key(g, "vit_ln_eps").c_str()); + if (patch_size <= 0 || c.heads <= 0 || c.hidden % c.heads || image_size % patch_size) { + std::fprintf(stderr, "vla(%s): bad vit geometry (image %lld patch %lld hidden %lld heads %lld)\n", + g.arch, (long long) image_size, (long long) patch_size, + (long long) c.hidden, (long long) c.heads); + return false; + } + const int64_t grid = image_size/patch_size; + if (grid * grid != n_tokens || n_tokens != n_img) { + std::fprintf(stderr, "vla(%s): vit geometry mismatch (grid^2=%lld n_img_tokens=%lld cfg.n_img=%lld)\n", + g.arch, (long long) (grid * grid), (long long) n_tokens, (long long) n_img); + return false; + } + c.head_dim = c.hidden/c.heads; + return true; + } + + void declare(WeightLoader & L) { + vit.declare(L, "vit", layers); + proj_w = L.gemm ("mm.proj.weight"); + proj_b = L.opt_f32("mm.proj.bias"); + } +}; + +inline bool load_pi_config(gguf_reader & g, const std::string & path, int64_t n_state, Config & cfg) { + if (path.size() < 5 || path.compare(path.size()-5, 5, ".gguf") != 0) { + std::fprintf(stderr, "vla(%s): ckpt must be a GGUF produced by scripts/convert_%s_to_gguf.py (got '%s')\n", + g.arch, g.arch, path.c_str()); + return false; + } + if (!g.open(path)) + return false; + + const std::string ak = pi_key(g, "architecture"); + if (g.str(ak.c_str()) != g.arch) { + std::fprintf(stderr, "vla(%s): '%s' is not a %s GGUF (%s missing/wrong)\n", g.arch, path.c_str(), g.arch, ak.c_str()); + return false; + } + for (const char * s : {"hidden", "intermediate", "n_q_heads", "n_kv_heads", "head_dim", "n_layers", + "expert_h", "expert_inter", "chunk_size", "num_steps", "max_state_dim", + "max_action_dim", "real_state_dim", "real_action_dim", "tokenizer_max_length", + "min_period", "max_period"}) { + const std::string key = pi_key(g, s); + if (!g.has(key.c_str())) { + std::fprintf(stderr, "vla(%s): gguf missing key %s\n", g.arch, key.c_str()); + return false; + } + } + cfg = Config{}; + cfg.hidden = g.u32(pi_key(g, "hidden").c_str()); + cfg.intermediate = g.u32(pi_key(g, "intermediate").c_str()); + cfg.n_q_heads = g.u32(pi_key(g, "n_q_heads").c_str()); + cfg.n_kv_heads = g.u32(pi_key(g, "n_kv_heads").c_str()); + cfg.head_dim = g.u32(pi_key(g, "head_dim").c_str()); + cfg.n_layers = g.u32(pi_key(g, "n_layers").c_str()); + cfg.expert_h = g.u32(pi_key(g, "expert_h").c_str()); + cfg.expert_inter = g.u32(pi_key(g, "expert_inter").c_str()); + cfg.n_suffix = g.u32(pi_key(g, "chunk_size").c_str()); + cfg.num_steps = g.u32(pi_key(g, "num_steps").c_str()); + cfg.max_state_dim = g.u32(pi_key(g, "max_state_dim").c_str()); + cfg.max_action_dim = g.u32(pi_key(g, "max_action_dim").c_str()); + cfg.real_state_dim = g.u32(pi_key(g, "real_state_dim").c_str()); + cfg.real_action_dim = g.u32(pi_key(g, "real_action_dim").c_str()); + cfg.n_lang = g.u32(pi_key(g, "tokenizer_max_length").c_str()); + cfg.min_period = g.f64(pi_key(g, "min_period").c_str()); + cfg.max_period = g.f64(pi_key(g, "max_period").c_str()); + if (cfg.num_steps < 1 || cfg.num_steps > 1000) { + std::fprintf(stderr, "vla(%s): num_steps %d out of range [1, 1000]\n", g.arch, cfg.num_steps); + return false; + } + + cfg.n_state = n_state; + cfg.n_img = 256; + cfg.q_full_dim = cfg.n_q_heads * cfg.head_dim; + cfg.kv_full_dim = cfg.n_kv_heads*cfg.head_dim; + cfg.self_attn_every_n = 0; + cfg.rms_eps = g.has(pi_key(g, "rms_norm_eps").c_str()) ? g.f32(pi_key(g, "rms_norm_eps").c_str()) : 1e-6f; + cfg.norm_eps = g.has(pi_key(g, "norm_eps").c_str()) ? g.f32(pi_key(g, "norm_eps").c_str()) : 1e-8f; + cfg.rope_mode = GGML_ROPE_TYPE_NEOX; + cfg.rope_n_dims = (int) cfg.head_dim; + cfg.rope_freq_base = g.has(pi_key(g, "rope_theta").c_str()) ? (float) g.f64(pi_key(g, "rope_theta").c_str()) : 10000.f; + cfg.n_prefix = 0; + cfg.n_full = 0; + return true; +} + +// Absent stats are a valid checkpoint: identity, carry on. Stats that are +// present but unreadable are not - falling back to identity there hands back +// un-denormalised actions with nothing in the log. Note stderr, not stdout: +// stdout is the action stream tests/predict_check.cpp diffs. +inline bool read_pi_stat(gguf_reader & g, const char * name, std::vector & dst) { + const ggml_tensor * t = g.meta(name); + if (!t) { + std::fprintf(stderr, "vla(%s): %s missing - identity\n", g.arch, name); + return true; + } + if (t->ne[0] != (int64_t) dst.size()) { + std::fprintf(stderr, "vla(%s): %s is %lld wide, expected %zu\n", + g.arch, name, (long long) t->ne[0], dst.size()); + return false; + } + if (!g.read_raw(name, dst.data(), dst.size()*sizeof(float))) { + std::fprintf(stderr, "vla(%s): %s read failed\n", g.arch, name); + return false; + } + return true; +} + } diff --git a/src/modules/preprocess.h b/src/modules/preprocess.h index cef47e9..016cc97 100644 --- a/src/modules/preprocess.h +++ b/src/modules/preprocess.h @@ -21,7 +21,6 @@ #include #include -#include #include namespace vla { @@ -32,34 +31,20 @@ inline bool view_is_side(const void * data, int w, int h, int64_t side) { return data != nullptr && (int64_t) w == side && (int64_t) h == side; } -// IDEFICS3/SmolVLM pixel-shuffle (space-to-depth), c-innermost channel order. -// src [embed, n_patches] row-major (patch p, channel e) -> dst [embed*s^2, (grid/s)^2]. -inline void pixel_shuffle_hf(const float * src, float * dst, - int64_t embed, int64_t grid, int64_t s) { - const int64_t g2 = grid/s, c4 = embed * s * s; - for (int64_t h2=0; h2 & out) { - if (v.w != (int) side || v.h != (int) side || !v.data) { - std::fprintf(stderr, "vla(%s): image view is %dx%d, expected %lldx%lld\n", - arch, v.w, v.h, (long long) side, (long long) side); + const float mean[3], const float std_[3], std::vector & out) { + if (!view_ok(arch, v, side)) return false; - } out.assign((size_t) 3*side * side, 0.0f); for (int64_t h=0; h & out) { + static const float half[3] = {0.5f, 0.5f, 0.5f}; + return preprocess_image_chw(arch, v, side, half, half, out); +} + inline bool preprocess_image_patches(const char * arch, const ImageView & v, int64_t side, int64_t ps, std::vector & out) { - if (v.w != (int) side || v.h != (int) side || !v.data) { - std::fprintf(stderr, "vla(%s): image view is %dx%d, expected %lldx%lld\n", - arch, v.w, v.h, (long long) side, (long long) side); + if (!view_ok(arch, v, side)) return false; - } const int64_t grid = side/ps, pd = 3*ps*ps, np = grid*grid; out.assign((size_t) pd*np, 0.0f); @@ -101,20 +90,4 @@ inline bool preprocess_image_patches(const char * arch, const ImageView & v, int return true; } -// c-outermost channel order, the inverse layout to pixel_shuffle_hf above. -inline void pixel_shuffle_back(const float * src, int64_t grid, int64_t hidden, int64_t r, float * dst) { - const int64_t g2 = grid/r, c4 = hidden*r*r; - for (int64_t y=0; y & out) { +bool fetch_embeds(gguf_reader & io, const Prompt & p, const float * img_emb, int64_t hidden, std::vector & out) { const int64_t seq = p.len(); out.assign((size_t) seq*hidden, 0.0f); if (!io.fetch_rows_f32("token_embd.weight", p.ids, out.data(), hidden)) @@ -70,8 +69,6 @@ bool fetch_embeds(const char * arch, gguf_reader & io, const Prompt & p, for (size_t k=0; k & out); +bool fetch_embeds(gguf_reader & io, const Prompt & p, const float * img_emb, int64_t hidden, std::vector & out); void init_noise(const Inputs & in, size_t n, std::vector & out); diff --git a/src/modules/qwen3vl_vit.h b/src/modules/qwen3vl_vit.h index e7a0c5f..a2293b9 100644 --- a/src/modules/qwen3vl_vit.h +++ b/src/modules/qwen3vl_vit.h @@ -17,10 +17,13 @@ #pragma once #include "backend.h" +#include "gguf_reader.h" #include "layers/attn.h" #include "loader.h" #include "layers/rope.h" #include "model.h" +#include "modules/preprocess.h" +#include "scratch_ctx.h" #include "ggml.h" #include "options.h" @@ -29,7 +32,7 @@ #include #include #include -#include +#include #include namespace vla { @@ -39,8 +42,14 @@ constexpr float QWEN3VL_STD [3] = {0.5f, 0.5f, 0.5f}; struct VitLayerW { ggml_tensor *ln1w,*ln1b,*ln2w,*ln2b,*Wqkv,*bqkv,*Wo,*bo,*Wfc1,*bfc1,*Wfc2,*bfc2; }; struct MergerW { ggml_tensor *nw,*nb,*fc1w,*fc1b,*fc2w,*fc2b; }; +struct VitIO { ggml_tensor *t_patches=nullptr,*t_pos=nullptr,*t_cos=nullptr,*t_sin=nullptr,*embeds=nullptr,*ds[3]={}; }; struct Qwen3VLTower { + int64_t hidden = 1024, layers = 24, heads = 16, patch = 16, temporal = 2, merge = 2; + int64_t num_pos = 2304, patch_flat = 1536, side = 256; + int64_t deepstack_idx[3] = {5, 11, 17}; + float ln_eps = 1e-6f, rope_theta = 10000.0f, conn_eps = 1e-6f; + std::vector blk; MergerW deepstack[3]; MergerW merger; @@ -48,7 +57,24 @@ struct Qwen3VLTower { ggml_tensor * patch_b = nullptr; ggml_tensor * pos = nullptr; - void declare(WeightLoader & L, const char * prefix, int64_t layers) { + std::vector row, col; + std::vector rope_cos, rope_sin, pos_interp; + + int64_t grid() const { + return side/patch; + } + int64_t n_tokens() const { + return (grid()/merge)*(grid()/merge); + } + + bool load_config(const char * arch, const gguf_reader & g, const char * ns); + + bool build_caches(const char * arch, gguf_reader & io); + + bool encode(const char * arch, ggml_backend_t backend, graph_cache & cache, const ImageView * images, + int64_t n_views, const float * patches_in, std::vector & emb, std::vector (&ds)[3]) const; + + void declare(WeightLoader & L, const char * prefix) { patch_w = L.gemm("%s.patch_embd.weight", prefix); patch_b = L.f32 ("%s.patch_embd.bias", prefix); pos = L.f32 ("%s.pos_embd", prefix); @@ -124,7 +150,7 @@ inline ggml_tensor * build_merger(ggml_context * C, const MergerW & w, ggml_tens m = ggml_add(C, ggml_mul(C, ggml_norm(C, mr, ln_eps), w.nw), w.nb); } ggml_tensor * z1 = ggml_add(C, ggml_mul_mat(C, w.fc1w, m), w.fc1b); - return ggml_add(C, ggml_mul_mat(C, w.fc2w, vla::gelu(C, z1)), w.fc2b); + return ggml_add(C, ggml_mul_mat(C, w.fc2w, ggml_gelu_erf(C, z1)), w.fc2b); } // Patch row/col after the spatial merge. @@ -160,9 +186,11 @@ inline void vit_rope_tables(const std::vector & row, const std::vector< } // Bilinear resample of the pretrained position table onto gh x gw. -inline void interp_pos_embed(const std::vector & table, int64_t num_side, int64_t hidden, +inline bool interp_pos_embed(const std::vector & table, int64_t num_side, int64_t hidden, const std::vector & row, const std::vector & col, int64_t gh, int64_t gw, std::vector & out) { + if (num_side <= 0 || (int64_t) table.size() != num_side*num_side*hidden) + return false; const int64_t S = (int64_t) row.size(); out.assign((size_t) S * hidden, 0.0f); auto src_coord = [&](int64_t k, int64_t g) -> double { return (g <= 1) ? 0.0 : (double) k * (double)(num_side-1)/(double)(g-1); }; @@ -180,17 +208,15 @@ inline void interp_pos_embed(const std::vector & table, int64_t num_side, for (int64_t c=0; c & row, const std::vector & col, std::vector & out) { - if (v.w != (int) side || v.h != (int) side || !v.data) { - std::fprintf(stderr, "vla(%s): image view is %dx%d, expected %lldx%lld\n", - arch, v.w, v.h, (long long) side, (long long) side); + if (!view_ok(arch, v, side)) return false; - } const int64_t S = (int64_t) row.size(), pf = 3*tps * ps * ps; out.assign((size_t) pf * S, 0.0f); auto px = [&](int64_t r, int64_t c, int64_t ch) -> float { @@ -209,4 +235,160 @@ inline bool preprocess_image_patches(const char * arch, const ImageView & v, int return true; } +inline bool mrope_positions(const char * arch, const std::vector & ids, int32_t image_token, int64_t grid, + std::vector & pp) { + const int64_t seq = (int64_t) ids.size(), g2 = grid*grid; + pp.assign((size_t) 4*seq, 0); + int64_t st = 0, st_idx = 0; + while (st < seq) { + int64_t img = st; + while (img < seq && ids[img] != image_token) + ++img; + for (int64_t i=st; i