From 209cf8c7efa1b99f75efe609ca016b75094c8e5e Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Mon, 28 Sep 2026 13:38:51 +0700 Subject: [PATCH 01/61] bump llama.cpp to b11223 --- CMakeLists.txt | 42 ++++--- docs/backend/hexagon-windows.md | 4 +- docs/backend/hexagon.md | 22 ++-- docs/backend/ov.md | 10 +- scripts/patch_ggml_cuda_ext_hook.py | 14 +-- scripts/patch_ggml_openvino.py | 182 +++++++++++++++------------- scripts/upstream_split.py | 6 +- src/layers/attn.h | 4 +- src/models/bitvla.cpp | 4 +- src/models/evo1.cpp | 8 +- src/models/octo.cpp | 6 +- src/models/openvla_oft.cpp | 2 +- src/models/pi0.cpp | 8 +- src/models/pi05.cpp | 6 +- src/models/smolvla.cpp | 12 +- src/models/vla_adapter.cpp | 4 +- src/modules/dual_tower.h | 2 +- 17 files changed, 178 insertions(+), 158 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 1066aea..495aa54 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -43,6 +43,30 @@ if(GGML_CUDA) find_package(Python3 COMPONENTS Interpreter REQUIRED) set(_vla_llama_patch PATCH_COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/scripts/patch_ggml_cuda_ext_hook.py ) + + find_package(CUDAToolkit REQUIRED) + if(NOT DEFINED CMAKE_CUDA_ARCHITECTURES) + # SASS per supported GPU, PTX only for the newest arch so future cards JIT. + set(_vla_cuda_archs 80-real 86-real 87-real 89-real 90-real) + set(_vla_cuda_ptx 90-virtual) + if(CUDAToolkit_VERSION VERSION_GREATER_EQUAL 12.8) + list(APPEND _vla_cuda_archs 100-real 120-real) + set(_vla_cuda_ptx 120-virtual) + else() + message(WARNING + "CUDA ${CUDAToolkit_VERSION} < 12.8: omitting Blackwell sm_100/sm_120. " + "Pass -DCMAKE_CUDA_ARCHITECTURES= explicitly to override.") + endif() + if(CUDAToolkit_VERSION VERSION_GREATER_EQUAL 12.9) + list(APPEND _vla_cuda_archs 121-real) + endif() + if(CUDAToolkit_VERSION VERSION_GREATER_EQUAL 13.0) + list(APPEND _vla_cuda_archs 110-real) + endif() + set(CMAKE_CUDA_ARCHITECTURES ${_vla_cuda_archs} ${_vla_cuda_ptx} + CACHE STRING "" FORCE) + endif() + set(GGML_CUDA_FA_QUANTS "f16-f16" CACHE STRING "") elseif(GGML_OPENVINO) find_package(Python3 COMPONENTS Interpreter REQUIRED) set(_vla_llama_patch PATCH_COMMAND ${Python3_EXECUTABLE} @@ -63,7 +87,7 @@ endif() # build dir (-DVLA_LLAMA_TAG=b10326) without editing this file. The patch # anchors in scripts/patch_ggml_cuda_ext_hook.py and scripts/patch_ggml_openvino.py # are checked against the default. -set(_vla_llama_tag_default "b10729") +set(_vla_llama_tag_default "b11223") set(VLA_LLAMA_TAG "${_vla_llama_tag_default}" CACHE STRING "llama.cpp tag to fetch") # A cache entry survives an edit to the line above, so an existing build dir keeps # the tag it was first configured with and quietly builds the wrong llama.cpp. @@ -200,22 +224,6 @@ if(GGML_CUDA) target_compile_definitions(vla_core PUBLIC GGML_USE_CUDA) enable_language(CUDA) - find_package(CUDAToolkit REQUIRED) - if(NOT DEFINED CMAKE_CUDA_ARCHITECTURES) - # SASS per supported GPU, PTX only for the newest arch so future cards JIT. - set(_vla_cuda_archs 80-real 86-real 87-real 89-real 90-real) - set(_vla_cuda_ptx 90-virtual) - if(CUDAToolkit_VERSION VERSION_GREATER_EQUAL 12.8) - list(APPEND _vla_cuda_archs 100-real 120-real) - set(_vla_cuda_ptx 120-virtual) - else() - message(WARNING - "CUDA ${CUDAToolkit_VERSION} < 12.8: omitting Blackwell sm_100/sm_120. " - "Pass -DCMAKE_CUDA_ARCHITECTURES= explicitly to override.") - endif() - set(CMAKE_CUDA_ARCHITECTURES ${_vla_cuda_archs} ${_vla_cuda_ptx} - CACHE STRING "" FORCE) - endif() add_library(bitvla_cuda_kernels STATIC src/kernels/bitvla/bitnet_kernels.cu src/kernels/bitvla/bitvla_lm_cuda.cu diff --git a/docs/backend/hexagon-windows.md b/docs/backend/hexagon-windows.md index 686f2dc..1a6d86d 100644 --- a/docs/backend/hexagon-windows.md +++ b/docs/backend/hexagon-windows.md @@ -1,7 +1,7 @@ # vla.cpp on Snapdragon X (Windows on Arm): Hexagon NPU, Adreno GPU and CPU Measured 2026-09 against llama.cpp build 11201 (`2145525a4`), passed in with -`-LlamaDir`. The `b10729` tag that `CMakeLists.txt` pins was not tested. +`-LlamaDir`. The `b11223` tag that `CMakeLists.txt` pins was not tested. ## Summary @@ -79,7 +79,7 @@ The build script sets up the Visual Studio shell, the compiler flags llama.cpp's Each build goes into `build-wos-`, with every binary and DLL in `build-wos-\bin`. `-NoServer` skips `vla-server`, Octo and their protobuf and ZeroMQ dependencies. -`-LlamaDir` points the build at an existing llama.cpp checkout through `FETCHCONTENT_SOURCE_DIR_LLAMA`. Without it, the `b10729` pin in `CMakeLists.txt` applies, which was not tested here. +`-LlamaDir` points the build at an existing llama.cpp checkout through `FETCHCONTENT_SOURCE_DIR_LLAMA`. Without it, the `b11223` pin in `CMakeLists.txt` applies, which was not tested here. The HTP build also signs `libggml-htp-v*.so` with the certificate and copies the skels and their catalog next to the binaries. At startup the Hexagon backend points `ADSP_LIBRARY_PATH` at the executable's own folder, but only if the variable is unset. If it is already set, for example by a llama.cpp install, the skels it names must come from the same llama.cpp commit. diff --git a/docs/backend/hexagon.md b/docs/backend/hexagon.md index ec1e123..d4767a6 100644 --- a/docs/backend/hexagon.md +++ b/docs/backend/hexagon.md @@ -19,8 +19,7 @@ to take once it ran. ## What "IQ9" and "IQ10" refer to -Two commits, both inside `b10729` (`458681e1`, 2026-09-01), and no Hexagon commit -lands after it as of this writing: +Two commits, both inside the pinned `b11223` (`4da63377`, 2026-09-27): | Commit | Date | What it actually did | |---|---|---| @@ -45,12 +44,12 @@ The generational split visible in the code, rather than in the marketing: supported without dynamic discovery"*. Two is what the fallback knows about; the discovery path is open-ended. -## What the backend gives you at `b10729` +## What the backend gives you at `b11223` - **Two libraries.** `libggml-hexagon.so` on the CPU side, `libggml-htp-vNN.so` - on the NPU side. Skels are built for v68, v69, v73, v75, v79 and v81 and the - right one is picked at runtime from `htpdrv_get_arch`; a failed query falls - back to v73. `GGML_HEXAGON_ARCH` overrides. + on the NPU side. Skels are built for v73, v75, v79 and v81 and the right one + is picked at runtime from `htpdrv_get_arch`; a failed query falls back to v73 + and older parts are capped to it. `GGML_HEXAGON_ARCH` overrides. - **Sessions are devices.** Each Hexagon process domain shows up to ggml as one device and behaves like a GPU for offload and model splitting. `GGML_HEXAGON_DEVICES` takes either a count or an explicit @@ -58,17 +57,18 @@ The generational split visible in the code, rather than in the marketing: - **~3.5 GB per session.** The backend now maps and unmaps execution buffers during graph execution to fit larger models into one session, and layer- or tensor-splitting across sessions is the alternative. -- **Repack buffers.** Q4_0, Q4_1, Q8_0, IQ4_NL and MXFP4 weights are repacked - into non-host buffers; since #26501 non-host is the default and - `GGML_HEXAGON_HOSTBUF=1` is the opt-out (needed to exercise `MUL_MAT` in - `test-backend-ops`). +- **Repack buffers.** Q4_0, Q4_1, Q8_0, IQ4_NL, MXFP4, Q4_K, Q5_K and Q6_K + weights are repacked into non-host buffers; since #26501 non-host is the + default and `GGML_HEXAGON_HOSTBUF=1` is the opt-out (needed to exercise + `MUL_MAT` in `test-backend-ops`). - **VTCM is the real budget.** `supports_op` precomputes kernel params for `MUL_MAT`, `FLASH_ATTN_EXT` and friends and returns false when the tile does not fit VTCM. An op is not rejected by shape rules so much as by whether it fits - which means coverage is a function of your tensor sizes, and has to be measured on the board, not predicted from a table. - **Fusion**, controlled by `GGML_HEXAGON_OPFUSION`: `RMS_NORM+MUL`, - `MUL_MAT+ADD`, N-way `MUL_MAT`, `ALLREDUCE+ADD`. + `MUL_MAT+ADD`, N-way `MUL_MAT` and `MUL_MAT_ID`, `ALLREDUCE+ADD`, + `GATED_DELTA_NET+CPY`. The knobs worth knowing on day one: diff --git a/docs/backend/ov.md b/docs/backend/ov.md index 103df65..2393349 100644 --- a/docs/backend/ov.md +++ b/docs/backend/ov.md @@ -26,6 +26,8 @@ Measured on an **Intel Core Ultra X7 358H** (Panther Lake) with the Arc B390 iGPU and the AI Boost NPU, Ubuntu 24.04, OpenVINO 2026.2.1, llama.cpp `b10729`, on the checkpoints under `vrfai/` on the Hub. Every **fidelity** number was re-measured on that pin; only the **latency** table still dates from `b10331`. +The build now pins llama.cpp `b11223` and `install_ov.sh` installs OpenVINO +2026.4; the numbers below have not been re-measured on that pin yet. ggml's backend translates a ggml compute graph into an OpenVINO model and hands it to the CPU, GPU or NPU plugin, which compiles and fuses it for the device. @@ -139,9 +141,9 @@ carry them across restarts; it produces silently wrong actions here - see one camera view, best of 4-6 iterations after 3 warmups. "CPU backend" is ggml's own CPU backend on the same 16-core host. No `GGML_OPENVINO_CACHE_DIR`. -Latencies were taken at `b10331` and have not been re-timed on `b10729`. Read the -GPU column for VLA-JEPA and GR00T N1.7 as the cost of a wrong answer at that pin; -both are correct now. +Latencies were taken at `b10331` and have not been re-timed on `b10729` or +`b11223`. Read the GPU column for VLA-JEPA and GR00T N1.7 as the cost of a wrong +answer at that pin; both are correct now. | Model | input | CPU backend | OpenVINO CPU | OpenVINO GPU | OpenVINO NPU | |---|---|---:|---:|---:|---:| @@ -268,7 +270,7 @@ the ggml contract, or fills a gap: | Folded weights padded to full rank | a 2-D weight becomes a rank-2 constant, but views index it at ggml rank | | CONCAT input ranks aligned | same rank-2 constants, and concat cannot broadcast rank | | Missing `GELU_ERF` translator | the exact-erf GELU op had no table entry at all, so a graph using it could not run | -| Naive-path graph cache | that path re-compiled the whole model on every graph_compute, and its `graph_key` is a node count plus two names, which two graphs can share | +| Naive-path graph cache | that path re-compiled the whole model on every graph_compute, and its `graph_key` is a node count plus tensor names, which two graphs of different shapes can share | | Interleaved-mrope sectors bounded | the sector cycle ignored `sections`, so the last few took the wrong stream | | Naive-path threshold settable | the 20-node constant is what picks the literal path | diff --git a/scripts/patch_ggml_cuda_ext_hook.py b/scripts/patch_ggml_cuda_ext_hook.py index d02bcb4..cf6aee4 100644 --- a/scripts/patch_ggml_cuda_ext_hook.py +++ b/scripts/patch_ggml_cuda_ext_hook.py @@ -33,10 +33,10 @@ 1. Two exported function pointers, null by default. 2. One call to the first at the top of ggml_cuda_compute_forward. Returning false means "not mine", and ggml runs the op exactly as before. - 3. The RMS_NORM+MUL fusion check GGML_ASSERTs F32 rather than declining, so a - BF16 rms_norm aborts the process before dispatch is ever reached. Those two - asserts become a return, which is what the surrounding checks already do - for every other unsupported type. + 3. The RMS_NORM+MUL and RMS_NORM+SCALE fusion checks GGML_ASSERT F32 rather + than declining, so a BF16 rms_norm aborts the process before dispatch is + ever reached. Each pair of asserts becomes a return, which is what the + surrounding checks already do for every other unsupported type. 4. One call to the second in the ADD/MUL fusion branch of ggml_cuda_try_fuse. Fusion happens in ggml_backend_cuda_graph_compute, upstream of ggml_cuda_compute_forward, so the hook in (2) never sees a fused node -- @@ -121,11 +121,11 @@ def main(): if MARKER in text: return # idempotent: re-configure over an already-patched tree - for old, new in (HOOK_DECL, FUSION_GUARD, FUSED_BINBCAST_GUARD): + for old, new, want in ((*HOOK_DECL, 1), (*FUSION_GUARD, 2), (*FUSED_BINBCAST_GUARD, 1)): n = text.count(old) - if n != 1: + if n != want: raise SystemExit( - f"{path}: anchor found {n} times, expected 1. The pinned llama.cpp " + f"{path}: anchor found {n} times, expected {want}. The pinned llama.cpp " f"probably moved; re-check this anchor against the new tag.\n" f"---\n{old[:400]}\n---" ) diff --git a/scripts/patch_ggml_openvino.py b/scripts/patch_ggml_openvino.py index 0bbffba..4860ffa 100755 --- a/scripts/patch_ggml_openvino.py +++ b/scripts/patch_ggml_openvino.py @@ -83,12 +83,12 @@ 1.4 s per prediction with the cache in place. A hit rebinds the cached decoder to the new graph through the existing update_io(), which is how the dynamic path already handles freshly built tensors. - The key is `naive_key`, not `graph_key`: the latter is n_nodes plus the - first and last node name, which two graphs of the same size can share, and - a compiled model is bound to the shapes it was built for. Reusing one - across a shape change returns another graph's answer with no error, so the - key mixes in every node's op and shape. The map is bounded; see the comment - on the flush. + The key is `naive_key`, not `graph_key`: the latter is n_nodes, the first + and last node name and the input names, which two graphs of the same size + can share, and a compiled model is bound to the shapes it was built for. + Reusing one across a shape change returns another graph's answer with no + error, so the key mixes in every node's op and shape. The map is bounded; + see the comment on the flush. 6. openvino/op_table.cpp - get both GELU flavours right. ggml has two: GGML_UNARY_OP_GELU is the tanh approximation, GGML_UNARY_OP_ @@ -241,11 +241,13 @@ NAIVE_COMPUTE_OLD = """enum ggml_status naive_compute(ggml_cgraph * cgraph, ov::Core & core, const std::string & device, - const ov::AnyMap & config) { + const ov::AnyMap & config, + ov_compiled_model_cache & cache) { if (cgraph->n_nodes == 1 && (cgraph->nodes[0]->op == GGML_OP_NONE || cgraph->nodes[0]->op == GGML_OP_VIEW)) { return GGML_STATUS_SUCCESS; } + std::unique_lock compile_lock(cache.mutex); bool naive = true; auto model_weights = GgmlOvDecoder::create_weight_nodes(cgraph, naive); auto decoder = std::make_shared(cgraph, model_weights); @@ -257,46 +259,59 @@ std::shared_ptr infer_request; auto remote_context = ggml_openvino_get_remote_context(); + ov::AnyMap compile_config = config; if (cgraph->nodes[0]->op == GGML_OP_MUL_MAT) { // TODO ACCURACY hint triggers a bug in GPU plugin/driver on Lunar Lake. Remove once CVS-182166 is resolved - core.set_property(device, ov::hint::execution_mode(ov::hint::ExecutionMode::PERFORMANCE)); + compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::PERFORMANCE; } else { - core.set_property(device, ov::hint::execution_mode(ov::hint::ExecutionMode::ACCURACY)); + compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::ACCURACY; } if (remote_context.has_value()) { infer_request = std::make_shared( - core.compile_model(model, remote_context.value(), config).create_infer_request()); + core.compile_model(model, remote_context.value(), compile_config).create_infer_request()); } else { - infer_request = - std::make_shared(core.compile_model(model, device, config).create_infer_request()); + infer_request = std::make_shared( + core.compile_model(model, device, compile_config).create_infer_request()); } - - auto ov_params = model->get_parameters();""" + std::vector input_names; + std::vector output_names; + for (const auto & param : model->get_parameters()) { + input_names.push_back(param->get_friendly_name()); + } + for (const auto & result : model->get_results()) { + output_names.push_back(result->get_friendly_name()); + } + // Destroy the frontend graph under the compilation lock as well: it can + // still own edges into the shared weight nodes. + model.reset(); + input_model.reset(); + decoder->clear_model_weights(); + model_weights.clear(); + compile_lock.unlock(); +""" NAIVE_COMPUTE_NEW = """enum ggml_status naive_compute(ggml_cgraph * cgraph, ov::Core & core, const std::string & device, const ov::AnyMap & config, - std::shared_ptr r_ctx) { + const std::shared_ptr & r_ctx) { if (cgraph->n_nodes == 1 && (cgraph->nodes[0]->op == GGML_OP_NONE || cgraph->nodes[0]->op == GGML_OP_VIEW)) { return GGML_STATUS_SUCCESS; } - // vla.cpp: reuse the decoder, the converted model and the compiled infer - // request across calls on the same graph, the way the dynamic and static - // paths already do. Conversion plus compile_model dominates a naive call, so - // without this every graph_compute pays it again. + // vla.cpp: reuse the decoder and the compiled infer request across calls on + // the same graph, the way the dynamic and static paths already do. + // Conversion plus compile_model dominates a naive call, so without this every + // graph_compute pays it again. static const bool cache_enabled = !ggml_openvino_getenv_int("GGML_OPENVINO_DISABLE_CACHE"); - const naive_key key(cgraph); std::shared_ptr entry; - bool cache_hit = false; - if (cache_enabled && r_ctx != nullptr) { + if (cache_enabled) { + const naive_key key(cgraph); std::lock_guard lock(r_ctx->ctx_mutex); auto it = r_ctx->naive_cache.find(key); if (it != r_ctx->naive_cache.end()) { entry = it->second; - cache_hit = true; } else { // Each entry holds a compiled model, so this cannot grow forever. // Flush rather than evict: a caller sees a handful of shapes, and an @@ -307,55 +322,68 @@ entry = std::make_shared(); r_ctx->naive_cache[key] = entry; } - } else { - entry = std::make_shared(); } - // One graph at a time: an ov::InferRequest is not re-entrant, and a hit - // rebinds the decoder to this cgraph. - std::lock_guard entry_lock(entry->mutex); - - bool naive = true; std::shared_ptr decoder; - std::shared_ptr model; std::shared_ptr infer_request; + std::vector input_names; + std::vector output_names; - if (cache_hit && entry->infer_request != nullptr) { + if (entry != nullptr && entry->infer_request != nullptr) { decoder = entry->decoder; - model = entry->model; infer_request = entry->infer_request; + input_names = entry->input_names; + output_names = entry->output_names; // Same shapes, new tensors: point the decoder at this call's graph. decoder->update_io(cgraph); } else { + std::unique_lock compile_lock(r_ctx->compiled_cache->mutex); + bool naive = true; auto model_weights = GgmlOvDecoder::create_weight_nodes(cgraph, naive); decoder = std::make_shared(cgraph, model_weights); auto input_model = std::make_shared(decoder); - model = ov::frontend::ggml::FrontEnd::convert(input_model, naive); + auto model = ov::frontend::ggml::FrontEnd::convert(input_model, naive); if (ggml_openvino_getenv_int("GGML_OPENVINO_DUMP_IR")) { ov::serialize(model, "IR_naive.xml"); } auto remote_context = ggml_openvino_get_remote_context(); + ov::AnyMap compile_config = config; if (cgraph->nodes[0]->op == GGML_OP_MUL_MAT) { // TODO ACCURACY hint triggers a bug in GPU plugin/driver on Lunar Lake. Remove once CVS-182166 is resolved - core.set_property(device, ov::hint::execution_mode(ov::hint::ExecutionMode::PERFORMANCE)); + compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::PERFORMANCE; } else { - core.set_property(device, ov::hint::execution_mode(ov::hint::ExecutionMode::ACCURACY)); + compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::ACCURACY; } if (remote_context.has_value()) { infer_request = std::make_shared( - core.compile_model(model, remote_context.value(), config).create_infer_request()); + core.compile_model(model, remote_context.value(), compile_config).create_infer_request()); } else { - infer_request = - std::make_shared(core.compile_model(model, device, config).create_infer_request()); + infer_request = std::make_shared( + core.compile_model(model, device, compile_config).create_infer_request()); + } + for (const auto & param : model->get_parameters()) { + input_names.push_back(param->get_friendly_name()); + } + for (const auto & result : model->get_results()) { + output_names.push_back(result->get_friendly_name()); + } + // Destroy the frontend graph under the compilation lock as well: it can + // still own edges into the shared weight nodes. + model.reset(); + input_model.reset(); + decoder->clear_model_weights(); + model_weights.clear(); + compile_lock.unlock(); + + if (entry != nullptr) { + entry->decoder = decoder; + entry->infer_request = infer_request; + entry->input_names = input_names; + entry->output_names = output_names; } - - entry->decoder = decoder; - entry->model = model; - entry->infer_request = infer_request; } - - auto ov_params = model->get_parameters();""" +""" # file -> [(anchor, replacement), ...]. Every anchor must match exactly once. EDITS = { @@ -383,11 +411,11 @@ (USM_LOOKUP % (("clEnqueueMemcpyINTEL",) * 2), USM_LOOKUP_NEW % (("clEnqueueMemcpyINTEL",) * 2)), ( """ "GGML_OPENVINO_LOG_UNSUPPORTED_OPS", - };""", +""", """ "GGML_OPENVINO_LOG_UNSUPPORTED_OPS", // vla.cpp: f16 (default) or f32 for the GPU plugin's inference precision. "GGML_OPENVINO_GPU_PRECISION", - };""", +""", ), ( """ } else if (cache_dir && strlen(cache_dir) > 0) { @@ -417,27 +445,20 @@ ], "ggml/src/ggml-openvino/openvino/op/add.cpp": [ ( - """ auto input_0 = process_view_input_new(context, 0); - auto input_1 = process_view_input_new(context, 1); - auto res = std::make_shared(input_0, input_1);""", - """ auto input_0 = process_view_input_new(context, 0); - auto input_1 = process_view_input_new(context, 1); - - // vla.cpp: re-hang the outer add on the inner one's non-GEMM operand so the + """ ov::Output res = std::make_shared(input_0, input_1);""", + """ // vla.cpp: re-hang the outer add on the inner one's non-GEMM operand so the // GEMM is left with a single post-op. Addition is associative. + ov::Output res; const int oc = context.get_op_case(); - if (oc == 2 || oc == 3) { - auto inner = input_0.get_node_shared_ptr(); - if (inner->get_input_size() == 2) { - const size_t keep = (oc == 2) ? 0 : 1; - const size_t fold = 1 - keep; - auto folded = std::make_shared(inner->input_value(fold), input_1); - auto res2 = std::make_shared(inner->input_value(keep), folded); - return rename_outputs_with_suffix({res2}, context.get_name()); - } - } - - auto res = std::make_shared(input_0, input_1);""", + auto inner = input_0.get_node_shared_ptr(); + if ((oc == 2 || oc == 3) && inner->get_input_size() == 2) { + const size_t keep = (oc == 2) ? 0 : 1; + const size_t fold = 1 - keep; + auto folded = std::make_shared(inner->input_value(fold), input_1); + res = std::make_shared(inner->input_value(keep), folded); + } else { + res = std::make_shared(input_0, input_1); + }""", ), ], "ggml/src/ggml-openvino/openvino/op/concat.cpp": [ @@ -642,7 +663,7 @@ }""", """ if (GgmlOvDecoder::is_inp_pos(tensor, op)) { // vla.cpp: this free function is the live naming path -- the - // GgmlOvDecoder member of the same intent is unreferenced at b10729. + // GgmlOvDecoder member of the same intent is unreferenced at b11223. // Only collapse ROPE position inputs onto one "inp_pos" parameter when // the graph really has one. See scripts/patch_ggml_openvino.py. return decoder->has_multiple_inp_pos() ? get_tensor_ov_name(cgraph, tensor) : std::string("inp_pos"); @@ -707,16 +728,16 @@ // it that path rebuilt the decoder, re-converted the model and called // compile_model() on every ggml_backend_graph_compute, which dominated runtime. struct naive_runtime_ctx { - std::mutex mutex; std::shared_ptr decoder; - std::shared_ptr model; std::shared_ptr infer_request; + std::vector input_names; + std::vector output_names; }; -// vla.cpp: graph_key is {n_nodes, first name, last name}, which two graphs of the -// same size can share. A compiled model is bound to the shapes it was built for, -// so reusing one across a shape change returns another graph's answer with no -// error. Mix the ops and shapes in as well. +// vla.cpp: graph_key is {n_nodes, first name, last name, input names}, which two +// graphs of the same size can share. A compiled model is bound to the shapes it +// was built for, so reusing one across a shape change returns another graph's +// answer with no error. Mix the ops and shapes in as well. inline uint64_t naive_graph_sig(const ggml_cgraph * cgraph) { uint64_t h = 1469598103934665603ull; auto mix = [&h](uint64_t v) { h = (h ^ v) * 1099511628211ull; }; @@ -768,22 +789,11 @@ naive_cache.clear(); infer_request_cache.clear();""", ), - ( - """enum ggml_status naive_compute(struct ggml_cgraph * cgraph, - ov::Core & core, - const std::string & device, - const ov::AnyMap & config);""", - """enum ggml_status naive_compute(struct ggml_cgraph * cgraph, - ov::Core & core, - const std::string & device, - const ov::AnyMap & config, - std::shared_ptr r_ctx);""", - ), ], "ggml/src/ggml-openvino/utils.cpp": [ ( """ if (!model_is_splitted) { - return naive_compute(cgraph, core, device, config); + return naive_compute(cgraph, core, device, config, *r_ctx->compiled_cache); }""", """ if (!model_is_splitted) { return naive_compute(cgraph, core, device, config, r_ctx); @@ -791,7 +801,7 @@ ), ( """ if (is_naive(cgraph)) { - return naive_compute(cgraph, core, device, config); + return naive_compute(cgraph, core, device, config, *r_ctx->compiled_cache); }""", """ if (is_naive(cgraph)) { return naive_compute(cgraph, core, device, config, r_ctx); diff --git a/scripts/upstream_split.py b/scripts/upstream_split.py index 2b328f7..7d2cbe2 100755 --- a/scripts/upstream_split.py +++ b/scripts/upstream_split.py @@ -62,12 +62,12 @@ def H(rel, key): "A hit rebinds the cached decoder through the existing update_io(), the same way\n" "the dynamic path handles freshly built tensors.\n\n" "The cache is keyed on naive_key rather than graph_key. graph_key is n_nodes plus\n" - "the first and last node name, which two graphs of the same size can share, and a\n" + "tensor names, which two graphs of the same size can share, and a\n" "compiled model is bound to the shapes it was built for, so a collision returns\n" "another graph's answer with no error. naive_key mixes in every node's op, type\n" "and shape. The map is bounded and flushed when full.", [H(D+"utils.h","struct decoder_runtime_ctx"),H(D+"utils.h","graph_key_hash> decoder_cache"), - H(D+"utils.h","decoder_cache.clear()"),H(D+"utils.h","enum ggml_status naive_compute"), + H(D+"utils.h","decoder_cache.clear()"), H(D+"utils.cpp","if (!model_is_splitted)"),H(D+"utils.cpp","if (is_naive(cgraph))"), H(D+"utils.cpp","enum ggml_status naive_compute")]), @@ -137,7 +137,7 @@ def H(rel, key): "of the inner add is the GEMM, since the order is not fixed.\n\n" "Same fusion path as the broadcast-DIV defect already handled in supports_op.", [H(D+"ggml-decoder.cpp","case GGML_OP_ADD: {"), - H(D+"openvino/op/add.cpp","auto input_0 = process_view_input_new(context, 0);")]), + H(D+"openvino/op/add.cpp","ov::Output res = std::make_shared")]), ("openvino-permute-op-case", "openvino: require a ROPE before taking PERMUTE op_case 2", diff --git a/src/layers/attn.h b/src/layers/attn.h index bef6af2..db77e5a 100644 --- a/src/layers/attn.h +++ b/src/layers/attn.h @@ -38,7 +38,7 @@ inline ggml_tensor * to_heads_v(ggml_context * C, ggml_tensor * p, int64_t hd, i inline ggml_tensor * attention(ggml_context * C, ggml_tensor * Q, ggml_tensor * K, ggml_tensor * V, ggml_tensor * mask, float scale, int64_t dim, int64_t T, int64_t nv = 1) { ggml_tensor * kq = ggml_mul_mat(C, K, Q); - ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_prec_set_acc(kq, GGML_PREC_F32); ggml_tensor * aw = ggml_soft_max_ext(C, kq, mask, scale, 0.0f); ggml_tensor * kqv = ggml_mul_mat(C, V, aw); @@ -52,7 +52,7 @@ inline ggml_tensor * flash_attention(ggml_context * C, ggml_tensor * Q, ggml_ten ggml_tensor * vf = V->type == GGML_TYPE_F16 ? V : ggml_cast(C, V, GGML_TYPE_F16); ggml_tensor * o = ggml_flash_attn_ext(C, Q, kf, vf, mask, scale, 0.0f, 0.0f); - ggml_flash_attn_ext_set_prec(o, GGML_PREC_F32); + ggml_prec_set_acc(o, GGML_PREC_F32); return ggml_reshape_2d(C, o, o->ne[0]*o->ne[1], o->ne[2]*o->ne[3]); } diff --git a/src/models/bitvla.cpp b/src/models/bitvla.cpp index 71df94f..f2f63b9 100644 --- a/src/models/bitvla.cpp +++ b/src/models/bitvla.cpp @@ -192,7 +192,7 @@ ggml_tensor * build_vit_layer(ggml_context * C, const VitLayerW & w, ggml_tensor ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, head_dim, heads, seq), 0, 2, 1, 3)); ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, head_dim, heads, seq), 0, 2, 1, 3)); ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_prec_set_acc(kq, GGML_PREC_F32); ggml_tensor * att= ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); ggml_tensor * y = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, att), 0, 2, 1, 3)), hidden, seq); ggml_tensor * o = bit_linear(C, w.Wo, w.bo, y); @@ -220,7 +220,7 @@ ggml_tensor * build_lm_layer(ggml_context * C, const BitvlaModelArch & m, const ggml_tensor * Q = ggml_cont(C, ggml_permute(C, qR, 0, 2, 1, 3)); ggml_tensor * K = ggml_cont(C, ggml_permute(C, kR, 0, 2, 1, 3)); ggml_tensor * V = ggml_cont(C, ggml_permute(C, v3, 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_prec_set_acc(kq, GGML_PREC_F32); // Unmasked on purpose, same as openvla_oft: BitVLA is fine-tuned with // OpenVLA-OFT's recipe, which swaps the causal mask for a bidirectional one // so the action chunk decodes in a single pass. diff --git a/src/models/evo1.cpp b/src/models/evo1.cpp index bdeb9a9..890e36b 100644 --- a/src/models/evo1.cpp +++ b/src/models/evo1.cpp @@ -146,7 +146,7 @@ ggml_tensor * build_qwen2_layer(ggml_context * C, const Evo1ModelArch & m, const ggml_tensor * Q = ggml_cont(C, ggml_permute(C, q_rope, 0, 2, 1, 3)); ggml_tensor * K = ggml_cont(C, ggml_permute(C, k_rope, 0, 2, 1, 3)); ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, vp, hd, n_kv, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_prec_set_acc(kq, GGML_PREC_F32); ggml_tensor * aw = ggml_soft_max_ext(C, kq, mask, scale, 0.0f); ggml_tensor * kqv = ggml_mul_mat(C, V, aw); ggml_tensor * att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, kqv, 0, 2, 1, 3)), hq, seq); @@ -216,7 +216,7 @@ ggml_tensor * evo1_flash_attn(ggml_context * C, ggml_tensor * q, ggml_tensor * k ggml_tensor * o = ggml_flash_attn_ext(C, q, k, v, nullptr, scale, 0.0f, 0.0f); // F32 accumulation keeps the softmax/AV reduction at the precision the // explicit path used, so switching kernels does not move the actions. - ggml_flash_attn_ext_set_prec(o, GGML_PREC_F32); + ggml_prec_set_acc(o, GGML_PREC_F32); return ggml_reshape_2d(C, o, hidden, N); } @@ -240,7 +240,7 @@ ggml_tensor * build_internvit_layer(ggml_context * C, const Evo1ModelArch & m, c att = evo1_flash_attn(C, Q, K, V, scale, H, N); } else { ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, hd, n_heads, N), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_prec_set_acc(kq, GGML_PREC_F32); ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); ggml_tensor * kqv = ggml_mul_mat(C, V, aw); att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, kqv, 0, 2, 1, 3)), H, N); @@ -735,7 +735,7 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { ggml_tensor * x_q = ggml_add(C, ggml_mul(C, ggml_norm(C, x, proj_ln_eps), w.n1w), w.n1b); ggml_tensor * qp = as_type(C, ggml_add(C, mm_act(C, c.Wq, x_q, at), c.bq), GGML_TYPE_F32); ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, qp, hd_dit, dit_heads, horizon), 0, 2, 1, 3)); - ggml_tensor * kq = ggml_mul_mat(C, c.K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * kq = ggml_mul_mat(C, c.K, Q); ggml_prec_set_acc(kq, GGML_PREC_F32); // The Evo-1 reference cross-attends over the full padded context // (no key mask), so the action queries see every LM position. ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale_dit, 0.0f); diff --git a/src/models/octo.cpp b/src/models/octo.cpp index 0bd5bc3..8690059 100644 --- a/src/models/octo.cpp +++ b/src/models/octo.cpp @@ -862,7 +862,7 @@ bool run_t5_encoder_graph(OctoRuntime& rt, ggml_tensor * Vh = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, V, head_dim, heads, seq), 1, 2, 0, 3)); ggml_tensor * scores = ggml_mul_mat(C, Kh, Qh); - ggml_mul_mat_set_prec(scores, GGML_PREC_F32); + ggml_prec_set_acc(scores, GGML_PREC_F32); // T5 folds 1/sqrt(d_k) into the weights, so the scale here is 1. ggml_tensor * probs = ggml_soft_max_ext(C, scores, mask, 1.0f, 0.0f); ggml_tensor * attended = ggml_mul_mat(C, Vh, probs); @@ -1113,7 +1113,7 @@ bool run_transformer_graph(OctoRuntime& rt, ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); ggml_tensor * scores = ggml_mul_mat(C, K, Q); - ggml_mul_mat_set_prec(scores, GGML_PREC_F32); + ggml_prec_set_acc(scores, GGML_PREC_F32); ggml_tensor * probs = ggml_soft_max_ext(C, scores, mask, attn_scale, 0.0f); ggml_tensor * attended = ggml_mul_mat(C, V, probs); ggml_tensor * merged = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, attended, 0, 2, 1, 3)), kHidden, seq); @@ -1445,7 +1445,7 @@ bool run_l1_action_head_graph(OctoRuntime& rt, ggml_tensor * Vh = ggml_cont(C, ggml_permute(C, ggml_reshape_4d(C, v, map_head_dim, map_heads, 1, width), 1, 2, 0, 3)); ggml_tensor * scores = ggml_mul_mat(C, Qh, Kh); - ggml_mul_mat_set_prec(scores, GGML_PREC_F32); + ggml_prec_set_acc(scores, GGML_PREC_F32); // One readout token per timestep, so this softmax is over a single logit // and always yields 1.0. Kept as the real op in case that changes. ggml_tensor * probs = ggml_soft_max_ext(C, scores, nullptr, 1.0f/std::sqrt((float) map_head_dim), 0.0f); diff --git a/src/models/openvla_oft.cpp b/src/models/openvla_oft.cpp index 735b1fe..4dc9140 100644 --- a/src/models/openvla_oft.cpp +++ b/src/models/openvla_oft.cpp @@ -360,7 +360,7 @@ std::vector OpenVlaOftModelArch::predict(const Inputs& in) { ggml_tensor*qr=ggml_rope_ext(C,qh,t_pos,nullptr,(int)lm_head_dim,GGML_ROPE_TYPE_NEOX,0,lm_rope_base,1.0f,0.0f,1.0f,32.0f,1.0f); ggml_tensor*kr=ggml_rope_ext(C,kh,t_pos,nullptr,(int)lm_head_dim,GGML_ROPE_TYPE_NEOX,0,lm_rope_base,1.0f,0.0f,1.0f,32.0f,1.0f); ggml_tensor*Q=ggml_cont(C,ggml_permute(C,qr,0,2,1,3)),*K=ggml_cont(C,ggml_permute(C,kr,0,2,1,3)),*V=ggml_cont(C,ggml_permute(C,vh,1,2,0,3)); - ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_mul_mat_set_prec(kq,GGML_PREC_F32); + ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_prec_set_acc(kq,GGML_PREC_F32); // Unmasked on purpose: OpenVLA-OFT patches transformers to replace the // causal mask across the whole sequence (modeling_llama.py:719-723). ggml_tensor*aw=ggml_soft_max_ext(C,kq,nullptr,lsc,0.0f); diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index 27f25ff..ed61142 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -141,11 +141,11 @@ ggml_tensor * build_siglip_layer(ggml_context * C, const EncBlockW & w, ggml_ten // kernel takes F16 K/V only; see fa_kv). ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 0, 2, 1, 3)); ggml_tensor * fa = ggml_flash_attn_ext(C, Q, vla::fa_kv(C, K), vla::fa_kv(C, V), nullptr, scale, 0.0f, 0.0f); - ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); + ggml_prec_set_acc(fa, GGML_PREC_F32); att = ggml_reshape_2d(C, fa, hidden, seq); } else { ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_prec_set_acc(kq, GGML_PREC_F32); ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); } @@ -211,12 +211,12 @@ ggml_tensor * build_gemma_layer( // -inf, both exactly representable in F16, so the cast is lossless. ggml_tensor * mask_f16 = mask ? ggml_cast(ctx, mask, GGML_TYPE_F16) : nullptr; ggml_tensor * fa = ggml_flash_attn_ext(ctx, Q, vla::fa_kv(ctx, K), vla::fa_kv(ctx, V), mask_f16, scale, 0.0f, 0.0f); - ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); + ggml_prec_set_acc(fa, GGML_PREC_F32); att_pre = ggml_reshape_2d(ctx, fa, qf, seq); } else { ggml_tensor * V = ggml_cont(ctx, ggml_permute(ctx, V_full, 1, 2, 0, 3)); ggml_tensor * kq = ggml_mul_mat(ctx, K, Q); - ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_prec_set_acc(kq, GGML_PREC_F32); ggml_tensor * attn = ggml_soft_max_ext(ctx, kq, mask, scale, 0.f); ggml_tensor * kqv = ggml_mul_mat(ctx, V, attn); att_pre = ggml_reshape_2d(ctx, diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index 02a5cce..1e348d8 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -138,7 +138,7 @@ ggml_tensor * build_siglip_layer(ggml_context * C, const EncBlockW & w, ggml_ten ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, head_dim, heads, seq), 0, 2, 1, 3)); ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, head_dim, heads, seq), 0, 2, 1, 3)); ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_prec_set_acc(kq, GGML_PREC_F32); ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); ggml_tensor * att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); ggml_tensor * h1 = ggml_add(C, x, ggml_add(C, ggml_mul_mat(C, w.Wo, att), w.bo)); @@ -187,7 +187,7 @@ ggml_tensor * build_vlm_layer( ggml_tensor * V = ggml_cont(ctx, ggml_permute(ctx, v_h, 1, 2, 0, 3)); ggml_tensor * kq = ggml_mul_mat(ctx, K, Q); - ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_prec_set_acc(kq, GGML_PREC_F32); const float scale = 1.f/std::sqrt((float) hd); ggml_tensor * attn = ggml_soft_max_ext(ctx, kq, nullptr, scale, 0.f); ggml_tensor * kqv = ggml_mul_mat(ctx, V, attn); @@ -260,7 +260,7 @@ ggml_tensor * build_expert_layer( ggml_tensor * V = ggml_cont(ctx, ggml_permute(ctx, V_full, 1, 2, 0, 3)); ggml_tensor * kq = ggml_mul_mat(ctx, K, Q); - ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_prec_set_acc(kq, GGML_PREC_F32); const float scale = 1.f/std::sqrt((float) hd); ggml_tensor * attn = ggml_soft_max_ext(ctx, kq, nullptr, scale, 0.f); ggml_tensor * kqv = ggml_mul_mat(ctx, V, attn); diff --git a/src/models/smolvla.cpp b/src/models/smolvla.cpp index d9fd0fc..51f6eef 100644 --- a/src/models/smolvla.cpp +++ b/src/models/smolvla.cpp @@ -442,11 +442,11 @@ ggml_tensor * build_siglip_layer(ggml_context * C, const EncBlockW & w, ggml_ten // this op the same way. ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 0, 2, 1, 3)); ggml_tensor * fa = ggml_flash_attn_ext(C, Q, vla::fa_kv(C, K), vla::fa_kv(C, V), nullptr, scale, 0.0f, 0.0f); - ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); + ggml_prec_set_acc(fa, GGML_PREC_F32); att = ggml_reshape_2d(C, fa, hidden, seq); } else { ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_prec_set_acc(kq, GGML_PREC_F32); ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); } @@ -802,7 +802,7 @@ static inline bool tower_mm_f32_prec() { static inline ggml_tensor * mm_w(ggml_context * ctx, ggml_tensor * w, ggml_tensor * x) { ggml_tensor * r = ggml_mul_mat(ctx, w, x); if (tower_mm_f32_prec()) - ggml_mul_mat_set_prec(r, GGML_PREC_F32); + ggml_prec_set_acc(r, GGML_PREC_F32); return r; } @@ -830,7 +830,7 @@ ggml_tensor * build_vlm_layer(ggml_context * ctx, const VlmLayerW & w, const float scale = 1.f/std::sqrt(static_cast(cfg.head_dim)); ggml_tensor * fa = ggml_flash_attn_ext(ctx, Q, K, V, mask, scale, 0.f, 0.f); - ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); + ggml_prec_set_acc(fa, GGML_PREC_F32); ggml_tensor * att_pre_o = ggml_reshape_2d(ctx, fa, cfg.q_full_dim, cfg.n_prefix); ggml_tensor * o_out = mm_w(ctx, w.Wo, att_pre_o); ggml_tensor * h1 = ggml_add(ctx, x_in, o_out); @@ -869,7 +869,7 @@ ggml_tensor * build_expert_self_attn_layer( const float scale = 1.f/std::sqrt(static_cast(cfg.head_dim)); ggml_tensor * fa = ggml_flash_attn_ext(ctx, Q, Kp, Vp, mask_full, scale, 0.f, 0.f); - ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); + ggml_prec_set_acc(fa, GGML_PREC_F32); ggml_tensor * att_pre_o = ggml_reshape_2d(ctx, fa, cfg.q_full_dim, cfg.n_suffix); ggml_tensor * h1 = ggml_add(ctx, x_in, mm_w(ctx, w.Wo, att_pre_o)); @@ -908,7 +908,7 @@ ggml_tensor * build_expert_cross_attn_layer( const float scale = 1.f/std::sqrt(static_cast(cfg.head_dim)); ggml_tensor * fa = ggml_flash_attn_ext(ctx, Q, Kp, Vp, mask_prefix_only, scale, 0.f, 0.f); - ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); + ggml_prec_set_acc(fa, GGML_PREC_F32); ggml_tensor * att_pre_o = ggml_reshape_2d(ctx, fa, cfg.q_full_dim, cfg.n_suffix); ggml_tensor * h1 = ggml_add(ctx, x_in, mm_w(ctx, w.Wo, att_pre_o)); diff --git a/src/models/vla_adapter.cpp b/src/models/vla_adapter.cpp index 3a308da..016241a 100644 --- a/src/models/vla_adapter.cpp +++ b/src/models/vla_adapter.cpp @@ -393,7 +393,7 @@ std::vector VlaAdapterModelArch::predict(const Inputs& in) { ggml_tensor*qr=ggml_rope_ext(C,qh,t_pos,nullptr,(int)lm_head_dim,GGML_ROPE_TYPE_NEOX,0,lm_rope_base,1.0f,0.0f,1.0f,32.0f,1.0f); ggml_tensor*kr=ggml_rope_ext(C,kh,t_pos,nullptr,(int)lm_head_dim,GGML_ROPE_TYPE_NEOX,0,lm_rope_base,1.0f,0.0f,1.0f,32.0f,1.0f); ggml_tensor*Q=ggml_cont(C,ggml_permute(C,qr,0,2,1,3)),*K=ggml_cont(C,ggml_permute(C,kr,0,2,1,3)),*V=ggml_cont(C,ggml_permute(C,vh,1,2,0,3)); - ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_mul_mat_set_prec(kq,GGML_PREC_F32); + ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_prec_set_acc(kq,GGML_PREC_F32); ggml_tensor*aw=ggml_soft_max_ext(C,kq,t_mask,lsc,0.0f); ggml_tensor*kqv=ggml_mul_mat(C,V,aw); ggml_tensor*att=ggml_reshape_2d(C,ggml_cont(C,ggml_permute(C,kqv,0,2,1,3)),HC,SEQ); @@ -439,7 +439,7 @@ std::vector VlaAdapterModelArch::predict(const Inputs& in) { auto tov=[&](ggml_tensor*pp){ int64_t L=pp->ne[1]; return ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,pp,HD,NH,L),1,2,0,3)); }; ggml_tensor*Vs=tov(vse),*Va=tov(vad),*VT=tov(vta); ggml_tensor*ss2=ggml_mul_mat(C,kse,q),*sa=ggml_mul_mat(C,kad,q),*sr=ggml_mul_mat(C,kta,q); - ggml_mul_mat_set_prec(ss2,GGML_PREC_F32); ggml_mul_mat_set_prec(sa,GGML_PREC_F32); ggml_mul_mat_set_prec(sr,GGML_PREC_F32); + ggml_prec_set_acc(ss2,GGML_PREC_F32); ggml_prec_set_acc(sa,GGML_PREC_F32); ggml_prec_set_acc(sr,GGML_PREC_F32); ggml_tensor*st2=ggml_scale(C,sr,w.rg); ggml_tensor*scr=ggml_concat(C,ggml_concat(C,ss2,sa,0),st2,0); ggml_tensor*attn=ggml_soft_max_ext(C,scr,nullptr,hsc,0.0f); diff --git a/src/modules/dual_tower.h b/src/modules/dual_tower.h index ffee524..51c0e1c 100644 --- a/src/modules/dual_tower.h +++ b/src/modules/dual_tower.h @@ -92,7 +92,7 @@ inline ggml_tensor* vit_block(ggml_context*C, const ViTLayerW&w, ggml_tensor*x, ggml_tensor*Q=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,q,hd,heads,N),0,2,1,3)); ggml_tensor*K=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,k,hd,heads,N),0,2,1,3)); ggml_tensor*V=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,v,hd,heads,N),1,2,0,3)); - ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_mul_mat_set_prec(kq,GGML_PREC_F32); + ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_prec_set_acc(kq,GGML_PREC_F32); ggml_tensor*aw=ggml_soft_max_ext(C,kq,nullptr,sc,0.0f); ggml_tensor*kqv=ggml_mul_mat(C,V,aw); ggml_tensor*att=ggml_reshape_2d(C,ggml_cont(C,ggml_permute(C,kqv,0,2,1,3)),hidden,N); From 3faea0aedfce81837f45012aff181bc061a335de Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Mon, 28 Sep 2026 13:38:51 +0700 Subject: [PATCH 02/61] bump sentencepiece to v0.2.1 --- CMakeLists.txt | 6 +----- cmake/patch_sentencepiece_fpic.cmake | 2 +- 2 files changed, 2 insertions(+), 6 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 495aa54..dd0224e 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -142,15 +142,11 @@ if(VLA_OCTO) endif() FetchContent_Declare(sentencepiece GIT_REPOSITORY https://github.com/google/sentencepiece - GIT_TAG v0.2.0 + GIT_TAG v0.2.1 GIT_SHALLOW TRUE ${_vla_spm_patch} ) - # sentencepiece v0.2.0 still asks for cmake_minimum_required(VERSION 3.1), - # which CMake 4 refuses outright. - set(CMAKE_POLICY_VERSION_MINIMUM 3.5) FetchContent_MakeAvailable(sentencepiece) - unset(CMAKE_POLICY_VERSION_MINIMUM) # Only the static library is linked; its CLI tools are dead weight. vla_exclude_fetched_targets(${sentencepiece_SOURCE_DIR}) # vcpkg's FindProtobuf shim reports the DLL, not its import library, as diff --git a/cmake/patch_sentencepiece_fpic.cmake b/cmake/patch_sentencepiece_fpic.cmake index 3e9eba6..3371619 100644 --- a/cmake/patch_sentencepiece_fpic.cmake +++ b/cmake/patch_sentencepiece_fpic.cmake @@ -1,4 +1,4 @@ -# sentencepiece v0.2.0 adds -fPIC for every compiler that is not MSVC, and +# sentencepiece v0.2.1 adds -fPIC for every compiler that is not MSVC, and # clang targeting arm64-pc-windows-msvc rejects the flag as a hard error. # Windows has no PIC to ask for, so drop it. Run as the FetchContent patch step # with the working directory at the sentencepiece source root. From eff454e07cf6681ceaa1f0750f0e6d5f71dc4327 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Mon, 28 Sep 2026 13:38:51 +0700 Subject: [PATCH 03/61] bump OpenVINO, macOS runner and python deps --- .github/workflows/release.yml | 2 +- CHANGELOG.md | 28 ++++++++++++--- bindings/python/pyproject.toml | 8 ++--- pyproject.toml | 18 +++++----- scripts/install_ov.sh | 62 +++++++++++++++++++++------------- 5 files changed, 77 insertions(+), 41 deletions(-) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 786f7f8..f6a76a5 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -81,7 +81,7 @@ jobs: path: '*.tar.gz' macos: - runs-on: macos-14 + runs-on: macos-15 steps: - uses: actions/checkout@v7 diff --git a/CHANGELOG.md b/CHANGELOG.md index f98d4c6..a33e1aa 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -118,11 +118,29 @@ Notable changes to vla.cpp. Format loosely follows [Keep a Changelog](https://ke ### Changed -- llama.cpp pinned at `b10729`, up from `b10331`. Brings OpenVINO 2026.3.1, the - IM2COL+MatMul to native-convolution fusion, and the `RELU`/`NEG`/`SQR` - translators, which the local patch no longer has to add. Byte-identical on the - CPU backend for all eleven archs. The build.yml cache key now reads the tag out - of `CMakeLists.txt` instead of repeating it. +- llama.cpp pinned at `b11223`, up from `b10331` (via `b10729`). Brings the + IM2COL+MatMul to native-convolution fusion, the `RELU`/`NEG`/`SQR` translators + the local patch no longer adds, CUDA RMS_NORM+SCALE fusion, fixes for a + divergent `__syncthreads` in the f16 flash-attention kernel and races in + mmf/mmid, and Metal and OpenCL correctness fixes. All thirteen archs are + byte-identical on CUDA (sm_120) against `b10729`. The + `ggml_mul_mat_set_prec`/`ggml_flash_attn_ext_set_prec` calls moved to + `ggml_prec_set_acc`, which writes the same op params. The build.yml cache key + now reads the tag out of `CMakeLists.txt` instead of repeating it. +- The default CUDA architecture list is set before ggml is configured, so + ggml-cuda and the in-tree kernels build the same set, and it gains sm_110 + (Jetson Thor, CUDA 13) and sm_121 (DGX Spark, CUDA 12.9). Only the f16 flash + attention vector kernels are built (`GGML_CUDA_FA_QUANTS=f16-f16`). +- SentencePiece `v0.2.1`, which fixes a heap overflow on a malformed + normalization model. Octo loads that model from GGUF bytes. +- OpenVINO 2026.4 in `scripts/install_ov.sh`, with newer Intel GPU and NPU + drivers on Ubuntu 24.04. Every downloaded archive and package is now + checksummed, including the NPU driver, level-zero and IGC packages. +- The release macOS job runs on `macos-15`; `macos-14` runners are retired. +- Python tooling: transformers 5.x is supported (the client needs `>=5.4` for + the VLA-JEPA processor), the client extra gains the modules it imports + (`torchvision`, `protobuf`, `opencv-python-headless`), and both pyprojects use + an SPDX license string. - `src/models/dit_common.h` is gone. It redefined six `vla::` functions that `src/layers/` already had, with both copies linked into `vla_core`. Every includer used only `sinusoidal_time_emb` or `build_causal_mask`, so they now diff --git a/bindings/python/pyproject.toml b/bindings/python/pyproject.toml index af666ad..caa6fa3 100644 --- a/bindings/python/pyproject.toml +++ b/bindings/python/pyproject.toml @@ -1,14 +1,14 @@ [build-system] -requires = ["setuptools>=68"] +requires = ["setuptools>=77"] build-backend = "setuptools.build_meta" [project] name = "vla-cpp" -version = "0.2.0" +version = "0.3.0" description = "Python bindings for vla.cpp, a C++ inference engine for Vision-Language-Action models." readme = "README.md" -requires-python = ">=3.9" -license = { text = "Apache-2.0" } +requires-python = ">=3.10" +license = "Apache-2.0" dependencies = [] [project.optional-dependencies] diff --git a/pyproject.toml b/pyproject.toml index ec97106..67f7e31 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,15 +4,16 @@ version = "0.3.0" description = "Python tooling for vla.cpp: HuggingFace -> GGUF converters and the ZeroMQ eval client." readme = "README.md" requires-python = ">=3.10" -license = { text = "Apache-2.0" } +license = "Apache-2.0" dependencies = ["numpy>=1.24"] [project.optional-dependencies] # scripts/convert_*_to_gguf.py and the gguf_common / gguf_blocks modules they share convert = [ "torch>=2.5", - "safetensors>=0.4", - "gguf>=0.10", + "safetensors>=0.4.3", + "gguf>=0.17", + "transformers>=4.57", "numpy>=1.24", ] # eval/client/* talks to vla-server over ZeroMQ. The simulators (LIBERO, SimplerEnv, @@ -20,13 +21,14 @@ convert = [ client = [ "pyzmq>=25", "msgpack>=1.1", - "msgpack-numpy>=0.4.8", "pillow>=10", "torch>=2.5", - # Only used for AutoTokenizer/AutoProcessor.from_pretrained (the pi0 PaliGemma - # tokenizer). Capped below 5.0: the v5 line is a breaking rewrite we have not - # tested against. - "transformers>=4.51,<5", + "torchvision>=0.20", + # AutoTokenizer/AutoProcessor only. vla_jepa passes processor_kwargs to + # apply_chat_template, which 5.4 is the first release to accept. + "transformers>=5.4", + "protobuf>=4.21", + "opencv-python-headless>=4.8", "numpy>=1.24", ] diff --git a/scripts/install_ov.sh b/scripts/install_ov.sh index 28d450c..310df2d 100644 --- a/scripts/install_ov.sh +++ b/scripts/install_ov.sh @@ -21,8 +21,8 @@ need_cmd() { # Digests of the two archives the defaults below pin. These land in /opt under # sudo, so a bad download is a root-level problem. -OPENVINO_SHA256_2204="d701a115d3dc18088ff75b5b8e67a51fbf780022a3d40ee8ee7f2adfbd9915e6" -OPENVINO_SHA256_2404="6931e5a3c9b1fc9cb170137196df2c40489625703f2d184f511b7add2c110ef8" +OPENVINO_SHA256_2204="d327ede0a5dd29ad6e73d156aa7fe43b7d8fc0eac5e789f17bf9d381ab333e7d" +OPENVINO_SHA256_2404="0bd86d578beb1e8805655f593315c69bf909008e212f801ed76e3a350927736f" # verify_sha256 . An overridden version has no # digest here, so fall back to the one the mirror publishes: that catches a @@ -127,6 +127,10 @@ install_gpu_2204() { local igc_base_url="https://github.com/intel/intel-graphics-compiler/releases/download/v2.10.8" local crt_base_url="https://github.com/intel/compute-runtime/releases/download/25.13.33276.16" local checksum_file="ww13.sum" + local igc_sums=( + "85bb185f5c9f0700321c6ce9362a4aa85b6cdf02f74a7b8685424d0be024fb0e intel-igc-core-2_2.10.8+18926_amd64.deb" + "21e5ee9f0b5798335d82a88f98e80bccd9539380782e60893d6d050eb801d322 intel-igc-opencl-2_2.10.8+18926_amd64.deb" + ) local packages=( "${igc_base_url}/intel-igc-core-2_2.10.8+18926_amd64.deb" "${igc_base_url}/intel-igc-opencl-2_2.10.8+18926_amd64.deb" @@ -146,7 +150,8 @@ install_gpu_2204() { done wget --no-continue "${crt_base_url}/${checksum_file}" - sha256sum --ignore-missing -c "${checksum_file}" + sha256sum -c "${checksum_file}" + printf '%s\n' "${igc_sums[@]}" | sha256sum -c - shopt -s nullglob local artifacts=( *.deb *.ddeb ) @@ -167,6 +172,8 @@ install_npu_2204() { local npu_url="https://github.com/intel/linux-npu-driver/releases/download/v1.26.0/${npu_tarball}" local level_zero_deb="level-zero_1.24.2+u22.04_amd64.deb" local level_zero_url="https://github.com/oneapi-src/level-zero/releases/download/v1.24.2/${level_zero_deb}" + local npu_sha256="cfdbcc9adc1ea20d498ebd9cbdb5c212f6fc940e1034ef7a72e239a8636f653a" + local level_zero_sha256="7c304e93835d96025c90f6a3d8f2ce5edf142c24da9f4871113a4f0225fef22e" log "Installing Intel NPU drivers for Ubuntu 22.04..." mkdir -p "${download_dir}" @@ -176,6 +183,8 @@ install_npu_2204() { # at all if the fetch fails. wget --no-continue "${npu_url}" wget --no-continue "${level_zero_url}" + verify_sha256 "${npu_tarball}" "${npu_sha256}" "${npu_url}" + verify_sha256 "${level_zero_deb}" "${level_zero_sha256}" "${level_zero_url}" tar -xf "${npu_tarball}" mapfile -t npu_debs < <(find . -type f -name '*.deb' ! -name 'level-zero*.deb' | sort) @@ -201,8 +210,8 @@ install_npu_2204() { install_runtime_2204() { local download_dir="${WORK_DIR}/openvino_runtime_2204" - local openvino_version="${OPENVINO_VERSION:-2025.3}" - local openvino_build="${OPENVINO_BUILD:-19807.44526285f24}" + local openvino_version="${OPENVINO_VERSION:-2026.4}" + local openvino_build="${OPENVINO_BUILD:-22959.99c81491cc3}" local openvino_archive="openvino_toolkit_ubuntu22_${openvino_version}.0.${openvino_build}_x86_64.tgz" local openvino_dirname="openvino_toolkit_ubuntu22_${openvino_version}.0.${openvino_build}_x86_64" local openvino_url="https://storage.openvinotoolkit.org/repositories/openvino/packages/${openvino_version}/linux/${openvino_archive}" @@ -211,7 +220,7 @@ install_runtime_2204() { local symlink_path="${install_root}/openvino" local archive_path="${download_dir}/openvino_${openvino_version}.tgz" local expected_sha="" - if [[ "${openvino_version}" == "2025.3" && "${openvino_build}" == "19807.44526285f24" ]]; then + if [[ "${openvino_version}" == "2026.4" && "${openvino_build}" == "22959.99c81491cc3" ]]; then expected_sha="${OPENVINO_SHA256_2204}" fi @@ -238,19 +247,23 @@ install_runtime_2204() { install_gpu_2404() { local download_dir="${WORK_DIR}/intel_gpu_2404" - local igc_base_url="https://github.com/intel/intel-graphics-compiler/releases/download/v2.36.3" - local crt_base_url="https://github.com/intel/compute-runtime/releases/download/26.22.38646.4" - local checksum_file="ww22.sum" + local igc_base_url="https://github.com/intel/intel-graphics-compiler/releases/download/v2.41.5" + local crt_base_url="https://github.com/intel/compute-runtime/releases/download/26.35.39758.10" + local checksum_file="ww35.sum" + local igc_sums=( + "0a6e64a663ae65a0fa02d6912ae3b6b37cf85b90c21cc423fd9fef70aaf4f628 intel-igc-core-2_2.41.5+22716_amd64.deb" + "779e1b9e88098eb25711e9a8f67c2752665bad22f134aa40ed5649f6e1b87058 intel-igc-opencl-2_2.41.5+22716_amd64.deb" + ) local packages=( - "${igc_base_url}/intel-igc-core-2_2.36.3+21719_amd64.deb" - "${igc_base_url}/intel-igc-opencl-2_2.36.3+21719_amd64.deb" - "${crt_base_url}/intel-ocloc-dbgsym_26.22.38646.4-0_amd64.ddeb" - "${crt_base_url}/intel-ocloc_26.22.38646.4-0_amd64.deb" - "${crt_base_url}/intel-opencl-icd-dbgsym_26.22.38646.4-0_amd64.ddeb" - "${crt_base_url}/intel-opencl-icd_26.22.38646.4-0_amd64.deb" + "${igc_base_url}/intel-igc-core-2_2.41.5+22716_amd64.deb" + "${igc_base_url}/intel-igc-opencl-2_2.41.5+22716_amd64.deb" + "${crt_base_url}/intel-ocloc-dbgsym_26.35.39758.10-0_amd64.ddeb" + "${crt_base_url}/intel-ocloc_26.35.39758.10-0_amd64.deb" + "${crt_base_url}/intel-opencl-icd-dbgsym_26.35.39758.10-0_amd64.ddeb" + "${crt_base_url}/intel-opencl-icd_26.35.39758.10-0_amd64.deb" "${crt_base_url}/libigdgmm12_22.10.0_amd64.deb" - "${crt_base_url}/libze-intel-gpu1-dbgsym_26.22.38646.4-0_amd64.ddeb" - "${crt_base_url}/libze-intel-gpu1_26.22.38646.4-0_amd64.deb" + "${crt_base_url}/libze-intel-gpu1-dbgsym_26.35.39758.10-0_amd64.ddeb" + "${crt_base_url}/libze-intel-gpu1_26.35.39758.10-0_amd64.deb" ) log "Installing Intel GPU drivers for Ubuntu 24.04..." @@ -262,7 +275,8 @@ install_gpu_2404() { done wget --no-continue "${crt_base_url}/${checksum_file}" - sha256sum --ignore-missing -c "${checksum_file}" + sha256sum -c "${checksum_file}" + printf '%s\n' "${igc_sums[@]}" | sha256sum -c - shopt -s nullglob local artifacts=( *.deb *.ddeb ) @@ -279,9 +293,10 @@ install_gpu_2404() { install_npu_2404() { local download_dir="${WORK_DIR}/intel_npu_2404" - local npu_release="v1.33.0" - local npu_archive="linux-npu-driver-v1.33.0.20260529-26625960453-ubuntu2404.tar.gz" + local npu_release="v1.38.0" + local npu_archive="linux-npu-driver-v1.38.0.20260910-34487311128-ubuntu2404.tar.gz" local npu_url="https://github.com/intel/linux-npu-driver/releases/download/${npu_release}/${npu_archive}" + local npu_sha256="1efcd4b60c22abee751d8f2705962cbcc2a569de45c7e0e670cf08afbfcdc1d2" local npu_packages=( intel-driver-compiler-npu intel-fw-npu @@ -296,6 +311,7 @@ install_npu_2404() { # Download before purging, so a failed fetch does not leave the machine with # no NPU driver at all. wget --no-continue "${npu_url}" + verify_sha256 "${npu_archive}" "${npu_sha256}" "${npu_url}" tar -xf "${npu_archive}" shopt -s nullglob @@ -321,8 +337,8 @@ install_npu_2404() { install_runtime_2404() { local download_dir="${WORK_DIR}/openvino_runtime_2404" - local openvino_version="${OPENVINO_VERSION:-2026.2.1}" - local openvino_build="${OPENVINO_BUILD:-21919.ede283a88e3}" + local openvino_version="${OPENVINO_VERSION:-2026.4}" + local openvino_build="${OPENVINO_BUILD:-0.22959.99c81491cc3}" local openvino_archive="openvino_toolkit_ubuntu24_${openvino_version}.${openvino_build}_x86_64.tgz" local openvino_dirname="openvino_toolkit_ubuntu24_${openvino_version}.${openvino_build}_x86_64" local openvino_url="https://storage.openvinotoolkit.org/repositories/openvino/packages/${openvino_version}/linux/${openvino_archive}" @@ -331,7 +347,7 @@ install_runtime_2404() { local symlink_path="${install_root}/openvino" local archive_path="${download_dir}/openvino_${openvino_version}.tgz" local expected_sha="" - if [[ "${openvino_version}" == "2026.2.1" && "${openvino_build}" == "21919.ede283a88e3" ]]; then + if [[ "${openvino_version}" == "2026.4" && "${openvino_build}" == "0.22959.99c81491cc3" ]]; then expected_sha="${OPENVINO_SHA256_2404}" fi From 79533b204d515442af800d0645300438821c4857 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Mon, 28 Sep 2026 14:53:57 +0700 Subject: [PATCH 04/61] harden vla-server and vlm-server --- examples/chat/vlm_pb2.py | 511 +++---------------------------------- src/serving/server.cpp | 84 +++--- src/serving/vla-bench.cpp | 1 + src/serving/vla-cli.cpp | 6 +- src/serving/vlm-server.cpp | 50 ++-- src/vlm/engine.cpp | 175 ++++++------- src/vlm/engine.h | 17 -- 7 files changed, 203 insertions(+), 641 deletions(-) diff --git a/examples/chat/vlm_pb2.py b/examples/chat/vlm_pb2.py index c25c443..03a12ae 100644 --- a/examples/chat/vlm_pb2.py +++ b/examples/chat/vlm_pb2.py @@ -20,491 +20,42 @@ # OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE # SOFTWARE. +# -*- coding: utf-8 -*- +# Generated by the protocol buffer compiler. DO NOT EDIT! +# source: vlm.proto +"""Generated protocol buffer code.""" +from google.protobuf.internal import builder as _builder from google.protobuf import descriptor as _descriptor -from google.protobuf import message as _message -from google.protobuf import reflection as _reflection +from google.protobuf import descriptor_pool as _descriptor_pool from google.protobuf import symbol_database as _symbol_database +# @@protoc_insertion_point(imports) _sym_db = _symbol_database.Default() -DESCRIPTOR = _descriptor.FileDescriptor( - name='vlm.proto', - package='vlm_chat', - syntax='proto3', - serialized_options=None, - create_key=_descriptor._internal_create_key, - serialized_pb=b'\n\tvlm.proto\x12\x08vlm_chat\"\x82\x01\n\x05Image\x12*\n\x08\x65ncoding\x18\x01 \x01(\x0e\x32\x18.vlm_chat.Image.Encoding\x12\r\n\x05width\x18\x02 \x01(\r\x12\x0e\n\x06height\x18\x03 \x01(\r\x12\x0c\n\x04\x64\x61ta\x18\x04 \x01(\x0c\" \n\x08\x45ncoding\x12\x08\n\x04JPEG\x10\x00\x12\n\n\x06RGB_U8\x10\x01\",\n\x0b\x43hatMessage\x12\x0c\n\x04role\x18\x01 \x01(\t\x12\x0f\n\x07\x63ontent\x18\x02 \x01(\t\"e\n\x0eSamplingParams\x12\x13\n\x0btemperature\x18\x01 \x01(\x02\x12\r\n\x05top_p\x18\x02 \x01(\x02\x12\r\n\x05top_k\x18\x03 \x01(\x05\x12\x12\n\nmax_tokens\x18\x04 \x01(\x05\x12\x0c\n\x04seed\x18\x05 \x01(\x04\"\xa7\x01\n\x0b\x43hatRequest\x12\'\n\x08messages\x18\x01 \x03(\x0b\x32\x15.vlm_chat.ChatMessage\x12\x1f\n\x06images\x18\x02 \x03(\x0b\x32\x0f.vlm_chat.Image\x12*\n\x08sampling\x18\x03 \x01(\x0b\x32\x18.vlm_chat.SamplingParams\x12\x0e\n\x06stream\x18\x04 \x01(\x08\x12\x12\n\nrequest_id\x18\x05 \x01(\x04\"\xd9\x01\n\x0c\x43hatResponse\x12\x12\n\nrequest_id\x18\x01 \x01(\x04\x12\x0c\n\x04text\x18\x02 \x01(\t\x12\x15\n\rfinish_reason\x18\x03 \x01(\t\x12\x15\n\rprompt_tokens\x18\x04 \x01(\r\x12\x19\n\x11\x63ompletion_tokens\x18\x05 \x01(\r\x12\x18\n\x10latency_ms_total\x18\x06 \x01(\x02\x12\x1a\n\x12latency_ms_prefill\x18\x07 \x01(\x02\x12\x19\n\x11latency_ms_decode\x18\x08 \x01(\x02\x12\r\n\x05\x65rror\x18\t \x01(\t\"4\n\x0f\x43hatStreamDelta\x12\x12\n\nrequest_id\x18\x01 \x01(\x04\x12\r\n\x05\x64\x65lta\x18\x02 \x01(\t\"l\n\rStreamMessage\x12*\n\x05\x64\x65lta\x18\x01 \x01(\x0b\x32\x19.vlm_chat.ChatStreamDeltaH\x00\x12\'\n\x05\x66inal\x18\x02 \x01(\x0b\x32\x16.vlm_chat.ChatResponseH\x00\x42\x06\n\x04kindb\x06proto3' -) -_IMAGE_ENCODING = _descriptor.EnumDescriptor( - name='Encoding', - full_name='vlm_chat.Image.Encoding', - filename=None, - file=DESCRIPTOR, - create_key=_descriptor._internal_create_key, - values=[ - _descriptor.EnumValueDescriptor( - name='JPEG', index=0, number=0, - serialized_options=None, - type=None, - create_key=_descriptor._internal_create_key), - _descriptor.EnumValueDescriptor( - name='RGB_U8', index=1, number=1, - serialized_options=None, - type=None, - create_key=_descriptor._internal_create_key), - ], - containing_type=None, - serialized_options=None, - serialized_start=122, - serialized_end=154, -) -_sym_db.RegisterEnumDescriptor(_IMAGE_ENCODING) -_IMAGE = _descriptor.Descriptor( - name='Image', - full_name='vlm_chat.Image', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='encoding', full_name='vlm_chat.Image.encoding', index=0, - number=1, type=14, cpp_type=8, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='width', full_name='vlm_chat.Image.width', index=1, - number=2, type=13, cpp_type=3, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='height', full_name='vlm_chat.Image.height', index=2, - number=3, type=13, cpp_type=3, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='data', full_name='vlm_chat.Image.data', index=3, - number=4, type=12, cpp_type=9, label=1, - has_default_value=False, default_value=b"", - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - _IMAGE_ENCODING, - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - ], - serialized_start=24, - serialized_end=154, -) -_CHATMESSAGE = _descriptor.Descriptor( - name='ChatMessage', - full_name='vlm_chat.ChatMessage', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='role', full_name='vlm_chat.ChatMessage.role', index=0, - number=1, type=9, cpp_type=9, label=1, - has_default_value=False, default_value=b"".decode('utf-8'), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='content', full_name='vlm_chat.ChatMessage.content', index=1, - number=2, type=9, cpp_type=9, label=1, - has_default_value=False, default_value=b"".decode('utf-8'), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - ], - serialized_start=156, - serialized_end=200, -) - -_SAMPLINGPARAMS = _descriptor.Descriptor( - name='SamplingParams', - full_name='vlm_chat.SamplingParams', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='temperature', full_name='vlm_chat.SamplingParams.temperature', index=0, - number=1, type=2, cpp_type=6, label=1, - has_default_value=False, default_value=float(0), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='top_p', full_name='vlm_chat.SamplingParams.top_p', index=1, - number=2, type=2, cpp_type=6, label=1, - has_default_value=False, default_value=float(0), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='top_k', full_name='vlm_chat.SamplingParams.top_k', index=2, - number=3, type=5, cpp_type=1, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='max_tokens', full_name='vlm_chat.SamplingParams.max_tokens', index=3, - number=4, type=5, cpp_type=1, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='seed', full_name='vlm_chat.SamplingParams.seed', index=4, - number=5, type=4, cpp_type=4, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - ], - serialized_start=202, - serialized_end=303, -) - -_CHATREQUEST = _descriptor.Descriptor( - name='ChatRequest', - full_name='vlm_chat.ChatRequest', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='messages', full_name='vlm_chat.ChatRequest.messages', index=0, - number=1, type=11, cpp_type=10, label=3, - has_default_value=False, default_value=[], - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='images', full_name='vlm_chat.ChatRequest.images', index=1, - number=2, type=11, cpp_type=10, label=3, - has_default_value=False, default_value=[], - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='sampling', full_name='vlm_chat.ChatRequest.sampling', index=2, - number=3, type=11, cpp_type=10, label=1, - has_default_value=False, default_value=None, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='stream', full_name='vlm_chat.ChatRequest.stream', index=3, - number=4, type=8, cpp_type=7, label=1, - has_default_value=False, default_value=False, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='request_id', full_name='vlm_chat.ChatRequest.request_id', index=4, - number=5, type=4, cpp_type=4, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - ], - serialized_start=306, - serialized_end=473, -) - -_CHATRESPONSE = _descriptor.Descriptor( - name='ChatResponse', - full_name='vlm_chat.ChatResponse', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='request_id', full_name='vlm_chat.ChatResponse.request_id', index=0, - number=1, type=4, cpp_type=4, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='text', full_name='vlm_chat.ChatResponse.text', index=1, - number=2, type=9, cpp_type=9, label=1, - has_default_value=False, default_value=b"".decode('utf-8'), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='finish_reason', full_name='vlm_chat.ChatResponse.finish_reason', index=2, - number=3, type=9, cpp_type=9, label=1, - has_default_value=False, default_value=b"".decode('utf-8'), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='prompt_tokens', full_name='vlm_chat.ChatResponse.prompt_tokens', index=3, - number=4, type=13, cpp_type=3, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='completion_tokens', full_name='vlm_chat.ChatResponse.completion_tokens', index=4, - number=5, type=13, cpp_type=3, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='latency_ms_total', full_name='vlm_chat.ChatResponse.latency_ms_total', index=5, - number=6, type=2, cpp_type=6, label=1, - has_default_value=False, default_value=float(0), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='latency_ms_prefill', full_name='vlm_chat.ChatResponse.latency_ms_prefill', index=6, - number=7, type=2, cpp_type=6, label=1, - has_default_value=False, default_value=float(0), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='latency_ms_decode', full_name='vlm_chat.ChatResponse.latency_ms_decode', index=7, - number=8, type=2, cpp_type=6, label=1, - has_default_value=False, default_value=float(0), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='error', full_name='vlm_chat.ChatResponse.error', index=8, - number=9, type=9, cpp_type=9, label=1, - has_default_value=False, default_value=b"".decode('utf-8'), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - ], - serialized_start=476, - serialized_end=693, -) - -_CHATSTREAMDELTA = _descriptor.Descriptor( - name='ChatStreamDelta', - full_name='vlm_chat.ChatStreamDelta', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='request_id', full_name='vlm_chat.ChatStreamDelta.request_id', index=0, - number=1, type=4, cpp_type=4, label=1, - has_default_value=False, default_value=0, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='delta', full_name='vlm_chat.ChatStreamDelta.delta', index=1, - number=2, type=9, cpp_type=9, label=1, - has_default_value=False, default_value=b"".decode('utf-8'), - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - ], - serialized_start=695, - serialized_end=747, -) - -_STREAMMESSAGE = _descriptor.Descriptor( - name='StreamMessage', - full_name='vlm_chat.StreamMessage', - filename=None, - file=DESCRIPTOR, - containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[ - _descriptor.FieldDescriptor( - name='delta', full_name='vlm_chat.StreamMessage.delta', index=0, - number=1, type=11, cpp_type=10, label=1, - has_default_value=False, default_value=None, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - _descriptor.FieldDescriptor( - name='final', full_name='vlm_chat.StreamMessage.final', index=1, - number=2, type=11, cpp_type=10, label=1, - has_default_value=False, default_value=None, - message_type=None, enum_type=None, containing_type=None, - is_extension=False, extension_scope=None, - serialized_options=None, file=DESCRIPTOR, create_key=_descriptor._internal_create_key), - ], - extensions=[ - ], - nested_types=[], - enum_types=[ - ], - serialized_options=None, - is_extendable=False, - syntax='proto3', - extension_ranges=[], - oneofs=[ - _descriptor.OneofDescriptor( - name='kind', full_name='vlm_chat.StreamMessage.kind', - index=0, containing_type=None, - create_key=_descriptor._internal_create_key, - fields=[]), - ], - serialized_start=749, - serialized_end=857, -) - -_IMAGE.fields_by_name['encoding'].enum_type = _IMAGE_ENCODING -_IMAGE_ENCODING.containing_type = _IMAGE -_CHATREQUEST.fields_by_name['messages'].message_type = _CHATMESSAGE -_CHATREQUEST.fields_by_name['images'].message_type = _IMAGE -_CHATREQUEST.fields_by_name['sampling'].message_type = _SAMPLINGPARAMS -_STREAMMESSAGE.fields_by_name['delta'].message_type = _CHATSTREAMDELTA -_STREAMMESSAGE.fields_by_name['final'].message_type = _CHATRESPONSE -_STREAMMESSAGE.oneofs_by_name['kind'].fields.append( - _STREAMMESSAGE.fields_by_name['delta']) -_STREAMMESSAGE.fields_by_name['delta'].containing_oneof = _STREAMMESSAGE.oneofs_by_name['kind'] -_STREAMMESSAGE.oneofs_by_name['kind'].fields.append( - _STREAMMESSAGE.fields_by_name['final']) -_STREAMMESSAGE.fields_by_name['final'].containing_oneof = _STREAMMESSAGE.oneofs_by_name['kind'] -DESCRIPTOR.message_types_by_name['Image'] = _IMAGE -DESCRIPTOR.message_types_by_name['ChatMessage'] = _CHATMESSAGE -DESCRIPTOR.message_types_by_name['SamplingParams'] = _SAMPLINGPARAMS -DESCRIPTOR.message_types_by_name['ChatRequest'] = _CHATREQUEST -DESCRIPTOR.message_types_by_name['ChatResponse'] = _CHATRESPONSE -DESCRIPTOR.message_types_by_name['ChatStreamDelta'] = _CHATSTREAMDELTA -DESCRIPTOR.message_types_by_name['StreamMessage'] = _STREAMMESSAGE -_sym_db.RegisterFileDescriptor(DESCRIPTOR) - -Image = _reflection.GeneratedProtocolMessageType('Image', (_message.Message,), { - 'DESCRIPTOR' : _IMAGE, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(Image) - -ChatMessage = _reflection.GeneratedProtocolMessageType('ChatMessage', (_message.Message,), { - 'DESCRIPTOR' : _CHATMESSAGE, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(ChatMessage) - -SamplingParams = _reflection.GeneratedProtocolMessageType('SamplingParams', (_message.Message,), { - 'DESCRIPTOR' : _SAMPLINGPARAMS, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(SamplingParams) - -ChatRequest = _reflection.GeneratedProtocolMessageType('ChatRequest', (_message.Message,), { - 'DESCRIPTOR' : _CHATREQUEST, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(ChatRequest) - -ChatResponse = _reflection.GeneratedProtocolMessageType('ChatResponse', (_message.Message,), { - 'DESCRIPTOR' : _CHATRESPONSE, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(ChatResponse) - -ChatStreamDelta = _reflection.GeneratedProtocolMessageType('ChatStreamDelta', (_message.Message,), { - 'DESCRIPTOR' : _CHATSTREAMDELTA, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(ChatStreamDelta) - -StreamMessage = _reflection.GeneratedProtocolMessageType('StreamMessage', (_message.Message,), { - 'DESCRIPTOR' : _STREAMMESSAGE, - '__module__' : 'vlm_pb2' - - }) -_sym_db.RegisterMessage(StreamMessage) +DESCRIPTOR = _descriptor_pool.Default().AddSerializedFile(b'\n\tvlm.proto\x12\x08vlm_chat\"\x82\x01\n\x05Image\x12*\n\x08\x65ncoding\x18\x01 \x01(\x0e\x32\x18.vlm_chat.Image.Encoding\x12\r\n\x05width\x18\x02 \x01(\r\x12\x0e\n\x06height\x18\x03 \x01(\r\x12\x0c\n\x04\x64\x61ta\x18\x04 \x01(\x0c\" \n\x08\x45ncoding\x12\x08\n\x04JPEG\x10\x00\x12\n\n\x06RGB_U8\x10\x01\",\n\x0b\x43hatMessage\x12\x0c\n\x04role\x18\x01 \x01(\t\x12\x0f\n\x07\x63ontent\x18\x02 \x01(\t\"e\n\x0eSamplingParams\x12\x13\n\x0btemperature\x18\x01 \x01(\x02\x12\r\n\x05top_p\x18\x02 \x01(\x02\x12\r\n\x05top_k\x18\x03 \x01(\x05\x12\x12\n\nmax_tokens\x18\x04 \x01(\x05\x12\x0c\n\x04seed\x18\x05 \x01(\x04\"\xa7\x01\n\x0b\x43hatRequest\x12\'\n\x08messages\x18\x01 \x03(\x0b\x32\x15.vlm_chat.ChatMessage\x12\x1f\n\x06images\x18\x02 \x03(\x0b\x32\x0f.vlm_chat.Image\x12*\n\x08sampling\x18\x03 \x01(\x0b\x32\x18.vlm_chat.SamplingParams\x12\x0e\n\x06stream\x18\x04 \x01(\x08\x12\x12\n\nrequest_id\x18\x05 \x01(\x04\"\xd9\x01\n\x0c\x43hatResponse\x12\x12\n\nrequest_id\x18\x01 \x01(\x04\x12\x0c\n\x04text\x18\x02 \x01(\t\x12\x15\n\rfinish_reason\x18\x03 \x01(\t\x12\x15\n\rprompt_tokens\x18\x04 \x01(\r\x12\x19\n\x11\x63ompletion_tokens\x18\x05 \x01(\r\x12\x18\n\x10latency_ms_total\x18\x06 \x01(\x02\x12\x1a\n\x12latency_ms_prefill\x18\x07 \x01(\x02\x12\x19\n\x11latency_ms_decode\x18\x08 \x01(\x02\x12\r\n\x05\x65rror\x18\t \x01(\t\"4\n\x0f\x43hatStreamDelta\x12\x12\n\nrequest_id\x18\x01 \x01(\x04\x12\r\n\x05\x64\x65lta\x18\x02 \x01(\t\"l\n\rStreamMessage\x12*\n\x05\x64\x65lta\x18\x01 \x01(\x0b\x32\x19.vlm_chat.ChatStreamDeltaH\x00\x12\'\n\x05\x66inal\x18\x02 \x01(\x0b\x32\x16.vlm_chat.ChatResponseH\x00\x42\x06\n\x04kindb\x06proto3') + +_builder.BuildMessageAndEnumDescriptors(DESCRIPTOR, globals()) +_builder.BuildTopDescriptorsAndMessages(DESCRIPTOR, 'vlm_pb2', globals()) +if _descriptor._USE_C_DESCRIPTORS == False: + + DESCRIPTOR._options = None + _IMAGE._serialized_start=24 + _IMAGE._serialized_end=154 + _IMAGE_ENCODING._serialized_start=122 + _IMAGE_ENCODING._serialized_end=154 + _CHATMESSAGE._serialized_start=156 + _CHATMESSAGE._serialized_end=200 + _SAMPLINGPARAMS._serialized_start=202 + _SAMPLINGPARAMS._serialized_end=303 + _CHATREQUEST._serialized_start=306 + _CHATREQUEST._serialized_end=473 + _CHATRESPONSE._serialized_start=476 + _CHATRESPONSE._serialized_end=693 + _CHATSTREAMDELTA._serialized_start=695 + _CHATSTREAMDELTA._serialized_end=747 + _STREAMMESSAGE._serialized_start=749 + _STREAMMESSAGE._serialized_end=857 +# @@protoc_insertion_point(module_scope) diff --git a/src/serving/server.cpp b/src/serving/server.cpp index 3c853dd..8f82688 100644 --- a/src/serving/server.cpp +++ b/src/serving/server.cpp @@ -19,6 +19,8 @@ #define STB_IMAGE_IMPLEMENTATION #define STB_IMAGE_STATIC +#define STBI_ONLY_JPEG +#define STBI_ONLY_PNG // stb ships its full implementation here; silence its unused-function noise so // our own -Wall -Wextra output stays meaningful. #pragma GCC diagnostic push @@ -48,6 +50,7 @@ void on_signal(int) { // Reject absurd image dimensions before any size arithmetic, so an untrusted // width or height cannot overflow size_t or truncate to a negative int. constexpr unsigned kMaxImageDim = 8192; +constexpr size_t kMaxTotalPixels = size_t(64) << 20; bool decode_image(const vla::Image & img, std::vector & u8, @@ -154,19 +157,15 @@ std::string make_error_response(uint64_t request_id, const std::string & msg) { } // Discard frames after the first. Must run to completion: a queued frame keeps -// REP in receive state and send throws EFSM. Stalled means the peer announced a -// frame it never sent, so no reply is possible until the rest arrives. -enum class Drain { Clean, Extra, Stalled }; - -Drain drain_extra_frames(zmq::socket_t & sock) { - Drain d = Drain::Clean; +// REP in receive state and send throws EFSM. +bool drain_extra_frames(zmq::socket_t & sock) { + bool extra = false; while (sock.get(zmq::sockopt::rcvmore)) { zmq::message_t junk; - if (!sock.recv(junk, zmq::recv_flags::none)) - return Drain::Stalled; - d = Drain::Extra; + (void) sock.recv(junk, zmq::recv_flags::none); + extra = true; } - return d; + return extra; } int find_non_finite(const float * data, int n) { @@ -180,13 +179,12 @@ int find_non_finite(const float * data, int n) { void usage(const char * prog) { std::fprintf(stderr, "usage: %s [--bind ADDR] [--timing-detail none|phase] [--config PATH] " - "[] ( | -hf user/repo[:file.gguf])\n" - " vision-tower mmproj GGUF (SigLIP / PaliGemma /\n" - " connector). Required for SmolVLA, π0, Evo-1, GR00T.\n" - " Omit for BitVLA - its vision tower is baked into\n" - " the combined ckpt GGUF.\n" + "([] | -hf user/repo[:file.gguf])\n" + " ignored; every arch bundles its vision tower in the\n" + " ckpt GGUF. Accepted so older command lines still work.\n" " -hf HuggingFace repo, user/repo[:file.gguf]; downloaded\n" - " on a miss and cached under $VLA_CACHE.\n" + " on a miss and cached under $VLA_CACHE. Not combined\n" + " with positional args.\n" " SmolVLA .safetensors or .gguf, or any of the other\n" " supported architectures' .gguf; the architecture is\n" " auto-detected from the checkpoint.\n" @@ -194,11 +192,13 @@ void usage(const char * prog) { " --timing-detail LEVEL per-request timing breakdown (default: none)\n" " 'none' : single ms_inference\n" " 'phase' : ms_prefill + ms_denoise broken out\n" - " (π0 currently reports only the combined ms_inference)\n" + " (only SmolVLA, BitVLA and VLA-JEPA report the split;\n" + " the others report ms_inference only)\n" "%s" - " --config PATH LeRobot policy config.json (SmolVLA safetensors only;\n" - " ignored for GGUF checkpoints). If omitted, uses\n" - " /config.json.\n", + " --config PATH policy config.json. Its \"runtime\" object sets the\n" + " precision flags above for any arch; flags given on\n" + " the command line win. SmolVLA safetensors also read\n" + " the policy from it (default /config.json).\n", prog, vla::Options::usage()); } @@ -261,7 +261,12 @@ int main(int argc, char ** argv) { positionals.push_back(std::move(a)); } } - if (!hf_spec.empty() && positionals.empty()) { + if (!hf_spec.empty()) { + if (!positionals.empty()) { + std::fprintf(stderr, "vla-server: pass or -hf, not both\n"); + usage(argv[0]); + return 1; + } ckpt_path = vla::hf_resolve(hf_spec); if (ckpt_path.empty()) return 1; @@ -272,9 +277,7 @@ int main(int argc, char ** argv) { ckpt_path = positionals[1]; } else { std::fprintf(stderr, - "vla-server: expected -hf, or 1 or 2 positional args " - "( for SmolVLA/π0/Evo-1/GR00T, " - "or just for BitVLA), got %zu\n", + "vla-server: expected -hf or [] , got %zu positional args\n", positionals.size()); usage(argv[0]); return 1; @@ -311,10 +314,12 @@ int main(int argc, char ** argv) { // 64 MiB is above any real request (16 views of 512x512 F32 RGB is ~50 MiB) and // low enough to bound protobuf's expansion during ParseFromArray. sock.set(zmq::sockopt::maxmsgsize, int64_t(64)*1024*1024); - // A peer that sends a frame with SNDMORE and then stalls would otherwise park - // this single-threaded loop in recv for good, starving every other client. - sock.set(zmq::sockopt::rcvtimeo, 5000); - sock.bind(bind_addr); + try { + sock.bind(bind_addr); + } catch (const zmq::error_t & e) { + std::fprintf(stderr, "vla-server: bind %s: %s\n", bind_addr.c_str(), e.what()); + return 1; + } std::printf("vla-server: bound to %s. ready.\n", bind_addr.c_str()); if (bind_addr.find("127.0.0.1") == std::string::npos && @@ -375,13 +380,7 @@ int main(int argc, char ** argv) { // Without this an unauthenticated client shuts the server down with one // two-frame request: the reply fails and send_reply sets g_shutdown. - const Drain drained = drain_extra_frames(sock); - if (drained == Drain::Stalled) { - // Back to the poll rather than blocking here, so shutdown still works. - std::fprintf(stderr, "vla-server: peer stalled mid-request\n"); - continue; - } - if (drained == Drain::Extra) { + if (drain_extra_frames(sock)) { send_reply(make_error_response(0, "expected a single-frame request")); continue; } @@ -457,8 +456,13 @@ int main(int argc, char ** argv) { precomputed_n_views = static_cast(req.precomputed_img_emb_n_views()); const int64_t per_view = cfg.n_img*cfg.hidden; const int64_t expected = per_view * static_cast(precomputed_n_views); - if (precomputed_n_views < 1 || - static_cast(req.precomputed_img_emb_size()) != expected) { + if (precomputed_n_views < 1 || precomputed_n_views > 16) { + char buf[96]; std::snprintf(buf, sizeof(buf), + "precomputed_img_emb_n_views %d out of range [1, 16]", precomputed_n_views); + send_reply(make_error_response(rid, buf)); + continue; + } + if (static_cast(req.precomputed_img_emb_size()) != expected) { char buf[160]; std::snprintf(buf, sizeof(buf), "precomputed_img_emb size %d != %lld (n_views=%d * n_img_per_view=%lld * hidden=%lld)", req.precomputed_img_emb_size(), (long long) expected, precomputed_n_views, @@ -480,6 +484,7 @@ int main(int argc, char ** argv) { } else { bool decode_ok = true; + size_t total_px = 0; for (int v=0; v kMaxTotalPixels) { + send_reply(make_error_response(rid, "images exceed the per-request pixel budget")); + decode_ok = false; + break; + } } if (!decode_ok) continue; diff --git a/src/serving/vla-bench.cpp b/src/serving/vla-bench.cpp index 0b4a603..3da2c27 100644 --- a/src/serving/vla-bench.cpp +++ b/src/serving/vla-bench.cpp @@ -36,6 +36,7 @@ void usage(const char * prog) { " [--label name] [--images N] [--size N] [--tokens N]\n" " [--extra-token ID] [--extra-count N] [--warmup N] [--reps N] [--markdown]\n" " [precision flags]\n" + " --mmproj ignored; every arch bundles its vision tower in the ckpt GGUF\n" " --label row label (default: the checkpoint filename)\n" " --images camera views (default 1)\n" " --size square input side in pixels (default 224)\n" diff --git a/src/serving/vla-cli.cpp b/src/serving/vla-cli.cpp index 95bf099..0ffd351 100644 --- a/src/serving/vla-cli.cpp +++ b/src/serving/vla-cli.cpp @@ -182,6 +182,10 @@ std::string tokenize_text(const std::string & ckpt, const std::string & text) { std::fprintf(stderr, "vla-cli: cannot detect the arch of %s for --text\n", ckpt.c_str()); return ""; } + if (arch == Arch::GR00T_N1_6 || arch == Arch::VLA_JEPA) { + std::fprintf(stderr, "vla-cli: --text is not supported for %s; pass --tokens\n", arch_slug(arch)); + return ""; + } if (!text_ok(text)) { std::fprintf(stderr, "vla-cli: --text takes plain prose (letters, digits, space . , - _ ')\n"); return ""; @@ -249,7 +253,7 @@ void usage(const char * prog) { std::fprintf(stderr, "usage: %s [--mmproj m.gguf] (--ckpt c.gguf | -hf user/repo) --image img.jpg [--image ...]\n" " (--text \"...\" | --tokens id,id,...) [--state f,f,...] [--pretty]\n" - " --mmproj vision-tower GGUF (SmolVLA/pi0/pi0.5); omit for baked-vision archs\n" + " --mmproj ignored; every arch bundles its vision tower in the ckpt GGUF\n" " --ckpt model checkpoint GGUF\n" " -hf HuggingFace repo, user/repo[:file.gguf], cached under $VLA_CACHE\n" " --image image file, repeat for multi-view (decoded via stb_image)\n" diff --git a/src/serving/vlm-server.cpp b/src/serving/vlm-server.cpp index 7a948f4..1adab30 100644 --- a/src/serving/vlm-server.cpp +++ b/src/serving/vlm-server.cpp @@ -15,9 +15,10 @@ #include "vlm/engine.h" #include "serving/vlm.pb.h" -// stbi_info_from_memory only, to preflight JPEG dimensions before mtmd decodes. #define STB_IMAGE_IMPLEMENTATION #define STB_IMAGE_STATIC +#define STBI_ONLY_JPEG +#define STBI_ONLY_PNG #pragma GCC diagnostic push #pragma GCC diagnostic ignored "-Wunused-function" #include "stb_image.h" @@ -43,6 +44,7 @@ void on_signal(int) { // Reject absurd image dimensions before any size arithmetic on untrusted input. constexpr unsigned kMaxImageDim = 8192; +constexpr size_t kMaxTotalPixels = size_t(64) << 20; std::string make_error_stream(uint64_t request_id, const std::string & msg) { vlm_chat::StreamMessage sm; @@ -106,6 +108,18 @@ int main(int argc, char ** argv) { lp.mmproj_path = positionals[0]; lp.model_path = positionals[1]; + zmq::context_t zctx( 1); + zmq::socket_t sock(zctx, zmq::socket_type::router); + sock.set(zmq::sockopt::linger, 0); + // Per frame only. + sock.set(zmq::sockopt::maxmsgsize, int64_t(64)*1024*1024); + try { + sock.bind(bind_addr); + } catch (const zmq::error_t & e) { + std::fprintf(stderr, "vlm-server: bind %s: %s\n", bind_addr.c_str(), e.what()); + return 1; + } + std::printf("vlm-server: loading model ...\n mmproj: %s\n lm: %s\n", lp.mmproj_path.c_str(), lp.model_path.c_str()); @@ -115,16 +129,6 @@ int main(int argc, char ** argv) { return 1; } std::printf("vlm-server: loaded. n_ctx=%d ngl=%d\n", lp.n_ctx, lp.n_gpu_layers); - - zmq::context_t zctx( 1); - zmq::socket_t sock(zctx, zmq::socket_type::router); - sock.set(zmq::sockopt::linger, 0); - // Per frame only; the recv loop caps the multipart total. - sock.set(zmq::sockopt::maxmsgsize, int64_t(64)*1024*1024); - // A peer that sends a frame with SNDMORE and then stalls would otherwise park - // this single-threaded loop in recv for good, starving every other client. - sock.set(zmq::sockopt::rcvtimeo, 5000); - sock.bind(bind_addr); std::printf("vlm-server: bound to %s. ready.\n", bind_addr.c_str()); if (bind_addr.find("127.0.0.1") == std::string::npos && @@ -156,8 +160,7 @@ int main(int argc, char ** argv) { if (!(poll[0].revents & ZMQ_POLLIN)) continue; - // maxmsgsize bounds each frame but not how many, so a peer could stream - // sub-limit frames until memory runs out. + // Caps the envelope copied and echoed back; libzmq has already buffered it. constexpr size_t kMaxEnvFrames = 8; constexpr size_t kMaxEnvBytes = 64*1024; @@ -248,12 +251,13 @@ int main(int argc, char ** argv) { std::vector images; images.reserve(req.images_size()); bool decode_ok = true; + size_t total_px = 0; for (int v=0; v size_t(INT_MAX) || !stbi_info_from_memory(reinterpret_cast(d.data()), @@ -265,12 +269,22 @@ int main(int argc, char ** argv) { send_reply(make_error_stream(rid, buf)); decode_ok = false; break; } - if (!engine.decode_image_buf( - reinterpret_cast(d.data()), d.size(), out)) { + if ((total_px += size_t(jw)*size_t(jh)) > kMaxTotalPixels) { + send_reply(make_error_stream(rid, "images exceed the per-request pixel budget")); + decode_ok = false; break; + } + int w = 0, h = 0, c = 0; + stbi_uc * px = stbi_load_from_memory(reinterpret_cast(d.data()), + static_cast(d.size()), &w, &h, &c, 3); + if (!px) { char buf[64]; std::snprintf(buf, sizeof(buf), "image[%d] JPEG decode failed", v); send_reply(make_error_stream(rid, buf)); decode_ok = false; break; } + out.width = uint32_t(w); + out.height = uint32_t(h); + out.rgb.assign(px, px+size_t(3)*w*h); + stbi_image_free(px); } else if (im.encoding() == vlm_chat::Image::RGB_U8) { if (im.width() == 0 || im.height() == 0 || im.width() > kMaxImageDim || im.height() > kMaxImageDim) { @@ -288,6 +302,10 @@ int main(int argc, char ** argv) { send_reply(make_error_stream(rid, buf)); decode_ok = false; break; } + if ((total_px += size_t(im.width())*im.height()) > kMaxTotalPixels) { + send_reply(make_error_stream(rid, "images exceed the per-request pixel budget")); + decode_ok = false; break; + } out.width = im.width(); out.height = im.height(); out.rgb.assign(im.data().begin(), im.data().end()); diff --git a/src/vlm/engine.cpp b/src/vlm/engine.cpp index ab956ff..8f995a6 100644 --- a/src/vlm/engine.cpp +++ b/src/vlm/engine.cpp @@ -24,6 +24,7 @@ #include #include +#include namespace vlm { @@ -37,6 +38,18 @@ void ensure_global_init() { done = true; } } + +size_t utf8_complete_prefix(const std::string & s) { + for (size_t i=1; i<=4 && i<=s.size(); ++i) { + const unsigned char c = s[s.size()-i]; + if ((c & 0xC0) == 0x80) { + continue; + } + const size_t need = (c & 0xE0) == 0xC0 ? 2 : (c & 0xF0) == 0xE0 ? 3 : (c & 0xF8) == 0xF0 ? 4 : 1; + return need > i ? s.size()-i : s.size(); + } + return s.size(); +} } struct Engine::Impl { @@ -82,6 +95,8 @@ bool Engine::load(const LoadParams & lp) { if (lp.n_threads > 0) { params.cpuparams.n_threads = lp.n_threads; } + postprocess_cpu_params(params.cpuparams, nullptr); + postprocess_cpu_params(params.cpuparams_batch, ¶ms.cpuparams); impl_->llama_init = common_init_from_params(params); impl_->model = impl_->llama_init->model(); @@ -118,37 +133,6 @@ bool Engine::load(const LoadParams & lp) { return true; } -namespace { - -bool bitmap_to_image(mtmd::bitmap & bmp, Image & out) { - if (!bmp.ptr || mtmd_bitmap_is_audio(bmp.ptr.get())) { - return false; - } - out.width = bmp.nx(); - out.height = bmp.ny(); - out.rgb.assign(bmp.data(), bmp.data()+bmp.n_bytes()); - return true; -} -} - -bool Engine::decode_image_file(const std::string & path, Image & out) const { - if (!loaded()) { - return false; - } - mtmd::bitmap bmp(mtmd_helper_bitmap_init_from_file(impl_->vision.get(), path.c_str(), false, - mtmd_helper_init_opt_default()).bitmap); - return bitmap_to_image(bmp, out); -} - -bool Engine::decode_image_buf(const uint8_t * data, size_t len, Image & out) const { - if (!loaded() || !data || len == 0) { - return false; - } - mtmd::bitmap bmp(mtmd_helper_bitmap_init_from_buf(impl_->vision.get(), data, len, false, - mtmd_helper_init_opt_default()).bitmap); - return bitmap_to_image(bmp, out); -} - ChatResult Engine::chat(const std::vector & messages, const std::vector & images, const SamplingParams & sampling, @@ -167,14 +151,13 @@ ChatResult Engine::chat(const std::vector & messages, llama_memory_clear(llama_get_memory(impl_->lctx), true); llama_pos n_past = 0; - std::vector chat_history; common_params_sampling sp; sp.seed = sampling.seed; sp.top_k = sampling.top_k; sp.top_p = sampling.top_p; sp.temp = sampling.temperature; - common_sampler * smpl = common_sampler_init(impl_->model, sp); + common_sampler_ptr smpl(common_sampler_init(impl_->model, sp)); if (!smpl) { res.finish_reason = "error"; res.error = "common_sampler_init failed"; @@ -195,76 +178,82 @@ ChatResult Engine::chat(const std::vector & messages, const char * marker = mtmd_default_marker(); const int64_t t_prefill_start = ggml_time_us(); - for (size_t i=0; iuse_jinja; + ti.add_generation_prompt = messages.back().role != "assistant"; + ti.add_bos = llama_vocab_get_add_bos(impl_->vocab); + ti.add_eos = llama_vocab_get_add_eos(impl_->vocab); + for (const auto & m : messages) { common_chat_msg msg; - msg.role = messages[i].role; - msg.content = messages[i].content; - - std::vector bmps; - if ((int) i == img_msg_idx && !images.empty()) { + msg.role = m.role; + msg.content = m.content; + ti.messages.push_back(std::move(msg)); + } - if (msg.content.find(marker) == std::string::npos) { - std::string prefix; - for (size_t k=0; k bmps; + if (!images.empty()) { + std::string & content = ti.messages[img_msg_idx].content; + if (content.find(marker) == std::string::npos) { + std::string prefix; + for (size_t k=0; ktmpls.get(), chat_history, msg, - msg.role == "user", impl_->use_jinja); - chat_history.push_back(msg); + std::string prompt; + try { + prompt = common_chat_templates_apply(impl_->tmpls.get(), ti).prompt; + } catch (const std::exception & e) { + res.finish_reason = "error"; + res.error = e.what(); + return res; + } - mtmd_input_text text; - text.text = formatted.c_str(); - text.add_special = add_special; - text.parse_special = true; + mtmd_input_text text{}; + text.text = prompt.c_str(); + text.text_len = prompt.size(); + text.add_special = true; + text.parse_special = true; - std::vector bmp_ptrs(bmps.size()); - for (size_t k=0; k bmp_ptrs(bmps.size()); + for (size_t k=0; kvision.get(), chunks.ptr.get(), &text, - bmp_ptrs.data(), bmp_ptrs.size()) != 0) { - res.finish_reason = "error"; - res.error = "mtmd_tokenize failed"; - common_sampler_free(smpl); - return res; - } + mtmd::input_chunks chunks(mtmd_input_chunks_init()); + if (mtmd_tokenize(impl_->vision.get(), chunks.ptr.get(), &text, + bmp_ptrs.data(), bmp_ptrs.size()) != 0) { + res.finish_reason = "error"; + res.error = "mtmd_tokenize failed"; + return res; + } - llama_pos new_n_past = n_past; - if (mtmd_helper_eval_chunks(impl_->vision.get(), impl_->lctx, chunks.ptr.get(), - n_past, 0, impl_->n_batch, - true, &new_n_past) != 0) { - res.finish_reason = "error"; - res.error = "mtmd_helper_eval_chunks failed"; - common_sampler_free(smpl); - return res; - } - n_past = new_n_past; + if (mtmd_helper_eval_chunks(impl_->vision.get(), impl_->lctx, chunks.ptr.get(), + 0, 0, impl_->n_batch, + true, &n_past) != 0) { + res.finish_reason = "error"; + res.error = "mtmd_helper_eval_chunks failed"; + return res; } - res.prompt_tokens = (int32_t) n_past; + res.prompt_tokens = (int32_t) mtmd_helper_get_n_tokens(chunks.ptr.get()); res.ms_prefill = (ggml_time_us()-t_prefill_start)/1000.0f; const int n_predict = sampling.max_tokens <= 0 ? INT_MAX : sampling.max_tokens; std::vector generated; + std::string pending; const int64_t t_decode_start = ggml_time_us(); res.finish_reason = "length"; for (int i=0; ilctx, -1); - common_sampler_accept(smpl, tok, true); + const llama_token tok = common_sampler_sample(smpl.get(), impl_->lctx, -1); + common_sampler_accept(smpl.get(), tok, true); generated.push_back(tok); if (llama_vocab_is_eog(impl_->vocab, tok)) { @@ -274,14 +263,21 @@ ChatResult Engine::chat(const std::vector & messages, const std::string piece = common_token_to_piece(impl_->lctx, tok); res.text += piece; - if (on_token && !on_token(piece)) { + pending += piece; + const size_t n = utf8_complete_prefix(pending); + if (on_token && n > 0 && !on_token(pending.substr(0, n))) { res.finish_reason = "cancelled"; break; } + pending.erase(0, n); common_batch_clear(impl_->batch); common_batch_add(impl_->batch, tok, n_past++, {0}, true); - if (llama_decode(impl_->lctx, impl_->batch) != 0) { + const int rc = llama_decode(impl_->lctx, impl_->batch); + if (rc == 1) { + break; + } + if (rc != 0) { res.finish_reason = "error"; res.error = "llama_decode failed"; break; @@ -293,8 +289,7 @@ ChatResult Engine::chat(const std::vector & messages, if (res.finish_reason != "error") { res.text = common_detokenize(impl_->lctx, generated, false); } - - common_sampler_free(smpl); + res.text.resize(utf8_complete_prefix(res.text)); return res; } diff --git a/src/vlm/engine.h b/src/vlm/engine.h index b8e2bae..9f520f9 100644 --- a/src/vlm/engine.h +++ b/src/vlm/engine.h @@ -123,23 +123,6 @@ class Engine { /// @return @c true once @ref load has succeeded. bool loaded() const; - /** - * @brief Decode an image file (jpg/png/...) into @p out. - * @param path Path to the encoded image. - * @param[out] out Decoded RGB image. - * @return @c true on success; @c false on decode error. - */ - bool decode_image_file(const std::string& path, Image& out) const; - - /** - * @brief Decode an in-memory encoded image into @p out. - * @param data Pointer to the encoded byte stream. - * @param len Length of @p data in bytes. - * @param[out] out Decoded RGB image. - * @return @c true on success; @c false on decode error. - */ - bool decode_image_buf(const uint8_t* data, size_t len, Image& out) const; - /** * @brief Run one chat completion. * From 66ce698e0608147861cc3f5d894d20e50a7de185 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Mon, 28 Sep 2026 14:53:57 +0700 Subject: [PATCH 05/61] fix loader fuse, option precedence and C API races --- bindings/python/vla_cpp/__init__.py | 35 +++++++++++++++----------- src/backend_fallback.cpp | 24 ++++++++++++------ src/loader.cpp | 14 ++++++++--- src/model.cpp | 20 ++++++++++++--- src/options.cpp | 39 +++++++++++++++++++++-------- src/scratch_ctx.h | 1 - src/vla_c_api.cpp | 6 ++++- 7 files changed, 97 insertions(+), 42 deletions(-) diff --git a/bindings/python/vla_cpp/__init__.py b/bindings/python/vla_cpp/__init__.py index 5893866..0ad260c 100644 --- a/bindings/python/vla_cpp/__init__.py +++ b/bindings/python/vla_cpp/__init__.py @@ -11,6 +11,7 @@ from __future__ import annotations import ctypes +import threading from ctypes import POINTER, c_float, c_int32, c_int64 from typing import Sequence @@ -53,6 +54,7 @@ class Model: """A loaded checkpoint. Free it with ``close()`` or a ``with`` block.""" def __init__(self, handle, lib): + self._mu = threading.Lock() self._h = handle self._lib = lib cfg = _ffi.Config() @@ -69,9 +71,10 @@ def __exit__(self, *exc): return False def close(self): - if getattr(self, "_h", None): - self._lib.vla_model_free(self._h) - self._h = None + with self._mu: + if getattr(self, "_h", None): + self._lib.vla_model_free(self._h) + self._h = None def __del__(self): self.close() @@ -83,9 +86,6 @@ def predict(self, images, tokens: Sequence[int], state=None, noise=None, images: one HWC array, or a sequence of them for multi-view. uint8 RGB by default; pass pixel_format=PIXEL_F32_RGB_01 for float RGB in [0, 1]. """ - if self._h is None: - raise RuntimeError("model is closed") - views = images if isinstance(images, (list, tuple)) else [images] if not views: raise ValueError("at least one image is required") @@ -99,6 +99,9 @@ def predict(self, images, tokens: Sequence[int], state=None, noise=None, shape = mv.shape if len(shape) != 3 or shape[2] != 3: raise ValueError(f"image must be HxWx3, got {shape}") + want = "f" if pixel_format == PIXEL_F32_RGB_01 else "B" + if mv.format != want: + raise ValueError(f"image format {mv.format!r} does not match pixel_format (expected {want!r})") raw = (ctypes.c_char * mv.nbytes).from_buffer_copy(mv) keep.append(raw) img_array[i].data = ctypes.cast(raw, ctypes.c_void_p) @@ -133,13 +136,16 @@ def predict(self, images, tokens: Sequence[int], state=None, noise=None, out = POINTER(c_float)() n = c_int64() - rc = self._lib.vla_predict(self._h, ctypes.byref(cin), ctypes.byref(out), ctypes.byref(n)) - if rc != _ffi.OK: - raise RuntimeError(f"vla_predict failed ({rc})") - try: - flat = [out[i] for i in range(n.value)] - finally: - self._lib.vla_free_actions(out) + with self._mu: + if self._h is None: + raise RuntimeError("model is closed") + rc = self._lib.vla_predict(self._h, ctypes.byref(cin), ctypes.byref(out), ctypes.byref(n)) + if rc != _ffi.OK: + raise RuntimeError(f"vla_predict failed ({rc})") + try: + flat = [out[i] for i in range(n.value)] + finally: + self._lib.vla_free_actions(out) cols = int(self.config.max_action_dim) or 1 rows = len(flat) // cols if cols else len(flat) @@ -151,7 +157,8 @@ def predict(self, images, tokens: Sequence[int], state=None, noise=None, def last_stats(self) -> _ffi.Stats: st = _ffi.Stats() - rc = self._lib.vla_last_stats(self._h, ctypes.byref(st)) + with self._mu: + rc = self._lib.vla_last_stats(self._h, ctypes.byref(st)) if rc != _ffi.OK: raise RuntimeError(f"vla_last_stats failed ({rc})") return st diff --git a/src/backend_fallback.cpp b/src/backend_fallback.cpp index 21d7b9a..7136bc7 100644 --- a/src/backend_fallback.cpp +++ b/src/backend_fallback.cpp @@ -45,9 +45,10 @@ namespace { enum class Where : uint8_t { Accel, Cpu, Either }; struct FallbackCtx { - ggml_backend_t accel = nullptr; - ggml_backend_t cpu = nullptr; - std::string name; + ggml_backend_t accel = nullptr; + ggml_backend_t cpu = nullptr; + ggml_threadpool_t tp = nullptr; + std::string name; // Host copies of weights some CPU-side op reads, made once. Keyed by tensor: // the wrapper lives exactly as long as the model that owns the weights. @@ -137,14 +138,15 @@ ggml_tensor * host_input(FallbackCtx * fc, ggml_context * meta, if (is_host(t)) { s->data = t->data; } else if (is_weight(t)) { - std::vector & w = fc->weights[t]; + const ggml_tensor * base = t->view_src ? t->view_src : t; + std::vector & w = fc->weights[base]; if (w.empty()) { // get_tensor undoes the accelerator's repacking, so this is plain ggml layout. - w.resize(n); - ggml_backend_tensor_get(t, w.data(), 0, n); - fc->bytes_in += n; + w.resize(ggml_nbytes(base)); + ggml_backend_tensor_get(base, w.data(), 0, w.size()); + fc->bytes_in += w.size(); } - s->data = w.data(); + s->data = w.data() + (t->view_src ? t->view_offs : 0); } else { s->data = take(fc, n); ggml_backend_tensor_get(t, s->data, 0, n); @@ -258,6 +260,7 @@ void fb_free(ggml_backend_t be) { } ggml_backend_free(fc->accel); ggml_backend_free(fc->cpu); + ggml_threadpool_free(fc->tp); delete fc; delete be; } @@ -418,6 +421,11 @@ ggml_backend_t fallback_backend_new(ggml_backend_t accel, int n_threads) { auto * fc = new FallbackCtx; fc->accel = accel; fc->cpu = cpu; + ggml_threadpool_params tpp = ggml_threadpool_params_default(n_threads); + tpp.poll = 0; + fc->tp = ggml_threadpool_new(&tpp); + if (fc->tp) + ggml_backend_cpu_set_threadpool(cpu, fc->tp); fc->name = std::string(ggml_backend_name(accel)) + "+CPU"; const char * st = std::getenv("VLA_FALLBACK_STATS"); fc->stats = st && st[0] == '1'; diff --git a/src/loader.cpp b/src/loader.cpp index 0137731..340fb2c 100644 --- a/src/loader.cpp +++ b/src/loader.cpp @@ -106,8 +106,9 @@ ggml_tensor * WeightLoader::fuse(ggml_type want, const char * out_name, const st return nullptr; } - const bool is1d = ggml_n_dims(first) == 1; - int64_t rows = 0; + const ggml_type rt = g_.resident_type(first, want); + const bool is1d = ggml_n_dims(first) == 1; + int64_t rows = 0; for (const std::string & s : srcs) { const ggml_tensor * gs = g_.meta(s.c_str()); if (!gs) { @@ -115,11 +116,16 @@ ggml_tensor * WeightLoader::fuse(ggml_type want, const char * out_name, const st ok_ = false; return nullptr; } + if (g_.resident_type(gs, want) != rt || (!is1d && gs->ne[0] != first->ne[0])) { + std::fprintf(stderr, "vla(%s): %s does not match %s for fusing\n", arch_, s.c_str(), srcs[0].c_str()); + ok_ = false; + return nullptr; + } rows += is1d ? gs->ne[0] : gs->ne[1]; } - ggml_tensor * t = is1d ? ggml_new_tensor_1d(ctx_, want, rows) - : ggml_new_tensor_2d(ctx_, want, first->ne[0], rows); + ggml_tensor * t = is1d ? ggml_new_tensor_1d(ctx_, rt, rows) + : ggml_new_tensor_2d(ctx_, rt, first->ne[0], rows); if (!t) { std::fprintf(stderr, "vla(%s): ggml_new_tensor failed for %s\n", arch_, out_name); ok_ = false; diff --git a/src/model.cpp b/src/model.cpp index 6bf4f55..f6e18c9 100644 --- a/src/model.cpp +++ b/src/model.cpp @@ -30,6 +30,8 @@ namespace vla { struct Model { std::unique_ptr impl; + bool fa = false; + bool mm = true; }; namespace { @@ -209,7 +211,13 @@ bool detect_arch_from_ckpt(const std::string& ckpt_path, Arch* out) { Model* model_load(const std::string& mmproj_path, const std::string& ckpt_path, const std::string& config_path) { - return model_load(mmproj_path, ckpt_path, config_path, Options{}); + Options o; + std::string err; + if (!o.load_json(config_path, err)) { + std::fprintf(stderr, "vla: %s\n", err.c_str()); + return nullptr; + } + return model_load(mmproj_path, ckpt_path, config_path, o); } Model* model_load(const std::string& mmproj_path, const std::string& ckpt_path, @@ -232,8 +240,10 @@ Model* model_load(const std::string& mmproj_path, const std::string& ckpt_path, } } - set_flash_attn(opts.flash_attn.value_or(default_flash_attn())); - set_mm_prec_f32(opts.mm_prec_f32.value_or(true)); + const bool fa = opts.flash_attn.value_or(default_flash_attn()); + const bool mm = opts.mm_prec_f32.value_or(true); + set_flash_attn(fa); + set_mm_prec_f32(mm); switch (arch) { case Arch::SMOLVLA: @@ -300,6 +310,8 @@ Model* model_load(const std::string& mmproj_path, const std::string& ckpt_path, auto* m = new Model(); m->impl = std::move(impl); + m->fa = fa; + m->mm = mm; return m; } @@ -317,6 +329,8 @@ const Stats& last_stats(const Model* m) { std::vector predict(Model* m, const Inputs& in) { if (!m || !m->impl) return {}; + set_flash_attn(m->fa); + set_mm_prec_f32(m->mm); return m->impl->predict(in); } diff --git a/src/options.cpp b/src/options.cpp index f3614c4..5157727 100644 --- a/src/options.cpp +++ b/src/options.cpp @@ -61,8 +61,8 @@ const char * dtype_name(ggml_type t) { } namespace { -bool g_flash_attn = false; -bool g_mm_prec_f32 = true; +thread_local bool g_flash_attn = false; +thread_local bool g_mm_prec_f32 = true; } void set_flash_attn(bool on) { @@ -199,15 +199,32 @@ bool Options::load_json(const std::string & path, std::string & err) { const nlohmann::json & r = j["runtime"]; try { - ggml_type t; - if (r.contains("weight_dtype") && parse_dtype(r["weight_dtype"].get(), t)) - weight_dtype = t; - if (r.contains("act_dtype") && parse_dtype(r["act_dtype"].get(), t)) - act_dtype = t; - if (r.contains("flash_attn")) - flash_attn = r["flash_attn"].get(); - if (r.contains("mm_prec")) - mm_prec_f32 = r["mm_prec"].get() == "f32"; + auto dtype = [&](const char * k, std::optional & dst, bool allow_f16) { + if (!r.contains(k)) + return true; + const std::string v = r[k].get(); + ggml_type t; + if (!parse_dtype(v, t) || (!allow_f16 && t == GGML_TYPE_F16)) { + err = std::string("config json runtime.")+k+": bad value '"+v+"'"; + return false; + } + if (!dst) + dst = t; + return true; + }; + if (!dtype("weight_dtype", weight_dtype, true) || !dtype("act_dtype", act_dtype, false)) + return false; + if (r.contains("flash_attn") && !flash_attn) + flash_attn = r["flash_attn"].get(); + if (r.contains("mm_prec")) { + const std::string v = r["mm_prec"].get(); + if (v != "default" && v != "f32") { + err = "config json runtime.mm_prec: expected default or f32, got '"+v+"'"; + return false; + } + if (!mm_prec_f32) + mm_prec_f32 = v == "f32"; + } } catch (const std::exception & e) { err = std::string("config json runtime: ")+e.what(); return false; diff --git a/src/scratch_ctx.h b/src/scratch_ctx.h index 4e6a67e..8e3d100 100644 --- a/src/scratch_ctx.h +++ b/src/scratch_ctx.h @@ -43,7 +43,6 @@ class scratch_ctx { ggml_context * reset(size_t arena) { // Growing matters: an arena sized from the input shape would otherwise // keep the first call's smaller pool and abort in ggml_new_tensor. - // Every call site passes a constant today. if (ctx_ && arena > arena_) { ggml_free(ctx_); ctx_ = nullptr; diff --git a/src/vla_c_api.cpp b/src/vla_c_api.cpp index 49572a0..fa9e781 100644 --- a/src/vla_c_api.cpp +++ b/src/vla_c_api.cpp @@ -21,13 +21,15 @@ #include #include #include +#include #include namespace { // vla_model is opaque to C, so it can just wrap the C++ handle. struct vla_model_impl { - vla::Model * m = nullptr; + vla::Model * m = nullptr; + mutable std::mutex mu; }; vla::PixelFormat to_pixel_format(int32_t f) { @@ -129,6 +131,7 @@ int32_t vla_predict(vla_model * h, const vla_inputs * in, return VLA_ERR_ARG; try { + std::lock_guard lk(h->mu); std::vector views((size_t) (in->n_images > 0 ? in->n_images : 0)); for (size_t i=0; iimages[i].data, @@ -177,6 +180,7 @@ int32_t vla_last_stats(const vla_model * h, vla_stats * out) { if (!h || !h->m || !out) return VLA_ERR_ARG; try { + std::lock_guard lk(h->mu); const vla::Stats & s = vla::last_stats(h->m); out->ms_total = s.ms_total; out->ms_vision = s.ms_vision; From 3a07dfcec0d135f3f248baf0a447a682e1120437 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Mon, 28 Sep 2026 14:53:57 +0700 Subject: [PATCH 06/61] size graphs from node counts, validate pi0/smolvla/oft input --- src/models/openvla_oft.cpp | 11 +++++++++-- src/models/pi0.cpp | 21 +++++++++++++++++++-- src/models/pi05.cpp | 21 +++++++++++++++++++-- src/models/smolvla.cpp | 33 ++++++++++++++++++++++++++++----- src/models/vla_adapter.cpp | 6 ++++-- 5 files changed, 79 insertions(+), 13 deletions(-) diff --git a/src/models/openvla_oft.cpp b/src/models/openvla_oft.cpp index 4dc9140..583b68b 100644 --- a/src/models/openvla_oft.cpp +++ b/src/models/openvla_oft.cpp @@ -188,6 +188,11 @@ std::unique_ptr openvla_oft_create(const std::string& mmproj_path // No empty_id: the reference zeroes the action-slot embeddings instead // (modeling_prismatic.py:891), which is what act0 below does. U("openvla_oft.tokens.stop_id",m->stop_id); + if (m->patch_size <= 0 || m->n_q <= 0) { + std::fprintf(stderr, "vla(openvla_oft): bad geometry (patch_size %lld q_heads %lld)\n", + (long long) m->patch_size, (long long) m->n_q); + return nullptr; + } if (m->lm_head_dim==0) m->lm_head_dim = m->lm_hidden/m->n_q; @@ -261,6 +266,7 @@ std::vector OpenVlaOftModelArch::predict(const Inputs& in) { const int64_t S=image_size, NP=n_patches, HC=lm_hidden; const int64_t n_views = in.n_images; if (in.precomputed_img_emb) { std::fprintf(stderr, "vla(openvla_oft): precomputed_img_emb is not supported; the DINOv2+SigLIP tower is baked into the GGUF, pass raw images\n"); return {}; } + if (in.n_lang < 1 || !in.lang_tokens) { std::fprintf(stderr, "vla(openvla_oft): need >=1 lang token\n"); return {}; } if (n_views < 1) { std::fprintf(stderr, "vla(openvla_oft): need >=1 image view\n"); return {}; } if (!in.images) { std::fprintf(stderr, "vla(openvla_oft): n_images=%d but the images pointer is null\n", in.n_images); return {}; } // towers read S*S*3 per view; reject any view that is not exactly SxS. @@ -281,7 +287,8 @@ std::vector OpenVlaOftModelArch::predict(const Inputs& in) { std::vector proj_host((size_t)HC*NPATCH); { const auto tv=clock::now(); - ggml_context*C=vision_scratch.reset((size_t)64*1024*1024); + const size_t max_nodes=(size_t)64*(d_layers+s_layers+1)*n_views+1024; + ggml_context*C=vision_scratch.reset(ggml_tensor_overhead()*max_nodes+ggml_graph_overhead_custom(max_nodes,false)); std::vector px_d(n_views), px_s(n_views), cmb(n_views); for(int v=0; v OpenVlaOftModelArch::predict(const Inputs& in) { ggml_tensor*ph=ggml_add(C,ggml_mul_mat(C,vis.pj_fc1w,allp),vis.pj_fc1b); ph=ggml_gelu_erf(C,ph); ph=ggml_add(C,ggml_mul_mat(C,vis.pj_fc2w,ph),vis.pj_fc2b); ph=ggml_gelu_erf(C,ph); ggml_tensor*proj=ggml_add(C,ggml_mul_mat(C,vis.pj_fc3w,ph),vis.pj_fc3b); ggml_set_output(proj); - ggml_cgraph*vg=ggml_new_graph_custom(C,16384,false); ggml_build_forward_expand(vg,proj); + ggml_cgraph*vg=ggml_new_graph_custom(C,max_nodes,false); ggml_build_forward_expand(vg,proj); if(!vision_scratch.alloc(backend,vg)){ std::fprintf(stderr,"vla(openvla_oft): vision gallocr failed\n"); return {}; } std::vector dbuf, sbuf; for(int v=0;v 1000) { + std::fprintf(stderr, "vla(pi0): num_steps %d out of range [1, 1000]\n", cfg.num_steps); + return false; + } cfg.n_state = 1; cfg.n_img = 256; @@ -411,6 +415,13 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, vu("pi0.patch_size", m->vit_patch_size); vu("pi0.n_img_tokens", m->vit_n_tokens); if (g.has("pi0.vit_ln_eps")) m->vit_ln_eps = g.f32("pi0.vit_ln_eps"); + if (m->vit_patch_size <= 0 || m->vit_heads <= 0 || m->vit_hidden % m->vit_heads || + m->vit_image_size % m->vit_patch_size) { + std::fprintf(stderr, "vla(pi0): bad vit geometry (image %lld patch %lld hidden %lld heads %lld)\n", + (long long) m->vit_image_size, (long long) m->vit_patch_size, + (long long) m->vit_hidden, (long long) m->vit_heads); + return nullptr; + } const int64_t grid = m->vit_image_size/m->vit_patch_size; if (grid * grid != m->vit_n_tokens || m->vit_n_tokens != cfg.n_img) { std::fprintf(stderr, "vla(pi0): vit geometry mismatch (grid^2=%lld n_img_tokens=%lld cfg.n_img=%lld)\n", @@ -474,6 +485,10 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { std::vector img_emb_host; int64_t n_img_tokens = 0; if (in.precomputed_img_emb) { + if (in.n_img_views < 1) { + std::fprintf(stderr, "vla(pi0): precomputed_img_emb set but n_img_views=%d\n", in.n_img_views); + return {}; + } n_img_tokens = (int64_t) in.n_img_views*cfg.n_img; img_emb_host.assign(in.precomputed_img_emb, in.precomputed_img_emb+(size_t) n_img_tokens * hidden_pl); @@ -542,7 +557,9 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { // Prefix + expert graph depends only on the token counts and step count. const MainKey mkey{ n_img_tokens, n_lang, num_steps }; - const bool built = main_graph.ensure(backend, mkey, (size_t) 64*1024*1024, + const size_t max_nodes = (size_t) 64*n_layers*(num_steps+1) + 1024; + const bool built = main_graph.ensure(backend, mkey, + ggml_tensor_overhead()*max_nodes + ggml_graph_overhead_custom(max_nodes, false), [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { ggml_tensor * t_image_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_img_tokens); ggml_set_input(t_image_emb); ggml_tensor * t_lang_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_lang); ggml_set_input(t_lang_emb); @@ -600,7 +617,7 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { gio.t_state=t_state; gio.t_x0=t_x0; gio.t_suffix_pos=t_suffix_pos; gio.t_full_mask=t_full_mask; gio.t_time=t_time; gio.x_final=x_final; - ggml_cgraph * gf = ggml_new_graph_custom(C, 16384, false); + ggml_cgraph * gf = ggml_new_graph_custom(C, max_nodes, false); ggml_build_forward_expand(gf, x_final); return gf; }); diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index 1e348d8..7f1e1a0 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -315,6 +315,10 @@ bool load_config(const gguf_reader & g, Config & cfg) { cfg.n_lang = g.u32("pi05.tokenizer_max_length"); cfg.min_period = g.f64("pi05.min_period"); cfg.max_period = g.f64("pi05.max_period"); + if (cfg.num_steps < 1 || cfg.num_steps > 1000) { + std::fprintf(stderr, "vla(pi05): num_steps %d out of range [1, 1000]\n", cfg.num_steps); + return false; + } cfg.n_state = 0; cfg.n_img = 256; @@ -442,6 +446,13 @@ std::unique_ptr pi05_create(const std::string& mmproj_path, vu("pi05.patch_size", m->vit_patch_size); vu("pi05.n_img_tokens", m->vit_n_tokens); if (g.has("pi05.vit_ln_eps")) m->vit_ln_eps = g.f32("pi05.vit_ln_eps"); + if (m->vit_patch_size <= 0 || m->vit_heads <= 0 || m->vit_hidden % m->vit_heads || + m->vit_image_size % m->vit_patch_size) { + std::fprintf(stderr, "vla(pi05): bad vit geometry (image %lld patch %lld hidden %lld heads %lld)\n", + (long long) m->vit_image_size, (long long) m->vit_patch_size, + (long long) m->vit_hidden, (long long) m->vit_heads); + return nullptr; + } const int64_t grid = m->vit_image_size/m->vit_patch_size; if (grid * grid != m->vit_n_tokens || m->vit_n_tokens != cfg.n_img) { std::fprintf(stderr, "vla(pi05): vit geometry mismatch (grid^2=%lld n_img_tokens=%lld cfg.n_img=%lld)\n", @@ -520,6 +531,10 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { std::vector img_emb_host; int64_t n_img_tokens = 0; if (in.precomputed_img_emb) { + if (in.n_img_views < 1) { + std::fprintf(stderr, "vla(pi05): precomputed_img_emb set but n_img_views=%d\n", in.n_img_views); + return {}; + } n_img_tokens = (int64_t) in.n_img_views*cfg.n_img; img_emb_host.assign(in.precomputed_img_emb, in.precomputed_img_emb+(size_t) n_img_tokens * hidden_pl); @@ -592,7 +607,9 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { // Prefix + expert graph depends only on the token counts and step count. const MainKey mkey{ n_img_tokens, n_lang, num_steps }; - const bool built = main_graph.ensure(backend, mkey, (size_t) 64*1024*1024, + const size_t max_nodes = (size_t) 64*n_layers*(num_steps+1) + 1024; + const bool built = main_graph.ensure(backend, mkey, + ggml_tensor_overhead()*max_nodes + ggml_graph_overhead_custom(max_nodes, false), [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { ggml_tensor * t_image_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_img_tokens); ggml_set_input(t_image_emb); ggml_tensor * t_lang_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_lang); ggml_set_input(t_lang_emb); @@ -641,7 +658,7 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { gio.t_image_emb=t_image_emb; gio.t_lang_emb=t_lang_emb; gio.t_prefix_pos=t_prefix_pos; gio.t_x0=t_x0; gio.t_suffix_pos=t_suffix_pos; gio.t_time=t_time; gio.x_final=x_final; - ggml_cgraph * gf = ggml_new_graph_custom(C, 16384, false); + ggml_cgraph * gf = ggml_new_graph_custom(C, max_nodes, false); ggml_build_forward_expand(gf, x_final); return gf; }); diff --git a/src/models/smolvla.cpp b/src/models/smolvla.cpp index 51f6eef..4e1a10d 100644 --- a/src/models/smolvla.cpp +++ b/src/models/smolvla.cpp @@ -471,6 +471,12 @@ bool load_config_from_json(const std::string & path, Config & cfg) { std::fprintf(stderr, "vla: failed to parse %s: %s\n", path.c_str(), e.what()); return false; } + for (const char * k : {"adapt_to_pi_aloha", "add_image_special_tokens"}) { + if (j.contains(k) && j[k].is_boolean() && j[k].get()) { + std::fprintf(stderr, "vla(smolvla): %s=true in %s is not supported\n", k, path.c_str()); + return false; + } + } cfg.hidden = 960; cfg.n_q_heads = 15; @@ -998,6 +1004,11 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, } std::printf("vla: config = %s\n", cfg_path.c_str()); } + if (m->cfg.num_steps < 1 || m->cfg.num_steps > 1000) { + std::fprintf(stderr, "vla(smolvla): num_steps %d out of range [1, 1000]\n", m->cfg.num_steps); + delete m; + return nullptr; + } { const Backend b = backend_init("vla", default_cpu_threads()); @@ -1024,6 +1035,15 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, m->vit_ln_eps = gst.get_f32("smolvla.vit_ln_eps"); } { + if (m->vit_patch <= 0 || m->vit_scale <= 0 || m->vit_heads <= 0 || + m->vit_image % m->vit_patch || (m->vit_image/m->vit_patch) % m->vit_scale || + m->vit_hidden % m->vit_heads) { + std::fprintf(stderr, "vla(smolvla): bad vit geometry (image %lld patch %lld shuffle %lld hidden %lld heads %lld)\n", + (long long) m->vit_image, (long long) m->vit_patch, (long long) m->vit_scale, + (long long) m->vit_hidden, (long long) m->vit_heads); + delete m; + return nullptr; + } const int64_t grid = m->vit_image/m->vit_patch; const int64_t k = grid/m->vit_scale; if (k * k != m->vit_n_tokens) { @@ -1210,7 +1230,7 @@ SmolVLAModelArch* smolvla_load_impl(ggml_type weight_dtype, m->expert_layers.resize(cfg.n_layers); for (int i=0; iexpert_layers[i]; - w.is_self_attn = (i%cfg.self_attn_every_n == 0); + w.is_self_attn = cfg.self_attn_every_n > 0 && i%cfg.self_attn_every_n == 0; w.Wln_in = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, cfg.expert_h); w.Wq = ggml_new_tensor_2d(ctx, wdt, cfg.expert_h, cfg.q_full_dim); if (w.is_self_attn) { @@ -1389,8 +1409,9 @@ bool build_compute_graph(SmolVLAModelArch* m, int n_views) { cfg.n_img = cfg_model.n_img*int64_t(n_views); + const size_t max_nodes = size_t(64)*cfg.n_layers*(cfg.num_steps+1) + 1024; ggml_init_params gparams = { - size_t(64)*1024*1024, + ggml_tensor_overhead()*max_nodes + ggml_graph_overhead_custom(max_nodes, false), nullptr, true, }; @@ -1488,7 +1509,7 @@ bool build_compute_graph(SmolVLAModelArch* m, int n_views) { ggml_set_output(x_t); - ggml_cgraph * gf = ggml_new_graph_custom(ctx, 16384, false); + ggml_cgraph * gf = ggml_new_graph_custom(ctx, max_nodes, false); ggml_build_forward_expand(gf, x_t); ggml_backend_buffer_type_t buft = ggml_backend_get_default_buffer_type(m->backend); @@ -1780,8 +1801,10 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { return out; } + const size_t max_nodes = size_t(64)*cfg.n_layers*(cfg.num_steps+1) + 1024; ggml_init_params gparams = { - size_t(64)*1024*1024, + ggml_tensor_overhead()*max_nodes + ggml_graph_overhead_custom(max_nodes, false) + + ggml_graph_overhead_custom(4096, false), nullptr, true, }; @@ -1995,7 +2018,7 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { } { - ggml_cgraph * gf = ggml_new_graph_custom(ctx, 16384, false); + ggml_cgraph * gf = ggml_new_graph_custom(ctx, max_nodes, false); ggml_build_forward_expand(gf, x_t); graph_unique_names(gf); const auto t0 = clock::now(); diff --git a/src/models/vla_adapter.cpp b/src/models/vla_adapter.cpp index 016241a..95159ee 100644 --- a/src/models/vla_adapter.cpp +++ b/src/models/vla_adapter.cpp @@ -304,6 +304,7 @@ std::vector VlaAdapterModelArch::predict(const Inputs& in) { const int64_t S=image_size, NP=n_patches, HC=lm_hidden, HD=head_dim, NH=head_heads; const int64_t n_views = in.n_images; if (in.precomputed_img_emb) { std::fprintf(stderr, "vla(vla_adapter): precomputed_img_emb is not supported; the DINOv2+SigLIP tower is baked into the GGUF, pass raw images\n"); return {}; } + if (in.n_lang < 1 || !in.lang_tokens) { std::fprintf(stderr, "vla(vla_adapter): need >=1 lang token\n"); return {}; } if (n_views < 1) { std::fprintf(stderr, "vla(vla_adapter): need >=1 image view\n"); return {}; } if (!in.images) { std::fprintf(stderr, "vla(vla_adapter): n_images=%d but the images pointer is null\n", in.n_images); return {}; } // towers read S*S*3 per view; reject any view that is not exactly SxS. @@ -323,7 +324,8 @@ std::vector VlaAdapterModelArch::predict(const Inputs& in) { std::vector proj_host((size_t)HC*NP*n_views); { const auto tv=clock::now(); - ggml_context*C=vision_scratch.reset((size_t)64*1024*1024); + const size_t max_nodes=(size_t)64*(d_layers+s_layers+1)*n_views+1024; + ggml_context*C=vision_scratch.reset(ggml_tensor_overhead()*max_nodes+ggml_graph_overhead_custom(max_nodes,false)); std::vector px_d(n_views), px_s(n_views); std::vector cmb(n_views); for(int v=0; v VlaAdapterModelArch::predict(const Inputs& in) { ggml_tensor*ph=ggml_add(C,ggml_mul_mat(C,vis.pj_fc1w,allp),vis.pj_fc1b); ph=ggml_gelu_erf(C,ph); ph=ggml_add(C,ggml_mul_mat(C,vis.pj_fc2w,ph),vis.pj_fc2b); ph=ggml_gelu_erf(C,ph); ggml_tensor*proj=ggml_add(C,ggml_mul_mat(C,vis.pj_fc3w,ph),vis.pj_fc3b); ggml_set_output(proj); - ggml_cgraph*vg=ggml_new_graph_custom(C,16384,false); ggml_build_forward_expand(vg,proj); + ggml_cgraph*vg=ggml_new_graph_custom(C,max_nodes,false); ggml_build_forward_expand(vg,proj); if(!vision_scratch.alloc(backend,vg)){ std::fprintf(stderr,"vla(vla_adapter): vision gallocr failed\n"); return {}; } std::vector dbuf, sbuf; for(int v=0;v Date: Mon, 28 Sep 2026 14:53:57 +0700 Subject: [PATCH 07/61] validate GR00T, Evo-1, VLA-JEPA and TurboVLA metadata --- docs/backend/hexagon-windows.md | 2 +- src/models/evo1.cpp | 14 +++++++++++--- src/models/gr00tn1d5.cpp | 9 ++++++++- src/models/gr00tn1d6.cpp | 7 ++++++- src/models/gr00tn1d7.cpp | 19 ++++++++++++++----- src/models/turbovla.cpp | 14 ++++++++++++-- src/models/vla_jepa.cpp | 21 +++++++++++++++++---- src/modules/qwen3vl_vit.h | 5 ++++- 8 files changed, 73 insertions(+), 18 deletions(-) diff --git a/docs/backend/hexagon-windows.md b/docs/backend/hexagon-windows.md index 1a6d86d..254dfa3 100644 --- a/docs/backend/hexagon-windows.md +++ b/docs/backend/hexagon-windows.md @@ -290,7 +290,7 @@ count on top of the host copy made during loading. | GR00T N1.6 | 9.16 → 7.99 GB | 4,930 ms, 7.9e-3 | 3,473 ms, 8.5e-3 | 6,509 ms, 6.6e-3 | | VLA-JEPA | 4.57 → 3.24 GB | 745 ms, 2.5e-2 | 1,477 ms, 4.9e-2 | 2,979 ms, 5.2e-2 | -SmolVLA's CPU Q8_0 run keeps its float weights at F16; the other rows keep the CPU default, BF16. The π0 Q8_0 result on the GPU (0.27) is far worse than the same file on the CPU, and was not investigated. Evo-1's Q8_0 file fails to load on every backend, including the CPU, with a `ggml_view` assertion; that is an arch bug, not a Snapdragon one. +SmolVLA's CPU Q8_0 run keeps its float weights at F16; the other rows keep the CPU default, BF16. The π0 Q8_0 result on the GPU (0.27) is far worse than the same file on the CPU, and was not investigated. Evo-1's Q8_0 file failed to load on every backend with a `ggml_view` assertion; it now loads on the CPU and the NPU. The Adreno GPU refuses one made by the old quantizer, because ggml-opencl ignores view offsets on quantized weights; requantize with the current `scripts/quantize_gguf.py`, which keeps the action expert float. ## Observations diff --git a/src/models/evo1.cpp b/src/models/evo1.cpp index 890e36b..7fdcb14 100644 --- a/src/models/evo1.cpp +++ b/src/models/evo1.cpp @@ -282,10 +282,10 @@ ggml_tensor * build_internvit_view(ggml_context * C, const Evo1ModelArch & m, gg } ggml_tensor * inproj_split_w(ggml_context * C, ggml_tensor * Win, int64_t E, int64_t k) { - return ggml_cont(C, ggml_view_2d(C, Win, E, E, Win->nb[1], (size_t) k * E * E * ggml_element_size(Win))); + return ggml_view_2d(C, Win, Win->ne[0], E, Win->nb[1], (size_t) k * E * Win->nb[1]); } ggml_tensor * inproj_split_b(ggml_context * C, ggml_tensor * bin, int64_t E, int64_t k) { - return ggml_cont(C, ggml_view_1d(C, bin, E, (size_t) k * E * ggml_element_size(bin))); + return ggml_view_1d(C, bin, E, (size_t) k * E * bin->nb[0]); } bool load_config(const gguf_reader & g, Evo1ModelArch & m, Config & cfg) { @@ -448,6 +448,14 @@ std::unique_ptr evo1_create(const std::string& mmproj_path, w.f1w = mk_mm(N("ff1.weight")); w.f1b = mk_f32(N("ff1.bias")); w.f2w = mk_mm(N("ff2.weight")); w.f2b = mk_f32(N("ff2.bias")); ok &= w.n1w && w.n1b && w.n2w && w.n2b && w.Win && w.bin && w.Wo && w.bo && w.f1w && w.f1b && w.f2w && w.f2b; +#ifdef GGML_USE_OPENCL + if (ok && ggml_is_quantized(w.Win->type) && + std::strcmp(ggml_backend_reg_name(ggml_backend_dev_backend_reg(ggml_backend_get_device(m->backend))), "OpenCL") == 0) { + std::fprintf(stderr, "vla(evo1): %s is %s; OpenCL cannot split a quantized attn_in, requantize with attn_in kept float\n", + N("attn_in.weight"), ggml_type_name(w.Win->type)); + return nullptr; + } +#endif } m->norm_out_w = mk_f32("aex.norm_out.weight"); m->norm_out_b = mk_f32("aex.norm_out.bias"); m->seq_pool_w = mk_mm("aex.seq_pool.weight"); m->seq_pool_b = mk_f32("aex.seq_pool.bias"); @@ -793,7 +801,7 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { ggml_backend_tensor_set(t_pos, pp.data(), 0, ggml_nbytes(t_pos)); } { std::vector mk((size_t) SEQ * SEQ); const float NEG = -std::numeric_limits::infinity(); - for (int64_t q=0; q Gr00tN1d7ModelArch::predict(const Inputs& in) { ggml_tensor * eagle = nullptr, * vl_embs = nullptr; std::vector lm_h_dump, vlsa_dump; const MainKey mkey{ SEQ, n_img, SEQ_TXT, num_steps, inject_deepstack }; - const bool built = mg.ensure(backend, mkey, (size_t) 256*1024*1024, + const size_t main_nodes = 65536 + (size_t) num_steps*64*(dit_layers+1); + const bool built = mg.ensure(backend, mkey, main_nodes*ggml_tensor_overhead() + ggml_graph_overhead_custom(main_nodes, false), [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { ggml_tensor * t_embeds = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, SEQ); ggml_set_input(t_embeds); ggml_tensor * t_pos = ggml_new_tensor_1d(C, GGML_TYPE_I32, 4*SEQ); ggml_set_input(t_pos); @@ -605,7 +614,7 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { gio.t_ds[0]=t_ds[0]; gio.t_ds[1]=t_ds[1]; gio.t_ds[2]=t_ds[2]; gio.t_img_idx=t_img_idx; gio.t_txt_idx=t_txt_idx; gio.t_tau=t_tau; gio.t_tproj=t_tproj; gio.actions=actions; - ggml_cgraph * gf = ggml_new_graph_custom(C, 65536, false); + ggml_cgraph * gf = ggml_new_graph_custom(C, main_nodes, false); ggml_build_forward_expand(gf, actions); return gf; }); diff --git a/src/models/turbovla.cpp b/src/models/turbovla.cpp index 236aa75..c574016 100644 --- a/src/models/turbovla.cpp +++ b/src/models/turbovla.cpp @@ -470,11 +470,21 @@ bool load_config(const gguf_reader & g, TurboVlaModelArch & m) { I("period_token_id", m.period_id); I("question_token_id", m.question_id); if (g.has("turbovla.rope_theta")) m.rope_theta = g.f32("turbovla.rope_theta"); + bool ok = std::isfinite(m.rope_theta) && m.rope_theta > 0.f && m.image_size <= 4096; + for (int64_t v : { m.hidden, m.n_views, m.image_size, m.patch, m.vit_dim, m.vit_layers, m.vit_heads, + m.bert_dim, m.bert_layers, m.bert_heads, m.vocab, m.fusion_layers, m.fusion_heads, + m.enh_heads, m.dec_layers, m.dec_heads, m.horizon, m.action_dim, m.state_dim, + m.n_state_tok, m.text_len_max }) + ok = ok && v >= 1; + if (!ok) { + std::fprintf(stderr, "vla(turbovla): inconsistent dimensions in GGUF metadata\n"); + return false; + } int64_t fhd = m.fusion_dim / m.fusion_heads; U("fusion_head_dim", fhd); m.fusion_dim = fhd * m.fusion_heads; - if (m.image_size % m.patch || m.vit_dim % m.vit_heads || (m.vit_dim / m.vit_heads) % 4 || + if (m.fusion_dim < 1 || m.image_size % m.patch || m.vit_dim % m.vit_heads || (m.vit_dim / m.vit_heads) % 4 || m.bert_dim % m.bert_heads || m.hidden % m.enh_heads || m.hidden % m.dec_heads) { std::fprintf(stderr, "vla(turbovla): inconsistent dimensions in GGUF metadata\n"); return false; @@ -681,7 +691,7 @@ bool load_weights(TurboVlaModelArch & m, gguf_reader & g) { if (m.vocab != m.word_emb->ne[1] || m.text_len_max > m.bert_max_pos || m.patch_w->ne[0] != 3*m.patch*m.patch || (m.reg_tok && m.reg_tok->ne[1] != m.n_reg) || ggml_nelements(m.view_emb) != m.hidden*m.n_views || m.act_q->ne[1] != m.horizon || - m.dec[0].cross_qkv_w->ne[1] != 3*m.hidden) { + m.act_q->type != GGML_TYPE_F32 || m.dec[0].cross_qkv_w->ne[1] != 3*m.hidden) { std::fprintf(stderr, "vla(turbovla): tensor shapes disagree with GGUF metadata\n"); return false; } diff --git a/src/models/vla_jepa.cpp b/src/models/vla_jepa.cpp index 00667ec..f8d5161 100644 --- a/src/models/vla_jepa.cpp +++ b/src/models/vla_jepa.cpp @@ -124,7 +124,7 @@ namespace { bool load_config(const gguf_reader & g, VlaJepaModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; auto F = [&](const char * k, float & dst) { if (g.has(k)) dst = g.f32(k); }; - auto fk = [&](const char * s) { static char b[64]; std::snprintf(b, sizeof(b), "vla_jepa.%s", s); return b; }; + auto fk = [&](const char * s) { thread_local char b[64]; std::snprintf(b, sizeof(b), "vla_jepa.%s", s); return b; }; U(fk("vit_hidden"), m.vit_hidden); U(fk("vit_layers"), m.vit_layers); U(fk("vit_heads"), m.vit_heads); U(fk("vit_inter"), m.vit_inter); U(fk("patch_size"), m.patch_size); U(fk("temporal_patch_size"), m.temporal_patch); U(fk("spatial_merge_size"), m.spatial_merge); U(fk("vit_num_position_embeddings"), m.vit_num_pos); U(fk("vit_patch_flat"), m.vit_patch_flat); U(fk("vit_merged_dim"), m.vit_merged_dim); @@ -157,6 +157,11 @@ bool load_config(const gguf_reader & g, VlaJepaModelArch & m, Config & cfg) { (long long) m.image_target_size, (long long) m.patch_size, (long long) m.spatial_merge); return false; } + if (m.vit_heads <= 0 || m.vit_patch_flat != 3*m.temporal_patch*m.patch_size*m.patch_size) { + std::fprintf(stderr, "vla(vla_jepa): vit_heads %lld or vit_patch_flat %lld is inconsistent\n", + (long long) m.vit_heads, (long long) m.vit_patch_flat); + return false; + } // timesteps_proj always emits 256 floats into the time-projection input. if (m.time_proj_dim != 256) { std::fprintf(stderr, "vla(vla_jepa): time_proj_dim %lld, expected 256\n", (long long) m.time_proj_dim); @@ -303,7 +308,9 @@ bool VlaJepaModelArch::build_caches() { if (pos_table.empty() || (int64_t) pos_table.size() != vit_num_pos * vit_hidden) { std::fprintf(stderr, "vla(vla_jepa): build_caches: vit.pos_embd unreadable\n"); return false; } - interp_pos_embed(pos_table, num_side, vit_hidden, c_grow, c_gcol, grid, grid, c_pos_interp); + if (!interp_pos_embed(pos_table, num_side, vit_hidden, c_grow, c_gcol, grid, grid, c_pos_interp)) { + std::fprintf(stderr, "vla(vla_jepa): build_caches: vit_num_position_embeddings %lld is not a square\n", (long long) vit_num_pos); return false; + } c_tau.assign((size_t) num_steps, {}); c_tproj.assign((size_t) num_steps, {}); for (int64_t s=0; s VlaJepaModelArch::predict(const Inputs& in) { } int64_t img_end = img_start; while (img_end < SEQ && input_ids[img_end] == (int32_t) image_token_index) ++img_end; const int64_t n_img_tokens = img_end-img_start; + if (n_img_tokens % (llm_grid * llm_grid) != 0) { + std::fprintf(stderr, "vla(vla_jepa): image run length %lld not a multiple of %lld (post-merge grid)\n", + (long long) n_img_tokens, (long long) (llm_grid * llm_grid)); + return {}; + } const int64_t this_t = n_img_tokens/(llm_grid * llm_grid); const int64_t image_offset = text_len+st_idx; for (int64_t tt=0; tt VlaJepaModelArch::predict(const Inputs& in) { if (dump_prefix) head_graph.release(); const HeadKey hkey{ num_steps }; - const bool head_built = head_graph.ensure(backend, hkey, (size_t) 256*1024*1024, + const size_t head_nodes = 8192 + (size_t) num_steps*64*(dit_layers+1); + const bool head_built = head_graph.ensure(backend, hkey, head_nodes*ggml_tensor_overhead() + ggml_graph_overhead_custom(head_nodes, false), [&](ggml_context * C, HeadIO & gio) -> ggml_cgraph * { ggml_tensor * t_cond = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, num_future); ggml_set_input(t_cond); ggml_tensor * t_state = ggml_new_tensor_2d(C, GGML_TYPE_F32, state_dim, 1); ggml_set_input(t_state); @@ -641,7 +654,7 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { gio.t_cond=t_cond; gio.t_state=t_state; gio.t_x0=t_x0; gio.actions=actions; gio.t_tau=t_tau; gio.t_tproj=t_tproj; - ggml_cgraph * hg = ggml_new_graph_custom(C, 65536, false); + ggml_cgraph * hg = ggml_new_graph_custom(C, head_nodes, false); ggml_build_forward_expand(hg, actions); if (dump_prefix) for (int64_t s=0; s & row, const std::vector< } // Bilinear resample of the pretrained position table onto gh x gw. -inline void interp_pos_embed(const std::vector & table, int64_t num_side, int64_t hidden, +inline bool interp_pos_embed(const std::vector & table, int64_t num_side, int64_t hidden, const std::vector & row, const std::vector & col, int64_t gh, int64_t gw, std::vector & out) { + if (num_side <= 0 || (int64_t) table.size() != num_side*num_side*hidden) + return false; const int64_t S = (int64_t) row.size(); out.assign((size_t) S * hidden, 0.0f); auto src_coord = [&](int64_t k, int64_t g) -> double { return (g <= 1) ? 0.0 : (double) k * (double)(num_side-1)/(double)(g-1); }; @@ -180,6 +182,7 @@ inline void interp_pos_embed(const std::vector & table, int64_t num_side, for (int64_t c=0; c Date: Mon, 28 Sep 2026 14:53:57 +0700 Subject: [PATCH 08/61] fix Octo crashes on bad tokens and stats --- src/models/octo.cpp | 55 ++++++++++++++++++++++++++++++--------------- 1 file changed, 37 insertions(+), 18 deletions(-) diff --git a/src/models/octo.cpp b/src/models/octo.cpp index 8690059..a807142 100644 --- a/src/models/octo.cpp +++ b/src/models/octo.cpp @@ -798,6 +798,16 @@ bool run_t5_encoder_graph(OctoRuntime& rt, std::fprintf(stderr, "vla(octo): T5 encoder expected %d input_ids/attention_mask\n", seq); return false; } + const ggml_tensor * te = rt.weight("octo.t5.tok_embd.weight"); + if (!te) + return false; + for (int i=0; i= te->ne[1]) { + std::fprintf(stderr, "vla(octo): lang_tokens[%d]=%d out of vocab range [0, %lld)\n", + i, input_ids[(size_t) i], (long long) te->ne[1]); + return false; + } + } std::vector bucket_idx((size_t) seq*seq); std::vector padmask((size_t) seq*seq); @@ -1557,7 +1567,8 @@ bool resolve_stats_block(const nlohmann::json& j, const std::string& dataset_key // Parses octo.dataset_statistics once and caches the blocks on `rt`: it is a // JSON blob in the metadata, ~25 datasets wide for the pretrain checkpoint, and // re-reading 21 floats out of it per request is pure overhead. -bool ensure_stats(OctoRuntime& rt, gguf_reader& g, const std::string& dataset_key_in, int64_t action_dim) { +bool ensure_stats(OctoRuntime& rt, gguf_reader& g, const std::string& dataset_key_in, int64_t action_dim, + bool need_proprio) { if (rt.stats_loaded && rt.stats_key == dataset_key_in) return true; @@ -1581,27 +1592,35 @@ bool ensure_stats(OctoRuntime& rt, gguf_reader& g, const std::string& dataset_ke const auto& act = (*block)["action"]; OctoRuntime::ActionStats a; - a.mean = act.at("mean").get>(); - a.stdv = act.at("std").get>(); - for (bool b : act.at("mask").get>()) - a.mask.push_back(b ? 1 : 0); + OctoRuntime::ProprioStats pr; + bool has_pr = false; + try { + a.mean = act.at("mean").get>(); + a.stdv = act.at("std").get>(); + if (act.contains("mask")) { + for (bool b : act["mask"].get>()) + a.mask.push_back(b ? 1 : 0); + } else { + a.mask.assign(a.mean.size(), 1); + } + if (need_proprio && block->contains("proprio")) { + const auto& p = (*block)["proprio"]; + pr.mean = p.at("mean").get>(); + pr.stdv = p.at("std").get>(); + has_pr = true; + } + } catch (const nlohmann::json::exception& e) { + std::fprintf(stderr, "vla(octo): bad dataset_statistics: %s\n", e.what()); + return false; + } if (a.mask.size() != (size_t) action_dim || a.mean.size() != (size_t) action_dim || a.stdv.size() != (size_t) action_dim) { std::fprintf(stderr, "vla(octo): dataset_statistics/action is not %lld-dim\n", (long long) action_dim); return false; } - - OctoRuntime::ProprioStats pr; - bool has_pr = false; - if (block->contains("proprio")) { - const auto& p = (*block)["proprio"]; - pr.mean = p.at("mean").get>(); - pr.stdv = p.at("std").get>(); - if (pr.mean.size() != (size_t) kProprioTokens || pr.stdv.size() != (size_t) kProprioTokens) { - std::fprintf(stderr, "vla(octo): dataset_statistics/proprio is not %d-dim\n", kProprioTokens); - return false; - } - has_pr = true; + if (has_pr && (pr.mean.size() != (size_t) kProprioTokens || pr.stdv.size() != (size_t) kProprioTokens)) { + std::fprintf(stderr, "vla(octo): dataset_statistics/proprio is not %d-dim\n", kProprioTokens); + return false; } rt.action_stats = std::move(a); @@ -1690,7 +1709,7 @@ bool run_pipeline(OctoModelArch& m, const OctoFrame& f, const int window_size = (int) m.window_size; const int action_total = (int) (m.action_horizon*m.action_dim); const int n_proprio = m.has_proprio ? kProprioTokens : 0; - if (!ensure_stats(rt, m.io, "", m.action_dim)) + if (!ensure_stats(rt, m.io, "", m.action_dim, m.has_proprio)) return false; // Cold start: history is filled with copies of the one live frame and every From 9bf951d45f7a7903b6ba320ffc6695e861b13145 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Mon, 28 Sep 2026 14:53:57 +0700 Subject: [PATCH 09/61] fix BitVLA uploads, device choice and kernel races --- src/kernels/bitvla/bitnet_kernels.cu | 40 ------ src/kernels/bitvla/bitnet_kernels.h | 88 ++---------- src/kernels/bitvla/bitvla_fp32head_cuda.cu | 5 +- src/kernels/bitvla/bitvla_lm_cuda.cu | 40 +----- src/kernels/bitvla/bitvla_lm_cuda.h | 21 --- src/kernels/bitvla/bitvla_vit_cuda.cu | 3 +- src/models/bitvla.cpp | 150 +++++++++++++++------ 7 files changed, 130 insertions(+), 217 deletions(-) diff --git a/src/kernels/bitvla/bitnet_kernels.cu b/src/kernels/bitvla/bitnet_kernels.cu index 2204474..350fc3e 100644 --- a/src/kernels/bitvla/bitnet_kernels.cu +++ b/src/kernels/bitvla/bitnet_kernels.cu @@ -28,42 +28,6 @@ std::abort(); } -extern "C" void bitlinear_int8xint2(int8_t* input0, int8_t* input1, __nv_bfloat16* output0, float* s, float* ws, int M, int N, int K, cudaStream_t stream){ - if (M == 1 && N == 3840 && K == 2560){ - ladder_int8xint2_kernel<1, 3840, 2560, 3, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if (M == 1 && N == 2560 && K == 2560){ - ladder_int8xint2_kernel<1, 2560, 2560, 1, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if (M == 1 && N == 13824 && K == 2560){ - ladder_int8xint2_kernel<1, 13824, 2560, 2, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if (M == 1 && N == 2560 && K == 6912){ - ladder_int8xint2_kernel<1, 2560, 6912, 1, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if(M == 1 && N == 4800 && K == 3200){ - ladder_int8xint2_kernel<1, 4800, 3200, 6, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if(M == 1 && N == 3200 && K == 3200){ - ladder_int8xint2_kernel<1, 3200, 3200, 1, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if(M == 1 && N == 20480 && K == 3200){ - ladder_int8xint2_kernel<1, 20480, 3200, 2, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if(M == 1 && N == 3200 && K == 10240){ - ladder_int8xint2_kernel<1, 3200, 10240, 1, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if(M == 1 && N == 5120 && K == 27648){ - ladder_int8xint2_kernel<1, 5120, 27648, 1, 8, 16><<>>(input0, input1, output0, s, ws); - } - else if(M == 1 && N == 55296 && K == 5120){ - ladder_int8xint2_kernel<1, 55296, 5120, 1, 8, 16><<>>(input0, input1, output0, s, ws); - } - else{ - bitlinear_unsupported_shape("bitlinear_int8xint2", M, N, K); - } -} - // The wide kernel amortises the shared A block over 4 column tiles instead of // 1 (see ladder_int8xint2_kernel_m_wide). It is the default; set // VLA_BITVLA_NARROW_GEMM=1 to fall back to the one-tile-per-CTA kernel, which @@ -92,8 +56,6 @@ extern "C" void bitlinear_int8xint2_m( else if (N == 1152 && K == 1152) WIDE(1152, 1152, 1); else if (N == 4304 && K == 1152) WIDE(4304, 1152, 1); else if (N == 1152 && K == 4352) WIDE(1152, 4352, 1); - - else if (N == 3840 && K == 2560) WIDE(3840, 2560, 3); else bitlinear_unsupported_shape("bitlinear_int8xint2_m", M, N, K); #undef WIDE @@ -109,8 +71,6 @@ extern "C" void bitlinear_int8xint2_m( else if (N == 1152 && K == 1152) launch_ladder_int8xint2_m<1152, 1152, 1, 128>(input0, input1, output0, s, ws, M, stream); else if (N == 4304 && K == 1152) launch_ladder_int8xint2_m<4304, 1152, 1, 128>(input0, input1, output0, s, ws, M, stream); else if (N == 1152 && K == 4352) launch_ladder_int8xint2_m<1152, 4352, 1, 128>(input0, input1, output0, s, ws, M, stream); - - else if (N == 3840 && K == 2560) launch_ladder_int8xint2_m<3840, 2560, 3, 128>(input0, input1, output0, s, ws, M, stream); else bitlinear_unsupported_shape("bitlinear_int8xint2_m", M, N, K); } diff --git a/src/kernels/bitvla/bitnet_kernels.h b/src/kernels/bitvla/bitnet_kernels.h index bae049d..5656900 100644 --- a/src/kernels/bitvla/bitnet_kernels.h +++ b/src/kernels/bitvla/bitnet_kernels.h @@ -23,11 +23,8 @@ * intrinsic to expand i2 to i8, then run a single warp-level @c wmma * fragment multiply per tile. * - * Two GEMM entry points are provided: - * * @ref ladder_int8xint2_kernel-single-row (M=1) decode kernel - * used for next-token/single-query inference. - * * @ref ladder_int8xint2_kernel_m + @ref launch_ladder_int8xint2_m - * - multi-row (M>1) variant for prefill and ViT batches. + * GEMM entry point: @ref ladder_int8xint2_kernel_m + @ref launch_ladder_int8xint2_m + * - multi-row (M>1) variant for prefill and ViT batches. * * This header is meant to be included by the per-tier CUDA `.cu` files * (@c bitvla_lm_cuda.cu, @c bitvla_vit_cuda.cu); it is not part of the @@ -38,23 +35,10 @@ #include #include #include -#include #include #include #include -#if (((__CUDACC_VER_MAJOR__ == 11) && (__CUDACC_VER_MINOR__ >= 4)) || (__CUDACC_VER_MAJOR__ > 11)) -#define TVM_ENABLE_L2_PREFETCH 1 -#else -#define TVM_ENABLE_L2_PREFETCH 0 -#endif - -#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ == 800 -#define TVM_ENBALE_EFFICIENT_SMEM_PTR_CAST 1 -#else -#define TVM_ENBALE_EFFICIENT_SMEM_PTR_CAST 0 -#endif - /** * @brief Decode a packed int2 word into N int8 values via @c lop3.b32. * @@ -89,61 +73,6 @@ __device__ void decode_i2s_to_i8s(T1 *_i2s, T2 *_i8s, const int N = 16) } } -/** - * @brief Single-row ternary GEMM kernel (M = 1). - * - * Computes one row of @c dtype_transform[0,:] = (A * B^T)/s[0]*ws, - * with @c A in int8, @c B packed as int2 (decoded on the fly), accumulated - * in int32 via @c __dp4a, then scaled back to bf16. The output bias - * @c ws is applied per @c ws_num column groups. - * - * @tparam M Always 1 in this overload; kept for symmetry with - * the multi-row kernel. - * @tparam N Output column count. - * @tparam K Reduction dimension. - * @tparam ws_num Number of column groups sharing one bias entry. - * @tparam K_block_size Threads collaborating along K (warp width). - * @tparam N_block_size Threads collaborating along N (warp height). - */ -template -__global__ void __launch_bounds__(128) ladder_int8xint2_kernel(int8_t* __restrict__ A, int8_t* __restrict__ B, __nv_bfloat16* __restrict__ dtype_transform, float* __restrict__ s, float* __restrict__ ws) { - constexpr int K_per_loop = 16; - constexpr int wmma_K = 32; - constexpr int wmma_N = 16; - int in_thread_C_local[1]; - signed char A_local[K_per_loop]; - int B_reshape_local[1]; - signed char B_decode_local[K_per_loop]; - int red_buf0[1]; - in_thread_C_local[0] = 0; - #pragma unroll - for (int k_0=0; k_0> 1)*wmma_K * wmma_N/4) + - ((((int)threadIdx.y) >> 3)*(wmma_K * wmma_N/2)/4) + - ((((int)threadIdx.x) & 1)*(wmma_K * wmma_N/4)/4) + - ((((int)threadIdx.y) & 7)*(wmma_K/2)/4) - ); - decode_i2s_to_i8s(B_reshape_local, B_decode_local, 16); - #pragma unroll - for (int k_2_0=0; k_2_0<4; ++k_2_0) { - in_thread_C_local[0] = __dp4a(*(int *)&A_local[((k_2_0*4))],*(int *)&B_decode_local[((k_2_0*4))], in_thread_C_local[0]); - } - } - red_buf0[0] = in_thread_C_local[0]; - #pragma unroll - for (int offset=K_block_size/2; offset>0; offset /= 2) { - red_buf0[0] += __shfl_down_sync(__activemask(), red_buf0[0], offset, K_block_size); - } - int out_idx = ((((int)blockIdx.x)*N_block_size)+((int)threadIdx.y)); - int ws_idx = out_idx/(N/ws_num); - if (threadIdx.x == 0) - dtype_transform[out_idx] = __float2bfloat16(((float)red_buf0[0])/s[0]*ws[ws_idx]); -} - /** * @brief Multi-row ternary GEMM kernel (M > 1) using @c wmma fragments. * @@ -180,9 +109,9 @@ __global__ void __launch_bounds__(128) ladder_int8xint2_kernel_m( const int warp = tid >> 5; const int m_base = (int)blockIdx.y*M_ROWS; - __shared__ signed char A_smem[M_ROWS][K_CHUNK]; - __shared__ signed char W_smem[16][K_CHUNK]; - __shared__ int C_smem[M_ROWS][16]; + __shared__ __align__(32) signed char A_smem[M_ROWS][K_CHUNK]; + __shared__ __align__(32) signed char W_smem[16][K_CHUNK]; + __shared__ __align__(32) int C_smem[M_ROWS][16]; int B_reshape_local[1]; signed char B_decode_local[K_per_loop]; @@ -316,9 +245,9 @@ __global__ void __launch_bounds__(128) ladder_int8xint2_kernel_m_wide( const int lane = tid & 31; const int m_base = (int)blockIdx.y*M_ROWS; - __shared__ signed char A_smem[M_ROWS][SM_STRIDE]; - __shared__ signed char W_smem[N_TILES][16][SM_STRIDE]; - __shared__ int C_smem[WARPS][16][16]; + __shared__ __align__(32) signed char A_smem[M_ROWS][SM_STRIDE]; + __shared__ __align__(32) signed char W_smem[N_TILES][16][SM_STRIDE]; + __shared__ __align__(32) int C_smem[WARPS][16][16]; // Column tile this warp owns. N is not always a multiple of 16*N_TILES // (the ViT's 4304 is 269 tiles), so tiles past the end are skipped rather @@ -430,7 +359,6 @@ static constexpr int bitvla_n_tiles_for(int N, int K) { : (N == 1152 && K == 1152) ? 2 // vit.q/k/v/o : (N == 4304 && K == 1152) ? 4 // vit.fc1 : (N == 1152 && K == 4352) ? 2 // vit.fc2 - : (N == 3840 && K == 2560) ? 4 // action head qkv : 2; } diff --git a/src/kernels/bitvla/bitvla_fp32head_cuda.cu b/src/kernels/bitvla/bitvla_fp32head_cuda.cu index 3bd1c24..6a43c98 100644 --- a/src/kernels/bitvla/bitvla_fp32head_cuda.cu +++ b/src/kernels/bitvla/bitvla_fp32head_cuda.cu @@ -22,7 +22,7 @@ return -1; } } while (0) #define CUDA_OK_NULL(c) do { cudaError_t _e = (c); if (_e != cudaSuccess) { \ std::fprintf(stderr, "vla(bitvla_fp32head): %s at %s:%d (%s)\n", cudaGetErrorString(_e), __FILE__, __LINE__, #c); \ - return nullptr; } } while (0) + bitvla_fp32head_cuda_free(ctx); return nullptr; } } while (0) struct bitvla_fp32head_cuda_ctx { cublasHandle_t cublas; @@ -117,6 +117,7 @@ __global__ void layernorm_fp32_kernel(const float* __restrict__ x, } __syncthreads(); const float mean = smem[0]/(float)K; + __syncthreads(); float vsum = 0.0f; for (int k=tid; kd_pp_out, (size_t)ctx->lm_hidden*sizeof(float), cudaMemcpyDeviceToHost, stream)); + CUDA_OK_RET(cudaGetLastError()); CUDA_OK_RET(cudaStreamSynchronize(stream)); return 0; } @@ -374,6 +376,7 @@ extern "C" int bitvla_fp32head_action_forward( CUDA_OK_RET(cudaMemcpyAsync(host_norm_actions, ctx->d_ah_out, (size_t)M * A * sizeof(float), cudaMemcpyDeviceToHost, stream)); + CUDA_OK_RET(cudaGetLastError()); CUDA_OK_RET(cudaStreamSynchronize(stream)); return 0; } diff --git a/src/kernels/bitvla/bitvla_lm_cuda.cu b/src/kernels/bitvla/bitvla_lm_cuda.cu index 7605a0e..91b6a41 100644 --- a/src/kernels/bitvla/bitvla_lm_cuda.cu +++ b/src/kernels/bitvla/bitvla_lm_cuda.cu @@ -22,6 +22,7 @@ #include #include #include +#include #include extern "C" void bitlinear_int8xint2_m(int8_t* A, int8_t* B, __nv_bfloat16* out, @@ -132,6 +133,7 @@ __global__ void softmax_scaled_bf16_kernel(__nv_bfloat16* __restrict__ inout, } __syncthreads(); const float max_v = smem[0]; + __syncthreads(); float s_sum = 0.0f; for (int i=tid; i= N) - return; - float gv = __bfloat162float(g[i]); - if (gv < 0.0f) - gv = 0.0f; - out[i] = __float2bfloat16(gv * gv * __bfloat162float(u[i])); -} - __global__ void add_bf16_kernel(const __nv_bfloat16* a, const __nv_bfloat16* b, __nv_bfloat16* out, int N) { const int i = (int)(blockIdx.x*blockDim.x+threadIdx.x); @@ -240,11 +230,6 @@ extern "C" void bitvla_softmax_scaled_bf16(__nv_bfloat16* inout, float scale, constexpr int B = 256; softmax_scaled_bf16_kernel<<>>(inout, scale, S); } -extern "C" void bitvla_squared_relu_mul_bf16(const __nv_bfloat16* g, const __nv_bfloat16* u, - __nv_bfloat16* out, int N, cudaStream_t stream) { - constexpr int B = 256; - squared_relu_mul_bf16_kernel<<>>(g, u, out, N); -} extern "C" void bitvla_add_bf16(const __nv_bfloat16* a, const __nv_bfloat16* b, __nv_bfloat16* out, int N, cudaStream_t stream) { constexpr int B = 256; @@ -277,7 +262,7 @@ extern "C" void bitvla_gather_rows_bf16(const __nv_bfloat16* in, __nv_bfloat16* return -1; } } while (0) #define CUDA_OKV(call) do { cudaError_t e = (call); if (e != cudaSuccess) { \ std::fprintf(stderr, "vla(bitvla_lm_cuda): %s @ %s:%d\n", cudaGetErrorString(e), __FILE__, __LINE__); \ - return nullptr; } } while (0) + bitvla_lm_cuda_free(ctx); return nullptr; } } while (0) struct bitvla_lm_cuda_ctx { int hidden, n_q, n_kv, head_dim, ffn, n_layers, max_seq; @@ -578,6 +563,7 @@ __global__ void layernorm_bias_bf16_kernel(const __nv_bfloat16* __restrict__ x, } __syncthreads(); const float mean = smem[0]/(float)K; + __syncthreads(); float vsum = 0.0f; for (int k=tid; k= total_cols) - return; - x[(size_t)m * total_cols+k] = __float2bfloat16(0.0f); -} - extern "C" void bitvla_layernorm_bf16(const __nv_bfloat16* x, const __nv_bfloat16* w, const __nv_bfloat16* b, __nv_bfloat16* out, float eps, int M, int K, cudaStream_t stream) { @@ -655,15 +633,6 @@ extern "C" void bitvla_add_bias_bf16(const __nv_bfloat16* x, const __nv_bfloat16 const int n_kb = (K+B-1)/B; add_bias_bf16_kernel<<>>(x, bias, out, M, K); } -extern "C" void bitvla_zero_tail_bf16(__nv_bfloat16* x, int M, int total_cols, - int start_col, cudaStream_t stream) { - if (start_col >= total_cols) - return; - constexpr int B = 128; - const int len = total_cols-start_col; - const int n_kb = (len+B-1)/B; - zero_tail_bf16_kernel<<>>(x, total_cols, start_col); -} extern "C" int bitvla_lm_cuda_forward(bitvla_lm_cuda_ctx* ctx, const __nv_bfloat16* d_in, @@ -720,5 +689,6 @@ extern "C" int bitvla_lm_cuda_forward(bitvla_lm_cuda_ctx* ctx, cudaStreamSynchronize(stream); dump_to_file("lm_final", d_out); } + CUDA_OK(cudaGetLastError()); return 0; } diff --git a/src/kernels/bitvla/bitvla_lm_cuda.h b/src/kernels/bitvla/bitvla_lm_cuda.h index ee79799..004fbaf 100644 --- a/src/kernels/bitvla/bitvla_lm_cuda.h +++ b/src/kernels/bitvla/bitvla_lm_cuda.h @@ -80,17 +80,6 @@ void bitvla_rope_neox_bf16(__nv_bfloat16* inout, const float* cos_tab, void bitvla_softmax_scaled_bf16(__nv_bfloat16* inout, float scale, int n_rows, int S, cudaStream_t stream); -/** - * @brief Elementwise @c relu(g)^2*u (BitVLA squared-ReLU FFN gate). - * @param g Gate input (N), bf16 device pointer. - * @param u Up input (N), bf16 device pointer. - * @param out Output (N), bf16 device pointer. - * @param N Element count. - * @param stream CUDA stream. - */ -void bitvla_squared_relu_mul_bf16(const __nv_bfloat16* g, const __nv_bfloat16* u, - __nv_bfloat16* out, int N, cudaStream_t stream); - /** * @brief Elementwise bf16 add (@p out = @p a + @p b). * @param a Length-@p N input, device pointer. @@ -191,16 +180,6 @@ void bitvla_gelu_tanh_bf16(const __nv_bfloat16* x, __nv_bfloat16* out, void bitvla_add_bias_bf16(const __nv_bfloat16* x, const __nv_bfloat16* bias, __nv_bfloat16* out, int M, int K, cudaStream_t stream); -/** - * @brief Zero out columns [@p start_col, @p total_cols) of an - * (M x @p total_cols) bf16 matrix, leaving the first @p start_col - * columns untouched. - * - * Used to mask out padding lanes added by the ternary GEMM's column tiling. - */ -void bitvla_zero_tail_bf16(__nv_bfloat16* x, int M, int total_cols, - int start_col, cudaStream_t stream); - /** * @brief Per-layer weight pointers for one BitVLA LM transformer block. * diff --git a/src/kernels/bitvla/bitvla_vit_cuda.cu b/src/kernels/bitvla/bitvla_vit_cuda.cu index 2f48e3e..5add370 100644 --- a/src/kernels/bitvla/bitvla_vit_cuda.cu +++ b/src/kernels/bitvla/bitvla_vit_cuda.cu @@ -48,7 +48,7 @@ static void gelu_erf_bf16(const __nv_bfloat16* in, __nv_bfloat16* out, int N, cu return -1; } } while (0) #define CUDA_OKV(call) do { cudaError_t e = (call); if (e != cudaSuccess) { \ std::fprintf(stderr, "vla(bitvla_vit_cuda): %s @ %s:%d\n", cudaGetErrorString(e), __FILE__, __LINE__); \ - return nullptr; } } while (0) + bitvla_vit_cuda_free(ctx); return nullptr; } } while (0) struct bitvla_vit_cuda_ctx { int n_layers, hidden, n_heads, head_dim, ffn, n_patches, patch_flat, mm_out; @@ -316,5 +316,6 @@ int bitvla_vit_cuda_forward(bitvla_vit_cuda_ctx* ctx, return -1; } bitvla_add_bias_bf16(d_out, ctx->mm_b2, d_out, seq, M, stream); + CUDA_OK(cudaGetLastError()); return 0; } diff --git a/src/models/bitvla.cpp b/src/models/bitvla.cpp index f2f63b9..6366148 100644 --- a/src/models/bitvla.cpp +++ b/src/models/bitvla.cpp @@ -133,7 +133,7 @@ struct BitvlaModelArch : public ModelArchBase { std::vector vit; ggml_tensor *mm_l1_w = nullptr, *mm_l1_b = nullptr, *mm_l2_w = nullptr, *mm_l2_b = nullptr; ggml_tensor *pp_fc1_w = nullptr, *pp_fc1_b = nullptr, *pp_fc2_w = nullptr, *pp_fc2_b = nullptr; - ggml_tensor *embed_tokens = nullptr, *lm_output_norm = nullptr; + ggml_tensor *lm_output_norm = nullptr; std::vector lm; ggml_tensor *ah_ln1_w = nullptr, *ah_ln1_b = nullptr, *ah_fc1_w = nullptr, *ah_fc1_b = nullptr; ggml_tensor *ah_b0_ln_w = nullptr, *ah_b0_ln_b = nullptr, *ah_b0_w = nullptr, *ah_b0_b = nullptr; @@ -151,6 +151,7 @@ struct BitvlaModelArch : public ModelArchBase { bool cuda_vit_ready = false; std::vector cpu_kept_ptrs; + int cuda_dev = 0; __nv_bfloat16* d_inputs_embeds = nullptr; __nv_bfloat16* d_last_hidden = nullptr; @@ -391,6 +392,11 @@ bool load_config(const gguf_reader & g, BitvlaModelArch & m, Config & cfg) { (long long) m.lm_q, (long long) m.lm_head_dim, (long long) m.lm_hidden); return false; } + if (m.vit_heads <= 0 || m.vit_head_dim <= 0 || m.vit_heads*m.vit_head_dim != m.vit_hidden) { + std::fprintf(stderr, "vla(bitvla): vit heads %lld x head_dim %lld does not match hidden %lld\n", + (long long) m.vit_heads, (long long) m.vit_head_dim, (long long) m.vit_hidden); + return false; + } const std::string js = g.str("bitvla.statistics_json"); if (js.empty()) { @@ -551,6 +557,8 @@ static int8_t* pack_and_upload_fused(const std::vector& wptrs, BitvlaModelArch::~BitvlaModelArch() { #ifdef VLA_BITVLA_CUDA_KERNELS + if (lm_cuda_ctx || !cuda_devptrs.empty()) + cudaSetDevice(cuda_dev); if (lm_cuda_ctx) bitvla_lm_cuda_free(lm_cuda_ctx); if (vit_cuda_ctx) @@ -603,9 +611,11 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, } if (!load_config(g, *m, m->cfg)) return nullptr; + if (m->packed_int2) + m->matmul_type = GGML_TYPE_F32; // Keep one reader open for the per-step token-embedding fetches (token_embd - // stays on disk under int2 packing) and cache the constant stop-token row, + // stays on disk) and cache the constant stop-token row, // so predict() no longer re-opens and re-parses the GGUF twice per call. if (!m->emb_reader.open(ckpt_path)) return nullptr; @@ -670,7 +680,6 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, m->pp_fc1_w=mk_f32("aex.proprio.fc1.weight"); m->pp_fc1_b=mk_f32("aex.proprio.fc1.bias"); m->pp_fc2_w=mk_f32("aex.proprio.fc2.weight"); m->pp_fc2_b=mk_f32("aex.proprio.fc2.bias"); - m->embed_tokens = m->packed_int2 ? nullptr : mk_mm("token_embd.weight"); m->lm_output_norm = mk_f32("lm.output_norm.weight"); m->lm.resize(m->lm_layers); for (int64_t i=0; ilm_layers && ok; ++i) { @@ -702,7 +711,7 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, m->ah_fc2_w =mk_mm ("aex.head.fc2.weight"); m->ah_fc2_b =mk_f32("aex.head.fc2.bias"); ok &= m->vit_patch_w&&m->vit_patch_b&&m->vit_pos&&m->mm_l1_w&&m->mm_l1_b&&m->mm_l2_w&&m->mm_l2_b&& - m->pp_fc1_w&&m->pp_fc1_b&&m->pp_fc2_w&&m->pp_fc2_b&&(m->embed_tokens||m->packed_int2)&&m->lm_output_norm&& + m->pp_fc1_w&&m->pp_fc1_b&&m->pp_fc2_w&&m->pp_fc2_b&&m->lm_output_norm&& m->ah_ln1_w&&m->ah_ln1_b&&m->ah_fc1_w&&m->ah_fc1_b&& m->ah_b0_ln_w&&m->ah_b0_ln_b&&m->ah_b0_w&&m->ah_b0_b&& m->ah_b1_ln_w&&m->ah_b1_ln_b&&m->ah_b1_w&&m->ah_b1_b&& @@ -731,25 +740,40 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, #ifdef VLA_BITVLA_CUDA_KERNELS - if ((m->packed_int2 || m->matmul_type == GGML_TYPE_F32) && !vla::env_flag("VLA_BITVLA_NO_CUDA_LM")) { + const ggml_tensor * quantized = nullptr; + for (ggml_tensor * t = ggml_get_first_tensor(m->ctx_weights); t && !quantized; t = ggml_get_next_tensor(m->ctx_weights, t)) + if (ggml_is_quantized(t->type)) + quantized = t; + if (quantized) + std::fprintf(stderr, "vla(bitvla): %s is quantized; the CUDA path needs F32 weights\n", ggml_get_name(quantized)); + + if (m->matmul_type == GGML_TYPE_F32 && !quantized && !vla::env_flag("VLA_BITVLA_NO_CUDA_LM")) { int dev_count = 0; - if (cudaGetDeviceCount(&dev_count) == cudaSuccess && dev_count > 0) { - cudaSetDevice(0); + m->cuda_dev = vla::backend_device_index(); + if (cudaGetDeviceCount(&dev_count) == cudaSuccess && m->cuda_dev < dev_count && + cudaSetDevice(m->cuda_dev) == cudaSuccess) { + cudaGetLastError(); // The ladder kernels dereference the scale pointer unconditionally, so // a missing sidecar is a device-side OOB read, not a soft failure. - bool scales_ok = true; + bool weights_ok = true; auto load_bit = [&](ggml_tensor * t, int64_t N, int64_t K) -> std::pair { + const size_t want = m->packed_int2 ? (size_t) (N*K/4) : (size_t) (N*K)*sizeof(float); + if (ggml_nbytes(t) != want) { + std::fprintf(stderr, "vla(bitvla): %s is %zu bytes, expected %zu\n", ggml_get_name(t), ggml_nbytes(t), want); + weights_ok = false; + return { nullptr, nullptr }; + } if (m->packed_int2) { int8_t * dp = upload_int8((const uint8_t*) t->data, ggml_nbytes(t), m->cuda_devptrs); std::string nm = ggml_get_name(t); std::string sn = nm.substr(0, nm.size()-7) + ".scale"; std::vector sc = g.read_f32(sn.c_str()); - if (sc.empty()) { - std::fprintf(stderr, "vla(bitvla): int2 tensor %s has no %s sidecar\n", + if (sc.size() != 1) { + std::fprintf(stderr, "vla(bitvla): int2 tensor %s needs a 1-element %s sidecar\n", nm.c_str(), sn.c_str()); - scales_ok = false; + weights_ok = false; return { dp, nullptr }; } float * dws = upload_f32_scales(sc.data(), (int) sc.size(), m->cuda_devptrs); @@ -766,7 +790,7 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, (int) m->lm_inter, (int) m->lm_layers, m->lm_rope_base, m->lm_rms_eps, max_seq); if (m->lm_cuda_ctx) { bool pack_ok = true; - for (int64_t L=0; Llm_layers && pack_ok && scales_ok; ++L) { + for (int64_t L=0; Llm_layers && pack_ok && weights_ok; ++L) { bitvla_lm_layer_cuda lyr{}; lyr.attn_norm_w = upload_bf16_from_f32((const float*) m->lm[L].attn_norm->data, m->lm_hidden, m->cuda_devptrs); @@ -799,15 +823,20 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, } if (m->packed_int2) { - lyr.gate_up_packed = upload_int8((const uint8_t*) m->lm[L].Wgate_up->data, ggml_nbytes(m->lm[L].Wgate_up), m->cuda_devptrs); const std::string sn = "lm.blk." + std::to_string(L) + ".ffn_gate_up.scale"; std::vector sc = g.read_f32(sn.c_str()); - if (sc.empty()) { - std::fprintf(stderr, "vla(bitvla): missing %s\n", sn.c_str()); - scales_ok = false; + if (ggml_nbytes(m->lm[L].Wgate_up) != (size_t) (2*m->lm_inter*m->lm_hidden/4) || sc.size() != 2) { + std::fprintf(stderr, "vla(bitvla): lm.blk.%lld.ffn_gate_up weight or its 2-element %s is malformed\n", + (long long) L, sn.c_str()); + weights_ok = false; } else { + lyr.gate_up_packed = upload_int8((const uint8_t*) m->lm[L].Wgate_up->data, ggml_nbytes(m->lm[L].Wgate_up), m->cuda_devptrs); lyr.gate_up_ws = upload_f32_scales(sc.data(), (int) sc.size(), m->cuda_devptrs); } + } else if (ggml_nbytes(m->lm[L].Wgate) != (size_t) (m->lm_inter*m->lm_hidden)*sizeof(float) || + ggml_nbytes(m->lm[L].Wup) != (size_t) (m->lm_inter*m->lm_hidden)*sizeof(float)) { + std::fprintf(stderr, "vla(bitvla): lm.blk.%lld ffn_gate/ffn_up size mismatch\n", (long long) L); + weights_ok = false; } else { std::vector ws2; lyr.gate_up_packed = pack_and_upload_fused( @@ -823,11 +852,12 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, lyr.down_ws = r.second; } bitvla_lm_cuda_set_layer(m->lm_cuda_ctx, (int) L, &lyr); + pack_ok = cudaGetLastError() == cudaSuccess; } - if (!scales_ok) { - std::fprintf(stderr, "vla(bitvla): int2 scale sidecars incomplete; refusing the CUDA LM\n"); + if (!weights_ok) { + std::fprintf(stderr, "vla(bitvla): LM weights or int2 scale sidecars invalid; refusing the CUDA LM\n"); } - if (pack_ok && scales_ok) { + if (pack_ok && weights_ok) { __nv_bfloat16* onorm = upload_bf16_from_f32((const float*) m->lm_output_norm->data, m->lm_hidden, m->cuda_devptrs); bitvla_lm_cuda_set_output_norm(m->lm_cuda_ctx, onorm); @@ -838,6 +868,8 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, lm_ce = cudaMalloc(&m->d_action_hidden, (size_t) (m->num_actions_chunk*m->action_dim)*m->lm_hidden*sizeof(__nv_bfloat16)); if (lm_ce == cudaSuccess) lm_ce = cudaMalloc(&m->d_action_ids, (size_t) (m->num_actions_chunk*m->action_dim)*sizeof(int32_t)); + if (lm_ce == cudaSuccess) + lm_ce = cudaGetLastError(); // only enable the CUDA LM once every work buffer is really allocated. if (lm_ce != cudaSuccess) { std::fprintf(stderr, "vla(bitvla): CUDA LM buffer alloc failed (%s); using CPU LM\n", @@ -865,6 +897,7 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, (int) m->vit_inter, (int) m->n_patches, patch_flat, m->vit_ln_eps, mm_out); if (m->vit_cuda_ctx) { + cudaGetLastError(); bool vit_ok = true; for (int64_t L=0; Lvit_layers && vit_ok; ++L) { bitvla_vit_layer_cuda vl{}; @@ -913,6 +946,9 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, vl.fc2_ws = r.second; } vl.fc2_b = upload_bf16_from_f32((const float*) m->vit[L].bfc2->data, m->vit_hidden, m->cuda_devptrs); + } else if (ggml_nbytes(m->vit[L].Wfc2) != (size_t) (m->vit_hidden*m->vit_inter)*sizeof(float)) { + std::fprintf(stderr, "vla(bitvla): vit.blk.%lld.fc2.weight size mismatch\n", (long long) L); + weights_ok = false; } else { const float* W = (const float*) m->vit[L].Wfc2->data; std::vector tern; @@ -932,9 +968,9 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, vl.fc2_b = upload_bf16_from_f32((const float*) m->vit[L].bfc2->data, m->vit_hidden, m->cuda_devptrs); } bitvla_vit_cuda_set_layer(m->vit_cuda_ctx, (int) L, &vl); + vit_ok = weights_ok && cudaGetLastError() == cudaSuccess; } if (vit_ok) { - __nv_bfloat16* pe_w = upload_bf16_from_f32((const float*) m->vit_patch_w->data, m->vit_hidden*patch_flat, m->cuda_devptrs); __nv_bfloat16* pe_b = upload_bf16_from_f32((const float*) m->vit_patch_b->data, m->vit_hidden, m->cuda_devptrs); __nv_bfloat16* pos_e = upload_bf16_from_f32((const float*) m->vit_pos->data, m->n_patches*m->vit_hidden, m->cuda_devptrs); @@ -946,8 +982,11 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, __nv_bfloat16* mm_b2 = upload_bf16_from_f32((const float*) m->mm_l2_b->data, mm_out, m->cuda_devptrs); bitvla_vit_cuda_set_mmproj(m->vit_cuda_ctx, mm_W1, mm_b1, mm_W2, mm_b2); - cudaMalloc(&m->d_vit_patches, (size_t) m->n_patches*patch_flat * sizeof(__nv_bfloat16)); - cudaMalloc(&m->d_vit_img_embeds, (size_t) m->n_patches*mm_out * sizeof(__nv_bfloat16)); + vit_ok = cudaMalloc(&m->d_vit_patches, (size_t) m->n_patches*patch_flat * sizeof(__nv_bfloat16)) == cudaSuccess && + cudaMalloc(&m->d_vit_img_embeds, (size_t) m->n_patches*mm_out * sizeof(__nv_bfloat16)) == cudaSuccess && + cudaGetLastError() == cudaSuccess; + } + if (vit_ok) { m->cuda_vit_ready = true; const size_t vit_packed_bytes = (size_t) m->vit_layers*( 4*(size_t) m->vit_hidden*m->vit_hidden/4 + @@ -958,6 +997,8 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, } else { bitvla_vit_cuda_free(m->vit_cuda_ctx); m->vit_cuda_ctx = nullptr; + cudaFree(m->d_vit_patches); cudaFree(m->d_vit_img_embeds); + m->d_vit_patches = nullptr; m->d_vit_img_embeds = nullptr; std::printf("vla(bitvla): CUDA ViT packing failed; falling back to CPU vision\n"); } } else { @@ -972,7 +1013,7 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, std::fprintf(stderr, "vla(bitvla): bitvla_lm_cuda_init failed; falling back to CPU LM\n"); } } else { - std::printf("vla(bitvla): no CUDA device - using CPU LM forward\n"); + std::printf("vla(bitvla): CUDA device %d unavailable (%d visible) - using CPU LM forward\n", m->cuda_dev, dev_count); } } @@ -1038,6 +1079,7 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, } std::memcpy(copy, t->data, nb); t->data = copy; + t->buffer = nullptr; m->cpu_kept_ptrs.push_back(copy); bytes_kept += nb; } @@ -1068,6 +1110,15 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { const auto t_start = clk::now(); stats = Stats{}; const bool timing_phase = (in.timing_detail == TimingDetail::PHASE); +#ifdef VLA_BITVLA_CUDA_KERNELS + if (cuda_lm_ready || cuda_vit_ready) { + if (cudaSetDevice(cuda_dev) != cudaSuccess) { + std::fprintf(stderr, "vla(bitvla): cudaSetDevice(%d) failed\n", cuda_dev); + return {}; + } + cudaGetLastError(); + } +#endif const char* _dump_dir = std::getenv("VLA_BITVLA_DUMP_DIR"); auto _dump_bin = [&](const char* name, const float* data, size_t nelem) { @@ -1152,11 +1203,13 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { std::vector patches_bf16((size_t) N * patch_flat); for (size_t i=0; i img_bf16((size_t) N * hidden_l); - cudaMemcpy(img_bf16.data(), d_vit_img_embeds, img_bf16.size()*sizeof(uint16_t), cudaMemcpyDeviceToHost); + if (cudaMemcpy(d_vit_patches, patches_bf16.data(), patches_bf16.size()*sizeof(uint16_t), cudaMemcpyHostToDevice) != cudaSuccess || + bitvla_vit_cuda_forward(vit_cuda_ctx, d_vit_patches, d_vit_img_embeds, 0) != 0 || + cudaMemcpy(img_bf16.data(), d_vit_img_embeds, img_bf16.size()*sizeof(uint16_t), cudaMemcpyDeviceToHost) != cudaSuccess) { + std::fprintf(stderr, "vla(bitvla): CUDA ViT forward failed (view %lld)\n", (long long) v); + return {}; + } float* dst = img_embeds_host.data()+(size_t) v * N * hidden_l; for (size_t i=0; i BitvlaModelArch::predict(const Inputs& in) { std::fprintf(stderr, "vla(bitvla): seq=%lld > lm_max_pos=%lld\n", (long long) seq, (long long) lm_max_pos); return {}; } +#ifdef VLA_BITVLA_CUDA_KERNELS + if (cuda_lm_ready && seq > cuda_max_seq && (packed_int2 || !weight_buf)) { + std::fprintf(stderr, "vla(bitvla): seq=%lld > CUDA LM max_seq=%d and no CPU LM weights to fall back on\n", + (long long) seq, cuda_max_seq); + return {}; + } +#endif // Both LM paths index the action slots as seq-2-n_action+i and neither // ggml_get_rows nor the CUDA gather bound-checks, so a short sequence would // read out of bounds and come back as plausible hidden states. @@ -1284,13 +1344,22 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { if (full_prefix) { - std::vector ids(in.lang_tokens, in.lang_tokens+n_lang_in); - for (int32_t id : ids) { + std::vector ids; + std::vector pos; + for (int64_t i=0; i= vocab_size) { std::fprintf(stderr, "vla(bitvla): prompt token %d out of vocab\n", id); return {}; } + ids.push_back(id); + pos.push_back(i); } - if (!emb_reader.fetch_rows_f32("token_embd.weight", ids, inputs_embeds.data(), hidden_l)) return {}; + std::vector rows(ids.size()*(size_t) hidden_l); + if (!emb_reader.fetch_rows_f32("token_embd.weight", ids, rows.data(), hidden_l)) return {}; + for (size_t k=0; k BitvlaModelArch::predict(const Inputs& in) { std::vector in_bf16((size_t) seq * hidden_l); for (size_t i=0; i aids(n_action); for (int64_t i=0; i out_bf16((size_t) n_action * hidden_l); - cudaMemcpy(out_bf16.data(), d_action_hidden, out_bf16.size()*sizeof(uint16_t), cudaMemcpyDeviceToHost); + if (cudaMemcpy(d_inputs_embeds, in_bf16.data(), in_bf16.size()*sizeof(uint16_t), cudaMemcpyHostToDevice) != cudaSuccess || + bitvla_lm_cuda_forward(lm_cuda_ctx, d_inputs_embeds, d_last_hidden, (int) seq, 0) != 0 || + cudaMemcpy(d_action_ids, aids.data(), n_action * sizeof(int32_t), cudaMemcpyHostToDevice) != cudaSuccess) { + std::fprintf(stderr, "vla(bitvla): CUDA LM forward failed\n"); + return {}; + } + bitvla_gather_rows_bf16(d_last_hidden, d_action_hidden, d_action_ids, (int) n_action, (int) hidden_l, 0); + if (cudaMemcpy(out_bf16.data(), d_action_hidden, out_bf16.size()*sizeof(uint16_t), cudaMemcpyDeviceToHost) != cudaSuccess || + cudaGetLastError() != cudaSuccess) { + std::fprintf(stderr, "vla(bitvla): CUDA LM forward failed\n"); + return {}; + } for (size_t i=0; i Date: Mon, 28 Sep 2026 14:53:57 +0700 Subject: [PATCH 10/61] fix BF16 CUDA hook overflow and races --- src/cuda/vla_cuda_bf16.cu | 99 +++++++++++++++++++++--------------- tests/test_bf16_cuda_ops.cpp | 19 ++++--- 2 files changed, 69 insertions(+), 49 deletions(-) diff --git a/src/cuda/vla_cuda_bf16.cu b/src/cuda/vla_cuda_bf16.cu index 70a3d63..70b3766 100644 --- a/src/cuda/vla_cuda_bf16.cu +++ b/src/cuda/vla_cuda_bf16.cu @@ -30,7 +30,9 @@ // cannot break it the way an anchored source patch would. // // Every entry point returns false for anything it does not handle, and ggml -// then runs the op exactly as it would have. Nothing here changes the F32 path. +// then runs the op exactly as it would have. The exception is mul_mat with a +// BF16 result: ggml has no fallback for it, so an unsupported one aborts. +// Nothing here changes the F32 path. // // Accumulation is float throughout: only operand and result *storage* is BF16, // never a reduction. @@ -44,6 +46,8 @@ #include #include +#include +#include // Must match the typedef the hook patch inserts into ggml-cuda.cu. extern "C" { @@ -266,7 +270,8 @@ bool bin_bcast(ggml_tensor * dst, cudaStream_t stream) { g.ok && dst->ne[0]%8 == 0 && es(src0, 0) == 1 && es(dst, 0) == 1 && es(src1, 0) == 1 && src1->ne[0] == dst->ne[0] && - es(src0, 1)%8 == 0 && es(dst, 1)%8 == 0 && es(src1, 1)%8 == 0 && + es(src0, 1)%8 == 0 && es(src0, 2)%8 == 0 && es(src0, 3)%8 == 0 && + es(dst, 1)%8 == 0 && es(dst, 2)%8 == 0 && es(dst, 3)%8 == 0 && es(src1, 1)%8 == 0 && ((uintptr_t) src0->data%16) == 0 && ((uintptr_t) dst->data%16) == 0; if (vec8_shape) { const int64_t nvec = dst->ne[0]/8; @@ -619,29 +624,32 @@ bool norm(ggml_tensor * dst, cudaStream_t stream) { // dst->data unconditionally, which for a BF16 dst is both wrong and twice the // bytes the allocator reserved. -cublasHandle_t g_handle = nullptr; - bool mul_mat(ggml_tensor * dst, cudaStream_t stream) { + if (dst->type != GGML_TYPE_BF16) + return false; const ggml_tensor * src0 = dst->src[0]; const ggml_tensor * src1 = dst->src[1]; - if (!src0 || !src1) - return false; - if (dst->type != GGML_TYPE_BF16 || src0->type != GGML_TYPE_BF16 || src1->type != GGML_TYPE_BF16) { - return false; - } - if (!ggml_is_contiguous(src0) || !ggml_is_contiguous(src1) || !ggml_is_contiguous(dst)) { - return false; + if (src0->type != GGML_TYPE_BF16 || src1->type != GGML_TYPE_BF16 || + !ggml_is_contiguous(src0) || !ggml_is_contiguous(src1) || !ggml_is_contiguous(dst)) { + GGML_ABORT("vla: unsupported BF16 mul_mat %s", dst->name); } // src0 is either shared across the whole batch or batched 1:1 with src1 const bool batch_ok = (src0->ne[2] == 1 && src0->ne[3] == 1) || (src0->ne[2] == src1->ne[2] && src0->ne[3] == src1->ne[3]); if (!batch_ok) - return false; - - if (!g_handle && cublasCreate(&g_handle) != CUBLAS_STATUS_SUCCESS) - return false; - if (cublasSetStream(g_handle, stream) != CUBLAS_STATUS_SUCCESS) - return false; + GGML_ABORT("vla: unsupported BF16 mul_mat %s", dst->name); + + int dev = 0; + if (cudaGetDevice(&dev) != cudaSuccess) + GGML_ABORT("vla: cudaGetDevice failed for %s", dst->name); + static std::mutex mu; + static std::map handles; + std::lock_guard lock(mu); + cublasHandle_t & handle = handles[dev]; + if ((!handle && cublasCreate(&handle) != CUBLAS_STATUS_SUCCESS) || + cublasSetStream(handle, stream) != CUBLAS_STATUS_SUCCESS) { + GGML_ABORT("vla: cuBLAS setup failed for %s", dst->name); + } const int64_t ne00 = src0->ne[0], ne01 = src0->ne[1]; const int64_t ne10 = src1->ne[0], ne11 = src1->ne[1]; @@ -657,7 +665,7 @@ bool mul_mat(ggml_tensor * dst, cudaStream_t stream) { cublasStatus_t st; if (n_batch == 1) { - st = cublasGemmEx(g_handle, CUBLAS_OP_T, CUBLAS_OP_N, + st = cublasGemmEx(handle, CUBLAS_OP_T, CUBLAS_OP_N, (int) ne01, (int) ne11, (int) ne10, &alpha, a, CUDA_R_16BF, (int) ne00, b, CUDA_R_16BF, (int) ne10, @@ -666,7 +674,7 @@ bool mul_mat(ggml_tensor * dst, cudaStream_t stream) { } else { // stride_a == 0 broadcasts one weight matrix across the batch const long long stride_a = (src0->ne[2] == 1 && src0->ne[3] == 1) ? 0 : (long long) ne00*ne01; - st = cublasGemmStridedBatchedEx(g_handle, CUBLAS_OP_T, CUBLAS_OP_N, + st = cublasGemmStridedBatchedEx(handle, CUBLAS_OP_T, CUBLAS_OP_N, (int) ne01, (int) ne11, (int) ne10, &alpha, a, CUDA_R_16BF, (int) ne00, stride_a, b, CUDA_R_16BF, (int) ne10, (long long) ne10*ne11, @@ -674,32 +682,19 @@ bool mul_mat(ggml_tensor * dst, cudaStream_t stream) { (int) n_batch, CUBLAS_COMPUTE_32F, CUBLAS_GEMM_DEFAULT_TENSOR_OP); } - return st == CUBLAS_STATUS_SUCCESS; + if (st != CUBLAS_STATUS_SUCCESS) + GGML_ABORT("vla: BF16 mul_mat %s failed (%d)", dst->name, (int) st); + return true; } -} // namespace - -// --------------------------------------------------------------------------- -// hook entry point -// --------------------------------------------------------------------------- - -extern "C" bool vla_cuda_bf16_fused_binbcast(ggml_tensor * dst, int n_fuse, void * stream_v) { - if (!dst) - return false; - cudaStream_t stream = (cudaStream_t) stream_v; - - switch (dst->op) { - case GGML_OP_ADD: return fused_bin_bcast(dst, n_fuse, stream); - case GGML_OP_MUL: return fused_bin_bcast(dst, n_fuse, stream); - default: return false; - } +bool launched(const ggml_tensor * dst) { + const cudaError_t err = cudaGetLastError(); + if (err != cudaSuccess) + GGML_ABORT("vla: %s %s failed: %s", ggml_op_desc(dst), dst->name, cudaGetErrorString(err)); + return true; } -extern "C" bool vla_cuda_bf16_forward(ggml_tensor * dst, void * stream_v) { - if (!dst) - return false; - cudaStream_t stream = (cudaStream_t) stream_v; - +bool forward(ggml_tensor * dst, cudaStream_t stream) { switch (dst->op) { case GGML_OP_MUL_MAT: return mul_mat(dst, stream); case GGML_OP_ADD: return bin_bcast(dst, stream); @@ -719,6 +714,28 @@ extern "C" bool vla_cuda_bf16_forward(ggml_tensor * dst, void * stream_v) { } } +} // namespace + +// --------------------------------------------------------------------------- +// hook entry point +// --------------------------------------------------------------------------- + +extern "C" bool vla_cuda_bf16_fused_binbcast(ggml_tensor * dst, int n_fuse, void * stream_v) { + if (!dst) + return false; + cudaStream_t stream = (cudaStream_t) stream_v; + + switch (dst->op) { + case GGML_OP_ADD: return fused_bin_bcast(dst, n_fuse, stream) && launched(dst); + case GGML_OP_MUL: return fused_bin_bcast(dst, n_fuse, stream) && launched(dst); + default: return false; + } +} + +extern "C" bool vla_cuda_bf16_forward(ggml_tensor * dst, void * stream_v) { + return dst && forward(dst, (cudaStream_t) stream_v) && launched(dst); +} + namespace vla { // Called once, after the CUDA backend is up. Idempotent. diff --git a/tests/test_bf16_cuda_ops.cpp b/tests/test_bf16_cuda_ops.cpp index b72b29a..a799cbc 100644 --- a/tests/test_bf16_cuda_ops.cpp +++ b/tests/test_bf16_cuda_ops.cpp @@ -40,17 +40,18 @@ namespace { constexpr int64_t K = 128; // reduction / feature dim constexpr int64_t M = 64; // rows -constexpr int64_t N = 96; // output features +constexpr int64_t N = 100; // output features constexpr float EPS = 1e-5f; // Builds the chain at `at`. Inputs are always F32 tensors; the BF16 run casts in // at the top and back out at the bottom, exactly as the models do. ggml_tensor * build_chain(ggml_context * C, ggml_tensor * x, ggml_tensor * w_norm, - ggml_tensor * W, ggml_tensor * bias, ggml_type at) { + ggml_tensor * W, ggml_tensor * bias, ggml_tensor * bias2, ggml_type at) { ggml_tensor * h = vla::as_type(C, x, at); h = ggml_mul(C, ggml_rms_norm(C, h, EPS), w_norm); // RMS_NORM (+ MUL, F32 weight) - h = vla::mm_act(C, W, h, at); // MUL_MAT - h = ggml_add(C, h, bias); // ADD, F32 bias + h = ggml_add(C, vla::mm_act(C, W, ggml_reshape_3d(C, h, K, M/2, 2), at), + ggml_reshape_3d(C, vla::mm_act(C, W, h, at), N, M/2, 2)); + h = ggml_add(C, ggml_add(C, h, bias), bias2); // ADD, F32 bias h = ggml_silu(C, h); // UNARY SILU h = ggml_scale(C, h, 0.5f); // SCALE h = ggml_add(C, ggml_norm(C, h, EPS), h); // NORM (+ ADD, BF16 x BF16) @@ -70,9 +71,10 @@ std::vector run(ggml_backend_t backend, ggml_type at, // typed-GEMM path; that mirrors a BF16 checkpoint. ggml_tensor * W = ggml_new_tensor_2d(C, at, K, N); ggml_tensor * bias = ggml_new_tensor_1d(C, GGML_TYPE_F32, N); - for (ggml_tensor * t : {x, w_norm, W, bias}) ggml_set_input(t); + ggml_tensor * bias2 = ggml_new_tensor_1d(C, GGML_TYPE_F32, N); + for (ggml_tensor * t : {x, w_norm, W, bias, bias2}) ggml_set_input(t); - ggml_tensor * out = build_chain(C, x, w_norm, W, bias, at); + ggml_tensor * out = build_chain(C, x, w_norm, W, bias, bias2, at); ggml_set_output(out); ggml_cgraph * gf = ggml_new_graph(C); @@ -87,6 +89,7 @@ std::vector run(ggml_backend_t backend, ggml_type at, ggml_backend_tensor_set(x, hx.data(), 0, ggml_nbytes(x)); ggml_backend_tensor_set(w_norm, hw.data(), 0, ggml_nbytes(w_norm)); ggml_backend_tensor_set(bias, hb.data(), 0, ggml_nbytes(bias)); + ggml_backend_tensor_set(bias2, hb.data()+N, 0, ggml_nbytes(bias2)); if (at == GGML_TYPE_BF16) { std::vector t(hW.size()); ggml_fp32_to_bf16_row(hW.data(), t.data(), (int64_t) hW.size()); @@ -119,11 +122,11 @@ int main() { } vla::cuda_register_bf16_ops(); - std::vector hx((size_t) K*M), hw((size_t) K), hW((size_t) K*N), hb((size_t) N); + std::vector hx((size_t) K*M), hw((size_t) K), hW((size_t) K*N), hb((size_t) N*2); for (size_t i = 0; i < hx.size(); ++i) hx[i] = ((float) ((i*37) % 23) - 11.0f) / 8.0f; for (size_t i = 0; i < hw.size(); ++i) hw[i] = 0.5f + ((float) ((i*11) % 7)) / 16.0f; for (size_t i = 0; i < hW.size(); ++i) hW[i] = ((float) ((i*53) % 19) - 9.0f) / 32.0f; - for (size_t i = 0; i < hb.size(); ++i) hb[i] = ((float) ((i*17) % 5) - 2.0f) / 16.0f; + for (size_t i = 0; i < hb.size(); ++i) hb[i] = ((float) ((i*17) % 7) - 3.0f) / 16.0f; const std::vector ref = run(backend, GGML_TYPE_F32, hx, hw, hW, hb); const std::vector got = run(backend, GGML_TYPE_BF16, hx, hw, hW, hb); From 06969f6f8780fda1669d24fae3b6b2605c2cc579 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Mon, 28 Sep 2026 14:53:57 +0700 Subject: [PATCH 11/61] fix quantizer, converters and eval client --- eval/client/benchmark.py | 4 + eval/client/run_ALOHA_client_direct.py | 19 ++-- eval/client/vla_cpp_client.py | 120 +++++++++--------------- pyproject.toml | 6 +- scripts/build_windows_snapdragon.ps1 | 2 +- scripts/convert_bitvla_to_gguf.py | 18 +--- scripts/convert_evo1_to_gguf.py | 4 +- scripts/convert_octo_to_gguf.py | 12 ++- scripts/convert_openvla_oft_to_gguf.py | 5 +- scripts/convert_pi05_to_gguf.py | 68 ++++++++------ scripts/convert_pi0_to_gguf.py | 125 ++++++------------------- scripts/convert_smolvla_to_gguf.py | 35 ++----- scripts/convert_turbovla_to_gguf.py | 83 +++++----------- scripts/gguf_blocks.py | 56 +++++++++-- scripts/gguf_common.py | 15 ++- scripts/quantize_gguf.py | 14 +-- tests/py/test_converters.py | 38 ++++++++ 17 files changed, 288 insertions(+), 336 deletions(-) diff --git a/eval/client/benchmark.py b/eval/client/benchmark.py index 41873b5..37b8c95 100644 --- a/eval/client/benchmark.py +++ b/eval/client/benchmark.py @@ -244,6 +244,9 @@ def main() -> int: if done or trunc: obs, _info = env.reset() + if args.backend == "vla-cpp" and inner is not None: + inner._last_response = None + # The PyTorch server times select_action internally (see # pytorch_ref/utils/service.py). Drop the samples accumulated during warmup # so the compiled variants aren't charged for their first-call compilation. @@ -278,6 +281,7 @@ def main() -> int: if args.backend == "vla-cpp" and inner is not None: r = inner._last_response if r is not None: + inner._last_response = None server_latencies.append({ "total": r.latency_ms_total, "vision": r.latency_ms_vision, diff --git a/eval/client/run_ALOHA_client_direct.py b/eval/client/run_ALOHA_client_direct.py index cb416f7..fef1d03 100644 --- a/eval/client/run_ALOHA_client_direct.py +++ b/eval/client/run_ALOHA_client_direct.py @@ -419,6 +419,7 @@ def __init__(self, args: argparse.Namespace): self._async_lock = Lock() self._async_chunk = None # prefetched raw chunk (HORIZON × action_dim) self._async_obs = None # obs snapshot captured at trigger time + self._async_gen = 0 self._async_err: Exception | None = None self._async_cached = None # chunk ready for next _run_inference call self._async_worker = Thread(target=self._async_infer_worker, daemon=True) @@ -498,7 +499,7 @@ def _async_infer_worker(self): self._async_trigger.clear() with self._async_lock: - obs_snap = self._async_obs + gen, obs_snap = self._async_gen, self._async_obs if obs_snap is None: continue @@ -513,6 +514,8 @@ def _async_infer_worker(self): err = e with self._async_lock: + if gen != self._async_gen: + continue self._async_chunk = chunk self._async_err = err self._async_ready.set() @@ -540,6 +543,7 @@ def _trigger_next_infer(self, predicted_left=None, predicted_right=None): right_state, ) with self._async_lock: + self._async_gen += 1 self._async_obs = obs_snap self._async_chunk = None self._async_err = None @@ -548,17 +552,12 @@ def _trigger_next_infer(self, predicted_left=None, predicted_right=None): def _collect_prefetched(self, timeout: float = 2.0): """ - Wait for the prefetched chunk. On timeout, clear the cache so the - next _run_inference falls back to a fresh synchronous request. + Wait for the prefetched chunk. Returns the raw chunk or raises on error. """ if not self._async_ready.wait(timeout=timeout): - self.log.warning( - f"async prefetch timed out after {timeout:.1f}s - " - "next call will re-trigger synchronously" - ) - self._async_cached = None - raise TimeoutError("async inference timed out") + self.log.warning(f"async prefetch slower than {timeout:.1f}s, waiting") + self._async_ready.wait() with self._async_lock: err = self._async_err chunk = self._async_chunk @@ -634,7 +633,7 @@ def _run_inference_async(self): self._async_cached = None self.log.debug("async: using prefetched chunk") else: - # First call or after a timeout fallback: trigger and wait. + # First call or after a failed prefetch: trigger and wait. with self.lock: if self.front_rgb is None or self.wrist_left_rgb is None or self.left_state is None: self._log_waiting(self.front_rgb is not None, diff --git a/eval/client/vla_cpp_client.py b/eval/client/vla_cpp_client.py index b295905..9c9c521 100644 --- a/eval/client/vla_cpp_client.py +++ b/eval/client/vla_cpp_client.py @@ -172,6 +172,8 @@ def __init__( self.sock = self.ctx.socket(zmq.REQ) self.sock.setsockopt(zmq.LINGER, 0) self.sock.setsockopt(zmq.RCVTIMEO, recv_timeout_ms) + self.sock.setsockopt(zmq.REQ_RELAXED, 1) + self.sock.setsockopt(zmq.REQ_CORRELATE, 1) self.sock.connect(vla_addr) print(f"vla-cpp-direct[arch={arch}]: connected to {vla_addr}", flush=True) @@ -381,6 +383,13 @@ def _action_unnorm(chunk, mn=action_min, mx=action_max): q01 = self._gr00t_quantile(action_stats, modalities, "q01") q99 = self._gr00t_quantile(action_stats, modalities, "q99") act_dim = int(q01.size) + state_stats = blob[key]["state"] + state_keys, state_dims = self._gr00t_modality_layout(state_stats) + state_cols = {} + s_off = 0 + for m, dim in zip(state_keys, state_dims): + state_cols[m] = slice(s_off, s_off + dim) + s_off += dim # Checkpoints trained with use_relative_action predict, for the # modalities listed in meta/relative_stats.json, the offset from the @@ -401,7 +410,7 @@ def _action_unnorm(chunk, mn=action_min, mx=action_max): horizon = min(len(rel_stats[m]["min"]) for m in rel_names) q01_t = np.tile(q01, (horizon, 1)).astype(np.float32) q99_t = np.tile(q99, (horizon, 1)).astype(np.float32) - is_rel = np.zeros(act_dim, dtype=bool) + rel_cols = [] off = 0 for m, dim in zip(modalities, mod_dims): if m in rel_stats: @@ -419,13 +428,18 @@ def _action_unnorm(chunk, mn=action_min, mx=action_max): raise ValueError( f"relative stats for {m!r} are {a.shape[1]}-wide, " f"statistics say {dim}") + sc = state_cols.get(m) + if sc is None or sc.stop - sc.start != dim: + raise ValueError( + f"relative modality {m!r} needs a {dim}-wide state.{m}, " + f"state statistics have {list(zip(state_keys, state_dims))}") q01_t[:, off:off + dim] = a q99_t[:, off:off + dim] = b - is_rel[off:off + dim] = True + rel_cols.append((slice(off, off + dim), sc)) off += dim rng_t = (q99_t - q01_t).astype(np.float32) - def _unnorm(chunk_132, q01_t=q01_t, rng_t=rng_t, is_rel=is_rel, + def _unnorm(chunk_132, q01_t=q01_t, rng_t=rng_t, rel_cols=tuple(rel_cols), act_dim=act_dim, horizon=horizon): n = min(len(chunk_132), horizon) norm = np.clip(chunk_132[:n, :act_dim].astype(np.float32), -1.0, 1.0) @@ -434,7 +448,9 @@ def _unnorm(chunk_132, q01_t=q01_t, rng_t=rng_t, is_rel=is_rel, if ref is None: raise RuntimeError("relative actions need the observation state; " "none was recorded for this request") - raw[:, is_rel] += np.asarray(ref, dtype=np.float32)[:act_dim][is_rel] + ref = np.asarray(ref, dtype=np.float32) + for a_cols, s_cols in rel_cols: + raw[:, a_cols] += ref[s_cols] return raw.astype(np.float32) else: rng = (q99 - q01).astype(np.float32) @@ -456,8 +472,6 @@ def _unnorm(chunk_132: np.ndarray, q01=q01, q99=q99, rng=rng, f"relative={rel_names or 'none'}]", flush=True) - state_stats = blob[key]["state"] - state_keys, state_dims = self._gr00t_modality_layout(state_stats) self._gr00t_state_keys = tuple(state_keys) self._gr00t_state_dims = tuple(state_dims) s_q01 = self._gr00t_quantile(state_stats, state_keys, "q01") @@ -1094,33 +1108,31 @@ def _predict_chunk_octo(self, observations: dict[str, Any]) -> np.ndarray: return (np.array(resp.action_chunk, dtype=np.float32) .reshape(resp.chunk_size, resp.action_dim)) + def _oft_image(self, observations: dict[str, Any], key: str) -> np.ndarray: + if key not in observations: + raise KeyError(f"image key '{key}' missing; got {list(observations.keys())}") + img = observations[key] + if isinstance(img, torch.Tensor): + img = img.numpy() + img = np.asarray(img, dtype=np.float32) + if img.ndim != 3 or img.shape[0] != 3: + raise ValueError(f"{key}: expected CHW float [3, H, W], got {img.shape}") + img_u8 = np.clip(np.transpose(img, (1, 2, 0)) * 255.0 + 0.5, 0, 255).astype(np.uint8) + if img_u8.shape[0] != self.image_size or img_u8.shape[1] != self.image_size: + img_u8 = np.array(Image.fromarray(img_u8, mode="RGB").resize( + (self.image_size, self.image_size), resample=Image.LANCZOS), dtype=np.uint8) + h, w = img_u8.shape[:2] + s = 0.9 ** 0.5 + new_h, new_w = int(round(h * s)), int(round(w * s)) + off_h, off_w = (h - new_h) // 2, (w - new_w) // 2 + cropped = img_u8[off_h:off_h + new_h, off_w:off_w + new_w] + img_u8 = np.array(Image.fromarray(cropped, mode="RGB").resize( + (w, h), resample=Image.BILINEAR), dtype=np.uint8) + return np.ascontiguousarray(img_u8, dtype=np.uint8) + def _predict_chunk_bitvla(self, observations: dict[str, Any]) -> np.ndarray: - images_u8: list[np.ndarray] = [] - for key in self.image_keys[:BITVLA_N_VIEWS]: - if key not in observations: - raise KeyError(f"image key '{key}' missing; got {list(observations.keys())}") - img = observations[key] - if isinstance(img, torch.Tensor): - img = img.numpy() - img = np.asarray(img, dtype=np.float32) - if img.ndim != 3 or img.shape[0] != 3: - raise ValueError(f"{key}: expected CHW float [3, H, W], got {img.shape}") - img_u8 = np.clip(np.transpose(img, (1, 2, 0)) * 255.0 + 0.5, 0, 255).astype(np.uint8) - - if img_u8.shape[0] != self.image_size or img_u8.shape[1] != self.image_size: - img_u8 = np.array(Image.fromarray(img_u8, mode="RGB").resize( - (self.image_size, self.image_size), resample=Image.LANCZOS), - dtype=np.uint8) - - h, w = img_u8.shape[:2] - s = 0.9 ** 0.5 - new_h, new_w = int(round(h * s)), int(round(w * s)) - off_h, off_w = (h - new_h) // 2, (w - new_w) // 2 - cropped = img_u8[off_h:off_h + new_h, off_w:off_w + new_w] - img_u8 = np.array(Image.fromarray(cropped, mode="RGB").resize( - (w, h), resample=Image.BILINEAR), dtype=np.uint8) - images_u8.append(np.ascontiguousarray(img_u8, dtype=np.uint8)) + images_u8 = [self._oft_image(observations, k) for k in self.image_keys[:BITVLA_N_VIEWS]] s = observations["observation.state"] if isinstance(s, torch.Tensor): @@ -1167,28 +1179,7 @@ def _predict_chunk_bitvla(self, observations: dict[str, Any]) -> np.ndarray: return chunk def _predict_chunk_vla_adapter(self, observations: dict[str, Any]) -> np.ndarray: - images_u8: list[np.ndarray] = [] - for key in self.image_keys[:VLA_ADAPTER_N_VIEWS]: - if key not in observations: - raise KeyError(f"image key '{key}' missing; got {list(observations.keys())}") - img = observations[key] - if isinstance(img, torch.Tensor): - img = img.numpy() - img = np.asarray(img, dtype=np.float32) - if img.ndim != 3 or img.shape[0] != 3: - raise ValueError(f"{key}: expected CHW float [3, H, W], got {img.shape}") - img_u8 = np.clip(np.transpose(img, (1, 2, 0)) * 255.0 + 0.5, 0, 255).astype(np.uint8) - if img_u8.shape[0] != self.image_size or img_u8.shape[1] != self.image_size: - img_u8 = np.array(Image.fromarray(img_u8, mode="RGB").resize( - (self.image_size, self.image_size), resample=Image.LANCZOS), dtype=np.uint8) - h, w = img_u8.shape[:2] - s = 0.9 ** 0.5 - new_h, new_w = int(round(h * s)), int(round(w * s)) - off_h, off_w = (h - new_h) // 2, (w - new_w) // 2 - cropped = img_u8[off_h:off_h + new_h, off_w:off_w + new_w] - img_u8 = np.array(Image.fromarray(cropped, mode="RGB").resize( - (w, h), resample=Image.BILINEAR), dtype=np.uint8) - images_u8.append(np.ascontiguousarray(img_u8, dtype=np.uint8)) + images_u8 = [self._oft_image(observations, k) for k in self.image_keys[:VLA_ADAPTER_N_VIEWS]] st = observations["observation.state"] if isinstance(st, torch.Tensor): @@ -1227,28 +1218,7 @@ def _predict_chunk_vla_adapter(self, observations: dict[str, Any]) -> np.ndarray return chunk def _predict_chunk_openvla_oft(self, observations: dict[str, Any]) -> np.ndarray: - images_u8: list[np.ndarray] = [] - for key in self.image_keys[:OPENVLA_OFT_N_VIEWS]: - if key not in observations: - raise KeyError(f"image key '{key}' missing; got {list(observations.keys())}") - img = observations[key] - if isinstance(img, torch.Tensor): - img = img.numpy() - img = np.asarray(img, dtype=np.float32) - if img.ndim != 3 or img.shape[0] != 3: - raise ValueError(f"{key}: expected CHW float [3, H, W], got {img.shape}") - img_u8 = np.clip(np.transpose(img, (1, 2, 0)) * 255.0 + 0.5, 0, 255).astype(np.uint8) - if img_u8.shape[0] != self.image_size or img_u8.shape[1] != self.image_size: - img_u8 = np.array(Image.fromarray(img_u8, mode="RGB").resize( - (self.image_size, self.image_size), resample=Image.LANCZOS), dtype=np.uint8) - h, w = img_u8.shape[:2] - s = 0.9 ** 0.5 - new_h, new_w = int(round(h * s)), int(round(w * s)) - off_h, off_w = (h - new_h) // 2, (w - new_w) // 2 - cropped = img_u8[off_h:off_h + new_h, off_w:off_w + new_w] - img_u8 = np.array(Image.fromarray(cropped, mode="RGB").resize( - (w, h), resample=Image.BILINEAR), dtype=np.uint8) - images_u8.append(np.ascontiguousarray(img_u8, dtype=np.uint8)) + images_u8 = [self._oft_image(observations, k) for k in self.image_keys[:OPENVLA_OFT_N_VIEWS]] st = observations["observation.state"] if isinstance(st, torch.Tensor): diff --git a/pyproject.toml b/pyproject.toml index 67f7e31..d318311 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -10,7 +10,7 @@ dependencies = ["numpy>=1.24"] [project.optional-dependencies] # scripts/convert_*_to_gguf.py and the gguf_common / gguf_blocks modules they share convert = [ - "torch>=2.5", + "torch>=2.6", "safetensors>=0.4.3", "gguf>=0.17", "transformers>=4.57", @@ -22,8 +22,8 @@ client = [ "pyzmq>=25", "msgpack>=1.1", "pillow>=10", - "torch>=2.5", - "torchvision>=0.20", + "torch>=2.6", + "torchvision>=0.21", # AutoTokenizer/AutoProcessor only. vla_jepa passes processor_kwargs to # apply_chat_template, which 5.4 is the first release to accept. "transformers>=5.4", diff --git a/scripts/build_windows_snapdragon.ps1 b/scripts/build_windows_snapdragon.ps1 index ed2de01..71c097b 100644 --- a/scripts/build_windows_snapdragon.ps1 +++ b/scripts/build_windows_snapdragon.ps1 @@ -70,7 +70,7 @@ Set-Location $root $dir = "build-wos-$Backend" $jobs = $env:NUMBER_OF_PROCESSORS -$flags = "-march=armv8.7a+fp16+dotprod+i8mm -fvectorize -ffp-model=fast -D_GNU_SOURCE" +$flags = "-march=armv8.7a+fp16+dotprod+i8mm -fvectorize -ffp-model=fast -fno-finite-math-only -D_GNU_SOURCE" # protobuf and abseil must be built by clang too, with the triplet in # cmake/vcpkg-triplets: clang code does not link against an MSVC-built protobuf. # ZeroMQ is a C API and comes from the stock triplet, searched second. diff --git a/scripts/convert_bitvla_to_gguf.py b/scripts/convert_bitvla_to_gguf.py index fe25dcf..3e29b74 100644 --- a/scripts/convert_bitvla_to_gguf.py +++ b/scripts/convert_bitvla_to_gguf.py @@ -15,9 +15,6 @@ from __future__ import annotations -import re -from pathlib import Path - import numpy as np import torch @@ -25,6 +22,7 @@ from gguf_common import ( add, arg_parser, + find_sidecar, finish, kv_f32, kv_prefix, @@ -196,14 +194,6 @@ def _add_bit_fused(writer, base: str, Ws: list[torch.Tensor]) -> None: packed, scales = pack_fused_projection(Ws) _add_packed(writer, base, packed, scales) -def _find_sidecar(ckpt: Path, stem: str) -> Path | None: - - cands = sorted( - ckpt.glob(f"{stem}--*_checkpoint.pt"), - key=lambda p: int(m.group(1)) if (m := re.search(r"--(\d+)_checkpoint\.pt$", p.name)) else -1, - ) - return cands[-1] if cands else None - def _add_kv(writer, statistics_json: str, processor_json: str, preproc_json: str) -> None: kv_u32( @@ -294,10 +284,8 @@ def main() -> int: W = load_safetensors(ckpt) print(f" {len(W)} main tensors") - ah_path = _find_sidecar(ckpt, "action_head") - pp_path = _find_sidecar(ckpt, "proprio_projector") - if ah_path is None or pp_path is None: - raise SystemExit(f"missing action_head/proprio_projector sidecars in {ckpt}") + ah_path = find_sidecar(ckpt, "action_head") + pp_path = find_sidecar(ckpt, "proprio_projector") print(f" sidecars: {ah_path.name}, {pp_path.name}") AH = load_pt_module(ah_path) PP = load_pt_module(pp_path) diff --git a/scripts/convert_evo1_to_gguf.py b/scripts/convert_evo1_to_gguf.py index 5ce88ee..7b0c1c4 100644 --- a/scripts/convert_evo1_to_gguf.py +++ b/scripts/convert_evo1_to_gguf.py @@ -137,13 +137,13 @@ def main() -> int: cfg["mlp_head_hidden"] = int(cfg_json.get("hidden_dim", 1024)) cfg["num_inference_timesteps"] = int(cfg_json.get("num_inference_timesteps", NUM_INFERENCE_TIMESTEPS)) cfg["image_size"] = int(cfg_json.get("image_size", VIT["image_size"])) - cfg["dit_heads"] = DIT_HEADS + cfg["dit_heads"] = int(cfg_json.get("num_heads", DIT_HEADS)) cfg["proj_ln_eps"] = PROJ_LN_EPS if cfg["action_dim"] != cfg["horizon"] * cfg["per_action_dim"]: raise SystemExit(f"action_dim {cfg['action_dim']} != horizon*per_action_dim {cfg['horizon']*cfg['per_action_dim']}") print(f"loading {pt_path} ...") - module = torch.load(pt_path, map_location="cpu", weights_only=False)["module"] + module = torch.load(pt_path, map_location="cpu", weights_only=True)["module"] keys = set(module.keys()) print(f" {len(module)} tensors") diff --git a/scripts/convert_octo_to_gguf.py b/scripts/convert_octo_to_gguf.py index cf4fb90..3d72402 100644 --- a/scripts/convert_octo_to_gguf.py +++ b/scripts/convert_octo_to_gguf.py @@ -16,6 +16,7 @@ import torch import gguf +from gguf_common import add_f32 ARCH = "octo" MODEL_ID = "hf://rail-berkeley/octo-small-1.5" @@ -90,10 +91,6 @@ def _add_meta(writer: gguf.GGUFWriter, key: str, value: Any) -> None: raise TypeError(f"unsupported metadata {full}={value!r}") -def _f32(t: torch.Tensor) -> np.ndarray: - return t.detach().to(dtype=torch.float32, device="cpu").contiguous().numpy() - - def _embed_tokenizer(writer: gguf.GGUFWriter, tokenizer_name: str = "t5-base") -> None: """Embed the raw T5 SentencePiece unigram model (spiece.model) as a UINT8 GGUF array (not a GGUF string: the serialized proto contains embedded NUL bytes, @@ -466,6 +463,11 @@ def main() -> int: OCTO_META["action.head_type"] = head_type OCTO_META["action.horizon"] = int(head_cfg["kwargs"]["action_horizon"]) OCTO_META["action.dim"] = int(head_cfg["kwargs"]["action_dim"]) + OCTO_META["diffusion.steps"] = int(head_cfg["kwargs"].get("diffusion_steps", 20)) + OCTO_META["diffusion.max_action"] = float(head_cfg["kwargs"].get("max_action", 5.0)) + mc = m.config["model"] + if not (mc.get("repeat_task_tokens") and mc.get("use_correct_attention")) or mc.get("readouts") != {"action": 1}: + raise SystemExit("vla.cpp Octo needs repeat_task_tokens, use_correct_attention and readouts={'action': 1}") print(f"action head: {head_cfg['name']} -> head_type={head_type} " f"horizon={OCTO_META['action.horizon']} dim={OCTO_META['action.dim']}") @@ -523,7 +525,7 @@ def main() -> int: rows = [] for src, dst in sorted(mapped.items(), key=lambda kv: kv[1]): tensor = sd[src] - writer.add_tensor(dst, _f32(tensor), raw_dtype=gguf.GGMLQuantizationType.F32) + add_f32(writer, dst, tensor) rows.append({"state_dict": src, "gguf": dst, "shape": list(tensor.shape)}) print(f"map {src} {tuple(tensor.shape)} -> {dst}") diff --git a/scripts/convert_openvla_oft_to_gguf.py b/scripts/convert_openvla_oft_to_gguf.py index 5bcf3a5..c6d882b 100644 --- a/scripts/convert_openvla_oft_to_gguf.py +++ b/scripts/convert_openvla_oft_to_gguf.py @@ -21,6 +21,7 @@ from gguf_common import ( add_bf16, arg_parser, + find_sidecar, finish, kv_prefix, kv_u32, @@ -101,8 +102,8 @@ def main() -> int: ckpt = args.ckpt.resolve() out = resolve_out(args, ckpt, ARCH) - ah_path = args.action_head or next(ckpt.glob("action_head--*checkpoint.pt")) - pp_path = args.proprio or next(ckpt.glob("proprio_projector--*checkpoint.pt")) + ah_path = args.action_head or find_sidecar(ckpt, "action_head") + pp_path = args.proprio or find_sidecar(ckpt, "proprio_projector") stats_path = ckpt / "dataset_statistics.json" require(stats_path) diff --git a/scripts/convert_pi05_to_gguf.py b/scripts/convert_pi05_to_gguf.py index e8e9280..48e8eb9 100644 --- a/scripts/convert_pi05_to_gguf.py +++ b/scripts/convert_pi05_to_gguf.py @@ -23,7 +23,9 @@ from safetensors import safe_open from gguf_blocks import ( + lerobot_stats, norm_eps, + pi_root, probe_paligemma_vision, write_decoder_blocks, write_paligemma_vision, @@ -89,11 +91,11 @@ "action_out_proj.bias", ] -def _write_adarms_blocks(writer, sf, n_layers: int) -> None: +def _write_adarms_blocks(writer, sf, aex: str, n_layers: int) -> None: for i in range(n_layers): for src_suf, dst_suf in AEX_MAP: - add(writer, f"aex.blk.{i}.{dst_suf}", sf.get_tensor(f"{PFX_AEX}.layers.{i}.{src_suf}")) + add(writer, f"aex.blk.{i}.{dst_suf}", sf.get_tensor(f"{aex}.layers.{i}.{src_suf}")) def _load_dataset_stats( stats_json: Optional[Path], @@ -173,6 +175,10 @@ def main() -> int: cfg_json = read_json(ckpt / "config.json") if cfg_json.get("type") != ARCH: raise SystemExit(f"config.json type is {cfg_json.get('type')!r}, expected 'pi05'") + norm_map = cfg_json.get("normalization_mapping") or {} + norm_mode = norm_map.get("ACTION", "QUANTILES") + if norm_mode not in ("QUANTILES", "MEAN_STD") or norm_map.get("STATE", "QUANTILES") != norm_mode: + raise SystemExit(f"unsupported pi05 normalization_mapping {norm_map}") cfg = dict(GEMMA_2B, **GEMMA_300M) cfg["paligemma_variant"] = str(cfg_json.get("paligemma_variant", "gemma_2b")) @@ -190,22 +196,25 @@ def main() -> int: cfg["rope_theta"] = ROPE_THETA cfg["rms_norm_eps"] = RMS_NORM_EPS cfg["norm_eps"] = norm_eps(ckpt) + cfg["norm_mode"] = norm_mode.lower() print(f"opening {sf_path}") sf = safe_open(sf_path, framework="pt") keys = set(sf.keys()) + root = pi_root(keys) + vlm, head, aex = root + PFX_VLM, root + PFX_VLM_HEAD, root + PFX_AEX - n_layers_vlm = max_layer(keys, f"{PFX_VLM}.layers.") - n_layers_aex = max_layer(keys, f"{PFX_AEX}.layers.") + n_layers_vlm = max_layer(keys, f"{vlm}.layers.") + n_layers_aex = max_layer(keys, f"{aex}.layers.") if n_layers_vlm <= 0: raise SystemExit("cannot find PaliGemma language-model layers in checkpoint") if n_layers_aex != n_layers_vlm: raise SystemExit(f"layer count mismatch: VLM={n_layers_vlm} expert={n_layers_aex}") cfg["n_layers"] = n_layers_vlm - q0 = sf.get_slice(f"{PFX_VLM}.layers.0.self_attn.q_proj.weight").get_shape() - kv0 = sf.get_slice(f"{PFX_VLM}.layers.0.self_attn.k_proj.weight").get_shape() - gate0 = sf.get_slice(f"{PFX_VLM}.layers.0.mlp.gate_proj.weight").get_shape() + q0 = sf.get_slice(f"{vlm}.layers.0.self_attn.q_proj.weight").get_shape() + kv0 = sf.get_slice(f"{vlm}.layers.0.self_attn.k_proj.weight").get_shape() + gate0 = sf.get_slice(f"{vlm}.layers.0.mlp.gate_proj.weight").get_shape() if q0[1] != cfg["hidden"]: raise SystemExit(f"hidden mismatch: cfg={cfg['hidden']} ckpt={q0[1]}") if q0[0] != cfg["n_q_heads"] * cfg["head_dim"]: @@ -215,15 +224,15 @@ def main() -> int: if gate0[0] != cfg["intermediate"]: raise SystemExit(f"intermediate mismatch: cfg={cfg['intermediate']} ckpt={gate0[0]}") - ada0 = sf.get_slice(f"{PFX_AEX}.layers.0.input_layernorm.dense.weight").get_shape() + ada0 = sf.get_slice(f"{aex}.layers.0.input_layernorm.dense.weight").get_shape() if ada0 != [3 * cfg["expert_h"], cfg["expert_h"]]: raise SystemExit(f"expert adaRMS dense shape {ada0} != [3*expert_h, expert_h] " f"{[3*cfg['expert_h'], cfg['expert_h']]}") - aex_o0 = sf.get_slice(f"{PFX_AEX}.layers.0.self_attn.o_proj.weight").get_shape() + aex_o0 = sf.get_slice(f"{aex}.layers.0.self_attn.o_proj.weight").get_shape() if aex_o0 != [cfg["expert_h"], cfg["n_q_heads"] * cfg["head_dim"]]: raise SystemExit(f"expert o_proj shape {aex_o0} unexpected") - cfg["vocab_size"] = int(sf.get_slice(PFX_VLM_HEAD).get_shape()[0]) + cfg["vocab_size"] = int(sf.get_slice(head).get_shape()[0]) print(f"resolved cfg: hidden={cfg['hidden']} n_layers={cfg['n_layers']} " f"expert_h={cfg['expert_h']} vocab={cfg['vocab_size']} chunk={cfg['chunk_size']} " @@ -231,35 +240,40 @@ def main() -> int: f"real_action={cfg['real_action_dim']} max_len={cfg['tokenizer_max_length']} " f"norm_eps={cfg['norm_eps']:g}") - cfg["vit"] = probe_paligemma_vision(sf, keys, cfg_json, PFX_VIS_CANDIDATES, PFX_MMP_CANDIDATES) + cfg["vit"] = probe_paligemma_vision(sf, keys, cfg_json, [root + p for p in PFX_VIS_CANDIDATES], + [root + p for p in PFX_MMP_CANDIDATES]) v = cfg["vit"] print(f"vision: SigLIP hidden={v['vit_hidden']} layers={v['vit_layers']} " f"heads={v['vit_heads']} image={v['image_size']} patch={v['patch_size']} " f"tokens={v['n_img_tokens']} ln_eps={v['vit_ln_eps']:g}") - print("loading dataset normalizer stats...") - stats = _load_dataset_stats( - args.dataset_stats, - args.dataset_repo, - cfg["real_state_dim"], - cfg["real_action_dim"] - ) - print(f" state_q01[:3]={stats['state_q01'][:3]} state_q99[:3]={stats['state_q99'][:3]}") - print(f" action_q01[:3]={stats['action_q01'][:3]} action_q99[:3]={stats['action_q99'][:3]} (QUANTILES)") + if norm_mode == "MEAN_STD": + print("loading normalizer stats...") + stats = lerobot_stats(sf, ckpt, cfg["real_state_dim"], cfg["real_action_dim"], norm_map) + else: + print("loading dataset normalizer stats...") + stats = _load_dataset_stats( + args.dataset_stats, + args.dataset_repo, + cfg["real_state_dim"], + cfg["real_action_dim"] + ) + print(f" state_q01[:3]={stats['state_q01'][:3]} state_q99[:3]={stats['state_q99'][:3]}") + print(f" action_q01[:3]={stats['action_q01'][:3]} action_q99[:3]={stats['action_q99'][:3]} (QUANTILES)") writer = open_writer(out, ARCH) write_pi_kv(writer, KV, cfg, adarms=True) - add(writer, "token_embd.weight", sf.get_tensor(PFX_VLM_HEAD)) - add(writer, "vlm.output_norm.weight", sf.get_tensor(f"{PFX_VLM}.norm.weight")) - write_decoder_blocks(writer, sf.get_tensor, PFX_VLM, "vlm", cfg["n_layers"]) + add(writer, "token_embd.weight", sf.get_tensor(head)) + add(writer, "vlm.output_norm.weight", sf.get_tensor(f"{vlm}.norm.weight")) + write_decoder_blocks(writer, sf.get_tensor, vlm, "vlm", cfg["n_layers"]) - add(writer, "aex.output_norm.weight", sf.get_tensor(f"{PFX_AEX}.norm.dense.weight")) - add(writer, "aex.output_norm.bias", sf.get_tensor(f"{PFX_AEX}.norm.dense.bias")) - _write_adarms_blocks(writer, sf, cfg["n_layers"]) + add(writer, "aex.output_norm.weight", sf.get_tensor(f"{aex}.norm.dense.weight")) + add(writer, "aex.output_norm.bias", sf.get_tensor(f"{aex}.norm.dense.bias")) + _write_adarms_blocks(writer, sf, aex, cfg["n_layers"]) for suf in PROJ_SUFFIXES: - add(writer, suf, sf.get_tensor(suf)) + add(writer, suf, sf.get_tensor(root + suf)) write_paligemma_vision(writer, sf, cfg["vit"]) diff --git a/scripts/convert_pi0_to_gguf.py b/scripts/convert_pi0_to_gguf.py index 45658f1..8276562 100644 --- a/scripts/convert_pi0_to_gguf.py +++ b/scripts/convert_pi0_to_gguf.py @@ -15,15 +15,12 @@ from __future__ import annotations -from pathlib import Path - -import numpy as np from safetensors import safe_open from gguf_blocks import ( - identity_stats, - load_processor_stats, + lerobot_stats, norm_eps, + pi_root, probe_paligemma_vision, write_decoder_blocks, write_paligemma_vision, @@ -45,18 +42,17 @@ ARCH = "pi0" KV = kv_prefix(ARCH) -PFX_VLM = "model.paligemma_with_expert.paligemma.model.language_model" -PFX_VLM_HEAD = "model.paligemma_with_expert.paligemma.lm_head.weight" -PFX_AEX = "model.paligemma_with_expert.gemma_expert.model" -PFX_PROJ = "model" +PFX_VLM = "paligemma_with_expert.paligemma.model.language_model" +PFX_VLM_HEAD = "paligemma_with_expert.paligemma.lm_head.weight" +PFX_AEX = "paligemma_with_expert.gemma_expert.model" PFX_VIS_CANDIDATES = [ - "model.paligemma_with_expert.paligemma.model.vision_tower.vision_model", - "model.paligemma_with_expert.paligemma.vision_tower.vision_model", + "paligemma_with_expert.paligemma.model.vision_tower.vision_model", + "paligemma_with_expert.paligemma.vision_tower.vision_model", ] PFX_MMP_CANDIDATES = [ - "model.paligemma_with_expert.paligemma.model.multi_modal_projector", - "model.paligemma_with_expert.paligemma.multi_modal_projector", + "paligemma_with_expert.paligemma.model.multi_modal_projector", + "paligemma_with_expert.paligemma.multi_modal_projector", ] GEMMA_2B = dict(hidden=2048, n_q_heads=8, n_kv_heads=1, head_dim=256, intermediate=16384) @@ -78,71 +74,6 @@ "action_out_proj.bias", ] -def _load_stats(sf, ckpt: Path, state_dim: int, action_dim: int) -> dict[str, np.ndarray]: - - out = identity_stats(state_dim, action_dim) - - got_state = load_processor_stats( - ckpt, - "policy_preprocessor.json", - "normalizer_processor", - "observation.state", - state_dim - ) - got_action = load_processor_stats( - ckpt, - "policy_postprocessor.json", - "unnormalizer_processor", - "action", - action_dim - ) - - keys = set(sf.keys()) - - def _legacy(mk: str, sk: str, mean_dst: str, std_dst: str, dim: int) -> None: - if mk not in keys or sk not in keys: - print(f" stats: legacy {mk} / {sk} missing - using identity for {mean_dst[:-5]}") - return - mean = sf.get_tensor(mk).float().numpy().reshape(-1) - std = sf.get_tensor(sk).float().numpy().reshape(-1) - if mean.size != dim or std.size != dim: - print(f" stats: legacy {mk} dim mismatch ({mean.size} vs {dim}) - using identity") - return - out[mean_dst] = mean.astype(np.float32, copy=False) - out[std_dst] = std .astype(np.float32, copy=False) - print(f" stats: loaded {mean_dst[:-5]} from model.safetensors ({mk}/{sk}) [legacy]") - - if got_state is not None: - out["state_mean"], out["state_std"] = got_state - else: - _legacy( - "normalize_inputs.buffer_observation_state.mean", - "normalize_inputs.buffer_observation_state.std", - "state_mean", - "state_std", - state_dim - ) - - if got_action is not None: - out["action_mean"], out["action_std"] = got_action - elif "unnormalize_outputs.buffer_action.mean" in keys: - _legacy( - "unnormalize_outputs.buffer_action.mean", - "unnormalize_outputs.buffer_action.std", - "action_mean", - "action_std", - action_dim - ) - else: - _legacy( - "normalize_targets.buffer_action.mean", - "normalize_targets.buffer_action.std", - "action_mean", - "action_std", - action_dim - ) - return out - def main() -> int: ap = arg_parser(ARCH, "lerobot π₀ checkpoint dir (model.safetensors + config.json + policy_*processor.json)") args = ap.parse_args() @@ -156,6 +87,8 @@ def main() -> int: if cfg_json.get("type") != ARCH: raise SystemExit(f"config.json type is {cfg_json.get('type')!r}, expected 'pi0' " f"(π0.5 / other variants are not handled by this converter)") + if cfg_json.get("adapt_to_pi_aloha"): + raise SystemExit("adapt_to_pi_aloha=true is not supported") cfg = dict(GEMMA_2B, **GEMMA_300M) cfg["paligemma_variant"] = str(cfg_json.get("paligemma_variant", "gemma_2b")) @@ -177,9 +110,11 @@ def main() -> int: print(f"opening {sf_path}") sf = safe_open(sf_path, framework="pt") keys = set(sf.keys()) + root = pi_root(keys) + vlm, head, aex = root + PFX_VLM, root + PFX_VLM_HEAD, root + PFX_AEX - n_layers_vlm = max_layer(keys, f"{PFX_VLM}.layers.") - n_layers_aex = max_layer(keys, f"{PFX_AEX}.layers.") + n_layers_vlm = max_layer(keys, f"{vlm}.layers.") + n_layers_aex = max_layer(keys, f"{aex}.layers.") if n_layers_vlm <= 0: raise SystemExit("cannot find PaliGemma language-model layers in checkpoint") if n_layers_aex != n_layers_vlm: @@ -187,9 +122,9 @@ def main() -> int: f"(π0 expects them equal)") cfg["n_layers"] = n_layers_vlm - q0 = sf.get_slice(f"{PFX_VLM}.layers.0.self_attn.q_proj.weight").get_shape() - kv0 = sf.get_slice(f"{PFX_VLM}.layers.0.self_attn.k_proj.weight").get_shape() - gate0 = sf.get_slice(f"{PFX_VLM}.layers.0.mlp.gate_proj.weight").get_shape() + q0 = sf.get_slice(f"{vlm}.layers.0.self_attn.q_proj.weight").get_shape() + kv0 = sf.get_slice(f"{vlm}.layers.0.self_attn.k_proj.weight").get_shape() + gate0 = sf.get_slice(f"{vlm}.layers.0.mlp.gate_proj.weight").get_shape() if q0[1] != cfg["hidden"]: raise SystemExit(f"hidden mismatch: cfg={cfg['hidden']} ckpt={q0[1]}") if q0[0] != cfg["n_q_heads"] * cfg["head_dim"]: @@ -199,17 +134,17 @@ def main() -> int: if gate0[0] != cfg["intermediate"]: raise SystemExit(f"intermediate mismatch: cfg={cfg['intermediate']} ckpt={gate0[0]}") - aex_gate0 = sf.get_slice(f"{PFX_AEX}.layers.0.mlp.gate_proj.weight").get_shape() + aex_gate0 = sf.get_slice(f"{aex}.layers.0.mlp.gate_proj.weight").get_shape() if aex_gate0[1] != cfg["expert_h"]: raise SystemExit(f"expert_h mismatch: cfg={cfg['expert_h']} ckpt={aex_gate0[1]}") if aex_gate0[0] != cfg["expert_inter"]: raise SystemExit(f"expert_inter mismatch: cfg={cfg['expert_inter']} ckpt={aex_gate0[0]}") - aex_o0 = sf.get_slice(f"{PFX_AEX}.layers.0.self_attn.o_proj.weight").get_shape() + aex_o0 = sf.get_slice(f"{aex}.layers.0.self_attn.o_proj.weight").get_shape() if aex_o0 != [cfg["expert_h"], cfg["n_q_heads"] * cfg["head_dim"]]: raise SystemExit(f"expert o_proj shape {aex_o0} != [expert_h, n_q*head_dim] " f"{[cfg['expert_h'], cfg['n_q_heads']*cfg['head_dim']]}") - head_w = sf.get_slice(PFX_VLM_HEAD).get_shape() + head_w = sf.get_slice(head).get_shape() if head_w[1] != cfg["hidden"]: raise SystemExit(f"lm_head hidden mismatch: cfg={cfg['hidden']} ckpt={head_w[1]}") cfg["vocab_size"] = int(head_w[0]) @@ -222,9 +157,11 @@ def main() -> int: f"norm_eps={cfg['norm_eps']:g}") print("loading normalizer stats...") - stats = _load_stats(sf, ckpt, cfg["real_state_dim"], cfg["real_action_dim"]) + stats = lerobot_stats(sf, ckpt, cfg["real_state_dim"], cfg["real_action_dim"], + cfg_json.get("normalization_mapping") or {}) - cfg["vit"] = probe_paligemma_vision(sf, keys, cfg_json, PFX_VIS_CANDIDATES, PFX_MMP_CANDIDATES) + cfg["vit"] = probe_paligemma_vision(sf, keys, cfg_json, [root + p for p in PFX_VIS_CANDIDATES], + [root + p for p in PFX_MMP_CANDIDATES]) v = cfg["vit"] print(f"vision: SigLIP hidden={v['vit_hidden']} layers={v['vit_layers']} " f"heads={v['vit_heads']} image={v['image_size']} patch={v['patch_size']} " @@ -233,15 +170,15 @@ def main() -> int: writer = open_writer(out, ARCH) write_pi_kv(writer, KV, cfg) - add(writer, "token_embd.weight", sf.get_tensor(PFX_VLM_HEAD)) - add(writer, "vlm.output_norm.weight", sf.get_tensor(f"{PFX_VLM}.norm.weight")) - write_decoder_blocks(writer, sf.get_tensor, PFX_VLM, "vlm", cfg["n_layers"]) + add(writer, "token_embd.weight", sf.get_tensor(head)) + add(writer, "vlm.output_norm.weight", sf.get_tensor(f"{vlm}.norm.weight")) + write_decoder_blocks(writer, sf.get_tensor, vlm, "vlm", cfg["n_layers"]) - add(writer, "aex.output_norm.weight", sf.get_tensor(f"{PFX_AEX}.norm.weight")) - write_decoder_blocks(writer, sf.get_tensor, PFX_AEX, "aex", cfg["n_layers"]) + add(writer, "aex.output_norm.weight", sf.get_tensor(f"{aex}.norm.weight")) + write_decoder_blocks(writer, sf.get_tensor, aex, "aex", cfg["n_layers"]) for suf in PROJ_SUFFIXES: - add(writer, suf, sf.get_tensor(f"{PFX_PROJ}.{suf}")) + add(writer, suf, sf.get_tensor(root + suf)) write_paligemma_vision(writer, sf, cfg["vit"]) diff --git a/scripts/convert_smolvla_to_gguf.py b/scripts/convert_smolvla_to_gguf.py index f7c6463..5c741d1 100644 --- a/scripts/convert_smolvla_to_gguf.py +++ b/scripts/convert_smolvla_to_gguf.py @@ -15,14 +15,10 @@ from __future__ import annotations -from pathlib import Path - -import numpy as np from safetensors import safe_open from gguf_blocks import ( - identity_stats, - load_processor_stats, + lerobot_stats, norm_eps, probe_siglip, write_decoder_blocks, @@ -77,29 +73,6 @@ def _probe_vision(sf, keys, cfg_json: dict) -> dict: v["n_img_tokens"] = (grid // scale) ** 2 return v -def _load_stats(ckpt: Path, state_dim: int, action_dim: int) -> dict[str, np.ndarray]: - - out = identity_stats(state_dim, action_dim) - got_state = load_processor_stats( - ckpt, - "policy_preprocessor.json", - "normalizer_processor", - "observation.state", - state_dim - ) - got_action = load_processor_stats( - ckpt, - "policy_postprocessor.json", - "unnormalizer_processor", - "action", - action_dim - ) - if got_state is not None: - out["state_mean"], out["state_std"] = got_state - if got_action is not None: - out["action_mean"], out["action_std"] = got_action - return out - def _add_kv(writer, cfg: dict) -> None: writer.add_uint32 (KV("hidden"), cfg["hidden"]) @@ -145,6 +118,9 @@ def main() -> int: require(sf_path) cfg_json = read_json(ckpt / "config.json") + for k in ("adapt_to_pi_aloha", "add_image_special_tokens"): + if cfg_json.get(k): + raise SystemExit(f"{k}=true is not supported") cfg = dict(SMOLLM2_500M) cfg["chunk_size"] = int(cfg_json["chunk_size"]) @@ -189,7 +165,8 @@ def main() -> int: f"vocab={cfg['vocab_size']} chunk={cfg['chunk_size']}") print("loading normalizer stats...") - stats = _load_stats(ckpt, cfg["real_state_dim"], cfg["real_action_dim"]) + stats = lerobot_stats(sf, ckpt, cfg["real_state_dim"], cfg["real_action_dim"], + cfg_json.get("normalization_mapping") or {}) cfg["vit"] = _probe_vision(sf, keys, cfg_json) v = cfg["vit"] diff --git a/scripts/convert_turbovla_to_gguf.py b/scripts/convert_turbovla_to_gguf.py index e910827..f731e79 100644 --- a/scripts/convert_turbovla_to_gguf.py +++ b/scripts/convert_turbovla_to_gguf.py @@ -28,16 +28,14 @@ import json from pathlib import Path -import numpy as np import torch from safetensors import safe_open import gguf - -F32 = gguf.GGMLQuantizationType.F32 -BF16 = gguf.GGMLQuantizationType.BF16 +from gguf_common import add as add_tensor, finish, kv_prefix, max_layer, open_writer ARCH = "turbovla" +kv = kv_prefix(ARCH) # Prefixes used by official TurboVLA checkpoints (without a "model." prefix). @@ -52,41 +50,9 @@ KEY_TEXT_PROJ = "text_encoder.text_projection" -def kv_prefix(name: str) -> str: - return f"{ARCH}.{name}" - - -def bf16_u16(t: torch.Tensor) -> np.ndarray: - return t.contiguous().view(torch.uint16).cpu().numpy() - - -def add_tensor(writer: gguf.GGUFWriter, name: str, t: torch.Tensor) -> None: - """Add tensor with preserved dtype.""" - if t.dtype == torch.float32: - writer.add_tensor(name, t.contiguous().cpu().numpy(), raw_dtype=F32) - elif t.dtype == torch.bfloat16: - writer.add_tensor(name, bf16_u16(t), raw_shape=list(t.shape), raw_dtype=BF16) - elif t.dtype == torch.float16: - writer.add_tensor(name, t.contiguous().cpu().numpy().astype(np.float32), raw_dtype=F32) - else: - raise NotImplementedError(f"unsupported dtype {t.dtype} for {name}") - - -def max_layer(keys: set[str], pfx: str) -> int: - """Count number of layers with given prefix.""" - m = -1 - for k in keys: - if k.startswith(pfx): - try: - m = max(m, int(k[len(pfx):].split(".", 1)[0])) - except ValueError: - pass - return m + 1 - - # facebook/dinov3-vitb16-pretrain-lvd1689m is gated, but only these architecture # values are needed: the fine-tuned weights ship inside the TurboVLA checkpoint. -DINOV3_VITB16 = {"rope_theta": 100.0, "num_register_tokens": 4} +DINOV3_VITB16 = {"rope_theta": 100.0, "num_register_tokens": 4, "num_attention_heads": 12} # Tensors the runtime never reads: DINOv3's final norm (TurboVLA taps # hidden_states[-1], before it), BERT's pooler, and the MAE mask token. @@ -108,7 +74,7 @@ def __getitem__(self, key): def load_checkpoint(ckpt: Path) -> tuple[TrackedTensors, dict]: """Return (state dict, TurboVLA model_config) from a .pth or a directory.""" if ckpt.is_file(): - blob = torch.load(str(ckpt), map_location="cpu", weights_only=False) + blob = torch.load(str(ckpt), map_location="cpu", weights_only=True) if not isinstance(blob, dict) or "model_state_dict" not in blob: raise SystemExit(f"{ckpt} is not a TurboVLA checkpoint (no model_state_dict)") state = blob["model_state_dict"] @@ -232,6 +198,9 @@ def __init__(self, tensors: dict[str, torch.Tensor], keys: set[str], cfg_json: d ) if int(self.dinov3_cfg.get("num_register_tokens", self.num_register_tokens)) != self.num_register_tokens: raise SystemExit("DINOv3 config num_register_tokens disagrees with checkpoint weights") + self.vit_heads = int(self.dinov3_cfg.get("num_attention_heads", 0)) + if self.vit_heads <= 0 or self.vit_dim % self.vit_heads: + raise SystemExit("DINOv3 config num_attention_heads is missing or does not divide the ViT width") word_emb = self._get(f"{PREFIX_TEXT}.embeddings.word_embeddings.weight") self.vocab_size = int(word_emb.shape[0]) @@ -508,7 +477,7 @@ def write_text_groups(writer: gguf.GGUFWriter, text_cfg: dict) -> None: sees token ids, so the table is keyed by them. """ groups = text_cfg.get("padding_length_by_instruction") or {} - writer.add_uint32(kv_prefix("text_groups.count"), len(groups)) + writer.add_uint32(kv("text_groups.count"), len(groups)) if not groups: return from transformers import AutoTokenizer @@ -520,9 +489,9 @@ def write_text_groups(writer: gguf.GGUFWriter, text_cfg: dict) -> None: ids += [int(t) for t in seq] lengths.append(len(seq)) pad_to.append(int(length)) - writer.add_array(kv_prefix("text_groups.tokens"), ids) - writer.add_array(kv_prefix("text_groups.lengths"), lengths) - writer.add_array(kv_prefix("text_groups.pad_to"), pad_to) + writer.add_array(kv("text_groups.tokens"), ids) + writer.add_array(kv("text_groups.lengths"), lengths) + writer.add_array(kv("text_groups.pad_to"), pad_to) def verify_consumed_tensors(tensors: TrackedTensors) -> None: @@ -563,17 +532,21 @@ def main() -> int: dims = TurboVLADims(tensors, keys, cfg_json, dinov3_cfg) print(f" Detected: {dims}") - print(f"Writing GGUF to {out}...") - out.parent.mkdir(parents=True, exist_ok=True) - writer = gguf.GGUFWriter(str(out), ARCH) - - kv = kv_prefix - writer.add_string(kv("architecture"), ARCH) + text_cfg = cfg_json.get("text", {}) + inter_cfg = cfg_json.get("interaction", {}) + if (inter_cfg.get("residual_style", "normalized") != "normalized" + or inter_cfg.get("padding_strategy", "key_padding_mask") != "key_padding_mask" + or text_cfg.get("zero_padded_tokens", False) + or not text_cfg.get("sub_sentence_present", True)): + raise SystemExit("unsupported TurboVLA variant: the runtime implements residual_style=normalized, " + "padding_strategy=key_padding_mask, zero_padded_tokens=false, sub_sentence_present=true") + + writer = open_writer(out, ARCH) writer.add_uint32(kv("hidden"), dims.hidden_dim) writer.add_uint32(kv("vit_dim"), dims.vit_dim) writer.add_uint32(kv("vit_layers"), dims.vit_layers) - writer.add_uint32(kv("vit_head_dim"), 64) - writer.add_uint32(kv("vit_heads"), 12) + writer.add_uint32(kv("vit_head_dim"), dims.vit_dim // dims.vit_heads) + writer.add_uint32(kv("vit_heads"), dims.vit_heads) writer.add_uint32(kv("text_dim"), dims.text_dim) writer.add_uint32(kv("text_layers"), dims.text_layers) writer.add_uint32(kv("text_head_dim"), 64) @@ -608,7 +581,6 @@ def main() -> int: # TurboVLA's BERT wrapper uses these exact punctuation IDs when it creates # sub-sentence attention masks. Persist them so the GGUF runtime does not # silently depend on a tokenizer installation. - text_cfg = cfg_json.get("text", {}) model_name = text_cfg.get("model_name_or_path", "bert-base-uncased") if model_name not in ("bert-base-uncased", "google-bert/bert-base-uncased"): raise SystemExit( @@ -650,14 +622,7 @@ def main() -> int: print(" Verifying tensor consumption...") verify_consumed_tensors(tensors) - writer.write_header_to_file() - writer.write_kv_data_to_file() - writer.write_tensors_to_file() - writer.close() - - size_mb = out.stat().st_size / (1024 * 1024) - print(f" Done! Output: {out} ({size_mb:.1f} MiB)") - return 0 + return finish(writer, out) if __name__ == "__main__": diff --git a/scripts/gguf_blocks.py b/scripts/gguf_blocks.py index 1a3536c..917c0f6 100644 --- a/scripts/gguf_blocks.py +++ b/scripts/gguf_blocks.py @@ -85,6 +85,9 @@ def write_siglip_tower( add_n(writer, "vit.post_ln.weight", g(f"{root}.post_layernorm.weight")); add_n(writer, "vit.post_ln.bias", g(f"{root}.post_layernorm.bias")) +def pi_root(keys) -> str: + return "model." if "model.paligemma_with_expert.paligemma.lm_head.weight" in keys else "" + def probe_paligemma_vision(sf, keys, cfg_json: dict, vis_candidates, mmp_candidates) -> dict: vis = next((p for p in vis_candidates if f"{p}.embeddings.patch_embedding.weight" in keys), None) @@ -125,7 +128,7 @@ def write_pi_kv(writer, kv, cfg: dict, adarms: bool = False) -> None: if adarms: writer.add_bool (kv("use_adarms_expert"), True) writer.add_uint32(kv("adarms_cond_dim"), cfg["expert_h"]) - writer.add_string(kv("norm_mode"), "quantiles") + writer.add_string(kv("norm_mode"), cfg["norm_mode"]) writer.add_float64 (kv("min_period"), cfg["min_period"]) writer.add_float64 (kv("max_period"), cfg["max_period"]) writer.add_float64 (kv("rope_theta"), cfg["rope_theta"]) @@ -335,12 +338,12 @@ def load_processor_stats(ckpt: Path, meta_json: str, registry: str, key: str, di meta_path = ckpt / meta_json if not meta_path.exists(): - print(f" stats: {meta_json} missing - using identity for {key}") + print(f" stats: {meta_json} missing") return None try: meta = json.loads(meta_path.read_text()) except Exception as e: - print(f" stats: {meta_json} parse failed ({e}) - using identity for {key}") + print(f" stats: {meta_json} parse failed ({e})") return None state_file = None @@ -349,29 +352,68 @@ def load_processor_stats(ckpt: Path, meta_json: str, registry: str, key: str, di state_file = step.get("state_file") break if not state_file: - print(f" stats: no {registry} step in {meta_json} - using identity for {key}") + print(f" stats: no {registry} step in {meta_json}") return None sf_path = ckpt / state_file if not sf_path.is_file(): - print(f" stats: {sf_path.name} referenced by {meta_json} but missing - using identity for {key}") + print(f" stats: {sf_path.name} referenced by {meta_json} but missing") return None with safe_open(str(sf_path), framework="pt") as f: keys = set(f.keys()) mk, sk = f"{key}.mean", f"{key}.std" if mk not in keys or sk not in keys: - print(f" stats: {sf_path.name} lacks {mk}/{sk} - using identity for {key}") + print(f" stats: {sf_path.name} lacks {mk}/{sk}") return None mean = f.get_tensor(mk).float().numpy().reshape(-1) std = f.get_tensor(sk).float().numpy().reshape(-1) if mean.size != dim or std.size != dim: - print(f" stats: {mk} dim mismatch ({mean.size} vs {dim}) in {sf_path.name} - using identity") + print(f" stats: {mk} dim mismatch ({mean.size} vs {dim}) in {sf_path.name}") return None print(f" stats: loaded {key} from {sf_path.name} ({mk}/{sk})") return mean.astype(np.float32, copy=False), std.astype(np.float32, copy=False) +def lerobot_stats(sf, ckpt: Path, state_dim: int, action_dim: int, norm_map: dict) -> dict[str, np.ndarray]: + + out = identity_stats(state_dim, action_dim) + keys = set(sf.keys()) + + def _legacy(pfx: str, dim: int): + mk, sk = f"{pfx}.mean", f"{pfx}.std" + if mk not in keys or sk not in keys: + return None + mean = sf.get_tensor(mk).float().numpy().reshape(-1) + std = sf.get_tensor(sk).float().numpy().reshape(-1) + if mean.size != dim or std.size != dim: + print(f" stats: legacy {mk} dim mismatch ({mean.size} vs {dim})") + return None + print(f" stats: loaded {pfx} from model.safetensors [legacy]") + return mean.astype(np.float32, copy=False), std.astype(np.float32, copy=False) + + feats = ( + ("STATE", "state", "policy_preprocessor.json", "normalizer_processor", "observation.state", state_dim, + ("normalize_inputs.buffer_observation_state",)), + ("ACTION", "action", "policy_postprocessor.json", "unnormalizer_processor", "action", action_dim, + ("unnormalize_outputs.buffer_action", "normalize_targets.buffer_action")), + ) + for ftype, dst, meta_json, registry, key, dim, legacy in feats: + mode = norm_map.get(ftype, "MEAN_STD") + if mode == "IDENTITY": + continue + if mode != "MEAN_STD": + raise SystemExit(f"normalization_mapping {ftype}={mode} is not supported (MEAN_STD or IDENTITY only)") + got = load_processor_stats(ckpt, meta_json, registry, key, dim) + for pfx in legacy: + if got is None: + got = _legacy(pfx, dim) + if got is None: + raise SystemExit(f"no {key} mean/std in {meta_json} or model.safetensors ({', '.join(legacy)}); " + f"refusing to bake identity stats for MEAN_STD {ftype}") + out[f"{dst}_mean"], out[f"{dst}_std"] = got + return out + def identity_stats(state_dim: int, action_dim: int) -> dict[str, np.ndarray]: return { "state_mean": np.zeros(state_dim, dtype=np.float32), diff --git a/scripts/gguf_common.py b/scripts/gguf_common.py index 37541d2..a8f05ea 100644 --- a/scripts/gguf_common.py +++ b/scripts/gguf_common.py @@ -20,6 +20,7 @@ import argparse import json +import re from pathlib import Path import numpy as np @@ -42,6 +43,8 @@ def add(writer: gguf.GGUFWriter, name: str, t: torch.Tensor) -> None: writer.add_tensor(name, t.contiguous().cpu().numpy(), raw_dtype=F32) elif t.dtype == torch.bfloat16: writer.add_tensor(name, bf16_u16(t), raw_shape=list(t.shape), raw_dtype=BF16) + elif t.dtype == torch.float16: + writer.add_tensor(name, t.contiguous().cpu().numpy().astype(np.float32), raw_dtype=F32) else: raise NotImplementedError(f"unsupported dtype {t.dtype} for {name}") @@ -87,10 +90,20 @@ def load_safetensors(ckpt: Path, keep: tuple[str, ...] | None = None) -> dict[st def load_pt_module(path: Path) -> dict[str, torch.Tensor]: - sd = torch.load(str(path), map_location="cpu", weights_only=False) + sd = torch.load(str(path), map_location="cpu", weights_only=True) pfx = "module." return {(k[len(pfx):] if k.startswith(pfx) else k): v.contiguous() for k, v in sd.items()} +def find_sidecar(ckpt: Path, stem: str) -> Path: + + cands = sorted( + ckpt.glob(f"{stem}--*checkpoint.pt"), + key=lambda p: int(m.group(1)) if (m := re.search(r"--(\d+)_checkpoint\.pt$", p.name)) else -1, + ) + if not cands: + raise SystemExit(f"no {stem}--*checkpoint.pt in {ckpt}") + return cands[-1] + def read_json(path: Path) -> dict: if not path.exists(): raise SystemExit(f"missing {path}") diff --git a/scripts/quantize_gguf.py b/scripts/quantize_gguf.py index 6e37647..f036248 100644 --- a/scripts/quantize_gguf.py +++ b/scripts/quantize_gguf.py @@ -32,6 +32,7 @@ # tower stays float too by default; add --vision to pack it as well. SKIP = ( "token_embd", + "tok_embd", "output.weight", "patch_embd", "norm", @@ -40,13 +41,16 @@ "cls", "action", "state", - "expert", + "aex.", + "ah.", + "act.", + "octo.head", "dit", "adaln", "ada_", "time" ) -SKIP_VISION = ("vit", "vision") +SKIP_VISION = ("vit", "vision", "vis.d.", "vis.s.", "octo.obs.") # Block size per row (ne0 must divide this). Only the types the gguf writer can # pack are offered; Q8_0 is near-lossless, Q4_0/Q4_1 are 4-bit. @@ -93,10 +97,8 @@ def main() -> None: for name, f in r.fields.items(): if name in meta: continue - if f.types and f.types[0] == gguf.GGUFValueType.ARRAY: - w.add_array(name, f.contents()) - else: - w.add_key_value(name, f.contents(), f.types[0]) + sub = f.types[-1] if f.types[0] == gguf.GGUFValueType.ARRAY else None + w.add_key_value(name, f.contents(), f.types[0], sub_type=sub) qtype = getattr(gguf.GGMLQuantizationType, args.type) F32, BF16 = gguf.GGMLQuantizationType.F32, gguf.GGMLQuantizationType.BF16 diff --git a/tests/py/test_converters.py b/tests/py/test_converters.py index ba403e2..18dcb6f 100644 --- a/tests/py/test_converters.py +++ b/tests/py/test_converters.py @@ -348,6 +348,44 @@ def test_turbovla_converter_remap(): } <= set(tensors.keys_read) +def test_quantize_skip_names(): + import importlib + Q = importlib.import_module("quantize_gguf") + default = Q.SKIP + Q.SKIP_VISION + keep = [ + "aex.blk.0.attn_q.weight", "aex.vlsa.0.ff0.weight", "aex.head.fc1.weight", "aex.seq_pool.weight", + "ah.act_enc.l2.weight", "act.dec.0.fc1.weight", "octo.head.diffusion.reverse.out.weight", + "octo.t5.tok_embd.weight", "vis.d.blk.0.qkv.weight", "vis.s.blk.0.fc1.weight", + "octo.obs.primary.proj.weight", + ] + pack = [ + "vlm.blk.0.attn_q.weight", "lm.blk.0.ffn_up.weight", "octo.t5.blk.0.attn_q.weight", + "text.encoder.layer.0.attention.self.query.weight", "mm.fc1.weight", "vis.proj.fc1.weight", + ] + for n in keep: + assert not Q.eligible(n, (64, 64), "Q8_0", default), n + for n in pack: + assert Q.eligible(n, (64, 64), "Q8_0", default), n + assert Q.eligible("vis.d.blk.0.qkv.weight", (64, 64), "Q8_0", Q.SKIP) + + +def test_find_sidecar(): + import tempfile + from gguf_common import find_sidecar + with tempfile.TemporaryDirectory() as d: + d = pathlib.Path(d) + try: + find_sidecar(d, "action_head") + raise AssertionError("missing sidecar must exit") + except SystemExit: + pass + (d / "action_head--checkpoint.pt").touch() + assert find_sidecar(d, "action_head").name == "action_head--checkpoint.pt" + for step in (5000, 30000, 10000): + (d / f"action_head--{step}_checkpoint.pt").touch() + assert find_sidecar(d, "action_head").name == "action_head--30000_checkpoint.pt" + + def test_every_converter_imports(): # every converter must resolve against the two shared modules import importlib From c1eaec59bfc7d804f36981cf855e80ac2ffb5992 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Mon, 28 Sep 2026 14:53:57 +0700 Subject: [PATCH 12/61] harden CI, add CUDA, OpenCL and Metal build checks --- .github/workflows/build.yml | 87 ++++++++++++++++++++++++++++-------- .github/workflows/vla-ci.yml | 25 +++++++---- CMakeLists.txt | 23 ++++++---- ci/README.md | 31 +++++++++---- ci/agent/agent.cpp | 18 ++------ ci/agent/ctl.cpp | 5 +-- ci/build_servers.sh | 14 ++++-- ci/check_commits.sh | 10 +++-- ci/check_thresholds.py | 34 +++++++++----- ci/orchestrate.sh | 4 ++ eval/Dockerfile.client | 15 ++----- tests/CMakeLists.txt | 5 +++ 12 files changed, 181 insertions(+), 90 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 6a96ce9..5f64f33 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -13,19 +13,6 @@ concurrency: cancel-in-progress: true jobs: - cpp-unit: - runs-on: ubuntu-24.04 - steps: - - uses: actions/checkout@v7 - # Compiled directly: pure, no llama.cpp or protobuf/zmq needed. - - name: pure unit tests - run: | - for t in test_vision_common test_rope_conventions; do - g++ -std=c++17 -Isrc -Wall -Wextra -fsanitize=address,undefined \ - -fno-omit-frame-pointer "tests/$t.cpp" -o "/tmp/$t" - "/tmp/$t" - done - py-tooling: runs-on: ubuntu-24.04 steps: @@ -56,17 +43,19 @@ jobs: - uses: actions/cache@v6 with: path: build/_deps - key: llama-${{ steps.pin.outputs.tag }}-${{ runner.os }} + key: llama-${{ steps.pin.outputs.tag }}-${{ runner.os }}-asan # Everything, not a target list: ctest registers tests this job must build, # and a named list goes stale the next time one is added. - - name: build + ctest (CPU, -Wall -Wextra) + - name: build + ctest (CPU, ASan+UBSan, -Werror) run: | - cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=OFF -DVLA_BUILD_TESTS=ON + san="-fsanitize=address,undefined -fno-sanitize-recover=undefined -fno-omit-frame-pointer" + cmake -B build -DCMAKE_BUILD_TYPE=RelWithDebInfo -DGGML_CUDA=OFF -DVLA_BUILD_TESTS=ON \ + -DVLA_WERROR=ON -DCMAKE_C_FLAGS="$san" -DCMAKE_CXX_FLAGS="$san" cmake --build build -j"$(nproc)" ctest --test-dir build --output-on-failure - # This job is the only one that fetches llama.cpp, and neither patch script - # runs on a CPU build, so their anchors would otherwise rot unnoticed until - # someone configures a CUDA or OpenVINO tree. Patch a copy: the real one is + # Neither patch script runs on a CPU build, and build-cuda patches only on a + # cache miss, so their anchors would otherwise rot unnoticed until someone + # configures a fresh CUDA or OpenVINO tree. Patch a copy: the real one is # cached. - name: patch anchors still apply run: | @@ -74,3 +63,63 @@ jobs: python3 scripts/patch_ggml_cuda_ext_hook.py /tmp/llama-patchtest python3 scripts/patch_ggml_openvino.py /tmp/llama-patchtest python3 scripts/patch_ggml_openvino.py /tmp/llama-patchtest # idempotent + + build-cuda: + runs-on: ubuntu-24.04 + container: nvidia/cuda:12.9.1-devel-ubuntu24.04 + steps: + - name: deps + run: | + apt-get update -qq + DEBIAN_FRONTEND=noninteractive apt-get install -y -qq --no-install-recommends \ + build-essential cmake git ca-certificates pkg-config python3 \ + libzmq3-dev cppzmq-dev libprotobuf-dev protobuf-compiler + - uses: actions/checkout@v7 + - name: read llama.cpp pin + id: pin + run: echo "tag=$(bash scripts/llama_tag.sh)" >> "$GITHUB_OUTPUT" + - uses: actions/cache@v6 + with: + path: build/_deps + key: llama-${{ steps.pin.outputs.tag }}-${{ runner.os }}-cuda-${{ hashFiles('scripts/patch_ggml_cuda_ext_hook.py') }} + - name: build (compile only, the runner has no GPU) + run: | + cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=89-real \ + -DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined -DVLA_BUILD_TESTS=ON -DVLA_WERROR=ON + cmake --build build -j"$(nproc)" + + build-backends: + runs-on: ${{ matrix.os }} + strategy: + fail-fast: false + matrix: + include: + - { name: opencl, os: ubuntu-24.04, cmake: -DGGML_OPENCL=ON -DVLA_WERROR=ON } + - { name: metal, os: macos-15, cmake: -DGGML_METAL=ON, test: true } + steps: + - uses: actions/checkout@v7 + - name: deps + if: runner.os == 'Linux' + run: | + sudo apt-get update -qq + sudo apt-get install -y -qq --no-install-recommends \ + build-essential cmake git ca-certificates pkg-config \ + libzmq3-dev cppzmq-dev libprotobuf-dev protobuf-compiler \ + ocl-icd-opencl-dev opencl-headers + - name: deps + if: runner.os == 'macOS' + run: brew install cmake zeromq cppzmq protobuf + - name: read llama.cpp pin + id: pin + run: echo "tag=$(bash scripts/llama_tag.sh)" >> "$GITHUB_OUTPUT" + - uses: actions/cache@v6 + with: + path: build/_deps + key: llama-${{ steps.pin.outputs.tag }}-${{ runner.os }}-${{ matrix.name }} + - name: build + run: | + cmake -B build -DCMAKE_BUILD_TYPE=Release ${{ matrix.cmake }} -DVLA_BUILD_TESTS=ON + cmake --build build -j"$(getconf _NPROCESSORS_ONLN)" + - name: ctest + if: matrix.test + run: ctest --test-dir build --output-on-failure diff --git a/.github/workflows/vla-ci.yml b/.github/workflows/vla-ci.yml index 57c88bd..c17da61 100644 --- a/.github/workflows/vla-ci.yml +++ b/.github/workflows/vla-ci.yml @@ -1,7 +1,9 @@ # Cross-platform VLA regression CI. -# Fires when a PR from `dev` targets `main`. Runs on the self-hosted runner -# labelled `vla-ci-orchestrator` (register your orchestrator host with that -# label - the actual hostname stays out of the repo, in ci/config/hosts.env). +# Fires on a push to `dev`, or by hand. Runs on the self-hosted runner labelled +# `vla-ci-orchestrator` (register your orchestrator host with that label - the +# actual hostname stays out of the repo, in the hosts.env that the runner's +# VLA_CI_HOSTS_ENV points at). Keep that runner in a runner group restricted to +# VinRobotics/vla.cpp/.github/workflows/vla-ci.yml@refs/heads/dev. # Every platform is a remote server reached over the LAN via vla-ci-agent (no # SSH); the orchestrator drives one LIBERO client per platform. The three # platforms are swept CONCURRENTLY in a single job (a self-hosted runner runs one @@ -9,8 +11,9 @@ name: vla-ci on: - pull_request: - branches: [main] + push: + branches: [dev] + workflow_dispatch: concurrency: group: vla-ci-${{ github.ref }} @@ -18,13 +21,19 @@ concurrency: jobs: ci: - # Only PRs whose source branch is `dev`. - if: github.head_ref == 'dev' runs-on: [self-hosted, vla-ci-orchestrator] timeout-minutes: 240 + env: + VLA_CI_EXPECTED_COMMIT: ${{ github.sha }} + CI_OUTPUT_ROOT: ${{ github.workspace }}/outputs/ci steps: - uses: actions/checkout@v7 - - name: Sweep all platforms (parallel) + gate + - name: Host config + control tools + run: | + cp "${VLA_CI_HOSTS_ENV:?set VLA_CI_HOSTS_ENV in the runner .env}" ci/config/hosts.env + cmake -S ci/agent -B ci/agent/build -DCMAKE_BUILD_TYPE=Release + cmake --build ci/agent/build -j + - name: Move servers to the pushed commit, build, sweep all platforms (parallel) + gate run: bash ci/orchestrate.sh all - uses: actions/upload-artifact@v7 if: always() diff --git a/CMakeLists.txt b/CMakeLists.txt index dd0224e..9ae0758 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -32,6 +32,11 @@ if(_vla_accel_n GREATER 0 AND GGML_BACKEND_DL) "GGML_BACKEND_DL=ON is not supported with ${_vla_accel}: the backend is " "built as a module and src/backend.h links its init directly.") endif() +foreach(_flag GGML_VULKAN GGML_HIP GGML_MUSA GGML_CANN GGML_WEBGPU) + if(${_flag}) + message(WARNING "${_flag}=ON reaches vlm-server through llama only; src/backend.h has no ${_flag} path, so the VLA archs run on CPU.") + endif() +endforeach() set(LLAMA_BUILD_COMMON ON CACHE BOOL "" FORCE) set(LLAMA_BUILD_TOOLS ON CACHE BOOL "" FORCE) @@ -293,7 +298,7 @@ endif() # headers for the backend vtable, hence the private include. if(GGML_HEXAGON OR GGML_OPENCL) target_sources(vla_core PRIVATE src/backend_fallback.cpp) - target_include_directories(vla_core PRIVATE ${llama_SOURCE_DIR}/ggml/src) + target_include_directories(vla_core SYSTEM PRIVATE ${llama_SOURCE_DIR}/ggml/src) if(GGML_HEXAGON) target_compile_definitions(vla_core PUBLIC GGML_USE_HEXAGON) else() @@ -304,13 +309,11 @@ endif() add_library(vlm_core ${VLA_CORE_LIB_TYPE} src/vlm/engine.cpp ) -target_include_directories(vlm_core - PUBLIC - ${CMAKE_CURRENT_SOURCE_DIR}/src - PRIVATE - ${llama_SOURCE_DIR}/common - ${llama_SOURCE_DIR}/tools/mtmd - ${llama_SOURCE_DIR}/vendor +target_include_directories(vlm_core PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/src) +target_include_directories(vlm_core SYSTEM PRIVATE + ${llama_SOURCE_DIR}/common + ${llama_SOURCE_DIR}/tools/mtmd + ${llama_SOURCE_DIR}/vendor ) target_link_libraries(vlm_core PUBLIC llama ggml mtmd llama-common) @@ -434,8 +437,10 @@ if(GGML_CUDA) list(APPEND VLA_FIRST_PARTY_TARGETS bitvla_cuda_kernels vla_cuda_ops) endif() +option(VLA_WERROR "Treat warnings in first-party code as errors" OFF) foreach(tgt IN LISTS VLA_FIRST_PARTY_TARGETS) - target_compile_options(${tgt} PRIVATE $<$:-Wall -Wextra>) + target_compile_options(${tgt} PRIVATE + $<$:-Wall -Wextra $<$:-Werror>>) endforeach() # llama.cpp sends its DLLs to /bin, and Windows has no rpath: a binary diff --git a/ci/README.md b/ci/README.md index ae36dfe..7696a3c 100644 --- a/ci/README.md +++ b/ci/README.md @@ -1,9 +1,9 @@ # vla.cpp cross-platform CI -Per-PR regression gate for `vla.cpp` across the three target platforms - RTX 3090 -(`rtx3090`), Jetson Orin Nano (`orin`), Apple M4 (`m4`). On a PR from `dev` into -`main`, each platform's `vla-server` is evaluated in LIBERO and gated on success -rate (client side) + latency / memory (server side). +Regression gate for `vla.cpp` across the three target platforms - RTX 3090 +(`rtx3090`), Jetson Orin Nano (`orin`), Apple M4 (`m4`). On every push to `dev`, +each platform's `vla-server` is evaluated in LIBERO and gated on success rate +(client side) + latency / memory (server side). Machines are referred to by **role** (`orchestrator`) and **platform key**; real hostnames / IPs live only in the gitignored `ci/config/hosts.env`. @@ -37,7 +37,9 @@ Every cell is **10 tasks × 1 episode**. ## Gating -- **SR** (client side) - must be **> 0**. +- **SR** (client side) - must be **≥ 0.5× `sr_reported`** of the baseline + (`sr_tolerance` in `ci/baselines/.json` overrides 0.5; **> 0** where a + model has no `sr_reported`). - **Server latency & memory** - must be **≤ 1.10× baseline** (`ci/baselines/.json`). M4 memory has no baseline (recorded, not gated). @@ -65,7 +67,8 @@ $EDITOR ci/config/hosts.env # LAN IPs, ctrl/data ports, repo paths, MODELS_R ### 3. Bring up servers + agents, check with ping On each server (its own git checkout, with `vla-server` already built), run the -agent as a service (systemd / launchd / nohup): +agent as a service (systemd / launchd / nohup). It binds 127.0.0.1 unless given +`--bind`: ```bash ci/agent/build/vla-ci-agent --bind 'tcp://*:5600' [--token "$VLA_CI_TOKEN"] @@ -113,5 +116,17 @@ python ci/check_thresholds.py --platform rtx3090 \ ## CI trigger `.github/workflows/vla-ci.yml` runs `ci/orchestrate.sh all` on a self-hosted -runner labelled `vla-ci-orchestrator` for PRs from `dev` into `main`, and uploads -`outputs/ci/` as an artifact. +runner labelled `vla-ci-orchestrator` on every push to `dev` (or by hand via +workflow_dispatch), and uploads `outputs/ci/` as an artifact. It sets +`VLA_CI_EXPECTED_COMMIT` to the pushed commit, so every server first fetches that +commit from its `origin`, checks it out and rebuilds (`ci/build_servers.sh`); the +gate fails if a server is on any other commit. The runner's `.env` must set +`VLA_CI_HOSTS_ENV` to a hosts.env outside the workspace (checkout wipes ignored +files). + +It does not run on `pull_request`: a PR run takes its workflow file from the PR's +merge commit, so a fork could drop any guard in it and reach the LAN agents. Put +the orchestrator runner in an org runner group whose workflow access is restricted +to `VinRobotics/vla.cpp/.github/workflows/vla-ci.yml@refs/heads/dev`. A path-only +restriction is not enough, because a fork edits the same path. With that group, +a manual dispatch gets a runner only when run from `dev`. diff --git a/ci/agent/agent.cpp b/ci/agent/agent.cpp index de8e101..67d164d 100644 --- a/ci/agent/agent.cpp +++ b/ci/agent/agent.cpp @@ -7,7 +7,7 @@ // // Single-threaded request loop (the orchestrator drives it serially). Spawned // servers are detached into their own session/group so `stop` can signal the -// whole group; dead detached children are reaped at the top of the loop. +// whole group; only `stop` or a respawn reaps them, so a dead one keeps its pgid. // // Security: with --token T (or env VLA_CI_TOKEN) every request must carry a // matching token. Bind to a LAN address only - this runs arbitrary commands by @@ -42,14 +42,6 @@ namespace { std::map g_spawned; // name -> session-leader pid -std::vector to_argv(const google::protobuf::RepeatedPtrField& a) { - std::vector v; - v.reserve(a.size() + 1); - for (const auto& s : a) v.push_back(const_cast(s.c_str())); - v.push_back(nullptr); - return v; -} - void apply_env(const google::protobuf::RepeatedPtrField& env) { for (const auto& kv : env) { auto p = kv.find('='); @@ -297,20 +289,19 @@ void handle(const Request& req, Reply& rep) { int main(int argc, char** argv) { GOOGLE_PROTOBUF_VERIFY_VERSION; - std::string bind = "tcp://*:5600"; + std::string bind = "tcp://127.0.0.1:5600"; std::string token = std::getenv("VLA_CI_TOKEN") ? std::getenv("VLA_CI_TOKEN") : ""; for (int i = 1; i < argc; ++i) { std::string a = argv[i]; if (a == "--bind" && i + 1 < argc) bind = argv[++i]; else if (a == "--token" && i + 1 < argc) token = argv[++i]; else if (a == "-h" || a == "--help") { - std::printf("usage: %s [--bind tcp://*:5600] [--token SECRET]\n", argv[0]); + std::printf("usage: %s [--bind tcp://127.0.0.1:5600] [--token SECRET]\n", argv[0]); return 0; } else { std::fprintf(stderr, "unknown arg: %s\n", a.c_str()); return 2; } } - // A dropped peer must not kill us with SIGPIPE; detached children are reaped - // at the top of the request loop instead. + // A dropped peer must not kill us with SIGPIPE. signal(SIGPIPE, SIG_IGN); zmq::context_t zctx(1); @@ -323,7 +314,6 @@ int main(int argc, char** argv) { zmq::pollitem_t poll[] = {{static_cast(sock), 0, ZMQ_POLLIN, 0}}; for (;;) { - while (waitpid(-1, nullptr, WNOHANG) > 0) {} // reap dead detached children try { zmq::poll(poll, 1, std::chrono::milliseconds(200)); } catch (const zmq::error_t&) { continue; } diff --git a/ci/agent/ctl.cpp b/ci/agent/ctl.cpp index b272543..e311c1d 100644 --- a/ci/agent/ctl.cpp +++ b/ci/agent/ctl.cpp @@ -8,7 +8,7 @@ // ping // put --src DIR --dst DIR [--prune] [--protect P]... (general; CI no longer deploys with it) // exec --cwd DIR -- ARGV... (general; CI no longer builds with it) -// build --cwd DIR [--flags ""] [--jobs N] [--no-patch] [--prelude ""] # build vla-server +// build --cwd DIR [--flags ""] [--jobs N] [--prelude ""] # build vla-server // spawn --name N --cwd DIR --log FILE -- ARGV... // stop --name N // get --remote PATH --local PATH @@ -158,8 +158,7 @@ int main(int argc, char** argv) { std::string jobs = rest("--jobs"); // optional -j value std::string jexpr = jobs.empty() ? "$(getconf _NPROCESSORS_ONLN)" : jobs; std::string prelude = rest("--prelude"); // shell run first, e.g. CUDA exports (Jetson) - std::string patch = has("--no-patch") ? "" : "bash patches/patch.sh; "; - std::string script = "set -e; " + (prelude.empty() ? std::string() : prelude + " ") + patch + + std::string script = "set -e; " + (prelude.empty() ? std::string() : prelude + " ") + "cmake -B build -DCMAKE_BUILD_TYPE=Release " + flags + "; " "cmake --build build -j" + jexpr; auto* e = req.mutable_exec(); diff --git a/ci/build_servers.sh b/ci/build_servers.sh index cae9401..a0e684f 100755 --- a/ci/build_servers.sh +++ b/ci/build_servers.sh @@ -4,8 +4,9 @@ # Build vla-server on the platform servers via the control agent (vla-ci-ctl # build), in parallel. A convenience for the self-managed model: it builds each # server's OWN checkout with that platform's CMake flags (from hosts.env) and does -# NOT change the server's commit. Run it after the servers are at the target -# commit, before `ci/orchestrate.sh all`. +# NOT change the server's commit, unless VLA_CI_EXPECTED_COMMIT is set: then each +# server first fetches that commit from its `origin` and checks it out. Otherwise +# run it after the servers are at the target commit, before `ci/orchestrate.sh all`. # # bash ci/build_servers.sh # all ALL_PLATFORMS, in parallel # bash ci/build_servers.sh rtx3090 orin # a subset @@ -30,8 +31,13 @@ for p in ${PLATFORMS}; do resolve_platform "${p}" || exit 1 ep="tcp://${SRV_HOST}:${CTRL_PORT}" echo "[build] ${p} -> ${ep} cwd=${RROOT} flags=[${CMAKE_FLAGS:-}] prelude=[${BUILD_ENV:+set}] log=${CI_OUTPUT_ROOT}/${p}.build.log" - ( "${CTL}" --endpoint "${ep}" build --cwd "${RROOT}" --flags "${CMAKE_FLAGS}" --prelude "${BUILD_ENV:-}" ) \ - >"${CI_OUTPUT_ROOT}/${p}.build.log" 2>&1 & + ( + if [[ -n "${VLA_CI_EXPECTED_COMMIT:-}" ]]; then + "${CTL}" --endpoint "${ep}" exec --cwd "${RROOT}" -- git fetch --quiet origin "${VLA_CI_EXPECTED_COMMIT}" + "${CTL}" --endpoint "${ep}" exec --cwd "${RROOT}" -- git checkout --quiet --detach "${VLA_CI_EXPECTED_COMMIT}" + fi + "${CTL}" --endpoint "${ep}" build --cwd "${RROOT}" --flags "${CMAKE_FLAGS}" --prelude "${BUILD_ENV:-}" + ) >"${CI_OUTPUT_ROOT}/${p}.build.log" 2>&1 & PID[$p]=$! done diff --git a/ci/check_commits.sh b/ci/check_commits.sh index 7e6b5cb..d8b6ce2 100755 --- a/ci/check_commits.sh +++ b/ci/check_commits.sh @@ -4,10 +4,11 @@ # Verify git-commit consistency across the tested platforms BEFORE running any # sim. Each server self-manages its checkout; the agent's `rev` reports that # server's real `git rev-parse HEAD`. The orchestrator's own commit need NOT -# match the servers. Rules (over the TESTED machines = the platform servers): +# match the servers unless VLA_CI_EXPECTED_COMMIT is set, which then replaces it. +# Rules (over the TESTED machines = the platform servers): # # all four equal (orchestrator + servers) -> INFO -# servers equal, orchestrator differs -> WARNING (proceed) +# servers equal, orchestrator differs -> WARNING (proceed); ERROR, exit 2 with VLA_CI_EXPECTED_COMMIT # servers disagree (or a rev is unreadable) -> ERROR, exit 2 (stop the CI) # # bash ci/check_commits.sh # all ALL_PLATFORMS (the CI default) @@ -26,7 +27,7 @@ CTL="${VLA_CI_CTL:-${CI_DIR}/agent/build/vla-ci-ctl}" export VLA_CI_TOKEN="${VLA_CI_TOKEN:-}" PLATFORMS="${*:-${ALL_PLATFORMS}}" -ORCH_COMMIT="$(git -C "${REPO_ROOT}" rev-parse HEAD 2>/dev/null || echo unknown)" +ORCH_COMMIT="${VLA_CI_EXPECTED_COMMIT:-$(git -C "${REPO_ROOT}" rev-parse HEAD 2>/dev/null || echo unknown)}" mkdir -p "${CI_OUTPUT_ROOT}" declare -A COMMIT @@ -64,6 +65,9 @@ if [[ ${servers_same} -eq 0 ]]; then fi if [[ "${first}" == "${ORCH_COMMIT}" ]]; then echo "INFO: orchestrator and all tested machines are at the same commit (${first})." +elif [[ -n "${VLA_CI_EXPECTED_COMMIT:-}" ]]; then + echo "ERROR: tested machines are at ${first}, not the expected ${VLA_CI_EXPECTED_COMMIT}; stopping CI." >&2 + exit 2 else echo "WARNING: orchestrator (${ORCH_COMMIT}) differs from the tested machines (${first}); proceeding." >&2 fi diff --git a/ci/check_thresholds.py b/ci/check_thresholds.py index fe80414..1dc6f73 100755 --- a/ci/check_thresholds.py +++ b/ci/check_thresholds.py @@ -20,7 +20,9 @@ verdict. Gates (all must pass for exit 0): - * SR > 0 per (model, suite) - "must be a positive number" + * SR >= sr_tol*sr_reported per (model, suite) - sr_tolerance in the + baseline, default 0.5; + SR > 0 without sr_reported * server latency <= tol*baseline per model - mean `total` ms / call * server memory <= tol*baseline per model - platform mem metric (skipped where the @@ -117,7 +119,8 @@ def suite_sr(model_dir: Path, suite: str) -> tuple[int, int, list[int]]: def check_model(name: str, model_dir: Path, logs_dir: Path, base: dict, - latency_metric: str, mem_metric: str | None, tol: float) -> dict: + latency_metric: str, mem_metric: str | None, tol: float, + sr_tol: float) -> dict: res: dict = {"model": name, "suites": {}, "checks": [], "ok": True} def gate(ok: bool, label: str, detail: str): @@ -125,7 +128,8 @@ def gate(ok: bool, label: str, detail: str): if not ok: res["ok"] = False - # ---- SR per suite (gate: > 0) ----------------------------------------- + # ---- SR per suite (gate: >= sr_tol * sr_reported, else > 0) ---------- + base_sr = base.get("sr_reported") suites = discover_suites(model_dir) if not suites: gate(False, "outputs", f"no summary.txt found under {model_dir}") @@ -135,8 +139,14 @@ def gate(ok: bool, label: str, detail: str): sr = succ / eps if eps else 0.0 res["suites"][suite] = {"successes": succ, "episodes": eps, "tasks": len(seen), "sr": sr} - gate(succ > 0, f"SR>0 [{suite}]", - f"{succ}/{eps} success ({sr:.1%}) over {len(seen)} tasks") + if base_sr: + floor = sr_tol * base_sr + gate(sr >= floor - 1e-9, f"SR [{suite}]", + f"{succ}/{eps} success ({sr:.1%}) over {len(seen)} tasks vs " + f"{sr_tol:g}*{base_sr:.1%}={floor:.1%} baseline") + else: + gate(succ > 0, f"SR>0 [{suite}]", + f"{succ}/{eps} success ({sr:.1%}) over {len(seen)} tasks") # ---- server latency (gate: <= tol * baseline) ------------------------- # Per-suite models run several server processes; aggregate (sample-weighted) @@ -188,10 +198,11 @@ def gate(ok: bool, label: str, detail: str): return res -def render_md(platform: str, tol: float, results: list[dict]) -> str: +def render_md(platform: str, tol: float, sr_tol: float, results: list[dict]) -> str: overall = all(r["ok"] for r in results) out = [f"# CI gate - `{platform}` {'PASS' if overall else 'FAIL'}", - "", f"Tolerance: actual ≤ {tol:g}× reported baseline. SR gate: > 0.", ""] + "", f"Tolerance: actual ≤ {tol:g}× reported baseline. " + f"SR gate: ≥ {sr_tol:g}× sr_reported (> 0 without one).", ""] for r in results: out.append(f"## `{r['model']}` {'PASS' if r['ok'] else 'FAIL'}") for c in r["checks"]: @@ -215,6 +226,7 @@ def main() -> int: spec = json.loads(args.baseline.read_text()) tol = float(spec.get("tolerance", 1.10)) + sr_tol = float(spec.get("sr_tolerance", 0.5)) lat_metric = spec.get("latency_metric", "server_total_ms") mem_metric = spec.get("mem_metric") base_models = spec["models"] @@ -238,16 +250,16 @@ def main() -> int: "suites": {}}) continue results.append(check_model(name, model_dir, logs_dir, base_models[name], - lat_metric, mem_metric, tol)) + lat_metric, mem_metric, tol, sr_tol)) overall = bool(results) and all(r["ok"] for r in results) - verdict = {"platform": args.platform, "tolerance": tol, "ok": overall, - "results": results} + verdict = {"platform": args.platform, "tolerance": tol, "sr_tolerance": sr_tol, + "ok": overall, "results": results} out_dir = args.out or args.sweep out_dir.mkdir(parents=True, exist_ok=True) (out_dir / "verdict.json").write_text(json.dumps(verdict, indent=2)) - md = render_md(args.platform, tol, results) + md = render_md(args.platform, tol, sr_tol, results) (out_dir / "verdict.md").write_text(md) print(md) print(f"\n[gate] {'PASS' if overall else 'FAIL'} - wrote {out_dir / 'verdict.json'}") diff --git a/ci/orchestrate.sh b/ci/orchestrate.sh index 99f226c..f0dcc6a 100755 --- a/ci/orchestrate.sh +++ b/ci/orchestrate.sh @@ -34,6 +34,8 @@ sweep_and_gate() { # One LIBERO client per platform runs on the orchestrator at once. Gated metrics # are server-side, so orchestrator load cannot bias them. if [[ "${PLATFORM}" == "all" ]]; then + [[ -z "${VLA_CI_EXPECTED_COMMIT:-}" ]] || bash "${CI_DIR}/build_servers.sh" \ + || { echo "[all] server checkout/build failed - aborting before sim." >&2; exit 1; } # Commit consistency across the tested machines; stops here if they disagree. bash "${CI_DIR}/check_commits.sh" || { echo "[all] commit check failed - aborting before sim." >&2; exit 1; } @@ -59,6 +61,8 @@ fi # ── one platform ──────────────────────────────────────────────────────────── case "${PLATFORM}" in rtx3090|orin|m4) + [[ -z "${VLA_CI_EXPECTED_COMMIT:-}" ]] || bash "${CI_DIR}/build_servers.sh" "${PLATFORM}" \ + || { echo "[${PLATFORM}] server checkout/build failed - aborting before sim." >&2; exit 1; } bash "${CI_DIR}/check_commits.sh" "${PLATFORM}" \ || { echo "[${PLATFORM}] commit check failed - aborting before sim." >&2; exit 1; } sweep_and_gate "${PLATFORM}" ;; diff --git a/eval/Dockerfile.client b/eval/Dockerfile.client index fcba0b6..cae14e4 100644 --- a/eval/Dockerfile.client +++ b/eval/Dockerfile.client @@ -30,6 +30,7 @@ ENV DEBIAN_FRONTEND=noninteractive \ LANG=C.UTF-8 \ LC_ALL=C.UTF-8 \ PROTOCOL_BUFFERS_PYTHON_IMPLEMENTATION=python \ + PIP_NO_CACHE_DIR=1 \ VLA_CPP_PROTO=/workspace/vla.cpp/src/serving/vla.proto # System dependencies: build tools, protobuf, ZMQ, EGL (for MuJoCo offscreen) @@ -66,8 +67,7 @@ RUN pip3 install "mujoco<3.0" # PyTorch CPU-only (the server handles GPU inference) # install it now to prevent later deps from pulling in a GPU version -RUN pip3 install torch==2.5.1 --index-url https://download.pytorch.org/whl/cpu -RUN pip3 install torchvision==0.20.1 --index-url https://download.pytorch.org/whl/cpu +RUN pip3 install torch==2.7.1 torchvision==0.22.1 --index-url https://download.pytorch.org/whl/cpu # LIBERO — clone and install. # We use editable install (following eval/sim/libero/setup_libero.sh) so that @@ -78,18 +78,14 @@ RUN pip3 install torchvision==0.20.1 --index-url https://download.pytorch.org/wh # The CMAKE_POLICY_VERSION_MINIMUM env-var is needed by egl-probe (a transitive # dep of robomimic/robosuite) which requests cmake_minimum_required < 3.5. RUN CMAKE_POLICY_VERSION_MINIMUM=3.5 pip3 install --upgrade pip setuptools -RUN git clone --depth 1 https://github.com/Lifelong-Robot-Learning/LIBERO.git /tmp/LIBERO +RUN git clone https://github.com/Lifelong-Robot-Learning/LIBERO.git /tmp/LIBERO && git -C /tmp/LIBERO checkout 8f1084e3132a39270c3a13ebe37270a43ece2a01 RUN CMAKE_POLICY_VERSION_MINIMUM=3.5 pip3 install -r /tmp/LIBERO/requirements.txt RUN CMAKE_POLICY_VERSION_MINIMUM=3.5 pip3 install \ -e /tmp/LIBERO --config-settings editable_mode=compat && \ rm -rf /root/.libero && \ echo "N" | python3 -c "import libero.libero" 2>/dev/null || true -# lerobot — install with dependencies but immediately re-pin torch so it is -# not upgraded from our pinned CPU-only version. -RUN pip3 install lerobot==0.4.3 && \ - pip3 install torch==2.5.1 --index-url https://download.pytorch.org/whl/cpu && \ - pip3 install torchvision==0.20.1 --index-url https://download.pytorch.org/whl/cpu +RUN pip3 install lerobot==0.4.3 # Remaining client dependencies RUN pip3 install \ @@ -120,9 +116,6 @@ RUN pip3 install numpy==1.26.4 RUN sed -i 's/np_core = np._core if is_numpy_available("2.0.0") else np.core/np_core = np.core/' \ /usr/local/lib/python3.10/dist-packages/accelerate/utils/other.py 2>/dev/null || true -# Clean up pip cache to keep image size small -RUN rm -rf /root/.cache/pip - # The repo goes last: it changes every commit, and everything above it is a # half-hour of installs worth caching. COPY . . diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index ddffcfd..50e049f 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -5,6 +5,9 @@ if(WIN32) set(CMAKE_RUNTIME_OUTPUT_DIRECTORY ${CMAKE_BINARY_DIR}/bin) endif() +if(VLA_WERROR) + add_compile_options($<$:-Werror>) +endif() add_executable(vla_predict_check predict_check.cpp) target_link_libraries(vla_predict_check PRIVATE vla_core) @@ -69,4 +72,6 @@ if(GGML_CUDA) target_link_libraries(test_bf16_cuda_ops PRIVATE vla_cuda_ops ggml) target_compile_options(test_bf16_cuda_ops PRIVATE -Wall -Wextra) add_test(NAME bf16_cuda_ops COMMAND test_bf16_cuda_ops) + add_test(NAME bf16_cuda_ops_flat COMMAND test_bf16_cuda_ops) + set_tests_properties(bf16_cuda_ops_flat PROPERTIES ENVIRONMENT VLA_BF16_FLAT=1) endif() From d90aee9678348a0dd0f3764b7262e0a400f833a8 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Mon, 28 Sep 2026 14:53:57 +0700 Subject: [PATCH 13/61] update changelog --- CHANGELOG.md | 46 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 46 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index a33e1aa..e584515 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -45,6 +45,52 @@ Notable changes to vla.cpp. Format loosely follows [Keep a Changelog](https://ke ### Fixed +- vla-server no longer dies on a bad request. An out-of-vocab Octo token, 9 to + 16 OpenVLA-OFT or VLA-Adapter views, a BitVLA prompt past 1024 tokens, or Octo + stats without a mask each abort the process today; they now get an error + reply. Images are decoded as JPEG or PNG only, a request is capped at 64 + megapixels in total, precomputed embeddings at 16 views, and a port that is + already in use exits with a message instead of SIGABRT after the model load. +- vlm-server segfaulted on every start at `b11223`, read an uninitialized prompt + length, died on a chat-template exception, corrupted multi-turn prompts, + streamed invalid UTF-8, and handed network bytes to the ffmpeg image fallback. + All fixed; a full context now ends with `finish_reason="length"`. +- Graph sizes follow the real node count in π0, π0.5, SmolVLA, OpenVLA-OFT, + VLA-Adapter, VLA-JEPA and GR00T N1.7, so more views or `VLA_NUM_STEPS` no + longer trip a fixed 16384/65536-node assert. +- Malformed GGUF metadata (zero heads or patch size, non-square position tables, + bad RoPE theta, mismatched patch shapes) is rejected at load for every arch + instead of dividing by zero or reading past a buffer. +- BitVLA: BF16/F16/Q8_0 weights were uploaded to CUDA as float (NaN or garbage + actions), `VLA_DEVICE` was ignored, CUDA allocation and launch errors were + dropped, three kernels had shared-memory races, and failed inits leaked up to + 250 MiB. +- The BF16 CUDA hook could write F32 into a BF16 buffer on a declined matmul, + lost launch errors, shared one cuBLAS handle across threads, and missed an + alignment check on batched views. +- Evo-1's Q8_0 file aborted on a view assert; its attention split now uses row + views and no longer copies 24 weight slices per call. +- `WeightLoader::fuse` read quantized sources as float, so a quantized TurboVLA + failed to load. +- A `--config` runtime block overrode flags given on the command line and + silently dropped bad values. The command line now wins, bad values are an + error, and `libvla` and the Python bindings apply the block too. +- Two models in one process shared the flash-attention and matmul-precision + flags of whichever loaded last. +- The C API and Python bindings are safe to call from several threads on one + handle, and the bindings reject an image whose dtype does not match its pixel + format instead of reading past it. +- The CPU fallback wrapper reuses one threadpool instead of spawning threads for + every split. +- `scripts/quantize_gguf.py` wrote array metadata as INT32, which broke Octo and + TurboVLA, and packed the action experts and vision towers its skip list meant + to keep float. +- Converters: both lerobot key layouts for π0/π0.5, π0.5's normalization mapping, + legacy SmolVLA stats, head counts from config, `torch.load(weights_only=True)` + on third-party checkpoints (torch >= 2.6), and unsupported config variants are + refused. +- The eval client's REQ socket stayed stuck after one timeout, and the ALOHA + prefetch could replay a chunk from an earlier observation. - A `scripts/quantize_gguf.py` file did not load for SmolVLA on any platform: its loader read every weight as float and refused the packed connector. It now keeps packed GEMM weights packed, like the other archs, and dequantizes the From 4a7a42390120eb86bd502b2712ec61b8b66ccb99 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Mon, 28 Sep 2026 15:17:27 +0700 Subject: [PATCH 14/61] pair LIBERO runs with seeded noise, add octo, turbovla, vla_jepa --- eval/client/run_sim_client_direct.py | 6 +++ eval/client/vla_cpp_client.py | 28 +++++++++-- eval/run_libero.sh | 69 ++++++++++++++++++++++++---- 3 files changed, 89 insertions(+), 14 deletions(-) diff --git a/eval/client/run_sim_client_direct.py b/eval/client/run_sim_client_direct.py index 0bb62f5..7f3f878 100644 --- a/eval/client/run_sim_client_direct.py +++ b/eval/client/run_sim_client_direct.py @@ -144,6 +144,7 @@ success_count, inference_times = 0.0, [] skipped = 0 + episodes = [] for episode in range(args.n_episodes): print(f"*** Episode {episode + 1}/{args.n_episodes}") @@ -184,6 +185,8 @@ if episode_aborted: skipped += 1 + episodes.append((episode, 0 if episode_aborted else int(bool(info.get("is_success", 0))), + step_id, int(episode_aborted))) env.close() counted = max(1, args.n_episodes - skipped) @@ -196,6 +199,9 @@ f.write(f"Success rate: {success_count / counted:.2%} ({int(success_count)}/{counted})\n") f.write(f"Skipped (terminated mid-step): {skipped}/{args.n_episodes}\n") f.write(f"Average inference time per step: {avg_inf_ms} ms\n") + with open(output_dir / "episodes.csv", "w") as f: + f.write("episode,success,steps,aborted\n") + f.writelines(f"{e},{s},{n},{a}\n" for e, s, n, a in episodes) print("*** All episodes completed.") print(f"- Success rate: {success_count / counted:.2%} ({int(success_count)}/{counted})") diff --git a/eval/client/vla_cpp_client.py b/eval/client/vla_cpp_client.py index 9c9c521..ad8363a 100644 --- a/eval/client/vla_cpp_client.py +++ b/eval/client/vla_cpp_client.py @@ -195,6 +195,7 @@ def __init__( self.image_keys = list(image_keys) self.max_length = max_length self._step = 0 + self._episode = 0 self._last_response = None if n_action_steps < 1: @@ -641,6 +642,8 @@ def ping(self) -> bool: def reset(self) -> None: self._action_queue.clear() + self._episode += 1 + self._step = 0 def get_action(self, observations: dict[str, Any]) -> np.ndarray: @@ -720,6 +723,7 @@ def _predict_chunk(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(lang.tolist()) req.state.extend(state_padded.tolist()) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() resp = self.pb.PredictResponse() @@ -870,6 +874,7 @@ def _predict_chunk_vla_jepa(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(int(t) for t in lang) req.state.extend(float(x) for x in state_padded) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) resp = self.pb.PredictResponse() resp.ParseFromString(self.sock.recv()) @@ -930,6 +935,7 @@ def _predict_chunk_pi05(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(lang.tolist()) req.state.extend([0.0] * self.max_state_dim) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() resp = self.pb.PredictResponse() @@ -945,9 +951,13 @@ def _predict_chunk_pi05(self, observations: dict[str, Any]) -> np.ndarray: _EVO1_IMG_CTX = "" _EVO1_NUM_IMAGE_TOKEN = 256 _EVO1_MAX_TEXT_LENGTH = 1024 - _EVO1_NOISE_LEN = 50 * 24 # horizon * per_action_dim + _FIXED_NOISE_LEN = { + "smolvla": 50 * 32, "pi0": 50 * 32, "pi05": 50 * 32, "evo1": 50 * 24, + "gr00t_n1_5": 16 * 32, "gr00t_n1_6": 50 * 128, "gr00t_n1_7": 40 * 132, + "vla_jepa": 7 * 7, "octo": 4 * 7, + } - def _maybe_add_fixed_noise(self, req, n: int | None) -> None: + def _maybe_add_fixed_noise(self, req) -> None: """Attach a reproducible noise vector when VLA_FIXED_NOISE_SEED is set. Without it the server draws flow-matching noise from a clock-seeded RNG, @@ -956,14 +966,18 @@ def _maybe_add_fixed_noise(self, req, n: int | None) -> None: verifiable: same inputs plus same noise must give the same actions. """ seed = os.environ.get("VLA_FIXED_NOISE_SEED") + n = self._FIXED_NOISE_LEN.get(self.arch) if seed is None or not n: return # Vary per step but reproducibly, so a replay of the same episode sends # the same sequence of noise vectors. - rng = np.random.default_rng(int(seed) + self._step) + rng = np.random.default_rng([int(seed), self._episode, self._step]) # Evo-1 is trained on uniform[-1,1]; matching that keeps the check in # the distribution the model actually sees. - req.noise.extend(rng.uniform(-1.0, 1.0, size=n).astype(np.float32).tolist()) + if self.arch == "evo1": + req.noise.extend(rng.uniform(-1.0, 1.0, size=n).astype(np.float32).tolist()) + else: + req.noise.extend(rng.standard_normal(n, dtype=np.float32).tolist()) def _predict_chunk_evo1(self, observations: dict[str, Any]) -> np.ndarray: @@ -1042,7 +1056,7 @@ def _predict_chunk_evo1(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(input_ids_full[:n_real].tolist()) req.state.extend(state_padded.tolist()) req.attention_mask.extend(attn_mask.tolist()) - self._maybe_add_fixed_noise(req, self._EVO1_NOISE_LEN) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() @@ -1095,6 +1109,7 @@ def _predict_chunk_octo(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(input_ids.tolist()) req.attention_mask.extend(attn_mask.tolist()) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() resp = self.pb.PredictResponse() @@ -1422,6 +1437,7 @@ def _predict_chunk_gr00t_n1_7(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(int(t) for t in lang) req.state.extend(float(x) for x in state_padded) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() resp = self.pb.PredictResponse() @@ -1542,6 +1558,7 @@ def _predict_chunk_gr00t_n1_6(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(int(t) for t in lang) req.state.extend(float(x) for x in state_padded) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() resp = self.pb.PredictResponse() @@ -1628,6 +1645,7 @@ def _predict_chunk_gr00t_n1_5(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(int(t) for t in lang) req.state.extend(float(x) for x in state_padded) + self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) body = self.sock.recv() resp = self.pb.PredictResponse() diff --git a/eval/run_libero.sh b/eval/run_libero.sh index 5c589ed..c14f661 100644 --- a/eval/run_libero.sh +++ b/eval/run_libero.sh @@ -34,12 +34,17 @@ Usage: $(basename "$0") -i [-o ] [-n ] [- -n N_EPISODES episodes per task-id (default: 1) -m MODEL which model to run: smol | pi0 | pi05 | bit | evo1 | vla_adapter | openvla_oft | - gr00t_n1_5 | gr00t_n1_6 | gr00t_n1_7 | all + gr00t_n1_5 | gr00t_n1_6 | gr00t_n1_7 | + octo | turbovla | vla_jepa | all (default: all) -h show this help Env overrides: BIND_ADDR, CLIENT_ADDR, BITVLA_TOKENIZER, GR00T_N1_6_TOKENIZER, - GR00T_N1_5_STATS, GR00T_N1_6_STATS, GR00T_N1_7_STATS + GR00T_N1_5_STATS, GR00T_N1_6_STATS, GR00T_N1_7_STATS, + PALIGEMMA_TOKENIZER (pi0/pi05), TURBOVLA_STATS, VLA_JEPA_STATS, + TASK_IDS (default "0 1 2 3 4 5 6 7 8 9"), + SERVER_BIN (prebuilt vla-server; setting it skips the build), + VLA_FIXED_NOISE_SEED (client-side noise, for paired A/B runs) EOF } @@ -62,9 +67,9 @@ done shift $((OPTIND - 1)) case "${MODEL}" in - smol|pi0|pi05|bit|evo1|vla_adapter|openvla_oft|gr00t_n1_5|gr00t_n1_6|gr00t_n1_7|all) ;; + smol|pi0|pi05|bit|evo1|vla_adapter|openvla_oft|gr00t_n1_5|gr00t_n1_6|gr00t_n1_7|octo|turbovla|vla_jepa|all) ;; *) - echo "ERROR: -m must be one of: smol | pi0 | pi05 | bit | evo1 | vla_adapter | openvla_oft | gr00t_n1_5 | gr00t_n1_6 | gr00t_n1_7 | all (got '${MODEL}')" >&2 + echo "ERROR: -m must be one of: smol | pi0 | pi05 | bit | evo1 | vla_adapter | openvla_oft | gr00t_n1_5 | gr00t_n1_6 | gr00t_n1_7 | octo | turbovla | vla_jepa | all (got '${MODEL}')" >&2 exit 1 ;; esac @@ -87,7 +92,12 @@ if ! [[ "${N_EPISODES}" =~ ^[1-9][0-9]*$ ]]; then exit 1 fi -SERVER_BIN="${REPO_ROOT}/build/vla-server" +if [[ -n "${SERVER_BIN:-}" ]]; then + SKIP_BUILD=1 +fi +SERVER_BIN="${SERVER_BIN:-${REPO_ROOT}/build/vla-server}" +TASK_IDS="${TASK_IDS:-0 1 2 3 4 5 6 7 8 9}" +PALIGEMMA_TOKENIZER="${PALIGEMMA_TOKENIZER:-}" VENV_PY="${REPO_ROOT}/eval/sim/libero/libero_uv/.venv/bin/python" CLIENT="${REPO_ROOT}/eval/client/run_sim_client_direct.py" BIND_ADDR="${BIND_ADDR:-tcp://*:5555}" @@ -121,6 +131,9 @@ N_ACTION_STEPS_BIT="${N_ACTION_STEPS_BIT:-8}" # BitVLA NUM_ACTION N_ACTION_STEPS_GR00T_N1_5="${N_ACTION_STEPS_GR00T_N1_5:-16}" # N1.5 lerobot closeout (10/10 on libero_object/task_0) N_ACTION_STEPS_GR00T_N1_6="${N_ACTION_STEPS_GR00T_N1_6:-16}" # N1.6 H4 closeout (10/10 on libero_object/task_0) N_ACTION_STEPS_GR00T_N1_7="${N_ACTION_STEPS_GR00T_N1_7:-16}" # N1.7 H4 closeout (10/10 on libero_object/task_0) +N_ACTION_STEPS_OCTO="${N_ACTION_STEPS_OCTO:-4}" +N_ACTION_STEPS_TURBOVLA="${N_ACTION_STEPS_TURBOVLA:-12}" +N_ACTION_STEPS_VLA_JEPA="${N_ACTION_STEPS_VLA_JEPA:-7}" mkdir -p "${OUTPUT_ROOT}" OUTPUT_ROOT="$(cd "${OUTPUT_ROOT}" && pwd)" @@ -132,11 +145,13 @@ echo "[config] MODELS_ROOT=${MODELS_ROOT}" echo "[config] OUTPUT_ROOT=${OUTPUT_ROOT}" echo "[config] N_EPISODES=${N_EPISODES}" echo "[config] MODEL=${MODEL}" +echo "[config] TASK_IDS=${TASK_IDS}" +echo "[config] SERVER_BIN=${SERVER_BIN}" cd "${REPO_ROOT}" if [[ "${SKIP_BUILD:-0}" == "1" ]]; then - echo "[build] skipped (SKIP_BUILD=1)" + echo "[build] skipped (SKIP_BUILD=1 or SERVER_BIN set)" else echo "[build] cmake --build build" cmake --build build -j"$(nproc)" @@ -294,6 +309,9 @@ run_model() { if [[ "${arch}" == "gr00t_n1_6" ]]; then client_extra+=(--tokenizer "${GR00T_N1_6_TOKENIZER:-${model_dir}}") fi + if [[ ( "${arch}" == pi0 || "${arch}" == pi05 ) && -n "${PALIGEMMA_TOKENIZER}" ]]; then + client_extra+=(--tokenizer "${PALIGEMMA_TOKENIZER}") + fi if [[ -n "${stats_json}" ]]; then client_extra+=(--stats-json "${stats_json}") fi @@ -326,6 +344,10 @@ run_model() { else unset VLA_OPENVLA_OFT_UNNORM_KEY fi + if [[ "${arch}" == octo ]]; then + export VLA_OCTO_UNNORM_DATASET="${VLA_OCTO_UNNORM_DATASET:-${TASK_SUITE}}" + echo "[${arch}] VLA_OCTO_UNNORM_DATASET=${VLA_OCTO_UNNORM_DATASET}" + fi local log="${LOG_DIR}/${arch}.log" echo "====================" @@ -336,7 +358,7 @@ run_model() { local out_dir="${OUTPUT_ROOT}/${arch}" mkdir -p "${out_dir}" - for task_id in $(seq 0 9); do + for task_id in ${TASK_IDS}; do echo "[${arch}] task_id=${task_id} episodes=${N_EPISODES}" "${VENV_PY}" "${CLIENT}" \ --arch "${arch}" \ @@ -405,10 +427,10 @@ fi # tokenizer auto-loads from the base ckpt on the Hub, stats baked into the GGUF. if should_run vla_adapter; then run_model vla_adapter \ - "${MODELS_ROOT}/vla-adapter-libero-object-gguf" \ + "${MODELS_ROOT}/vla-adapter-libero-gguf" \ "${N_ACTION_STEPS_VLA_ADAPTER}" \ "" \ - "${MODELS_ROOT}/vla-adapter-libero-object-gguf/libero_object/vla-adapter-libero-object.gguf" + "${MODELS_ROOT}/vla-adapter-libero-gguf/libero_object/vla-adapter-libero-object.gguf" fi # openvla_oft: Llama-2-7B + MLPResNet head; vision baked in (no mmproj). Needs the @@ -477,5 +499,34 @@ if should_run gr00t_n1_7; then fi fi +if should_run octo; then + run_model octo \ + "${MODELS_ROOT}/octo-small-libero-gguf" \ + "${N_ACTION_STEPS_OCTO}" \ + "" \ + "${MODELS_ROOT}/octo-small-libero-gguf/octo-small-libero-f32.gguf" +fi + +if should_run turbovla; then + run_model turbovla \ + "${MODELS_ROOT}/turbovla-libero-gguf" \ + "${N_ACTION_STEPS_TURBOVLA}" \ + "${TURBOVLA_STATS:-}" \ + "${MODELS_ROOT}/turbovla-libero-gguf/turbovla-libero-f32.gguf" +fi + +if should_run vla_jepa; then + jepa_stats="${VLA_JEPA_STATS:-${MODELS_ROOT}/vla-jepa-libero}" + if [[ -f "${jepa_stats}/policy_preprocessor_step_3_normalizer_processor.safetensors" ]]; then + run_model vla_jepa \ + "${MODELS_ROOT}/vla-jepa-libero" \ + "${N_ACTION_STEPS_VLA_JEPA}" \ + "${jepa_stats}" \ + "${MODELS_ROOT}/vla-jepa-libero/vla-jepa.gguf" + else + echo "[skip] vla_jepa: policy_{pre,post}processor safetensors not found in ${jepa_stats}; set VLA_JEPA_STATS to override" + fi +fi + echo "====================" echo "Done. Results under ${OUTPUT_ROOT}" From e020e490d452e517f6754e3b25b348f554f852b5 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Mon, 28 Sep 2026 16:46:01 +0700 Subject: [PATCH 15/61] move GR00T and VLA-JEPA onto shared modules --- src/models/gr00tn1d5.cpp | 68 +----- src/models/gr00tn1d6.cpp | 87 +------ src/models/gr00tn1d7.cpp | 414 +++++----------------------------- src/models/vla_jepa.cpp | 296 ++++-------------------- src/modules/action_expert.cpp | 75 ++++++ src/modules/action_expert.h | 11 + src/modules/dit_head.cpp | 33 +++ src/modules/dit_head.h | 10 + src/modules/prompt.cpp | 5 +- src/modules/prompt.h | 3 +- src/modules/qwen3vl_vit.h | 182 ++++++++++++++- tests/test_qwen3vl_vit.cpp | 14 ++ 12 files changed, 439 insertions(+), 759 deletions(-) diff --git a/src/models/gr00tn1d5.cpp b/src/models/gr00tn1d5.cpp index d692d6c..5a87c1f 100644 --- a/src/models/gr00tn1d5.cpp +++ b/src/models/gr00tn1d5.cpp @@ -15,7 +15,6 @@ #include "arch.h" #include "options.h" #include "backend.h" -#include "env_flag.h" #include "gguf_reader.h" #include "layers/embed.h" #include "layers/linear.h" @@ -36,10 +35,7 @@ #include #include #include -#include -#include #include -#include #include #include @@ -57,6 +53,7 @@ struct Gr00tN1d5ModelArch : public ModelArchBase { ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; scratch_ctx vision_scratch; + FlowTimes times; struct MainKey { int64_t seq=-1, nsteps=-1; @@ -82,7 +79,7 @@ struct Gr00tN1d5ModelArch : public ModelArchBase { int64_t vit_layers=27, vit_inter=4304, image_size=224, patch_size=14, n_img_tokens=256; int64_t lm_inter=6144, vocab=151680, image_token_index=151669; int64_t bb_embed_dim=2048, in_embed_dim=1536, dit_interleave=1, vlsa_layers=4; - int64_t num_future=32, action_horizon=16, action_dim=32, max_state_dim=64; + int64_t action_horizon=16, action_dim=32, max_state_dim=64; int64_t num_steps=4, num_buckets=1000, max_embodiments=32, max_seq_len=1024; float vlln_eps=1e-5f; @@ -121,7 +118,6 @@ bool load_config(const gguf_reader & g, Gr00tN1d5ModelArch & m, Config & cfg) { U(fk("vlsa_layers" ), m.vlsa_layers); U(fk("vlsa_heads" ), m.vlsa.cfg.heads); U(fk("vlsa_head_dim" ), m.vlsa.cfg.head_dim); - U(fk("num_target_vision_tokens"), m.num_future); U(fk("action_horizon" ), m.action_horizon); U(fk("action_dim" ), m.action_dim); U(fk("max_state_dim" ), m.max_state_dim); @@ -152,25 +148,8 @@ bool load_config(const gguf_reader & g, Gr00tN1d5ModelArch & m, Config & cfg) { m.lm.cfg.rope.n_dims = (int) m.lm.cfg.head_dim; m.aex.embodiment_id = 24; - if (const char * e = std::getenv("VLA_GR00T_EMBODIMENT")) { - char * end = nullptr; - const long v = std::strtol(e, &end, 10); - if (end && *end == '\0') { - m.aex.embodiment_id = (int64_t) v; - } else { - const std::string js = g.str(fk("embodiment_tag_mapping")); - const std::string key = std::string("\"")+e+"\":"; - const size_t p = js.find(key); - if (p != std::string::npos) - m.aex.embodiment_id = std::strtol(js.c_str()+p+key.size(), nullptr, 10); - else std::fprintf(stderr, "vla(gr00tn1d5): embodiment tag '%s' not in embodiment_tag_mapping; using id %lld\n", e, (long long) m.aex.embodiment_id); - } - } - if (m.aex.embodiment_id < 0 || m.aex.embodiment_id >= m.max_embodiments) { - std::fprintf(stderr, "vla(gr00tn1d5): embodiment id %lld out of range [0,%lld)\n", - (long long) m.aex.embodiment_id, (long long) m.max_embodiments); + if (!resolve_embodiment("gr00tn1d5", g.str(fk("embodiment_tag_mapping")), nullptr, m.max_embodiments, m.aex.embodiment_id)) return false; - } cfg = Config{}; cfg.n_img = m.n_img_tokens; @@ -229,6 +208,7 @@ std::unique_ptr gr00t_n1_5_create(const std::string& mmproj_path, } if (!load_config(g, *m, m->cfg)) return nullptr; + m->times.build(m->num_steps, m->num_buckets, m->in_embed_dim, m->action_horizon); std::printf("vla(gr00tn1d5): vit=%lldd×%lldL×%lldh n_img_tok=%lld lm=Qwen3 %lldd×%lldL (%lldq/%lldkv×%lld) " "dit=%lldL×%lldh×%lld(inner %lld) interleave=%lld vlsa=%lldL×%lldh×%lld in_emb=%lld horizon=%lld action_dim=%lld N_steps=%lld embodiment=%lld resident=%s\n", @@ -285,7 +265,6 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { const int64_t E = in_embed_dim; const int64_t AD = action_dim; const int64_t AH = action_horizon; - const int64_t Nsa = 1+num_future+AH; int64_t n_views = 0; std::vector img_emb_host; @@ -338,7 +317,7 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { const int64_t SEQ = prompt.len(); std::vector inputs_embeds; - if (!fetch_embeds("gr00tn1d5", io, prompt, img_emb_ptr, H, inputs_embeds)) return {}; + if (!fetch_embeds(io, prompt, img_emb_ptr, H, inputs_embeds)) return {}; std::vector x_init; init_noise(in, (size_t) AH*AD, x_init); @@ -362,31 +341,8 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { ggml_tensor * vl = layer_norm(C, eagle, vlln_w, vlln_b, vlln_eps); ggml_tensor * vl_embs = vlsa.build(C, vl, SEQ); - ggml_tensor * state_features = aex.encode_state(C, t_state); - - std::vector Kc(dit.cfg.layers, nullptr), Vc(dit.cfg.layers, nullptr); - for (int64_t i=0; inb[1], (size_t)(Nsa-AH)*pred->nb[1])); - actions = ggml_add(C, actions, ggml_scale(C, vel, dt)); - } + ggml_tensor * actions = aex.denoise(C, dit, dit_interleave != 0, 1, t_state, future_tokens, vl_embs, vl_embs, + t_x0, t_tau, t_tproj); ggml_set_name(actions, "action_pred"); ggml_set_output(actions); @@ -418,15 +374,7 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { ggml_backend_tensor_set(gio.t_state, st.data(), 0, ggml_nbytes(gio.t_state)); ggml_backend_tensor_set(gio.t_x0, x_init.data(), 0, ggml_nbytes(gio.t_x0)); - - for (int64_t s=0; s tau, tpr; - action_sinusoid(bucket, E, AH, tau); - timesteps_proj(bucket, tpr); - ggml_backend_tensor_set(gio.t_tau[s], tau.data(), 0, ggml_nbytes(gio.t_tau[s])); - ggml_backend_tensor_set(gio.t_tproj[s], tpr.data(), 0, ggml_nbytes(gio.t_tproj[s])); - } + times.upload(gio.t_tau, gio.t_tproj); graph_unique_names(main_graph.graph()); const auto tc0 = std::chrono::steady_clock::now(); diff --git a/src/models/gr00tn1d6.cpp b/src/models/gr00tn1d6.cpp index 6617dea..a14bd4c 100644 --- a/src/models/gr00tn1d6.cpp +++ b/src/models/gr00tn1d6.cpp @@ -15,7 +15,6 @@ #include "arch.h" #include "options.h" #include "backend.h" -#include "env_flag.h" #include "gguf_reader.h" #include "layers/embed.h" #include "layers/ffn.h" @@ -36,10 +35,8 @@ #include #include #include -#include #include #include -#include #include #include @@ -58,6 +55,7 @@ struct Gr00tN1d6ModelArch : public ModelArchBase { ggml_type matmul_type = GGML_TYPE_F32; scratch_ctx vision_scratch; scratch_ctx merge_scratch; + FlowTimes times; struct MainKey { int64_t seq=-1, n_img=-1, seq_txt=-1, nsteps=-1; @@ -151,41 +149,8 @@ bool load_config(const gguf_reader & g, Gr00tN1d6ModelArch & m, Config & cfg) { m.lm.cfg.rope.n_dims = (int) m.lm.cfg.head_dim; m.aex.embodiment_id = 20; - { - const std::string js = g.str(fk("embodiment_id_mapping")); - auto lookup = [&](const char * key) -> long { - const std::string k = std::string("\"")+key+"\""; - size_t p = js.find(k); - if (p == std::string::npos) - return -1; - p = js.find(':', p+k.size()); - if (p == std::string::npos) - return -1; - return std::strtol(js.c_str()+p+1, nullptr, 10); - }; - - const long gr1 = lookup("gr1"); - if (gr1 >= 0) - m.aex.embodiment_id = gr1; - - if (const char * e = std::getenv("VLA_GR00T_EMBODIMENT")) { - char * end = nullptr; - const long v = std::strtol(e, &end, 10); - if (end && *end == '\0') { - m.aex.embodiment_id = v; - } else { - const long id = lookup(e); - if (id >= 0) - m.aex.embodiment_id = id; - else std::fprintf(stderr, "vla(gr00tn1d6): embodiment tag '%s' not in embodiment_id_mapping; using id %lld\n", e, (long long) m.aex.embodiment_id); - } - } - } - if (m.aex.embodiment_id < 0 || m.aex.embodiment_id >= m.max_embodiments) { - std::fprintf(stderr, "vla(gr00tn1d6): embodiment id %lld out of range [0,%lld)\n", - (long long) m.aex.embodiment_id, (long long) m.max_embodiments); + if (!resolve_embodiment("gr00tn1d6", g.str(fk("embodiment_id_mapping")), "gr1", m.max_embodiments, m.aex.embodiment_id)) return false; - } // pixel_shuffle_back writes (grid/shuffle)^2 tokens into a buffer sized from // n_img_tokens, so the KV has to agree with the grid it is derived from. @@ -261,6 +226,7 @@ std::unique_ptr gr00t_n1_6_create(const std::string& mmproj_path, } if (!load_config(g, *m, m->cfg)) return nullptr; + m->times.build(m->num_steps, m->num_buckets, m->in_embed_dim, m->action_horizon); std::printf("vla(gr00tn1d6): vit=%lldd×%lldL×%lldh (Linear patch embed) pixel_shuffle÷%lld ⇒ n_img_tok=%lld mlp1=LN(%lld)→Linear→GELU→Linear " "lm=Qwen3 %lldd×%lldL (%lldq/%lldkv×%lld) dit=AlternateVLDiT %lldL×%lldh×%lld(inner %lld) attend_text_every_n=%lld in_emb=%lld " @@ -325,7 +291,6 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { const int64_t c4 = vit.enc.cfg.hidden*r*r; const int64_t AD = action_dim; const int64_t AH = action_horizon; - const int64_t Nsa = 1+AH; int64_t n_views = 0; std::vector img_emb_host; @@ -415,7 +380,7 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { const int64_t SEQ_TXT = prompt.n_text(); std::vector inputs_embeds; - if (!fetch_embeds("gr00tn1d6", io, prompt, img_emb_ptr, H, inputs_embeds)) return {}; + if (!fetch_embeds(io, prompt, img_emb_ptr, H, inputs_embeds)) return {}; std::vector x_init; init_noise(in, (size_t) AH*AD, x_init); @@ -446,38 +411,8 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { ggml_tensor * vl_img = ggml_get_rows(C, vl_embs, t_img_idx); ggml_tensor * vl_txt = t_txt_idx ? ggml_get_rows(C, vl_embs, t_txt_idx) : vl_img; - ggml_tensor * state_features = aex.encode_state(C, t_state); - - const float dt = 1.0f/(float) num_steps; - const int64_t every2 = 2*attend_text_every_n; - - std::vector Kc(dit.cfg.layers, nullptr), Vc(dit.cfg.layers, nullptr); - for (int64_t i=0; inb[1], (size_t)(Nsa-AH)*pred->nb[1])); - actions = ggml_add(C, actions, ggml_scale(C, vel, dt)); - } + ggml_tensor * actions = aex.denoise(C, dit, dit_interleave != 0, 2*attend_text_every_n, t_state, nullptr, + vl_txt, vl_img, t_x0, t_tau, t_tproj); ggml_set_name(actions, "action_pred"); ggml_set_output(actions); @@ -512,15 +447,7 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { ggml_backend_tensor_set(gio.t_img_idx, prompt.image_pos.data(), 0, ggml_nbytes(gio.t_img_idx)); if (gio.t_txt_idx) ggml_backend_tensor_set(gio.t_txt_idx, prompt.text_pos.data(), 0, ggml_nbytes(gio.t_txt_idx)); - - for (int64_t s=0; s tau, tpr; - action_sinusoid(bucket, E, AH, tau); - timesteps_proj(bucket, tpr); - ggml_backend_tensor_set(gio.t_tau[s], tau.data(), 0, ggml_nbytes(gio.t_tau[s])); - ggml_backend_tensor_set(gio.t_tproj[s], tpr.data(), 0, ggml_nbytes(gio.t_tproj[s])); - } + times.upload(gio.t_tau, gio.t_tproj); graph_unique_names(main_graph.graph()); const auto tc0 = std::chrono::steady_clock::now(); diff --git a/src/models/gr00tn1d7.cpp b/src/models/gr00tn1d7.cpp index 9d7dc8d..eb24f6f 100644 --- a/src/models/gr00tn1d7.cpp +++ b/src/models/gr00tn1d7.cpp @@ -13,53 +13,38 @@ // limitations under the License. #include "arch.h" -#include "layers/attn.h" -#include "layers/linear.h" #include "layers/norm.h" #include "modules/action_expert.h" #include "modules/dit_head.h" #include "modules/encoder.h" +#include "modules/prompt.h" #include "modules/qwen3_lm.h" #include "options.h" #include "model.h" #include "ggml.h" -#include "ggml-cpu.h" #include "ggml-backend.h" #include "backend.h" -#include "gguf.h" #include "gguf_reader.h" #include "scratch_ctx.h" #include "layers/embed.h" #include "modules/qwen3vl_vit.h" -#include "env_flag.h" #include -#include #include #include #include #include -#include #include -#include -#include #include #include namespace vla { -namespace { - - -struct VlsaLayerW { ggml_tensor *n1w,*n1b,*n3w,*n3b,*Wq,*bq,*Wk,*bk,*Wv,*bv,*Wo,*bo,*Wff0,*bff0,*Wff2,*bff2; }; - -} struct Gr00tN1d7ModelArch : public ModelArchBase { Gr00tN1d7ModelArch() : ModelArchBase(Arch::GR00T_N1_7) {} ~Gr00tN1d7ModelArch() override; - std::string gguf_path; ggml_backend_t backend = nullptr; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; @@ -67,18 +52,13 @@ struct Gr00tN1d7ModelArch : public ModelArchBase { ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; - int64_t vit_hidden=1024, vit_layers=24, vit_heads=16, vit_inter=4096; - int64_t patch_size=16, temporal_patch=2, spatial_merge=2, vit_num_pos=2304, vit_patch_flat=1536, vit_merged_dim=4096; - int64_t deepstack_idx[3] = {5, 11, 17}; int64_t lm_hidden=2048, lm_layers=16, n_q=16, n_kv=8, lm_head_dim=128, lm_inter=6144, vocab=151936, image_token_index=151655; int64_t vlsa_layers=4, vlsa_heads=32, vlsa_head_dim=64, vlsa_ff_inner=8192; int64_t bb_embed_dim=2048, in_embed_dim=1536, dit_hidden=1536, dit_heads=32, dit_head_dim=48, dit_layers=32, dit_interleave=1, attend_text_every_n=2; int64_t action_horizon=40, action_dim=132, max_state_dim=132; int64_t num_steps=4, num_buckets=1000, max_embodiments=32, max_seq_len=1024; - int64_t image_target_size=256; - float vit_ln_eps=1e-6f, vit_rope_base=10000.0f, lm_rms_eps=1e-6f, lm_rope_base=5000000.0f; - float vlln_eps=1e-5f, vlsa_ln_eps=1e-5f, ln_eps=1e-5f, norm_out_eps=1e-6f, connector_ln_eps=1e-6f; - int64_t embodiment_id = 2; + float lm_rms_eps=1e-6f, lm_rope_base=5000000.0f; + float vlln_eps=1e-5f, vlsa_ln_eps=1e-5f, ln_eps=1e-5f, norm_out_eps=1e-6f; Qwen3VLTower vit; Qwen3LM lm; @@ -87,14 +67,9 @@ struct Gr00tN1d7ModelArch : public ModelArchBase { DitHead dit; ggml_tensor *vlln_w=nullptr,*vlln_b=nullptr; - bool caches_ready = false; - std::vector c_grow, c_gcol; - std::vector c_rope_cos, c_rope_sin; - std::vector c_pos_interp; - std::vector> c_tau, c_tproj; - std::vector c_mask; int64_t c_mask_seq = -1; - gguf_reader io; - bool build_caches(); + FlowTimes times; + std::vector c_mask; int64_t c_mask_seq = -1; + gguf_reader io{"gr00tn1d7"}; struct MainKey { int64_t seq=-1, n_img=-1, seq_txt=-1, nsteps=-1; bool deepstack=false; @@ -116,18 +91,12 @@ struct Gr00tN1d7ModelArch : public ModelArchBase { namespace { - - - - bool load_config(const gguf_reader & g, Gr00tN1d7ModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; auto F = [&](const char * k, float & dst) { if (g.has(k)) dst = g.f32(k); }; auto fk = [&](const char * s) { thread_local char b[64]; std::snprintf(b, sizeof(b), "gr00t_n1_7.%s", s); return b; }; - U(fk("vit_hidden"), m.vit_hidden); U(fk("vit_layers"), m.vit_layers); U(fk("vit_heads"), m.vit_heads); U(fk("vit_inter"), m.vit_inter); - U(fk("patch_size"), m.patch_size); U(fk("temporal_patch_size"), m.temporal_patch); U(fk("spatial_merge_size"), m.spatial_merge); - U(fk("vit_num_position_embeddings"), m.vit_num_pos); U(fk("vit_patch_flat"), m.vit_patch_flat); U(fk("vit_merged_dim"), m.vit_merged_dim); - U(fk("deepstack_idx_0"), m.deepstack_idx[0]); U(fk("deepstack_idx_1"), m.deepstack_idx[1]); U(fk("deepstack_idx_2"), m.deepstack_idx[2]); + if (!m.vit.load_config("gr00tn1d7", g, "gr00t_n1_7")) + return false; U(fk("lm_hidden"), m.lm_hidden); U(fk("lm_layers_used"), m.lm_layers); U(fk("lm_q_heads"), m.n_q); U(fk("lm_kv_heads"), m.n_kv); U(fk("lm_head_dim"), m.lm_head_dim); U(fk("lm_inter"), m.lm_inter); U(fk("vocab_size"), m.vocab); U(fk("image_token_index"), m.image_token_index); U(fk("vlsa_layers"), m.vlsa_layers); U(fk("vlsa_heads"), m.vlsa_heads); U(fk("vlsa_head_dim"), m.vlsa_head_dim); U(fk("vlsa_ff_inner"), m.vlsa_ff_inner); @@ -136,31 +105,15 @@ bool load_config(const gguf_reader & g, Gr00tN1d7ModelArch & m, Config & cfg) { U(fk("attend_text_every_n_blocks"), m.attend_text_every_n); U(fk("action_horizon"), m.action_horizon); U(fk("action_dim"), m.action_dim); U(fk("max_state_dim"), m.max_state_dim); U(fk("num_inference_timesteps"), m.num_steps); U(fk("num_timestep_buckets"), m.num_buckets); U(fk("max_num_embodiments"), m.max_embodiments); U(fk("max_seq_len"), m.max_seq_len); - U(fk("image_target_size"), m.image_target_size); - - // merge_block_coords only enumerates the patch grid exactly when the spatial - // merge divides it; otherwise it emits rows past the position table. - if (m.patch_size <= 0 || m.spatial_merge <= 0 || m.image_target_size%m.patch_size != 0 || - (m.image_target_size/m.patch_size)%m.spatial_merge != 0) { - std::fprintf(stderr, "vla(gr00tn1d7): image %lld / patch %lld / merge %lld do not divide evenly\n", - (long long) m.image_target_size, (long long) m.patch_size, (long long) m.spatial_merge); - return false; - } - if (m.vit_heads <= 0 || m.attend_text_every_n <= 0 || m.vit_patch_flat != 3*m.temporal_patch*m.patch_size*m.patch_size) { - std::fprintf(stderr, "vla(gr00tn1d7): vit_heads %lld, attend_text_every_n_blocks %lld or vit_patch_flat %lld is inconsistent\n", - (long long) m.vit_heads, (long long) m.attend_text_every_n, (long long) m.vit_patch_flat); + + if (m.attend_text_every_n <= 0) { + std::fprintf(stderr, "vla(gr00tn1d7): attend_text_every_n_blocks %lld must be positive\n", (long long) m.attend_text_every_n); return false; } - if (const char * ns = std::getenv("VLA_NUM_STEPS")) { - char * end = nullptr; long v = std::strtol(ns, &end, 10); - if (end && *end == '\0' && v >= 1) { - m.num_steps = (int64_t) v; - std::fprintf(stderr, "vla(gr00tn1d7): VLA_NUM_STEPS override → num_steps=%lld\n", (long long) v); - } - } - F(fk("vit_ln_eps"), m.vit_ln_eps); F(fk("lm_rms_eps"), m.lm_rms_eps); F(fk("ln_eps"), m.ln_eps); F(fk("norm_out_eps"), m.norm_out_eps); - F(fk("vlln_eps"), m.vlln_eps); F(fk("vlsa_ln_eps"), m.vlsa_ln_eps); F(fk("connector_ln_eps"), m.connector_ln_eps); F(fk("vit_rope_theta"), m.vit_rope_base); + env_num_steps("gr00tn1d7", m.num_steps); + F(fk("lm_rms_eps"), m.lm_rms_eps); F(fk("ln_eps"), m.ln_eps); F(fk("norm_out_eps"), m.norm_out_eps); + F(fk("vlln_eps"), m.vlln_eps); F(fk("vlsa_ln_eps"), m.vlsa_ln_eps); if (g.has(fk("lm_rope_theta"))) m.lm_rope_base = (float) g.f64(fk("lm_rope_theta")); @@ -194,29 +147,11 @@ bool load_config(const gguf_reader & g, Gr00tN1d7ModelArch & m, Config & cfg) { m.dit.cfg.norm_out_eps = m.norm_out_eps; m.aex.embodiment_id = 2; - { - const std::string js = g.str(fk("embodiment_id_mapping")); - auto lookup = [&](const char * key) -> long { - const std::string k = std::string("\"")+key + "\""; - size_t p = js.find(k); if (p == std::string::npos) return -1; - p = js.find(':', p+k.size()); if (p == std::string::npos) return -1; - return std::strtol(js.c_str()+p+1, nullptr, 10); - }; - long ls = lookup("libero_sim"); if (ls >= 0) m.aex.embodiment_id = ls; - if (const char * e = std::getenv("VLA_GR00T_EMBODIMENT")) { - char * end = nullptr; long v = std::strtol(e, &end, 10); - if (end && *end == '\0') - m.aex.embodiment_id = v; - else { long id = lookup(e); if (id >= 0) m.aex.embodiment_id = id; else std::fprintf(stderr, "vla(gr00tn1d7): embodiment tag '%s' not in embodiment_id_mapping; using id %lld\n", e, (long long) m.aex.embodiment_id); } - } - } - if (m.aex.embodiment_id < 0 || m.aex.embodiment_id >= m.max_embodiments) { - std::fprintf(stderr, "vla(gr00tn1d7): embodiment id %lld out of range [0,%lld)\n", (long long) m.aex.embodiment_id, (long long) m.max_embodiments); + if (!resolve_embodiment("gr00tn1d7", g.str(fk("embodiment_id_mapping")), "libero_sim", m.max_embodiments, m.aex.embodiment_id)) return false; - } cfg = Config{}; - cfg.n_img = (m.image_target_size/m.patch_size/m.spatial_merge)*(m.image_target_size/m.patch_size/m.spatial_merge); + cfg.n_img = m.vit.n_tokens(); cfg.n_lang = m.max_seq_len; cfg.n_state = 1; cfg.n_suffix = m.action_horizon; cfg.max_state_dim = m.max_state_dim; cfg.max_action_dim = m.action_dim; cfg.real_state_dim = m.max_state_dim; cfg.real_action_dim = m.action_dim; @@ -250,23 +185,23 @@ std::unique_ptr gr00t_n1_7_create(const std::string& mmproj_path, std::printf("vla(gr00tn1d7): note - mmproj '%s' is ignored (the vision tower is bundled in the combined GGUF)\n", mmproj_path.c_str()); auto m = std::make_unique(); - m->gguf_path = ckpt_path; m->matmul_type = opts.weight_dtype.value_or(vla::default_weight_dtype(GGML_TYPE_BF16)); - gguf_reader g("gr00tn1d7"); - if (!g.open(ckpt_path)) + if (!m->io.open(ckpt_path)) return nullptr; + gguf_reader & g = m->io; if (!g.has("gr00t_n1_7.architecture")) { std::fprintf(stderr, "vla(gr00tn1d7): %s is not a gr00t_n1_7 GGUF\n", ckpt_path.c_str()); return nullptr; } if (!load_config(g, *m, m->cfg)) return nullptr; + m->times.build(m->num_steps, m->num_buckets, m->in_embed_dim, m->action_horizon); std::printf("vla(gr00tn1d7): vit=Qwen3-VL %lldd×%lldL×%lldh (Conv3d patch %lld², temporal %lld; learned pos %lld + 2D rope; deepstack@{%lld,%lld,%lld}; merge÷%lld) " "lm=Qwen3-VL %lldd×%lldL (%lldq/%lldkv×%lld, θ=%g) vlsa=%lldL×%lldh×%lld dit=AlternateVLDiT %lldL×%lldh×%lld(inner %lld) attend_text_every_n=%lld " "in_emb=%lld horizon=%lld action_dim=%lld max_state=%lld N_steps=%lld embodiment=%lld resident=%s\n", - (long long) m->vit_hidden, (long long) m->vit_layers, (long long) m->vit_heads, (long long) m->patch_size, (long long) m->temporal_patch, - (long long) m->vit_num_pos, (long long) m->deepstack_idx[0], (long long) m->deepstack_idx[1], (long long) m->deepstack_idx[2], (long long) m->spatial_merge, + (long long) m->vit.hidden, (long long) m->vit.layers, (long long) m->vit.heads, (long long) m->vit.patch, (long long) m->vit.temporal, + (long long) m->vit.num_pos, (long long) m->vit.deepstack_idx[0], (long long) m->vit.deepstack_idx[1], (long long) m->vit.deepstack_idx[2], (long long) m->vit.merge, (long long) m->lm_hidden, (long long) m->lm_layers, (long long) m->n_q, (long long) m->n_kv, (long long) m->lm_head_dim, (double) m->lm_rope_base, (long long) m->vlsa_layers, (long long) m->vlsa_heads, (long long) m->vlsa_head_dim, (long long) m->dit_layers, (long long) m->dit_heads, (long long) m->dit_head_dim, (long long) m->dit_hidden, (long long) m->attend_text_every_n, (long long) m->in_embed_dim, @@ -290,7 +225,7 @@ std::unique_ptr gr00t_n1_7_create(const std::string& mmproj_path, WeightLoader L("gr00tn1d7", g, m->ctx_weights, m->matmul_type); - m->vit.declare(L, "vit", m->vit_layers); + m->vit.declare(L, "vit"); m->lm.declare(L, "vlm"); m->vlln_w = L.f32("aex.vlln.weight"); @@ -307,131 +242,29 @@ std::unique_ptr gr00t_n1_7_create(const std::string& mmproj_path, std::printf("vla(gr00tn1d7): weights resident in %.2f GiB (%s) - incl. Qwen3-VL vision tower + deepstack + vl_self_attention; embodiment id %lld\n", ggml_backend_buffer_get_size(m->weight_buf)/(1024.0*1024.0*1024.0), dtype_name(m->matmul_type), (long long) m->aex.embodiment_id); - if (!m->build_caches()) { - std::fprintf(stderr, "vla(gr00tn1d7): build_caches failed\n"); + if (!m->vit.build_caches("gr00tn1d7", m->io)) return nullptr; - } return m; } -bool Gr00tN1d7ModelArch::build_caches() { - if (caches_ready) - return true; - const int64_t side = image_target_size, ps = patch_size, m2 = spatial_merge; - const int64_t grid = side/ps; - const int64_t hd_vit = vit_hidden/vit_heads; - const int64_t num_side = (int64_t) std::lround(std::sqrt((double) vit_num_pos)); - const int64_t E = in_embed_dim, AH = action_horizon; - - merge_block_coords(grid, grid, m2, c_grow, c_gcol); - vit_rope_tables(c_grow, c_gcol, hd_vit, (double) vit_rope_base, c_rope_cos, c_rope_sin); - - if (!io.open(gguf_path)) { - std::fprintf(stderr, "vla(gr00tn1d7): build_caches: io.open(%s) failed\n", gguf_path.c_str()); - return false; - } - std::vector pos_table = io.read_f32("vit.pos_embd"); - if (pos_table.empty() || (int64_t) pos_table.size() != vit_num_pos * vit_hidden) { - std::fprintf(stderr, "vla(gr00tn1d7): build_caches: vit.pos_embd unreadable\n"); return false; - } - if (!interp_pos_embed(pos_table, num_side, vit_hidden, c_grow, c_gcol, grid, grid, c_pos_interp)) { - std::fprintf(stderr, "vla(gr00tn1d7): build_caches: vit_num_position_embeddings %lld is not a square\n", (long long) vit_num_pos); return false; - } - - c_tau.assign((size_t) num_steps, {}); c_tproj.assign((size_t) num_steps, {}); - for (int64_t s=0; s Gr00tN1d7ModelArch::predict(const Inputs& in) { const auto t0 = std::chrono::steady_clock::now(); stats = Stats{}; const int64_t H = lm_hidden, E = in_embed_dim; - const int64_t side = image_target_size; - const int64_t ps = patch_size, m2 = spatial_merge; - const int64_t grid = side/ps; - const int64_t n_patches = grid * grid; - const int64_t K = (grid/m2)*(grid/m2); - const int64_t hd_vit = vit_hidden/vit_heads; - const int64_t AD = action_dim, AH = action_horizon, Nsa = 1+AH; + const int64_t K = vit.n_tokens(); + const int64_t AD = action_dim, AH = action_horizon; const bool do_dump = (std::getenv("VLA_GR00T_N17_DUMP") != nullptr); - if (!caches_ready) { std::fprintf(stderr, "vla(gr00tn1d7): caches not ready\n"); return {}; } - const std::vector & grow = c_grow, & gcol = c_gcol; - const std::vector & rope_cos = c_rope_cos, & rope_sin = c_rope_sin, & pos_interp = c_pos_interp; - int64_t n_views = 0; std::vector img_emb_host, ds_host[3]; const float * img_emb_ptr = nullptr; if (in.precomputed_img_emb && in.n_img_views > 0) { n_views = in.n_img_views; img_emb_ptr = in.precomputed_img_emb; - - for (int j=0; j<3; ++j) - ds_host[j].assign((size_t) n_views * K * H, 0.0f); } else if (in.images && in.n_images > 0) { n_views = in.n_images; - img_emb_host.assign((size_t) n_views * K * H, 0.0f); - for (int j=0; j<3; ++j) - ds_host[j].assign((size_t) n_views * K * H, 0.0f); - - ggml_context * VC = vision_scratch.reset((size_t) 512*1024*1024); - if (!VC) { std::fprintf(stderr, "vla(gr00tn1d7): ggml_init(vision ctx) failed\n"); return {}; } - ggml_tensor * t_patches = ggml_new_tensor_2d(VC, GGML_TYPE_F32, vit_patch_flat, n_patches); ggml_set_input(t_patches); - ggml_tensor * t_pos = ggml_new_tensor_2d(VC, GGML_TYPE_F32, vit_hidden, n_patches); ggml_set_input(t_pos); - ggml_tensor * t_cos = ggml_new_tensor_2d(VC, GGML_TYPE_F32, hd_vit, n_patches); ggml_set_input(t_cos); - ggml_tensor * t_sin = ggml_new_tensor_2d(VC, GGML_TYPE_F32, hd_vit, n_patches); ggml_set_input(t_sin); - ggml_tensor * h = ggml_add(VC, ggml_add(VC, ggml_mul_mat(VC, vit.patch_w, t_patches), vit.patch_b), t_pos); - - ggml_set_output(h); - ggml_tensor * stash[3] = {nullptr, nullptr, nullptr}; - for (int64_t i=0; i patches; - bool vok = true; - for (int64_t v=0; v(std::chrono::steady_clock::now()-tv0).count(); if (!vok) return {}; img_emb_ptr = img_emb_host.data(); @@ -440,68 +273,28 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { } const int64_t n_img = n_views * K; - std::vector input_ids; - int64_t n_img_slots = 0; - for (int j=0; j max_seq_len) { std::fprintf(stderr, "vla(gr00tn1d7): prompt too long (%lld > %lld)\n", (long long) SEQ, (long long) max_seq_len); return {}; } - - std::vector inputs_embeds((size_t) SEQ * H); - if (!io.fetch_rows_f32("token_embd.weight", input_ids, inputs_embeds.data(), H)) return {}; - { int64_t k = 0; - for (int64_t p=0; p= n_img) { std::fprintf(stderr, "vla(gr00tn1d7): more tokens than ViT embeds\n"); return {}; } - std::memcpy(inputs_embeds.data()+p * H, img_emb_ptr+k * H, H * sizeof(float)); ++k; - } - } + Prompt prompt; + if (!build_prompt("gr00tn1d7", in, n_img, (int32_t) image_token_index, max_seq_len, prompt)) return {}; + const int64_t SEQ = prompt.len(), SEQ_TXT = prompt.n_text(); - std::vector image_pos_idx, text_pos_idx; - image_pos_idx.reserve((size_t) n_img); text_pos_idx.reserve((size_t) (SEQ-n_img)); - for (int64_t p=0; p inputs_embeds; + if (!fetch_embeds(io, prompt, img_emb_ptr, H, inputs_embeds)) return {}; + + std::vector pp; + if (!mrope_positions("gr00tn1d7", prompt.ids, (int32_t) image_token_index, vit.grid()/vit.merge, pp)) return {}; std::vector> ds_pad(3); - const bool inject_deepstack = (in.images && in.n_images > 0); + const bool inject_deepstack = !img_emb_host.empty(); if (inject_deepstack) for (int j=0; j<3; ++j) { ds_pad[j].assign((size_t) SEQ * H, 0.0f); for (int64_t k=0; k x_init((size_t) AH * AD); - if (in.noise) - std::memcpy(x_init.data(), in.noise, x_init.size()*sizeof(float)); - else { - std::mt19937 rng((uint32_t) std::chrono::steady_clock::now().time_since_epoch().count()); - std::normal_distribution nd(0.f, 1.f); - for (auto & v : x_init) - v = nd(rng); - } + std::vector x_init; + init_noise(in, (size_t) AH*AD, x_init); // On by default: 16% faster, bit-identical. Set VLA_GR00T_GRAPH_CACHE=0 to opt out. // Dumping adds graph outputs, so it always rebuilds. @@ -552,7 +345,7 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { eagle = h; ggml_set_name(eagle, "eagle"); ggml_set_output(eagle); - vl_embs = ggml_add(C, ggml_mul(C, ggml_norm(C, eagle, vlln_eps), vlln_w), vlln_b); + vl_embs = layer_norm(C, eagle, vlln_w, vlln_b, vlln_eps); if (do_dump) { ggml_set_output(vl_embs); vlsa_dump.push_back(vl_embs); @@ -569,45 +362,8 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { ggml_tensor * vl_img = ggml_get_rows(C, vl_embs, t_img_idx); ggml_tensor * vl_txt = (t_txt_idx ? ggml_get_rows(C, vl_embs, t_txt_idx) : vl_img); - ggml_tensor * state_features = cat_linear(C, aex.se_l2W, aex.se_l2b, aex.embodiment_id, ggml_relu(C, cat_linear(C, aex.se_l1W, aex.se_l1b, aex.embodiment_id, t_state))); - - const float dt = 1.0f/(float) num_steps; - const int64_t every2 = 2*attend_text_every_n; - - std::vector Kc(dit_layers, nullptr), Vc(dit_layers, nullptr); - for (int64_t i=0; inb[1], 0)); - ggml_tensor * sa = ggml_concat(C, state_features, af, 1); - ggml_tensor * hh = sa; - for (int64_t i=0; inb[1], (size_t) (Nsa-AH)*pred->nb[1])); - actions = ggml_add(C, actions, ggml_scale(C, vel, dt)); - } + ggml_tensor * actions = aex.denoise(C, dit, dit_interleave != 0, 2*attend_text_every_n, t_state, nullptr, + vl_txt, vl_img, t_x0, t_tau, t_tproj); ggml_set_name(actions, "action_pred"); ggml_set_output(actions); gio.t_embeds=t_embeds; gio.t_pos=t_pos; gio.t_lmmask=t_lmmask; gio.t_state=t_state; gio.t_x0=t_x0; @@ -622,92 +378,26 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { MainIO & gio = mg.io(); ggml_cgraph * gf = mg.graph(); - ggml_tensor * t_embeds = gio.t_embeds, * t_pos = gio.t_pos, * t_lmmask = gio.t_lmmask, * t_state = gio.t_state, * t_x0 = gio.t_x0; - ggml_tensor * t_ds[3] = { gio.t_ds[0], gio.t_ds[1], gio.t_ds[2] }; - ggml_tensor * t_img_idx = gio.t_img_idx, * t_txt_idx = gio.t_txt_idx, * actions = gio.actions; - std::vector & t_tau = gio.t_tau; std::vector & t_tproj = gio.t_tproj; - - ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); - { - - const int64_t llm_grid_h = image_target_size/patch_size/spatial_merge; - const int64_t llm_grid_w = llm_grid_h; - - std::vector pp((size_t) 4*SEQ, 0); - int64_t st = 0, st_idx = 0; - while (st < SEQ) { - int64_t img_start = -1; - for (int64_t i=st; i max_image_pos) - max_image_pos = llm_grid_h-1; - if (llm_grid_w-1 > max_image_pos) - max_image_pos = llm_grid_w-1; - st_idx = image_offset+max_image_pos+1; - st = img_end; - } - std::memcpy(pp.data()+(size_t) 3*SEQ, pp.data()+(size_t) 0*SEQ, (size_t) SEQ * sizeof(int32_t)); - ggml_backend_tensor_set(t_pos, pp.data(), 0, ggml_nbytes(t_pos)); - } + ggml_backend_tensor_set(gio.t_embeds, inputs_embeds.data(), 0, ggml_nbytes(gio.t_embeds)); + ggml_backend_tensor_set(gio.t_pos, pp.data(), 0, ggml_nbytes(gio.t_pos)); if (c_mask_seq != SEQ) { build_causal_mask(SEQ, c_mask); c_mask_seq = SEQ; } - ggml_backend_tensor_set(t_lmmask, c_mask.data(), 0, ggml_nbytes(t_lmmask)); + ggml_backend_tensor_set(gio.t_lmmask, c_mask.data(), 0, ggml_nbytes(gio.t_lmmask)); { std::vector st(max_state_dim, 0.0f); for (int64_t i=0; i Gr00tN1d7ModelArch::predict(const Inputs& in) { stats.ms_inference = std::chrono::duration(tc1-tc0).count(); std::vector out((size_t) AH * AD); - ggml_backend_tensor_get(actions, out.data(), 0, out.size()*sizeof(float)); + ggml_backend_tensor_get(gio.actions, out.data(), 0, out.size()*sizeof(float)); if (const char * dump = std::getenv("VLA_GR00T_N17_DUMP")) { auto dump_t = [&](const char * name, ggml_tensor * t) { diff --git a/src/models/vla_jepa.cpp b/src/models/vla_jepa.cpp index f8d5161..27e81e7 100644 --- a/src/models/vla_jepa.cpp +++ b/src/models/vla_jepa.cpp @@ -17,44 +17,32 @@ #include "model.h" #include "ggml.h" -#include "ggml-cpu.h" #include "ggml-backend.h" #include "backend.h" -#include "gguf.h" #include "gguf_reader.h" #include "scratch_ctx.h" #include "layers/embed.h" -#include "layers/linear.h" -#include "layers/norm.h" +#include "layers/ffn.h" #include "modules/dit_head.h" +#include "modules/prompt.h" #include "modules/qwen3_lm.h" #include "modules/qwen3vl_vit.h" -#include "env_flag.h" #include -#include #include #include #include #include -#include #include -#include -#include #include #include namespace vla { -namespace { - - -} struct VlaJepaModelArch : public ModelArchBase { VlaJepaModelArch() : ModelArchBase(Arch::VLA_JEPA) {} ~VlaJepaModelArch() override; - std::string gguf_path; ggml_backend_t backend = nullptr; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; @@ -85,16 +73,12 @@ struct VlaJepaModelArch : public ModelArchBase { ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; - int64_t vit_hidden=1024, vit_layers=24, vit_heads=16, vit_inter=4096; - int64_t patch_size=16, temporal_patch=2, spatial_merge=2, vit_num_pos=2304, vit_patch_flat=1536, vit_merged_dim=4096; - int64_t deepstack_idx[3] = {5, 11, 17}; int64_t lm_hidden=2048, lm_layers=28, n_q=16, n_kv=8, lm_head_dim=128, lm_inter=6144, vocab=151936; int64_t image_token_index=151655, embodied_token_id=151697; - int64_t image_target_size=256; int64_t dit_hidden=768, dit_heads=12, dit_head_dim=64, dit_layers=16, cross_dim=2048, output_dim=1024, time_proj_dim=256; int64_t action_dim=7, state_dim=8, action_horizon=7, num_future=32, num_steps=4, num_buckets=1000; - float vit_ln_eps=1e-6f, vit_rope_base=10000.0f, lm_rms_eps=1e-6f, lm_rope_base=5000000.0f, connector_ln_eps=1e-6f; + float lm_rms_eps=1e-6f, lm_rope_base=5000000.0f; float dit_ln_eps=1e-5f, dit_norm_out_eps=1e-6f; Qwen3VLTower vit; @@ -106,62 +90,33 @@ struct VlaJepaModelArch : public ModelArchBase { ggml_tensor *ad_l1W=nullptr,*ad_l1b=nullptr,*ad_l2W=nullptr,*ad_l2b=nullptr; ggml_tensor *future_tokens=nullptr,*pos_embd=nullptr; - bool caches_ready = false; - std::vector c_grow, c_gcol; - std::vector c_rope_cos, c_rope_sin, c_pos_interp; - std::vector> c_tau, c_tproj; - std::vector c_mask; int64_t c_mask_seq = -1; - gguf_reader io; - bool build_caches(); + FlowTimes times; + std::vector c_mask; int64_t c_mask_seq = -1; + gguf_reader io{"vla_jepa"}; std::vector predict(const Inputs& in) override; }; namespace { - - bool load_config(const gguf_reader & g, VlaJepaModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; auto F = [&](const char * k, float & dst) { if (g.has(k)) dst = g.f32(k); }; auto fk = [&](const char * s) { thread_local char b[64]; std::snprintf(b, sizeof(b), "vla_jepa.%s", s); return b; }; - U(fk("vit_hidden"), m.vit_hidden); U(fk("vit_layers"), m.vit_layers); U(fk("vit_heads"), m.vit_heads); U(fk("vit_inter"), m.vit_inter); - U(fk("patch_size"), m.patch_size); U(fk("temporal_patch_size"), m.temporal_patch); U(fk("spatial_merge_size"), m.spatial_merge); - U(fk("vit_num_position_embeddings"), m.vit_num_pos); U(fk("vit_patch_flat"), m.vit_patch_flat); U(fk("vit_merged_dim"), m.vit_merged_dim); - U(fk("deepstack_idx_0"), m.deepstack_idx[0]); U(fk("deepstack_idx_1"), m.deepstack_idx[1]); U(fk("deepstack_idx_2"), m.deepstack_idx[2]); + if (!m.vit.load_config("vla_jepa", g, "vla_jepa")) + return false; U(fk("lm_hidden"), m.lm_hidden); U(fk("lm_layers"), m.lm_layers); U(fk("lm_q_heads"), m.n_q); U(fk("lm_kv_heads"), m.n_kv); U(fk("lm_head_dim"), m.lm_head_dim); U(fk("lm_inter"), m.lm_inter); U(fk("vocab_size"), m.vocab); U(fk("image_token_index"), m.image_token_index); U(fk("embodied_action_token_id"), m.embodied_token_id); - U(fk("image_target_size"), m.image_target_size); U(fk("dit_hidden"), m.dit_hidden); U(fk("dit_heads"), m.dit_heads); U(fk("dit_head_dim"), m.dit_head_dim); U(fk("dit_layers"), m.dit_layers); U(fk("cross_dim"), m.cross_dim); U(fk("output_dim"), m.output_dim); U(fk("time_proj_dim"), m.time_proj_dim); U(fk("action_dim"), m.action_dim); U(fk("state_dim"), m.state_dim); U(fk("action_horizon"), m.action_horizon); U(fk("num_future_tokens"), m.num_future); U(fk("num_inference_timesteps"), m.num_steps); U(fk("num_timestep_buckets"), m.num_buckets); - if (const char * ns = std::getenv("VLA_NUM_STEPS")) { - char * end = nullptr; long v = std::strtol(ns, &end, 10); - if (end && *end == '\0' && v >= 1) { - m.num_steps = (int64_t) v; - std::fprintf(stderr, "vla(vla_jepa): VLA_NUM_STEPS override → num_steps=%lld\n", (long long) v); - } - } - F(fk("vit_ln_eps"), m.vit_ln_eps); F(fk("lm_rms_eps"), m.lm_rms_eps); F(fk("connector_ln_eps"), m.connector_ln_eps); - F(fk("vit_rope_theta"), m.vit_rope_base); F(fk("dit_ln_eps"), m.dit_ln_eps); F(fk("dit_norm_out_eps"), m.dit_norm_out_eps); + env_num_steps("vla_jepa", m.num_steps); + F(fk("lm_rms_eps"), m.lm_rms_eps); F(fk("dit_ln_eps"), m.dit_ln_eps); F(fk("dit_norm_out_eps"), m.dit_norm_out_eps); if (g.has(fk("lm_rope_theta"))) m.lm_rope_base = (float) g.f64(fk("lm_rope_theta")); - // merge_block_coords only enumerates the patch grid exactly when the spatial - // merge divides it; otherwise it emits rows past the position table. - if (m.patch_size <= 0 || m.spatial_merge <= 0 || m.image_target_size%m.patch_size != 0 || - (m.image_target_size/m.patch_size)%m.spatial_merge != 0) { - std::fprintf(stderr, "vla(vla_jepa): image %lld / patch %lld / merge %lld do not divide evenly\n", - (long long) m.image_target_size, (long long) m.patch_size, (long long) m.spatial_merge); - return false; - } - if (m.vit_heads <= 0 || m.vit_patch_flat != 3*m.temporal_patch*m.patch_size*m.patch_size) { - std::fprintf(stderr, "vla(vla_jepa): vit_heads %lld or vit_patch_flat %lld is inconsistent\n", - (long long) m.vit_heads, (long long) m.vit_patch_flat); - return false; - } // timesteps_proj always emits 256 floats into the time-projection input. if (m.time_proj_dim != 256) { std::fprintf(stderr, "vla(vla_jepa): time_proj_dim %lld, expected 256\n", (long long) m.time_proj_dim); @@ -192,7 +147,7 @@ bool load_config(const gguf_reader & g, VlaJepaModelArch & m, Config & cfg) { m.dit.cfg.norm_out_eps = m.dit_norm_out_eps; cfg = Config{}; - cfg.n_img = (m.image_target_size/m.patch_size/m.spatial_merge)*(m.image_target_size/m.patch_size/m.spatial_merge); + cfg.n_img = m.vit.n_tokens(); cfg.n_lang = 1024; cfg.n_state = 1; cfg.n_suffix = m.action_horizon; cfg.max_state_dim = m.state_dim; cfg.max_action_dim = m.action_dim; cfg.real_state_dim = m.state_dim; cfg.real_action_dim = m.action_dim; @@ -225,21 +180,21 @@ std::unique_ptr vla_jepa_create(const std::string& mmproj_path, std::printf("vla(vla_jepa): note - mmproj '%s' is ignored (the vision tower is bundled in the combined GGUF)\n", mmproj_path.c_str()); auto m = std::make_unique(); - m->gguf_path = ckpt_path; m->matmul_type = opts.weight_dtype.value_or(vla::default_weight_dtype(GGML_TYPE_BF16)); - gguf_reader g("vla_jepa"); - if (!g.open(ckpt_path)) + if (!m->io.open(ckpt_path)) return nullptr; + gguf_reader & g = m->io; if (!g.has("vla_jepa.architecture")) { std::fprintf(stderr, "vla(vla_jepa): %s is not a vla_jepa GGUF\n", ckpt_path.c_str()); return nullptr; } if (!load_config(g, *m, m->cfg)) return nullptr; + m->times.build(m->num_steps, m->num_buckets, m->dit_hidden, m->action_horizon); std::printf("vla(vla_jepa): vit=Qwen3-VL %lldd×%lldL (deepstack@{%lld,%lld,%lld}, merge÷%lld) lm=Qwen3-VL %lldd×%lldL (%lldq/%lldkv×%lld, θ=%g) " "dit-B %lldL×%lldh×%lld(inner %lld, cross %lld, out %lld) horizon=%lld action_dim=%lld state_dim=%lld future=%lld N_steps=%lld resident=%s\n", - (long long) m->vit_hidden, (long long) m->vit_layers, (long long) m->deepstack_idx[0], (long long) m->deepstack_idx[1], (long long) m->deepstack_idx[2], (long long) m->spatial_merge, + (long long) m->vit.hidden, (long long) m->vit.layers, (long long) m->vit.deepstack_idx[0], (long long) m->vit.deepstack_idx[1], (long long) m->vit.deepstack_idx[2], (long long) m->vit.merge, (long long) m->lm_hidden, (long long) m->lm_layers, (long long) m->n_q, (long long) m->n_kv, (long long) m->lm_head_dim, (double) m->lm_rope_base, (long long) m->dit_layers, (long long) m->dit_heads, (long long) m->dit_head_dim, (long long) m->dit_hidden, (long long) m->cross_dim, (long long) m->output_dim, (long long) m->action_horizon, (long long) m->action_dim, (long long) m->state_dim, (long long) m->num_future, (long long) m->num_steps, @@ -262,7 +217,7 @@ std::unique_ptr vla_jepa_create(const std::string& mmproj_path, WeightLoader L("vla_jepa", g, m->ctx_weights, m->matmul_type); - m->vit.declare(L, "vit", m->vit_layers); + m->vit.declare(L, "vit"); m->lm.declare(L, "vlm"); m->ae_l1W = L.f32("ah.act_enc.l1.weight"); m->ae_l1b = L.f32("ah.act_enc.l1.bias"); @@ -282,57 +237,19 @@ std::unique_ptr vla_jepa_create(const std::string& mmproj_path, std::printf("vla(vla_jepa): weights resident in %.2f GiB (%s)\n", ggml_backend_buffer_get_size(m->weight_buf)/(1024.0*1024.0*1024.0), dtype_name(m->matmul_type)); - if (!m->build_caches()) { - std::fprintf(stderr, "vla(vla_jepa): build_caches failed\n"); + if (!m->vit.build_caches("vla_jepa", m->io)) return nullptr; - } return m; } -bool VlaJepaModelArch::build_caches() { - if (caches_ready) - return true; - const int64_t side = image_target_size, ps = patch_size, m2 = spatial_merge; - const int64_t grid = side/ps; - const int64_t hd_vit = vit_hidden/vit_heads; - const int64_t num_side = (int64_t) std::lround(std::sqrt((double) vit_num_pos)); - - merge_block_coords(grid, grid, m2, c_grow, c_gcol); - vit_rope_tables(c_grow, c_gcol, hd_vit, (double) vit_rope_base, c_rope_cos, c_rope_sin); - - if (!io.open(gguf_path)) { - std::fprintf(stderr, "vla(vla_jepa): build_caches: io.open(%s) failed\n", gguf_path.c_str()); - return false; - } - std::vector pos_table = io.read_f32("vit.pos_embd"); - if (pos_table.empty() || (int64_t) pos_table.size() != vit_num_pos * vit_hidden) { - std::fprintf(stderr, "vla(vla_jepa): build_caches: vit.pos_embd unreadable\n"); return false; - } - if (!interp_pos_embed(pos_table, num_side, vit_hidden, c_grow, c_gcol, grid, grid, c_pos_interp)) { - std::fprintf(stderr, "vla(vla_jepa): build_caches: vit_num_position_embeddings %lld is not a square\n", (long long) vit_num_pos); return false; - } - - c_tau.assign((size_t) num_steps, {}); c_tproj.assign((size_t) num_steps, {}); - for (int64_t s=0; s VlaJepaModelArch::predict(const Inputs& in) { const auto t0 = std::chrono::steady_clock::now(); stats = Stats{}; const int64_t H = lm_hidden, E = dit_hidden, AD = action_dim, AH = action_horizon, OUTD = output_dim; - const int64_t side = image_target_size, ps = patch_size, m2 = spatial_merge; - const int64_t grid = side/ps, n_patches = grid * grid, K = (grid/m2)*(grid/m2); - const int64_t hd_vit = vit_hidden/vit_heads; + const int64_t K = vit.n_tokens(), n_patches = vit.grid()*vit.grid(); const int64_t Nseq = 1+num_future+AH; const char * dump_prefix = std::getenv("VLA_JEPA_DUMP"); - if (!caches_ready) { std::fprintf(stderr, "vla(vla_jepa): caches not ready\n"); return {}; } auto dump_t = [&](const char * name, ggml_tensor * t) { if (!dump_prefix) @@ -347,15 +264,8 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { } }; - std::vector x_init((size_t) AH * AD); - if (in.noise) - std::memcpy(x_init.data(), in.noise, x_init.size()*sizeof(float)); - else { - std::mt19937 rng((uint32_t) std::chrono::steady_clock::now().time_since_epoch().count()); - std::normal_distribution nd(0.f, 1.f); - for (auto & v : x_init) - v = nd(rng); - } + std::vector x_init; + init_noise(in, (size_t) AH*AD, x_init); std::vector cond_host((size_t) H * num_future, 0.0f); const char * cond_file = std::getenv("VLA_JEPA_COND"); @@ -374,116 +284,51 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { } int64_t n_views = in.n_images; if (n_views <= 0) { std::fprintf(stderr, "vla(vla_jepa): no images in the request\n"); return {}; } - std::vector img_emb_host((size_t) n_views * K * H), ds_host[3]; - for (int j=0; j<3; ++j) - ds_host[j].assign((size_t) n_views * K * H, 0.0f); + std::vector img_emb_host, ds_host[3]; std::vector inj_patches; const char * patches_file = std::getenv("VLA_JEPA_PATCHES"); if (patches_file) { FILE * fp = std::fopen(patches_file, "rb"); if (!fp) { std::fprintf(stderr, "vla(vla_jepa): VLA_JEPA_PATCHES open failed\n"); return {}; } - inj_patches.resize((size_t) n_views * n_patches * vit_patch_flat); + inj_patches.resize((size_t) n_views * n_patches * vit.patch_flat); if (std::fread(inj_patches.data(), sizeof(float), inj_patches.size(), fp) != inj_patches.size()) { std::fprintf(stderr, "vla(vla_jepa): VLA_JEPA_PATCHES short read\n"); std::fclose(fp); return {}; } std::fclose(fp); std::printf("vla(vla_jepa): pixel_values injected from %s\n", patches_file); } if (inj_patches.empty() && !in.images) { std::fprintf(stderr, "vla(vla_jepa): n_images=%d but the images pointer is null\n", in.n_images); return {}; } - ggml_context * VC = vision_scratch.reset((size_t) 512*1024*1024); - if (!VC) { std::fprintf(stderr, "vla(vla_jepa): ggml_init(vision ctx) failed\n"); return {}; } - ggml_tensor * t_patches = ggml_new_tensor_2d(VC, GGML_TYPE_F32, vit_patch_flat, n_patches); ggml_set_input(t_patches); - ggml_tensor * t_pos = ggml_new_tensor_2d(VC, GGML_TYPE_F32, vit_hidden, n_patches); ggml_set_input(t_pos); - ggml_tensor * t_cos = ggml_new_tensor_2d(VC, GGML_TYPE_F32, hd_vit, n_patches); ggml_set_input(t_cos); - ggml_tensor * t_sin = ggml_new_tensor_2d(VC, GGML_TYPE_F32, hd_vit, n_patches); ggml_set_input(t_sin); - ggml_tensor * h = ggml_add(VC, ggml_add(VC, ggml_mul_mat(VC, vit.patch_w, t_patches), vit.patch_b), t_pos); - ggml_set_output(h); - ggml_tensor * stash[3] = {nullptr, nullptr, nullptr}; - for (int64_t i=0; i patches; - bool vok = true; - for (int64_t v=0; v(std::chrono::steady_clock::now()-tv0).count(); if (!vok) return {}; + if (dump_prefix) for (int64_t v=0; v input_ids; - int64_t n_img_slots = 0; - for (int j=0; j inputs_embeds((size_t) SEQ * H); - if (!io.fetch_rows_f32("token_embd.weight", input_ids, inputs_embeds.data(), H)) return {}; - { int64_t k = 0; for (int64_t p=0; p inputs_embeds; + if (!fetch_embeds(io, prompt, img_emb_host.data(), H, inputs_embeds)) return {}; - std::vector image_pos_idx, emb_pos_idx; - for (int64_t p=0; p emb_pos_idx; + for (int64_t p=0; p pp; + if (!mrope_positions("vla_jepa", prompt.ids, (int32_t) image_token_index, vit.grid()/vit.merge, pp)) return {}; + std::vector> ds_pad(3); for (int j=0; j<3; ++j) { ds_pad[j].assign((size_t) SEQ * H, 0.0f); for (int64_t k=0; k VlaJepaModelArch::predict(const Inputs& in) { ggml_tensor * eagle = gio.eagle, * conditioning = gio.conditioning; ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); - - { - const int64_t llm_grid = side/ps/m2; - std::vector pp((size_t) 4*SEQ, 0); - int64_t st = 0, st_idx = 0; - while (st < SEQ) { - int64_t img_start = -1; - for (int64_t i=st; i max_image_pos) max_image_pos = llm_grid-1; - st_idx = image_offset+max_image_pos+1; st = img_end; - } - std::memcpy(pp.data()+(size_t) 3*SEQ, pp.data(), (size_t) SEQ * sizeof(int32_t)); - ggml_backend_tensor_set(t_pos2, pp.data(), 0, ggml_nbytes(t_pos2)); - } + ggml_backend_tensor_set(t_pos2, pp.data(), 0, ggml_nbytes(t_pos2)); if (c_mask_seq != SEQ) { build_causal_mask(SEQ, c_mask); c_mask_seq = SEQ; @@ -607,7 +411,7 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { ggml_set_input(t_tproj[s]); } - ggml_tensor * state_features = ggml_add(C, ggml_mul_mat(C, se_l2W, ggml_relu(C, ggml_add(C, ggml_mul_mat(C, se_l1W, t_state), se_l1b))), se_l2b); + ggml_tensor * state_features = ffn_relu(C, se_l1W, se_l1b, se_l2W, se_l2b, t_state); ggml_tensor * future = future_tokens; const float dt = 1.0f/(float) num_steps; step_seq.assign(num_steps, nullptr); step_pred.assign(num_steps, nullptr); @@ -615,8 +419,7 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { ggml_tensor * actions = t_x0; for (int64_t s=0; s VlaJepaModelArch::predict(const Inputs& in) { x = dit.block(C, dit.blk[i], x, temb, enc); } - ggml_tensor * po = ggml_add(C, ggml_mul_mat(C, dit.po1W, ggml_silu(C, temb)), dit.po1b); - ggml_tensor * sh = ggml_view_1d(C, po, dit_hidden, 0), * sc = ggml_view_1d(C, po, dit_hidden, (size_t) dit_hidden * sizeof(float)); - ggml_tensor * xn = ggml_norm(C, x, dit_norm_out_eps); - ggml_tensor * h_mod = ggml_add(C, ggml_add(C, xn, ggml_mul(C, xn, sc)), sh); - ggml_tensor * model_output = ggml_add(C, ggml_mul_mat(C, dit.po2W, h_mod), dit.po2b); + ggml_tensor * model_output = dit.proj_out(C, x, temb); step_pred[s] = model_output; ggml_tensor * last = ggml_cont(C, ggml_view_2d(C, model_output, OUTD, AH, model_output->nb[1], (size_t) (Nseq-AH)*model_output->nb[1])); - ggml_tensor * vel = ggml_add(C, ggml_mul_mat(C, ad_l2W, ggml_relu(C, ggml_add(C, ggml_mul_mat(C, ad_l1W, last), ad_l1b))), ad_l2b); + ggml_tensor * vel = ffn_relu(C, ad_l1W, ad_l1b, ad_l2W, ad_l2b, last); step_vel[s] = vel; actions = ggml_add(C, actions, ggml_scale(C, vel, dt)); step_act[s] = actions; @@ -679,10 +478,7 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { ggml_backend_tensor_set(t_state, st.data(), 0, ggml_nbytes(t_state)); } ggml_backend_tensor_set(t_x0, x_init.data(), 0, ggml_nbytes(t_x0)); - for (int64_t s=0; s +#include + namespace vla { void ActionExpert::declare(WeightLoader & L, const char * prefix) { @@ -57,4 +60,76 @@ ggml_tensor * ActionExpert::decode(ggml_context * C, ggml_tensor * model_out) co return cat_linear(C, ad_l2W, ad_l2b, embodiment_id, h); } +ggml_tensor * ActionExpert::denoise(ggml_context * C, const DitHead & dit, bool interleave, int64_t every2, + ggml_tensor * state, ggml_tensor * future, ggml_tensor * txt, ggml_tensor * img, + ggml_tensor * x0, const std::vector & tau, + const std::vector & tproj) const { + const int64_t n_layers = dit.cfg.layers, AD = x0->ne[0], AH = x0->ne[1]; + ggml_tensor * state_features = encode_state(C, state); + + std::vector enc(n_layers, nullptr), Kc(n_layers, nullptr), Vc(n_layers, nullptr); + for (int64_t i=0; ine[0], AH); + ggml_tensor * sa = future ? ggml_concat(C, state_features, future, 1) : state_features; + ggml_tensor * hh = ggml_concat(C, sa, af, 1); + + for (int64_t i=0; inb[1], (size_t)(pred->ne[1]-AH)*pred->nb[1])); + actions = ggml_add(C, actions, ggml_scale(C, vel, dt)); + } + return actions; +} + +bool resolve_embodiment(const char * arch, const std::string & mapping, const char * default_tag, + int64_t max_id, int64_t & id) { + auto lookup = [&](const char * key) -> long { + const std::string k = std::string("\"")+key+"\""; + size_t p = mapping.find(k); + if (p == std::string::npos) + return -1; + p = mapping.find(':', p+k.size()); + if (p == std::string::npos) + return -1; + return std::strtol(mapping.c_str()+p+1, nullptr, 10); + }; + + if (default_tag) { + const long d = lookup(default_tag); + if (d >= 0) + id = d; + } + if (const char * e = std::getenv("VLA_GR00T_EMBODIMENT")) { + char * end = nullptr; + const long v = std::strtol(e, &end, 10); + if (end && *end == '\0') { + id = v; + } else { + const long t = lookup(e); + if (t >= 0) + id = t; + else + std::fprintf(stderr, "vla(%s): embodiment tag '%s' not in the GGUF embodiment mapping; using id %lld\n", + arch, e, (long long) id); + } + } + if (id < 0 || id >= max_id) { + std::fprintf(stderr, "vla(%s): embodiment id %lld out of range [0,%lld)\n", arch, (long long) id, (long long) max_id); + return false; + } + return true; +} + } diff --git a/src/modules/action_expert.h b/src/modules/action_expert.h index 0804a74..f9f4ef6 100644 --- a/src/modules/action_expert.h +++ b/src/modules/action_expert.h @@ -17,10 +17,13 @@ #pragma once #include "loader.h" +#include "modules/dit_head.h" #include "ggml.h" #include +#include +#include namespace vla { @@ -41,6 +44,14 @@ struct ActionExpert { int64_t embed_dim, int64_t horizon) const; ggml_tensor * decode(ggml_context * C, ggml_tensor * model_out) const; + + ggml_tensor * denoise(ggml_context * C, const DitHead & dit, bool interleave, int64_t every2, + ggml_tensor * state, ggml_tensor * future, ggml_tensor * txt, ggml_tensor * img, + ggml_tensor * x0, const std::vector & tau, + const std::vector & tproj) const; }; +bool resolve_embodiment(const char * arch, const std::string & mapping, const char * default_tag, + int64_t max_id, int64_t & id); + } diff --git a/src/modules/dit_head.cpp b/src/modules/dit_head.cpp index f081288..a9b60ad 100644 --- a/src/modules/dit_head.cpp +++ b/src/modules/dit_head.cpp @@ -15,12 +15,16 @@ #include "modules/dit_head.h" #include "layers/attn.h" +#include "layers/embed.h" #include "layers/ffn.h" #include "layers/linear.h" #include "layers/norm.h" +#include "ggml-backend.h" + #include #include +#include namespace vla { @@ -146,4 +150,33 @@ ggml_tensor * DitHead::proj_out(ggml_context * C, ggml_tensor * h, ggml_tensor * return linear(C, po2W, po2b, h_mod); } +void FlowTimes::build(int64_t steps, int64_t buckets, int64_t embed_dim, int64_t horizon) { + tau.assign((size_t) steps, {}); + tproj.assign((size_t) steps, {}); + for (int64_t s=0; s & t_tau, const std::vector & t_tproj) const { + for (size_t s=0; s= 1) { + steps = (int64_t) v; + std::fprintf(stderr, "vla(%s): VLA_NUM_STEPS override → num_steps=%lld\n", arch, (long long) v); + } +} + } diff --git a/src/modules/dit_head.h b/src/modules/dit_head.h index 6c00b2c..a055a73 100644 --- a/src/modules/dit_head.h +++ b/src/modules/dit_head.h @@ -67,4 +67,14 @@ struct DitHead { ggml_tensor * proj_out(ggml_context * C, ggml_tensor * h, ggml_tensor * temb) const; }; +struct FlowTimes { + std::vector> tau, tproj; + + void build(int64_t steps, int64_t buckets, int64_t embed_dim, int64_t horizon); + + void upload(const std::vector & t_tau, const std::vector & t_tproj) const; +}; + +void env_num_steps(const char * arch, int64_t & steps); + } diff --git a/src/modules/prompt.cpp b/src/modules/prompt.cpp index 99ab626..3fe87f4 100644 --- a/src/modules/prompt.cpp +++ b/src/modules/prompt.cpp @@ -61,8 +61,7 @@ bool build_prompt(const char * arch, const Inputs & in, int64_t n_img, return true; } -bool fetch_embeds(const char * arch, gguf_reader & io, const Prompt & p, - const float * img_emb, int64_t hidden, std::vector & out) { +bool fetch_embeds(gguf_reader & io, const Prompt & p, const float * img_emb, int64_t hidden, std::vector & out) { const int64_t seq = p.len(); out.assign((size_t) seq*hidden, 0.0f); if (!io.fetch_rows_f32("token_embd.weight", p.ids, out.data(), hidden)) @@ -70,8 +69,6 @@ bool fetch_embeds(const char * arch, gguf_reader & io, const Prompt & p, for (size_t k=0; k & out); +bool fetch_embeds(gguf_reader & io, const Prompt & p, const float * img_emb, int64_t hidden, std::vector & out); void init_noise(const Inputs & in, size_t n, std::vector & out); diff --git a/src/modules/qwen3vl_vit.h b/src/modules/qwen3vl_vit.h index ff35eff..691e917 100644 --- a/src/modules/qwen3vl_vit.h +++ b/src/modules/qwen3vl_vit.h @@ -17,10 +17,12 @@ #pragma once #include "backend.h" +#include "gguf_reader.h" #include "layers/attn.h" #include "loader.h" #include "layers/rope.h" #include "model.h" +#include "scratch_ctx.h" #include "ggml.h" #include "options.h" @@ -30,6 +32,7 @@ #include #include #include +#include #include namespace vla { @@ -41,6 +44,11 @@ struct VitLayerW { ggml_tensor *ln1w,*ln1b,*ln2w,*ln2b,*Wqkv,*bqkv,*Wo,*bo,*Wfc1 struct MergerW { ggml_tensor *nw,*nb,*fc1w,*fc1b,*fc2w,*fc2b; }; struct Qwen3VLTower { + int64_t hidden = 1024, layers = 24, heads = 16, patch = 16, temporal = 2, merge = 2; + int64_t num_pos = 2304, patch_flat = 1536, side = 256; + int64_t deepstack_idx[3] = {5, 11, 17}; + float ln_eps = 1e-6f, rope_theta = 10000.0f, conn_eps = 1e-6f; + std::vector blk; MergerW deepstack[3]; MergerW merger; @@ -48,7 +56,24 @@ struct Qwen3VLTower { ggml_tensor * patch_b = nullptr; ggml_tensor * pos = nullptr; - void declare(WeightLoader & L, const char * prefix, int64_t layers) { + std::vector row, col; + std::vector rope_cos, rope_sin, pos_interp; + + int64_t grid() const { + return side/patch; + } + int64_t n_tokens() const { + return (grid()/merge)*(grid()/merge); + } + + bool load_config(const char * arch, const gguf_reader & g, const char * ns); + + bool build_caches(const char * arch, gguf_reader & io); + + bool encode(const char * arch, ggml_backend_t backend, scratch_ctx & scratch, const ImageView * images, + int64_t n_views, const float * patches_in, std::vector & emb, std::vector (&ds)[3]) const; + + void declare(WeightLoader & L, const char * prefix) { patch_w = L.gemm("%s.patch_embd.weight", prefix); patch_b = L.f32 ("%s.patch_embd.bias", prefix); pos = L.f32 ("%s.pos_embd", prefix); @@ -212,4 +237,159 @@ inline bool preprocess_image_patches(const char * arch, const ImageView & v, int return true; } +inline bool mrope_positions(const char * arch, const std::vector & ids, int32_t image_token, int64_t grid, + std::vector & pp) { + const int64_t seq = (int64_t) ids.size(), g2 = grid*grid; + pp.assign((size_t) 4*seq, 0); + int64_t st = 0, st_idx = 0; + while (st < seq) { + int64_t img = st; + while (img < seq && ids[img] != image_token) + ++img; + for (int64_t i=st; i