diff --git a/CHANGELOG.md b/CHANGELOG.md index 07b6029..a04005d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,14 @@ Notable changes to vla.cpp. Format loosely follows [Keep a Changelog](https://ke ### Added +- **New model: ACT** (LeRobot `policies/act`). `scripts/convert_act_to_gguf.py` + converts a LeRobot `pretrained_model` directory, folding the ResNet's frozen + batch norms and dropping the training-only VAE encoder. No language input, so + `vla-server` and `vla-cli` accept requests without tokens for it. The ResNet + runs at the cameras' own size, all views in one batch; on CUDA (Turing and + newer) it is ggml's direct convolution on F16 kernels, unless + `--weight-dtype f32` is given. `vla-bench --height` and `VLA_IMG_H` in + `vla_predict_check` give non-square inputs. - **FoldQuant W8A8 / W4A4 inference** for GR00T N1.5 / N1.6 / N1.7 and π0.5. A FoldQuant GGUF carries the language backbone and the action module as INT8 or INT4 codes with per-row scales in a block-Hadamard, diff --git a/CMakeLists.txt b/CMakeLists.txt index 2a01c88..c726cf4 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -249,6 +249,7 @@ add_library(vla_core ${VLA_CORE_LIB_TYPE} src/models/openvla_oft.cpp src/models/vla_jepa.cpp src/models/turbovla.cpp + src/models/act.cpp src/models/octo.cpp src/tokenizer.cpp ) diff --git a/README.md b/README.md index 2f73c8e..ba9ab35 100644 --- a/README.md +++ b/README.md @@ -163,6 +163,7 @@ supported (released and benchmarked), `~` = in progress, `-` = planned. | [VLA-JEPA](https://hf.co/vrfai/vla-jepa-libero) | Y | Y | - | Y | Y | ~ | | [Octo-Small](https://hf.co/vrfai/octo-small-libero-gguf) | Y | Y | Y | Y | - | Y | | [TurboVLA](https://hf.co/vrfai/turbovla-libero-gguf) | Y | Y | Y | Y | Y | Y | +| [ACT](https://hf.co/ravediamond/act-aloha-sim-transfer-cube-gguf) | Y | Y | - | Y | - | - | --- diff --git a/scripts/convert_act_to_gguf.py b/scripts/convert_act_to_gguf.py new file mode 100644 index 0000000..6ea22a1 --- /dev/null +++ b/scripts/convert_act_to_gguf.py @@ -0,0 +1,265 @@ +#!/usr/bin/env python3 +# Copyright 2026 VinRobotics +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Convert a LeRobot ACT checkpoint (policies/act) to GGUF. + +Reads a pretrained_model directory: config.json, model.safetensors and, for +checkpoints saved by LeRobot >= 0.4, the normalizer processor files. Older +checkpoints (lerobot/act_aloha_sim_transfer_cube_human) keep their stats in +model.safetensors, which is read as a fallback. + +At inference the VAE encoder is unused (the latent is zeros), so it is not +written. The ResNet's frozen batch norms are folded into the convolutions. + +Usage: + python convert_act_to_gguf.py --ckpt [--out act.gguf] +""" + +from __future__ import annotations + +import json +from pathlib import Path + +import numpy as np +import torch +from safetensors import safe_open + +from gguf_blocks import load_processor_stats +from gguf_common import ( + add_array, + add_f32, + arg_parser, + finish, + kv_prefix, + kv_u32, + open_writer, + read_json, + resolve_out, +) + +ARCH = "act" +KV = kv_prefix(ARCH) + +# torchvision BasicBlock counts per stage; ACT's backbone option is a torchvision name. +RESNET_BLOCKS = {"resnet18": (2, 2, 2, 2), "resnet34": (3, 4, 6, 3)} +BN_EPS = 1e-5 # torchvision FrozenBatchNorm2d default + +# Read only at training time. +UNUSED = ("model.vae_encoder", "normalize_targets.", "model.backbone.fc.") + + +class TrackedTensors(dict): + """A state dict that records which keys the writers read.""" + + def __init__(self, *a, **k): + super().__init__(*a, **k) + self.read = set() + + def __getitem__(self, key): + self.read.add(key) + return super().__getitem__(key) + + +def fold_bn(conv_w: torch.Tensor, bn: str, t: dict) -> tuple[torch.Tensor, torch.Tensor]: + """Conv (no bias) followed by FrozenBatchNorm2d, as one conv with a bias.""" + scale = t[f"{bn}.weight"].double() * torch.rsqrt(t[f"{bn}.running_var"].double() + BN_EPS) + w = conv_w.double() * scale.view(-1, 1, 1, 1) + b = t[f"{bn}.bias"].double() - t[f"{bn}.running_mean"].double() * scale + return w.float(), b.float() + + +def write_conv_bn(writer, t: dict, conv: str, bn: str, dst: str) -> None: + w, b = fold_bn(t[f"{conv}.weight"], bn, t) + add_f32(writer, f"{dst}.weight", w) + add_f32(writer, f"{dst}.bias", b) + + +def write_backbone(writer, t: dict, blocks: tuple[int, ...]) -> None: + pfx = "model.backbone" + write_conv_bn(writer, t, f"{pfx}.conv1", f"{pfx}.bn1", "bb.conv1") + for li, n in enumerate(blocks, start=1): + for bi in range(n): + src, dst = f"{pfx}.layer{li}.{bi}", f"bb.layer{li}.{bi}" + write_conv_bn(writer, t, f"{src}.conv1", f"{src}.bn1", f"{dst}.conv1") + write_conv_bn(writer, t, f"{src}.conv2", f"{src}.bn2", f"{dst}.conv2") + if f"{src}.downsample.0.weight" in t: + write_conv_bn(writer, t, f"{src}.downsample.0", f"{src}.downsample.1", f"{dst}.down") + + +def write_attn(writer, t: dict, src: str, dst: str, dim: int) -> None: + """nn.MultiheadAttention: in_proj is [W_q; W_k; W_v]. Split, since ACT adds + the position embedding to the query and key inputs but not to the value.""" + w, b = t[f"{src}.in_proj_weight"], t[f"{src}.in_proj_bias"] + for i, part in enumerate("qkv"): + add_f32(writer, f"{dst}_{part}.weight", w[i*dim:(i + 1)*dim]) + add_f32(writer, f"{dst}_{part}.bias", b[i*dim:(i + 1)*dim]) + add_f32(writer, f"{dst}_o.weight", t[f"{src}.out_proj.weight"]) + add_f32(writer, f"{dst}_o.bias", t[f"{src}.out_proj.bias"]) + + +def write_ffn_norms(writer, t: dict, src: str, dst: str, n_norms: int) -> None: + for name, part in (("fc1", "linear1"), ("fc2", "linear2")): + add_f32(writer, f"{dst}.{name}.weight", t[f"{src}.{part}.weight"]) + add_f32(writer, f"{dst}.{name}.bias", t[f"{src}.{part}.bias"]) + for i in range(1, n_norms + 1): + add_f32(writer, f"{dst}.ln{i}.weight", t[f"{src}.norm{i}.weight"]) + add_f32(writer, f"{dst}.ln{i}.bias", t[f"{src}.norm{i}.bias"]) + + +def write_transformer(writer, t: dict, n_enc: int, n_dec: int, dim: int, pre_norm: bool) -> None: + for i in range(n_enc): + src, dst = f"model.encoder.layers.{i}", f"enc.blk.{i}" + write_attn(writer, t, f"{src}.self_attn", f"{dst}.attn", dim) + write_ffn_norms(writer, t, src, dst, 2) + if pre_norm: + add_f32(writer, "enc.norm.weight", t["model.encoder.norm.weight"]) + add_f32(writer, "enc.norm.bias", t["model.encoder.norm.bias"]) + for i in range(n_dec): + src, dst = f"model.decoder.layers.{i}", f"dec.blk.{i}" + write_attn(writer, t, f"{src}.self_attn", f"{dst}.attn", dim) + write_attn(writer, t, f"{src}.multihead_attn", f"{dst}.cross", dim) + write_ffn_norms(writer, t, src, dst, 3) + add_f32(writer, "dec.norm.weight", t["model.decoder.norm.weight"]) + add_f32(writer, "dec.norm.bias", t["model.decoder.norm.bias"]) + + +def write_io(writer, t: dict, has_state: bool) -> None: + # The latent is zeros at inference, so its projection is just the bias. + add_f32(writer, "latent_tok", t["model.encoder_latent_input_proj.bias"]) + t["model.encoder_latent_input_proj.weight"] # noqa: B018 (read for --verify) + if has_state: + add_f32(writer, "state_proj.weight", t["model.encoder_robot_state_input_proj.weight"]) + add_f32(writer, "state_proj.bias", t["model.encoder_robot_state_input_proj.bias"]) + add_f32(writer, "pos_1d", t["model.encoder_1d_feature_pos_embed.weight"]) + w = t["model.encoder_img_feat_input_proj.weight"] + add_f32(writer, "img_proj.weight", w.reshape(w.shape[0], w.shape[1])) + add_f32(writer, "img_proj.bias", t["model.encoder_img_feat_input_proj.bias"]) + add_f32(writer, "dec.pos", t["model.decoder_pos_embed.weight"]) + add_f32(writer, "action_head.weight", t["model.action_head.weight"]) + add_f32(writer, "action_head.bias", t["model.action_head.bias"]) + + +def legacy_key(feature: str) -> str: + return feature.replace(".", "_") + + +def read_stats(ckpt: Path, t: dict, feature: str, dim: int, legacy: tuple[str, ...]) -> tuple[np.ndarray, np.ndarray]: + """mean/std of one feature: the processor files first, then model.safetensors buffers.""" + meta, registry = ("policy_postprocessor.json", "unnormalizer_processor") if feature == "action" \ + else ("policy_preprocessor.json", "normalizer_processor") + got = load_processor_stats(ckpt, meta, registry, feature, dim) + for pfx in legacy: + if got is None and f"{pfx}.mean" in t: + got = (t[f"{pfx}.mean"].float().numpy().reshape(-1), t[f"{pfx}.std"].float().numpy().reshape(-1)) + print(f" stats: loaded {feature} from model.safetensors ({pfx})") + if got is None or got[0].size != dim: + raise SystemExit(f"no mean/std of {feature} (dim {dim}) in the processor files or model.safetensors") + return got + + +def main() -> int: + ap = arg_parser(ARCH, "LeRobot ACT pretrained_model directory (config.json + model.safetensors)", + description=__doc__) + ap.add_argument("--verify", action="store_true", help="fail if a checkpoint tensor is left unconverted") + args = ap.parse_args() + ckpt = args.ckpt.resolve() + out = resolve_out(args, ckpt, ARCH) + + cfg = read_json(ckpt / "config.json") + if cfg.get("type") != "act": + raise SystemExit(f"{ckpt}/config.json is not an ACT config (type={cfg.get('type')!r})") + feats = cfg.get("input_features") or {} + images = [k for k, v in feats.items() if v.get("type") == "VISUAL"] + has_state = "observation.state" in feats + if any(v.get("type") == "ENV" for v in feats.values()): + raise SystemExit("observation.environment_state is not supported") + if not images: + raise SystemExit("ACT without cameras is not supported") + shapes = {tuple(feats[k]["shape"]) for k in images} + if len(shapes) != 1: + raise SystemExit(f"cameras of different sizes are not supported: {sorted(shapes)}") + _, img_h, img_w = shapes.pop() + backbone = cfg.get("vision_backbone", "resnet18") + if backbone not in RESNET_BLOCKS: + raise SystemExit(f"vision_backbone {backbone!r} is not supported ({', '.join(RESNET_BLOCKS)})") + if cfg.get("replace_final_stride_with_dilation"): + raise SystemExit("replace_final_stride_with_dilation is not supported") + act_fn = cfg.get("feedforward_activation", "relu") + if act_fn not in ("relu", "gelu"): + raise SystemExit(f"feedforward_activation {act_fn!r} is not supported (relu, gelu)") + if cfg.get("temporal_ensemble_coeff") is not None: + print(" note: temporal_ensemble_coeff is a client-side policy; the GGUF predicts the full chunk") + norm_map = cfg.get("normalization_mapping") or {} + for ftype in ("VISUAL", "STATE", "ACTION"): + if norm_map.get(ftype, "MEAN_STD") != "MEAN_STD": + raise SystemExit(f"normalization_mapping {ftype}={norm_map[ftype]} is not supported (MEAN_STD only)") + + with safe_open(str(ckpt / "model.safetensors"), framework="pt") as f: + t = TrackedTensors({k: f.get_tensor(k) for k in f.keys()}) + + dim = int(cfg["dim_model"]) + pre_norm = bool(cfg.get("pre_norm", False)) + n_enc, n_dec = int(cfg["n_encoder_layers"]), int(cfg["n_decoder_layers"]) + state_dim = int(feats["observation.state"]["shape"][0]) if has_state else 0 + action_dim = int(cfg["output_features"]["action"]["shape"][0]) + chunk = int(cfg["chunk_size"]) + if t["model.decoder_pos_embed.weight"].shape != (chunk, dim) or t["model.action_head.weight"].shape != (action_dim, dim): + raise SystemExit("config.json disagrees with the checkpoint's tensor shapes") + + writer = open_writer(out, ARCH) + kv_u32(writer, KV, { + "dim_model": dim, "n_heads": cfg["n_heads"], "dim_feedforward": cfg["dim_feedforward"], + "n_encoder_layers": n_enc, "n_decoder_layers": n_dec, "pre_norm": int(pre_norm), + "gelu": int(act_fn == "gelu"), "chunk_size": chunk, "n_action_steps": cfg.get("n_action_steps", chunk), + "state_dim": state_dim, "action_dim": action_dim, "num_views": len(images), + "image_height": img_h, "image_width": img_w, + }) + writer.add_array(KV("backbone_blocks"), list(RESNET_BLOCKS[backbone])) + writer.add_array(KV("cameras"), images) + + print(f" {backbone}, {len(images)} camera(s) at {img_h}x{img_w}, state {state_dim}, action {action_dim}, " + f"chunk {chunk}, {n_enc}+{n_dec} layers, dim {dim}") + write_backbone(writer, t, RESNET_BLOCKS[backbone]) + write_transformer(writer, t, n_enc, n_dec, dim, pre_norm) + write_io(writer, t, has_state) + + img_mean = np.zeros((len(images), 3), dtype=np.float32) + img_std = np.ones((len(images), 3), dtype=np.float32) + for i, k in enumerate(images): + img_mean[i], img_std[i] = read_stats(ckpt, t, k, 3, (f"normalize_inputs.buffer_{legacy_key(k)}",)) + add_array(writer, "image_mean", img_mean) + add_array(writer, "image_std", img_std) + if has_state: + for name, vec in zip(("state_mean", "state_std"), + read_stats(ckpt, t, "observation.state", state_dim, + ("normalize_inputs.buffer_observation_state",))): + add_array(writer, name, vec) + for name, vec in zip(("action_mean", "action_std"), + read_stats(ckpt, t, "action", action_dim, ("unnormalize_outputs.buffer_action",))): + add_array(writer, name, vec) + writer.add_string(KV("config_json"), json.dumps(cfg)) + + if args.verify: + left = sorted(k for k in t if k not in t.read and not k.startswith(UNUSED) + and not k.startswith("normalize_inputs.") and not k.startswith("unnormalize_outputs.")) + if left: + raise SystemExit(f"{len(left)} checkpoint tensors not converted: {left[:20]}") + print(f" all {len(t.read)} used tensors consumed") + + return finish(writer, out) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/arch.h b/src/arch.h index 146b6e4..5387e5d 100644 --- a/src/arch.h +++ b/src/arch.h @@ -72,6 +72,7 @@ enum class Arch { OPENVLA_OFT,// DINOv2-L/14-reg4 + SigLIP-so400m/14 +Llama-2-7B + MLPResNet. VLA_JEPA, // LeRobot Qwen3-VL-2B-Instruct+V-JEPÀ+DiT-B FM. TURBOVLA, // TurboVLA (DINOv3 + BERT + VL Fusion + ACT decoder). + ACT, // LeRobot ACT (ResNet + transformer encoder-decoder, no language). }; /** @@ -225,6 +226,15 @@ std::unique_ptr turbovla_create(const std::string& mmproj_path, const std::string& config_path, const Options& opts); +/** + * @brief Build an ACT model. The ResNet backbone is baked into @p ckpt_path. + * @copydetails smolvla_create + */ +std::unique_ptr act_create(const std::string& mmproj_path, + const std::string& ckpt_path, + const std::string& config_path, + const Options& opts); + /** * @brief Inspect a GGUF and identify the architecture tag. * diff --git a/src/layers/conv.h b/src/layers/conv.h new file mode 100644 index 0000000..dc37381 --- /dev/null +++ b/src/layers/conv.h @@ -0,0 +1,37 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + + +#pragma once + +#include "ggml.h" + +namespace vla { + +// x [W, H, C, N] with w [KW, KH, C, OC] to [OW, OH, OC, N], no bias, as im2col +// and one F32 GEMM. The patches are the GEMM's first operand, so the result +// lands as [OW*OH*N, OC]: one image is already in the input layout, a batch is +// one permute away from it. +inline ggml_tensor * conv_2d_f32(ggml_context * C, ggml_tensor * w, ggml_tensor * x, int stride, int pad) { + ggml_tensor * col = ggml_im2col(C, w, x, stride, stride, pad, pad, 1, 1, true, GGML_TYPE_F32); + ggml_tensor * y = ggml_mul_mat(C, + ggml_reshape_2d(C, col, col->ne[0], col->ne[3]*col->ne[2]*col->ne[1]), + ggml_reshape_2d(C, w, w->ne[0]*w->ne[1]*w->ne[2], w->ne[3])); + if (col->ne[3] == 1) + return ggml_reshape_4d(C, y, col->ne[1], col->ne[2], w->ne[3], 1); + y = ggml_reshape_4d(C, y, col->ne[1], col->ne[2], col->ne[3], w->ne[3]); + return ggml_cont(C, ggml_permute(C, y, 0, 1, 3, 2)); +} + +} diff --git a/src/model.cpp b/src/model.cpp index ecbecdf..0e5b34a 100644 --- a/src/model.cpp +++ b/src/model.cpp @@ -80,7 +80,8 @@ bool detect_arch_gguf(const std::string& path, Arch* out) { try_str("openvla_oft.architecture", arch_str) || try_str("vla_jepa.architecture", arch_str) || try_str("vla_adapter.architecture", arch_str) || - try_str("turbovla.architecture", arch_str)) { + try_str("turbovla.architecture", arch_str) || + try_str("act.architecture", arch_str)) { if (arch_str == "smolvla") { *out = Arch::SMOLVLA; ok = true; @@ -133,6 +134,10 @@ bool detect_arch_gguf(const std::string& path, Arch* out) { *out = Arch::TURBOVLA; ok = true; } + else if (arch_str == "act") { + *out = Arch::ACT; + ok = true; + } } gguf_free(gctx); @@ -243,7 +248,8 @@ Model* model_load(const std::string& mmproj_path, const std::string& ckpt_path, : arch == Arch::BITVLA ? "bitvla" : arch == Arch::VLA_ADAPTER ? "vla_adapter" : arch == Arch::OPENVLA_OFT ? "openvla_oft" - : arch == Arch::TURBOVLA ? "turbovla" : nullptr; + : arch == Arch::TURBOVLA ? "turbovla" + : arch == Arch::ACT ? "act" : nullptr; if (name) { std::fprintf(stderr, "vla(%s): num_steps is not supported\n", name); return nullptr; @@ -312,6 +318,10 @@ Model* model_load(const std::string& mmproj_path, const std::string& ckpt_path, std::printf("vla: arch = turbovla\n"); impl = turbovla_create(mmproj_path, ckpt_path, config_path, opts); break; + case Arch::ACT: + std::printf("vla: arch = act\n"); + impl = act_create(mmproj_path, ckpt_path, config_path, opts); + break; } if (!impl) return nullptr; diff --git a/src/models/act.cpp b/src/models/act.cpp new file mode 100644 index 0000000..da633dd --- /dev/null +++ b/src/models/act.cpp @@ -0,0 +1,635 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// ACT (LeRobot policies/act, from Zhao et al. 2023): a ResNet over each camera, +// a transformer encoder over [latent, state, image features], and a decoder +// whose chunk_size zero queries cross-attend the encoder output. No language +// and no denoising loop. At inference the VAE latent is zeros, so the latent +// token is its projection's bias and the VAE encoder is not converted. +// +// The ResNet runs at whatever size the cameras send, like the PyTorch policy, +// over all views as one batch; the graph is cached per input size. + +#include "arch.h" +#include "backend.h" +#include "gguf.h" +#include "gguf_reader.h" +#include "loader.h" +#include "model.h" +#include "options.h" +#include "scratch_ctx.h" +#include "layers/attn.h" +#include "layers/conv.h" +#include "layers/ffn.h" +#include "layers/linear.h" +#include "layers/norm.h" +#include "modules/preprocess.h" + +#include "ggml.h" +#include "ggml-backend.h" + +#ifdef GGML_USE_CUDA +#include +#endif + +#include +#include +#include +#include +#include +#include +#include + +namespace vla { +namespace { + +constexpr float kLnEps = 1e-5f; // nn.LayerNorm default +constexpr float kNormEps = 1e-8f; // LeRobot's NormalizerProcessorStep: (x - mean) / (std + eps) + +struct ConvW { + ggml_tensor *w = nullptr, *b = nullptr; +}; + +struct BlockW { + ConvW conv1, conv2, down; + int stride = 1; +}; + +// q and k share their input (tokens plus position), so they are one GEMM. +struct SelfAttnW { + ggml_tensor *qk_w, *qk_b, *v_w, *v_b, *o_w, *o_b; +}; + +struct CrossAttnW { + ggml_tensor *q_w, *q_b, *k_w, *k_b, *v_w, *v_b, *o_w, *o_b; +}; + +struct EncLayerW { + SelfAttnW attn; + ggml_tensor *fc1_w, *fc1_b, *fc2_w, *fc2_b, *ln1_w, *ln1_b, *ln2_w, *ln2_b; +}; + +struct DecLayerW { + SelfAttnW attn; + CrossAttnW cross; + ggml_tensor *fc1_w, *fc1_b, *fc2_w, *fc2_b, *ln1_w, *ln1_b, *ln2_w, *ln2_b, *ln3_w, *ln3_b; +}; + +// torchvision ResNet output size of a 3x3 (or 7x7) conv / pool at stride 2. +int64_t down2(int64_t n, int64_t k, int64_t pad) { + return (n + 2*pad - k) / 2 + 1; +} + +// ggml-cuda's direct convolution is an implicit GEMM only where it has MMA +// (Turing and newer). Before that, Volta's Jetson Xavier included, it falls back +// to a naive kernel several times slower than im2col. +bool cuda_conv_has_mma() { +#ifdef GGML_USE_CUDA + int major = 0, minor = 0; + const int dev = backend_device_index(); + return cudaDeviceGetAttribute(&major, cudaDevAttrComputeCapabilityMajor, dev) == cudaSuccess && + cudaDeviceGetAttribute(&minor, cudaDevAttrComputeCapabilityMinor, dev) == cudaSuccess && + major*10 + minor >= 75; +#else + return false; +#endif +} + +} // namespace + +struct ActModelArch : public ModelArchBase { + ActModelArch() : ModelArchBase(Arch::ACT) {} + ~ActModelArch() override { + graph.release(); + if (weight_buf) ggml_backend_buffer_free(weight_buf); + if (ctx_weights) ggml_free(ctx_weights); + if (backend) ggml_backend_free(backend); + } + + ggml_backend_t backend = nullptr; + int n_threads = default_cpu_threads(); + ggml_context * ctx_weights = nullptr; + ggml_backend_buffer_t weight_buf = nullptr; + ggml_type mt = GGML_TYPE_F32; + // CUDA (Turing+) runs the ResNet as ggml's direct convolution on F16 + // kernels: an implicit GEMM on tensor cores with F32 accumulation and no + // im2col buffer. Elsewhere the kernels stay F32 and go through im2col, which + // is the faster of the two on CPU. + bool conv_direct = false; + + int64_t dim = 512, heads = 8, ff = 3200, enc_layers = 4, dec_layers = 1, chunk = 100; + int64_t state_dim = 0, action_dim = 0, n_views = 1, img_h = 480, img_w = 640; + bool pre_norm = false, gelu = false; + std::vector blocks; + std::string cameras; // the training feature names, in the order views must arrive + + ConvW stem; + std::vector res; + ggml_tensor *img_proj_w = nullptr, *img_proj_b = nullptr; + ggml_tensor *latent_tok = nullptr, *state_w = nullptr, *state_b = nullptr; + std::vector enc; + ggml_tensor *enc_norm_w = nullptr, *enc_norm_b = nullptr; + std::vector dec; + ggml_tensor *dec_norm_w = nullptr, *dec_norm_b = nullptr, *dec_pos = nullptr; + ggml_tensor *head_w = nullptr, *head_b = nullptr; + + std::vector pos_1d; // [n_1d, dim], host copy for the encoder position table + // img_std and the other stds already carry LeRobot's eps. + std::vector img_mean, img_std, state_mean, state_std, action_mean, action_std; + + struct Key { + int64_t h = -1, w = -1; + bool operator==(const Key & o) const { return h == o.h && w == o.w; } + }; + struct IO { + ggml_tensor *pixels = nullptr, *state = nullptr, *enc_pos = nullptr, *actions = nullptr; + }; + graph_cache graph; + std::vector pixels; // host staging for io.pixels, kept so a call does not fault in fresh pages + + int64_t n_1d() const { return state_dim > 0 ? 2 : 1; } + static void feature_size(int64_t h, int64_t w, int64_t & fh, int64_t & fw) { + fh = down2(down2(h, 7, 3), 3, 1); // conv1, maxpool + fw = down2(down2(w, 7, 3), 3, 1); + for (int i = 0; i < 3; ++i) { // layer2..4 open with a stride-2 block + fh = down2(fh, 3, 1); + fw = down2(fw, 3, 1); + } + } + + ggml_cgraph * build(ggml_context * C, IO & io, int64_t h, int64_t w) const; + std::vector enc_pos_table(int64_t fh, int64_t fw) const; + + std::vector predict(const Inputs& in) override; +}; + +namespace { + +// [w, h, c, view] to [ow, oh, oc, view]. The kernel's type picks the path; see +// ActModelArch::conv_direct. +ggml_tensor * conv2d(ggml_context * C, const ConvW & cw, ggml_tensor * x, int stride, int pad) { + ggml_tensor * y = cw.w->type == GGML_TYPE_F16 + ? ggml_conv_2d_direct(C, cw.w, x, stride, stride, pad, pad, 1, 1) + : conv_2d_f32(C, cw.w, x, stride, pad); + return ggml_add(C, y, ggml_reshape_4d(C, cw.b, 1, 1, cw.b->ne[0], 1)); +} + +// [D, T] attention with separate query and key/value sources. q_in and k_in +// already carry their position embeddings; v_in does not. +ggml_tensor * mha(ggml_context * C, ggml_tensor * q, ggml_tensor * k, ggml_tensor * v, + int64_t hd, int64_t heads, int64_t Tq, int64_t Tk) { + const float scale = 1.0f / std::sqrt((float) hd); + if (flash_attn_enabled()) { + auto h = [&](ggml_tensor * p, int64_t T) { + return ggml_permute(C, ggml_reshape_3d(C, ggml_cont(C, p), hd, heads, T), 0, 2, 1, 3); + }; + return flash_attention(C, h(q, Tq), h(k, Tk), h(v, Tk), nullptr, scale); + } + return ggml_reshape_2d(C, attention(C, to_heads(C, ggml_cont(C, q), hd, heads, Tq), + to_heads(C, ggml_cont(C, k), hd, heads, Tk), + to_heads_v(C, ggml_cont(C, v), hd, heads, Tk), + nullptr, scale, hd*heads, Tq), hd*heads, Tq); +} + +ggml_tensor * self_attn(ggml_context * C, const SelfAttnW & a, ggml_tensor * x, ggml_tensor * pos, + int64_t dim, int64_t heads, int64_t T) { + ggml_tensor * qk = linear(C, a.qk_w, a.qk_b, ggml_add(C, x, pos)); + ggml_tensor * q = ggml_view_2d(C, qk, dim, T, qk->nb[1], 0); + ggml_tensor * k = ggml_view_2d(C, qk, dim, T, qk->nb[1], (size_t) dim*ggml_element_size(qk)); + ggml_tensor * v = linear(C, a.v_w, a.v_b, x); + return linear(C, a.o_w, a.o_b, mha(C, q, k, v, dim / heads, heads, T, T)); +} + +ggml_tensor * cross_attn(ggml_context * C, const CrossAttnW & a, ggml_tensor * x, ggml_tensor * x_pos, + ggml_tensor * mem, ggml_tensor * mem_pos, int64_t dim, int64_t heads, + int64_t Tq, int64_t Tk) { + ggml_tensor * q = linear(C, a.q_w, a.q_b, ggml_add(C, x, x_pos)); + ggml_tensor * k = linear(C, a.k_w, a.k_b, ggml_add(C, mem, mem_pos)); + ggml_tensor * v = linear(C, a.v_w, a.v_b, mem); + return linear(C, a.o_w, a.o_b, mha(C, q, k, v, dim / heads, heads, Tq, Tk)); +} + +} // namespace + +ggml_cgraph * ActModelArch::build(ggml_context * C, IO & io, int64_t h, int64_t w) const { + int64_t fh = 0, fw = 0; + feature_size(h, w, fh, fw); + const int64_t NF = fh*fw, T = n_1d() + n_views*NF; + auto ffn = [&](ggml_tensor * f1w, ggml_tensor * f1b, ggml_tensor * f2w, ggml_tensor * f2b, ggml_tensor * x) { + return gelu ? ffn_gelu_erf(C, f1w, f1b, f2w, f2b, x) : ffn_relu(C, f1w, f1b, f2w, f2b, x); + }; + + // ResNet over the views as one batch: [w, h, c, view]. The stem's ReLU + // commutes with the max pool, so it runs on the pooled map, a quarter the size. + io.pixels = ggml_new_tensor_4d(C, GGML_TYPE_F32, w, h, 3, n_views); + ggml_set_input(io.pixels); + ggml_tensor * x = ggml_pool_2d(C, conv2d(C, stem, io.pixels, 2, 3), GGML_OP_POOL_MAX, 3, 3, 2, 2, 1, 1); + x = ggml_relu(C, x); + for (const BlockW & b : res) { + ggml_tensor * y = ggml_relu(C, conv2d(C, b.conv1, x, b.stride, 1)); + y = conv2d(C, b.conv2, y, 1, 1); + ggml_tensor * skip = b.down.w ? conv2d(C, b.down, x, b.stride, 0) : x; + x = ggml_relu(C, ggml_add(C, y, skip)); + } + // 1x1 projection to the model width, tokens in (view h w) order. Only this + // last, small feature map is transposed to channels-first. + const int64_t fc = x->ne[2]; + ggml_tensor * f = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, x, NF, fc, n_views), 1, 0, 2, 3)); + ggml_tensor * img = linear(C, img_proj_w, img_proj_b, ggml_reshape_2d(C, f, fc, NF*n_views)); + + ggml_tensor * tok = ggml_reshape_2d(C, latent_tok, dim, 1); + if (state_dim > 0) { + io.state = ggml_new_tensor_1d(C, GGML_TYPE_F32, state_dim); + ggml_set_input(io.state); + tok = ggml_concat(C, tok, ggml_reshape_2d(C, linear(C, state_w, state_b, io.state), dim, 1), 1); + } + tok = ggml_concat(C, tok, img, 1); + + // Uploaded once per graph, so it is an output too: gallocr never reuses an + // output's memory for a later node. + io.enc_pos = ggml_new_tensor_2d(C, GGML_TYPE_F32, dim, T); + ggml_set_input(io.enc_pos); + ggml_set_output(io.enc_pos); + for (const EncLayerW & l : enc) { + if (pre_norm) { + ggml_tensor * n = layer_norm(C, tok, l.ln1_w, l.ln1_b, kLnEps); + tok = ggml_add(C, tok, self_attn(C, l.attn, n, io.enc_pos, dim, heads, T)); + n = layer_norm(C, tok, l.ln2_w, l.ln2_b, kLnEps); + tok = ggml_add(C, tok, ffn(l.fc1_w, l.fc1_b, l.fc2_w, l.fc2_b, n)); + } else { + tok = layer_norm(C, ggml_add(C, tok, self_attn(C, l.attn, tok, io.enc_pos, dim, heads, T)), + l.ln1_w, l.ln1_b, kLnEps); + tok = layer_norm(C, ggml_add(C, tok, ffn(l.fc1_w, l.fc1_b, l.fc2_w, l.fc2_b, tok)), + l.ln2_w, l.ln2_b, kLnEps); + } + } + if (enc_norm_w) + tok = layer_norm(C, tok, enc_norm_w, enc_norm_b, kLnEps); + + // Decoder: the queries start at zero and carry only their position. + ggml_tensor * a = ggml_scale(C, dec_pos, 0.0f); + for (const DecLayerW & l : dec) { + if (pre_norm) { + ggml_tensor * n = layer_norm(C, a, l.ln1_w, l.ln1_b, kLnEps); + a = ggml_add(C, a, self_attn(C, l.attn, n, dec_pos, dim, heads, chunk)); + n = layer_norm(C, a, l.ln2_w, l.ln2_b, kLnEps); + a = ggml_add(C, a, cross_attn(C, l.cross, n, dec_pos, tok, io.enc_pos, dim, heads, chunk, T)); + n = layer_norm(C, a, l.ln3_w, l.ln3_b, kLnEps); + a = ggml_add(C, a, ffn(l.fc1_w, l.fc1_b, l.fc2_w, l.fc2_b, n)); + } else { + a = layer_norm(C, ggml_add(C, a, self_attn(C, l.attn, a, dec_pos, dim, heads, chunk)), + l.ln1_w, l.ln1_b, kLnEps); + a = layer_norm(C, ggml_add(C, a, cross_attn(C, l.cross, a, dec_pos, tok, io.enc_pos, dim, heads, chunk, T)), + l.ln2_w, l.ln2_b, kLnEps); + a = layer_norm(C, ggml_add(C, a, ffn(l.fc1_w, l.fc1_b, l.fc2_w, l.fc2_b, a)), l.ln3_w, l.ln3_b, kLnEps); + } + } + a = layer_norm(C, a, dec_norm_w, dec_norm_b, kLnEps); + io.actions = linear(C, head_w, head_b, a); + ggml_set_output(io.actions); + + ggml_cgraph * gf = ggml_new_graph_custom(C, 4096, false); + ggml_build_forward_expand(gf, io.actions); + return gf; +} + +// Encoder positions: the learned 1D rows for the latent and state tokens, then +// ACTSinusoidalPositionEmbedding2d over the feature grid, the same for every +// camera. Channels are [y | x], each half interleaving sin and cos of the +// normalized coordinate (index+1)/(n+eps)*2pi over periods 10000^(2*(i/2)/half). +std::vector ActModelArch::enc_pos_table(int64_t fh, int64_t fw) const { + const int64_t half = dim / 2, NF = fh*fw, T = n_1d() + n_views*NF; + std::vector table((size_t) dim*T, 0.0f); + std::copy(pos_1d.begin(), pos_1d.end(), table.begin()); + const float two_pi = 6.28318530717958647692f, eps = 1e-6f; + std::vector inv((size_t) half); + for (int64_t i = 0; i < half; ++i) + inv[(size_t) i] = std::pow(10000.0f, (float) (2*(i/2)) / (float) half); + std::vector grid((size_t) dim*NF); + for (int64_t y = 0; y < fh; ++y) + for (int64_t x = 0; x < fw; ++x) { + const float yr = (float) (y + 1) / ((float) fh + eps) * two_pi; + const float xr = (float) (x + 1) / ((float) fw + eps) * two_pi; + float * row = grid.data() + (size_t) (y*fw + x)*dim; + for (int64_t i = 0; i < half; ++i) { + row[i] = (i % 2 == 0) ? std::sin(yr / inv[(size_t) i]) : std::cos(yr / inv[(size_t) i]); + row[half + i] = (i % 2 == 0) ? std::sin(xr / inv[(size_t) i]) : std::cos(xr / inv[(size_t) i]); + } + } + for (int64_t v = 0; v < n_views; ++v) + std::copy(grid.begin(), grid.end(), table.begin() + (size_t) (n_1d() + v*NF)*dim); + return table; +} + +namespace { + +bool load_config(const gguf_reader & g, ActModelArch & m) { + auto U = [&](const char * k, int64_t & dst) { + char key[96]; + std::snprintf(key, sizeof(key), "act.%s", k); + if (g.has(key)) + dst = (int64_t) g.u32(key); + }; + int64_t pre = 0, gelu = 0; + U("dim_model", m.dim); U("n_heads", m.heads); U("dim_feedforward", m.ff); + U("n_encoder_layers", m.enc_layers); U("n_decoder_layers", m.dec_layers); + U("chunk_size", m.chunk); U("state_dim", m.state_dim); U("action_dim", m.action_dim); + U("num_views", m.n_views); U("image_height", m.img_h); U("image_width", m.img_w); + U("pre_norm", pre); U("gelu", gelu); + m.pre_norm = pre != 0; + m.gelu = gelu != 0; + + const int64_t kb = gguf_find_key(g.gctx, "act.backbone_blocks"); + if (kb < 0 || gguf_get_kv_type(g.gctx, kb) != GGUF_TYPE_ARRAY || gguf_get_arr_type(g.gctx, kb) != GGUF_TYPE_INT32 || + gguf_get_arr_n(g.gctx, kb) != 4) { + std::fprintf(stderr, "vla(act): missing or malformed act.backbone_blocks\n"); + return false; + } + const int32_t * nb = (const int32_t *) gguf_get_arr_data(g.gctx, kb); + m.blocks.assign(nb, nb + 4); + + const int64_t kc = gguf_find_key(g.gctx, "act.cameras"); + if (kc >= 0 && gguf_get_kv_type(g.gctx, kc) == GGUF_TYPE_ARRAY && gguf_get_arr_type(g.gctx, kc) == GGUF_TYPE_STRING) + for (size_t i = 0; i < gguf_get_arr_n(g.gctx, kc); ++i) + m.cameras += std::string(i ? ", " : "") + gguf_get_arr_str(g.gctx, kc, i); + + bool ok = m.state_dim >= 0 && m.dim % m.heads == 0 && m.dim % 4 == 0; + for (int64_t v : { m.dim, m.heads, m.ff, m.enc_layers, m.dec_layers, m.chunk, m.action_dim, m.n_views, + m.blocks[0], m.blocks[1], m.blocks[2], m.blocks[3] }) + ok = ok && v >= 1; + if (!ok) { + std::fprintf(stderr, "vla(act): inconsistent dimensions in GGUF metadata\n"); + return false; + } + return true; +} + +SelfAttnW load_self_attn(WeightLoader & L, const std::string & p) { + SelfAttnW a{}; + a.qk_w = L.fuse_gemm((p + "_qk.weight").c_str(), { p + "_q.weight", p + "_k.weight" }); + a.qk_b = L.fuse_f32((p + "_qk.bias").c_str(), { p + "_q.bias", p + "_k.bias" }); + a.v_w = L.gemm("%s_v.weight", p.c_str()); + a.v_b = L.f32("%s_v.bias", p.c_str()); + a.o_w = L.gemm("%s_o.weight", p.c_str()); + a.o_b = L.f32("%s_o.bias", p.c_str()); + return a; +} + +bool load_weights(ActModelArch & m, gguf_reader & g) { + ggml_init_params wp = { (size_t) 4*1024*1024, nullptr, true }; + m.ctx_weights = ggml_init(wp); + if (!m.ctx_weights) + return false; + WeightLoader L("act", g, m.ctx_weights, m.mt); + + const ggml_type kt = m.conv_direct ? GGML_TYPE_F16 : GGML_TYPE_F32; + auto conv = [&](const std::string & p) { + return ConvW{ L.typed(kt, "%s.weight", p.c_str()), L.f32("%s.bias", p.c_str()) }; + }; + m.stem = conv("bb.conv1"); + for (int64_t li = 0; li < 4; ++li) + for (int64_t bi = 0; bi < m.blocks[(size_t) li]; ++bi) { + const std::string p = "bb.layer" + std::to_string(li + 1) + "." + std::to_string(bi); + BlockW b; + b.stride = (li > 0 && bi == 0) ? 2 : 1; + b.conv1 = conv(p + ".conv1"); + b.conv2 = conv(p + ".conv2"); + if (gguf_find_tensor(g.gctx, (p + ".down.weight").c_str()) >= 0) + b.down = conv(p + ".down"); + m.res.push_back(b); + } + m.img_proj_w = L.gemm("img_proj.weight"); + m.img_proj_b = L.f32("img_proj.bias"); + + m.latent_tok = L.f32("latent_tok"); + if (m.state_dim > 0) { + m.state_w = L.gemm("state_proj.weight"); + m.state_b = L.f32("state_proj.bias"); + } + + m.enc.resize((size_t) m.enc_layers); + for (int64_t i = 0; i < m.enc_layers; ++i) { + EncLayerW & e = m.enc[(size_t) i]; + const std::string p = "enc.blk." + std::to_string(i); + e.attn = load_self_attn(L, p + ".attn"); + e.fc1_w = L.gemm("%s.fc1.weight", p.c_str()); + e.fc1_b = L.f32("%s.fc1.bias", p.c_str()); + e.fc2_w = L.gemm("%s.fc2.weight", p.c_str()); + e.fc2_b = L.f32("%s.fc2.bias", p.c_str()); + e.ln1_w = L.f32("%s.ln1.weight", p.c_str()); + e.ln1_b = L.f32("%s.ln1.bias", p.c_str()); + e.ln2_w = L.f32("%s.ln2.weight", p.c_str()); + e.ln2_b = L.f32("%s.ln2.bias", p.c_str()); + } + if (m.pre_norm) { + m.enc_norm_w = L.f32("enc.norm.weight"); + m.enc_norm_b = L.f32("enc.norm.bias"); + } + + m.dec.resize((size_t) m.dec_layers); + for (int64_t i = 0; i < m.dec_layers; ++i) { + DecLayerW & d = m.dec[(size_t) i]; + const std::string p = "dec.blk." + std::to_string(i); + d.attn = load_self_attn(L, p + ".attn"); + for (const char * part : { "q", "k", "v", "o" }) { + ggml_tensor * w = L.gemm("%s.cross_%s.weight", p.c_str(), part); + ggml_tensor * b = L.f32("%s.cross_%s.bias", p.c_str(), part); + switch (part[0]) { + case 'q': d.cross.q_w = w; d.cross.q_b = b; break; + case 'k': d.cross.k_w = w; d.cross.k_b = b; break; + case 'v': d.cross.v_w = w; d.cross.v_b = b; break; + default: d.cross.o_w = w; d.cross.o_b = b; break; + } + } + d.fc1_w = L.gemm("%s.fc1.weight", p.c_str()); + d.fc1_b = L.f32("%s.fc1.bias", p.c_str()); + d.fc2_w = L.gemm("%s.fc2.weight", p.c_str()); + d.fc2_b = L.f32("%s.fc2.bias", p.c_str()); + d.ln1_w = L.f32("%s.ln1.weight", p.c_str()); + d.ln1_b = L.f32("%s.ln1.bias", p.c_str()); + d.ln2_w = L.f32("%s.ln2.weight", p.c_str()); + d.ln2_b = L.f32("%s.ln2.bias", p.c_str()); + d.ln3_w = L.f32("%s.ln3.weight", p.c_str()); + d.ln3_b = L.f32("%s.ln3.bias", p.c_str()); + } + m.dec_norm_w = L.f32("dec.norm.weight"); + m.dec_norm_b = L.f32("dec.norm.bias"); + m.dec_pos = L.f32("dec.pos"); + m.head_w = L.gemm("action_head.weight"); + m.head_b = L.f32("action_head.bias"); + + if (!L.upload(m.backend, &m.weight_buf)) + return false; + + if (m.stem.w->ne[2] != 3 || m.img_proj_w->ne[0] != m.res.back().conv2.w->ne[3] || + m.img_proj_w->ne[1] != m.dim || m.dec_pos->ne[0] != m.dim || m.dec_pos->ne[1] != m.chunk || + m.head_w->ne[1] != m.action_dim || m.enc[0].fc1_w->ne[1] != m.ff || + (m.state_w && m.state_w->ne[0] != m.state_dim)) { + std::fprintf(stderr, "vla(act): tensor shapes disagree with GGUF metadata\n"); + return false; + } + + // Host-side tables: the 1D position rows and the normalization stats. + m.pos_1d = g.read_f32("pos_1d"); + m.img_mean = g.read_f32("image_mean"); + m.img_std = g.read_f32("image_std"); + m.action_mean = g.read_f32("action_mean"); + m.action_std = g.read_f32("action_std"); + if (m.state_dim > 0) { + m.state_mean = g.read_f32("state_mean"); + m.state_std = g.read_f32("state_std"); + } + // Folded in once, so every use divides by the stored std. + for (std::vector * sd : { &m.img_std, &m.state_std, &m.action_std }) + for (float & s : *sd) + s += kNormEps; + if (m.pos_1d.size() != (size_t) (m.n_1d()*m.dim) || m.img_mean.size() != (size_t) (3*m.n_views) || + m.img_std.size() != m.img_mean.size() || m.action_mean.size() != (size_t) m.action_dim || + m.action_std.size() != (size_t) m.action_dim || m.state_mean.size() != (size_t) m.state_dim || + m.state_std.size() != (size_t) m.state_dim) { + std::fprintf(stderr, "vla(act): position or normalization tables disagree with GGUF metadata\n"); + return false; + } + return true; +} + +} // namespace + +std::unique_ptr act_create(const std::string& mmproj_path, + const std::string& ckpt_path, + const std::string&, + const Options& opts) { + if (!mmproj_path.empty()) + std::printf("vla(act): note - mmproj '%s' is ignored (vision is baked into the GGUF)\n", + mmproj_path.c_str()); + + auto m = std::make_unique(); + m->mt = opts.weight_dtype.value_or(GGML_TYPE_F32); + + gguf_reader g("act"); + if (!g.open(ckpt_path)) + return nullptr; + if (!g.has("act.architecture")) { + std::fprintf(stderr, "vla(act): %s is not an ACT GGUF\n", ckpt_path.c_str()); + return nullptr; + } + if (!load_config(g, *m)) + return nullptr; + + const Backend b = backend_init("vla(act)", m->n_threads); + if (!b.handle) + return nullptr; + m->backend = b.handle; + // An explicit --weight-dtype f32 keeps the convolutions F32 too. + m->conv_direct = b.is_cuda && cuda_conv_has_mma() && opts.weight_dtype.value_or(GGML_TYPE_F16) != GGML_TYPE_F32; + + if (!load_weights(*m, g)) + return nullptr; + + int64_t fh = 0, fw = 0; + ActModelArch::feature_size(m->img_h, m->img_w, fh, fw); + m->cfg.n_img = m->n_views * fh * fw; + m->cfg.n_lang = 0; + m->cfg.n_state = m->state_dim > 0 ? 1 : 0; + m->cfg.n_suffix = m->chunk; + m->cfg.hidden = m->dim; + m->cfg.max_state_dim = m->state_dim; + m->cfg.real_state_dim = m->state_dim; + m->cfg.max_action_dim = m->action_dim; + m->cfg.real_action_dim = m->action_dim; + + std::printf("vla(act): weights resident %.1f MiB (%s, %s convs) - ResNet x%lld views at %lldx%lld, " + "%lld+%lld layers, dim %lld, chunk %lld\n", + ggml_backend_buffer_get_size(m->weight_buf)/(1024.0*1024.0), dtype_name(m->mt), + m->conv_direct ? "f16 direct" : "f32 im2col", + (long long) m->n_views, (long long) m->img_h, (long long) m->img_w, (long long) m->enc_layers, (long long) m->dec_layers, (long long) m->dim, (long long) m->chunk); + if (!m->cameras.empty()) + std::printf("vla(act): send the views in this order: %s\n", m->cameras.c_str()); + return m; +} + +std::vector ActModelArch::predict(const Inputs& in) { + using clock = std::chrono::steady_clock; + const auto t0 = clock::now(); + stats = Stats{}; + + if (in.precomputed_img_emb) { + std::fprintf(stderr, "vla(act): precomputed_img_emb is not supported; pass raw images\n"); + return {}; + } + if (!in.images || in.n_images != n_views) { + std::fprintf(stderr, "vla(act): expected %lld views, got %d\n", (long long) n_views, in.n_images); + return {}; + } + const int64_t h = in.images[0].h, w = in.images[0].w; + for (int64_t i = 0; i < n_views; ++i) { + const ImageView & iv = in.images[i]; + if (!iv.data || iv.w != w || iv.h != h || w < 32 || h < 32) { + std::fprintf(stderr, "vla(act): view %lld is %dx%d; all views must share one size of at least 32x32\n", + (long long) i, iv.w, iv.h); + return {}; + } + } + if (state_dim > 0 && !in.state) { + std::fprintf(stderr, "vla(act): the checkpoint needs a %lld-dim state\n", (long long) state_dim); + return {}; + } + + bool fresh = false; + const size_t arena = ggml_tensor_overhead()*4096 + ggml_graph_overhead_custom(4096, false); + if (!graph.ensure(backend, Key{h, w}, arena, + [&](ggml_context * C, IO & io) { fresh = true; return build(C, io, h, w); })) { + std::fprintf(stderr, "vla(act): graph build/alloc failed\n"); + return {}; + } + IO & io = graph.io(); + if (fresh) { + int64_t fh = 0, fw = 0; + feature_size(h, w, fh, fw); + const std::vector table = enc_pos_table(fh, fw); + ggml_backend_tensor_set(io.enc_pos, table.data(), 0, ggml_nbytes(io.enc_pos)); + } + + // HWC to [w, h, c, view], normalized per camera with the dataset stats. + pixels.resize((size_t) w*h*3*n_views); + for (int64_t v = 0; v < n_views; ++v) + image_to_chw(in.images[v], &img_mean[(size_t) v*3], &img_std[(size_t) v*3], pixels.data() + (size_t) v*3*h*w); + ggml_backend_tensor_set(io.pixels, pixels.data(), 0, ggml_nbytes(io.pixels)); + if (state_dim > 0) { + std::vector s((size_t) state_dim); + for (int64_t i = 0; i < state_dim; ++i) + s[(size_t) i] = (in.state[i] - state_mean[(size_t) i]) / state_std[(size_t) i]; + ggml_backend_tensor_set(io.state, s.data(), 0, ggml_nbytes(io.state)); + } + + const auto tc = clock::now(); + graph_unique_names(graph.graph()); + if (ggml_backend_graph_compute(backend, graph.graph()) != GGML_STATUS_SUCCESS) { + std::fprintf(stderr, "vla(act): compute failed\n"); + return {}; + } + std::vector out((size_t) (chunk*action_dim)); + ggml_backend_tensor_get(io.actions, out.data(), 0, out.size()*sizeof(float)); + for (int64_t t = 0; t < chunk; ++t) + for (int64_t j = 0; j < action_dim; ++j) { + float & a = out[(size_t) (t*action_dim + j)]; + a = a * action_std[(size_t) j] + action_mean[(size_t) j]; + } + + stats.ms_inference = std::chrono::duration(clock::now() - tc).count(); + stats.ms_total = std::chrono::duration(clock::now() - t0).count(); + return out; +} + +} // namespace vla diff --git a/src/models/octo.cpp b/src/models/octo.cpp index fca0b31..aa453d8 100644 --- a/src/models/octo.cpp +++ b/src/models/octo.cpp @@ -16,6 +16,7 @@ #include "backend.h" #include "gguf_reader.h" #include "layers/attn.h" +#include "layers/conv.h" #include "layers/ffn.h" #include "layers/linear.h" #include "layers/norm.h" @@ -505,15 +506,6 @@ std::vector tensor_to_vec(const ggml_tensor * t) { return out; } -ggml_tensor * conv_2d_f32(ggml_context * C, ggml_tensor * w, ggml_tensor * x, int stride, int pad) { - ggml_tensor * col = ggml_im2col(C, w, x, stride, stride, pad, pad, 1, 1, true, GGML_TYPE_F32); - ggml_tensor * y = ggml_mul_mat(C, - ggml_reshape_2d(C, col, col->ne[0], col->ne[3]*col->ne[2]*col->ne[1]), - ggml_reshape_2d(C, w, w->ne[0]*w->ne[1]*w->ne[2], w->ne[3])); - y = ggml_reshape_4d(C, y, col->ne[1], col->ne[2], col->ne[3], w->ne[3]); - return ggml_cont(C, ggml_permute(C, y, 0, 1, 3, 2)); -} - // SmallStem16 for one camera view: four standardized-conv + GroupNorm + ReLU // stages at stride 2, a 1x1 patch embedding, a projection to the model width, // and the per-timestep position embedding. diff --git a/src/modules/preprocess.h b/src/modules/preprocess.h index 016cc97..b977e7b 100644 --- a/src/modules/preprocess.h +++ b/src/modules/preprocess.h @@ -39,23 +39,37 @@ inline bool view_ok(const char * arch, const ImageView & v, int64_t side) { return false; } -// HWC to CHW planar with per-channel mean/std. No resize: the view must -// already be side x side. arch only labels the error. +// HWC to CHW planar with per-channel mean/std, into 3*v.w*v.h floats at out. +// Any size; the caller has checked the view holds real data. +inline void image_to_chw(const ImageView & v, const float mean[3], const float std_[3], float * out) { + const int64_t n = (int64_t) v.w * v.h; + for (int64_t c=0; c<3; ++c) { + float * dst = out + c*n; + if (v.format == PixelFormat::U8) { + // A byte has 256 values: a table of the exact results replaces a + // per-pixel divide, which does not vectorize for a stride-3 read. + float lut[256]; + for (int k=0; k<256; ++k) + lut[k] = (k/255.0f-mean[c])/std_[c]; + const uint8_t * src = (const uint8_t *) v.data + c; + for (int64_t i=0; i & out) { if (!view_ok(arch, v, side)) return false; - out.assign((size_t) 3*side * side, 0.0f); - for (int64_t h=0; h int(cfg.n_lang)) { + // ACT takes no instruction (n_lang 0), so it accepts no tokens at all. + if (cfg.n_lang == 0 && req.lang_tokens_size() != 0) { + send_reply(make_error_response(rid, "this model takes no language tokens")); + continue; + } + if (cfg.n_lang > 0 && (req.lang_tokens_size() < 1 || req.lang_tokens_size() > int(cfg.n_lang))) { char buf[128]; std::snprintf(buf, sizeof(buf), "lang_tokens length %d out of range [1, %lld]", req.lang_tokens_size(), (long long) cfg.n_lang); diff --git a/src/serving/vla-bench.cpp b/src/serving/vla-bench.cpp index 3da2c27..f7f602c 100644 --- a/src/serving/vla-bench.cpp +++ b/src/serving/vla-bench.cpp @@ -33,13 +33,15 @@ namespace { void usage(const char * prog) { std::fprintf(stderr, "usage: %s (--ckpt c.gguf | -hf user/repo) [--mmproj m.gguf]\n" - " [--label name] [--images N] [--size N] [--tokens N]\n" + " [--label name] [--images N] [--size N] [--height N] [--tokens N]\n" " [--extra-token ID] [--extra-count N] [--warmup N] [--reps N] [--markdown]\n" " [precision flags]\n" " --mmproj ignored; every arch bundles its vision tower in the ckpt GGUF\n" " --label row label (default: the checkpoint filename)\n" " --images camera views (default 1)\n" - " --size square input side in pixels (default 224)\n" + " --size input width in pixels (default 224)\n" + " --height input height in pixels (default: the width; ACT runs at the\n" + " camera's own size, e.g. --size 640 --height 480)\n" " --tokens language token count (default 16)\n" " --extra-token token id appended --extra-count times (VLA-JEPA needs its\n" " tokens)\n" @@ -63,7 +65,7 @@ double percentile(const std::vector & v, double p) { int main(int argc, char ** argv) { std::string ckpt, mmproj, hf, label; - int n_images = 1, side = 224, n_tokens = 16, warmup = 3, reps = 20; + int n_images = 1, side = 224, height = 0, n_tokens = 16, warmup = 3, reps = 20; int extra_token = -1, extra_count = 0; bool markdown = false; vla::Options opts; @@ -85,6 +87,7 @@ int main(int argc, char ** argv) { else if (a == "--label") label = need("--label"); else if (a == "--images") n_images = std::atoi(need("--images")); else if (a == "--size") side = std::atoi(need("--size")); + else if (a == "--height") height = std::atoi(need("--height")); else if (a == "--tokens") n_tokens = std::atoi(need("--tokens")); else if (a == "--extra-token") extra_token = std::atoi(need("--extra-token")); else if (a == "--extra-count") extra_count = std::atoi(need("--extra-count")); @@ -121,8 +124,10 @@ int main(int argc, char ** argv) { usage(argv[0]); return 1; } - if (n_images < 1 || side < 16 || n_tokens < 1 || warmup < 0 || reps < 1) { - std::fprintf(stderr, "vla-bench: --images/--size/--tokens/--reps must be positive\n"); + if (height == 0) + height = side; + if (n_images < 1 || side < 16 || height < 16 || n_tokens < 1 || warmup < 0 || reps < 1) { + std::fprintf(stderr, "vla-bench: --images/--size/--height/--tokens/--reps must be positive\n"); return 1; } if (label.empty()) { @@ -137,14 +142,14 @@ int main(int argc, char ** argv) { } const vla::Config & cfg = vla::model_config(m); - std::vector> pixels(n_images, std::vector((size_t) 3*side * side)); + std::vector> pixels(n_images, std::vector((size_t) 3*side * height)); std::vector views(n_images); for (int v=0; v lang((size_t) n_tokens); @@ -205,7 +210,7 @@ int main(int argc, char ** argv) { label.c_str(), n_images, side, n_tokens, lo, mean, p50, p90, vision); } else { std::printf("%s: min %.1f ms mean %.1f ms p50 %.1f ms p90 %.1f ms vision %.1f ms (%d views, %dx%d, %d tokens, %d reps)\n", - label.c_str(), lo, mean, p50, p90, vision, n_images, side, side, n_tokens, reps); + label.c_str(), lo, mean, p50, p90, vision, n_images, side, height, n_tokens, reps); } vla::model_free(m); diff --git a/src/serving/vla-cli.cpp b/src/serving/vla-cli.cpp index 3f0a715..b9074ad 100644 --- a/src/serving/vla-cli.cpp +++ b/src/serving/vla-cli.cpp @@ -17,7 +17,7 @@ // No server, no simulator. --text is tokenized in-process when the GGUF carries a // SentencePiece tokenizer (Octo, or pi0/pi05/OpenVLA-OFT after // scripts/add_tokenizer_to_gguf.py) and shells out to scripts/tokenize_prompt.py -// otherwise; --tokens takes ids directly. +// otherwise; --tokens takes ids directly. ACT takes neither. // // vla-cli [--mmproj m.gguf] --ckpt c.gguf --image img.jpg [--image img2.jpg] // (--text "pick up the bowl" | --tokens id,id,...) [--state f,f,...] [--pretty] @@ -131,6 +131,7 @@ const char * arch_slug(Arch a) { case Arch::VLA_JEPA: return "vla_jepa"; case Arch::OCTO: return "octo"; case Arch::TURBOVLA: return "turbovla"; + case Arch::ACT: return "act"; } return ""; } @@ -251,6 +252,7 @@ void usage(const char * prog) { " --text instruction; tokenized in-process when the GGUF carries its\n" " tokenizer, else by scripts/tokenize_prompt.py (needs transformers)\n" " --tokens language token ids, comma-separated, if you tokenized already\n" + " (ACT reads no instruction: pass neither --text nor --tokens)\n" " --state proprioception floats, comma-separated (default zeros); pi05\n" " --text needs it, since the state is part of the prompt\n" " --pretty print one action row (max_action_dim values) per line\n" @@ -317,7 +319,7 @@ int main(int argc, char ** argv) { if (ckpt.empty()) return 1; } - if (ckpt.empty() || image_paths.empty() || (tokens_s.empty() && text_s.empty())) { + if (ckpt.empty() || image_paths.empty()) { usage(argv[0]); return 1; } @@ -325,6 +327,21 @@ int main(int argc, char ** argv) { std::fprintf(stderr, "vla-cli: pass --text or --tokens, not both\n"); return 1; } + Arch arch; + if (!detect_arch_from_ckpt(ckpt, &arch)) { + std::fprintf(stderr, "vla-cli: cannot detect the arch of %s\n", ckpt.c_str()); + return 1; + } + // ACT reads no instruction; every other arch needs one. + const bool takes_lang = arch != Arch::ACT; + if (takes_lang && tokens_s.empty() && text_s.empty()) { + usage(argv[0]); + return 1; + } + if (!takes_lang && (!tokens_s.empty() || !text_s.empty())) { + std::fprintf(stderr, "vla-cli: %s takes no instruction; drop --text/--tokens\n", arch_slug(arch)); + return 1; + } // Validate the cheap args before loading the model. std::vector lang; std::vector attn; // Octo only; empty leaves Inputs::attention_mask null. @@ -333,11 +350,6 @@ int main(int argc, char ** argv) { if (!parse_floats(state_s, state)) return 1; if (!text_s.empty()) { - Arch arch; - if (!detect_arch_from_ckpt(ckpt, &arch)) { - std::fprintf(stderr, "vla-cli: cannot detect the arch of %s for --text\n", ckpt.c_str()); - return 1; - } if (has_spm_tokenizer(ckpt, arch_slug(arch))) { if (!tokenize_prompt(ckpt, arch_slug(arch), text_s, state, lang, attn)) return 1; @@ -356,7 +368,7 @@ int main(int argc, char ** argv) { } if (!parse_ints(tokens_s, lang)) return 1; - if (lang.empty()) { + if (takes_lang && lang.empty()) { std::fprintf(stderr, "vla-cli: --tokens parsed to nothing\n"); return 1; } diff --git a/tests/predict_check.cpp b/tests/predict_check.cpp index 831ef0a..ba21968 100644 --- a/tests/predict_check.cpp +++ b/tests/predict_check.cpp @@ -18,7 +18,8 @@ // others read it, so fixed noise is reproducible for all archs). // // predict_check [mmproj.gguf] [n_images] -// env: VLA_IMG_SIZE (square input, default 224), VLA_BENCH_ITERS (>0 = time it), +// env: VLA_IMG_SIZE (input width, default 224), VLA_IMG_H (height, default the +// width), VLA_BENCH_ITERS (>0 = time it), // VLA_TIMING=phase, VLA_EXTRA_TOKEN / VLA_EXTRA_COUNT #include "model.h" @@ -67,7 +68,8 @@ int main(int argc, char** argv) { (long long)cfg.n_suffix, cfg.num_steps); const char* isz = std::getenv("VLA_IMG_SIZE"); - const int W = isz ? std::atoi(isz) : 224, H = W; + const char* ih = std::getenv("VLA_IMG_H"); + const int W = isz ? std::atoi(isz) : 224, H = ih ? std::atoi(ih) : W; std::vector> imgbuf(n_images, std::vector((size_t)3 * W * H)); std::vector views(n_images); for (int v = 0; v < n_images; ++v) { diff --git a/tests/py/test_converters.py b/tests/py/test_converters.py index 9f3fdfe..4c7febd 100644 --- a/tests/py/test_converters.py +++ b/tests/py/test_converters.py @@ -349,6 +349,44 @@ def test_turbovla_converter_remap(): } <= set(tensors.keys_read) +def test_act_converter_remap(): + import importlib + + A = importlib.import_module("convert_act_to_gguf") + tensors = _SourceTensors() + writer = _Writer() + # Folding needs real tensors; the remap only needs to know which ones were read. + fold_bn, A.fold_bn = A.fold_bn, lambda w, bn, t: (t[f"{bn}.weight"], w) + try: + A.write_backbone(writer, tensors, (1, 1, 1, 1)) + finally: + A.fold_bn = fold_bn + assert writer.names[:4] == ["bb.conv1.weight", "bb.conv1.bias", "bb.layer1.0.conv1.weight", "bb.layer1.0.conv1.bias"] + assert writer.names[-2:] == ["bb.layer4.0.down.weight", "bb.layer4.0.down.bias"] + assert {"model.backbone.conv1.weight", "model.backbone.bn1.weight", "model.backbone.layer4.0.downsample.0.weight", + "model.backbone.layer4.0.downsample.1.weight"} <= set(tensors.keys_read) + + class _Rows(_Tensor): + def __getitem__(self, s): return self # in_proj is sliced into q, k, v + + class _SlicedSource(_SourceTensors): + def __getitem__(self, key): + self.keys_read.append(key) + return _Rows() + + tensors = _SlicedSource() + writer = _Writer() + A.write_transformer(writer, tensors, 1, 1, 4, pre_norm=False) + attn = ["{}_" + f"{x}.{y}" for x in "qkvo" for y in ("weight", "bias")] + enc = [a.format("enc.blk.0.attn") for a in attn] + ffn = lambda p, n: [f"{p}.fc1.weight", f"{p}.fc1.bias", f"{p}.fc2.weight", f"{p}.fc2.bias"] + \ + [f"{p}.ln{i}.{y}" for i in range(1, n + 1) for y in ("weight", "bias")] + assert writer.names == enc + ffn("enc.blk.0", 2) + [a.format("dec.blk.0.attn") for a in attn] + \ + [a.format("dec.blk.0.cross") for a in attn] + ffn("dec.blk.0", 3) + ["dec.norm.weight", "dec.norm.bias"] + assert "model.decoder.layers.0.multihead_attn.in_proj_weight" in tensors.keys_read + assert "model.encoder.norm.weight" not in tensors.keys_read + + def test_quantize_skip_names(): import importlib Q = importlib.import_module("quantize_gguf")