diff --git a/CHANGELOG.md b/CHANGELOG.md index 07b6029..7838106 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -27,6 +27,19 @@ Notable changes to vla.cpp. Format loosely follows [Keep a Changelog](https://ke - `scripts/foldquant_fake_export.py` (uncalibrated file for bring-up), `scripts/inspect_gguf_quant.py` (contract check) and `scripts/foldquant_ref.py` (numpy reference). +- **PicoVLA** (`fast_smolvla`): DINOv3 ConvNeXt-T with a language-gated 2x2 token + merge, a five-layer Llama backbone with 16 record tokens, and a four-layer + action expert that reads the backbone's per-layer prefix K/V; three MeanFlow + steps. `scripts/convert_picovla_to_gguf.py` converts the LeRobot checkpoint + (the latent head is dropped). The record tokens are a one-frame memory that + the model keeps between calls; the LIBERO client sets `reset_memory` at each + episode start and applies the reference server's cross-chunk blending. + +### Changed + +- `vla_inputs` gains `reset_memory` (C ABI version 2), as do `vla::Inputs`, the + `PredictRequest` proto and the Python bindings' `predict`. It clears the + cross-call memory of the archs that keep one (PicoVLA); the others ignore it. ## [0.4.0] - 2026-09-30 diff --git a/CMakeLists.txt b/CMakeLists.txt index 2a01c88..0074928 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -249,6 +249,7 @@ add_library(vla_core ${VLA_CORE_LIB_TYPE} src/models/openvla_oft.cpp src/models/vla_jepa.cpp src/models/turbovla.cpp + src/models/picovla.cpp src/models/octo.cpp src/tokenizer.cpp ) diff --git a/README.md b/README.md index 2f73c8e..3efac87 100644 --- a/README.md +++ b/README.md @@ -163,6 +163,7 @@ supported (released and benchmarked), `~` = in progress, `-` = planned. | [VLA-JEPA](https://hf.co/vrfai/vla-jepa-libero) | Y | Y | - | Y | Y | ~ | | [Octo-Small](https://hf.co/vrfai/octo-small-libero-gguf) | Y | Y | Y | Y | - | Y | | [TurboVLA](https://hf.co/vrfai/turbovla-libero-gguf) | Y | Y | Y | Y | Y | Y | +| [PicoVLA](https://hf.co/khanhnd61/picovla-libero-pretrained-gguf) | Y | Y | - | - | - | - | --- diff --git a/bindings/python/vla_cpp/__init__.py b/bindings/python/vla_cpp/__init__.py index 10f02a3..9063a57 100644 --- a/bindings/python/vla_cpp/__init__.py +++ b/bindings/python/vla_cpp/__init__.py @@ -80,11 +80,13 @@ def __del__(self): self.close() def predict(self, images, tokens: Sequence[int], state=None, noise=None, - pixel_format: int = PIXEL_U8, timing: int = TIMING_NONE): + pixel_format: int = PIXEL_U8, timing: int = TIMING_NONE, reset_memory: bool = False): """Run one forward pass. images: one HWC array, or a sequence of them for multi-view. uint8 RGB by default; pass pixel_format=PIXEL_F32_RGB_01 for float RGB in [0, 1]. + reset_memory: set on an episode's first call; clears the cross-call + memory of the archs that keep one (PicoVLA). """ views = images if isinstance(images, (list, tuple)) else [images] if not views: @@ -133,6 +135,7 @@ def predict(self, images, tokens: Sequence[int], state=None, noise=None, if noise_ptr is not None: cin.noise = ctypes.cast(noise_ptr, POINTER(c_float)) cin.timing_detail = int(timing) + cin.reset_memory = int(bool(reset_memory)) out = POINTER(c_float)() n = c_int64() diff --git a/bindings/python/vla_cpp/_ffi.py b/bindings/python/vla_cpp/_ffi.py index 41ae61d..fab8a78 100644 --- a/bindings/python/vla_cpp/_ffi.py +++ b/bindings/python/vla_cpp/_ffi.py @@ -15,7 +15,7 @@ c_void_p, ) -ABI_VERSION = 1 +ABI_VERSION = 2 OK = 0 ERR_ARG = -1 @@ -86,6 +86,7 @@ class Inputs(ctypes.Structure): ("attention_mask", POINTER(c_int32)), ("attention_mask_n", c_int32), ("timing_detail", c_int32), + ("reset_memory", c_int32), ] diff --git a/docs/EVAL.md b/docs/EVAL.md index d6091bd..1b7c1ab 100644 --- a/docs/EVAL.md +++ b/docs/EVAL.md @@ -58,6 +58,11 @@ The GR00T models need two extras: - server side: `VLA_GR00T_EMBODIMENT` (`new_embodiment` for N1.5, `libero_panda` for N1.6, `libero_sim` for N1.7). +PicoVLA (`--arch picovla`) matches its reference client with `--n-action-steps 1`: it re-plans every +step, the client blends each chunk with the previous one, and `reset()` clears +the server's memory at the next request. With more steps per chunk the blending +is off. + ### SimplerEnv So far only **GR00T-N1.6** is wired (the `gr00t-n1d6-bridge` checkpoint with the diff --git a/docs/MODELS.md b/docs/MODELS.md index 75b8b3c..ec1aa8a 100644 --- a/docs/MODELS.md +++ b/docs/MODELS.md @@ -25,7 +25,7 @@ python scripts/convert_smolvla_to_gguf.py \ ## Quantization -Most shipped GGUFs are BF16. π0.5, Octo and TurboVLA ship F32, and GR00T N1.5 +Most shipped GGUFs are BF16. π0.5, Octo, TurboVLA and PicoVLA ship F32, and GR00T N1.5 and N1.6 are mostly F32. `scripts/quantize_gguf.py` repacks the LM-backbone weight matrices to a smaller type and copies everything else unchanged. The loader keeps the packed weights, so the file loads and runs like the original. diff --git a/docs/USAGE.md b/docs/USAGE.md index 52b5d44..cbda8e9 100644 --- a/docs/USAGE.md +++ b/docs/USAGE.md @@ -60,6 +60,11 @@ vla-server: bound to tcp://*:5555. ready. Use `--bind` to change the address and port. Stop the server with `Ctrl-C`. `vla-server` also takes `-hf user/repo[:file.gguf|:tag]` in place of a checkpoint path. +PicoVLA keeps a one-frame memory between requests (its record tokens), so a +server holds one episode's state: serve one client per server, and set +`reset_memory` on the first `PredictRequest` of every episode. The other +architectures are stateless and ignore the field. + Clients: the LIBERO and SimplerEnv runners in [EVAL.md](EVAL.md), and the real-robot client in the README's [Rollout on a real robot](../README.md#rollout-on-a-real-robot). diff --git a/eval/client/vla_cpp_client.py b/eval/client/vla_cpp_client.py index 382cd32..2bef750 100644 --- a/eval/client/vla_cpp_client.py +++ b/eval/client/vla_cpp_client.py @@ -57,6 +57,8 @@ "max_length": 21, }, "gr00t_n1_7": {"image_size": 256, "tokenizer": "nvidia/Cosmos-Reason2-2B", "max_state_dim": 132}, + # 256x256 LIBERO frames upsampled to the ConvNeXt's 448, as its DINOv3 processor does. + "picovla": {"image_size": 448, "tokenizer": "HuggingFaceTB/SmolVLM2-500M-Instruct", "max_state_dim": 32}, "gr00t_n1_5": {"image_size": 224, "tokenizer": "lerobot/eagle2hg-processor-groot-n1p5", "max_state_dim": 64, "trust_remote_code": True}, @@ -210,6 +212,15 @@ def __init__( self.n_action_steps = n_action_steps self._action_queue: deque = deque(maxlen=n_action_steps) + # PicoVLA keeps a one-frame memory on the server, cleared by the first + # request of each episode, and its reference server blends every chunk + # with the previous one (_picovla_blend). + self._picovla_prev = None + self._picovla_calls = 0 + if arch == "picovla" and n_action_steps != 1: + print(f"vla-cpp-direct[arch=picovla]: n_action_steps={n_action_steps}, cross-chunk " + "blending off (the reference re-plans every step)", flush=True) + self._bitvla_proprio_norm = None self._bitvla_unnorm_key = None if arch in ("bitvla", "vla_adapter"): @@ -648,6 +659,8 @@ def reset(self) -> None: self._action_queue.clear() self._episode += 1 self._step = 0 + self._picovla_prev = None + self._picovla_calls = 0 def get_action(self, observations: dict[str, Any]) -> np.ndarray: @@ -662,6 +675,8 @@ def get_action(self, observations: dict[str, Any]) -> np.ndarray: chunk = self._predict_chunk_octo(observations) else: chunk = self._predict_chunk(observations) + if self.arch == "picovla": + chunk = self._picovla_blend(chunk) for row in chunk[: self.n_action_steps, : self.real_action_dim]: self._action_queue.append(np.ascontiguousarray(row, dtype=np.float32)) return self._action_queue.popleft() @@ -726,6 +741,8 @@ def _predict_chunk(self, observations: dict[str, Any]) -> np.ndarray: ip.data = img.tobytes() req.lang_tokens.extend(lang.tolist()) req.state.extend(state_padded.tolist()) + if self.arch == "picovla": + req.reset_memory = self._picovla_calls == 0 self._maybe_add_fixed_noise(req) self.sock.send(req.SerializeToString()) @@ -740,6 +757,23 @@ def _predict_chunk(self, observations: dict[str, Any]) -> np.ndarray: return (np.array(resp.action_chunk, dtype=np.float32) .reshape(resp.chunk_size, resp.action_dim)) + def _picovla_blend(self, chunk: np.ndarray) -> np.ndarray: + """PicoVLA's cross-chunk blending (service/server.py, utils/policy_utils.py + make_prev_action_chunk): from an episode's third call on, the new chunk + is mixed with the previous one shifted by the one step executed since, + w_h = arccos(1 - 2(H-h-1)/H) / pi, 0.84 on the first step down to 0 on + the last. The reference blends normalized actions; the un-normalization + is affine per dimension, so world units blend the same.""" + if self._picovla_calls > 1 and self.n_action_steps == 1: + horizon = chunk.shape[0] + prev = np.zeros_like(chunk) + prev[:-1] = self._picovla_prev[1:] + w = (np.arccos(1.0 - 2.0 * (horizon - np.arange(horizon) - 1) / horizon) / np.pi)[:, None] + chunk = ((1.0 - w) * chunk + w * prev).astype(np.float32) + self._picovla_prev = chunk.copy() + self._picovla_calls += 1 + return chunk + def _predict_chunk_turbovla(self, observations: dict[str, Any]) -> np.ndarray: if self._turbovla_state_norm is None or self._turbovla_action_unnorm is None: raise RuntimeError("TurboVLA statistics were not initialized") @@ -957,7 +991,7 @@ def _predict_chunk_pi05(self, observations: dict[str, Any]) -> np.ndarray: _EVO1_MAX_TEXT_LENGTH = 1024 _FIXED_NOISE_ARCHS = { "smolvla", "pi0", "pi05", "evo1", "gr00t_n1_5", "gr00t_n1_6", "gr00t_n1_7", - "vla_jepa", "octo", + "vla_jepa", "octo", "picovla", } def _maybe_add_fixed_noise(self, req) -> None: diff --git a/eval/run_libero.sh b/eval/run_libero.sh index 6b90d6c..20ad155 100644 --- a/eval/run_libero.sh +++ b/eval/run_libero.sh @@ -38,7 +38,7 @@ Usage: $(basename "$0") -i [-o ] [-n ] [- -m MODEL which model to run: smol | pi0 | pi05 | bit | evo1 | vla_adapter | openvla_oft | gr00t_n1_5 | gr00t_n1_6 | gr00t_n1_7 | - octo | turbovla | vla_jepa | all + octo | turbovla | vla_jepa | picovla | all (default: all) -h show this help @@ -74,9 +74,9 @@ done shift $((OPTIND - 1)) case "${MODEL}" in - smol|pi0|pi05|bit|evo1|vla_adapter|openvla_oft|gr00t_n1_5|gr00t_n1_6|gr00t_n1_7|octo|turbovla|vla_jepa|all) ;; + smol|pi0|pi05|bit|evo1|vla_adapter|openvla_oft|gr00t_n1_5|gr00t_n1_6|gr00t_n1_7|octo|turbovla|vla_jepa|picovla|all) ;; *) - echo "ERROR: -m must be one of: smol | pi0 | pi05 | bit | evo1 | vla_adapter | openvla_oft | gr00t_n1_5 | gr00t_n1_6 | gr00t_n1_7 | octo | turbovla | vla_jepa | all (got '${MODEL}')" >&2 + echo "ERROR: -m must be one of: smol | pi0 | pi05 | bit | evo1 | vla_adapter | openvla_oft | gr00t_n1_5 | gr00t_n1_6 | gr00t_n1_7 | octo | turbovla | vla_jepa | picovla | all (got '${MODEL}')" >&2 exit 1 ;; esac @@ -147,6 +147,7 @@ N_ACTION_STEPS_GR00T_N1_7="${N_ACTION_STEPS_GR00T_N1_7:-16}" # N1.7 H4 closeout N_ACTION_STEPS_OCTO="${N_ACTION_STEPS_OCTO:-4}" N_ACTION_STEPS_TURBOVLA="${N_ACTION_STEPS_TURBOVLA:-12}" N_ACTION_STEPS_VLA_JEPA="${N_ACTION_STEPS_VLA_JEPA:-7}" +N_ACTION_STEPS_PICOVLA="${N_ACTION_STEPS_PICOVLA:-1}" # reference sync client re-plans every step mkdir -p "${OUTPUT_ROOT}" OUTPUT_ROOT="$(cd "${OUTPUT_ROOT}" && pwd)" @@ -552,5 +553,13 @@ if should_run vla_jepa; then fi fi +if should_run picovla; then + run_model picovla \ + "${MODELS_ROOT}/picovla-libero-pretrained-gguf" \ + "${N_ACTION_STEPS_PICOVLA}" \ + "" \ + "${MODELS_ROOT}/picovla-libero-pretrained-gguf/picovla-libero-f32.gguf" +fi + echo "====================" echo "Done. Results under ${OUTPUT_ROOT}" diff --git a/include/vla.h b/include/vla.h index f278f46..9ce6b00 100644 --- a/include/vla.h +++ b/include/vla.h @@ -34,7 +34,7 @@ extern "C" { #endif // Bumped on any incompatible change to the structs or functions below. -#define VLA_ABI_VERSION 1 +#define VLA_ABI_VERSION 2 typedef struct vla_model vla_model; @@ -124,6 +124,10 @@ typedef struct { int32_t attention_mask_n; int32_t timing_detail; ///< A vla_timing_detail value. + + /// Non-zero at an episode start: clear the model's cross-call memory first. + /// Only PicoVLA keeps any; other archs ignore it. + int32_t reset_memory; } vla_inputs; /// Milliseconds. Phase fields are zero unless timing_detail was VLA_TIMING_PHASE. diff --git a/scripts/convert_picovla_to_gguf.py b/scripts/convert_picovla_to_gguf.py new file mode 100644 index 0000000..214f9fd --- /dev/null +++ b/scripts/convert_picovla_to_gguf.py @@ -0,0 +1,232 @@ +#!/usr/bin/env python3 +# Copyright 2026 VinRobotics +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Convert a picovla (fast_smolvla) LeRobot checkpoint to GGUF. + +Reads a directory holding model.safetensors, config.json and the LeRobot +policy_{pre,post}processor normalizer sidecars, e.g. a download of +khanhnd61/picovla-libero-pretrained-gguf. The latent head (the offline-RL +actor) is dropped: the runtime implements the supervised MeanFlow policy only. + +Usage: + python convert_picovla_to_gguf.py --ckpt --out picovla-libero-f32.gguf +""" + +from __future__ import annotations + +import numpy as np +from safetensors import safe_open + +from gguf_blocks import lerobot_stats, norm_eps, write_decoder_blocks +from gguf_common import add, add_array, arg_parser, finish, kv_prefix, max_layer, open_writer, read_json, resolve_out + +ARCH = "picovla" +KV = kv_prefix(ARCH) + +PFX_BASE = "base_model" +PFX_LM = "base_model.backbone.context_model" +PFX_AEX = "base_model.expert_head" +PFX_VIS = "base_model.backbone.vision_encoder.vision_model" +PFX_CONN = "base_model.backbone.vision_encoder.connector" + +# Read from the bundled configs/{context_model,vision_model}/config.json of the +# picovla package; the checkpoint carries neither. RoPE is policy_utils.apply_rope, +# whose max_wavelength is 10000: the context config's rope_theta is never read. +LLAMA = dict(n_heads=8, n_kv_heads=4, head_dim=48, rms_eps=1e-5, rope_base=10000.0, pad_id=2) +CONVNEXT = dict(image_size=448, eps=1e-6, stem=4) + +PROJ = [ + ("state_proj.weight", "state_proj.weight"), + ("state_proj.bias", "state_proj.bias"), + ("action_in_proj.weight", "action_in_proj.weight"), + ("action_in_proj.bias", "action_in_proj.bias"), + ("action_out_proj.0.weight", "action_out_proj.0.weight"), + ("action_out_proj.0.bias", "action_out_proj.0.bias"), + ("action_out_proj.2.weight", "action_out_proj.2.weight"), + ("action_out_proj.2.bias", "action_out_proj.2.bias"), + ("time_emb_proj.0.weight", "time_mlp_in.weight"), + ("time_emb_proj.0.bias", "time_mlp_in.bias"), + ("time_emb_proj.2.weight", "time_mlp_out.weight"), + ("time_emb_proj.2.bias", "time_mlp_out.bias"), +] + +# Expert tensors beyond the plain decoder map: the cross projections that read +# the backbone's cached prefix K/V, and the AdaRMSNorm modulations. +EXPERT_EXTRA = [ + ("self_attn.cross_k_proj.weight", "cross_k.weight"), + ("self_attn.cross_v_proj.weight", "cross_v.weight"), + ("input_adanorm.linear.weight", "ada_attn.weight"), + ("input_adanorm.linear.bias", "ada_attn.bias"), + ("post_attention_adanorm.linear.weight", "ada_ffn.weight"), + ("post_attention_adanorm.linear.bias", "ada_ffn.bias"), +] + + +def s2d_weight(w): + """[OC, IC, k, k] conv weight as a [OC, k*k*IC] GEMM over (ky, kx, ic). + + The runtime keeps the ConvNeXt channels-last, so a stride-k k x k conv is a + space-to-depth view of k x k pixels (channel fastest, then kx, then ky) + followed by one matmul. + """ + return w.permute(0, 2, 3, 1).reshape(w.shape[0], -1).contiguous() + + +def dw_weight(w): + """[C, 1, 7, 7] depthwise weight as [ky, kx, C]: ggml's channels-last + depthwise kernel reads the taps channel-fastest.""" + return w[:, 0].permute(1, 2, 0).contiguous() + + +def write_vision(writer, g, depths: list[int]) -> None: + add(writer, "vis.stem.weight", s2d_weight(g(f"{PFX_VIS}.model.stages.0.downsample_layers.0.weight"))) + add(writer, "vis.stem.bias", g(f"{PFX_VIS}.model.stages.0.downsample_layers.0.bias")) + add(writer, "vis.stem_norm.weight", g(f"{PFX_VIS}.model.stages.0.downsample_layers.1.weight")) + add(writer, "vis.stem_norm.bias", g(f"{PFX_VIS}.model.stages.0.downsample_layers.1.bias")) + for s, depth in enumerate(depths): + st = f"{PFX_VIS}.model.stages.{s}" + if s > 0: + add(writer, f"vis.down.{s}.norm.weight", g(f"{st}.downsample_layers.0.weight")) + add(writer, f"vis.down.{s}.norm.bias", g(f"{st}.downsample_layers.0.bias")) + add(writer, f"vis.down.{s}.weight", s2d_weight(g(f"{st}.downsample_layers.1.weight"))) + add(writer, f"vis.down.{s}.bias", g(f"{st}.downsample_layers.1.bias")) + for i in range(depth): + lr, dst = f"{st}.layers.{i}", f"vis.blk.{s}.{i}" + gamma = g(f"{lr}.gamma") + add(writer, f"{dst}.dw.weight", dw_weight(g(f"{lr}.depthwise_conv.weight"))) + add(writer, f"{dst}.dw.bias", g(f"{lr}.depthwise_conv.bias")) + add(writer, f"{dst}.ln.weight", g(f"{lr}.layer_norm.weight")) + add(writer, f"{dst}.ln.bias", g(f"{lr}.layer_norm.bias")) + add(writer, f"{dst}.fc1.weight", g(f"{lr}.pointwise_conv1.weight")) + add(writer, f"{dst}.fc1.bias", g(f"{lr}.pointwise_conv1.bias")) + # Layer scale folds into the second pointwise conv. + add(writer, f"{dst}.fc2.weight", g(f"{lr}.pointwise_conv2.weight") * gamma.view(-1, 1)) + add(writer, f"{dst}.fc2.bias", g(f"{lr}.pointwise_conv2.bias") * gamma) + add(writer, "vis.norm.weight", g(f"{PFX_VIS}.layer_norm.weight")) + add(writer, "vis.norm.bias", g(f"{PFX_VIS}.layer_norm.bias")) + add(writer, "conn.task_proj.weight", g(f"{PFX_CONN}.task_proj.weight")) + add(writer, "conn.vision_proj.weight", g(f"{PFX_CONN}.vision_proj.weight")) + add(writer, "conn.proj.weight", g(f"{PFX_CONN}.proj.weight")) + + +def write_expert(writer, g, n_layers: int) -> None: + add(writer, "aex.cond_proj.weight", g(f"{PFX_AEX}.cond_proj.weight")) + add(writer, "aex.cond_proj.bias", g(f"{PFX_AEX}.cond_proj.bias")) + write_decoder_blocks(writer, g, f"{PFX_AEX}.model", "aex", n_layers) + for i in range(n_layers): + for src, dst in EXPERT_EXTRA: + add(writer, f"aex.blk.{i}.{dst}", g(f"{PFX_AEX}.model.layers.{i}.{src}")) + add(writer, "aex.output_ada.weight", g(f"{PFX_AEX}.model.norm.linear.weight")) + add(writer, "aex.output_ada.bias", g(f"{PFX_AEX}.model.norm.linear.bias")) + + +def compact_vocab(used_mask: np.ndarray, pad_id: int) -> tuple[np.ndarray, np.ndarray]: + """CompactTaskEmbedding.compact() (policy/base_modules.py): keep the tokens + seen in training plus padding, and send every other id to padding. + + Returns (rows of the original table to keep, original id -> compact id).""" + used = used_mask.astype(bool) + used[pad_id] = True + rows = np.flatnonzero(used) + remap = np.full(used.shape[0], -1, dtype=np.int64) + remap[rows] = np.arange(rows.size) + remap[remap < 0] = remap[pad_id] + return rows, remap.astype(np.int32) + + +def main() -> int: + ap = arg_parser(ARCH, "picovla checkpoint directory (model.safetensors + config.json)", __doc__) + args = ap.parse_args() + ckpt = args.ckpt.resolve() + out = resolve_out(args, ckpt, ARCH) + + cfg_json = read_json(ckpt / "config.json") + if cfg_json.get("type") != "fast_smolvla": + raise SystemExit(f"{ckpt} is not a fast_smolvla checkpoint (type={cfg_json.get('type')!r})") + if cfg_json.get("use_latent_head"): + raise SystemExit("use_latent_head=true is not supported: the runtime samples Gaussian noise") + for k, want in (("empty_cameras", 0), ("prefix_length", -1), ("num_language_layers", 1), ("n_obs_steps", 1)): + if cfg_json.get(k, want) != want: + raise SystemExit(f"{k}={cfg_json[k]} is not supported (expected {want})") + images = [k for k, f in cfg_json["input_features"].items() if f["type"] == "VISUAL"] + + sf = safe_open(str(ckpt / "model.safetensors"), framework="pt") + keys = set(sf.keys()) + g = sf.get_tensor + + n_layers = max_layer(keys, f"{PFX_LM}.layers.") - 1 + n_expert = max_layer(keys, f"{PFX_AEX}.model.layers.") + if n_layers < 1 or n_expert != n_layers: + raise SystemExit(f"backbone has {n_layers} joint layers but the expert has {n_expert}") + depths = [max_layer(keys, f"{PFX_VIS}.model.stages.{s}.layers.") for s in range(4)] + dims = [int(sf.get_slice(f"{PFX_VIS}.model.stages.{s}.layers.0.depthwise_conv.weight").get_shape()[0]) + for s in range(4)] + + hidden = int(sf.get_slice(f"{PFX_LM}.norm.weight").get_shape()[0]) + inter = int(sf.get_slice(f"{PFX_LM}.layers.0.mlp.gate_proj.weight").get_shape()[0]) + expert_inter, expert_h = sf.get_slice(f"{PFX_AEX}.model.layers.0.mlp.gate_proj.weight").get_shape() + n_record = int(sf.get_slice(f"{PFX_BASE}.record_tokens").get_shape()[1]) + q_dim = int(sf.get_slice(f"{PFX_LM}.layers.0.self_attn.q_proj.weight").get_shape()[0]) + if q_dim != LLAMA["n_heads"] * LLAMA["head_dim"]: + raise SystemExit(f"q_proj width {q_dim} != {LLAMA['n_heads']}x{LLAMA['head_dim']}") + + real_state = int(cfg_json["input_features"]["observation.state"]["shape"][0]) + real_action = int(cfg_json["output_features"]["action"]["shape"][0]) + lo, hi = (float(v) for v in cfg_json["transport_range"]) + + print(f"resolved: hidden={hidden} layers=1+{n_layers} expert={expert_h}/{expert_inter} " + f"convnext depths={depths} dims={dims} views={len(images)} records={n_record}") + + emb = g(f"{PFX_LM}.embed_tokens.token_embedding.weight") + rows, remap = compact_vocab(g(f"{PFX_LM}.embed_tokens.token_id_map").numpy(), LLAMA["pad_id"]) + print(f"vocabulary: {rows.size} of {emb.shape[0]} tokens seen in training") + + print("loading normalizer stats...") + stats = lerobot_stats(sf, ckpt, real_state, real_action, cfg_json.get("normalization_mapping") or {}) + + writer = open_writer(out, ARCH) + for k, v in dict(hidden=hidden, intermediate=inter, n_layers=n_layers, n_heads=LLAMA["n_heads"], + n_kv_heads=LLAMA["n_kv_heads"], head_dim=LLAMA["head_dim"], expert_hidden=int(expert_h), + expert_intermediate=int(expert_inter), n_record=n_record, + chunk_size=int(cfg_json["chunk_size"]), num_steps=int(cfg_json["n_denoise_steps"]), + max_state_dim=int(cfg_json["max_state_dim"]), max_action_dim=int(cfg_json["max_action_dim"]), + real_state_dim=real_state, real_action_dim=real_action, num_views=len(images), + image_size=CONVNEXT["image_size"], stem_patch=CONVNEXT["stem"], + tokenizer_max_length=int(cfg_json["tokenizer_max_length"]), vocab_size=int(emb.shape[0])).items(): + writer.add_uint32(KV(k), int(v)) + for k, v in dict(rms_eps=LLAMA["rms_eps"], rope_base=LLAMA["rope_base"], vis_eps=CONVNEXT["eps"], + transport_low=lo, transport_high=hi, min_period=float(cfg_json["min_period"]), + max_period=float(cfg_json["max_period"]), norm_eps=norm_eps(ckpt)).items(): + writer.add_float32(KV(k), float(v)) + writer.add_array(KV("vis_depths"), depths) + writer.add_array(KV("vis_dims"), dims) + writer.add_array(KV("token_map"), remap.tolist()) + + add(writer, "token_embd.weight", emb[rows].contiguous()) + add(writer, "vlm.output_norm.weight", g(f"{PFX_LM}.norm.weight")) + write_decoder_blocks(writer, g, PFX_LM, "vlm", 1 + n_layers) + add(writer, "record_tokens", g(f"{PFX_BASE}.record_tokens")[0]) + for src, dst in PROJ: + add(writer, dst, g(f"{PFX_BASE}.{src}")) + write_expert(writer, g, n_layers) + write_vision(writer, g, depths) + + for name, vec in stats.items(): + add_array(writer, name, vec) + return finish(writer, out) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/tokenize_prompt.py b/scripts/tokenize_prompt.py index 7bd7d9c..c0d6b0e 100644 --- a/scripts/tokenize_prompt.py +++ b/scripts/tokenize_prompt.py @@ -45,9 +45,10 @@ "gr00t_n1_6": "vrfai/gr00tn1d6-libero-gguf", "gr00t_n1_7": "nvidia/Cosmos-Reason2-2B", "turbovla": "bert-base-uncased", + "picovla": "HuggingFaceTB/SmolVLM2-500M-Instruct", } TRUST_REMOTE_CODE = {"evo1", "gr00t_n1_5"} -MAX_LENGTH = {"smolvla": 48, "pi0": 48, "pi05": 200, "turbovla": 21} +MAX_LENGTH = {"smolvla": 48, "pi0": 48, "pi05": 200, "turbovla": 21, "picovla": 48} VIEWS = {"evo1": 3, "bitvla": 2, "vla_jepa": 2, "gr00t_n1_5": 2, "gr00t_n1_6": 2, "gr00t_n1_7": 2} BITVLA_PROMPT = "What action should the robot take to {}?" @@ -92,7 +93,7 @@ def pi05_prompt(text: str, state: str, stats: str) -> str: def token_ids(arch: str, text: str, tok, args) -> list: views = args.views - if arch in ("smolvla", "pi0"): + if arch in ("smolvla", "pi0", "picovla"): text = text if text.endswith("\n") else text + "\n" if arch == "pi05": text = pi05_prompt(text, args.state, args.stats) diff --git a/src/arch.h b/src/arch.h index 146b6e4..65a5cbc 100644 --- a/src/arch.h +++ b/src/arch.h @@ -72,6 +72,7 @@ enum class Arch { OPENVLA_OFT,// DINOv2-L/14-reg4 + SigLIP-so400m/14 +Llama-2-7B + MLPResNet. VLA_JEPA, // LeRobot Qwen3-VL-2B-Instruct+V-JEPÀ+DiT-B FM. TURBOVLA, // TurboVLA (DINOv3 + BERT + VL Fusion + ACT decoder). + PICOVLA, // PicoVLA (DINOv3 ConvNeXt-T + Llama with record memory + MeanFlow expert). }; /** @@ -225,6 +226,17 @@ std::unique_ptr turbovla_create(const std::string& mmproj_path, const std::string& config_path, const Options& opts); +/** + * @brief Build a PicoVLA model. Vision is baked into @p ckpt_path. The model + * keeps a one-frame memory between predict calls; see + * @ref Inputs::reset_memory. + * @copydetails smolvla_create + */ +std::unique_ptr picovla_create(const std::string& mmproj_path, + const std::string& ckpt_path, + const std::string& config_path, + const Options& opts); + /** * @brief Inspect a GGUF and identify the architecture tag. * diff --git a/src/model.cpp b/src/model.cpp index ecbecdf..068f71b 100644 --- a/src/model.cpp +++ b/src/model.cpp @@ -80,7 +80,8 @@ bool detect_arch_gguf(const std::string& path, Arch* out) { try_str("openvla_oft.architecture", arch_str) || try_str("vla_jepa.architecture", arch_str) || try_str("vla_adapter.architecture", arch_str) || - try_str("turbovla.architecture", arch_str)) { + try_str("turbovla.architecture", arch_str) || + try_str("picovla.architecture", arch_str)) { if (arch_str == "smolvla") { *out = Arch::SMOLVLA; ok = true; @@ -133,6 +134,10 @@ bool detect_arch_gguf(const std::string& path, Arch* out) { *out = Arch::TURBOVLA; ok = true; } + else if (arch_str == "picovla") { + *out = Arch::PICOVLA; + ok = true; + } } gguf_free(gctx); @@ -312,6 +317,10 @@ Model* model_load(const std::string& mmproj_path, const std::string& ckpt_path, std::printf("vla: arch = turbovla\n"); impl = turbovla_create(mmproj_path, ckpt_path, config_path, opts); break; + case Arch::PICOVLA: + std::printf("vla: arch = picovla\n"); + impl = picovla_create(mmproj_path, ckpt_path, config_path, opts); + break; } if (!impl) return nullptr; diff --git a/src/model.h b/src/model.h index e6b6622..bd4135e 100644 --- a/src/model.h +++ b/src/model.h @@ -151,6 +151,11 @@ struct Inputs { int attention_mask_n = 0; ///< Length of @ref attention_mask. TimingDetail timing_detail = TimingDetail::NONE; + + /// Start of a new episode: clear what the model carries between calls before + /// predicting. Only PicoVLA keeps such state (its record-token memory); the + /// other architectures are stateless and ignore it. + bool reset_memory = false; }; /** diff --git a/src/models/picovla.cpp b/src/models/picovla.cpp new file mode 100644 index 0000000..75b7d19 --- /dev/null +++ b/src/models/picovla.cpp @@ -0,0 +1,764 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// PicoVLA (fast_smolvla @ 70d79c8): DINOv3 ConvNeXt-T over each camera, merged +// 2x2 by a softmax gate conditioned on the pooled task embedding; a Llama +// backbone whose layer 0 reads the instruction alone and whose four joint +// layers read [state | images | language | 16 record tokens | 16 past slots]; +// and a four-layer action expert that reads, layer for layer, the backbone's +// prefix K/V through small cross projections. MeanFlow in three steps. +// +// The record tokens are a one-frame memory: each call's record outputs are the +// next call's past slots, so the model keeps them between calls and +// Inputs::reset_memory clears them at an episode start. Everything else is one +// cached graph, keyed by the instruction length. + +#include "arch.h" +#include "backend.h" +#include "gguf.h" +#include "gguf_reader.h" +#include "loader.h" +#include "model.h" +#include "options.h" +#include "scratch_ctx.h" +#include "layers/attn.h" +#include "layers/embed.h" +#include "layers/ffn.h" +#include "layers/linear.h" +#include "layers/norm.h" +#include "modules/preprocess.h" + +#include "ggml.h" +#include "ggml-backend.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace vla { +namespace { + +constexpr float kImagenetMean[3] = {0.485f, 0.456f, 0.406f}; +constexpr float kImagenetStd[3] = {0.229f, 0.224f, 0.225f}; + +// nn.LayerNorm's default, used by the backbone's non-affine skip_norm and +// out_norm, which picovla builds without passing an eps. +constexpr float kLnEps = 1e-5f; + +struct LlamaLayerW { + ggml_tensor *attn_norm, *qkv, *o, *ffn_norm, *gate, *up, *down; +}; + +struct ExpertLayerW { + LlamaLayerW l; + ggml_tensor * cross_k, * cross_v; + ggml_tensor * ada_attn_w, * ada_attn_b, * ada_ffn_w, * ada_ffn_b; +}; + +struct BlockW { + ggml_tensor *dw_w, *dw_b, *ln_w, *ln_b, *fc1_w, *fc1_b, *fc2_w, *fc2_b; +}; + +struct StageW { + ggml_tensor *norm_w = nullptr, *norm_b = nullptr, *down_w = nullptr, *down_b = nullptr; + std::vector blocks; +}; + +} // namespace + +struct PicoVlaModelArch : public ModelArchBase { + PicoVlaModelArch() : ModelArchBase(Arch::PICOVLA) {} + ~PicoVlaModelArch() override { + graph.release(); + if (weight_buf) ggml_backend_buffer_free(weight_buf); + if (ctx_weights) ggml_free(ctx_weights); + if (backend) ggml_backend_free(backend); + } + + ggml_backend_t backend = nullptr; + int n_threads = default_cpu_threads(); + ggml_context * ctx_weights = nullptr; + ggml_backend_buffer_t weight_buf = nullptr; + ggml_type mt = GGML_TYPE_F32; + + int64_t hidden = 384, inter = 768, n_layers = 4, n_heads = 8, n_kv = 4, head_dim = 48; + int64_t ex_hidden = 256, ex_inter = 512, n_record = 16, chunk = 16; + int64_t max_state = 32, max_action = 32, real_state = 8, real_action = 7; + int64_t n_views = 2, image_size = 448, stem = 4, max_lang = 48, vocab = 49280; + int num_steps = 3; + float rms_eps = 1e-5f, rope_base = 10000.0f, vis_eps = 1e-6f; + float t_lo = 0.01f, t_hi = 0.99f, min_period = 0.1f, max_period = 4.0f, norm_eps = 1e-8f; + std::vector vis_depths, vis_dims; + std::vector token_map; // original id -> row of the compacted table + + ggml_tensor *tok_embd = nullptr, *lm_norm = nullptr, *record_tokens = nullptr; + std::vector vlm; // [0] is the language-only layer + std::vector expert; + ggml_tensor *state_w = nullptr, *state_b = nullptr, *act_in_w = nullptr, *act_in_b = nullptr; + ggml_tensor *act_out0_w = nullptr, *act_out0_b = nullptr, *act_out2_w = nullptr, *act_out2_b = nullptr; + ggml_tensor *time_in_w = nullptr, *time_in_b = nullptr, *time_out_w = nullptr, *time_out_b = nullptr; + ggml_tensor *cond_w = nullptr, *cond_b = nullptr, *out_ada_w = nullptr, *out_ada_b = nullptr; + ggml_tensor *stem_w = nullptr, *stem_b = nullptr, *stem_norm_w = nullptr, *stem_norm_b = nullptr; + std::vector stages; + ggml_tensor *vis_norm_w = nullptr, *vis_norm_b = nullptr; + ggml_tensor *task_proj = nullptr, *vision_proj = nullptr, *conn_proj = nullptr; + std::vector state_mean, state_std, action_mean, action_std; + + // The record memory: the previous call's record outputs (out_norm applied), + // fed back as the past slots. Zero after load and after reset_memory, which + // is what the reference feeds when it has no past (past_mask false). + std::vector past; + std::mt19937 rng{std::random_device{}()}; + + struct Key { + int64_t n_lang = -1; + bool operator==(const Key & o) const { return n_lang == o.n_lang; } + }; + struct IO { + ggml_tensor *pixels = nullptr, *ids = nullptr, *pos = nullptr, *mask = nullptr, *state = nullptr; + ggml_tensor *past = nullptr, *noise = nullptr, *time = nullptr, *dt = nullptr; + ggml_tensor *actions = nullptr, *record = nullptr; + }; + graph_cache graph; + + int64_t grid() const { return (image_size / stem) >> (int) (vis_dims.size() - 1); } // 448 -> 14 + int64_t img_tokens() const { return (grid() / 2) * (grid() / 2); } + int64_t prefix_len(int64_t L) const { return 1 + n_views*img_tokens() + L; } + + ggml_cgraph * build(ggml_context * C, IO & io, int64_t L) const; + std::vector predict(const Inputs& in) override; +}; + +namespace { + +// [C, W, H, N] channels-last to [k*k*C, W/k, H/k, N]: each column gathers a +// k x k pixel block, channel fastest, then kx, then ky. The converter orders +// the stride-k conv weights the same way, so the conv is one matmul. +ggml_tensor * space_to_depth(ggml_context * C, ggml_tensor * x, int64_t k) { + const int64_t c = x->ne[0], w = x->ne[1], h = x->ne[2], n = x->ne[3]; + ggml_tensor * t = ggml_reshape_4d(C, x, c*k, w/k, k, (h/k)*n); + t = ggml_cont(C, ggml_permute(C, t, 0, 2, 1, 3)); + return ggml_reshape_4d(C, t, c*k*k, w/k, h/k, n); +} + +ggml_tensor * stride_conv(ggml_context * C, ggml_tensor * x, ggml_tensor * w, ggml_tensor * b, int64_t k) { + ggml_tensor * s = space_to_depth(C, x, k); + const int64_t ow = s->ne[1], oh = s->ne[2], n = s->ne[3]; + ggml_tensor * y = linear(C, w, b, ggml_reshape_2d(C, s, s->ne[0], ow*oh*n)); + return ggml_reshape_4d(C, y, w->ne[1], ow, oh, n); +} + +// ConvNeXt block on a channels-last [C, W, H, N] map. The depthwise 7x7 runs +// on a permuted view: ggml's channels-last depthwise kernel reads the taps +// channel-fastest, which is how the converter stores them. +ggml_tensor * convnext_block(ggml_context * C, ggml_tensor * x, const BlockW & b, float eps) { + const int64_t ch = x->ne[0]; + ggml_tensor * k = ggml_permute(C, ggml_reshape_4d(C, b.dw_w, ch, 1, 7, 7), 3, 2, 0, 1); + ggml_tensor * y = ggml_conv_2d_dw_direct(C, k, ggml_permute(C, x, 2, 0, 1, 3), 1, 1, 3, 3, 1, 1); + y = ggml_add(C, ggml_permute(C, y, 1, 2, 0, 3), b.dw_b); + y = layer_norm(C, y, b.ln_w, b.ln_b, eps); + y = ffn_gelu_erf(C, b.fc1_w, b.fc1_b, b.fc2_w, b.fc2_b, y); + return ggml_add(C, x, y); +} + +// Columns [first, first+n) of a [D, T] activation. +ggml_tensor * cols(ggml_context * C, ggml_tensor * x, int64_t first, int64_t n) { + return ggml_view_2d(C, x, x->ne[0], n, x->nb[1], (size_t) first*x->nb[1]); +} + +// Head-split part `part` (0 q, 1 k, 2 v) of a fused [q | k | v] projection. +ggml_tensor * qkv_part(ggml_context * C, ggml_tensor * qkv, int64_t hd, int64_t nq, int64_t nkv, int part) { + const size_t es = ggml_element_size(qkv); + const int64_t heads = part == 0 ? nq : nkv; + const size_t off = part == 0 ? 0 : (size_t) (nq + (part - 1)*nkv)*hd*es; + return ggml_view_3d(C, qkv, hd, heads, qkv->ne[1], (size_t) hd*es, qkv->nb[1], off); +} + +// q/k/v arrive as [hd, heads, T]. GQA by ggml_mul_mat broadcast, which pairs +// query head h with K/V head h / (nq/nkv), as the reference's expand+reshape +// does (VLTrBase.attention_forward). +ggml_tensor * attend(ggml_context * C, ggml_tensor * q, ggml_tensor * k, ggml_tensor * v, ggml_tensor * mask, + int64_t hd, int64_t nq) { + ggml_tensor * Q = ggml_cont(C, ggml_permute(C, q, 0, 2, 1, 3)); + ggml_tensor * K = ggml_cont(C, ggml_permute(C, k, 0, 2, 1, 3)); + ggml_tensor * V = ggml_cont(C, ggml_permute(C, v, 1, 2, 0, 3)); + return attention(C, Q, K, V, mask, 1.0f / std::sqrt((float) hd), hd*nq, q->ne[2]); +} + +// policy_utils.apply_rope: the half-split (NeoX) rotation, base 10000. +ggml_tensor * rope_neox(ggml_context * C, ggml_tensor * x, ggml_tensor * pos, int64_t hd, float base) { + return ggml_rope_ext(C, x, pos, nullptr, (int) hd, GGML_ROPE_TYPE_NEOX, 0, base, 1.0f, 0.0f, 1.0f, 0.0f, 0.0f); +} + +ggml_tensor * swiglu(ggml_context * C, const LlamaLayerW & l, ggml_tensor * x) { + return ffn_swiglu(C, l.gate, nullptr, l.up, nullptr, l.down, nullptr, x); +} + +// RMSNorm without affine, scaled by (1 + Linear(silu(temb))): AdaRMSNorm. +ggml_tensor * ada_rms(ggml_context * C, ggml_tensor * x, ggml_tensor * w, ggml_tensor * b, ggml_tensor * st, + float eps) { + return ggml_mul(C, ggml_rms_norm(C, x, eps), ggml_scale_bias(C, linear(C, w, b, st), 1.0f, 1.0f)); +} + +} // namespace + +ggml_cgraph * PicoVlaModelArch::build(ggml_context * C, IO & io, int64_t L) const { + const int64_t P = prefix_len(L), R = n_record, T = P + 2*R, hd = head_dim; + const int64_t NI = img_tokens(); + + // Positions run on past the backbone into the expert's action rows. + io.pos = ggml_new_tensor_1d(C, GGML_TYPE_I32, P + R + std::max(R, chunk)); + ggml_set_input(io.pos); + auto pos = [&](int64_t first, int64_t n) { + return ggml_view_1d(C, io.pos, n, (size_t) first*ggml_element_size(io.pos)); + }; + + // Language: Llama layer 0 over the instruction alone, bidirectional. Its + // output (no final norm) is the language part of the prefix; the masked + // mean of its normed output is the task embedding that gates the merge. + io.ids = ggml_new_tensor_1d(C, GGML_TYPE_I32, L); + ggml_set_input(io.ids); + ggml_tensor * lang = ggml_get_rows(C, tok_embd, io.ids); + { + const LlamaLayerW & l = vlm[0]; + ggml_tensor * qkv = linear(C, l.qkv, nullptr, rms_norm(C, lang, l.attn_norm, rms_eps)); + ggml_tensor * q = rope_neox(C, qkv_part(C, qkv, hd, n_heads, n_kv, 0), pos(0, L), hd, rope_base); + ggml_tensor * k = rope_neox(C, qkv_part(C, qkv, hd, n_heads, n_kv, 1), pos(0, L), hd, rope_base); + ggml_tensor * att = attend(C, q, k, qkv_part(C, qkv, hd, n_heads, n_kv, 2), nullptr, hd, n_heads); + lang = ggml_add(C, lang, linear(C, l.o, nullptr, ggml_reshape_2d(C, att, hidden, L))); + lang = ggml_add(C, lang, swiglu(C, l, rms_norm(C, lang, l.ffn_norm, rms_eps))); + } + ggml_tensor * task = rms_norm(C, lang, lm_norm, rms_eps); + task = ggml_reshape_1d(C, ggml_mean(C, ggml_cont(C, ggml_transpose(C, task))), hidden); + + // DINOv3 ConvNeXt-T, channels-last [C, W, H, views]. Pixels arrive + // normalized and interleaved, which already is that layout. + io.pixels = ggml_new_tensor_4d(C, GGML_TYPE_F32, 3, image_size, image_size, n_views); + ggml_set_input(io.pixels); + ggml_tensor * x = stride_conv(C, io.pixels, stem_w, stem_b, stem); + x = layer_norm(C, x, stem_norm_w, stem_norm_b, vis_eps); + for (size_t s = 0; s < stages.size(); ++s) { + const StageW & st = stages[s]; + if (s > 0) + x = stride_conv(C, layer_norm(C, x, st.norm_w, st.norm_b, vis_eps), st.down_w, st.down_b, 2); + for (const BlockW & b : st.blocks) + x = convnext_block(C, x, b, vis_eps); + } + x = layer_norm(C, x, vis_norm_w, vis_norm_b, vis_eps); + + // TaskAwareDINOv3Connector: the four tokens of each 2x2 block are weighted + // by a softmax against the task query (times 4, to keep the scale), then + // concatenated and projected. space_to_depth orders them dx-fastest, as + // the reference's reshape+permute does. + const int64_t vd = x->ne[0]; + ggml_tensor * merged = ggml_reshape_3d(C, space_to_depth(C, x, 2), vd, 4, NI*n_views); + ggml_tensor * keys = linear(C, vision_proj, nullptr, ggml_reshape_2d(C, merged, vd, 4*NI*n_views)); + ggml_tensor * query = linear(C, task_proj, nullptr, ggml_reshape_2d(C, task, hidden, 1)); + ggml_tensor * scores = ggml_reshape_2d(C, ggml_mul_mat(C, keys, query), 4, NI*n_views); + ggml_tensor * gates = ggml_scale(C, ggml_soft_max_ext(C, scores, nullptr, 1.0f / std::sqrt((float) query->ne[0]), + 0.0f), 4.0f); + merged = ggml_mul(C, merged, ggml_reshape_3d(C, gates, 1, 4, NI*n_views)); + ggml_tensor * img = linear(C, conn_proj, nullptr, ggml_reshape_2d(C, merged, 4*vd, NI*n_views)); + + // Backbone: [state | images | language | records | past], positions 0..T-1. + io.state = ggml_new_tensor_1d(C, GGML_TYPE_F32, max_state); + io.past = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden, R); + io.mask = ggml_new_tensor_2d(C, GGML_TYPE_F32, T, T); + ggml_set_input(io.state); + ggml_set_input(io.past); + ggml_set_input(io.mask); + ggml_tensor * h = linear(C, state_w, state_b, ggml_reshape_2d(C, io.state, max_state, 1)); + h = ggml_concat(C, ggml_concat(C, h, img, 1), lang, 1); + h = ggml_concat(C, ggml_concat(C, h, record_tokens, 1), io.past, 1); + + // Per joint layer, the prefix part of K (before RoPE) and V through a + // non-affine LayerNorm across all KV heads: the expert's cache. + std::vector kc, vc; + for (int64_t i = 1; i <= n_layers; ++i) { + const LlamaLayerW & l = vlm[(size_t) i]; + ggml_tensor * qkv = linear(C, l.qkv, nullptr, rms_norm(C, h, l.attn_norm, rms_eps)); + const size_t es = ggml_element_size(qkv); + kc.push_back(ggml_norm(C, ggml_cont(C, ggml_view_2d(C, qkv, n_kv*hd, P, qkv->nb[1], (size_t) n_heads*hd*es)), + kLnEps)); + vc.push_back(ggml_norm(C, ggml_cont(C, ggml_view_2d(C, qkv, n_kv*hd, P, qkv->nb[1], + (size_t) (n_heads + n_kv)*hd*es)), kLnEps)); + ggml_tensor * q = rope_neox(C, qkv_part(C, qkv, hd, n_heads, n_kv, 0), pos(0, T), hd, rope_base); + ggml_tensor * k = rope_neox(C, qkv_part(C, qkv, hd, n_heads, n_kv, 1), pos(0, T), hd, rope_base); + ggml_tensor * att = attend(C, q, k, qkv_part(C, qkv, hd, n_heads, n_kv, 2), io.mask, hd, n_heads); + h = ggml_add(C, h, linear(C, l.o, nullptr, ggml_reshape_2d(C, att, hidden, T))); + h = ggml_add(C, h, swiglu(C, l, rms_norm(C, h, l.ffn_norm, rms_eps))); + } + // No final RMSNorm: records and past slots leave through out_norm. + io.record = ggml_norm(C, ggml_cont(C, cols(C, h, P, R)), kLnEps); + ggml_set_output(io.record); + ggml_tensor * cond = ggml_norm(C, ggml_cont(C, cols(C, h, P + R, R)), kLnEps); + + // Expert, cond rows. They see the prefix and each other but never the + // action rows, and carry no time, so they run once per call: what each + // layer leaves is the RoPE'd [prefix | cond] K/V the action rows attend. + std::vector ctx_k, ctx_v; + ggml_tensor * c = linear(C, cond_w, cond_b, cond); + for (int64_t i = 0; i < n_layers; ++i) { + const ExpertLayerW & e = expert[(size_t) i]; + ggml_tensor * pk = ggml_reshape_3d(C, ggml_mul_mat(C, e.cross_k, kc[(size_t) i]), hd, n_kv, P); + ggml_tensor * pv = ggml_reshape_3d(C, ggml_mul_mat(C, e.cross_v, vc[(size_t) i]), hd, n_kv, P); + ggml_tensor * qkv = linear(C, e.l.qkv, nullptr, rms_norm(C, c, e.l.attn_norm, rms_eps)); + ggml_tensor * q = rope_neox(C, qkv_part(C, qkv, hd, n_heads, n_kv, 0), pos(P, R), hd, rope_base); + ggml_tensor * k = rope_neox(C, qkv_part(C, qkv, hd, n_heads, n_kv, 1), pos(P, R), hd, rope_base); + ggml_tensor * v = ggml_cont(C, qkv_part(C, qkv, hd, n_heads, n_kv, 2)); + ctx_k.push_back(ggml_concat(C, rope_neox(C, pk, pos(0, P), hd, rope_base), k, 2)); + ctx_v.push_back(ggml_concat(C, pv, v, 2)); + ggml_tensor * att = attend(C, q, ctx_k.back(), ctx_v.back(), nullptr, hd, n_heads); + c = ggml_add(C, c, linear(C, e.l.o, nullptr, ggml_reshape_2d(C, att, n_heads*hd, R))); + c = ggml_add(C, c, swiglu(C, e.l, rms_norm(C, c, e.l.ffn_norm, rms_eps))); + } + + // MeanFlow: x <- x - (t - r) * u(x, r, t), every step unrolled. io.time + // holds each step's [sincos(r) | sincos(t)], io.dt its t - r. + const int64_t A = chunk, tdim = time_in_w->ne[0]; + io.noise = ggml_new_tensor_2d(C, GGML_TYPE_F32, max_action, A); + io.time = ggml_new_tensor_2d(C, GGML_TYPE_F32, tdim, num_steps); + io.dt = ggml_new_tensor_1d(C, GGML_TYPE_F32, num_steps); + ggml_set_input(io.noise); + ggml_set_input(io.time); + ggml_set_input(io.dt); + ggml_tensor * xt = io.noise; + for (int s = 0; s < num_steps; ++s) { + ggml_tensor * tin = ggml_view_2d(C, io.time, tdim, 1, io.time->nb[1], (size_t) s*io.time->nb[1]); + ggml_tensor * dt = ggml_view_1d(C, io.dt, 1, (size_t) s*sizeof(float)); + ggml_tensor * temb = linear(C, time_out_w, time_out_b, ggml_silu(C, linear(C, time_in_w, time_in_b, tin))); + ggml_tensor * st = ggml_silu(C, temb); + + ggml_tensor * a = linear(C, act_in_w, act_in_b, xt); + for (int64_t i = 0; i < n_layers; ++i) { + const ExpertLayerW & e = expert[(size_t) i]; + ggml_tensor * qkv = linear(C, e.l.qkv, nullptr, ada_rms(C, a, e.ada_attn_w, e.ada_attn_b, st, rms_eps)); + ggml_tensor * q = rope_neox(C, qkv_part(C, qkv, hd, n_heads, n_kv, 0), pos(P + R, A), hd, rope_base); + ggml_tensor * k = rope_neox(C, qkv_part(C, qkv, hd, n_heads, n_kv, 1), pos(P + R, A), hd, rope_base); + ggml_tensor * v = ggml_cont(C, qkv_part(C, qkv, hd, n_heads, n_kv, 2)); + ggml_tensor * att = attend(C, q, ggml_concat(C, ctx_k[(size_t) i], k, 2), + ggml_concat(C, ctx_v[(size_t) i], v, 2), nullptr, hd, n_heads); + a = ggml_add(C, a, linear(C, e.l.o, nullptr, ggml_reshape_2d(C, att, n_heads*hd, A))); + a = ggml_add(C, a, swiglu(C, e.l, ada_rms(C, a, e.ada_ffn_w, e.ada_ffn_b, st, rms_eps))); + } + ggml_tensor * u = ada_rms(C, a, out_ada_w, out_ada_b, st, rms_eps); + u = linear(C, act_out2_w, act_out2_b, ggml_silu(C, linear(C, act_out0_w, act_out0_b, u))); + xt = ggml_sub(C, xt, ggml_mul(C, u, dt)); + } + io.actions = xt; + ggml_set_output(io.actions); + + ggml_cgraph * gf = ggml_new_graph_custom(C, 8192, false); + ggml_build_forward_expand(gf, io.record); + ggml_build_forward_expand(gf, io.actions); + return gf; +} + +namespace { + +bool read_i64_array(const gguf_reader & g, const char * key, std::vector & out) { + const int64_t id = gguf_find_key(g.gctx, key); + if (id < 0 || gguf_get_kv_type(g.gctx, id) != GGUF_TYPE_ARRAY || gguf_get_arr_type(g.gctx, id) != GGUF_TYPE_INT32) + return false; + const int32_t * d = (const int32_t *) gguf_get_arr_data(g.gctx, id); + out.assign(d, d + gguf_get_arr_n(g.gctx, id)); + return true; +} + +bool load_config(const gguf_reader & g, PicoVlaModelArch & m) { + auto U = [&](const char * k, int64_t & dst) { + char key[96]; + std::snprintf(key, sizeof(key), "picovla.%s", k); + if (g.has(key)) + dst = (int64_t) g.u32(key); + }; + auto F = [&](const char * k, float & dst) { + char key[96]; + std::snprintf(key, sizeof(key), "picovla.%s", k); + if (g.has(key)) + dst = g.f32(key); + }; + int64_t steps = m.num_steps; + U("hidden", m.hidden); U("intermediate", m.inter); U("n_layers", m.n_layers); + U("n_heads", m.n_heads); U("n_kv_heads", m.n_kv); U("head_dim", m.head_dim); + U("expert_hidden", m.ex_hidden); U("expert_intermediate", m.ex_inter); + U("n_record", m.n_record); U("chunk_size", m.chunk); U("num_steps", steps); + U("max_state_dim", m.max_state); U("max_action_dim", m.max_action); + U("real_state_dim", m.real_state); U("real_action_dim", m.real_action); + U("num_views", m.n_views); U("image_size", m.image_size); U("stem_patch", m.stem); + U("tokenizer_max_length", m.max_lang); U("vocab_size", m.vocab); + F("rms_eps", m.rms_eps); F("rope_base", m.rope_base); F("vis_eps", m.vis_eps); + F("transport_low", m.t_lo); F("transport_high", m.t_hi); + F("min_period", m.min_period); F("max_period", m.max_period); F("norm_eps", m.norm_eps); + m.num_steps = (int) steps; + + std::vector map; + if (!read_i64_array(g, "picovla.vis_depths", m.vis_depths) || !read_i64_array(g, "picovla.vis_dims", m.vis_dims) || + !read_i64_array(g, "picovla.token_map", map)) { + std::fprintf(stderr, "vla(picovla): GGUF lacks vis_depths / vis_dims / token_map\n"); + return false; + } + m.token_map.assign(map.begin(), map.end()); + + bool ok = m.vis_depths.size() == m.vis_dims.size() && !m.vis_dims.empty() && m.stem == 4 && + (int64_t) m.token_map.size() == m.vocab && m.image_size % (m.stem << (m.vis_dims.size() - 1)) == 0 && + m.grid() % 2 == 0 && m.n_heads % m.n_kv == 0 && m.num_steps >= 1 && m.t_lo < m.t_hi && + m.real_state <= m.max_state && m.real_action <= m.max_action; + for (int64_t v : { m.hidden, m.inter, m.n_layers, m.n_heads, m.n_kv, m.head_dim, m.ex_hidden, m.ex_inter, + m.n_record, m.chunk, m.real_state, m.real_action, m.n_views, m.max_lang }) + ok = ok && v >= 1; + if (!ok) { + std::fprintf(stderr, "vla(picovla): inconsistent dimensions in GGUF metadata\n"); + return false; + } + return true; +} + +LlamaLayerW load_llama(WeightLoader & L, const char * pfx, int64_t i) { + char n[4][96]; + auto N = [&](int slot, const char * s) { + std::snprintf(n[slot], sizeof(n[slot]), "%s.blk.%lld.%s", pfx, (long long) i, s); + return std::string(n[slot]); + }; + LlamaLayerW w; + w.attn_norm = L.f32("%s", N(0, "attn_norm.weight").c_str()); + w.qkv = L.fuse_gemm(N(0, "attn_qkv.weight").c_str(), + { N(1, "attn_q.weight"), N(2, "attn_k.weight"), N(3, "attn_v.weight") }); + w.o = L.gemm("%s", N(0, "attn_o.weight").c_str()); + w.ffn_norm = L.f32("%s", N(0, "ffn_norm.weight").c_str()); + w.gate = L.gemm("%s", N(0, "ffn_gate.weight").c_str()); + w.up = L.gemm("%s", N(0, "ffn_up.weight").c_str()); + w.down = L.gemm("%s", N(0, "ffn_down.weight").c_str()); + return w; +} + +bool load_weights(PicoVlaModelArch & m, gguf_reader & g) { + ggml_init_params wp = { (size_t) 16*1024*1024, nullptr, true }; + m.ctx_weights = ggml_init(wp); + if (!m.ctx_weights) + return false; + WeightLoader L("picovla", g, m.ctx_weights, m.mt); + + m.tok_embd = L.f32("token_embd.weight"); + m.lm_norm = L.f32("vlm.output_norm.weight"); + m.record_tokens = L.f32("record_tokens"); + for (int64_t i = 0; i <= m.n_layers; ++i) + m.vlm.push_back(load_llama(L, "vlm", i)); + + m.state_w = L.gemm("state_proj.weight"); + m.state_b = L.f32("state_proj.bias"); + m.act_in_w = L.gemm("action_in_proj.weight"); + m.act_in_b = L.f32("action_in_proj.bias"); + m.act_out0_w = L.gemm("action_out_proj.0.weight"); + m.act_out0_b = L.f32("action_out_proj.0.bias"); + m.act_out2_w = L.gemm("action_out_proj.2.weight"); + m.act_out2_b = L.f32("action_out_proj.2.bias"); + m.time_in_w = L.gemm("time_mlp_in.weight"); + m.time_in_b = L.f32("time_mlp_in.bias"); + m.time_out_w = L.gemm("time_mlp_out.weight"); + m.time_out_b = L.f32("time_mlp_out.bias"); + + m.cond_w = L.gemm("aex.cond_proj.weight"); + m.cond_b = L.f32("aex.cond_proj.bias"); + m.out_ada_w = L.gemm("aex.output_ada.weight"); + m.out_ada_b = L.f32("aex.output_ada.bias"); + for (int64_t i = 0; i < m.n_layers; ++i) { + ExpertLayerW e; + e.l = load_llama(L, "aex", i); + e.cross_k = L.gemm("aex.blk.%lld.cross_k.weight", (long long) i); + e.cross_v = L.gemm("aex.blk.%lld.cross_v.weight", (long long) i); + e.ada_attn_w = L.gemm("aex.blk.%lld.ada_attn.weight", (long long) i); + e.ada_attn_b = L.f32("aex.blk.%lld.ada_attn.bias", (long long) i); + e.ada_ffn_w = L.gemm("aex.blk.%lld.ada_ffn.weight", (long long) i); + e.ada_ffn_b = L.f32("aex.blk.%lld.ada_ffn.bias", (long long) i); + m.expert.push_back(e); + } + + // The depthwise taps and the norms stay F32: ggml's depthwise conv reads + // F32 or F16 taps only, and they are a rounding error of the tower's size. + m.stem_w = L.gemm("vis.stem.weight"); + m.stem_b = L.f32("vis.stem.bias"); + m.stem_norm_w = L.f32("vis.stem_norm.weight"); + m.stem_norm_b = L.f32("vis.stem_norm.bias"); + m.stages.resize(m.vis_depths.size()); + for (size_t s = 0; s < m.stages.size(); ++s) { + StageW & st = m.stages[s]; + const long long S = (long long) s; + if (s > 0) { + st.norm_w = L.f32("vis.down.%lld.norm.weight", S); + st.norm_b = L.f32("vis.down.%lld.norm.bias", S); + st.down_w = L.gemm("vis.down.%lld.weight", S); + st.down_b = L.f32("vis.down.%lld.bias", S); + } + for (int64_t i = 0; i < m.vis_depths[s]; ++i) { + const long long I = (long long) i; + BlockW b; + b.dw_w = L.f32("vis.blk.%lld.%lld.dw.weight", S, I); + b.dw_b = L.f32("vis.blk.%lld.%lld.dw.bias", S, I); + b.ln_w = L.f32("vis.blk.%lld.%lld.ln.weight", S, I); + b.ln_b = L.f32("vis.blk.%lld.%lld.ln.bias", S, I); + b.fc1_w = L.gemm("vis.blk.%lld.%lld.fc1.weight", S, I); + b.fc1_b = L.f32("vis.blk.%lld.%lld.fc1.bias", S, I); + b.fc2_w = L.gemm("vis.blk.%lld.%lld.fc2.weight", S, I); + b.fc2_b = L.f32("vis.blk.%lld.%lld.fc2.bias", S, I); + st.blocks.push_back(b); + } + } + m.vis_norm_w = L.f32("vis.norm.weight"); + m.vis_norm_b = L.f32("vis.norm.bias"); + m.task_proj = L.gemm("conn.task_proj.weight"); + m.vision_proj = L.gemm("conn.vision_proj.weight"); + m.conn_proj = L.gemm("conn.proj.weight"); + + if (!L.upload(m.backend, &m.weight_buf)) + return false; + + const int64_t vd = m.vis_dims.back(); + bool ok = m.tok_embd->ne[0] == m.hidden && m.record_tokens->ne[0] == m.hidden && + m.record_tokens->ne[1] == m.n_record && m.state_w->ne[0] == m.max_state && + m.act_in_w->ne[0] == m.max_action && m.act_out2_w->ne[1] == m.max_action && + m.vlm[0].qkv->ne[1] == (m.n_heads + 2*m.n_kv)*m.head_dim && m.time_in_w->ne[0] % 4 == 0 && + m.stem_w->ne[0] == 3*m.stem*m.stem && m.stem_w->ne[1] == m.vis_dims[0] && + m.conn_proj->ne[0] == 4*vd && m.conn_proj->ne[1] == m.hidden && m.vision_proj->ne[0] == vd && + m.task_proj->ne[0] == m.hidden && m.task_proj->ne[1] == m.vision_proj->ne[1] && + m.expert[0].cross_k->ne[0] == m.n_kv*m.head_dim && m.cond_w->ne[1] == m.ex_hidden; + for (size_t s = 0; s < m.stages.size(); ++s) + for (const BlockW & b : m.stages[s].blocks) + ok = ok && b.dw_w->ne[0] == m.vis_dims[s] && b.dw_w->ne[1] == 7 && b.dw_w->ne[2] == 7; + const int64_t n_rows = m.tok_embd->ne[1]; + for (int32_t r : m.token_map) + ok = ok && r >= 0 && r < n_rows; + if (!ok) { + std::fprintf(stderr, "vla(picovla): tensor shapes disagree with GGUF metadata\n"); + return false; + } + + auto stats = [&](const char * name, int64_t n, std::vector & dst) { + dst = g.read_f32(name); + return (int64_t) dst.size() == n; + }; + if (!stats("state_mean", m.real_state, m.state_mean) || !stats("state_std", m.real_state, m.state_std) || + !stats("action_mean", m.real_action, m.action_mean) || !stats("action_std", m.real_action, m.action_std)) { + std::fprintf(stderr, "vla(picovla): GGUF normalizer stats are missing or the wrong size\n"); + return false; + } + return true; +} + +} // namespace + +std::unique_ptr picovla_create(const std::string& mmproj_path, + const std::string& ckpt_path, + const std::string&, + const Options& opts) { + if (!mmproj_path.empty()) + std::printf("vla(picovla): note - mmproj '%s' is ignored (vision is baked into the GGUF)\n", + mmproj_path.c_str()); + + auto m = std::make_unique(); + m->mt = opts.weight_dtype.value_or(GGML_TYPE_F32); + + gguf_reader g("picovla"); + if (!g.open(ckpt_path)) + return nullptr; + if (!g.has("picovla.architecture")) { + std::fprintf(stderr, "vla(picovla): %s is not a picovla GGUF\n", ckpt_path.c_str()); + return nullptr; + } + if (!load_config(g, *m) || !resolve_num_steps("picovla", opts, m->num_steps)) + return nullptr; + + const Backend b = backend_init("vla(picovla)", m->n_threads); + if (!b.handle) + return nullptr; + m->backend = b.handle; + + if (!load_weights(*m, g)) + return nullptr; + m->past.assign((size_t) (m->hidden*m->n_record), 0.0f); + + m->cfg.n_img = m->n_views * m->img_tokens(); + m->cfg.n_lang = m->max_lang; + m->cfg.n_state = 1; + m->cfg.n_suffix = m->chunk; + m->cfg.hidden = m->hidden; + m->cfg.expert_h = m->ex_hidden; + m->cfg.intermediate = m->inter; + m->cfg.expert_inter = m->ex_inter; + m->cfg.n_q_heads = m->n_heads; + m->cfg.n_kv_heads = m->n_kv; + m->cfg.head_dim = m->head_dim; + m->cfg.n_layers = m->n_layers; + m->cfg.max_state_dim = m->max_state; + m->cfg.max_action_dim = m->max_action; + m->cfg.real_state_dim = m->real_state; + m->cfg.real_action_dim = m->real_action; + m->cfg.norm_eps = m->norm_eps; + m->cfg.rms_eps = m->rms_eps; + m->cfg.min_period = m->min_period; + m->cfg.max_period = m->max_period; + m->cfg.num_steps = m->num_steps; + m->cfg.rope_n_dims = (int) m->head_dim; + m->cfg.rope_mode = GGML_ROPE_TYPE_NEOX; + m->cfg.rope_freq_base = m->rope_base; + + std::printf("vla(picovla): weights resident %.2f GiB (%s) - ConvNeXt x%lld views + Llama 1+%lld layers + " + "expert %lld, chunk %lld, %d MeanFlow steps, %zu-token vocabulary\n", + ggml_backend_buffer_get_size(m->weight_buf)/(1024.0*1024.0*1024.0), dtype_name(m->mt), + (long long) m->n_views, (long long) m->n_layers, (long long) m->ex_hidden, (long long) m->chunk, + m->num_steps, (size_t) m->tok_embd->ne[1]); + return m; +} + +std::vector PicoVlaModelArch::predict(const Inputs& in) { + using clock = std::chrono::steady_clock; + const auto t0 = clock::now(); + stats = Stats{}; + + if (in.precomputed_img_emb) { + std::fprintf(stderr, "vla(picovla): precomputed_img_emb is not supported; pass raw images\n"); + return {}; + } + if (!in.images || in.n_images != n_views) { + std::fprintf(stderr, "vla(picovla): expected %lld views (agentview, wrist), got %d\n", + (long long) n_views, in.n_images); + return {}; + } + for (int64_t i = 0; i < n_views; ++i) + if (!view_ok("picovla", in.images[i], image_size)) + return {}; + if (!in.lang_tokens || in.n_lang < 1 || in.n_lang > max_lang) { + std::fprintf(stderr, "vla(picovla): %d language tokens; need 1..%lld\n", in.n_lang, (long long) max_lang); + return {}; + } + // Ids the training data never used share the padding row, as after + // CompactTaskEmbedding.compact(). + const int64_t L = in.n_lang; + std::vector ids((size_t) L); + for (int64_t i = 0; i < L; ++i) { + const int32_t t = in.lang_tokens[i]; + if (t < 0 || t >= vocab) { + std::fprintf(stderr, "vla(picovla): token %d out of vocab\n", t); + return {}; + } + ids[(size_t) i] = token_map[(size_t) t]; + } + + const size_t arena = ggml_tensor_overhead()*8192 + ggml_graph_overhead_custom(8192, false); + if (!graph.ensure(backend, Key{L}, arena, [&](ggml_context * C, IO & io) { return build(C, io, L); })) { + std::fprintf(stderr, "vla(picovla): graph build/alloc failed\n"); + return {}; + } + IO & io = graph.io(); + + // Pixels: ImageNet-normalized, kept interleaved (channels-last). + const int64_t side = image_size, npx = side*side; + std::vector pixels((size_t) (3*npx*n_views)); + for (int64_t v = 0; v < n_views; ++v) { + const ImageView & iv = in.images[v]; + float * dst = pixels.data() + (size_t) (v*3*npx); + for (int64_t p = 0; p < 3*npx; ++p) { + const int ch = (int) (p % 3); + const float px = iv.format == PixelFormat::U8 ? ((const uint8_t *) iv.data)[p] / 255.0f + : ((const float *) iv.data)[p]; + dst[p] = (px - kImagenetMean[ch]) / kImagenetStd[ch]; + } + } + + // LeRobot MEAN_STD, then zero padding to max_state_dim. + std::vector state((size_t) max_state, 0.0f); + if (in.state) + for (int64_t i = 0; i < real_state; ++i) + state[(size_t) i] = (in.state[i] - state_mean[(size_t) i]) / (state_std[(size_t) i] + norm_eps); + + const int64_t P = prefix_len(L), R = n_record, T = P + 2*R; + std::vector pos((size_t) io.pos->ne[0]); + for (size_t i = 0; i < pos.size(); ++i) + pos[i] = (int32_t) i; + // Prefix rows see the prefix, record rows the prefix and the records, past + // rows everything: record tokens never read the past, so memory does not + // compound across calls. + std::vector mask((size_t) (T*T), 0.0f); + for (int64_t q = 0; q < T; ++q) { + const int64_t visible = q < P ? P : q < P + R ? P + R : T; + for (int64_t k = visible; k < T; ++k) + mask[(size_t) (q*T + k)] = -INFINITY; + } + + std::vector noise((size_t) (chunk*max_action)); + if (in.noise) { + std::copy(in.noise, in.noise + noise.size(), noise.begin()); + } else { + std::normal_distribution dist(0.0f, 1.0f); + for (float & v : noise) + v = dist(rng); + } + + // Step s integrates from t = 1 - periods[s] down to r = 1 - periods[s+1], + // periods clamped to the transport range (VLAMeanFlow.sample_actions). The + // reference builds t and r as float32 tensors, so they round to float here. + const int64_t tdim = time_in_w->ne[0], half = tdim / 2; + std::vector time((size_t) (tdim*num_steps)), dt((size_t) num_steps); + auto period = [&](int s) { return std::min(std::max((float) s / (float) num_steps, t_lo), t_hi); }; + for (int s = 0; s < num_steps; ++s) { + const float t = 1.0f - period(s), r = 1.0f - period(s + 1); + const std::vector er = sinusoidal_time_emb(r, half, min_period, max_period); + const std::vector et = sinusoidal_time_emb(t, half, min_period, max_period); + float * row = time.data() + (size_t) (s*tdim); + std::copy(er.begin(), er.end(), row); + std::copy(et.begin(), et.end(), row + half); + dt[(size_t) s] = t - r; + } + + if (in.reset_memory) + std::fill(past.begin(), past.end(), 0.0f); + + ggml_backend_tensor_set(io.pixels, pixels.data(), 0, ggml_nbytes(io.pixels)); + ggml_backend_tensor_set(io.ids, ids.data(), 0, ggml_nbytes(io.ids)); + ggml_backend_tensor_set(io.pos, pos.data(), 0, ggml_nbytes(io.pos)); + ggml_backend_tensor_set(io.mask, mask.data(), 0, ggml_nbytes(io.mask)); + ggml_backend_tensor_set(io.state, state.data(), 0, ggml_nbytes(io.state)); + ggml_backend_tensor_set(io.past, past.data(), 0, ggml_nbytes(io.past)); + ggml_backend_tensor_set(io.noise, noise.data(), 0, ggml_nbytes(io.noise)); + ggml_backend_tensor_set(io.time, time.data(), 0, ggml_nbytes(io.time)); + ggml_backend_tensor_set(io.dt, dt.data(), 0, ggml_nbytes(io.dt)); + + const auto tc = clock::now(); + graph_unique_names(graph.graph()); + if (ggml_backend_graph_compute(backend, graph.graph()) != GGML_STATUS_SUCCESS) { + std::fprintf(stderr, "vla(picovla): compute failed\n"); + return {}; + } + std::vector out((size_t) (chunk*max_action)); + ggml_backend_tensor_get(io.actions, out.data(), 0, out.size()*sizeof(float)); + ggml_backend_tensor_get(io.record, past.data(), 0, past.size()*sizeof(float)); + + for (int64_t a = 0; a < chunk; ++a) { + float * row = out.data() + (size_t) (a*max_action); + for (int64_t j = 0; j < max_action; ++j) + row[j] = j < real_action ? row[j]*action_std[(size_t) j] + action_mean[(size_t) j] : 0.0f; + } + + stats.ms_inference = std::chrono::duration(clock::now() - tc).count(); + stats.ms_total = std::chrono::duration(clock::now() - t0).count(); + return out; +} + +} // namespace vla diff --git a/src/serving/server.cpp b/src/serving/server.cpp index d65a6ed..d68b212 100644 --- a/src/serving/server.cpp +++ b/src/serving/server.cpp @@ -549,6 +549,7 @@ int main(int argc, char ** argv) { in.attention_mask = attn_mask_vec.empty() ? nullptr : attn_mask_vec.data(); in.attention_mask_n = static_cast(attn_mask_vec.size()); in.timing_detail = timing_detail; + in.reset_memory = req.reset_memory(); std::vector action_chunk = vla::predict(model, in); const auto & st = vla::last_stats(model); diff --git a/src/serving/vla-cli.cpp b/src/serving/vla-cli.cpp index 3f0a715..24699e1 100644 --- a/src/serving/vla-cli.cpp +++ b/src/serving/vla-cli.cpp @@ -131,6 +131,7 @@ const char * arch_slug(Arch a) { case Arch::VLA_JEPA: return "vla_jepa"; case Arch::OCTO: return "octo"; case Arch::TURBOVLA: return "turbovla"; + case Arch::PICOVLA: return "picovla"; } return ""; } diff --git a/src/serving/vla.proto b/src/serving/vla.proto index 60f64e4..9b742ee 100644 --- a/src/serving/vla.proto +++ b/src/serving/vla.proto @@ -31,6 +31,10 @@ message PredictRequest { uint32 precomputed_img_emb_n_views = 7; repeated uint32 attention_mask = 8; + + // Episode start: the model clears its cross-call memory before predicting. + // Only PicoVLA keeps any. + bool reset_memory = 9; } message PredictResponse { diff --git a/src/vla_c_api.cpp b/src/vla_c_api.cpp index fa9e781..ac2c2be 100644 --- a/src/vla_c_api.cpp +++ b/src/vla_c_api.cpp @@ -154,6 +154,7 @@ int32_t vla_predict(vla_model * h, const vla_inputs * in, ci.timing_detail = in->timing_detail == VLA_TIMING_PHASE ? vla::TimingDetail::PHASE : vla::TimingDetail::NONE; + ci.reset_memory = in->reset_memory != 0; const std::vector act = vla::predict(h->m, ci); if (act.empty()) diff --git a/tests/py/test_converters.py b/tests/py/test_converters.py index 9f3fdfe..6c01bc8 100644 --- a/tests/py/test_converters.py +++ b/tests/py/test_converters.py @@ -62,6 +62,8 @@ def float(self): return _F32Tensor() def to(self, *a): return self def reshape(self, *a): return self def squeeze(self, *a): return self + def permute(self, *a): return self + def __getitem__(self, key): return self def clone(self): return self def __mul__(self, other): return self @@ -349,6 +351,47 @@ def test_turbovla_converter_remap(): } <= set(tensors.keys_read) +def test_picovla_converter_remap(): + import importlib + + P = importlib.import_module("convert_picovla_to_gguf") + tensors = _SourceTensors() + writer = _Writer() + P.write_vision(writer, tensors.__getitem__, [1, 1, 1, 1]) + P.write_expert(writer, tensors.__getitem__, 1) + + def block(s): + return [f"vis.blk.{s}.0.{n}" for n in ("dw.weight", "dw.bias", "ln.weight", "ln.bias", + "fc1.weight", "fc1.bias", "fc2.weight", "fc2.bias")] + + def down(s): + return [f"vis.down.{s}.norm.weight", f"vis.down.{s}.norm.bias", f"vis.down.{s}.weight", f"vis.down.{s}.bias"] + + assert writer.names == [ + "vis.stem.weight", "vis.stem.bias", "vis.stem_norm.weight", "vis.stem_norm.bias", + *block(0), *down(1), *block(1), *down(2), *block(2), *down(3), *block(3), + "vis.norm.weight", "vis.norm.bias", "conn.task_proj.weight", "conn.vision_proj.weight", "conn.proj.weight", + "aex.cond_proj.weight", "aex.cond_proj.bias", + "aex.blk.0.attn_norm.weight", "aex.blk.0.attn_q.weight", "aex.blk.0.attn_k.weight", + "aex.blk.0.attn_v.weight", "aex.blk.0.attn_o.weight", "aex.blk.0.ffn_norm.weight", + "aex.blk.0.ffn_gate.weight", "aex.blk.0.ffn_up.weight", "aex.blk.0.ffn_down.weight", + "aex.blk.0.cross_k.weight", "aex.blk.0.cross_v.weight", "aex.blk.0.ada_attn.weight", + "aex.blk.0.ada_attn.bias", "aex.blk.0.ada_ffn.weight", "aex.blk.0.ada_ffn.bias", + "aex.output_ada.weight", "aex.output_ada.bias", + ] + vis = "base_model.backbone.vision_encoder" + assert { + f"{vis}.vision_model.model.stages.0.downsample_layers.0.weight", + f"{vis}.vision_model.model.stages.1.downsample_layers.1.weight", + f"{vis}.vision_model.model.stages.3.layers.0.gamma", + f"{vis}.vision_model.layer_norm.weight", + f"{vis}.connector.proj.weight", + "base_model.expert_head.model.layers.0.self_attn.cross_k_proj.weight", + "base_model.expert_head.model.layers.0.post_attention_adanorm.linear.bias", + "base_model.expert_head.model.norm.linear.weight", + } <= set(tensors.keys_read) + + def test_quantize_skip_names(): import importlib Q = importlib.import_module("quantize_gguf")