|
| 1 | +# Copyright (c) Meta Platforms, Inc. and affiliates. |
| 2 | +# All rights reserved. |
| 3 | +# |
| 4 | +# This source code is licensed under the BSD-style license found in the |
| 5 | +# LICENSE file in the root directory of this source tree. |
| 6 | + |
| 7 | +# pyre-strict |
| 8 | + |
| 9 | +""" |
| 10 | +Custom ops for the native backend. |
| 11 | +
|
| 12 | +Registers ``native::rope``, the fused rotary position embedding op produced by |
| 13 | +``FuseRoPEPass``. Registration happens as an import side effect, so this module is |
| 14 | +imported from the backend's entry points (``passes`` / ``partitioner.py``) to |
| 15 | +guarantee the op exists before lowering. The reference ``CompositeExplicitAutograd`` |
| 16 | +body is only used for eager execution; AOT export/serialization uses the fake |
| 17 | +(meta) impl. |
| 18 | +""" |
| 19 | + |
| 20 | +import torch |
| 21 | +from torch import Tensor |
| 22 | + |
| 23 | +lib: torch.library.Library = torch.library.Library("native", "DEF") |
| 24 | + |
| 25 | + |
| 26 | +lib.define( |
| 27 | + "rope(Tensor input, Tensor position_ids, Tensor inv_freq, " |
| 28 | + "bool interleaved=False, float attention_scale=1.0) -> Tensor" |
| 29 | +) |
| 30 | + |
| 31 | + |
| 32 | +def _check_rope_shapes(input: Tensor, position_ids: Tensor, inv_freq: Tensor) -> None: |
| 33 | + torch._check(input.is_floating_point(), lambda: "rope input must be floating point") |
| 34 | + torch._check(input.dim() == 4, lambda: "rope input must have shape [B, H, T, D]") |
| 35 | + torch._check( |
| 36 | + position_ids.dim() == 2, lambda: "rope position_ids must have shape [Bpos, T]" |
| 37 | + ) |
| 38 | + torch._check(inv_freq.dim() == 1, lambda: "rope inv_freq must have shape [R / 2]") |
| 39 | + torch._check( |
| 40 | + (position_ids.shape[0] == 1) | (position_ids.shape[0] == input.shape[0]), |
| 41 | + lambda: "rope position batch must be 1 or match the input batch", |
| 42 | + ) |
| 43 | + torch._check( |
| 44 | + position_ids.shape[1] == input.shape[2], |
| 45 | + lambda: "rope position sequence length must match the input", |
| 46 | + ) |
| 47 | + torch._check( |
| 48 | + 2 * inv_freq.shape[0] <= input.shape[3], |
| 49 | + lambda: "rope rotary width must not exceed the input width", |
| 50 | + ) |
| 51 | + |
| 52 | + |
| 53 | +@torch.library.register_fake("native::rope", lib=lib) |
| 54 | +def _rope_fake( |
| 55 | + input: Tensor, |
| 56 | + position_ids: Tensor, |
| 57 | + inv_freq: Tensor, |
| 58 | + interleaved: bool = False, |
| 59 | + attention_scale: float = 1.0, |
| 60 | +) -> Tensor: |
| 61 | + _check_rope_shapes(input, position_ids, inv_freq) |
| 62 | + return input.new_empty(input.shape, dtype=input.dtype) |
| 63 | + |
| 64 | + |
| 65 | +@torch.library.impl("native::rope", "CompositeExplicitAutograd", lib=lib) |
| 66 | +def _rope_impl( |
| 67 | + input: Tensor, |
| 68 | + position_ids: Tensor, |
| 69 | + inv_freq: Tensor, |
| 70 | + interleaved: bool = False, |
| 71 | + attention_scale: float = 1.0, |
| 72 | +) -> Tensor: |
| 73 | + """Rotate the leading R channels of [B, H, T, D], preserving the tail. |
| 74 | +
|
| 75 | + Positions [Bpos, T] broadcast over heads and, when Bpos is 1, batches. |
| 76 | + inv_freq [R / 2] determines the rotary width. Phase, trig and attention |
| 77 | + scaling use FP32; the tables are then cast to the input dtype. |
| 78 | + """ |
| 79 | + _check_rope_shapes(input, position_ids, inv_freq) |
| 80 | + # Elementwise multiplication keeps phase construction in FP32 under autocast, |
| 81 | + # unlike the equivalent batched matmul used by HF's source pattern. |
| 82 | + phase = position_ids.float().unsqueeze(-1) * inv_freq.float() |
| 83 | + if interleaved: |
| 84 | + phase = phase.repeat_interleave(2, dim=-1) |
| 85 | + else: |
| 86 | + phase = torch.cat((phase, phase), dim=-1) |
| 87 | + cos = (phase.cos() * attention_scale).to(input.dtype).unsqueeze(1) |
| 88 | + sin = (phase.sin() * attention_scale).to(input.dtype).unsqueeze(1) |
| 89 | + rotary = 2 * inv_freq.shape[0] |
| 90 | + x_rot = input[..., :rotary] |
| 91 | + x_pass = input[..., rotary:] |
| 92 | + if interleaved: |
| 93 | + x1 = x_rot[..., 0::2] |
| 94 | + x2 = x_rot[..., 1::2] |
| 95 | + rotated = torch.stack((-x2, x1), dim=-1).flatten(-2) |
| 96 | + else: |
| 97 | + half = rotary // 2 |
| 98 | + x1 = x_rot[..., :half] |
| 99 | + x2 = x_rot[..., half:] |
| 100 | + rotated = torch.cat((-x2, x1), dim=-1) |
| 101 | + out = x_rot * cos + rotated * sin |
| 102 | + if x_pass.shape[-1] == 0: |
| 103 | + return out |
| 104 | + return torch.cat((out, x_pass), dim=-1) |
| 105 | + |
| 106 | + |
| 107 | +rope_op: torch._ops.OpOverload = torch.ops.native.rope.default |
0 commit comments